From db255ab4dd956ca649cd6d0a9437db938a6ddaad Mon Sep 17 00:00:00 2001 From: andrej Date: Thu, 24 Sep 2026 17:41:22 -0600 Subject: [PATCH 01/17] Make the Gemma 4 and shared host shared library code public Co-authored-by: Alfred Co-authored-by: ngdxzy Co-authored-by: joeldushouyu Co-authored-by: alfxu_amdeng Co-authored-by: Alfred Co-authored-by: Abhishek Varma Co-authored-by: alfxu_amd Co-authored-by: unknown Co-authored-by: ngdymx Co-authored-by: alfxu-amd Co-authored-by: Michelle Yu Co-authored-by: shdu Co-authored-by: Shouyu <65317431+joeldushouyu@users.noreply.github.com> Co-authored-by: zaneni6 Co-authored-by: ngdymx <58451250+ngdymx@users.noreply.github.com> Co-authored-by: Jorn Tuyls --- src/CMakeLists.txt | 33 +- src/detail/CMakeLists.txt | 107 + src/detail/dequant/dequant.cpp | 330 + src/detail/dequant/dequant_detail.hpp | 175 + src/detail/gemm/gemm.cpp | 454 ++ src/detail/gemm/gemm_detail.hpp | 94 + src/detail/gemma4e_npu/avx512_util.hpp | 296 + src/detail/gemma4e_npu/conv1d_prefill.hpp | 230 + src/detail/gemma4e_npu/embedding_q8_0.hpp | 85 + src/detail/gemma4e_npu/gemma4e_audio.cpp | 2280 +++++++ src/detail/gemma4e_npu/gemma4e_audio.hpp | 199 + .../gemma4e_npu/gemma4e_audio_attention.cpp | 440 ++ .../gemma4e_npu/gemma4e_audio_attention.hpp | 111 + .../gemma4e_npu/gemma4e_cpu_functions.hpp | 374 ++ src/detail/gemma4e_npu/gemma4e_image.cpp | 1853 ++++++ src/detail/gemma4e_npu/gemma4e_image.hpp | 136 + src/detail/gemma4e_npu/gemma4e_npu.cpp | 858 +++ src/detail/gemma4e_npu/gemma4e_npu_def.hpp | 558 ++ src/detail/gemma4e_npu/gemma4e_npu_detail.hpp | 181 + .../gemma4e_npu/gemma4e_npu_sequence.cpp | 1279 ++++ .../gemma4e_npu/gemma4e_npu_sequence.hpp | 216 + src/detail/gemma4e_npu/gemma4e_prefill.cpp | 584 ++ src/detail/gemma4e_npu/gemma4e_prefill.hpp | 521 ++ .../gemma4e_vision_prefill_helper.cpp | 1320 ++++ .../gemma4e_vision_prefill_helper.hpp | 173 + src/detail/gemma4e_npu/mmRuntimeSequence.hpp | 550 ++ src/detail/gemma4e_npu/reorder_cpy.hpp | 37 + src/detail/gemma4e_npu/rot_pos_emb.cpp | 219 + src/detail/gemma4e_npu/rot_pos_emb.hpp | 82 + src/detail/gemma4e_npu/seq_gen.hpp | 327 + src/detail/include/aiebu/aiebu.h | 88 + src/detail/include/aiebu/aiebu_assembler.h | 123 + src/detail/include/aiebu/aiebu_error.h | 45 + src/detail/include/base64.hpp | 697 +++ src/detail/include/biovault_bfloat16.h | 210 + src/detail/include/buffer.hpp | 1075 ++++ src/detail/include/causal_lm.hpp | 64 + src/detail/include/embedding_model.hpp | 29 + src/detail/include/flm_override.hpp | 30 + src/detail/include/hrx_cpp/hrx_cpp.hpp | 528 ++ src/detail/include/lm_config.hpp | 89 + src/detail/include/metrices.hpp | 149 + src/detail/include/model_list.hpp | 191 + src/detail/include/models/gemma/gemma_npu.hpp | 74 + .../models/gemma/gemma_npu_sequence.hpp | 56 + .../models/gemma4_12b/gemma4_12b_npu.hpp | 216 + .../include/models/gemma4e/gemma4e_npu.hpp | 228 + .../models/gemma4e_flash/gemma4e_flash.hpp | 156 + .../gemma_embedding/gemma_embedding.hpp | 44 + .../models/gemma_text/gemma_text_dequant.hpp | 43 + .../models/gemma_text/gemma_text_gemm.hpp | 42 + .../models/gemma_text/gemma_text_lm_head.hpp | 44 + .../models/gemma_text/gemma_text_npu.hpp | 74 + .../gemma_text/gemma_text_npu_sequence.hpp | 50 + .../include/models/gpt_oss/gpt_oss_npu.hpp | 75 + .../models/gpt_oss/gpt_oss_npu_sequence.hpp | 51 + .../include/models/hunyuan/hunyuan_npu.hpp | 73 + src/detail/include/models/lfm2/lfm2_npu.hpp | 74 + src/detail/include/models/llama/llama_npu.hpp | 74 + .../models/llama/llama_npu_sequence.hpp | 69 + .../include/models/nanbeige/nanbeige_npu.hpp | 74 + .../models/nanbeige/nanbeige_npu_sequence.hpp | 70 + src/detail/include/models/phi4/phi4_npu.hpp | 74 + .../include/models/phi4/phi4_npu_sequence.hpp | 67 + src/detail/include/models/qwen2/qwen2_npu.hpp | 73 + .../include/models/qwen2vl/qwen2vl_npu.hpp | 109 + src/detail/include/models/qwen3/qwen3_npu.hpp | 74 + .../models/qwen3/qwen3_npu_sequence.hpp | 69 + .../models/qwen3_5_omni/qwen3_5_omni.hpp | 95 + .../models/qwen3_5vl/qwen3_5vl_npu.hpp | 145 + .../models/qwen3_6_moe/qwen3_6_moe_npu.hpp | 145 + .../include/models/qwen3vl/qwen3vl_npu.hpp | 108 + .../models/qwen3vl_flash/qwen3vl_flash.hpp | 72 + .../include/models/whisper/whisper_npu.hpp | 50 + src/detail/include/modules/dequant.hpp | 45 + src/detail/include/modules/embedding.hpp | 57 + src/detail/include/modules/gemm.hpp | 52 + src/detail/include/modules/lm_head.hpp | 44 + src/detail/include/modules/mha.hpp | 47 + src/detail/include/modules/sampler.hpp | 69 + .../include/nlohmann/adl_serializer.hpp | 55 + .../nlohmann/byte_container_with_subtype.hpp | 103 + .../include/nlohmann/detail/abi_macros.hpp | 111 + .../nlohmann/detail/conversions/from_json.hpp | 583 ++ .../nlohmann/detail/conversions/to_chars.hpp | 1118 ++++ .../nlohmann/detail/conversions/to_json.hpp | 486 ++ .../include/nlohmann/detail/exceptions.hpp | 291 + src/detail/include/nlohmann/detail/hash.hpp | 129 + .../nlohmann/detail/input/binary_reader.hpp | 3081 ++++++++++ .../nlohmann/detail/input/input_adapters.hpp | 549 ++ .../nlohmann/detail/input/json_sax.hpp | 986 +++ .../include/nlohmann/detail/input/lexer.hpp | 1643 +++++ .../include/nlohmann/detail/input/parser.hpp | 519 ++ .../nlohmann/detail/input/position_t.hpp | 37 + .../detail/iterators/internal_iterator.hpp | 35 + .../nlohmann/detail/iterators/iter_impl.hpp | 760 +++ .../detail/iterators/iteration_proxy.hpp | 235 + .../detail/iterators/iterator_traits.hpp | 61 + .../iterators/json_reverse_iterator.hpp | 130 + .../detail/iterators/primitive_iterator.hpp | 132 + .../detail/json_custom_base_class.hpp | 39 + .../include/nlohmann/detail/json_pointer.hpp | 988 +++ .../include/nlohmann/detail/json_ref.hpp | 78 + .../include/nlohmann/detail/macro_scope.hpp | 601 ++ .../include/nlohmann/detail/macro_unscope.hpp | 47 + .../nlohmann/detail/meta/call_std/begin.hpp | 17 + .../nlohmann/detail/meta/call_std/end.hpp | 17 + .../nlohmann/detail/meta/cpp_future.hpp | 171 + .../include/nlohmann/detail/meta/detected.hpp | 70 + .../nlohmann/detail/meta/identity_tag.hpp | 21 + .../include/nlohmann/detail/meta/is_sax.hpp | 159 + .../include/nlohmann/detail/meta/std_fs.hpp | 29 + .../nlohmann/detail/meta/type_traits.hpp | 821 +++ .../include/nlohmann/detail/meta/void_t.hpp | 24 + .../nlohmann/detail/output/binary_writer.hpp | 1863 ++++++ .../detail/output/output_adapters.hpp | 147 + .../nlohmann/detail/output/serializer.hpp | 988 +++ .../include/nlohmann/detail/string_concat.hpp | 146 + .../include/nlohmann/detail/string_escape.hpp | 72 + .../include/nlohmann/detail/string_utils.hpp | 37 + .../include/nlohmann/detail/value_t.hpp | 118 + src/detail/include/nlohmann/json.hpp | 5309 +++++++++++++++++ src/detail/include/nlohmann/json_fwd.hpp | 75 + src/detail/include/nlohmann/ordered_map.hpp | 359 ++ .../nlohmann/thirdparty/hedley/hedley.hpp | 2045 +++++++ .../thirdparty/hedley/hedley_undef.hpp | 158 + .../GateDeltaNetPrefillRuntimeSequence.hpp | 555 ++ .../image_attention_sequence.hpp | 1458 +++++ .../npu_sequences/mmRuntimeSequence.hpp | 581 ++ .../window_attention_sequence.hpp | 484 ++ src/detail/include/npu_utils/amdxdna_accel.h | 632 ++ src/detail/include/npu_utils/flm_runtime.hpp | 17 + .../include/npu_utils/instr_utils/npu_cmd.hpp | 220 + .../npu_utils/instr_utils/npu_cmd_ddr.hpp | 92 + .../instr_utils/npu_cmd_issue_token.hpp | 77 + .../instr_utils/npu_cmd_maskwrite.hpp | 69 + .../instr_utils/npu_cmd_preemption.hpp | 41 + .../npu_utils/instr_utils/npu_cmd_wait.hpp | 66 + .../npu_utils/instr_utils/npu_cmd_write.hpp | 124 + .../instr_utils/npu_cmd_write_dma.hpp | 259 + .../include/npu_utils/npu_instr_utils.hpp | 1462 +++++ src/detail/include/npu_utils/npu_utils.hpp | 1316 ++++ src/detail/include/tensor_2d.hpp | 79 + .../include/tensor_utils/q4_npu_eXpress.hpp | 60 + .../include/tensor_utils/safe_tensors.hpp | 88 + src/detail/include/tokenizer/tokenizer.hpp | 81 + src/detail/include/tokenizer/tokenizers_c.h | 55 + src/detail/include/tokenizer/tokenizers_cpp.h | 110 + src/detail/include/typedef.hpp | 146 + src/detail/include/utils/avx512_util.hpp | 287 + src/detail/include/utils/debug_utils.hpp | 188 + src/detail/include/utils/error_measure.hpp | 157 + src/detail/include/utils/profiler.hpp | 82 + src/detail/include/utils/utils.hpp | 394 ++ src/detail/include/utils/vm_args.hpp | 46 + .../include/vision/conv3dPatchEmbedding.hpp | 127 + src/detail/include/vision/norm.hpp | 48 + .../include/vision/pos_embed_interpolate.hpp | 24 + src/detail/include/vision/rot_pos_emb.hpp | 96 + .../include/vision/vision_prefill_helper.hpp | 44 + src/detail/include/weight_desc.hpp | 340 ++ src/detail/lm_head/lm_head.cpp | 191 + src/detail/lm_head/lm_head_detail.hpp | 45 + src/detail/vision_common/norm.cpp | 184 + 164 files changed, 56169 insertions(+), 1 deletion(-) create mode 100644 src/detail/CMakeLists.txt create mode 100644 src/detail/dequant/dequant.cpp create mode 100644 src/detail/dequant/dequant_detail.hpp create mode 100644 src/detail/gemm/gemm.cpp create mode 100644 src/detail/gemm/gemm_detail.hpp create mode 100644 src/detail/gemma4e_npu/avx512_util.hpp create mode 100644 src/detail/gemma4e_npu/conv1d_prefill.hpp create mode 100644 src/detail/gemma4e_npu/embedding_q8_0.hpp create mode 100644 src/detail/gemma4e_npu/gemma4e_audio.cpp create mode 100644 src/detail/gemma4e_npu/gemma4e_audio.hpp create mode 100644 src/detail/gemma4e_npu/gemma4e_audio_attention.cpp create mode 100644 src/detail/gemma4e_npu/gemma4e_audio_attention.hpp create mode 100644 src/detail/gemma4e_npu/gemma4e_cpu_functions.hpp create mode 100644 src/detail/gemma4e_npu/gemma4e_image.cpp create mode 100644 src/detail/gemma4e_npu/gemma4e_image.hpp create mode 100644 src/detail/gemma4e_npu/gemma4e_npu.cpp create mode 100644 src/detail/gemma4e_npu/gemma4e_npu_def.hpp create mode 100644 src/detail/gemma4e_npu/gemma4e_npu_detail.hpp create mode 100644 src/detail/gemma4e_npu/gemma4e_npu_sequence.cpp create mode 100644 src/detail/gemma4e_npu/gemma4e_npu_sequence.hpp create mode 100644 src/detail/gemma4e_npu/gemma4e_prefill.cpp create mode 100644 src/detail/gemma4e_npu/gemma4e_prefill.hpp create mode 100644 src/detail/gemma4e_npu/gemma4e_vision_prefill_helper.cpp create mode 100644 src/detail/gemma4e_npu/gemma4e_vision_prefill_helper.hpp create mode 100644 src/detail/gemma4e_npu/mmRuntimeSequence.hpp create mode 100644 src/detail/gemma4e_npu/reorder_cpy.hpp create mode 100644 src/detail/gemma4e_npu/rot_pos_emb.cpp create mode 100644 src/detail/gemma4e_npu/rot_pos_emb.hpp create mode 100644 src/detail/gemma4e_npu/seq_gen.hpp create mode 100644 src/detail/include/aiebu/aiebu.h create mode 100644 src/detail/include/aiebu/aiebu_assembler.h create mode 100644 src/detail/include/aiebu/aiebu_error.h create mode 100644 src/detail/include/base64.hpp create mode 100644 src/detail/include/biovault_bfloat16.h create mode 100644 src/detail/include/buffer.hpp create mode 100644 src/detail/include/causal_lm.hpp create mode 100644 src/detail/include/embedding_model.hpp create mode 100644 src/detail/include/flm_override.hpp create mode 100644 src/detail/include/hrx_cpp/hrx_cpp.hpp create mode 100644 src/detail/include/lm_config.hpp create mode 100644 src/detail/include/metrices.hpp create mode 100644 src/detail/include/model_list.hpp create mode 100644 src/detail/include/models/gemma/gemma_npu.hpp create mode 100644 src/detail/include/models/gemma/gemma_npu_sequence.hpp create mode 100644 src/detail/include/models/gemma4_12b/gemma4_12b_npu.hpp create mode 100644 src/detail/include/models/gemma4e/gemma4e_npu.hpp create mode 100644 src/detail/include/models/gemma4e_flash/gemma4e_flash.hpp create mode 100644 src/detail/include/models/gemma_embedding/gemma_embedding.hpp create mode 100644 src/detail/include/models/gemma_text/gemma_text_dequant.hpp create mode 100644 src/detail/include/models/gemma_text/gemma_text_gemm.hpp create mode 100644 src/detail/include/models/gemma_text/gemma_text_lm_head.hpp create mode 100644 src/detail/include/models/gemma_text/gemma_text_npu.hpp create mode 100644 src/detail/include/models/gemma_text/gemma_text_npu_sequence.hpp create mode 100644 src/detail/include/models/gpt_oss/gpt_oss_npu.hpp create mode 100644 src/detail/include/models/gpt_oss/gpt_oss_npu_sequence.hpp create mode 100644 src/detail/include/models/hunyuan/hunyuan_npu.hpp create mode 100644 src/detail/include/models/lfm2/lfm2_npu.hpp create mode 100644 src/detail/include/models/llama/llama_npu.hpp create mode 100644 src/detail/include/models/llama/llama_npu_sequence.hpp create mode 100644 src/detail/include/models/nanbeige/nanbeige_npu.hpp create mode 100644 src/detail/include/models/nanbeige/nanbeige_npu_sequence.hpp create mode 100644 src/detail/include/models/phi4/phi4_npu.hpp create mode 100644 src/detail/include/models/phi4/phi4_npu_sequence.hpp create mode 100644 src/detail/include/models/qwen2/qwen2_npu.hpp create mode 100644 src/detail/include/models/qwen2vl/qwen2vl_npu.hpp create mode 100644 src/detail/include/models/qwen3/qwen3_npu.hpp create mode 100644 src/detail/include/models/qwen3/qwen3_npu_sequence.hpp create mode 100644 src/detail/include/models/qwen3_5_omni/qwen3_5_omni.hpp create mode 100644 src/detail/include/models/qwen3_5vl/qwen3_5vl_npu.hpp create mode 100644 src/detail/include/models/qwen3_6_moe/qwen3_6_moe_npu.hpp create mode 100644 src/detail/include/models/qwen3vl/qwen3vl_npu.hpp create mode 100644 src/detail/include/models/qwen3vl_flash/qwen3vl_flash.hpp create mode 100644 src/detail/include/models/whisper/whisper_npu.hpp create mode 100644 src/detail/include/modules/dequant.hpp create mode 100644 src/detail/include/modules/embedding.hpp create mode 100644 src/detail/include/modules/gemm.hpp create mode 100644 src/detail/include/modules/lm_head.hpp create mode 100644 src/detail/include/modules/mha.hpp create mode 100644 src/detail/include/modules/sampler.hpp create mode 100644 src/detail/include/nlohmann/adl_serializer.hpp create mode 100644 src/detail/include/nlohmann/byte_container_with_subtype.hpp create mode 100644 src/detail/include/nlohmann/detail/abi_macros.hpp create mode 100644 src/detail/include/nlohmann/detail/conversions/from_json.hpp create mode 100644 src/detail/include/nlohmann/detail/conversions/to_chars.hpp create mode 100644 src/detail/include/nlohmann/detail/conversions/to_json.hpp create mode 100644 src/detail/include/nlohmann/detail/exceptions.hpp create mode 100644 src/detail/include/nlohmann/detail/hash.hpp create mode 100644 src/detail/include/nlohmann/detail/input/binary_reader.hpp create mode 100644 src/detail/include/nlohmann/detail/input/input_adapters.hpp create mode 100644 src/detail/include/nlohmann/detail/input/json_sax.hpp create mode 100644 src/detail/include/nlohmann/detail/input/lexer.hpp create mode 100644 src/detail/include/nlohmann/detail/input/parser.hpp create mode 100644 src/detail/include/nlohmann/detail/input/position_t.hpp create mode 100644 src/detail/include/nlohmann/detail/iterators/internal_iterator.hpp create mode 100644 src/detail/include/nlohmann/detail/iterators/iter_impl.hpp create mode 100644 src/detail/include/nlohmann/detail/iterators/iteration_proxy.hpp create mode 100644 src/detail/include/nlohmann/detail/iterators/iterator_traits.hpp create mode 100644 src/detail/include/nlohmann/detail/iterators/json_reverse_iterator.hpp create mode 100644 src/detail/include/nlohmann/detail/iterators/primitive_iterator.hpp create mode 100644 src/detail/include/nlohmann/detail/json_custom_base_class.hpp create mode 100644 src/detail/include/nlohmann/detail/json_pointer.hpp create mode 100644 src/detail/include/nlohmann/detail/json_ref.hpp create mode 100644 src/detail/include/nlohmann/detail/macro_scope.hpp create mode 100644 src/detail/include/nlohmann/detail/macro_unscope.hpp create mode 100644 src/detail/include/nlohmann/detail/meta/call_std/begin.hpp create mode 100644 src/detail/include/nlohmann/detail/meta/call_std/end.hpp create mode 100644 src/detail/include/nlohmann/detail/meta/cpp_future.hpp create mode 100644 src/detail/include/nlohmann/detail/meta/detected.hpp create mode 100644 src/detail/include/nlohmann/detail/meta/identity_tag.hpp create mode 100644 src/detail/include/nlohmann/detail/meta/is_sax.hpp create mode 100644 src/detail/include/nlohmann/detail/meta/std_fs.hpp create mode 100644 src/detail/include/nlohmann/detail/meta/type_traits.hpp create mode 100644 src/detail/include/nlohmann/detail/meta/void_t.hpp create mode 100644 src/detail/include/nlohmann/detail/output/binary_writer.hpp create mode 100644 src/detail/include/nlohmann/detail/output/output_adapters.hpp create mode 100644 src/detail/include/nlohmann/detail/output/serializer.hpp create mode 100644 src/detail/include/nlohmann/detail/string_concat.hpp create mode 100644 src/detail/include/nlohmann/detail/string_escape.hpp create mode 100644 src/detail/include/nlohmann/detail/string_utils.hpp create mode 100644 src/detail/include/nlohmann/detail/value_t.hpp create mode 100644 src/detail/include/nlohmann/json.hpp create mode 100644 src/detail/include/nlohmann/json_fwd.hpp create mode 100644 src/detail/include/nlohmann/ordered_map.hpp create mode 100644 src/detail/include/nlohmann/thirdparty/hedley/hedley.hpp create mode 100644 src/detail/include/nlohmann/thirdparty/hedley/hedley_undef.hpp create mode 100644 src/detail/include/npu_sequences/GateDeltaNetPrefillRuntimeSequence.hpp create mode 100644 src/detail/include/npu_sequences/image_attention_sequence.hpp create mode 100644 src/detail/include/npu_sequences/mmRuntimeSequence.hpp create mode 100644 src/detail/include/npu_sequences/window_attention_sequence.hpp create mode 100644 src/detail/include/npu_utils/amdxdna_accel.h create mode 100644 src/detail/include/npu_utils/flm_runtime.hpp create mode 100644 src/detail/include/npu_utils/instr_utils/npu_cmd.hpp create mode 100644 src/detail/include/npu_utils/instr_utils/npu_cmd_ddr.hpp create mode 100644 src/detail/include/npu_utils/instr_utils/npu_cmd_issue_token.hpp create mode 100644 src/detail/include/npu_utils/instr_utils/npu_cmd_maskwrite.hpp create mode 100644 src/detail/include/npu_utils/instr_utils/npu_cmd_preemption.hpp create mode 100644 src/detail/include/npu_utils/instr_utils/npu_cmd_wait.hpp create mode 100644 src/detail/include/npu_utils/instr_utils/npu_cmd_write.hpp create mode 100644 src/detail/include/npu_utils/instr_utils/npu_cmd_write_dma.hpp create mode 100644 src/detail/include/npu_utils/npu_instr_utils.hpp create mode 100644 src/detail/include/npu_utils/npu_utils.hpp create mode 100644 src/detail/include/tensor_2d.hpp create mode 100644 src/detail/include/tensor_utils/q4_npu_eXpress.hpp create mode 100644 src/detail/include/tensor_utils/safe_tensors.hpp create mode 100644 src/detail/include/tokenizer/tokenizer.hpp create mode 100644 src/detail/include/tokenizer/tokenizers_c.h create mode 100644 src/detail/include/tokenizer/tokenizers_cpp.h create mode 100644 src/detail/include/typedef.hpp create mode 100644 src/detail/include/utils/avx512_util.hpp create mode 100644 src/detail/include/utils/debug_utils.hpp create mode 100644 src/detail/include/utils/error_measure.hpp create mode 100644 src/detail/include/utils/profiler.hpp create mode 100644 src/detail/include/utils/utils.hpp create mode 100644 src/detail/include/utils/vm_args.hpp create mode 100644 src/detail/include/vision/conv3dPatchEmbedding.hpp create mode 100644 src/detail/include/vision/norm.hpp create mode 100644 src/detail/include/vision/pos_embed_interpolate.hpp create mode 100644 src/detail/include/vision/rot_pos_emb.hpp create mode 100644 src/detail/include/vision/vision_prefill_helper.hpp create mode 100644 src/detail/include/weight_desc.hpp create mode 100644 src/detail/lm_head/lm_head.cpp create mode 100644 src/detail/lm_head/lm_head_detail.hpp create mode 100644 src/detail/vision_common/norm.cpp diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index 48910cc06..9b1a0dea6 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -110,6 +110,12 @@ set(FLM_ENGINE_LIB_DIR "${CMAKE_SOURCE_DIR}/lib/${FLM_RUNTIME_NAME}") # The flash engine libs (qwen3vl_flash, gemma4e_flash) are always shipped in # src/lib/${FLM_RUNTIME_NAME} for every backend, so their model families are # always compiled in -- no probe/gate/macro needed. +# Gemma4e is the one engine built from source here; the rest stay prebuilt. +option(FLM_BUILD_GEMMA4E "Build the Gemma4e engine from src/detail instead of using the prebuilt" ON) +option(FLM_ENGINE_NATIVE_ARCH "Build the from-source engine with -march=native" ON) +set(FLM_ENGINE_VERBOSE 0 CACHE STRING "VERBOSE level for the from-source engine") +set(FLM_ENGINE_DEBUG_LEVEL 0 CACHE STRING "DEBUG_LEVEL for the from-source engine") +set(FLM_OVERRIDE_FLAGS "" CACHE STRING "Extra compile flags for the from-source engine (see include/flm_override.hpp)") # ——————————————————————————————————————————————— # NPU runtime discovery. @@ -487,6 +493,12 @@ if(MSVC) target_link_libraries(flm PUBLIC ${STATIC_LIBS}) endif() +# Defines the gemma4e_npu target, which the link list below then resolves to +# instead of the prebuilt of the same name. +if(FLM_BUILD_GEMMA4E) + add_subdirectory(detail) +endif() + set(FLM_ENGINE_LINK_LIBS q4_npu_eXpress llama_npu @@ -567,6 +579,17 @@ else() endif() endif() +# The source-built engine sits in the build tree rather than lib/, and +# has to precede it: a prebuilt of the same name is still sitting there. +if(FLM_BUILD_GEMMA4E AND NOT WIN32) + get_target_property(_flm_build_rpath flm BUILD_RPATH) + if(NOT _flm_build_rpath) + set(_flm_build_rpath "") + endif() + set_target_properties(flm PROPERTIES + BUILD_RPATH "${CMAKE_BINARY_DIR}/engines;${_flm_build_rpath}") +endif() + if(WIN32 AND VCPKG_TOOLCHAIN) # Local/managed vcpkg: link the imported targets from the CONFIG packages # found above (versioned import-lib names resolved automatically). @@ -723,7 +746,15 @@ else() endif() file(GLOB so_libs "${FLM_ENGINE_LIB_DIR}/*.so*") - install(FILES ${so_libs} DESTINATION "${FLM_ENGINE_LIB_DESTINATION}") + # Installed from the glob unless a source build supersedes one of them. The + # name still has to reach _flm_engine_names below, so filter a copy. + set(_flm_prebuilt_libs ${so_libs}) + if(FLM_BUILD_GEMMA4E) + list(FILTER _flm_prebuilt_libs EXCLUDE REGEX "/libgemma4e_npu\\.so[^/]*$") + set_target_properties(gemma4e_npu PROPERTIES INSTALL_RPATH "${FLM_ENGINE_INSTALL_RPATH}") + install(TARGETS gemma4e_npu LIBRARY DESTINATION "${FLM_ENGINE_LIB_DESTINATION}") + endif() + install(FILES ${_flm_prebuilt_libs} DESTINATION "${FLM_ENGINE_LIB_DESTINATION}") set_target_properties(flm PROPERTIES INSTALL_RPATH "${FLM_FLM_INSTALL_RPATH}") # Engine .so file names, used below to keep them out of the flm dependency diff --git a/src/detail/CMakeLists.txt b/src/detail/CMakeLists.txt new file mode 100644 index 000000000..0f6cd453e --- /dev/null +++ b/src/detail/CMakeLists.txt @@ -0,0 +1,107 @@ +# Builds the Gemma4e engine from source, in place of the prebuilt +# lib//libgemma4e_npu.so that ships for every other model. +# +# The engine does not compile against ../include. It carries its own generation +# of the runtime headers under detail/include, which is what the prebuilt +# engines were built against: ../include lacks weight_desc.hpp entirely and its +# npu_utils.hpp and buffer.hpp differ by several hundred lines. Putting +# detail/include first keeps this target on one consistent set. + +set(GEMMA4E_ENGINE_DIR "${CMAKE_CURRENT_SOURCE_DIR}") + +file(GLOB GEMMA4E_SOURCES "${GEMMA4E_ENGINE_DIR}/gemma4e_npu/*.cpp") +# Every engine compiles its own copy of these four; they also ship as separate +# libraries for flm itself to link. +file(GLOB GEMMA4E_SHARED_SOURCES + "${GEMMA4E_ENGINE_DIR}/dequant/*.cpp" + "${GEMMA4E_ENGINE_DIR}/gemm/*.cpp" + "${GEMMA4E_ENGINE_DIR}/lm_head/*.cpp" +) +list(APPEND GEMMA4E_SHARED_SOURCES "${GEMMA4E_ENGINE_DIR}/vision_common/norm.cpp") + +add_library(gemma4e_npu SHARED ${GEMMA4E_SOURCES} ${GEMMA4E_SHARED_SOURCES}) + +target_include_directories(gemma4e_npu BEFORE PRIVATE + "${GEMMA4E_ENGINE_DIR}/include" +) +if(NOT WIN32) + target_include_directories(gemma4e_npu PRIVATE /opt/xilinx/xrt/include) +else() + target_include_directories(gemma4e_npu PRIVATE ${XRT_INCLUDE_DIR}) +endif() + +target_compile_features(gemma4e_npu PRIVATE cxx_std_20) +target_compile_definitions(gemma4e_npu PRIVATE + VERBOSE=${FLM_ENGINE_VERBOSE} + DEBUG_LEVEL=${FLM_ENGINE_DEBUG_LEVEL} +) + +if(NOT MSVC) + target_compile_options(gemma4e_npu PRIVATE + -O3 -Wall -fmax-errors=1 + -mavx512f -mavx512vl -mavx512bw -mavx512dq -mfma + -ffast-math + ) + # The Makefile builds with -march=native. It bakes the build host's ISA into + # the library, so a redistributable build has to turn it off. + if(FLM_ENGINE_NATIVE_ARCH) + target_compile_options(gemma4e_npu PRIVATE -march=native) + endif() +endif() + +# Point FLM_OVERRIDES at a header that redefines FLM_OVERRIDE to dispatch an +# operator elsewhere; see include/flm_override.hpp. Empty leaves the build +# byte-identical to an unannotated one. +if(FLM_OVERRIDE_FLAGS) + separate_arguments(_gemma4e_override_flags NATIVE_COMMAND "${FLM_OVERRIDE_FLAGS}") + target_compile_options(gemma4e_npu PRIVATE ${_gemma4e_override_flags}) +endif() + +find_package(OpenMP REQUIRED) +target_link_libraries(gemma4e_npu PRIVATE OpenMP::OpenMP_CXX) + +target_link_directories(gemma4e_npu PRIVATE "${FLM_ENGINE_LIB_DIR}") +if(FLM_USE_HRX) + target_link_libraries(gemma4e_npu PRIVATE hrx::hrx) +else() + if(NOT WIN32) + target_link_directories(gemma4e_npu PRIVATE /opt/xilinx/xrt/lib) + target_link_libraries(gemma4e_npu PRIVATE xrt_coreutil aiebu) + else() + target_link_libraries(gemma4e_npu PRIVATE xrt_coreutil aiebu_static) + endif() + # The HRX path links no Boost: no engine source uses it there. + find_package(Boost CONFIG REQUIRED COMPONENTS program_options filesystem) + target_link_libraries(gemma4e_npu PRIVATE + Boost::program_options + Boost::filesystem + ) +endif() + +if(NOT WIN32) + # Each engine .so carries its own copy of the shared runtime (npu_sequence, + # npu_app_manager, weight_desc_t, buffer, ...) at default visibility. flm + # loads every engine at once, so without this flag ELF interposition binds + # references -- including calls from inside this .so -- to whichever engine + # comes first in DT_NEEDED, running another model's implementation. + # Functions only: typeinfo and vtables stay interposable so dynamic_cast + # across the boundary keeps working. + target_link_options(gemma4e_npu PRIVATE -Wl,-Bsymbolic-functions) + # q4nx and mha stay prebuilt; flm resolves them when it links the engine. + target_link_options(gemma4e_npu PRIVATE -Wl,--allow-shlib-undefined) +endif() + +# Built into the build tree, leaving lib//libgemma4e_npu.so untouched +# so FLM_BUILD_GEMMA4E=OFF still has a prebuilt to fall back on. The parent adds +# this directory to flm's BUILD_RPATH and installs this target in place of the +# prebuilt. +set_target_properties(gemma4e_npu PROPERTIES + LIBRARY_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/engines" + RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/engines" +) +foreach(_cfg RELEASE DEBUG RELWITHDEBINFO MINSIZEREL) + set_target_properties(gemma4e_npu PROPERTIES + LIBRARY_OUTPUT_DIRECTORY_${_cfg} "${CMAKE_BINARY_DIR}/engines" + RUNTIME_OUTPUT_DIRECTORY_${_cfg} "${CMAKE_BINARY_DIR}/engines" + ) +endforeach() diff --git a/src/detail/dequant/dequant.cpp b/src/detail/dequant/dequant.cpp new file mode 100644 index 000000000..1a2d51c2d --- /dev/null +++ b/src/detail/dequant/dequant.cpp @@ -0,0 +1,330 @@ +#include "dequant_detail.hpp" + +// constructors +Dequant::Impl::Impl(LM_Config& config) : config(config){ + // Initialize any dequant-specific configuration here +} + +Dequant::Impl::~Impl() = default; + +// methods +/// @brief generate the dequant sequence +/// @param seq: the sequence +/// @param D_in: input dimension of the projection weight +/// @param D_out: output dimension of the projection weight +/// @param weight_offset: the weight offset in byte +/// @param mode: dequant output mode +void Dequant::Impl::generate_dequant_q80_packed_in_q4nx_seq(npu_sequence* seq_ptr, const u32 D_in, const u32 D_out, const u32 weight_offset, dequant_output_mode_t output_mode){ + std::cout << "generate_dequant_q80_packed_in_q4nx_seq, D_in: " << D_in << ", D_out: " << D_out << ", weight_offset: " << weight_offset << std::endl; + if (D_in % k_tile_q4 != 0) { + std::cerr << "D_in % k_tile_q4 != 0" << std::endl; + exit(1); + } + + int bd_wait_counter[8] = {0, 0, 0, 0, 0, 0, 0, 0}; + // although each data block is in mxk block, but the data block could be reorder in col-stride on block view + /* + For example, quant_block_col_stride = 2 means + + //This is the logical view of the data block, each block of m_tile_q4 x k_tile_q4 + [block0, block1, ...... blockD, + blockD+1, blockD+2, ...... + ] + + But in memory order, the data block is arrange as block0, blockD+1, block1, blockD+2 .... + + */ + + if(D_in % desired_k_dequant != 0){ + std::cerr << "D_in % desired_k_dequant != 0" << std::endl; + exit(1); + } + + const uint32_t blocks_per_row = D_in / k_tile_q4 * 2; + std::cout << "blocks_per_row: " << blocks_per_row << std::endl; + + if(D_out % desired_m_dequant != 0 ){ + std::cerr << "D_out % desired_m_dequant != 0" << std::endl; + exit(1); + } + + const int quant_in_per_column = (desired_m_dequant / m_tile_q4) * blocks_per_row * block_size_in_byte_q4_1; + const int total_column_rounds = D_out / (desired_m_dequant); + + const int row_per_round = desired_m_dequant * total_cols; + // down rounds, go though D_out + const int down_rounds = (D_out + row_per_round - 1) / row_per_round; + + npu_sequence& seq = *seq_ptr; + seq.clear_cmds(); + + uint32_t input_offset = weight_offset; + + if(output_mode == dequant_output_mode_t::GATE_MATRIX){ + input_offset += (gate_up_m_interleave_size / m_tile_q4) * blocks_per_row * block_size_in_byte_q4_1; + } + uint32_t gate_up_interleave_counter= 0; + + // first, the dequant of down + for(int i = 0; i < down_rounds; i++){ + for(int col = 0; col < 8; col++){ + uint32_t bd_offset = (i % 2) * 8; + uint32_t round_offset = i * 8 + col; + if(round_offset < total_column_rounds){ + seq.npu_dma_memcpy_nd( + sizeof(char), + qw_in_arg_idx, + MM2S, + IT[col], + (npu_bd_id)(0+bd_offset), + it_channel_0, + {0, 0, 0, input_offset}, + //NOTE: this for now only work if desired_m_dequant == quant_block_col_stride*m_tile_q4 + { + blocks_per_row, + (desired_m_dequant / m_tile_q4) / quant_block_col_stride, + quant_block_col_stride * block_size_in_byte_q4_1 / 512, + 512 + }, + { + quant_block_col_stride * block_size_in_byte_q4_1, + quant_block_col_stride * block_size_in_byte_q4_1 * blocks_per_row, + 512, + 1 + }, + -1 ,0, false + ); + + if(output_mode == dequant_output_mode_t::NORMAL_DEQUANT){ + std::cout << "Use normal output!" << std::endl; + input_offset += quant_in_per_column; + } + else{ + gate_up_interleave_counter++; + input_offset += quant_in_per_column; + if(gate_up_interleave_counter == (gate_up_m_interleave_size / desired_m_dequant) ){ + gate_up_interleave_counter = 0; + input_offset += (gate_up_m_interleave_size / m_tile_q4) * blocks_per_row * block_size_in_byte_q4_1; + } + } + + // Each port receive 2*Q4NX_ROWx D_Q4NX_BLOCK_PER_ROW*Q4NX_COL + uint32_t output_offset_0 = round_offset * desired_m_dequant * D_in; + + seq.npu_dma_memcpy_nd( + sizeof(uint16_t),//bf16 outpout + w_out_arg_idx, + S2MM, + IT[col], + (npu_bd_id)(1+bd_offset), + it_channel_0, + {0, 0, 0, output_offset_0}, + { + (uint32_t)D_in/desired_k_dequant, + desired_k_dequant/k_tile_q4, + desired_m_dequant, + k_tile_q4 + }, + { + desired_m_dequant * desired_k_dequant, + k_tile_q4, + desired_k_dequant, + 1 + }, + -1, 0, true, + aggressive_cache + ); + bd_wait_counter[col]++; + } + } + // note: for now + for(int col = 0; col < 8; col++){ + if(bd_wait_counter[col] == 2){ + seq.npu_dma_wait(IT[col], S2MM, it_channel_0); + bd_wait_counter[col]--; + } + } + } + + for(int col = 0; col < 8; col++){ + while(bd_wait_counter[col] != 0){ + seq.npu_dma_wait(IT[col], S2MM, it_channel_0); + bd_wait_counter[col]--; + } + } + seq.cmds2seq(); +} + +/// @brief generate the dequant sequence +/// @param seq: the sequence +/// @param D_in: input dimension of the projection weight +/// @param D_out: output dimension of the projection weight +/// @param weight_offset: the weight offset in byte +/// @param mode: dequant output mode +void Dequant::Impl::generate_dequant_q4_1_seq(npu_sequence* seq_ptr, const u32 D_in, const u32 D_out, const u32 weight_offset, dequant_output_mode_t output_mode){ + if (D_in % k_tile_q4 != 0) { + std::cerr << "D_in % k_tile_q4 != 0" << std::endl; + exit(1); + } + + int bd_wait_counter[8] = {0, 0, 0, 0, 0, 0, 0, 0}; + // although each data block is in mxk block, but the data block could be reorder in col-stride on block view + /* + For example, quant_block_col_stride = 2 means + + //This is the logical view of the data block, each block of m_tile_q4 x k_tile_q4 + [block0, block1, ...... blockD, + blockD+1, blockD+2, ...... + ] + + But in memory order, the data block is arrange as block0, blockD+1, block1, blockD+2 .... + + */ + + if(D_in % desired_k_dequant != 0){ + std::cerr << "D_in % desired_k_dequant != 0" << std::endl; + exit(1); + } + + const uint32_t blocks_per_row = D_in / k_tile_q4; + + if(D_out % desired_m_dequant != 0 ){ + std::cerr << "D_out % desired_m_dequant != 0" << std::endl; + exit(1); + } + + const int quant_in_per_column = (desired_m_dequant / m_tile_q4) * blocks_per_row * block_size_in_byte_q4_1; + const int total_column_rounds = D_out / (desired_m_dequant); + + const int row_per_round = desired_m_dequant * total_cols; + // down rounds, go though D_out + const int down_rounds = (D_out + row_per_round - 1) / row_per_round; + + npu_sequence& seq = *seq_ptr; + seq.clear_cmds(); + + uint32_t input_offset = weight_offset; + + if(output_mode == dequant_output_mode_t::GATE_MATRIX){ + input_offset += (gate_up_m_interleave_size / m_tile_q4) * blocks_per_row * block_size_in_byte_q4_1; + } + uint32_t gate_up_interleave_counter= 0; + + // first, the dequant of down + for(int i = 0; i < down_rounds; i++){ + for(int col = 0; col < 8; col++){ + uint32_t bd_offset = (i % 2) * 8; + uint32_t round_offset = i * 8 + col; + if(round_offset < total_column_rounds){ + + seq.npu_dma_memcpy_nd( + sizeof(char), + qw_in_arg_idx, + MM2S, + IT[col], + (npu_bd_id)(0+bd_offset), + it_channel_0, + {0, 0, 0, input_offset}, + //NOTE: this for now only work if desired_m_dequant == quant_block_col_stride*m_tile_q4 + { + blocks_per_row, + (desired_m_dequant / m_tile_q4) / quant_block_col_stride, + quant_block_col_stride * block_size_in_byte_q4_1 / 512, + 512 + }, + { + quant_block_col_stride * block_size_in_byte_q4_1, + quant_block_col_stride * block_size_in_byte_q4_1 * blocks_per_row, + 512, + 1 + }, + -1 ,0, false + ); + + if(output_mode == dequant_output_mode_t::NORMAL_DEQUANT){ + input_offset += quant_in_per_column; + } + else{ + gate_up_interleave_counter++; + input_offset += quant_in_per_column; + if(gate_up_interleave_counter == (gate_up_m_interleave_size / desired_m_dequant) ){ + gate_up_interleave_counter = 0; + input_offset += (gate_up_m_interleave_size / m_tile_q4) * blocks_per_row * block_size_in_byte_q4_1; + } + } + + // Each port receive 2*Q4NX_ROWx D_Q4NX_BLOCK_PER_ROW*Q4NX_COL + uint32_t output_offset_0 = round_offset * desired_m_dequant * D_in; + + seq.npu_dma_memcpy_nd( + sizeof(uint16_t),//bf16 outpout + w_out_arg_idx, + S2MM, + IT[col], + (npu_bd_id)(1+bd_offset), + it_channel_0, + {0, 0, 0, output_offset_0}, + { + (uint32_t)D_in/desired_k_dequant, + desired_k_dequant/k_tile_q4, + desired_m_dequant, + k_tile_q4 + }, + { + desired_m_dequant * desired_k_dequant, + k_tile_q4, + desired_k_dequant, + 1 + }, + -1, 0, true, + aggressive_cache + ); + bd_wait_counter[col]++; + } + } + // note: for now + for(int col = 0; col < 8; col++){ + if(bd_wait_counter[col] == 2){ + seq.npu_dma_wait(IT[col], S2MM, it_channel_0); + bd_wait_counter[col]--; + } + } + } + + for(int col = 0; col < 8; col++){ + while(bd_wait_counter[col] != 0){ + seq.npu_dma_wait(IT[col], S2MM, it_channel_0); + bd_wait_counter[col]--; + } + } + seq.cmds2seq(); +} + +// wrappers +Dequant::Dequant(LM_Config& config) : _impl(new Impl(config)){} +Dequant::~Dequant(){ + delete _impl; +} +void Dequant::generate_dequant_q80_packed_in_q4nx_seq(npu_sequence* seq, const u32 D_in, const u32 D_out, const u32 weight_offset, int mode){ + _impl->generate_dequant_q80_packed_in_q4nx_seq(seq, D_in, D_out, weight_offset, (Dequant::Impl::dequant_output_mode_t)mode); +} + +void Dequant::generate_dequant_q4_1_seq(npu_sequence* seq, const u32 D_in, const u32 D_out, const u32 weight_offset, int mode){ + _impl->generate_dequant_q4_1_seq(seq, D_in, D_out, weight_offset, (Dequant::Impl::dequant_output_mode_t)mode); +} + +void Dequant::reorder_cpy( + u8 *dst, buffer &src, + quant_block_t quant_block_type, + const int quant_matrix_row, + const int quant_matrix_col, + const int vertical_blocks , + const int vetrical_block_interleave_byte_size + +){ + + _impl->reorder_cpy( + dst, src, quant_block_type, quant_matrix_row, quant_matrix_col, + vertical_blocks, vetrical_block_interleave_byte_size + ); +} diff --git a/src/detail/dequant/dequant_detail.hpp b/src/detail/dequant/dequant_detail.hpp new file mode 100644 index 000000000..cdef9f1c0 --- /dev/null +++ b/src/detail/dequant/dequant_detail.hpp @@ -0,0 +1,175 @@ +#pragma once +#include "modules/dequant.hpp" + +struct Dequant::Impl{ +private: + static constexpr npu_tiles IT[] = {IT0, IT1, IT2, IT3, IT4, IT5, IT6, IT7}; + + static constexpr u32 total_cols = 8; + static constexpr u32 total_rows = 4; + + static constexpr int w_out_arg_idx = 0; + static constexpr int qw_in_arg_idx = 1; + + static constexpr int m_tile_q4 = 32; + static constexpr int k_tile_q4 = 256; + + static constexpr uint32_t block_size_in_byte_q4_0 = ((m_tile_q4 * k_tile_q4 * 4.5) / 8.0); + static constexpr uint32_t block_size_in_byte_q4_1 = m_tile_q4 * k_tile_q4 * 5 / 8; + + static constexpr int m_tile_q8 = 32; + static constexpr int k_tile_q8 = 256; + + static constexpr uint32_t block_size_in_byte_q8_0 = (m_tile_q8 * k_tile_q8 * (8.5) )/8.0; + static constexpr uint32_t block_size_in_byte_q8_1 = (m_tile_q8 * k_tile_q8 * (9) )/8.0; + static constexpr int quant_block_col_stride = 2; + static constexpr int quant_block_interleave_byte_size = 512; + + static constexpr int desired_k_dequant = 512; + static constexpr int desired_m_dequant = 128; + + static constexpr int glu_slice = 1024; + static constexpr int gate_up_m_interleave_size = glu_slice / 2; + +public: + /// @brief dequant output mode, as the quantized weight of UP and GATE are interleaved in memory, now we want to seperate them. + /// @note NORMAL_DEQUANT: normal dequant output + /// @note UP_MATRIX: up projection matrix output + /// @note GATE_MATRIX: gate projection matrix output + typedef enum: int{ + NORMAL_DEQUANT = 0, + UP_MATRIX = 1, + GATE_MATRIX = 2 + } dequant_output_mode_t; + + Impl(){} + Impl(LM_Config& config); + ~Impl(); + /// @brief generate the dequant sequence + /// @param seq: the sequence + /// @param D_in: input dimension of the projection weight + /// @param D_out: output dimension of the projection weight + /// @param weight_offset: the weight offset in byte + /// @param mode: dequant output mode + void generate_dequant_q4_1_seq(npu_sequence* seq, const u32 D_in, const u32 D_out, const u32 weight_offset, dequant_output_mode_t mode); + void generate_dequant_q80_packed_in_q4nx_seq(npu_sequence* seq, const u32 D_in, const u32 D_out, const u32 weight_offset, dequant_output_mode_t mode); + LM_Config config; + + void reorder_cpy(u8 *dst, buffer &src, + Dequant::quant_block_t quant_block_type, + const int quant_matrix_row, + const int quant_matrix_col, + const int vertical_blocks, + const int vetrical_block_interleave_byte_size) + { + int a_block_size = 0; + int block_col_size = 0; + int block_row_size = 0; + switch(quant_block_type){ + case Q4_1: + a_block_size = block_size_in_byte_q4_1; + block_col_size = k_tile_q4; + block_row_size = m_tile_q4; + break; + case Q8_0: + a_block_size= block_size_in_byte_q8_0; + block_col_size = k_tile_q8; + block_row_size = m_tile_q8; + break; + default: + std::cerr << "Unsupport type for now"; + exit(-1); + break; + } + + assert( quant_matrix_col % block_col_size== 0); + const int blocks_per_row = quant_matrix_col / block_col_size; + + assert(quant_matrix_row%(block_row_size* vertical_blocks) == 0 ); + + const int rows = src.size() / a_block_size / blocks_per_row; + + u8 *dst_ptr = dst; + std::vector src_ptr(vertical_blocks); + for (int i = 0; i < vertical_blocks; i++) + { + src_ptr[i] = src.data() + i * a_block_size * blocks_per_row; + } + for (int r = 0; r < rows; r += vertical_blocks) + { + for (int c = 0; c < blocks_per_row; c++) + { + for (int i = 0; i < vertical_blocks; i++) + { + memcpy(dst_ptr, src_ptr[i], a_block_size); + dst_ptr += a_block_size; + src_ptr[i] += a_block_size; + } + } + for (int i = 0; i < vertical_blocks; i++) + { + src_ptr[i] += (vertical_blocks - 1) * a_block_size * blocks_per_row; + if (src_ptr[i] + a_block_size * blocks_per_row > src.end()) + { + src_ptr[i] = src.data(); // useless padding + } + } + } + + // At this step, the blocks are now reorder with vertical blocks + + // For example + // IF previous are row-block order + /* + [A, B, C + D, E, F] + Where blocks are ordered as A, B, C, D, E, F + + With the vertical_blocks =2, + blocks are reorder as A, D, B, E, C, F + + */ + + if(vetrical_block_interleave_byte_size <=0){ + return ; // no need this step + } + // Apply vetrical_block_interleave_byte_size reorder + + // From example above, now A, D blocks are continousy in memory at block level + + // However, we want to do a byte-block level mixing + /** + For example, If A, D block are Block size of 2K and vetrical_block_interleave_byte_size = 1024 + + In memory, the data are layout as A(1-1204) A(1025-2048), D(1-1024), D(1025-2048) + + After the block level reorder, we have + A(1-1204), D(1-1024), A(1025-2048), D(1025-2048) + + */ + size_t num_data_block = (quant_matrix_row/block_row_size) * (quant_matrix_col/block_col_size); + std::vector temp_buffer(vertical_blocks*a_block_size ); + + size_t num_byte_data_chunk = (a_block_size) / vetrical_block_interleave_byte_size; + assert(a_block_size % vetrical_block_interleave_byte_size == 0); + + for(int i = 0; i < num_data_block; i+= vertical_blocks){ + uint8_t* cur_ptr = dst + i*a_block_size; + memcpy( temp_buffer.data(), cur_ptr, temp_buffer.size() ); + + uint8_t* chunk_dst_ptr = cur_ptr; + + for(int byte_chunk_idx = 0; byte_chunk_idx +#include // Required for std::max +#include + +// // constructors +Gemm::Impl::Impl() { + // mapping of valid shimtile index for A -> index offset + valid_A_MT_shimtile_index[0] = 0; + valid_A_MT_shimtile_index[2] = 1; + valid_A_MT_shimtile_index[4] = 2; + valid_A_MT_shimtile_index[6] = 3; +} + +Gemm::Impl::~Impl() = default; + +/// \brief Generate the sequence +/// \param seq the npu sequence +/// \param M the M dimension +/// \param K the K dimension +/// \param N the N dimension +/// \param weight_offset the weight offset +/// \param ADD_BIAS whether to add bias +/// \param OUTPUT_MODE the output activation mode +/// \param bias_offset the bias offset +void Gemm::Impl::generate_seq( + npu_sequence *seq, + uint32_t M, + uint32_t K, + uint32_t N, + const uint32_t weight_offset, + bool ADD_BIAS, + Activation_Type_t OUTPUT_MODE, + const uint32_t bias_offset, + uint32_t C_const_offset +){ + uint32_t A_const_offset = 0; + uint32_t B_const_offset = weight_offset; + + const int K_div_k = K/k; + + // some sanity checks + if (M % (m * total_rows) != 0) { + std::cerr << "GEMM M size not aligned with total npu rows"<< std::endl; + exit(1); + } + if (K % k != 0) { + std::cerr << "GEMM K size not aligned with k"<< std::endl; + exit(1); + } + if (N % n != 0) { + std::cerr << "GEMM N size not aligned with n"<< std::endl; + exit(1); + } + + seq->clear_cmds(); + + std::vector list_C_shim_queue; // int counter of how many DMA_Wait for C + std::vector list_A_shim_queue; // int counter of how many DMA_Wait for A + std::vector list_B_shim_queue; // int counter of how many DMA_Wait for B + for(size_t i = 0; i < shimtile_size; i++){ + list_C_shim_queue.push_back(0); + list_A_shim_queue.push_back(0); + list_B_shim_queue.push_back(0); + } + + // first, setup the rtp buffer and the rtp locks + for(size_t row_idx = 0; row_idx < total_rows; row_idx++){ + for(size_t col_idx = 0; col_idx< total_cols; col_idx++){ + auto CT_tile = get_tile(row_idx + 2, col_idx); + // set RTP value + seq->rtp_write(CT_tile, CT_rtp_address, K_div_k); + seq->rtp_write(CT_tile, CT_rtp_address + 4, M); + seq->rtp_write(CT_tile, CT_rtp_address + 8, N); + if(ADD_BIAS){ + seq->rtp_write( CT_tile, CT_rtp_address + 12, 1 ); + }else{ + seq->rtp_write( CT_tile, CT_rtp_address + 12, 0 ); + } + seq->rtp_write( CT_tile, CT_rtp_address + 16, OUTPUT_MODE ); // OUTPUT MODE + // set RTP lock, enable running + seq->rtp_write(CT_tile, CT_lock_address_base + 16 * CT_rtp_sync_lock_id, 1); // set lock to 1 + } + } + + generate_runtime_sequence( + seq, + A_const_offset, B_const_offset, C_const_offset, bias_offset, + M, N, K, + list_A_shim_queue, list_B_shim_queue, + list_C_shim_queue, + IS_B_ROW_MAJOR, ENABLE_AXI4, B_in_K_N_block_col_major_order, + ADD_BIAS + ); + + int max_C_remain = 0; + for( auto li: list_C_shim_queue){ + max_C_remain = std::max(max_C_remain, li); + } + + for(size_t k = 0; k< max_C_remain; k++){ + for(size_t shim_index = 0; shim_index < total_cols; shim_index++){ + + if(list_A_shim_queue.at(shim_index) > 0){ + seq->npu_dma_wait( + shim_tiles[shim_index], + MM2S, + it_channel_0 + ); + list_A_shim_queue.at(shim_index)--; + } + if(list_B_shim_queue.at(shim_index) > 0){ + seq->npu_dma_wait( + shim_tiles[shim_index], + MM2S, + it_channel_1 + ); + list_B_shim_queue.at(shim_index)--; + } + if(list_C_shim_queue.at(shim_index) > 0){ + seq->npu_dma_wait( + shim_tiles[shim_index], + S2MM, + it_channel_0 + + ); + list_C_shim_queue.at(shim_index)--; + } + } + } + + seq->cmds2seq(); +} + +template +void Gemm::Impl::generate_runtime_sequence( + npu_sequence* seq, + uint32_t A_const_offset, uint32_t B_const_offset, uint32_t C_const_offset, uint32_t Bias_const_offset, + uint32_t M_size, uint32_t N_size, uint32_t K_size, + std::vector &list_A_shim_queue, + std::vector &list_B_shim_queue, + std::vector &list_C_shim_queue, + bool IS_B_ROW_MAJOR, + bool ENABLE_AXI4, + bool B_in_K_N_block_col_major_order, + bool ADD_BIAS +){ + uint32_t M_div_num_row_m = M_size/(m * total_rows); + uint32_t N_div_num_col_n = N_size/(n * total_cols); + + uint32_t N_div_num_col_n_remainder_blocks = (N_size % (n * total_cols)) / n; + + std::vector list_A_BD_pingpong_flag(shimtile_size, 0); + std::vector list_B_BD_pingpong_flag(shimtile_size, 0); + std::vector list_C_BD_pingpong_flag(shimtile_size, 0); + + uint32_t col_block_range = N_div_num_col_n; + if (N_div_num_col_n_remainder_blocks!= 0){ + col_block_range += 1; + } + + for(uint32_t mega_block_col_idx = 0; mega_block_col_idx < col_block_range; mega_block_col_idx++){ + for (uint32_t mega_block_row_idx = 0; mega_block_row_idx < M_div_num_row_m; mega_block_row_idx++) { + for (uint32_t shim_index = 0; shim_index < shimtile_size; shim_index++) { + bool SEND_ADD_BIAS = false; + + if (mega_block_row_idx == 0 &&ADD_BIAS){ + SEND_ADD_BIAS = true; + } + + if (N_div_num_col_n_remainder_blocks!= 0 && mega_block_col_idx == N_div_num_col_n){ + if (shim_index < N_div_num_col_n_remainder_blocks){ + generate_shimtile_sequence_per_k_block( + seq, + shim_index, + mega_block_row_idx, mega_block_col_idx, + M_size, K_size, N_size, + A_const_offset, B_const_offset, C_const_offset, Bias_const_offset, + list_A_shim_queue, list_B_shim_queue, list_C_shim_queue, + list_A_BD_pingpong_flag, list_B_BD_pingpong_flag, list_C_BD_pingpong_flag, + IS_B_ROW_MAJOR, ENABLE_AXI4, + B_in_K_N_block_col_major_order, + true, + ADD_BIAS, SEND_ADD_BIAS + ); + } + + else{ + generate_shimtile_sequence_per_k_block( + seq, + shim_index, + mega_block_row_idx, mega_block_col_idx, + M_size, K_size, N_size, + A_const_offset, B_const_offset, C_const_offset, Bias_const_offset, + list_A_shim_queue, list_B_shim_queue, list_C_shim_queue, + list_A_BD_pingpong_flag, list_B_BD_pingpong_flag, list_C_BD_pingpong_flag, + IS_B_ROW_MAJOR, ENABLE_AXI4, + B_in_K_N_block_col_major_order, + false, + ADD_BIAS, SEND_ADD_BIAS + ); + } + } + else{ + generate_shimtile_sequence_per_k_block( + seq, + shim_index, + mega_block_row_idx, mega_block_col_idx, + M_size, K_size, N_size, + A_const_offset, B_const_offset, C_const_offset, Bias_const_offset, + list_A_shim_queue, list_B_shim_queue, list_C_shim_queue, + list_A_BD_pingpong_flag, list_B_BD_pingpong_flag, list_C_BD_pingpong_flag, + IS_B_ROW_MAJOR, ENABLE_AXI4, + B_in_K_N_block_col_major_order, + true, + ADD_BIAS, SEND_ADD_BIAS + ); + } + } + } + } +} + +template +void Gemm::Impl::generate_shimtile_sequence_per_k_block( + npu_sequence*seq, + uint32_t shim_index, + uint32_t mega_block_row_idx, uint32_t mega_block_col_idx, + uint32_t M_size, uint32_t K_size, uint32_t N_size, + uint32_t A_const_offset, uint32_t B_const_offset, uint32_t C_const_offset, uint32_t Bias_const_offset, + std::vector &list_A_shim_queue, std::vector &list_B_shim_queue, std::vector &list_C_shim_queue, + std::vector &list_A_bd_pingpong_flag, std::vector &list_B_bd_pingpong_flag, std::vector &list_C_bd_pingpong_flag, + bool IS_B_ROW_MAJOR, bool ENABLE_AXI4, + bool B_in_K_N_block_col_major_order, + bool VALID_COLUMN, + bool ADD_BIAS, bool SEND_BIAS +){ + + if(B_in_K_N_block_col_major_order){ + assert(IS_B_ROW_MAJOR == false); // on valid for B in col major order + if (IS_B_ROW_MAJOR){ + std::cerr << "Error: When B_in_K_N_block_col_major_order is set to true, IS_B_ROW_MAJOR cannot be true." << std::endl; + exit(1); + } + } + + // When B_in_K_N_block_col_major_order is set to true, it mean + // B is col-major order && + // B is rearrange into kxn blocks, where blocks are in col-major. Moreover, the data in each blocks is + // also in col-major order. + + // Basically,B as a col-major matrix goes through + // stride: [N_size/n,K_size/k ,n, k] + // offset: [K_size*n,k ,K_SIZE, 1] + + uint32_t K_div_k = K_size/k; + + npu_tiles cur_shimtile = shim_tiles[shim_index]; + + if (valid_A_MT_shimtile_index.contains(shim_index) && valid_A_MT_shimtile_index[shim_index] < total_rows){ + + if (list_A_shim_queue.at(shim_index) == 2) { + seq->npu_dma_wait( + cur_shimtile, MM2S, it_channel_0 + ); + list_A_shim_queue.at(shim_index)--; + } + + uint32_t A_offset = mega_block_row_idx * (total_rows * m) * K_size; + A_offset += valid_A_MT_shimtile_index[shim_index] * (m * K_size); + npu_bd_id A_bd_id; + if (list_A_bd_pingpong_flag.at(shim_index) ==0){ + A_bd_id = bd_0; + list_A_bd_pingpong_flag.at(shim_index) =1; + }else{ + A_bd_id = bd_1; + list_A_bd_pingpong_flag.at(shim_index) =0; + } + + seq->npu_dma_memcpy_nd( + sizeof(T_in), // bfloat16 + Arg_A, + MM2S, + cur_shimtile, + A_bd_id, + it_channel_0, + {0,0,0,A_offset+ A_const_offset}, + {1, K_div_k, m,k}, + {0, k, K_size, 1}, + -1, 0, true, + ENABLE_AXI4 ? aggressive_cache : normal_cache + ); + + list_A_shim_queue.at(shim_index)++; + } + + if((shim_index < total_cols) && VALID_COLUMN){ + if(SEND_BIAS){ + if (list_B_shim_queue[shim_index] == 2){ + + seq->npu_dma_wait( + cur_shimtile, MM2S, it_channel_1 + ); + list_B_shim_queue[shim_index] -= 1; + } + uint32_t _BIAS_DATA_OFFSET = Bias_const_offset + mega_block_col_idx * (total_cols * n) + shim_index *n; + seq->npu_dma_memcpy_nd( + sizeof(T_in), + Arg_Bias, + MM2S, + cur_shimtile, + npu_bd_id(bd_6), //reserved for sending bias + it_channel_1, + {0,0,0,_BIAS_DATA_OFFSET}, + {1, 1,1, k*n}, + {0, 0, 0, 1}, + -1, 0, true, + ENABLE_AXI4 ? aggressive_cache : normal_cache + ); + list_B_shim_queue[shim_index]++; + } + + npu_bd_id b_bd_id; + if (list_B_shim_queue[shim_index] == 2){ + seq->npu_dma_wait( + cur_shimtile, MM2S, it_channel_1 + ); + list_B_shim_queue[shim_index] -= 1; + } + + if (list_B_bd_pingpong_flag[shim_index] == 0) { + b_bd_id = bd_2; + list_B_bd_pingpong_flag[shim_index] = 1; + } else { + b_bd_id = bd_3; + list_B_bd_pingpong_flag[shim_index] = 0; + } + + if (IS_B_ROW_MAJOR){ + uint32_t B_offset = mega_block_col_idx* (total_cols) * n; + B_offset += shim_index * n; + seq->npu_dma_memcpy_nd( + sizeof(T_in), + Arg_B, + MM2S, + cur_shimtile, + b_bd_id, + it_channel_1, + {0,0,0,B_offset+ B_const_offset }, + {1, K_div_k, k, n}, + {0, k*N_size, N_size, 1}, + -1, 0, true, + ENABLE_AXI4 ? aggressive_cache : normal_cache + ); + }else{ + uint32_t B_offset = mega_block_col_idx * (total_cols * n) * K_size; + B_offset += shim_index * n * K_size; + if(B_in_K_N_block_col_major_order){ + seq->npu_dma_memcpy_nd( + sizeof(T_in), + Arg_B, + MM2S, + cur_shimtile, + b_bd_id, + it_channel_1, + {0,0,0,B_offset+ B_const_offset }, + {1, 1,1, K_div_k* n*k}, + {0, 0, 0, 1}, + -1, 0, true, + ENABLE_AXI4 ? aggressive_cache : normal_cache + ); + }else{ + seq->npu_dma_memcpy_nd( + sizeof(T_in), + Arg_B, + MM2S, + cur_shimtile, + b_bd_id, + it_channel_1, + {0,0,0,B_offset+ B_const_offset }, + {1, K_div_k, n, k}, + {0, k, K_size, 1}, + -1, 0, true, + ENABLE_AXI4 ? aggressive_cache : normal_cache + ); + } + } + list_B_shim_queue[shim_index]++; + } + + uint32_t C_offset = mega_block_col_idx * n * total_cols; + C_offset += mega_block_row_idx * m * total_rows * N_size; + C_offset += shim_index * n; + + if (shim_index < total_cols && VALID_COLUMN){ + + if (list_C_shim_queue.at(shim_index) == 2) { + seq->npu_dma_wait( + cur_shimtile, S2MM, it_channel_0 + ); + list_C_shim_queue.at(shim_index)--; + } + + npu_bd_id c_bd_id; + if(list_C_bd_pingpong_flag.at(shim_index) ==0){ + c_bd_id = bd_14; + list_C_bd_pingpong_flag.at(shim_index) =1; + }else{ + c_bd_id = bd_15; + list_C_bd_pingpong_flag.at(shim_index) =0; + } + + seq->npu_dma_memcpy_nd( + sizeof(T_out), + Arg_C, + S2MM, + cur_shimtile, + c_bd_id, + it_channel_0, + {0,0,0, C_offset+ C_const_offset}, + {1,1,4*m, n}, + {0,0,N_size, 1}, + -1, 0, true, + normal_cache + ); + list_C_shim_queue.at(shim_index)++; + } +} + +// wrappers +Gemm::Gemm(LM_Config& config) : _impl(new Impl()){} +Gemm::~Gemm(){ + delete _impl; +} + +uint32_t Gemm::get_m() const{ + return _impl->m; +} +uint32_t Gemm::get_k() const{ + return _impl->k; +} +uint32_t Gemm::get_n() const{ + return _impl->n; +} + +void Gemm::generate_seq(npu_sequence* seq, const uint32_t M, const uint32_t K, const uint32_t N, const uint32_t weight_offset, bool ADD_BIAS, Activation_Type_t OUTPUT_MODE, const uint32_t bias_offset){ + _impl->generate_seq(seq, M, K, N, weight_offset, ADD_BIAS, OUTPUT_MODE, bias_offset, 0); +} + +void Gemm::generate_seq(npu_sequence* seq, const uint32_t M, const uint32_t K, const uint32_t N, const uint32_t weight_offset, bool ADD_BIAS, Activation_Type_t OUTPUT_MODE, const uint32_t bias_offset, + const uint32_t output_offset +){ + _impl->generate_seq(seq, M, K, N, weight_offset, ADD_BIAS, OUTPUT_MODE, bias_offset, output_offset); +} diff --git a/src/detail/gemm/gemm_detail.hpp b/src/detail/gemm/gemm_detail.hpp new file mode 100644 index 000000000..01f5aab11 --- /dev/null +++ b/src/detail/gemm/gemm_detail.hpp @@ -0,0 +1,94 @@ +#ifndef __gemm_detail__ +#define __gemm_detail__ +#include "modules/gemm.hpp" + +struct Gemm::Impl +{ +private: + static constexpr int Arg_C = 0; + static constexpr int Arg_A = 1; + static constexpr int Arg_B = 2; + static constexpr int Arg_Bias = 3; + static constexpr int CT_lock_address_base = 0x000001F000; + + static constexpr int mm_y_group_id = 3; + static constexpr int mm_x_group_id = 4; + static constexpr int mm_w_group_id = 5; + static constexpr npu_tiles shim_tiles[] = {IT0, IT1, IT2, IT3, IT4, IT5, IT6, IT7}; + + static constexpr uint32_t total_cols = 8; + static constexpr uint32_t total_rows = 4; + + static constexpr uint32_t shimtile_size = total_cols > total_rows ? total_cols : total_rows; // max of the two + + static constexpr uint32_t CT_rtp_address = 4096; // for 128 + + static constexpr int CT_rtp_sync_lock_id = 10; // for now hard coded + + std::map valid_A_MT_shimtile_index; + +public: + static constexpr uint32_t m = 64; + static constexpr uint32_t k = 512; + static constexpr uint32_t n = 128; // for now + static constexpr uint32_t r = 8; + static constexpr uint32_t s = 8; + static constexpr uint32_t t = 8; + bool IS_B_ROW_MAJOR = false; // for language model, B is always in col-major order + bool B_in_K_N_block_col_major_order = true; // for language model, B is default in kxn block col-major order + bool ENABLE_AXI4 = true; // for language model, B is always in kxn block col-major order + + Impl(); + ~Impl(); + + /// \brief Generate the sequence + /// \param seq the npu sequence + /// \param M the M dimension + /// \param K the K dimension + /// \param N the N dimension + /// \param weight_offset the weight offset + /// \param ADD_BIAS whether to add bias + /// \param OUTPUT_MODE the output activation mode + /// \param bias_offset the bias offset + void generate_seq( + npu_sequence *seq, + uint32_t M, + uint32_t K, + uint32_t N, + const uint32_t weight_offset, + bool ADD_BIAS, + Activation_Type_t OUTPUT_MODE, + const uint32_t bias_offset, + uint32_t C_const_offset + ); + + template + void generate_runtime_sequence( + npu_sequence* seq, + uint32_t A_const_offset, uint32_t B_const_offset, uint32_t C_const_offset, uint32_t Bias_const_offset, + uint32_t M_size, uint32_t N_size, uint32_t K_size, + std::vector &list_A_shim_queue, + std::vector &list_B_shim_queue, + std::vector &list_C_shim_queue, + bool IS_B_ROW_MAJOR, + bool ENABLE_AXI4, + bool B_in_K_N_block_col_major_order, + bool ADD_BIAS + ); + + template + void generate_shimtile_sequence_per_k_block( + npu_sequence*seq, + uint32_t shim_index, + uint32_t mega_block_row_idx, uint32_t mega_block_col_idx, + uint32_t M_size, uint32_t K_size, uint32_t N_size, + uint32_t A_const_offset, uint32_t B_const_offset, uint32_t C_const_offset, uint32_t Bias_const_offset, + std::vector &list_A_shim_queue, std::vector &list_B_shim_queue, std::vector &list_C_shim_queue, + std::vector &list_A_bd_pingpong_flag, std::vector &list_B_bd_pingpong_flag, std::vector &list_C_bd_pingpong_flag, + bool IS_B_ROW_MAJOR, bool ENABLE_AXI4, + bool B_in_K_N_block_col_major_order, + bool VALID_COLUMN, + bool ADD_BIAS, bool SEND_BIAS + ); +}; +#endif diff --git a/src/detail/gemma4e_npu/avx512_util.hpp b/src/detail/gemma4e_npu/avx512_util.hpp new file mode 100644 index 000000000..60a18ac57 --- /dev/null +++ b/src/detail/gemma4e_npu/avx512_util.hpp @@ -0,0 +1,296 @@ +#pragma once +#include +#include "typedef.hpp" +#include +#include +#include + +// Maximum number of threads for prefill/encode SIMD operations +constexpr int max_prefill_threads = 1; + +/** + * @brief Helper function to load 16 bfloat16 values and convert to __m512 (float). + */ +inline __m512 load_bfloat16_to_m512(const bf16* ptr) { + // Load 16 bfloat16 values (32 bytes) into a __m256i + __m256i bf16_data = _mm256_loadu_si256(reinterpret_cast(ptr)); + + // Convert bfloat16 to float by shifting left 16 bits (bfloat16 is upper 16 bits of float) + __m512i shifted = _mm512_cvtepu16_epi32(bf16_data); + shifted = _mm512_slli_epi32(shifted, 16); + + return _mm512_castsi512_ps(shifted); +} + +/** + * @brief Helper function to store __m512 (float) as 16 bfloat16 values. + * Uses truncation (no rounding). + */ +inline void store_m512_to_bfloat16(bf16* ptr, __m512 data) { + // Convert float to bfloat16 by extracting upper 16 bits + __m512i int_data = _mm512_castps_si512(data); + __m512i shifted = _mm512_srli_epi32(int_data, 16); + __m256i bf16_data = _mm512_cvtepi32_epi16(shifted); + + _mm256_storeu_si256(reinterpret_cast<__m256i*>(ptr), bf16_data); +} + +/** + * @brief Helper function to store __m512 (float) as 16 bfloat16 values with rounding. + * Uses round-to-nearest-even for better accuracy. + */ +inline void store_m512_to_bfloat16_rne(bf16* ptr, __m512 data) { + // Convert float to bfloat16 with rounding to nearest even + __m512i int_data = _mm512_castps_si512(data); + + // Add 0x7FFF for round-to-nearest-even + __m512i rounding = _mm512_set1_epi32(0x7FFF); + __m512i rounded = _mm512_add_epi32(int_data, rounding); + + // Shift right by 16 to get bf16 in lower 16 bits + __m512i shifted = _mm512_srli_epi32(rounded, 16); + + // Pack to 16-bit values + __m256i bf16_data = _mm512_cvtepi32_epi16(shifted); + + _mm256_storeu_si256(reinterpret_cast<__m256i*>(ptr), bf16_data); +} + +// Fast, corrected AVX-512 exp approximation (single-precision). +// Notes: +// - Input x is clamped to [-88, 88] to avoid overflow/underflow. +// - Uses range reduction x = n*ln2 + r, where n is rounded to nearest int. +// - Uses a degree-5 polynomial for exp(r) evaluated with Horner + FMAs. +// - Constructs 2^n by writing the biased exponent field; the biased exponent +// is clamped to [0,255] as a safety measure. +// +// This is an approximation (not fully IEEE-754 accurate for all cases). +inline __m512 _mm512_exp_ps_corrected(__m512 x) { + // clamp x to a reasonable range to avoid overflow/underflow + const __m512 max_val = _mm512_set1_ps(88.0f); + const __m512 min_val = _mm512_set1_ps(-88.0f); + x = _mm512_min_ps(x, max_val); + x = _mm512_max_ps(x, min_val); + + // constants: 1/ln2 and split ln2 = ln2_hi + ln2_lo for extra precision + const __m512 ln2_inv = _mm512_set1_ps(1.44269504088896341f); // 1/ln(2) + const __m512 ln2_hi = _mm512_set1_ps(0.6931471824645996f); // hi part + const __m512 ln2_lo = _mm512_set1_ps(1.9082149292705877e-10f);// lo part + + // compute fx = x * (1/ln2) + __m512 fx = _mm512_mul_ps(x, ln2_inv); + + // round to nearest integer (using rounding intrinsic), storing integer-valued floats + fx = _mm512_roundscale_ps(fx, _MM_FROUND_TO_NEAREST_INT | _MM_FROUND_NO_EXC); + + // convert to int32 (safe since fx holds integer values after rounding) + __m512i emm0 = _mm512_cvttps_epi32(fx); + + // convert back to float for range-reduction arithmetic + __m512 n_ps = _mm512_cvtepi32_ps(emm0); + + // r = x - n * ln2 (use fnmadd to compute c - a*b robustly) + // first r1 = x - n*ln2_hi + __m512 r = _mm512_fnmadd_ps(n_ps, ln2_hi, x); // r = x - n*ln2_hi + // then r = r - n*ln2_lo + r = _mm512_fnmadd_ps(n_ps, ln2_lo, r); // r = x - n*(ln2_hi + ln2_lo) + + // polynomial coefficients for exp(r) ~ 1 + r + r^2/2 + r^3/6 + r^4/24 + r^5/120 + const __m512 c5 = _mm512_set1_ps(0.008333333333333333f); // 1/120 + const __m512 c4 = _mm512_set1_ps(0.041666666666666664f); // 1/24 + const __m512 c3 = _mm512_set1_ps(0.16666666666666666f); // 1/6 + const __m512 c2 = _mm512_set1_ps(0.5f); // 1/2 + const __m512 c1 = _mm512_set1_ps(1.0f); + const __m512 one = _mm512_set1_ps(1.0f); + + // Horner evaluation using FMA: (((c5*r + c4)*r + c3)*r + c2)*r + c1 ; then final *r + 1 + __m512 y = _mm512_fmadd_ps(c5, r, c4); + y = _mm512_fmadd_ps(y, r, c3); + y = _mm512_fmadd_ps(y, r, c2); + y = _mm512_fmadd_ps(y, r, c1); + y = _mm512_fmadd_ps(y, r, one); // y now approximates exp(r) + + // Build 2^n by inserting biased exponent into float bits: + // biased = n + 127 + __m512i biased = _mm512_add_epi32(emm0, _mm512_set1_epi32(127)); + + // clamp biased exponent to [0,255] to avoid invalid bit patterns + biased = _mm512_max_epi32(biased, _mm512_set1_epi32(0)); + biased = _mm512_min_epi32(biased, _mm512_set1_epi32(255)); + + // shift into exponent position (bits 23..30) and reinterpret as float + biased = _mm512_slli_epi32(biased, 23); + __m512 pow2n = _mm512_castsi512_ps(biased); + + // final result: exp(x) ≈ exp(r) * 2^n + return _mm512_mul_ps(y, pow2n); +} + +// Fast AVX-512 log approximation (single-precision). +// Input x must be strictly positive. +inline __m512 _mm512_log_ps_approx(__m512 x) { + const __m512i inv_mant_mask = _mm512_set1_epi32(~0x7f800000); + const __m512i min_norm_pos = _mm512_set1_epi32(0x00800000); + const __m512i exponent_mask = _mm512_set1_epi32(0x7f800000); + const __m512 one = _mm512_set1_ps(1.0f); + + // Extract exponent + __m512i vx = _mm512_castps_si512(x); + __m512i emm0 = _mm512_srli_epi32(vx, 23); + emm0 = _mm512_sub_epi32(emm0, _mm512_set1_epi32(127)); + __m512 e = _mm512_cvtepi32_ps(emm0); + + // Extract mantissa and force exponent to 0 (which means range [1.0, 2.0)) + __m512i m_bits = _mm512_and_si512(vx, inv_mant_mask); + m_bits = _mm512_or_si512(m_bits, _mm512_set1_epi32(0x3f800000)); + __m512 m = _mm512_castsi512_ps(m_bits); + + // Map m from [1, 2) to a symmetric range using p = (m - 1) / (m + 1) + __m512 p1 = _mm512_sub_ps(m, one); + __m512 p2 = _mm512_add_ps(m, one); + __m512 p = _mm512_div_ps(p1, p2); + __m512 p_sq = _mm512_mul_ps(p, p); + + // Evaluate Taylor series for log((1+p)/(1-p)) = 2 * (p + p^3/3 + p^5/5 + p^7/7) + const __m512 c7 = _mm512_set1_ps(2.0f / 7.0f); + const __m512 c5 = _mm512_set1_ps(2.0f / 5.0f); + const __m512 c3 = _mm512_set1_ps(2.0f / 3.0f); + const __m512 c1 = _mm512_set1_ps(2.0f); + + __m512 res = _mm512_fmadd_ps(c7, p_sq, c5); + res = _mm512_fmadd_ps(res, p_sq, c3); + res = _mm512_fmadd_ps(res, p_sq, c1); + res = _mm512_mul_ps(res, p); + + // log(x) = res + e * ln(2) + const __m512 ln2 = _mm512_set1_ps(0.6931471805599453f); + return _mm512_fmadd_ps(e, ln2, res); +} + +// AVX-512 GELU tanh-based, now using the corrected exp function +inline __m512 gelu_tanh_avx512_simd(__m512 gate_vec_fp32) { + // ---- GELU(gate) with tanh approximation: 0.5 * gate * (1 + tanh(sqrt(2/pi) * (gate + 0.044715 * gate^3))) ---- + const __m512 half = _mm512_set1_ps(0.5f); + const __m512 one = _mm512_set1_ps(1.0f); + const __m512 sqrt_2_pi = _mm512_set1_ps(0.7978845608f); // sqrt(2/pi) + const __m512 coeff = _mm512_set1_ps(0.044715f); + + // Compute gate^3 + __m512 gate_squared = _mm512_mul_ps(gate_vec_fp32, gate_vec_fp32); + __m512 gate_cubed = _mm512_mul_ps(gate_squared, gate_vec_fp32); + + // Compute gate + 0.044715 * gate^3 + __m512 inner_term = _mm512_fmadd_ps(coeff, gate_cubed, gate_vec_fp32); + + // Compute sqrt(2/pi) * (gate + 0.044715 * gate^3) + __m512 scaled_term = _mm512_mul_ps(sqrt_2_pi, inner_term); + + // Compute tanh using the corrected exp function: tanh(x) ≈ (exp(x) - exp(-x)) / (exp(x) + exp(-x)) + __m512 exp_pos = _mm512_exp_ps_corrected(scaled_term); + __m512 exp_neg = _mm512_exp_ps_corrected(_mm512_sub_ps(_mm512_setzero_ps(), scaled_term)); + + __m512 numerator = _mm512_sub_ps(exp_pos, exp_neg); + __m512 denominator = _mm512_add_ps(exp_pos, exp_neg); + __m512 tanh_approx = _mm512_div_ps(numerator, denominator); + + // Compute 1 + tanh(...) + __m512 one_plus_tanh = _mm512_add_ps(one, tanh_approx); + + // Compute 0.5 * gate * (1 + tanh(...)) + __m512 gelu = _mm512_mul_ps(half, _mm512_mul_ps(gate_vec_fp32, one_plus_tanh)); + return gelu; +} + +// AVX-512 sigmoid: 1 / (1 + exp(-x)) +inline __m512 sigmoid_avx512(__m512 x) { + const __m512 one = _mm512_set1_ps(1.0f); + __m512 neg_x = _mm512_sub_ps(_mm512_setzero_ps(), x); + __m512 exp_neg_x = _mm512_exp_ps_corrected(neg_x); + return _mm512_div_ps(one, _mm512_add_ps(one, exp_neg_x)); +} + +// AVX-512 SiLU (Swish): x * sigmoid(x) = x / (1 + exp(-x)) +inline __m512 silu_avx512(__m512 x) { + return _mm512_mul_ps(x, sigmoid_avx512(x)); +} + +// Vectorized gaussian function for 16 floats using AVX-512 exp approximation +inline __m512 gaussian_avx512(__m512 x, __m512 sigma) { + const __m512 one = _mm512_set1_ps(1.0f); + const __m512 two = _mm512_set1_ps(2.0f); + const __m512 zero = _mm512_setzero_ps(); + + // Check if sigma <= 0 + __mmask16 mask_zero_sigma = _mm512_cmp_ps_mask(sigma, zero, _CMP_LE_OQ); + + // Compute exp(-(x*x)/(2*sigma*sigma)) using fast AVX-512 approximation + __m512 x_sq = _mm512_mul_ps(x, x); + __m512 sigma_sq = _mm512_mul_ps(sigma, sigma); + __m512 two_sigma_sq = _mm512_mul_ps(two, sigma_sq); + + // Compute -(x*x)/(2*sigma*sigma) + __m512 neg_x_sq_over_2sigma_sq = _mm512_div_ps(_mm512_sub_ps(zero, x_sq), two_sigma_sq); + + // Apply fast exponential + __m512 exp_result = _mm512_exp_ps_corrected(neg_x_sq_over_2sigma_sq); + + // Return 1.0 if sigma <= 0, otherwise exp result + return _mm512_mask_blend_ps(mask_zero_sigma, exp_result, one); +} + +// Fast conversion from uint8 to float with normalization +inline void convert_uint8_to_float_avx512(const uint8_t* src, float* dst, size_t count) { + const size_t simd_count = count & ~15; // Process in chunks of 16 + + for (size_t i = 0; i < simd_count; i += 16) { + // Load 16 uint8 values + __m128i u8_vec = _mm_loadu_si128(reinterpret_cast(src + i)); + + // Convert to 32-bit integers + __m512i i32_vec = _mm512_cvtepu8_epi32(u8_vec); + + // Convert to float + __m512 f32_vec = _mm512_cvtepi32_ps(i32_vec); + + // Store result + _mm512_storeu_ps(dst + i, f32_vec); + } + + // Handle remaining elements + for (size_t i = simd_count; i < count; ++i) { + dst[i] = static_cast(src[i]); + } +} + +// Fast conversion from float to uint8 with clamping +inline void convert_float_to_uint8_avx512(const float* src, uint8_t* dst, size_t count) { + const __m512 zero = _mm512_setzero_ps(); + const __m512 max_val = _mm512_set1_ps(255.0f); + const size_t simd_count = count & ~15; // Process in chunks of 16 + + for (size_t i = 0; i < simd_count; i += 16) { + // Load 16 float values + __m512 f32_vec = _mm512_loadu_ps(src + i); + + // Round to nearest integer + f32_vec = _mm512_roundscale_ps(f32_vec, _MM_FROUND_TO_NEAREST_INT); + + // Clamp to [0, 255] + f32_vec = _mm512_max_ps(f32_vec, zero); + f32_vec = _mm512_min_ps(f32_vec, max_val); + + // Convert to 32-bit integers + __m512i i32_vec = _mm512_cvtps_epi32(f32_vec); + + // Pack to uint8 (with saturation) + __m128i u8_vec = _mm512_cvtusepi32_epi8(i32_vec); + + // Store result + _mm_storeu_si128(reinterpret_cast<__m128i*>(dst + i), u8_vec); + } + + // Handle remaining elements + for (size_t i = simd_count; i < count; ++i) { + dst[i] = static_cast(std::clamp(std::round(src[i]), 0.0f, 255.0f)); + } +} diff --git a/src/detail/gemma4e_npu/conv1d_prefill.hpp b/src/detail/gemma4e_npu/conv1d_prefill.hpp new file mode 100644 index 000000000..2ecbe5044 --- /dev/null +++ b/src/detail/gemma4e_npu/conv1d_prefill.hpp @@ -0,0 +1,230 @@ +#pragma once +#include +#include // Required for std::max +#include +#include "npu_utils/npu_instr_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +void conv1d_prefill( + npu_sequence* seq, + const uint32_t L_OUT, + bf16 min_value, + bf16 max_value, + const uint32_t external_x_offset, // in bf16 + const uint32_t external_o_offset, + const uint32_t d +){ + + auto round_up_to_multiple = [](int x, int multiple) -> int { + if (multiple == 0) { + return x; // Cannot divide by zero + } + // This uses integer division to achieve the rounding + return ((x + multiple - 1) / multiple) * multiple; + }; + + npu_tiles IT[8] = {IT0, IT1, IT2, IT3, IT4, IT5, IT6, IT7}; + seq->clear_cmds(); + + const int l_address = 49664; + const int round_address = 8704; + const int min_address = 27136; + const int max_address = 35840; + constexpr int CT_lock_address_base = 0x000001F000; + constexpr int CT_rtp_sync_lock_id = 6; + + assert(d == 1024); + + const int num_col = 8; + float max_float = (float)max_value; + float min_float = (float)min_value; + + const int l_column = 256; + int ROUND = L_OUT / (l_column * num_col); + int remaining = L_OUT % (l_column * num_col); + int needed_col = 0; + int l_left_last_col = 0; + + if (remaining > 0){ + // calculate how many columns are needed for the remaining part + needed_col = remaining / l_column; + if (remaining % l_column != 0) { + // calculate how many rows are needed for the remaining part in the last column + l_left_last_col = remaining - needed_col * l_column; + } + } + + if(ROUND > 0){ + int num_col = 8; + for (int row = 2; row < 6; row++){ + for (int col = 0; col < num_col; col++){ + npu_tiles tile = get_tile(row, col); + seq->rtp_write(tile, l_address, l_column); + seq->rtp_write(tile, round_address, ROUND); + seq->rtp_write(tile, max_address, *(uint32_t*)(&max_float)); + seq->rtp_write(tile, min_address, *(uint32_t*)(&min_float)); + seq->rtp_write(tile, CT_lock_address_base + 16 * CT_rtp_sync_lock_id, 1); // set lock to 1 + } + } + for (int col = 0; col < num_col; col++){ + // send w + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[col], + (npu_bd_id)(1), it_channel_1, + {0, 0, 0, (uint32_t)0}, + {1, 1, (uint32_t)1, (uint32_t)(5 * d)}, + {0, 0, (uint32_t)0, (uint32_t)1}, + -1, 0, false + ); + } + for (int round = 0; round < ROUND; round++){ + int bd_offset = (round % 2) * 8; + for (int col = 0; col < num_col; col++){ + // send x + uint32_t x_offset = external_x_offset + round * l_column * num_col * d + col * l_column * d; + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[col], + (npu_bd_id)(bd_offset + 2), it_channel_0, + {0, 0, 0, x_offset}, + {1, 1, 1, (uint32_t)((l_column + 4) * d)}, + {0, 0, 0, (uint32_t)1}, + -1, 0, false + ); + // receive o + uint32_t o_offset = external_o_offset + round * l_column * num_col * d + col * l_column * d; + seq->npu_dma_memcpy_nd( + 2, 0, + S2MM, IT[col], + (npu_bd_id)(bd_offset + 0), it_channel_0, + {0, 0, 0, o_offset}, + {1, 1, 1, (uint32_t)(l_column * d)}, + {0, 0, 0, (uint32_t)1}, + -1, 0, true + ); + } + if (round > 0){ + for (int col = 0; col < num_col; col++){ + seq->npu_dma_wait( + IT[col], + S2MM, + it_channel_0 + ); + } + } + } + for (int col = 0; col < num_col; col++){ + seq->npu_dma_wait( + IT[col], + S2MM, + it_channel_0 + ); + } + } + + if(remaining > 0){ + int num_col = needed_col; + + for (int row = 2; row < 6; row++){ + for (int col = 0; col < num_col; col++){ + npu_tiles tile = get_tile(row, col); + seq->rtp_write(tile, l_address, l_column); + seq->rtp_write(tile, round_address, 1); + seq->rtp_write(tile, max_address, *(uint32_t*)(&max_float)); + seq->rtp_write(tile, min_address, *(uint32_t*)(&min_float)); + seq->rtp_write(tile, CT_lock_address_base + 16 * CT_rtp_sync_lock_id, 1); // set lock to 1 + } + } + for (int col = 0; col < num_col; col++){ + // send w + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[col], + (npu_bd_id)(1), it_channel_1, + {0, 0, 0, (uint32_t)0}, + {1, 1, (uint32_t)1, (uint32_t)(5 * d)}, + {0, 0, (uint32_t)0, (uint32_t)1}, + -1, 0, false + ); + } + for (int col = 0; col < num_col; col++){ + // send x + uint32_t x_offset = external_x_offset + ROUND * l_column * 8 * d + col * l_column * d; + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[col], + (npu_bd_id)(2), it_channel_0, + {0, 0, 0, x_offset}, + {1, 1, 1, (uint32_t)((l_column + 4) * d)}, + {0, 0, 0, (uint32_t)1}, + -1, 0, false + ); + // receive o + uint32_t o_offset = external_o_offset + ROUND * l_column * 8 * d + col * l_column * d; + seq->npu_dma_memcpy_nd( + 2, 0, + S2MM, IT[col], + (npu_bd_id)(0), it_channel_0, + {0, 0, 0, o_offset}, + {1, 1, 1, (uint32_t)(l_column * d)}, + {0, 0, 0, (uint32_t)1}, + -1, 0, true + ); + seq->npu_dma_wait( + IT[col], + S2MM, + it_channel_0 + ); + } + + if (remaining % l_column != 0){ + for (int row = 2; row < 6; row++){ + npu_tiles tile = get_tile(row, num_col); + seq->rtp_write(tile, l_address, l_left_last_col); + seq->rtp_write(tile, round_address, 1); + seq->rtp_write(tile, max_address, *(uint32_t*)(&max_float)); + seq->rtp_write(tile, min_address, *(uint32_t*)(&min_float)); + seq->rtp_write(tile, CT_lock_address_base + 16 * CT_rtp_sync_lock_id, 1); // set lock to 1 + } + // send w + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[num_col], + (npu_bd_id)(1), it_channel_1, + {0, 0, 0, (uint32_t)0}, + {1, 1, (uint32_t)1, (uint32_t)(5 * d)}, + {0, 0, (uint32_t)0, (uint32_t)1}, + -1, 0, false + ); + // send x + uint32_t x_offset = external_x_offset + ROUND * l_column * 8 * d + num_col * l_column * d; + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[num_col], + (npu_bd_id)(2), it_channel_0, + {0, 0, 0, x_offset}, + {1, 1, 1, (uint32_t)((l_left_last_col + 4) * d)}, + {0, 0, 0, (uint32_t)1}, + -1, 0, false + ); + // receive o + uint32_t o_offset = external_o_offset +ROUND * l_column * 8 * d + num_col * l_column * d; + seq->npu_dma_memcpy_nd( + 2, 0, + S2MM, IT[num_col], + (npu_bd_id)(0), it_channel_0, + {0, 0, 0, o_offset}, + {1, 1, 1, (uint32_t)(l_left_last_col * d)}, + {0, 0, 0, (uint32_t)1}, + -1, 0, true + ); + seq->npu_dma_wait( + IT[num_col], + S2MM, + it_channel_0 + ); + } + } + + seq->cmds2seq(); +} diff --git a/src/detail/gemma4e_npu/embedding_q8_0.hpp b/src/detail/gemma4e_npu/embedding_q8_0.hpp new file mode 100644 index 000000000..5f1565a69 --- /dev/null +++ b/src/detail/gemma4e_npu/embedding_q8_0.hpp @@ -0,0 +1,85 @@ +#ifndef __EMBEDDING_Q8_0_HPP__ +#define __EMBEDDING_Q8_0_HPP__ +#include "buffer.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "tensor_2d.hpp" +#include + +class embedding_q8_0{ +private: + static constexpr int Q8_0_GROUP_SIZE = 32; + buffer scale; + buffer qweight; + size_t vocabe_size; + size_t dim; + tensor_2d tensor_qweight; + tensor_2d tensor_scale; + + buffer out_buffer; + + public: + embedding_q8_0(size_t vocab_size, size_t dim) { + this->vocabe_size = vocab_size; + this->dim = dim; + this->scale = buffer(vocab_size * dim / Q8_0_GROUP_SIZE); + this->qweight = buffer(vocab_size * dim); + this->out_buffer = buffer(dim); + this->tensor_qweight = tensor_2d(qweight, dim); + this->tensor_scale = tensor_2d(scale, dim / Q8_0_GROUP_SIZE); + } + + void init_weights(Q4NX& q4nx, const std::string& weight_name){ + q4nx.load_weights(this->scale, weight_name + ".weight.scale"); + q4nx.load_weights(this->qweight, weight_name + ".weight"); + } + + inline void dequant_row_avx512(const int8_t* __restrict qw, const float* __restrict sc, bf16* __restrict dst) { + const size_t num_groups = dim / Q8_0_GROUP_SIZE; + for (size_t i = 0; i < num_groups; i++) { + const int8_t* group_ptr = qw + i * Q8_0_GROUP_SIZE; + bf16* out_ptr = dst + i * Q8_0_GROUP_SIZE; + __m512 scale_vec = _mm512_set1_ps(sc[i]); + + // First 16 elements + __m128i q8_lo = _mm_loadu_si128((const __m128i*)group_ptr); + __m512i q32_lo = _mm512_cvtepi8_epi32(q8_lo); + __m512 f32_lo = _mm512_cvtepi32_ps(q32_lo); + f32_lo = _mm512_mul_ps(f32_lo, scale_vec); + __m512i bf16_lo = _mm512_srli_epi32(_mm512_castps_si512( + _mm512_add_ps(f32_lo, _mm512_castsi512_ps( + _mm512_add_epi32(_mm512_set1_epi32(0x7FFF), + _mm512_and_si512(_mm512_srli_epi32(_mm512_castps_si512(f32_lo), 16), + _mm512_set1_epi32(1)))))), 16); + __m256i out_lo = _mm512_cvtepi32_epi16(bf16_lo); + _mm256_storeu_si256((__m256i*)out_ptr, out_lo); + + // Next 16 elements + __m128i q8_hi = _mm_loadu_si128((const __m128i*)(group_ptr + 16)); + __m512i q32_hi = _mm512_cvtepi8_epi32(q8_hi); + __m512 f32_hi = _mm512_cvtepi32_ps(q32_hi); + f32_hi = _mm512_mul_ps(f32_hi, scale_vec); + __m512i bf16_hi = _mm512_srli_epi32(_mm512_castps_si512( + _mm512_add_ps(f32_hi, _mm512_castsi512_ps( + _mm512_add_epi32(_mm512_set1_epi32(0x7FFF), + _mm512_and_si512(_mm512_srli_epi32(_mm512_castps_si512(f32_hi), 16), + _mm512_set1_epi32(1)))))), 16); + __m256i out_hi = _mm512_cvtepi32_epi16(bf16_hi); + _mm256_storeu_si256((__m256i*)(out_ptr + 16), out_hi); + } + } + + buffer forward(int idx){ + buffer qweight_row = tensor_qweight[idx]; + buffer scale_row = tensor_scale[idx]; + dequant_row_avx512(qweight_row.data(), scale_row.data(), out_buffer.data()); + return out_buffer; + } + + void forward(int idx, buffer& out){ + buffer qweight_row = tensor_qweight[idx]; + buffer scale_row = tensor_scale[idx]; + dequant_row_avx512(qweight_row.data(), scale_row.data(), out.data()); + } +}; + +#endif diff --git a/src/detail/gemma4e_npu/gemma4e_audio.cpp b/src/detail/gemma4e_npu/gemma4e_audio.cpp new file mode 100644 index 000000000..0d493d4d9 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_audio.cpp @@ -0,0 +1,2280 @@ +#include "gemma4e_audio.hpp" + +#include +#include +#include +#ifdef _WIN32 +#include +#endif +#include "utils/debug_utils.hpp" +#include "utils/error_measure.hpp" + +#include "gemma4e_vision_prefill_helper.hpp" + +#include "vision/norm.hpp" +#include "mmRuntimeSequence.hpp" +#include "rot_pos_emb.hpp" + +#include "gemma4e_audio_attention.hpp" +#include "conv1d_prefill.hpp" +#include "utils/utils.hpp" +#include +// #define DEBUG_PRINT_ENCODE_ERROR_METRICS 1 + +Gemma4e_AudioEncoder::~Gemma4e_AudioEncoder() {} + +Gemma4e_AudioEncoder::Gemma4e_AudioEncoder(LM_Config config, npu_xclbin_manager *npu_instance, gemma4e_npu* parent_npu_ptr) + : config(config), npu(npu_instance), model_path(config.model_path), parent_npu_ptr(parent_npu_ptr) +{ + + // load parameters from json file + + { + MM_tile_M = config.sub("audio_config").value("Audio_MM_TILE_M", -1); + MM_tile_K = config.sub("audio_config").value("Audio_MM_TILE_K", -1); + MM_tile_N = config.sub("audio_config").value("Audio_MM_TILE_N", -1); + + seq_len_pad_requirement_for_MM = MM_ROW_SIZE*MM_tile_M; + assert( MM_tile_K % MM_tile_N == 0); + + Gemma4E_Audio_residual_weight = config.sub("audio_config").value("Gemma4E_Audio_residual_weight", 0.0); + assert(this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE % this->parent_npu_ptr->Gemma4E_Audio_num_attention_heads == 0); + Gemma4E_Audio_attention_head_dim = this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE / this->parent_npu_ptr->Gemma4E_Audio_num_attention_heads; + } + + Gemma4E_Audio_q_scale = (1/std::sqrt(Gemma4E_Audio_attention_head_dim)) / std::log(2); + Gemma4E_Audio_k_scale = std::log(1 + std::numbers::e) / std::log(2); + + Gemma4E_Audio_padded_requirement_for_conv1d = this->parent_npu_ptr->Gemma4E_Audio_conv1d_kernel_size\ + - this->parent_npu_ptr->Gemma4E_Audio_conv1d_stride; + DEBUG_BLOCK(1, + std::cout << "Audio q scale: " << Gemma4E_Audio_q_scale << ", k_scale: " << Gemma4E_Audio_k_scale << std::endl; + std::cout << "Gemma4E_Audio_padded_requirement_for_conv1d : " << Gemma4E_Audio_padded_requirement_for_conv1d << std::endl; + ) + + Padded_GEMMA4E_Audio_HIDDEN_SIZE = round_up_to_multiple(this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, MM_tile_K); + Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE = round_up_to_multiple(this->parent_npu_ptr->Gemma4E_Audio_INTERMEDIATE_SIZE, MM_tile_K); + Padded_Gemma4E_Audio_Multimodal_Output_SIZE = round_up_to_multiple(this->parent_npu_ptr->Gemma4E_Audio_Multimodal_Output_SIZE, MM_tile_K); + Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE = round_up_to_multiple( + this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE*2, MM_tile_K + ); + assert(Padded_GEMMA4E_Audio_HIDDEN_SIZE %Gemma4E_Audio_attention_head_dim == 0); + Padded_Gemma4E_Audio_num_attention_heads = Padded_GEMMA4E_Audio_HIDDEN_SIZE / Gemma4E_Audio_attention_head_dim; + assert(Padded_Gemma4E_Audio_num_attention_heads >= this->parent_npu_ptr->Gemma4E_Audio_num_attention_heads); + + this->proj = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "vision_mm.xclbin")); + this->proj_high_precision = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "vision_mm_high_precision.xclbin")); + this->conv1d = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "audio_conv1d.xclbin")); + + this->sub_sampleConvProjection_app = this->proj_high_precision->create_app(); + this->q_proj_app = this->proj->create_app(); + this->k_proj_app = this->proj->create_app(); + this->k_relative_proj_app = this->proj->create_app(); + this->v_proj_app = this->proj->create_app(); + this->o_proj_app = this->proj->create_app(); + this->ffn_down_proj_app = this->proj->create_app(); + this->ffn_up_proj_app = this->proj->create_app(); + this->conv1d_start_proj_app = this->proj->create_app(); + this->conv1d_end_proj_app = this->proj->create_app(); + this->conv1d_app = this->conv1d->create_app(); + audio_pre_encode_proj_app = this->proj->create_app(); + audio_to_language_proj_app = this->proj->create_app(); + + this->audio_attn_k_rel_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_attn_k_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_attn_v_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_attn_q_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_attn_o_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_conv1d_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_conv_pw_1_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_conv_pw_2_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_ffn_down_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_ffn_up_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_ffn_down_1_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->audio_ffn_up_1_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->attn_post_norm_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->attn_pre_norm_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->conv_norm_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->norm_conv_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->ffn_norm_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->ffn_norm_1_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->ffn_post_norm_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->ffn_post_norm_1_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->norm2_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); + this->per_dim_scale_with_softplus_weight.resize(this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers); +} + +void Gemma4e_AudioEncoder::conv1d_layer( + + int layer_idx, + int seq_len, int seq_len_padded, + std::vector &seq_len_per_audio, std::vector &start_seq_len_index_per_audio, + buffer &conv1d_start_proj_input, buffer &conv1d_start_proj_output, + buffer &conv1d_input, buffer &conv1d_output, + buffer &conv1d_end_proj_input, buffer &conv1d_end_proj_output, + + bf16 conv1d_start_input_min, bf16 conv1d_start_input_max, + bf16 conv1d_start_output_min, bf16 conv1d_start_output_max, + + bf16 conv1d_end_input_min, bf16 conv1d_end_input_max, + bf16 conv1d_end_output_min, bf16 conv1d_end_output_max, + + gemma4e_audio_payload_t* audio_payload, + SafeTensors *reference_safetensor +){ + + memcpy(residual.data(), hidden_state.data(), seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16)); + + // compare the norm weigths + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Comparing conv1d_layer " << layer_idx << " norm weights..." << std::endl; + buffer ref_per_layer_norm_weight; + reference_safetensor->load_weights( + ref_per_layer_norm_weight, + "Gemma4AudioLightConv1d_"+std::to_string(layer_idx)+ "_pre_layer_norm_weight" + ); + print_error_metrics( + this->conv_norm_weight[layer_idx].data(), ref_per_layer_norm_weight.data(), + 1, + ref_per_layer_norm_weight.size(), 1, + ref_per_layer_norm_weight.size(), 1 + ); + + std::cout << "Comparing conv1d_layer " << layer_idx << " conv norm weights..." << std::endl; + buffer ref_conv_norm_weight; + reference_safetensor->load_weights( + ref_conv_norm_weight, + "Gemma4AudioLightConv1d_"+std::to_string(layer_idx)+ "_conv_norm_weight" + ); + print_error_metrics( + this->norm_conv_weight[layer_idx].data(), ref_conv_norm_weight.data(), + 1, + ref_conv_norm_weight.size(), 1, + ref_conv_norm_weight.size(), 1 + ); + } + #endif + + simd_rms_norm( + hidden_state.data(), this->conv_norm_weight[layer_idx].data(), conv1d_start_proj_input.data(), + seq_len, this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, 1e-6f + ); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Comparing conv1d_layer " << layer_idx << " pre-attention norm output..." << std::endl; + buffer ref_Gemma4AudioLightConv1d_layer_idx_hidden_states_after_pre_layer_norm; + reference_safetensor->load_weights( + ref_Gemma4AudioLightConv1d_layer_idx_hidden_states_after_pre_layer_norm, + "Gemma4AudioLightConv1d_"+std::to_string(layer_idx)+ "_hidden_states_after_pre_layer_norm" + ); + size_t ref_offset_per_audio = ref_Gemma4AudioLightConv1d_layer_idx_hidden_states_after_pre_layer_norm.size() / audio_payload->num_audios; + for(int i = 0; i < audio_payload->num_audios; i++){ + print_error_metrics( + conv1d_start_proj_input.data() + start_seq_len_index_per_audio[i]*Padded_GEMMA4E_Audio_HIDDEN_SIZE, + ref_Gemma4AudioLightConv1d_layer_idx_hidden_states_after_pre_layer_norm.data() + i*ref_offset_per_audio, + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + + simd_clamp( + conv1d_start_proj_input.data(), conv1d_start_proj_input.data(), + conv1d_start_input_min, conv1d_start_input_max, + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + { + + generate_mm_sequence( + *this->conv1d_start_proj_app.seq(), + seq_len_padded, this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, //no bias, no activation + 1, conv1d_start_output_min, conv1d_start_output_max, // enable clamp in output + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + } + DEBUG_BLOCK(1, + std::cout << "DEBUG: Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE: " << Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE << std::endl; + ) + conv1d_start_proj_input.sync_to_device(); + audio_conv_pw_1_weight[layer_idx].sync_to_device(); + conv1d_start_proj_app(conv1d_start_proj_input, audio_conv_pw_1_weight[layer_idx], conv1d_start_proj_output); + conv1d_start_proj_output.sync_from_device(); + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Comparing conv1d_layer " << layer_idx << " conv1d linear output..." << std::endl; + buffer ref_conv1d_start_proj_output; + reference_safetensor->load_weights( + ref_conv1d_start_proj_output, + "Gemma4AudioLightConv1d_" + std::to_string(layer_idx) +"_hidden_states_after_linear_start" + ); + size_t ref_offset_per_audio = ref_conv1d_start_proj_output.size() / audio_payload->num_audios; + for(int i = 0; i < audio_payload->num_audios; i++){ + print_error_metrics( + conv1d_start_proj_output.data() + start_seq_len_index_per_audio[i]*Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE, + ref_conv1d_start_proj_output.data() + i*ref_offset_per_audio, + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE * 2, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE + ); + } + } + #endif + + assert(Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE == (Padded_GEMMA4E_Audio_HIDDEN_SIZE*2) ); + + for(int i = 0, seq_len_offset = 0; i < audio_payload->num_audios; i++){ + + seq_len_offset += Gemma4E_Audio_padded_requirement_for_conv1d; + simd_glu( + conv1d_start_proj_output.data() + start_seq_len_index_per_audio[i] * Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE, + conv1d_input.data() + (seq_len_offset * Padded_GEMMA4E_Audio_HIDDEN_SIZE), + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + + seq_len_offset += (seq_len_per_audio[i] ); + } + conv1d_start_proj_output.sync_to_device(); + + // compare with error metrics + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Comparing conv1d_layer " << layer_idx << " conv1d GLU output..." << std::endl; + buffer ref_conv1d_glu_output; + reference_safetensor->load_weights( + ref_conv1d_glu_output, + "Gemma4AudioLightConv1d_" + std::to_string(layer_idx) +"_hidden_states_after_glu" + ); + + size_t ref_offset_per_audio = ref_conv1d_glu_output.size() / audio_payload->num_audios; + + for(int i = 0, seq_len_offset = 0; i < audio_payload->num_audios; i++){ + seq_len_offset += Gemma4E_Audio_padded_requirement_for_conv1d; + + print_error_metrics( + conv1d_input.data() + seq_len_offset*Padded_GEMMA4E_Audio_HIDDEN_SIZE, + ref_conv1d_glu_output.data() + i*ref_offset_per_audio, + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + + seq_len_offset += seq_len_per_audio[i]; + } + } + #endif + + // conv1d + for(int i = 0, seq_len_offset = 0; i < audio_payload->num_audios; i++){ + + conv1d_prefill( + this->conv1d_app.seq(), + seq_len_per_audio[i], + -1.18e30f, 3.38e30f, // no clamping, + seq_len_offset * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + start_seq_len_index_per_audio[i] *Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + + // scalar_conv1d( + // this->parent_npu_ptr->Gemma4E_Audio_conv1d_kernel_size, + // this->parent_npu_ptr->Gemma4E_Audio_conv1d_stride, + // conv1d_input.data() + seq_len_offset * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + // audio_conv1d_weight[layer_idx].data(), + // conv1d_output.data() + start_seq_len_index_per_audio[i] *Padded_GEMMA4E_Audio_HIDDEN_SIZE, + // seq_len_per_audio[i], + // Padded_GEMMA4E_Audio_HIDDEN_SIZE + // ); + + seq_len_offset += (Gemma4E_Audio_padded_requirement_for_conv1d + seq_len_per_audio[i]); + + conv1d_input.sync_to_device(); + audio_conv1d_weight[layer_idx].sync_to_device(); + conv1d_app(conv1d_output, conv1d_input,audio_conv1d_weight[layer_idx] ); + conv1d_output.sync_from_device(); + } + + // utils::print_matrix( + // audio_conv1d_weight[layer_idx], Padded_GEMMA4E_Audio_HIDDEN_SIZE + + // ); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Comparing conv1d_layer " << layer_idx << " conv1d output..." << std::endl; + buffer ref_conv1d_output; + reference_safetensor->load_weights( + ref_conv1d_output, + "Gemma4AudioLightConv1d_" + std::to_string(layer_idx) +"_hidden_states_after_depthwise_conv1d" + ); + + size_t ref_offset_per_audio = ref_conv1d_output.size() / audio_payload->num_audios; + for(int i = 0; i < audio_payload->num_audios; i++){ + print_error_metrics( + conv1d_output.data() + start_seq_len_index_per_audio[i]*Padded_GEMMA4E_Audio_HIDDEN_SIZE, + ref_conv1d_output.data() + i*ref_offset_per_audio, + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + + simd_rms_norm( + conv1d_output.data(), this->norm_conv_weight[layer_idx].data(), conv1d_output.data(), + seq_len, this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, 1e-6f + ); + + simd_silu( + conv1d_output.data(), conv1d_end_proj_input.data(), + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + conv1d_output.sync_to_device(); + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer ref_hidden_states_after_conv_act; + reference_safetensor->load_weights( + ref_hidden_states_after_conv_act, + "Gemma4AudioLightConv1d_" + std::to_string(layer_idx) +"_hidden_states_after_conv_act" + ); + + size_t ref_offset_per_audio = ref_hidden_states_after_conv_act.size() / audio_payload->num_audios; + + for(int i = 0; i < audio_payload->num_audios; i++){ + std::cout << "Comparing conv1d_layer " << layer_idx << " conv1d activation output for audio " << i << std::endl; + print_error_metrics( + conv1d_end_proj_input.data() + start_seq_len_index_per_audio[i]*Padded_GEMMA4E_Audio_HIDDEN_SIZE, + ref_hidden_states_after_conv_act.data() + i*ref_offset_per_audio, + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + + simd_clamp( + conv1d_end_proj_input.data(), conv1d_end_proj_input.data(), + conv1d_end_input_min, conv1d_end_input_max, + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + + { + generate_mm_sequence( + *this->conv1d_end_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, Padded_GEMMA4E_Audio_HIDDEN_SIZE, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, //no bias, no activation + 1, conv1d_end_output_min,conv1d_end_output_max, // enable clamp in output + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + } + + conv1d_end_proj_input.sync_to_device(); + audio_conv_pw_2_weight[layer_idx].sync_to_device(); + conv1d_end_proj_app(conv1d_end_proj_input, audio_conv_pw_2_weight[layer_idx], conv1d_end_proj_output); + conv1d_end_proj_output.sync_from_device(); + + simd_add(conv1d_end_proj_output.data(), residual.data(), hidden_state.data(), + seq_len*Padded_GEMMA4E_Audio_HIDDEN_SIZE); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer ref_conv1d_final_output; + reference_safetensor->load_weights( + ref_conv1d_final_output, + "Gemma4AudioLightConv1d_" + std::to_string(layer_idx) +"_hidden_states_after_residual" + ); + + size_t ref_offset_per_audio = ref_conv1d_final_output.size() / audio_payload->num_audios; + for(int i = 0; i < audio_payload->num_audios; i++){ + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_audio[i]*Padded_GEMMA4E_Audio_HIDDEN_SIZE, + ref_conv1d_final_output.data() + i*ref_offset_per_audio, + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif +} + +void Gemma4e_AudioEncoder::ffn_layer( + + buffer &ffn_up_proj_input, + buffer &ffn_up_proj_output_down_input, + buffer &ffn_down_proj_output, + + buffer &cur_ffn_norm_weight, // ffn_norm or ffn_norm_1 + buffer &cur_ffn_post_norm_weight,// ffn_post_norm_weight or ffn_post_norm_1_weight + buffer &cur_ffn_up_weight, // audio_ffn_up_weight or audio_ffn_up_1_weight + buffer &cur_ffn_down_weight, // audio_ffn_down_weight + int seq_len, int seq_len_padded, + + bf16 cur_audio_ffn_up_input_min, bf16 cur_audio_ffn_up_input_max, + bf16 cur_audio_ffn_up_output_min, bf16 cur_audio_ffn_up_output_max, + bf16 cur_audio_ffn_down_input_min, bf16 cur_audio_ffn_down_input_max, + bf16 cur_audio_ffn_down_output_min, bf16 cur_audio_ffn_down_output_max + +){ + + memcpy(residual.data(), hidden_state.data(), seq_len * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16)); + + generate_mm_sequence( + *this->ffn_up_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 2, //no bias, silu activation + 1, cur_audio_ffn_up_output_min,cur_audio_ffn_up_output_max, // with clamp + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + + simd_rms_norm( + hidden_state.data(), + cur_ffn_norm_weight.data(), + ffn_up_proj_input.data(), + seq_len, + this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 1e-6f + ); + simd_clamp(ffn_up_proj_input.data(), + ffn_up_proj_input.data(), + cur_audio_ffn_up_input_min, cur_audio_ffn_up_input_max, + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + + //ffn_up_proj_app(ffn_up_proj_input, cur_ffn_up_weight, ffn_up_proj_output_down_input); + auto ffn_up_proj_run = ffn_up_proj_app.create_run( + ffn_up_proj_input,cur_ffn_up_weight, ffn_up_proj_output_down_input + ); + ffn_up_proj_input.sync_to_device(); + cur_ffn_up_weight.sync_to_device(); + ffn_up_proj_run.start(); + + generate_mm_sequence( + *this->ffn_down_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE, Padded_GEMMA4E_Audio_HIDDEN_SIZE, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, //no bias, no activation + 1, cur_audio_ffn_down_output_min,cur_audio_ffn_down_output_max, // with clamp + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + + ffn_up_proj_run.wait(); + ffn_up_proj_output_down_input.sync_from_device(); + + simd_clamp( + ffn_up_proj_output_down_input.data(), + ffn_up_proj_output_down_input.data(), + cur_audio_ffn_down_input_min, cur_audio_ffn_down_input_max, + seq_len * Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE + ); + + ffn_up_proj_output_down_input.sync_to_device(); + cur_ffn_down_weight.sync_to_device(); + ffn_down_proj_app(ffn_up_proj_output_down_input,cur_ffn_down_weight, ffn_down_proj_output ); + ffn_down_proj_output.sync_from_device(); + + // perform a post_layer_nrom + simd_rms_norm( + ffn_down_proj_output.data(), + cur_ffn_post_norm_weight.data(), + ffn_down_proj_output.data(), + seq_len, + this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 1e-6f + ); + + simd_add( + ffn_down_proj_output.data(), residual.data(), hidden_state.data(), + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + ffn_down_proj_output.sync_to_device(); +} + +void Gemma4e_AudioEncoder::init_weights(SafeTensors &q4nx){ + DEBUG_BLOCK(1, + std::cout << "Initializing Gemma4e_AudioEncoder weights from model path: " << model_path << std::endl; + ) + + q4nx.load_weights(this->audio_subsample_conv2d_weight_0,"model.audio.subsample.conv_layer0.weight"); + q4nx.load_weights(this->audio_subsample_conv2d_norm_weight_0,"model.audio.subsample.conv_layer0.norm.weight"); + q4nx.load_weights(this->audio_subsample_conv2d_weight_1,"model.audio.subsample.conv_layer1.weight"); + q4nx.load_weights(this->audio_subsample_conv2d_norm_weight_1,"model.audio.subsample.conv_layer1.norm.weight"); + { + buffer temp_buffer; + q4nx.load_weights(temp_buffer, "model.audio.encode_input_projection.weight"); + this->audio_embedding_projection_weight = this->sub_sampleConvProjection_app.create_bo_buffer(temp_buffer.size()); + memcpy( + this->audio_embedding_projection_weight.data(), + temp_buffer.data(), + temp_buffer.size() * sizeof(bf16) + ); + } + + audio_pre_encode_weight = this->audio_pre_encode_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_HIDDEN_SIZE * Padded_Gemma4E_Audio_Multimodal_Output_SIZE + Padded_Gemma4E_Audio_Multimodal_Output_SIZE + + ); + DEBUG_BLOCK(1, + std::cout <<"Padded_Gemma4E_Audio_Multimodal_Output_SIZE: " << Padded_Gemma4E_Audio_Multimodal_Output_SIZE << std::endl; + ) + q4nx.load_weights( + this->audio_pre_encode_weight, + "model.audio.pre_encoder.bias", 0 // the bias + ); + q4nx.load_weights( + this->audio_pre_encode_weight, + "model.audio.pre_encoder.weight", + sizeof(bf16) * Padded_Gemma4E_Audio_Multimodal_Output_SIZE // the weight + ); + + audio_to_language_projection_weight = this->audio_to_language_proj_app.create_bo_buffer( + Padded_Gemma4E_Audio_Multimodal_Output_SIZE * parent_npu_ptr->Gemma4E_Audio_language_projection_output_size + ); + assert( + parent_npu_ptr->Gemma4E_Audio_language_projection_output_size% MM_tile_N == 0 + ); + q4nx.load_weights( + audio_to_language_projection_weight, + "model.audio.embedding_projection.weight" + ); + + for(int layer_id =0; layer_id < this->parent_npu_ptr->Gemma4E_Audio_num_attention_layers; layer_id ++){ + + this->audio_attn_k_weight[layer_id] = this->k_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_HIDDEN_SIZE*Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + q4nx.load_weights( + this->audio_attn_k_weight[layer_id], + "model.audio." + std::to_string(layer_id)+ ".attn_k_proj.weight" + ); + + buffer k_input_max; + q4nx.load_weights(k_input_max, + "model.audio."+std::to_string(layer_id)+ ".attn_k_proj.input_max"); + assert(k_input_max.size() == 1); + this->audio_k_input_max.push_back(k_input_max[0]); + + buffer k_input_min; + q4nx.load_weights(k_input_min, + "model.audio."+std::to_string(layer_id)+ ".attn_k_proj.input_min"); + assert(k_input_min.size() == 1); + this->audio_k_input_min.push_back(k_input_min[0]); + + buffer k_output_max; + q4nx.load_weights(k_output_max, + "model.audio."+std::to_string(layer_id)+ ".attn_k_proj.output_max"); + assert(k_output_max.size() == 1); + this->audio_k_output_max.push_back(k_output_max[0]); + + buffer k_output_min; + q4nx.load_weights(k_output_min, + "model.audio."+std::to_string(layer_id)+ ".attn_k_proj.output_min"); + assert(k_output_min.size() == 1); + this->audio_k_output_min.push_back(k_output_min[0]); + + this->audio_attn_k_rel_weight[layer_id] = this->k_relative_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_HIDDEN_SIZE*Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + q4nx.load_weights( + this->audio_attn_k_rel_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".attn_k_proj.rel_weight" + ); + + this->audio_attn_o_weight[layer_id] = this->o_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_HIDDEN_SIZE*Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + q4nx.load_weights( + this->audio_attn_o_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".attn_out_proj.weight" + ); + + buffer o_input_max; + q4nx.load_weights(o_input_max, + "model.audio."+std::to_string(layer_id)+ ".attn_out_proj.input_max"); + assert(o_input_max.size() == 1); + + buffer o_input_min; + q4nx.load_weights(o_input_min, + "model.audio."+std::to_string(layer_id)+ ".attn_out_proj.input_min"); + assert(o_input_min.size() == 1); + this->audio_o_input_min.push_back(o_input_min[0]); + + this->audio_o_input_max.push_back(o_input_max[0]); + buffer o_output_max; + q4nx.load_weights(o_output_max, + "model.audio."+std::to_string(layer_id)+ ".attn_out_proj.output_max"); + assert(o_output_max.size() == 1); + + this->audio_o_output_max.push_back(o_output_max[0]); + buffer o_output_min; + q4nx.load_weights(o_output_min, + "model.audio."+std::to_string(layer_id)+ ".attn_out_proj.output_min"); + assert(o_output_min.size() == 1); + this->audio_o_output_min.push_back(o_output_min[0]); + + q4nx.load_weights( + this->attn_post_norm_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".attn_post_norm.weight" + ); + q4nx.load_weights( + this->attn_pre_norm_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".attn_pre_norm.weight" + ); + + this->audio_attn_q_weight[layer_id] = this->q_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_HIDDEN_SIZE*Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + q4nx.load_weights( + this->audio_attn_q_weight[layer_id], + "model.audio."+ std::to_string(layer_id)+".attn_q_proj.weight" + ); + buffer q_input_max; + q4nx.load_weights(q_input_max, + "model.audio."+std::to_string(layer_id)+ ".attn_q_proj.input_max"); + assert(q_input_max.size() == 1); + this->audio_q_input_max.push_back(q_input_max[0]); + + buffer q_input_min; + q4nx.load_weights(q_input_min, + "model.audio."+std::to_string(layer_id)+ ".attn_q_proj.input_min"); + assert(q_input_min.size() == 1); + this->audio_q_input_min.push_back(q_input_min[0]); + + buffer q_output_max; + q4nx.load_weights(q_output_max, + "model.audio."+std::to_string(layer_id)+ ".attn_q_proj.output_max"); + assert(q_output_max.size() == 1); + this->audio_q_output_max.push_back(q_output_max[0]); + + buffer q_output_min; + q4nx.load_weights(q_output_min, + "model.audio."+std::to_string(layer_id)+ ".attn_q_proj.output_min"); + assert(q_output_min.size() == 1); + this->audio_q_output_min.push_back(q_output_min[0]); + + this->audio_attn_v_weight[layer_id] = this->v_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_HIDDEN_SIZE*Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + q4nx.load_weights( + this->audio_attn_v_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".attn_v_proj.weight"); + buffer v_input_max; + q4nx.load_weights(v_input_max, + "model.audio."+std::to_string(layer_id)+ ".attn_v_proj.input_max"); + assert(v_input_max.size() == 1); + this->audio_v_input_max.push_back(v_input_max[0]); + + buffer v_input_min; + q4nx.load_weights(v_input_min, + "model.audio."+std::to_string(layer_id)+ ".attn_v_proj.input_min"); + assert(v_input_min.size() == 1); + this->audio_v_input_min.push_back(v_input_min[0]); + + buffer v_output_max; + q4nx.load_weights(v_output_max, + "model.audio."+std::to_string(layer_id)+ ".attn_v_proj.output_max"); + assert(v_output_max.size() == 1); + this->audio_v_output_max.push_back(v_output_max[0]); + + buffer v_output_min; + q4nx.load_weights(v_output_min, + "model.audio."+std::to_string(layer_id)+ ".attn_v_proj.output_min"); + assert(v_output_min.size() == 1); + this->audio_v_output_min.push_back(v_output_min[0]); + + this->audio_conv1d_weight[layer_id] = this->conv1d_app.create_bo_buffer( + this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE * this->parent_npu_ptr->Gemma4E_Audio_conv1d_kernel_size + ); + q4nx.load_weights( + this->audio_conv1d_weight[layer_id], + "model.audio." +std::to_string(layer_id) + ".conv_dw.weight" + ); + + q4nx.load_weights( + this->conv_norm_weight[layer_id], + "model.audio."+std::to_string(layer_id) +".conv_norm.weight" + ); + q4nx.load_weights( + this->norm_conv_weight[layer_id], + "model.audio."+std::to_string(layer_id) +".norm_conv.weight" + ); + + // + this->audio_conv_pw_1_weight[layer_id] = this->conv1d_start_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_HIDDEN_SIZE * Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE + ); + q4nx.load_weights( + this->audio_conv_pw_1_weight[layer_id], + "model.audio." +std::to_string(layer_id) + ".conv_pw_1.weight" + ); + buffer conv_pw_1_input_max; + q4nx.load_weights(conv_pw_1_input_max, + "model.audio."+std::to_string(layer_id)+ ".conv_pw_1.input_max"); + assert(conv_pw_1_input_max.size() == 1); + this->audio_conv_pw1_input_max.push_back(conv_pw_1_input_max[0]); + + buffer conv_pw_1_input_min; + q4nx.load_weights(conv_pw_1_input_min, + "model.audio."+std::to_string(layer_id)+ ".conv_pw_1.input_min"); + assert(conv_pw_1_input_min.size() == 1); + this->audio_conv_pw1_input_min.push_back(conv_pw_1_input_min[0]); + + buffer conv_pw_1_output_max; + q4nx.load_weights(conv_pw_1_output_max, + "model.audio."+std::to_string(layer_id)+ ".conv_pw_1.output_max"); + assert(conv_pw_1_output_max.size() == 1); + this->audio_conv_pw1_output_max.push_back(conv_pw_1_output_max[0]); + + buffer conv_pw_1_output_min; + q4nx.load_weights(conv_pw_1_output_min, + "model.audio."+std::to_string(layer_id)+ ".conv_pw_1.output_min"); + assert(conv_pw_1_output_min.size() == 1); + this->audio_conv_pw1_output_min.push_back(conv_pw_1_output_min[0]); + + this->audio_conv_pw_2_weight[layer_id] = this->conv1d_end_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_HIDDEN_SIZE * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + q4nx.load_weights( + this->audio_conv_pw_2_weight[layer_id], + "model.audio." +std::to_string(layer_id) + ".conv_pw_2.weight" + ); + buffer conv_pw_2_input_max; + q4nx.load_weights(conv_pw_2_input_max, + "model.audio."+std::to_string(layer_id)+ ".conv_pw_2.input_max"); + assert(conv_pw_2_input_max.size() == 1); + this->audio_conv_pw2_input_max.push_back(conv_pw_2_input_max[0]); + + buffer conv_pw_2_input_min; + q4nx.load_weights(conv_pw_2_input_min, + "model.audio."+std::to_string(layer_id)+ ".conv_pw_2.input_min"); + assert(conv_pw_2_input_min.size() == 1); + this->audio_conv_pw2_input_min.push_back(conv_pw_2_input_min[0]); + + buffer conv_pw_2_output_max; + q4nx.load_weights(conv_pw_2_output_max, + "model.audio."+std::to_string(layer_id)+ ".conv_pw_2.output_max"); + assert(conv_pw_2_output_max.size() == 1); + this->audio_conv_pw2_output_max.push_back(conv_pw_2_output_max[0]); + + buffer conv_pw_2_output_min; + q4nx.load_weights(conv_pw_2_output_min, + "model.audio."+std::to_string(layer_id)+ ".conv_pw_2.output_min"); + assert(conv_pw_2_output_min.size() == 1); + this->audio_conv_pw2_output_min.push_back(conv_pw_2_output_min[0]); + + this->audio_ffn_down_weight[layer_id] = this->ffn_down_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_HIDDEN_SIZE * Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE + ); + q4nx.load_weights( + this->audio_ffn_down_weight[layer_id], + "model.audio." +std::to_string(layer_id) + ".ffn.down_proj.weight" + ); + buffer ffn_down_input_max; + q4nx.load_weights(ffn_down_input_max, + "model.audio."+std::to_string(layer_id)+ ".ffn.down_proj.input_max"); + assert(ffn_down_input_max.size() == 1); + this->audio_ffn_down_input_max.push_back(ffn_down_input_max[0]); + + buffer ffn_down_input_min; + q4nx.load_weights(ffn_down_input_min, + "model.audio."+std::to_string(layer_id)+ ".ffn.down_proj.input_min"); + assert(ffn_down_input_min.size() == 1); + this->audio_ffn_down_input_min.push_back(ffn_down_input_min[0]); + + buffer ffn_down_output_max; + q4nx.load_weights(ffn_down_output_max, + "model.audio."+std::to_string(layer_id)+ ".ffn.down_proj.output_max"); + assert(ffn_down_output_max.size() == 1); + this->audio_ffn_down_output_max.push_back(ffn_down_output_max[0]); + + buffer ffn_down_output_min; + q4nx.load_weights(ffn_down_output_min, + "model.audio."+std::to_string(layer_id)+ ".ffn.down_proj.output_min"); + assert(ffn_down_output_min.size() == 1); + this->audio_ffn_down_output_min.push_back(ffn_down_output_min[0]); + + this->audio_ffn_down_1_weight[layer_id] = this->ffn_down_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_HIDDEN_SIZE * Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE + ); + q4nx.load_weights( + this->audio_ffn_down_1_weight[layer_id], + "model.audio." +std::to_string(layer_id) + ".ffn.down_proj_1.weight" + ); + buffer ffn_down_1_input_max; + q4nx.load_weights(ffn_down_1_input_max, + "model.audio."+std::to_string(layer_id)+ ".ffn.down_proj_1.input_max"); + assert(ffn_down_1_input_max.size() == 1); + this->audio_ffn_down_1_input_max.push_back(ffn_down_1_input_max[0]); + + buffer ffn_down_1_input_min; + q4nx.load_weights(ffn_down_1_input_min, + "model.audio."+std::to_string(layer_id)+ ".ffn.down_proj_1.input_min"); + assert(ffn_down_1_input_min.size() == 1); + this->audio_ffn_down_1_input_min.push_back(ffn_down_1_input_min[0]); + + buffer ffn_down_1_output_max; + q4nx.load_weights(ffn_down_1_output_max, + "model.audio."+std::to_string(layer_id)+ ".ffn.down_proj_1.output_max"); + assert(ffn_down_1_output_max.size() == 1); + this->audio_ffn_down_1_output_max.push_back(ffn_down_1_output_max[0]); + + buffer ffn_down_1_output_min; + q4nx.load_weights(ffn_down_1_output_min, + "model.audio."+std::to_string(layer_id)+ ".ffn.down_proj_1.output_min"); + assert(ffn_down_1_output_min.size() == 1); + this->audio_ffn_down_1_output_min.push_back(ffn_down_1_output_min[0]); + + q4nx.load_weights( + this->ffn_norm_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".ffn_norm.weight" + ); + q4nx.load_weights( + this->ffn_norm_1_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".ffn_norm_1.weight" + ); + q4nx.load_weights( + this->ffn_post_norm_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".ffn_post_norm.weight" + ); + //NOTE: Optimization in multiple ffn_post_norm_weight with this.Gemma4E_Audio_residual_weight + for(int i = 0; i < this->ffn_post_norm_weight[layer_id].size(); i++){ + this->ffn_post_norm_weight[layer_id][i] = this->ffn_post_norm_weight[layer_id][i] * this->Gemma4E_Audio_residual_weight; + } + + q4nx.load_weights( + this->ffn_post_norm_1_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".ffn_post_norm_1.weight" + ); + //NOTE: Optimization in multiple ffn_post_norm_1_weight with this.Gemma4E_Audio_residual_weight + for(int i = 0; i < this->ffn_post_norm_1_weight[layer_id].size(); i++){ + this->ffn_post_norm_1_weight[layer_id][i] = this->ffn_post_norm_1_weight[layer_id][i] * this->Gemma4E_Audio_residual_weight; + } + + this->audio_ffn_up_weight[layer_id] = this->ffn_up_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + q4nx.load_weights( + this->audio_ffn_up_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".ffn.up_proj.weight" + ); + + buffer ffn_up_input_max; + q4nx.load_weights(ffn_up_input_max, + "model.audio."+std::to_string(layer_id)+ ".ffn.up_proj.input_max"); + assert(ffn_up_input_max.size() == 1); + this->audio_ffn_up_input_max.push_back(ffn_up_input_max[0]); + + buffer ffn_up_input_min; + q4nx.load_weights(ffn_up_input_min, + "model.audio."+std::to_string(layer_id)+ ".ffn.up_proj.input_min"); + assert(ffn_up_input_min.size() == 1); + this->audio_ffn_up_input_min.push_back(ffn_up_input_min[0]); + + buffer ffn_up_output_max; + q4nx.load_weights(ffn_up_output_max, + "model.audio."+std::to_string(layer_id)+ ".ffn.up_proj.output_max"); + assert(ffn_up_output_max.size() == 1); + this->audio_ffn_up_output_max.push_back(ffn_up_output_max[0]); + + buffer ffn_up_output_min; + q4nx.load_weights(ffn_up_output_min, + "model.audio."+std::to_string(layer_id)+ ".ffn.up_proj.output_min"); + assert(ffn_up_output_min.size() == 1); + this->audio_ffn_up_output_min.push_back(ffn_up_output_min[0]); + + this->audio_ffn_up_1_weight[layer_id] = this->ffn_up_proj_app.create_bo_buffer( + Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + q4nx.load_weights( + this->audio_ffn_up_1_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".ffn.up_proj_1.weight" + ); + buffer ffn_up_1_input_max; + q4nx.load_weights(ffn_up_1_input_max, + "model.audio."+std::to_string(layer_id)+ ".ffn.up_proj_1.input_max"); + assert(ffn_up_1_input_max.size() == 1); + this->audio_ffn_up_1_input_max.push_back(ffn_up_1_input_max[0]); + + buffer ffn_up_1_input_min; + q4nx.load_weights(ffn_up_1_input_min, + "model.audio."+std::to_string(layer_id)+ ".ffn.up_proj_1.input_min"); + assert(ffn_up_1_input_min.size() == 1); + this->audio_ffn_up_1_input_min.push_back(ffn_up_1_input_min[0]); + + buffer ffn_up_1_output_max; + q4nx.load_weights(ffn_up_1_output_max, + "model.audio."+std::to_string(layer_id)+ ".ffn.up_proj_1.output_max"); + assert(ffn_up_1_output_max.size() == 1); + this->audio_ffn_up_1_output_max.push_back(ffn_up_1_output_max[0]); + + buffer ffn_up_1_output_min; + q4nx.load_weights(ffn_up_1_output_min, + "model.audio."+std::to_string(layer_id)+ ".ffn.up_proj_1.output_min"); + assert(ffn_up_1_output_min.size() == 1); + this->audio_ffn_up_1_output_min.push_back(ffn_up_1_output_min[0]); + + q4nx.load_weights( + this->norm2_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".ln2.weight" + ); + + q4nx.load_weights( + this->per_dim_scale_with_softplus_weight[layer_id], + "model.audio."+std::to_string(layer_id)+".pre_dim_scale.weight" + ); + // Any optimization, multiple per_dim_scale with self.q_scale at load weights + for(int i = 0; i < this->per_dim_scale_with_softplus_weight[layer_id].size(); i++){ + this->per_dim_scale_with_softplus_weight[layer_id][i] = this->per_dim_scale_with_softplus_weight[layer_id][i] * Gemma4E_Audio_q_scale; + } + } +} + +std::vector Gemma4e_AudioEncoder::encode(void* audio_payload_ptr){ + + DEBUG_BLOCK(1, + std::cout << "Gemma4e_AudioEncoder::encode called with audio_payload_ptr: " << audio_payload_ptr << std::endl; + ) + + //DEBUG + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + SafeTensors reference_tensors( + this->model_path + "/audio_reference_data.safetensors" + ); + + #endif + + gemma4e_audio_payload_t* audio_payload = static_cast(audio_payload_ptr); + assert(audio_payload != nullptr); + + /* + // compare the weights + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "DEBUG: Start computing error metrics for audio encoder weights" << std::endl; + buffer reference_conv_weight_0; + buffer reference_conv_weight_1; + reference_tensors.load_weights(reference_conv_weight_0,"2D_convolution_weights_0"); + reference_tensors.load_weights(reference_conv_weight_1,"2D_convolution_weights_1"); + + assert(audio_subsample_conv2d_weight_0.size() == reference_conv_weight_0.size()); + assert(audio_subsample_conv2d_weight_1.size() == reference_conv_weight_1.size()); + print_error_metrics( + this->audio_subsample_conv2d_weight_0.data(), reference_conv_weight_0.data(), + 1, + this->audio_subsample_conv2d_weight_0.size(),1, + this->audio_subsample_conv2d_weight_0.size(),1 + ); + print_error_metrics( + this->audio_subsample_conv2d_weight_1.data(), reference_conv_weight_1.data(), + 1, + this->audio_subsample_conv2d_weight_1.size(), 1, + this->audio_subsample_conv2d_weight_1.size(), 1 + ); + } + #endif + */ + + // now, compare with Gemma4AudioSubSampleConvProjection_hidden_states_before_conv + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "DEBUG: Start computing error metrics for audio encoder input" << std::endl; + buffer reference_projection_input; + reference_tensors.load_weights(reference_projection_input,"Gemma4AudioSubSampleConvProjection_hidden_states_before_conv"); + size_t reference_projection_input_num_elements = reference_projection_input.size() / audio_payload->num_audios ; + + for(int i = 0; i < audio_payload->num_audios; i++){ + print_error_metrics( + audio_payload->mel_spectrograms[i].data(), + reference_projection_input.data() + i* reference_projection_input_num_elements , + 1, + audio_payload->mel_spectrogram_frames_per_audio[i], audio_payload->mel_spectrogram_bins_per_audio[i], + audio_payload->mel_spectrogram_frames_per_audio[i], audio_payload->mel_spectrogram_bins_per_audio[i] + ); + } + + //TODO: FIXME: remove it later + for(int i = 0; i < audio_payload->num_audios; i++){ + + for(int l = 0; l< audio_payload->mel_spectrogram_frames_per_audio[i]*audio_payload->mel_spectrogram_bins_per_audio[i]; l++ ){ + audio_payload->mel_spectrograms[i][l] = (bf16)(reference_projection_input[i* reference_projection_input_num_elements + l]); + } + } + } + #endif + + // sanity checks, ensure to be the same + int initial_audio_bins = audio_payload->mel_spectrogram_bins_per_audio[0]; + for(int i = 1; i < audio_payload->num_audios; i++){ + assert(initial_audio_bins == audio_payload->mel_spectrogram_bins_per_audio[i]); + } + + // recall the equation for calculation conv2d output as the following + //H_out = (H_in + 2*padding - K) / stride + 1 + //W_out = (W_in + 2*padding - K) / stride + 1 + auto calc_conv2d_out = [](int h_in, int padding, int k, int stride) -> int { + return (h_in + 2 * padding - k) / stride + 1; + }; + + std::vector> subSample_conv_layer_0_res( audio_payload->num_audios); + //does the first SubSample convlution operation + int audio_bin_after_conv = calc_conv2d_out( + initial_audio_bins, this->parent_npu_ptr->Gemma4e_Audio_conv2d_Padding, + this->parent_npu_ptr->Gemma4E_Audio_conv2d_kernel_size, this->parent_npu_ptr->Gemma4E_Audio_conv2d_Stride + ); + std::vector audio_frames_after_conv2d_0(audio_payload->num_audios); + + { + + for(int i = 0; i num_audios; i++){ + audio_frames_after_conv2d_0[i] = calc_conv2d_out( + audio_payload->mel_spectrogram_frames_per_audio[i], this->parent_npu_ptr->Gemma4e_Audio_conv2d_Padding, + this->parent_npu_ptr->Gemma4E_Audio_conv2d_kernel_size, this->parent_npu_ptr->Gemma4E_Audio_conv2d_Stride + ); + std::vector conv_out( + this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0 * audio_frames_after_conv2d_0[i] * audio_bin_after_conv + ); + simd_conv2d( + audio_payload->mel_spectrograms[i].data(), + this->audio_subsample_conv2d_weight_0.data(), + conv_out.data(), + 1, + audio_payload->mel_spectrogram_frames_per_audio[i], audio_payload->mel_spectrogram_bins_per_audio[i], + this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0, + this->parent_npu_ptr->Gemma4E_Audio_conv2d_kernel_size, + this->parent_npu_ptr->Gemma4E_Audio_conv2d_Stride, + this->parent_npu_ptr->Gemma4e_Audio_conv2d_Padding + + ); + // scalar_conv2d( + // audio_payload->mel_spectrograms[i].data(), + // this->audio_subsample_conv2d_weight_0.data(), + // conv_out.data(), + // 1, + // audio_payload->mel_spectrogram_frames_per_audio[i], audio_payload->mel_spectrogram_bins_per_audio[i], + // this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0, + // this->parent_npu_ptr->Gemma4E_Audio_conv2d_kernel_size, + // this->parent_npu_ptr->Gemma4E_Audio_conv2d_Stride, + // this->parent_npu_ptr->Gemma4e_Audio_conv2d_Padding + // ); + subSample_conv_layer_0_res[i] = std::move(conv_out); + } + + /* + // #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + // { + + // size_t reference_projection_input_num_elements_per_chanel = reference_projection_input_num_elements / this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0; + + // print_error_metrics( + // subSample_conv_layer_0_res[i].data() + c* audio_frames_after_conv2d_0[i]* audio_bin_after_conv, + // Gemma4AudioSubSampleConvProjectionLayer_0_hidden_states_after_conv.data() + i* reference_projection_input_num_elements + c* reference_projection_input_num_elements_per_chanel, + // 1, + // audio_frames_after_conv2d_0[i], audio_bin_after_conv, + // audio_frames_after_conv2d_0[i], audio_bin_after_conv + // ); + + // } + + // } + + // // //TODO: FIXME: + // // for(int i = 0; i < audio_payload->num_audios; i++){ + // // for(int c = 0; c parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0; c++ ){ + // // memcpy( + // // subSample_conv_layer_0_res[i].data() + c* audio_frames_after_conv2d_0[i]* audio_bin_after_conv, + // // Gemma4AudioSubSampleConvProjectionLayer_0_hidden_states_after_conv.data() + i* reference_projection_input_num_elements + c* reference_projection_input_num_elements_per_chanel, + // // audio_frames_after_conv2d_0[i]* audio_bin_after_conv * sizeof(bf16) + // // ); + // // } + + // // } + + // } + + // #endif + + */ + + // now, perform the reorder with layernorm + // Python reference does: act(norm(hidden_states.permute(0,2,3,1)).permute(0,3,1,2)) + // Instead of permuting NCHW→NHWC, norming, permuting back, we apply LayerNorm + // directly over the channel dimension (stride = H*W) in NCHW layout, then ReLU. + + for(int i = 0; i < audio_payload->num_audios; i++){ + int H = audio_frames_after_conv2d_0[i]; + int W = audio_bin_after_conv; + int HW = H * W; + bf16* data = subSample_conv_layer_0_res[i].data(); + + // LayerNorm over C channels at each (h,w), then ReLU, all in NCHW layout + layernorm_relu_nchw(data, this->audio_subsample_conv2d_norm_weight_0.data(), + this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0,HW, 1e-6f); + } + + // #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + // { + + // size_t reference_projection_input_num_elements_per_chanel = reference_projection_input_num_elements / this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0; + + // print_error_metrics( + // subSample_conv_layer_0_res[i].data() + c* audio_frames_after_conv2d_0[i]* audio_bin_after_conv, + // Gemma4AudioSubSampleConvProjectionLayer_0_hidden_states_after_act.data() + i* reference_projection_input_num_elements + c* reference_projection_input_num_elements_per_chanel, + // 1, + // audio_frames_after_conv2d_0[i], audio_bin_after_conv, + // audio_frames_after_conv2d_0[i], audio_bin_after_conv + // ); + + // } + + // } + + // } + // #endif + } + + //subSample_conv_layer_0_res is [num_audio, this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0, + // audio_frames_after_conv2d_0[i], audio_bin_after_conv ] + + // now, the second subSample_conv_layer_1 + int audio_bin_after_conv_1 = calc_conv2d_out( + audio_bin_after_conv, this->parent_npu_ptr->Gemma4e_Audio_conv2d_Padding, + this->parent_npu_ptr->Gemma4E_Audio_conv2d_kernel_size, this->parent_npu_ptr->Gemma4E_Audio_conv2d_Stride + ); + std::vector audio_frames_after_conv2d_1( audio_payload->num_audios); + std::vector> subSample_conv_layer_1_res( audio_payload->num_audios); + { + + for(int i = 0; i < audio_payload->num_audios; i++){ + audio_frames_after_conv2d_1[i] = calc_conv2d_out( + audio_frames_after_conv2d_0[i], this->parent_npu_ptr->Gemma4e_Audio_conv2d_Padding, + this->parent_npu_ptr->Gemma4E_Audio_conv2d_kernel_size, this->parent_npu_ptr->Gemma4E_Audio_conv2d_Stride + ); + + std::vector conv_out( + this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1 * audio_frames_after_conv2d_1[i] * audio_bin_after_conv_1 + ); + simd_conv2d( + subSample_conv_layer_0_res[i].data(), + this->audio_subsample_conv2d_weight_1.data(), + conv_out.data(), + this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0, + audio_frames_after_conv2d_0[i], audio_bin_after_conv, + this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1, + this->parent_npu_ptr->Gemma4E_Audio_conv2d_kernel_size, + this->parent_npu_ptr->Gemma4E_Audio_conv2d_Stride, + this->parent_npu_ptr->Gemma4e_Audio_conv2d_Padding + ); + + // scalar_conv2d( + // subSample_conv_layer_0_res[i].data(), + // this->audio_subsample_conv2d_weight_1.data(), + // conv_out.data(), + // this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0, + // audio_frames_after_conv2d_0[i], audio_bin_after_conv, + // this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1, + // this->parent_npu_ptr->Gemma4E_Audio_conv2d_kernel_size, + // this->parent_npu_ptr->Gemma4E_Audio_conv2d_Stride, + // this->parent_npu_ptr->Gemma4e_Audio_conv2d_Padding + // ); + subSample_conv_layer_1_res[i] = std::move(conv_out); + } + + // #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + // { + // buffer Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_conv; + // reference_tensors.load_weights(Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_conv,"Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_conv"); + // size_t reference_projection_input_num_elements = Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_conv.size() / audio_payload->num_audios ; + // std::cout << "DEBUG: Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_conv" << std::endl; + + // size_t reference_projection_input_num_elements_per_chanel = reference_projection_input_num_elements / this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1; + // for(int i = 0; i < audio_payload->num_audios; i++){ + // for(int c = 0; c parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1; c++ ){ + // print_error_metrics( + // subSample_conv_layer_1_res[i].data() + c* audio_frames_after_conv2d_1[i]* audio_bin_after_conv_1, + // Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_conv.data() + i* reference_projection_input_num_elements + c* reference_projection_input_num_elements_per_chanel, + // 1, + // audio_frames_after_conv2d_1[i], audio_bin_after_conv_1, + // audio_frames_after_conv2d_1[i], audio_bin_after_conv_1 + // ); + // } + // } + + // } + // #endif + + for(int i = 0; i < audio_payload->num_audios; i++){ + int H = audio_frames_after_conv2d_1[i]; + int W = audio_bin_after_conv_1; + int HW = H * W; + bf16* data = subSample_conv_layer_1_res[i].data(); + + // LayerNorm over C channels at each (h,w), then ReLU, all in NCHW layout + layernorm_relu_nchw(data, this->audio_subsample_conv2d_norm_weight_1.data(), + this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1,HW, 1e-6f); + } + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_act; + reference_tensors.load_weights(Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_act,"Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_act"); + size_t reference_projection_input_num_elements = Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_act.size() / audio_payload->num_audios ; + std::cout << "DEBUG: Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_act" << std::endl; + + size_t reference_projection_input_num_elements_per_chanel = reference_projection_input_num_elements / this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1; + + for(int i = 0; i < audio_payload->num_audios; i++){ + for(int c = 0; c parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1; c++ ){ + print_error_metrics( + subSample_conv_layer_1_res[i].data() + c* audio_frames_after_conv2d_1[i]* audio_bin_after_conv_1, + Gemma4AudioSubSampleConvProjectionLayer_1_hidden_states_after_act.data() + i* reference_projection_input_num_elements + c* reference_projection_input_num_elements_per_chanel, + 1, + audio_frames_after_conv2d_1[i], audio_bin_after_conv_1, + audio_frames_after_conv2d_1[i], audio_bin_after_conv_1 + ); + } + } + } + #endif + } + + // now, subSample_conv_layer_1_res is shape of [ num_audio, this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1, audio_frames_after_conv2d_1[i], audio_bin_after_conv_1 ] + + // reorder subSample_conv_layer_1_res.permute(0, 2, 3, 1).contiguous().reshape(batch_size, seq_len, -1) + // aka seq_len = audio_frames_after_conv2d_1[i] + // the hidden_Size is audio_bin_after_conv_1 *this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1 + // the reorder output to audio_embedding_projection_input, which is [num_audio, seq_len_per_audio, Padded_Gemma4E_Audio_Multimodal_Output_SIZE ] + // NOTE: Padded_Gemma4E_Audio_Multimodal_Output_SIZE>= audio_bin_after_conv_1*this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1 == Gemma4E_Audio_HIDDEN_SIZE + + std::vector seq_len_per_audio(audio_payload->num_audios); + std::vector start_seq_len_index_per_audio(audio_payload->num_audios); + int seq_len = 0; + int seq_len_of_last_audio = 0; + + for(int i = 0; i < audio_payload->num_audios; i++){ + seq_len_per_audio[i] = audio_frames_after_conv2d_1[i] ; + + seq_len += seq_len_per_audio[i]; + seq_len_of_last_audio = seq_len_per_audio[i]; + + start_seq_len_index_per_audio[i] = seq_len - seq_len_per_audio[i]; + } + int seq_len_padded = 0; + + //TODO: FIXME: padding for conv1d + + seq_len_padded = round_up_to_multiple(seq_len, seq_len_pad_requirement_for_MM); + + assert(this->parent_npu_ptr->Gemma4E_Audio_Multimodal_Output_SIZE % MM_tile_K == 0); + assert(this->parent_npu_ptr->Gemma4E_Audio_Multimodal_Output_SIZE % MM_tile_N == 0); + assert(this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE == audio_bin_after_conv_1*this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1); + assert(this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_0/4 * this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1 == + this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + for(int i = 0; i < audio_payload->num_audios; i++) { + std::cout << "DEBUG: seq_len_per_audio[" << i << "]: " << seq_len_per_audio[i] << std::endl; + std::cout << "DEBUG: start_seq_len_index_per_audio[" << i << "]: " << start_seq_len_index_per_audio[i] << std::endl; + } + std::cout << "DEBUG: seq_len: " << seq_len << std::endl; + std::cout << "DEBUG: seq_len_padded: " << seq_len_padded << std::endl; + } + + #endif + + buffer audio_embedding_projection_input = this->sub_sampleConvProjection_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(audio_embedding_projection_input.data() + seq_len * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + audio_embedding_projection_input.sync_to_device(); + + buffer audio_embedding_projection_output = this->sub_sampleConvProjection_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(audio_embedding_projection_output.data() + seq_len * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + audio_embedding_projection_output.sync_to_device(); + + buffer ffn_up_proj_input = this->ffn_up_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(ffn_up_proj_input.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + ffn_up_proj_input.sync_to_device(); + + buffer ffn_up_proj_output_down_input = this->ffn_up_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE + ); + memset(ffn_up_proj_output_down_input.data() + seq_len * Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE * sizeof(bf16) + ); + ffn_up_proj_output_down_input.sync_to_device(); + + buffer ffn_down_proj_output = this->ffn_down_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(ffn_down_proj_output.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + ffn_down_proj_output.sync_to_device(); + + buffer q_proj_input = this->q_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(q_proj_input.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + q_proj_input.sync_to_device(); + + buffer q_proj_output = this->q_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(q_proj_output.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + q_proj_output.sync_to_device(); + + buffer k_proj_input = this->k_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(k_proj_input.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + k_proj_input.sync_to_device(); + + buffer k_proj_output = this->k_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(k_proj_output.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + k_proj_output.sync_to_device(); + + buffer v_proj_input = this->v_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(v_proj_input.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + v_proj_input.sync_to_device(); + + buffer v_proj_output = this->v_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(v_proj_output.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + v_proj_output.sync_to_device(); + + buffer o_output_proj_input = this->o_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(o_output_proj_input.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + o_output_proj_input.sync_to_device(); + + buffer o_output_proj_output = this->o_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(o_output_proj_output.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + o_output_proj_output.sync_to_device(); + + buffer conv1d_start_proj_input = this->conv1d_start_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(conv1d_start_proj_input.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + conv1d_start_proj_input.sync_to_device(); + + buffer conv1d_start_proj_output = this->conv1d_start_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE + ); + memset(conv1d_start_proj_output.data() + seq_len * Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE * sizeof(bf16) + ); + conv1d_start_proj_output.sync_to_device(); + + buffer audio_conv1d_input = this->conv1d_app.create_bo_buffer( + (seq_len_padded + audio_payload->num_audios* this->Gemma4E_Audio_padded_requirement_for_conv1d) * Padded_GEMMA4E_Audio_HIDDEN_SIZE + + ); + memset(audio_conv1d_input.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len + audio_payload->num_audios* this->Gemma4E_Audio_padded_requirement_for_conv1d) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + audio_conv1d_input.sync_to_device(); + + buffer audio_conv1d_output = this->conv1d_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + + ); + memset(audio_conv1d_output.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + audio_conv1d_output.sync_to_device(); + + buffer conv1d_end_proj_input = this->conv1d_end_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(conv1d_end_proj_input.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + conv1d_end_proj_input.sync_to_device(); + + buffer conv1d_end_proj_output = this->conv1d_end_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(conv1d_end_proj_output.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + conv1d_end_proj_output.sync_to_device(); + + buffer audio_pre_encode_input = this->audio_pre_encode_proj_app.create_bo_buffer( + seq_len_padded * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + memset(audio_pre_encode_input.data() + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 0, (seq_len_padded - seq_len) * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + ); + audio_pre_encode_input.sync_to_device(); + + buffer audio_pre_encode_output = this->audio_pre_encode_proj_app.create_bo_buffer( + seq_len_padded * Padded_Gemma4E_Audio_Multimodal_Output_SIZE + ); + memset(audio_pre_encode_output.data() + seq_len * Padded_Gemma4E_Audio_Multimodal_Output_SIZE, + 0, (seq_len_padded - seq_len) * Padded_Gemma4E_Audio_Multimodal_Output_SIZE * sizeof(bf16) + ); + audio_pre_encode_output.sync_to_device(); + + buffer audio_to_language_project_input = this->audio_to_language_proj_app.create_bo_buffer( + seq_len_padded * Padded_Gemma4E_Audio_Multimodal_Output_SIZE + ); + memset(audio_to_language_project_input.data() + seq_len * Padded_Gemma4E_Audio_Multimodal_Output_SIZE, + 0, (seq_len_padded - seq_len) * Padded_Gemma4E_Audio_Multimodal_Output_SIZE * sizeof(bf16) + ); + audio_to_language_project_input.sync_to_device(); + + buffer audio_to_language_project_output = this->audio_to_language_proj_app.create_bo_buffer( + seq_len_padded * parent_npu_ptr->Gemma4E_Audio_language_projection_output_size + ); + memset(audio_to_language_project_output.data() + seq_len * parent_npu_ptr->Gemma4E_Audio_language_projection_output_size, + 0, (seq_len_padded - seq_len) * parent_npu_ptr->Gemma4E_Audio_language_projection_output_size * sizeof(bf16) + ); + audio_to_language_project_output.sync_to_device(); + assert(parent_npu_ptr->Gemma4E_Audio_language_projection_output_size % MM_tile_N == 0); + + { + generate_mm_sequence( + *this->sub_sampleConvProjection_app.seq(), + seq_len_padded, this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, //no bias, no activation + 0, -10000.0, 1000000.0, // do not clamp on output + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + + generate_mm_sequence( + *this->audio_pre_encode_proj_app.seq(), + seq_len_padded, this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, this->Padded_Gemma4E_Audio_Multimodal_Output_SIZE, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + true, 0, //no bias, no activation + 0, -10000.0, 1000000.0, // do not clamp on output + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + generate_mm_sequence( + *this->audio_to_language_proj_app.seq(), + seq_len_padded, Padded_Gemma4E_Audio_Multimodal_Output_SIZE, parent_npu_ptr->Gemma4E_Audio_language_projection_output_size, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, //no bias, no activation + 0, -10000.0, 1000000.0, // do not clamp on output + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + } + + memset(audio_embedding_projection_input.data(), 0, audio_embedding_projection_input.size() * sizeof(bf16)); + + // reorder of subSample_conv_layer_1_res.permute(0, 2, 3, 1).contiguous().reshape(batch_size, seq_len, -1) + for(int i = 0; i < audio_payload->num_audios; i++){ + + for(int h = 0; h < audio_frames_after_conv2d_1[i]; h++){ + for(int w = 0; w < audio_bin_after_conv_1; w++){ + for(int c = 0; c < this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1; c++){ + size_t input_index = c* audio_frames_after_conv2d_1[i]* audio_bin_after_conv_1 + h* audio_bin_after_conv_1 + w; + + size_t output_index = start_seq_len_index_per_audio[i] * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE + h* this->Padded_GEMMA4E_Audio_HIDDEN_SIZE\ + + w* this->parent_npu_ptr->Gemma4E_Audio_subsampling_conv_channels_1 + c; + audio_embedding_projection_input[output_index] = subSample_conv_layer_1_res[i][input_index]; + } + } + } + } + // subSample_conv_layer_1_res is now shape of [ seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE ] + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioSubSampleConvProjection_hidden_states_before_linear; + reference_tensors.load_weights(Gemma4AudioSubSampleConvProjection_hidden_states_before_linear,"Gemma4AudioSubSampleConvProjection_hidden_states_before_linear"); + size_t reference_projection_input_num_elements = Gemma4AudioSubSampleConvProjection_hidden_states_before_linear.size() / audio_payload->num_audios ; + std::cout << "DEBUG: Gemma4AudioSubSampleConvProjection_hidden_states_before_linear" << std::endl; + + // for(int i = 0; i < audio_payload->num_audios; i++){ + // print_error_metrics( + // audio_embedding_projection_input.data() + start_seq_len_index_per_audio[i] * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + // Gemma4AudioSubSampleConvProjection_hidden_states_before_linear.data() + i* reference_projection_input_num_elements , + // 1, + // seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + // seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + // ); + // } + // //TODO:FIXME: remove it later + + // memcpy( + // seq_len_per_audio[i]* Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + } + #endif + + // debugt + assert(audio_embedding_projection_weight.size() == this->Padded_GEMMA4E_Audio_HIDDEN_SIZE* this->Padded_GEMMA4E_Audio_HIDDEN_SIZE); + DEBUG_BLOCK(1, + std::cout << "Padded_GEMMA4E_Audio_HIDDEN_SIZE : " << this->Padded_GEMMA4E_Audio_HIDDEN_SIZE << std::endl; + ) + + audio_embedding_projection_input.sync_to_device(); + this->audio_embedding_projection_weight.sync_to_device(); + sub_sampleConvProjection_app( audio_embedding_projection_input, audio_embedding_projection_weight, audio_embedding_projection_output); + audio_embedding_projection_output.sync_from_device(); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioSubSampleConvProjection_hidden_states_after_linear; + reference_tensors.load_weights(Gemma4AudioSubSampleConvProjection_hidden_states_after_linear,"Gemma4AudioSubSampleConvProjection_hidden_states_after_linear"); + size_t reference_projection_input_num_elements = Gemma4AudioSubSampleConvProjection_hidden_states_after_linear.size() / audio_payload->num_audios ; + std::cout << "DEBUG: Gemma4AudioSubSampleConvProjection_hidden_states_after_linear" << std::endl; + + // for(int i = 0; i < audio_payload->num_audios; i++){ + // print_error_metrics( + // audio_embedding_projection_output.data() + start_seq_len_index_per_audio[i] * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + // Gemma4AudioSubSampleConvProjection_hidden_states_after_linear.data() + i* reference_projection_input_num_elements , + // 1, + // seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + // seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + // ); + // } + } + #endif + + std::vector position_embedding(13 *this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE ); + generate_gemma4_audio_rotary_pos_emb( + this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + this->parent_npu_ptr->Gemma4E_Audio_attention_chunk_size, + this->parent_npu_ptr->Gemma4E_Audio_attention_context_left, + this->parent_npu_ptr->Gemma4E_Audio_attention_context_right, + position_embedding + ); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer ref_position_embeddings; + reference_tensors.load_weights(ref_position_embeddings,"position_embeddings"); + + std::cout << "DEBUG: position_embeddings" << std::endl; + + print_error_metrics( + position_embedding.data(), ref_position_embeddings.data(), + 1, + position_embedding.size(), 1, + position_embedding.size(), 1 + ); + } + #endif + + std::vector> audio_sliding_window_attention_mask(audio_payload->num_audios); + for(int i = 0; i < audio_payload->num_audios; i++){ + create_sliding_window_attention_mask( + seq_len_per_audio[i], + this->parent_npu_ptr->Gemma4E_Audio_attention_context_left-1, + this->parent_npu_ptr->Gemma4E_Audio_attention_context_right, + audio_sliding_window_attention_mask[i] + + ); + } + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer attention_mask_before_block_5d; + reference_tensors.load_weights(attention_mask_before_block_5d,"attention_mask_before_block_5d"); + + size_t reference_attention_mask_num_elements = attention_mask_before_block_5d.size() / audio_payload->num_audios ; + size_t reference_attention_mask_length = std::sqrt(reference_attention_mask_num_elements); + + std::cout << "DEBUG: attention_mask_before_block_5d" << std::endl; + + //NOTE: because the reference attention mask is >= each seq_len_per_audio[i] + // For ease of comparison, we create same mask size and load audio_sliding_windo_attention_mask to it + std::vector expanded_attention_mask(attention_mask_before_block_5d.size(), 0); + + for(int i = 0; i < audio_payload->num_audios; i++){ + int* mask_start_ptr = expanded_attention_mask.data() + i* reference_attention_mask_num_elements; + // Fill valid rows (0 to seq_len_per_audio[i]-1) from the C++ sliding window mask + for(int l = 0; l < seq_len_per_audio[i]; l++){ + memcpy( + mask_start_ptr + l* reference_attention_mask_length, + audio_sliding_window_attention_mask[i].data() + l* seq_len_per_audio[i], + seq_len_per_audio[i] * sizeof(int) + ); + } + // Fill padding rows (seq_len_per_audio[i] to reference_attention_mask_length-1) + // Python's create_bidirectional_mask doesn't mask queries, only keys. + // So padding rows still have 1s for valid columns within the sliding window. + int sliding_window_left = this->parent_npu_ptr->Gemma4E_Audio_attention_context_left - 1; + int sliding_window_right = this->parent_npu_ptr->Gemma4E_Audio_attention_context_right; + for(int l = seq_len_per_audio[i]; l < (int)reference_attention_mask_length; l++){ + for(int c = 0; c < seq_len_per_audio[i]; c++){ + int dist = l - c; + bool left_mask = (dist >= 0) && (dist < sliding_window_left); + bool right_mask = (dist < 0) && (-dist < sliding_window_right); + if(left_mask || right_mask){ + mask_start_ptr[l * reference_attention_mask_length + c] = 1; + } + } + } + } + print_error_metrics( + expanded_attention_mask.data(), attention_mask_before_block_5d.data(), + 1, + expanded_attention_mask.size(), 1, + expanded_attention_mask.size(), 1 + ); + } + #endif + + // now, convert to block attention mask + // [batch_Size, num_blocks, chunk_size, context_size] + std::vector> block_attention_mask_per_audio(audio_payload->num_audios); + std::vector num_blocks_per_audio(audio_payload->num_audios); + std::vector context_size_per_audio(audio_payload->num_audios); + for(int i = 0; i < audio_payload->num_audios; i++){ + convert_mask_to_blocked( + audio_sliding_window_attention_mask[i], + this->parent_npu_ptr->Gemma4E_Audio_attention_chunk_size, + this->parent_npu_ptr->Gemma4E_Audio_attention_context_left, + this->parent_npu_ptr->Gemma4E_Audio_attention_context_right, + block_attention_mask_per_audio[i], + num_blocks_per_audio[i], context_size_per_audio[i] + ); + } + + hidden_state.resize(seq_len_padded * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, 0.0f); + residual.resize(seq_len_padded * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, 0.0f); + + memcpy(hidden_state.data(), audio_embedding_projection_output.data(), seq_len * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16)); + for(int layer_id = 0; layer_id Gemma4E_Audio_num_attention_layers; layer_id++){ + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + //Gemma4AudioLayer_{layer_idx}_hidden_states_before_ffw1 + buffer Gemma4AudioLayer_hidden_State_before_ffw1; + reference_tensors.load_weights(Gemma4AudioLayer_hidden_State_before_ffw1,"Gemma4AudioLayer_"+std::to_string(layer_id)+"_hidden_states_before_ffw1"); + size_t reference_ffn_input_num_elements = Gemma4AudioLayer_hidden_State_before_ffw1.size() / audio_payload->num_audios ; + std::cout << "DEBUG: Gemma4AudioLayer_hidden_State_before_ffw1" << std::endl; + + for(int i = 0; i < audio_payload->num_audios; i++){ + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_audio[i] * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioLayer_hidden_State_before_ffw1.data() + i* reference_ffn_input_num_elements , + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + + // // TODO: FIXME: remove later + + // for(int i = 0; i < audio_payload->num_audios; i++){ + // memcpy( + // hidden_state.data() + start_seq_len_index_per_audio[i] * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + // Gemma4AudioLayer_hidden_State_before_ffw1.data() + i* reference_ffn_input_num_elements , + // seq_len_per_audio[i] * Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16) + // ); + // } + // memset(hidden_state.data() + seq_len * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, 0, + // (seq_len_padded - seq_len)* this->Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16)); + } + #endif + + ffn_layer( + ffn_up_proj_input, ffn_up_proj_output_down_input, ffn_down_proj_output, + ffn_norm_weight[layer_id], + ffn_post_norm_weight[layer_id], + audio_ffn_up_weight[layer_id], audio_ffn_down_weight[layer_id], + seq_len, seq_len_padded, + + audio_ffn_up_input_min[layer_id], audio_ffn_up_input_max[layer_id], + audio_ffn_up_output_min[layer_id], audio_ffn_up_output_max[layer_id], + audio_ffn_down_input_min[layer_id], audio_ffn_down_input_max[layer_id], + audio_ffn_down_output_min[layer_id], audio_ffn_down_output_max[layer_id] + + ); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioLayer_hidden_states_after_ffn1; + reference_tensors.load_weights(Gemma4AudioLayer_hidden_states_after_ffn1, + "Gemma4AudioLayer_"+std::to_string(layer_id)+"_hidden_states_after_ffw1"); + size_t reference_attn_output_num_elements = + Gemma4AudioLayer_hidden_states_after_ffn1.size() / audio_payload->num_audios; + std::cout << "DEBUG: Gemma4AudioLayer " +std::to_string(layer_id)+ "hidden_states_after_ffw1" << std::endl; + + for (int i = 0; i < audio_payload->num_audios; i++) { + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_audio[i] * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioLayer_hidden_states_after_ffn1.data() + i * reference_attn_output_num_elements, + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + + memcpy(residual.data(), hidden_state.data(), seq_len * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16)); + + simd_rms_norm( + hidden_state.data(), + this->attn_pre_norm_weight[layer_id].data(), + hidden_state.data(), + seq_len, + this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + + memcpy(q_proj_input.data(), hidden_state.data(), seq_len * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16)); + simd_clamp( + q_proj_input.data(), + q_proj_input.data(), + audio_q_input_min[layer_id], audio_q_input_max[layer_id], + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + + { + generate_mm_sequence( + *this->q_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, Padded_GEMMA4E_Audio_HIDDEN_SIZE, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, //no bias, no activation + 1, audio_q_output_min[layer_id], audio_q_output_max[layer_id], // do not clamp on output + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + } + + auto q_proj_run = q_proj_app.create_run( + q_proj_input, this->audio_attn_q_weight[layer_id], q_proj_output + ); + + q_proj_input.sync_to_device(); + this->audio_attn_q_weight[layer_id].sync_to_device(); + q_proj_run.start(); + + // setup for K + generate_mm_sequence( + *this->k_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, Padded_GEMMA4E_Audio_HIDDEN_SIZE, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, //no bias, no activation + 1, audio_k_output_min[layer_id], audio_k_output_max[layer_id], // do not clamp on output + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + memcpy(k_proj_input.data(), hidden_state.data(), seq_len * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16)); + simd_clamp( + k_proj_input.data(), + k_proj_input.data(), + audio_k_input_min[layer_id], audio_k_input_max[layer_id], + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + k_proj_input.sync_to_device(); + + auto k_proj_run = k_proj_app.create_run( + k_proj_input, this->audio_attn_k_weight[layer_id], k_proj_output + ); + + q_proj_run.wait(); + q_proj_output.sync_from_device(); + + k_proj_input.sync_to_device(); + this->audio_attn_k_weight[layer_id].sync_to_device(); + k_proj_run.start(); + + generate_mm_sequence( + *this->v_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, Padded_GEMMA4E_Audio_HIDDEN_SIZE, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, //no bias, no activation + 1, audio_v_output_min[layer_id], audio_v_output_max[layer_id], // do not clamp on output + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + memcpy(v_proj_input.data(), hidden_state.data(), seq_len * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE * sizeof(bf16)); + simd_clamp( + v_proj_input.data(), + v_proj_input.data(), + audio_v_input_min[layer_id], audio_v_input_max[layer_id], + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + v_proj_input.sync_to_device(); + auto v_proj_run = v_proj_app.create_run( + v_proj_input, this->audio_attn_v_weight[layer_id], v_proj_output + ); + k_proj_run.wait(); + k_proj_output.sync_from_device(); + + v_proj_input.sync_to_device(); + this->audio_attn_v_weight[layer_id].sync_to_device(); + v_proj_run.start(); + // At this point, q, k proj_output is shape of [num_audio, seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE] + // But can also be viewed as [num_audio, seq_len_per_audio[i], Padded_Gemma4E_Audio_num_attention_heads, Gemma4E_Audio_attention_head_dim] + for(int b = 0; b < audio_payload->num_audios; b++){ + + for(int s= 0; s< seq_len_per_audio[b]; s++){ + + for(int h_idx = 0; h_idx < parent_npu_ptr->Gemma4E_Audio_num_attention_heads; h_idx++){ + + size_t offset = (start_seq_len_index_per_audio[b]+s) * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE \ + + h_idx * Gemma4E_Audio_attention_head_dim; + simd_mul( + q_proj_output.data() + offset, + per_dim_scale_with_softplus_weight[layer_id].data(), + q_proj_output.data() + offset, + Gemma4E_Audio_attention_head_dim + ); + } + } + } + q_proj_output.sync_to_device(); + simd_mul( + k_proj_output.data(), + this->Gemma4E_Audio_k_scale, + k_proj_output.data(), + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + k_proj_output.sync_to_device(); + + // setup o_projection + generate_mm_sequence( + *this->o_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, Padded_GEMMA4E_Audio_HIDDEN_SIZE, + MM_tile_M, MM_tile_K, MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, //no bias, no activation + 1, audio_o_output_min[layer_id], audio_o_output_max[layer_id], // do not clamp on output + ENABLE_QKV_REORDER, 0// since we don't need it anymore + + ); + v_proj_run.wait(); + v_proj_output.sync_from_device(); + + // now, compare q, k, v with + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioAttention_query_states_after_scaling; + buffer Gemma4AudioAttention_key_states_after_scaling; + buffer Gemma4AudioAttention_value_states_after_scaling; + + reference_tensors.load_weights(Gemma4AudioAttention_query_states_after_scaling,"Gemma4AudioAttention_"+std::to_string(layer_id)+"_query_states_after_scaling"); + reference_tensors.load_weights(Gemma4AudioAttention_key_states_after_scaling,"Gemma4AudioAttention_"+std::to_string(layer_id)+"_key_states_after_scaling"); + reference_tensors.load_weights(Gemma4AudioAttention_value_states_after_scaling,"Gemma4AudioAttention_"+std::to_string(layer_id)+"_value_states_after_scaling"); + + size_t reference_qkv_num_elements = Gemma4AudioAttention_query_states_after_scaling.size() / audio_payload->num_audios ; + std::cout << "DEBUG: Gemma4AudioAttention_query_states_after_scaling" << std::endl; + + for(int i = 0; i < audio_payload->num_audios; i++){ + print_error_metrics( + q_proj_output.data() + start_seq_len_index_per_audio[i] * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioAttention_query_states_after_scaling.data() + i* reference_qkv_num_elements , + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + print_error_metrics( + k_proj_output.data() + start_seq_len_index_per_audio[i] * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioAttention_key_states_after_scaling.data() + i* reference_qkv_num_elements , + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + print_error_metrics( + v_proj_output.data() + start_seq_len_index_per_audio[i] * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioAttention_value_states_after_scaling.data() + i* reference_qkv_num_elements , + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + + int hidden_size = this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE; + int num_positions = (int)(position_embedding.size() / hidden_size); + + memset(o_output_proj_input.data(), 0, o_output_proj_input.size() * sizeof(bf16)); + + compute_audio_self_attention( + q_proj_output.data(), + k_proj_output.data(), + v_proj_output.data(), + this->audio_attn_k_rel_weight[layer_id].data(), + position_embedding.data(), + block_attention_mask_per_audio, + num_blocks_per_audio, + context_size_per_audio, + seq_len_per_audio, + start_seq_len_index_per_audio, + o_output_proj_input.data(), + audio_payload->num_audios, + this->parent_npu_ptr->Gemma4E_Audio_attention_chunk_size, + this->parent_npu_ptr->Gemma4E_Audio_attention_context_left, + this->parent_npu_ptr->Gemma4E_Audio_attention_context_right, + this->parent_npu_ptr->Gemma4E_Audio_num_attention_heads, + Gemma4E_Audio_attention_head_dim, + hidden_size, + Padded_GEMMA4E_Audio_HIDDEN_SIZE, + num_positions, + parent_npu_ptr->Gemma4E_Audio_attention_softcap, + -1e9f // invalid_logits_value + ); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioAttention_attn_output_before_post; + reference_tensors.load_weights(Gemma4AudioAttention_attn_output_before_post, + "Gemma4AudioAttention_"+std::to_string(layer_id)+"_attn_output_before_post"); + size_t reference_attn_output_num_elements = + Gemma4AudioAttention_attn_output_before_post.size() / audio_payload->num_audios; + std::cout << "DEBUG: Gemma4AudioAttention_attn_output_before_post" << std::endl; + + for (int i = 0; i < audio_payload->num_audios; i++) { + print_error_metrics( + o_output_proj_input.data() + start_seq_len_index_per_audio[i] * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioAttention_attn_output_before_post.data() + i * reference_attn_output_num_elements, + 1, + seq_len_per_audio[i], hidden_size, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + + // o_proj + simd_clamp( + o_output_proj_input.data(), + o_output_proj_input.data(), + audio_o_input_min[layer_id], audio_o_input_max[layer_id], + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + o_output_proj_input.sync_to_device(); + this->audio_attn_o_weight[layer_id].sync_to_device(); + o_proj_app(o_output_proj_input, this->audio_attn_o_weight[layer_id], o_output_proj_output); + o_output_proj_output.sync_from_device(); + + simd_rms_norm( + o_output_proj_output.data(), + this->attn_post_norm_weight[layer_id].data(), + hidden_state.data(), + seq_len, + this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 1e-6f + ); + + simd_add(hidden_state.data(), residual.data(), hidden_state.data(), + seq_len * Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + + // now compare it + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioAttention_hidden_states_before_conv1d; + reference_tensors.load_weights(Gemma4AudioAttention_hidden_states_before_conv1d, + "Gemma4AudioLayer_"+std::to_string(layer_id)+"_hidden_states_before_conv1d"); + size_t reference_attn_output_num_elements = + Gemma4AudioAttention_hidden_states_before_conv1d.size() / audio_payload->num_audios; + std::cout << "DEBUG: Gemma4AudioLayer__hidden_states_before_conv1d" << std::endl; + + for (int i = 0; i < audio_payload->num_audios; i++) { + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_audio[i] * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioAttention_hidden_states_before_conv1d.data() + i * reference_attn_output_num_elements, + 1, + seq_len_per_audio[i], hidden_size, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + conv1d_layer( + layer_id, + seq_len, seq_len_padded, + seq_len_per_audio, start_seq_len_index_per_audio, + conv1d_start_proj_input, conv1d_start_proj_output, + audio_conv1d_input, audio_conv1d_output, + conv1d_end_proj_input, conv1d_end_proj_output, + this->audio_conv_pw1_input_min[layer_id], this->audio_conv_pw1_input_max[layer_id], + this->audio_conv_pw1_output_min[layer_id], this->audio_conv_pw1_output_max[layer_id], + this->audio_conv_pw2_input_min[layer_id], this->audio_conv_pw2_input_max[layer_id], + this->audio_conv_pw2_output_min[layer_id], this->audio_conv_pw2_output_max[layer_id], + audio_payload, + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + &reference_tensors + #else + nullptr + #endif + ); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioLayer_hidden_states_after_conv1d; + reference_tensors.load_weights(Gemma4AudioLayer_hidden_states_after_conv1d, + "Gemma4AudioLayer_"+std::to_string(layer_id)+"_hidden_states_after_conv1d"); + size_t reference_attn_output_num_elements = + Gemma4AudioLayer_hidden_states_after_conv1d.size() / audio_payload->num_audios; + std::cout << "DEBUG: Gemma4AudioLayer_hidden_states_after_conv1d" << std::endl; + + for (int i = 0; i < audio_payload->num_audios; i++) { + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_audio[i] * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioLayer_hidden_states_after_conv1d.data() + i * reference_attn_output_num_elements, + 1, + seq_len_per_audio[i], hidden_size, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + + ffn_layer( + + ffn_up_proj_input, ffn_up_proj_output_down_input, ffn_down_proj_output, + ffn_norm_1_weight[layer_id], + ffn_post_norm_1_weight[layer_id], + audio_ffn_up_1_weight[layer_id], audio_ffn_down_1_weight[layer_id], + seq_len, seq_len_padded, + + audio_ffn_up_1_input_min[layer_id], audio_ffn_up_1_input_max[layer_id], + audio_ffn_up_1_output_min[layer_id], audio_ffn_up_1_output_max[layer_id], + audio_ffn_down_1_input_min[layer_id], audio_ffn_down_1_input_max[layer_id], + audio_ffn_down_1_output_min[layer_id], audio_ffn_down_1_output_max[layer_id] + + ); + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioLayer_hidden_states_after_ffn1; + reference_tensors.load_weights(Gemma4AudioLayer_hidden_states_after_ffn1, + "Gemma4AudioLayer_"+std::to_string(layer_id)+"_hidden_states_after_ffw2"); + size_t reference_attn_output_num_elements = + Gemma4AudioLayer_hidden_states_after_ffn1.size() / audio_payload->num_audios; + std::cout << "DEBUG: Gemma4AudioLayer_hidden_states_after_ffw2" << std::endl; + + for (int i = 0; i < audio_payload->num_audios; i++) { + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_audio[i] * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioLayer_hidden_states_after_ffn1.data() + i * reference_attn_output_num_elements, + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + + simd_rms_norm( + hidden_state.data(), norm2_weight[layer_id].data(), hidden_state.data(), + seq_len, this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, seq_len_padded, Padded_GEMMA4E_Audio_HIDDEN_SIZE, + 1e-6f + ); + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioLayer_hidden_states_after_norm_out; + reference_tensors.load_weights(Gemma4AudioLayer_hidden_states_after_norm_out, + "Gemma4AudioLayer_"+std::to_string(layer_id)+"_hidden_states_after_norm_out"); + size_t reference_attn_output_num_elements = + Gemma4AudioLayer_hidden_states_after_norm_out.size() / audio_payload->num_audios; + std::cout << "DEBUG: Gemma4AudioLayer" + std::to_string(layer_id)+ "_hidden_states_after_norm_out" << std::endl; + + for (int i = 0; i < audio_payload->num_audios; i++) { + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_audio[i] * Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioLayer_hidden_states_after_norm_out.data() + i * reference_attn_output_num_elements, + 1, + seq_len_per_audio[i], hidden_size, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + } + + memcpy(audio_pre_encode_input.data(), hidden_state.data(), + seq_len* this->Padded_GEMMA4E_Audio_HIDDEN_SIZE* sizeof(bf16) + ); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioPreEncodeProjection_hidden_states_before_linear; + reference_tensors.load_weights(Gemma4AudioPreEncodeProjection_hidden_states_before_linear,"hidden_states_before_output_proj"); + size_t reference_projection_input_num_elements = Gemma4AudioPreEncodeProjection_hidden_states_before_linear.size() / audio_payload->num_audios ; + std::cout << "DEBUG: hidden_states_before_output_proj" << std::endl; + + for(int i = 0; i < audio_payload->num_audios; i++){ + print_error_metrics( + audio_pre_encode_input.data() + start_seq_len_index_per_audio[i] * this->Padded_GEMMA4E_Audio_HIDDEN_SIZE, + Gemma4AudioPreEncodeProjection_hidden_states_before_linear.data() + i* reference_projection_input_num_elements , + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_HIDDEN_SIZE, + seq_len_per_audio[i], Padded_GEMMA4E_Audio_HIDDEN_SIZE + ); + } + } + #endif + + audio_pre_encode_input.sync_to_device(); + this->audio_pre_encode_weight.sync_to_device(); + audio_pre_encode_proj_app( audio_pre_encode_input, audio_pre_encode_weight, audio_pre_encode_output); + audio_pre_encode_output.sync_from_device(); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer Gemma4AudioPreEncodeProjection_output; + reference_tensors.load_weights(Gemma4AudioPreEncodeProjection_output,"audio_model_output_proj_result"); + size_t reference_projection_input_num_elements = Gemma4AudioPreEncodeProjection_output.size() / audio_payload->num_audios ; + std::cout << "DEBUG: audio_model_output_proj_result" << std::endl; + + for(int i = 0; i < audio_payload->num_audios; i++){ + print_error_metrics( + audio_pre_encode_output.data() + start_seq_len_index_per_audio[i] * this->Padded_Gemma4E_Audio_Multimodal_Output_SIZE, + Gemma4AudioPreEncodeProjection_output.data() + i* reference_projection_input_num_elements , + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_Multimodal_Output_SIZE, + seq_len_per_audio[i], Padded_Gemma4E_Audio_Multimodal_Output_SIZE + ); + } + } + #endif + + simd_rms_norm( + audio_pre_encode_output.data(), + audio_to_language_project_input.data(), + seq_len, this->parent_npu_ptr->Gemma4E_Audio_Multimodal_Output_SIZE, seq_len_padded, Padded_Gemma4E_Audio_Multimodal_Output_SIZE, + 1e-6f + ); + audio_pre_encode_output.sync_to_device(); + + audio_to_language_project_input.sync_to_device(); + this->audio_to_language_projection_weight.sync_to_device(); + audio_to_language_proj_app(audio_to_language_project_input, this->audio_to_language_projection_weight, audio_to_language_project_output); + audio_to_language_project_output.sync_from_device(); + + #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + { + buffer audio_to_language_projection_output_ref; + reference_tensors.load_weights(audio_to_language_projection_output_ref,"audio_final_embs_after_embed_audio"); + size_t reference_projection_input_num_elements = audio_to_language_projection_output_ref.size() / audio_payload->num_audios ; + std::cout << "DEBUG: audio_final_embs_after_embed_audio" << std::endl; + + for(int i = 0; i < audio_payload->num_audios; i++){ + print_error_metrics( + audio_to_language_project_output.data() + start_seq_len_index_per_audio[i] * this->parent_npu_ptr->Gemma4E_Audio_language_projection_output_size, + audio_to_language_projection_output_ref.data() + i* reference_projection_input_num_elements , + 1, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_language_projection_output_size, + seq_len_per_audio[i], this->parent_npu_ptr->Gemma4E_Audio_language_projection_output_size + ); + } + } + #endif + + std::vector dummy_output(audio_to_language_project_output.size()); // return half of the hidden size as dummy output + memcpy(dummy_output.data(), audio_to_language_project_output.data(), audio_to_language_project_output.size() * sizeof(bf16)); + return dummy_output; +} + diff --git a/src/detail/gemma4e_npu/gemma4e_audio.hpp b/src/detail/gemma4e_npu/gemma4e_audio.hpp new file mode 100644 index 000000000..17fb5a430 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_audio.hpp @@ -0,0 +1,199 @@ +#pragma once +#include "typedef.hpp" +#include +#include +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/gemma4e/gemma4e_npu.hpp" + +#include "vision/norm.hpp" + +class Gemma4e_AudioEncoder{ + public: + ~Gemma4e_AudioEncoder(); + void init_weights(SafeTensors &q4nx); + Gemma4e_AudioEncoder(LM_Config config, npu_xclbin_manager *npu_instance, gemma4e_npu* parent_npu_ptr); + + void ffn_layer( + + buffer &ffn_up_proj_input, + buffer &ffn_up_proj_output_down_input, + buffer &ffn_down_proj_output, + + buffer &cur_ffn_norm_weight, // ffn_norm or ffn_norm_1 + buffer &cur_ffn_post_norm_weight,// ffn_post_norm_weight or ffn_post_norm_1_weight + buffer &cur_ffn_up_weight, // audio_ffn_up_weight or audio_ffn_up_1_weight + buffer &cur_ffn_down_weight, // audio_ffn_down_weight + int seq_len, int seq_len_padded, + + bf16 cur_audio_ffn_up_input_min, bf16 cur_audio_ffn_up_input_max, + bf16 cur_audio_ffn_up_output_min, bf16 cur_audio_ffn_up_output_max, + bf16 cur_audio_ffn_down_input_min, bf16 cur_audio_ffn_down_input_max, + bf16 cur_audio_ffn_down_output_min, bf16 cur_audio_ffn_down_output_max + + ); + void conv1d_layer( + int layer_idx, + int seq_len, int seq_len_padded, + std::vector &seq_len_per_audio, std::vector &start_seq_len_index_per_audio, + buffer &conv1d_start_proj_input, buffer &conv1d_start_proj_output, + buffer &conv1d_input, buffer &conv1d_output, + buffer &conv1d_end_proj_input, buffer &conv1d_end_proj_output, + + bf16 conv1d_start_input_min, bf16 conv1d_start_input_max, + bf16 conv1d_start_output_min, bf16 conv1d_start_output_max, + bf16 conv1d_end_input_min, bf16 conv1d_end_input_max, + bf16 conv1d_end_output_min, bf16 conv1d_end_output_max, + + gemma4e_audio_payload_t* audio_payload, + SafeTensors *reference_safetensor + ); + + std::vector encode( void* audio_payload_ptr); + + LM_Config config; + npu_xclbin_manager *npu; + gemma4e_npu* parent_npu_ptr; + + uint32_t MM_tile_M; + uint32_t MM_tile_K; + uint32_t MM_tile_N; + + float Gemma4E_Audio_residual_weight; + float Gemma4E_Audio_q_scale; + float Gemma4E_Audio_k_scale; + int Gemma4E_Audio_attention_head_dim; + int Gemma4E_Audio_padded_requirement_for_conv1d; + + uint32_t seq_len_pad_requirement_for_MM; + uint32_t MM_ROW_SIZE = 4; + uint32_t MM_COL_SIZE = 8; + + unsigned int Padded_GEMMA4E_Audio_HIDDEN_SIZE; + unsigned int Padded_GEMMA4E_Audio_MLP_INTERMEDIATE_SIZE; + unsigned int Padded_Gemma4E_Audio_Multimodal_Output_SIZE; + unsigned int Padded_GEMMA4E_Audio_Conv1d_Linear_OUTPUT_SIZE; + unsigned int Padded_Gemma4E_Audio_num_attention_heads; + + inline int round_up_to_multiple (int x, int multiple) + { + return ((x + multiple - 1) / multiple) * multiple; + }; + + // no reorder for MM + bool ENABLE_QKV_REORDER = false; + + // for MM runtime sequence + uint32_t rtp_address = 4096; // offset right after stack size + uint32_t rtp_sync_lock_id = 10; // the rtp sync lock + bool ENABLE_AXI4 = true; + bool IS_B_ROW_MAJOR = false; + + std::string model_path; + npu_app_manager *proj; + npu_app_manager *proj_high_precision; + npu_app_manager *conv1d; + + npu_app conv1d_app; + + npu_app sub_sampleConvProjection_app; + npu_app q_proj_app; + npu_app k_proj_app; + npu_app k_relative_proj_app; + npu_app v_proj_app; + npu_app o_proj_app; + + npu_app ffn_down_proj_app; // share for both first and second FFN in the layer + npu_app ffn_up_proj_app; + npu_app conv1d_start_proj_app; + npu_app conv1d_end_proj_app; + npu_app audio_pre_encode_proj_app; + npu_app audio_to_language_proj_app; + + std::vector residual; + std::vector hidden_state; + + std::vector q_blocks; + std::vector k_blocks; + std::vector v_blocks; + + // the weights + buffer audio_embedding_projection_weight; + + buffer audio_subsample_conv2d_weight_0; + buffer audio_subsample_conv2d_norm_weight_0; + buffer audio_subsample_conv2d_weight_1; + buffer audio_subsample_conv2d_norm_weight_1; + + buffer audio_pre_encode_weight; + buffer audio_to_language_projection_weight; + // weights for each layer + std::vector> audio_attn_k_weight; + std::vector> audio_attn_k_rel_weight; + std::vector> audio_attn_v_weight; + std::vector> audio_attn_q_weight; + std::vector> audio_attn_o_weight; + std::vector> audio_conv1d_weight; + + std::vector> audio_conv_pw_1_weight; + std::vector> audio_conv_pw_2_weight; + + std::vector> audio_ffn_down_weight; + std::vector> audio_ffn_up_weight; + std::vector> audio_ffn_down_1_weight; + std::vector> audio_ffn_up_1_weight; + + std::vector> attn_post_norm_weight; + std::vector> attn_pre_norm_weight; + std::vector> conv_norm_weight; + std::vector> norm_conv_weight; + std::vector> ffn_norm_weight; + std::vector> ffn_norm_1_weight; + std::vector> ffn_post_norm_weight; + std::vector> ffn_post_norm_1_weight; + std::vector> norm2_weight; + std::vector> per_dim_scale_with_softplus_weight; + + std::vector audio_k_input_min; + std::vector audio_k_input_max; + std::vector audio_v_input_min; + std::vector audio_v_input_max; + std::vector audio_q_input_min; + std::vector audio_q_input_max; + std::vector audio_o_input_min; + std::vector audio_o_input_max; + std::vector audio_conv_pw1_input_min; + std::vector audio_conv_pw1_input_max; + std::vector audio_conv_pw2_input_min; + std::vector audio_conv_pw2_input_max; + std::vector audio_ffn_down_input_min; + std::vector audio_ffn_down_input_max; + std::vector audio_ffn_up_input_min; + std::vector audio_ffn_up_input_max; + std::vector audio_ffn_down_1_input_min; + std::vector audio_ffn_down_1_input_max; + std::vector audio_ffn_up_1_input_min; + std::vector audio_ffn_up_1_input_max; + + std::vector audio_k_output_min; + std::vector audio_k_output_max; + std::vector audio_v_output_min; + std::vector audio_v_output_max; + std::vector audio_q_output_min; + std::vector audio_q_output_max; + std::vector audio_o_output_min; + std::vector audio_o_output_max; + std::vector audio_conv1d_output_min; + std::vector audio_conv1d_output_max; + std::vector audio_conv_pw1_output_min; + std::vector audio_conv_pw1_output_max; + std::vector audio_conv_pw2_output_min; + std::vector audio_conv_pw2_output_max; + std::vector audio_ffn_down_output_min; + std::vector audio_ffn_down_output_max; + std::vector audio_ffn_up_output_min; + std::vector audio_ffn_up_output_max; + std::vector audio_ffn_down_1_output_min; + std::vector audio_ffn_down_1_output_max; + std::vector audio_ffn_up_1_output_min; + std::vector audio_ffn_up_1_output_max; +}; diff --git a/src/detail/gemma4e_npu/gemma4e_audio_attention.cpp b/src/detail/gemma4e_npu/gemma4e_audio_attention.cpp new file mode 100644 index 000000000..21b912d35 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_audio_attention.cpp @@ -0,0 +1,440 @@ +#include "gemma4e_audio_attention.hpp" +#include "gemma4e_vision_prefill_helper.hpp" +#include "avx512_util.hpp" +#include +#include +#include +#include +#include +#include + +void create_sliding_window_attention_mask( + int seq_len, + int sliding_window_left, + int sliding_window_right, + std::vector &mask + +){ + mask.assign(seq_len * seq_len, 0); + for (int q_idx = 0; q_idx < seq_len; ++q_idx) { + for (int kv_idx = 0; kv_idx < seq_len; ++kv_idx) { + int dist = q_idx - kv_idx; + bool left_mask = (dist >= 0) && (dist < sliding_window_left); + bool right_mask = (dist < 0) && (-dist < sliding_window_right); + + if (left_mask || right_mask) { + mask[q_idx * seq_len + kv_idx] = 1; + } else { + mask[q_idx * seq_len + kv_idx] = 0; + } + } + } +} +void convert_mask_to_blocked( + std::vector &input_mask, // [seq_len, seq_len] + int chunk_size, + int attention_context_left, + int attention_context_right, + std::vector &blocked_mask, // [num_chunks, chunk_size, context_size], + int &num_blocks, + int &context_size +) { + int seq_len = std::round(std::sqrt(input_mask.size())); + int max_past_horizon = attention_context_left - 1; + int max_future_horizon = attention_context_right; + + num_blocks = (seq_len + chunk_size - 1) / chunk_size; + context_size = chunk_size + max_past_horizon + max_future_horizon; + + blocked_mask.assign(num_blocks * chunk_size * context_size, 0); + + for (int b = 0; b < num_blocks; ++b) { + for (int c = 0; c < chunk_size; ++c) { + int q_idx = b * chunk_size + c; + for (int ctx = 0; ctx < context_size; ++ctx) { + int kv_idx = b * chunk_size + ctx - max_past_horizon; + int out_idx = b * chunk_size * context_size + c * context_size + ctx; + + if (q_idx >= 0 && q_idx < seq_len && kv_idx >= 0 && kv_idx < seq_len) { + blocked_mask[out_idx] = input_mask[q_idx * seq_len + kv_idx]; + } else { + blocked_mask[out_idx] = 0; + } + } + } + } +} + +void convert_to_block( + const bf16* input, + bf16* output, + int seq_len, + int chunk_size, + int row_stride, + int num_blocks +) { + // Output is expected to be zero-initialized by caller. + // The reshape from [num_blocks*chunk_size, row_stride] to [num_blocks, chunk_size, row_stride] + // is a no-op in row-major memory. We just copy valid rows and leave padding as zeros. + int total_output_rows = num_blocks * chunk_size; + int rows_to_copy = std::min(seq_len, total_output_rows); + memcpy(output, input, (size_t)rows_to_copy * row_stride * sizeof(bf16)); +} + +void extract_block_context( + const bf16* input, + bf16* output, + int seq_len, + int chunk_size, + int max_past_horizon, + int max_future_horizon, + int row_stride, + int num_blocks, + int context_size +) { + // 1. Create padded buffer: [max_past_horizon + seq_len + max_future_horizon + chunk_size - 1, row_stride] + // Left padding: max_past_horizon zeros, Right padding: max_future_horizon + chunk_size - 1 zeros + int padded_len = max_past_horizon + seq_len + max_future_horizon + chunk_size - 1; + std::vector padded((size_t)padded_len * row_stride, bf16(0)); + + // Copy input into padded buffer at offset max_past_horizon + memcpy( + padded.data() + (size_t)max_past_horizon * row_stride, + input, + (size_t)seq_len * row_stride * sizeof(bf16) + ); + + // 2. Extract overlapping windows (unfold): for block b, copy context_size rows starting at b*chunk_size + for (int b = 0; b < num_blocks; b++) { + int src_start = b * chunk_size; + memcpy( + output + (size_t)(b * context_size) * row_stride, + padded.data() + (size_t)src_start * row_stride, + (size_t)context_size * row_stride * sizeof(bf16) + ); + } +} + +// ============================================================================ +// AVX512 dot product of bf16 vectors (returns float32) +// ============================================================================ +static inline float avx512_dot_bf16(const bf16* a, const bf16* b, int len) { + __m512 acc0 = _mm512_setzero_ps(); + __m512 acc1 = _mm512_setzero_ps(); + int k = 0; + for (; k + 32 <= len; k += 32) { + acc0 = _mm512_fmadd_ps(load_bfloat16_to_m512(a + k), + load_bfloat16_to_m512(b + k), acc0); + acc1 = _mm512_fmadd_ps(load_bfloat16_to_m512(a + k + 16), + load_bfloat16_to_m512(b + k + 16), acc1); + } + __m512 acc = _mm512_add_ps(acc0, acc1); + for (; k + 16 <= len; k += 16) { + acc = _mm512_fmadd_ps(load_bfloat16_to_m512(a + k), + load_bfloat16_to_m512(b + k), acc); + } + float sum = _mm512_reduce_add_ps(acc); + for (; k < len; k++) { + sum += (float)a[k] * (float)b[k]; + } + return sum; +} + +// ============================================================================ +// _rel_shift for one block: [chunk_size, position_length] -> [chunk_size, context_size] +// Python equivalent: +// x = F.pad(x, (0, context_size + 1 - position_length)) +// x = x.view(..., block_size * (context_size + 1)) +// x = x[..., : block_size * context_size] +// x = x.view(..., block_size, context_size) +// ============================================================================ +static void rel_shift_block( + const float* input, // [chunk_size, position_length] + float* output, // [chunk_size, context_size] + int chunk_size, + int position_length, + int context_size +) { + int padded_col = context_size + 1; + int padded_size = chunk_size * padded_col; + // Use stack buffer for small sizes (typical: 12*25=300 floats = 1.2KB), heap fallback otherwise + float stack_buf[512]; + float* padded = (padded_size <= 512) ? stack_buf : new float[padded_size]; + memset(padded, 0, padded_size * sizeof(float)); + for (int c = 0; c < chunk_size; c++) { + memcpy(&padded[c * padded_col], &input[c * position_length], + position_length * sizeof(float)); + } + memcpy(output, padded, chunk_size * context_size * sizeof(float)); + if (padded_size > 512) delete[] padded; +} + +void compute_audio_blocked_attention( + const bf16* q_blocked, + const bf16* k_blocked, + const bf16* v_blocked, + const bf16* relative_key_states, + const int* block_attention_mask, + bf16* output, + int seq_len, + int num_blocks, + int chunk_size, + int context_size, + int num_positions, + int num_heads, + int head_dim, + int hidden_size, + int padded_hidden_size, + float softcap, + float invalid_logits_value +) { + memset(output, 0, (size_t)seq_len * padded_hidden_size * sizeof(bf16)); + + int total_q_rows = num_blocks * chunk_size; + float inv_softcap = 1.0f / softcap; + + // Pre-allocate per-thread buffers to avoid repeated heap allocation inside OMP loop + std::vector> thread_matrix_bd(num_heads, std::vector(total_q_rows * num_positions)); + std::vector> thread_shifted_bd(num_heads, std::vector(total_q_rows * context_size)); + std::vector> thread_attn_weights(num_heads, std::vector(chunk_size * context_size)); + + // Parallelize across heads — each head writes to non-overlapping columns + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) + for (int h = 0; h < num_heads; h++) { + int h_offset = h * head_dim; + int hd_vecs = head_dim / 16; + + float* matrix_bd = thread_matrix_bd[h].data(); + float* shifted_bd = thread_shifted_bd[h].data(); + float* attn_weights = thread_attn_weights[h].data(); + + // ---- Step 1: matrix_bd = queries_flat_h @ rel_K_h^T ---- + // queries_flat_h: [total_q_rows, head_dim] (head slice of q_blocked) + // rel_K_h: [num_positions, head_dim] (head slice of relative_key_states) + // matrix_bd: [total_q_rows, num_positions] + for (int i = 0; i < total_q_rows; i++) { + const bf16* q_row = q_blocked + (size_t)i * padded_hidden_size + h_offset; + for (int p = 0; p < num_positions; p++) { + const bf16* rk_row = relative_key_states + (size_t)p * hidden_size + h_offset; + matrix_bd[i * num_positions + p] = avx512_dot_bf16(q_row, rk_row, head_dim); + } + } + + // ---- Step 2: Reshape to [num_blocks, chunk_size, num_positions] then _rel_shift per block ---- + // _rel_shift: [chunk_size, num_positions] -> [chunk_size, context_size] + for (int blk = 0; blk < num_blocks; blk++) { + rel_shift_block( + &matrix_bd[blk * chunk_size * num_positions], + &shifted_bd[blk * chunk_size * context_size], + chunk_size, + num_positions, + context_size + ); + } + + // ---- Step 3: Per-block attention (matrix_ac + bd, softcap, mask, softmax, attn@V) ---- + for (int blk = 0; blk < num_blocks; blk++) { + // 3a. Compute combined logits: matrix_ac + shifted_bd, apply softcap + mask + for (int c = 0; c < chunk_size; c++) { + const bf16* q_row = q_blocked + + (size_t)(blk * chunk_size + c) * padded_hidden_size + h_offset; + + for (int ctx = 0; ctx < context_size; ctx++) { + const bf16* k_row = k_blocked + + (size_t)(blk * context_size + ctx) * padded_hidden_size + h_offset; + + // matrix_ac = Q_row · K_row + float w = avx512_dot_bf16(q_row, k_row, head_dim); + + // + shifted_bd + w += shifted_bd[(blk * chunk_size + c) * context_size + ctx]; + + // Softcap: tanh(w / softcap) * softcap + w = std::tanh(w * inv_softcap) * softcap; + + // Mask + int mask_idx = blk * chunk_size * context_size + c * context_size + ctx; + if (!block_attention_mask[mask_idx]) { + w = invalid_logits_value; + } + + attn_weights[c * context_size + ctx] = w; + } + } + + // 3b. Softmax over context_size dimension (per chunk row) + for (int c = 0; c < chunk_size; c++) { + float* row = &attn_weights[c * context_size]; + + // Find max for numerical stability + __m512 max_vec = _mm512_set1_ps(-1e30f); + int ctx = 0; + for (; ctx + 16 <= context_size; ctx += 16) { + max_vec = _mm512_max_ps(max_vec, _mm512_loadu_ps(row + ctx)); + } + float max_val = _mm512_reduce_max_ps(max_vec); + for (; ctx < context_size; ctx++) { + max_val = std::max(max_val, row[ctx]); + } + + // Exp and sum + __m512 sum_vec = _mm512_setzero_ps(); + __m512 max_broadcast = _mm512_set1_ps(max_val); + ctx = 0; + for (; ctx + 16 <= context_size; ctx += 16) { + __m512 val = _mm512_loadu_ps(row + ctx); + __m512 exp_val = _mm512_exp_ps_corrected(_mm512_sub_ps(val, max_broadcast)); + _mm512_storeu_ps(row + ctx, exp_val); + sum_vec = _mm512_add_ps(sum_vec, exp_val); + } + float sum_exp = _mm512_reduce_add_ps(sum_vec); + for (; ctx < context_size; ctx++) { + row[ctx] = std::exp(row[ctx] - max_val); + sum_exp += row[ctx]; + } + + // Normalize (guard against division by zero for fully-masked rows) + float inv_sum = (sum_exp > 0.0f) ? (1.0f / sum_exp) : 0.0f; + __m512 inv_sum_vec = _mm512_set1_ps(inv_sum); + ctx = 0; + for (; ctx + 16 <= context_size; ctx += 16) { + _mm512_storeu_ps(row + ctx, + _mm512_mul_ps(_mm512_loadu_ps(row + ctx), inv_sum_vec)); + } + for (; ctx < context_size; ctx++) { + row[ctx] *= inv_sum; + } + } + + // 3c. attn_output = attn_weights @ V_block (write to output) + for (int c = 0; c < chunk_size; c++) { + int out_seq_idx = blk * chunk_size + c; + if (out_seq_idx >= seq_len) break; + + bf16* out_ptr = output + (size_t)out_seq_idx * padded_hidden_size + h_offset; + + // Accumulate over context_size using AVX512 across head_dim + __m512 acc[16]; // supports up to head_dim=256 + for (int v = 0; v < hd_vecs; v++) acc[v] = _mm512_setzero_ps(); + + for (int ctx = 0; ctx < context_size; ctx++) { + __m512 w_broadcast = _mm512_set1_ps(attn_weights[c * context_size + ctx]); + const bf16* v_row = v_blocked + + (size_t)(blk * context_size + ctx) * padded_hidden_size + h_offset; + + for (int v = 0; v < hd_vecs; v++) { + __m512 v_vec = load_bfloat16_to_m512(v_row + v * 16); + acc[v] = _mm512_fmadd_ps(w_broadcast, v_vec, acc[v]); + } + } + + // Store bf16 result + for (int v = 0; v < hd_vecs; v++) { + store_m512_to_bfloat16(out_ptr + v * 16, acc[v]); + } + + // Scalar tail (head_dim not multiple of 16) + int tail_start = hd_vecs * 16; + for (int d = tail_start; d < head_dim; d++) { + float sum = 0.0f; + for (int ctx = 0; ctx < context_size; ctx++) { + float v_val = (float)v_blocked[ + (size_t)(blk * context_size + ctx) * padded_hidden_size + h_offset + d]; + sum += attn_weights[c * context_size + ctx] * v_val; + } + out_ptr[d] = (bf16)sum; + } + } + } + } +} + +void compute_audio_self_attention( + const bf16* q_proj_output, + const bf16* k_proj_output, + const bf16* v_proj_output, + const bf16* k_rel_weight, + const bf16* position_embedding, + const std::vector>& block_attention_mask_per_audio, + const std::vector& num_blocks_per_audio, + const std::vector& context_size_per_audio, + const std::vector& seq_len_per_audio, + const std::vector& start_seq_len_index_per_audio, + bf16* output, + int num_audios, + int chunk_size, + int context_left, + int context_right, + int num_heads, + int head_dim, + int hidden_size, + int padded_hidden_size, + int num_positions, + float softcap, + float invalid_logits_value +) { + int max_past_horizon = context_left - 1; + int max_future_horizon = context_right; + + // ---- Block Q/K/V per audio ---- + std::vector> blocked_q_per_audio(num_audios); + std::vector> blocked_k_per_audio(num_audios); + std::vector> blocked_v_per_audio(num_audios); + + for (int i = 0; i < num_audios; i++) { + int nb = num_blocks_per_audio[i]; + int cs = context_size_per_audio[i]; + + blocked_q_per_audio[i].assign((size_t)nb * chunk_size * padded_hidden_size, bf16(0)); + convert_to_block( + q_proj_output + start_seq_len_index_per_audio[i] * padded_hidden_size, + blocked_q_per_audio[i].data(), + seq_len_per_audio[i], chunk_size, padded_hidden_size, nb + ); + + blocked_k_per_audio[i].assign((size_t)nb * cs * padded_hidden_size, bf16(0)); + extract_block_context( + k_proj_output + start_seq_len_index_per_audio[i] * padded_hidden_size, + blocked_k_per_audio[i].data(), + seq_len_per_audio[i], chunk_size, max_past_horizon, max_future_horizon, + padded_hidden_size, nb, cs + ); + + blocked_v_per_audio[i].assign((size_t)nb * cs * padded_hidden_size, bf16(0)); + extract_block_context( + v_proj_output + start_seq_len_index_per_audio[i] * padded_hidden_size, + blocked_v_per_audio[i].data(), + seq_len_per_audio[i], chunk_size, max_past_horizon, max_future_horizon, + padded_hidden_size, nb, cs + ); + } + + // ---- Compute relative_key_states = position_embedding @ k_rel_weight^T ---- + std::vector relative_key_states(num_positions * hidden_size, bf16(0)); + simd_gemm_abt_bf16( + position_embedding, + k_rel_weight, + relative_key_states.data(), + num_positions, hidden_size, hidden_size, + hidden_size, padded_hidden_size, hidden_size + ); + + // ---- Compute blocked attention per audio ---- + for (int i = 0; i < num_audios; i++) { + int nb = num_blocks_per_audio[i]; + int cs = context_size_per_audio[i]; + + compute_audio_blocked_attention( + blocked_q_per_audio[i].data(), + blocked_k_per_audio[i].data(), + blocked_v_per_audio[i].data(), + relative_key_states.data(), + block_attention_mask_per_audio[i].data(), + output + start_seq_len_index_per_audio[i] * padded_hidden_size, + seq_len_per_audio[i], + nb, chunk_size, cs, num_positions, + num_heads, head_dim, hidden_size, padded_hidden_size, + softcap, invalid_logits_value + ); + } +} diff --git a/src/detail/gemma4e_npu/gemma4e_audio_attention.hpp b/src/detail/gemma4e_npu/gemma4e_audio_attention.hpp new file mode 100644 index 000000000..e8263fd24 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_audio_attention.hpp @@ -0,0 +1,111 @@ +#pragma once + +#include +#include "typedef.hpp" +#include "buffer.hpp" + +void create_sliding_window_attention_mask( + int seq_len, + int sliding_window_left, + int sliding_window_right, + std::vector &mask + +); + +void convert_mask_to_blocked( + std::vector &input_mask, // [seq_len, seq_len] + int chunk_size, + int attention_context_left, + int attention_context_right, + std::vector &blocked_mask, // [num_chunks, chunk_size, context_size], + int &num_chunks, + int &context_size +); + +// Splits [seq_len, row_stride] into [num_blocks * chunk_size, row_stride] with zero-padding +// Equivalent to Python: F.pad(hidden_states, (0,0,0,0,0,pad)).reshape(batch, num_blocks, chunk_size, num_heads, head_dim) +void convert_to_block( + const bf16* input, // [seq_len, row_stride] + bf16* output, // [num_blocks * chunk_size, row_stride], caller must pre-allocate & zero-init + int seq_len, + int chunk_size, + int row_stride, // Padded hidden size (num_heads_padded * head_dim) + int num_blocks // = ceil(seq_len / chunk_size) +); + +// Extracts overlapping context windows for blocked attention +// Equivalent to Python: F.pad(...).unfold(1, context_size, chunk_size).movedim(-1, 2) +// Output shape per audio: [num_blocks, context_size, num_heads_padded, head_dim] +void extract_block_context( + const bf16* input, // [seq_len, row_stride] + bf16* output, // [num_blocks * context_size, row_stride], caller must pre-allocate & zero-init + int seq_len, + int chunk_size, + int max_past_horizon, // attention_context_left - 1 + int max_future_horizon, // attention_context_right + int row_stride, // Padded hidden size + int num_blocks, + int context_size // = chunk_size + max_past_horizon + max_future_horizon +); + +// Full blocked audio attention computation for a single audio sample. +// Implements the Python forward() from after Q/K/V scaling through attn_output (before post projection). +// +// q_blocked: [num_blocks * chunk_size, padded_hidden_size] (from convert_to_block) +// k_blocked: [num_blocks * context_size, padded_hidden_size] (from extract_block_context) +// v_blocked: [num_blocks * context_size, padded_hidden_size] (from extract_block_context) +// relative_key_states: [num_positions, hidden_size] dense row-major (= position_embedding @ k_rel_weight^T) +// block_attention_mask: [num_blocks * chunk_size * context_size] (1=attend, 0=mask) +// output: [seq_len, padded_hidden_size] row-major (caller pre-allocates) +void compute_audio_blocked_attention( + const bf16* q_blocked, + const bf16* k_blocked, + const bf16* v_blocked, + const bf16* relative_key_states, + const int* block_attention_mask, + bf16* output, + int seq_len, + int num_blocks, + int chunk_size, + int context_size, + int num_positions, + int num_heads, + int head_dim, + int hidden_size, + int padded_hidden_size, + float softcap, + float invalid_logits_value +); + +// Top-level audio self-attention: blocking + relative key computation + blocked attention + output assembly. +// Replaces the inline code in encode() that does convert_to_block, extract_block_context, +// simd_gemm_abt_bf16 for relative_key_states, and compute_audio_blocked_attention per audio. +// +// q_proj_output / k_proj_output / v_proj_output: [seq_len_padded, padded_hidden_size] packed for all audios +// k_rel_weight: [padded_hidden_size, padded_hidden_size] row-major +// position_embedding: [num_positions, hidden_size] dense +// output: [seq_len_padded, padded_hidden_size] pre-zeroed by caller +void compute_audio_self_attention( + const bf16* q_proj_output, + const bf16* k_proj_output, + const bf16* v_proj_output, + const bf16* k_rel_weight, + const bf16* position_embedding, + const std::vector>& block_attention_mask_per_audio, + const std::vector& num_blocks_per_audio, + const std::vector& context_size_per_audio, + const std::vector& seq_len_per_audio, + const std::vector& start_seq_len_index_per_audio, + bf16* output, + int num_audios, + int chunk_size, + int context_left, + int context_right, + int num_heads, + int head_dim, + int hidden_size, + int padded_hidden_size, + int num_positions, + float softcap, + float invalid_logits_value +); diff --git a/src/detail/gemma4e_npu/gemma4e_cpu_functions.hpp b/src/detail/gemma4e_npu/gemma4e_cpu_functions.hpp new file mode 100644 index 000000000..ac6b2955d --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_cpu_functions.hpp @@ -0,0 +1,374 @@ +#ifndef __GEMMA4E_CPU_FUNCTIONS_HPP__ +#define __GEMMA4E_CPU_FUNCTIONS_HPP__ +#include +#include +#include +#include "typedef.hpp" +#include "buffer.hpp" +#include "models/gemma4e/gemma4e_npu.hpp" +#include "avx512_util.hpp" + +/// @brief Host-side batched kernels used by the gemma4e prefill path. +/// @note These used to be private members of gemma4e_npu::Impl; they are free +/// functions here so the prefill blocks can call them without a handle +/// on the model. Every one of them walks whole rows of a padded batch, +/// hence the L_offset_* arguments: row 0 of the buffer is the first +/// chunk-padding row, and the first live token sits at L_offset. +namespace gemma4e_cpu_func { + +/// @brief Threads the batched kernels fan out over. +static constexpr int MAX_PREFILL_THREAD = 4; + +/// @brief Rope frequency tables, one per attention flavour. +/// @note Both are zero-padded up to the head dim the kernels stride over. +inline constexpr f32 inv_freq_global[] = { + 1.0000e+00, 9.4746e-01, 8.9769e-01, 8.5053e-01, 8.0584e-01, 7.6351e-01, 7.2339e-01, 6.8539e-01, + 6.4938e-01, 6.1527e-01, 5.8294e-01, 5.5232e-01, 5.2330e-01, 4.9581e-01, 4.6976e-01, 4.4508e-01, + 4.2170e-01, 3.9954e-01, 3.7855e-01, 3.5866e-01, 3.3982e-01, 3.2197e-01, 3.0505e-01, 2.8903e-01, + 2.7384e-01, 2.5946e-01, 2.4582e-01, 2.3291e-01, 2.2067e-01, 2.0908e-01, 1.9810e-01, 1.8769e-01, + 1.7783e-01, 1.6849e-01, 1.5963e-01, 1.5125e-01, 1.4330e-01, 1.3577e-01, 1.2864e-01, 1.2188e-01, + 1.1548e-01, 1.0941e-01, 1.0366e-01, 9.8217e-02, 9.3057e-02, 8.8168e-02, 8.3536e-02, 7.9148e-02, + 7.4989e-02, 7.1050e-02, 6.7317e-02, 6.3780e-02, 6.0430e-02, 5.7255e-02, 5.4247e-02, 5.1397e-02, + 4.8697e-02, 4.6138e-02, 4.3714e-02, 4.1418e-02, 3.9242e-02, 3.7180e-02, 3.5227e-02, 3.3376e-02, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, + 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00, 0.0000e+00 +}; +inline constexpr f32 inv_freq_swa[] = { + 1.0000e+00, 9.3057e-01, 8.6596e-01, 8.0584e-01, 7.4989e-01, 6.9783e-01, 6.4938e-01, 6.0430e-01, + 5.6234e-01, 5.2330e-01, 4.8697e-01, 4.5316e-01, 4.2170e-01, 3.9242e-01, 3.6517e-01, 3.3982e-01, + 3.1623e-01, 2.9427e-01, 2.7384e-01, 2.5483e-01, 2.3714e-01, 2.2067e-01, 2.0535e-01, 1.9110e-01, + 1.7783e-01, 1.6548e-01, 1.5399e-01, 1.4330e-01, 1.3335e-01, 1.2409e-01, 1.1548e-01, 1.0746e-01, + 1.0000e-01, 9.3057e-02, 8.6596e-02, 8.0584e-02, 7.4989e-02, 6.9783e-02, 6.4938e-02, 6.0430e-02, + 5.6234e-02, 5.2330e-02, 4.8697e-02, 4.5316e-02, 4.2170e-02, 3.9242e-02, 3.6517e-02, 3.3982e-02, + 3.1623e-02, 2.9427e-02, 2.7384e-02, 2.5483e-02, 2.3714e-02, 2.2067e-02, 2.0535e-02, 1.9110e-02, + 1.7783e-02, 1.6548e-02, 1.5399e-02, 1.4330e-02, 1.3335e-02, 1.2409e-02, 1.1548e-02, 1.0746e-02, + 1.0000e-02, 9.3057e-03, 8.6596e-03, 8.0584e-03, 7.4989e-03, 6.9783e-03, 6.4938e-03, 6.0430e-03, + 5.6234e-03, 5.2330e-03, 4.8697e-03, 4.5316e-03, 4.2170e-03, 3.9242e-03, 3.6517e-03, 3.3982e-03, + 3.1623e-03, 2.9427e-03, 2.7384e-03, 2.5483e-03, 2.3714e-03, 2.2067e-03, 2.0535e-03, 1.9110e-03, + 1.7783e-03, 1.6548e-03, 1.5399e-03, 1.4330e-03, 1.3335e-03, 1.2409e-03, 1.1548e-03, 1.0746e-03, + 1.0000e-03, 9.3057e-04, 8.6596e-04, 8.0584e-04, 7.4989e-04, 6.9783e-04, 6.4938e-04, 6.0430e-04, + 5.6234e-04, 5.2330e-04, 4.8697e-04, 4.5316e-04, 4.2170e-04, 3.9242e-04, 3.6517e-04, 3.3982e-04, + 3.1623e-04, 2.9427e-04, 2.7384e-04, 2.5483e-04, 2.3714e-04, 2.2067e-04, 2.0535e-04, 1.9110e-04, + 1.7783e-04, 1.6548e-04, 1.5399e-04, 1.4330e-04, 1.3335e-04, 1.2409e-04, 1.1548e-04, 1.0746e-04 +}; + +/// @brief y[l] = rms_norm(x[l]) * w, over L rows of a D-wide batch. +/// @param w_ptr may be nullptr, which normalizes without a learned weight. +inline void _rms_norm_batch(bf16* y, bf16* x, const bf16* w_ptr, int norm_D, int D, int L, int L_offset_dest, int L_offset_input){ + + bf16* x_base = x + (size_t)L_offset_input * D; + bf16* y_base = y + (size_t)L_offset_dest * D; + const int simd_width = 16; + + #pragma omp parallel for num_threads(MAX_PREFILL_THREAD) schedule(static) + for (int l = 0; l < L; l++) { + bf16* x_row = x_base + (size_t)l * D; + bf16* y_row = y_base + (size_t)l * D; + for (int d = 0; d < D / norm_D; d++){ + // Step 1: sum of squares over norm_D elements + bf16* x_row_inner = x_row + d * norm_D; + bf16* y_row_inner = y_row + d * norm_D; + __m512 sum_xx_vec = _mm512_setzero_ps(); + int i = 0; + for (; i + simd_width <= norm_D; i += simd_width) { + __m256i bf16_vals = _mm256_loadu_si256((const __m256i*)(x_row_inner + i)); + __m512 fp32_vals = bf16o_fp32_512(bf16_vals); + sum_xx_vec = _mm512_fmadd_ps(fp32_vals, fp32_vals, sum_xx_vec); + } + f32 sum_xx = _mm512_reduce_add_ps(sum_xx_vec); + for (; i < norm_D; i++) { f32 v = static_cast(x_row_inner[i]); sum_xx += v * v; } + + f32 inv_rms_x = 1.0f / sqrtf(sum_xx / (f32)norm_D + 1e-6f); + + // Step 2: y[d] = w[d] * x[d] * inv_rms_x + __m512 inv_rms_vec = _mm512_set1_ps(inv_rms_x); + i = 0; + for (; i + simd_width <= norm_D; i += simd_width) { + __m256i bf16_x = _mm256_loadu_si256((const __m256i*)(x_row_inner + i)); + if (w_ptr != nullptr){ + __m256i bf16_w = _mm256_loadu_si256((const __m256i*)(w_ptr + i)); + __m512 fp32_x = bf16o_fp32_512(bf16_x); + __m512 fp32_w = bf16o_fp32_512(bf16_w); + __m512 y_vec = _mm512_mul_ps(_mm512_mul_ps(fp32_w, fp32_x), inv_rms_vec); + _mm256_storeu_si256((__m256i*)(y_row_inner + i), f32o_bf16_512(y_vec)); + } + else{ + __m512 fp32_x = bf16o_fp32_512(bf16_x); + __m512 y_vec = _mm512_mul_ps(fp32_x, inv_rms_vec); + _mm256_storeu_si256((__m256i*)(y_row_inner + i), f32o_bf16_512(y_vec)); + } + } + for (; i < norm_D; i++) { + y_row_inner[i] = static_cast(static_cast(w_ptr[i]) * static_cast(x_row_inner[i]) * inv_rms_x); + } + } + } +} + +/// @brief x[l] *= scale, over L rows of a D-wide batch. +inline void _elementwise_scale_batch(bf16* x, float scale, int D, int L, int L_offset){ + bf16* dest_base = x + (size_t)L_offset * D; + int total_elements = L * D; + int i = 0; + + __m512 scale_vec = _mm512_set1_ps(scale); + for (; i + 15 < total_elements; i += 16) { + __m256i bf16_vals = _mm256_loadu_si256((const __m256i*)(dest_base + i)); + __m512 fp32_vals = bf16o_fp32_512(bf16_vals); + __m512 result = _mm512_mul_ps(fp32_vals, scale_vec); + _mm256_storeu_si256((__m256i*)(dest_base + i), f32o_bf16_512(result)); + } + for (; i < total_elements; i++){ + dest_base[i] = static_cast(static_cast(dest_base[i]) * scale); + } +} + +/// @brief Applies rope to q or k and the per-head rms norm, over L_effective rows. +/// @note The frequency table is chosen by layer type: swa layers use a shorter +/// wavelength set than global ones. _DH is the head dimension of this +/// layer type, which the caller reads off the model descriptor. +inline void _rope_rms_batch(bf16* x, int D, int L_offset, int L_begin, int L_effective, bf16* rms_weight, gemma4e_layer_type_t layer_type, int _DH){ + const float* inv_freq = is_swa_layer(layer_type) ? inv_freq_swa : inv_freq_global; + const int heads = D / _DH; + bf16* x_base = x + (size_t)L_offset * D; + #pragma omp parallel for num_threads(MAX_PREFILL_THREAD) schedule(static) + for (int ll = 0; ll < L_effective; ll++){ + // Compute thread-local sin/cos values + std::vector local_cos(_DH / 2); + std::vector local_sin(_DH / 2); + for (int j = 0; j < _DH / 2; j++){ + float angle = inv_freq[j] * (L_begin + ll); + local_cos[j] = cosf(angle); + local_sin[j] = sinf(angle); + } + + bf16* x_token = x_base + ll * D; + for (int h = 0; h < heads; h++){ + bf16* x_head = x_token + h * _DH; + // inline rms_norm on single head + { + const int simd_width = 16; + __m512 sum_xx_vec = _mm512_setzero_ps(); + int i = 0; + for (; i + simd_width <= (int)_DH; i += simd_width) { + __m256i bf16_vals = _mm256_loadu_si256((const __m256i*)(x_head + i)); + __m512 fp32_vals = bf16o_fp32_512(bf16_vals); + sum_xx_vec = _mm512_fmadd_ps(fp32_vals, fp32_vals, sum_xx_vec); + } + f32 sum_xx = _mm512_reduce_add_ps(sum_xx_vec); + for (; i < (int)_DH; i++) { f32 v = static_cast(x_head[i]); sum_xx += v * v; } + f32 inv_rms = 1.0f / sqrtf(sum_xx / (f32)_DH + 1e-6f); + __m512 inv_rms_vec = _mm512_set1_ps(inv_rms); + i = 0; + for (; i + simd_width <= (int)_DH; i += simd_width) { + __m256i bf16_x = _mm256_loadu_si256((const __m256i*)(x_head + i)); + __m256i bf16_w = _mm256_loadu_si256((const __m256i*)(rms_weight + i)); + __m512 y_vec = _mm512_mul_ps(_mm512_mul_ps(bf16o_fp32_512(bf16_x), bf16o_fp32_512(bf16_w)), inv_rms_vec); + _mm256_storeu_si256((__m256i*)(x_head + i), f32o_bf16_512(y_vec)); + } + for (; i < (int)_DH; i++) { + x_head[i] = static_cast(static_cast(rms_weight[i]) * static_cast(x_head[i]) * inv_rms); + } + } + // apply rope rotation + bf16* x_left = x_head; + bf16* x_right = x_head + _DH / 2; + int i = 0, simd = 16; + for (; i + simd <= _DH / 2; i += simd) { + __m256i left_bf16 = _mm256_loadu_si256((__m256i*)(x_left + i)); + __m256i right_bf16 = _mm256_loadu_si256((__m256i*)(x_right + i)); + __m512 Lv = bf16o_fp32_512(left_bf16); + __m512 Rv = bf16o_fp32_512(right_bf16); + __m512 C = _mm512_loadu_ps(local_cos.data() + i); + __m512 S = _mm512_loadu_ps(local_sin.data() + i); + __m512 newL = _mm512_sub_ps(_mm512_mul_ps(Lv, C), _mm512_mul_ps(Rv, S)); + __m512 newR = _mm512_add_ps(_mm512_mul_ps(Lv, S), _mm512_mul_ps(Rv, C)); + _mm256_storeu_si256((__m256i*)(x_left + i), f32o_bf16_512(newL)); + _mm256_storeu_si256((__m256i*)(x_right + i), f32o_bf16_512(newR)); + } + } + } +} + +/// @brief dest = x + residual, over L rows of a D-wide batch. +inline void _residual_add_batch(bf16* dest, bf16* x, bf16* residual, int D, int L, int L_offset_dest, int L_offset_x, int L_offset_r){ + bf16* dest_base = dest + (size_t)L_offset_dest * D; + const bf16* src_base = x + (size_t)L_offset_x * D; + const bf16* res_base = residual + (size_t)L_offset_r * D; + + const int simd_width = 16; + + #pragma omp parallel for num_threads(MAX_PREFILL_THREAD) schedule(static) + for (int l = 0; l < L; l++) { + bf16* dest_row = dest_base + (size_t)l * D; + const bf16* src_row = src_base + (size_t)l * D; + const bf16* res_row = res_base + (size_t)l * D; + + int i = 0; + for (; i + simd_width <= D; i += simd_width) { + __m256i bf16_src = _mm256_loadu_si256((const __m256i*)(src_row + i)); + __m256i bf16_res = _mm256_loadu_si256((const __m256i*)(res_row + i)); + __m512 result = _mm512_add_ps(bf16o_fp32_512(bf16_src), bf16o_fp32_512(bf16_res)); + _mm256_storeu_si256((__m256i*)(dest_row + i), f32o_bf16_512(result)); + } + for (; i < D; i++) { + dest_row[i] = static_cast(static_cast(src_row[i]) + static_cast(res_row[i])); + } + } +} + +/// @brief dest = a * b, over L rows of a D-wide batch. +inline void _elementwise_mul_batch(bf16* dest, bf16* a,bf16* b, int D, int L, int L_offset_dest, int L_offset_a, int L_offset_b){ + + bf16* dest_base = dest + (size_t)L_offset_dest * D; + const bf16* a_base = a + (size_t)L_offset_a * D; + const bf16* b_base = b + (size_t)L_offset_b * D; + + const int simd_width = 16; + const int unroll = simd_width * 4; + + #pragma omp parallel for num_threads(MAX_PREFILL_THREAD) schedule(static) + for (int l = 0; l < L; l++) { + bf16* dest_row = dest_base + (size_t)l * D; + const bf16* a_row = a_base + (size_t)l * D; + const bf16* b_row = b_base + (size_t)l * D; + + int i = 0; + // 4-way unrolled loop for maximum throughput + for (; i + unroll <= D; i += unroll){ + __m256i bf16_vals_a0 = _mm256_loadu_si256((const __m256i*)(a_row + i)); + __m256i bf16_vals_a1 = _mm256_loadu_si256((const __m256i*)(a_row + i + simd_width)); + __m256i bf16_vals_a2 = _mm256_loadu_si256((const __m256i*)(a_row + i + simd_width * 2)); + __m256i bf16_vals_a3 = _mm256_loadu_si256((const __m256i*)(a_row + i + simd_width * 3)); + + __m256i bf16_vals_b0 = _mm256_loadu_si256((const __m256i*)(b_row + i)); + __m256i bf16_vals_b1 = _mm256_loadu_si256((const __m256i*)(b_row + i + simd_width)); + __m256i bf16_vals_b2 = _mm256_loadu_si256((const __m256i*)(b_row + i + simd_width * 2)); + __m256i bf16_vals_b3 = _mm256_loadu_si256((const __m256i*)(b_row + i + simd_width * 3)); + + __m512 fp32_vals_a0 = bf16o_fp32_512(bf16_vals_a0); + __m512 fp32_vals_a1 = bf16o_fp32_512(bf16_vals_a1); + __m512 fp32_vals_a2 = bf16o_fp32_512(bf16_vals_a2); + __m512 fp32_vals_a3 = bf16o_fp32_512(bf16_vals_a3); + + __m512 fp32_vals_b0 = bf16o_fp32_512(bf16_vals_b0); + __m512 fp32_vals_b1 = bf16o_fp32_512(bf16_vals_b1); + __m512 fp32_vals_b2 = bf16o_fp32_512(bf16_vals_b2); + __m512 fp32_vals_b3 = bf16o_fp32_512(bf16_vals_b3); + + __m512 result_vec0 = _mm512_mul_ps(fp32_vals_a0, fp32_vals_b0); + __m512 result_vec1 = _mm512_mul_ps(fp32_vals_a1, fp32_vals_b1); + __m512 result_vec2 = _mm512_mul_ps(fp32_vals_a2, fp32_vals_b2); + __m512 result_vec3 = _mm512_mul_ps(fp32_vals_a3, fp32_vals_b3); + + _mm256_storeu_si256((__m256i*)(dest_row + i), f32o_bf16_512(result_vec0)); + _mm256_storeu_si256((__m256i*)(dest_row + i + simd_width), f32o_bf16_512(result_vec1)); + _mm256_storeu_si256((__m256i*)(dest_row + i + simd_width * 2), f32o_bf16_512(result_vec2)); + _mm256_storeu_si256((__m256i*)(dest_row + i + simd_width * 3), f32o_bf16_512(result_vec3)); + } + + // Handle remaining SIMD-width chunks + for (; i + simd_width <= D; i += simd_width){ + __m256i bf16_vals_a = _mm256_loadu_si256((const __m256i*)(a_row + i)); + __m256i bf16_vals_b = _mm256_loadu_si256((const __m256i*)(b_row + i)); + __m512 result_vec = _mm512_mul_ps(bf16o_fp32_512(bf16_vals_a), bf16o_fp32_512(bf16_vals_b)); + _mm256_storeu_si256((__m256i*)(dest_row + i), f32o_bf16_512(result_vec)); + } + + // Scalar tail + for (; i < D; i++){ + dest_row[i] = static_cast(static_cast(a_row[i]) * static_cast(b_row[i])); + } + } +} + +/// @brief dest = gate * b, where b is a PLI_D-wide slice of a D-wide row. +inline void _elementwise_mul_batch(bf16* dest, bf16* gate,bf16* b, int PLI_D, int D, int L, int L_offset_dest, int L_offset_a, int L_offset_b){ + + bf16* dest_base = dest + (size_t)L_offset_dest * PLI_D; + const bf16* a_base = gate + (size_t)L_offset_a * PLI_D; + const bf16* b_base = b + (size_t)L_offset_b * D; + + const int simd_width = 16; + const int unroll = simd_width * 4; + + #pragma omp parallel for num_threads(MAX_PREFILL_THREAD) schedule(static) + for (int l = 0; l < L; l++) { + bf16* dest_row = dest_base + (size_t)l * PLI_D; + const bf16* a_row = a_base + (size_t)l * PLI_D; + const bf16* b_row = b_base + (size_t)l * D; + + int i = 0; + // 4-way unrolled loop for maximum throughput + for (; i + unroll <= PLI_D; i += unroll){ + __m256i bf16_vals_a0 = _mm256_loadu_si256((const __m256i*)(a_row + i)); + __m256i bf16_vals_a1 = _mm256_loadu_si256((const __m256i*)(a_row + i + simd_width)); + __m256i bf16_vals_a2 = _mm256_loadu_si256((const __m256i*)(a_row + i + simd_width * 2)); + __m256i bf16_vals_a3 = _mm256_loadu_si256((const __m256i*)(a_row + i + simd_width * 3)); + + __m256i bf16_vals_b0 = _mm256_loadu_si256((const __m256i*)(b_row + i)); + __m256i bf16_vals_b1 = _mm256_loadu_si256((const __m256i*)(b_row + i + simd_width)); + __m256i bf16_vals_b2 = _mm256_loadu_si256((const __m256i*)(b_row + i + simd_width * 2)); + __m256i bf16_vals_b3 = _mm256_loadu_si256((const __m256i*)(b_row + i + simd_width * 3)); + + __m512 fp32_vals_a0 = bf16o_fp32_512(bf16_vals_a0); + __m512 fp32_vals_a1 = bf16o_fp32_512(bf16_vals_a1); + __m512 fp32_vals_a2 = bf16o_fp32_512(bf16_vals_a2); + __m512 fp32_vals_a3 = bf16o_fp32_512(bf16_vals_a3); + + __m512 fp32_vals_b0 = bf16o_fp32_512(bf16_vals_b0); + __m512 fp32_vals_b1 = bf16o_fp32_512(bf16_vals_b1); + __m512 fp32_vals_b2 = bf16o_fp32_512(bf16_vals_b2); + __m512 fp32_vals_b3 = bf16o_fp32_512(bf16_vals_b3); + + __m512 result_vec0 = _mm512_mul_ps(fp32_vals_a0, fp32_vals_b0); + __m512 result_vec1 = _mm512_mul_ps(fp32_vals_a1, fp32_vals_b1); + __m512 result_vec2 = _mm512_mul_ps(fp32_vals_a2, fp32_vals_b2); + __m512 result_vec3 = _mm512_mul_ps(fp32_vals_a3, fp32_vals_b3); + + _mm256_storeu_si256((__m256i*)(dest_row + i), f32o_bf16_512(result_vec0)); + _mm256_storeu_si256((__m256i*)(dest_row + i + simd_width), f32o_bf16_512(result_vec1)); + _mm256_storeu_si256((__m256i*)(dest_row + i + simd_width * 2), f32o_bf16_512(result_vec2)); + _mm256_storeu_si256((__m256i*)(dest_row + i + simd_width * 3), f32o_bf16_512(result_vec3)); + } + + // Handle remaining SIMD-width chunks + for (; i + simd_width <= PLI_D; i += simd_width){ + __m256i bf16_vals_a = _mm256_loadu_si256((const __m256i*)(a_row + i)); + __m256i bf16_vals_b = _mm256_loadu_si256((const __m256i*)(b_row + i)); + __m512 result_vec = _mm512_mul_ps(bf16o_fp32_512(bf16_vals_a), bf16o_fp32_512(bf16_vals_b)); + _mm256_storeu_si256((__m256i*)(dest_row + i), f32o_bf16_512(result_vec)); + } + + // Scalar tail + for (; i < PLI_D; i++){ + dest_row[i] = static_cast(static_cast(a_row[i]) * static_cast(b_row[i])); + } + } +} + +} // namespace gemma4e_cpu_func + +#endif // __GEMMA4E_CPU_FUNCTIONS_HPP__ diff --git a/src/detail/gemma4e_npu/gemma4e_image.cpp b/src/detail/gemma4e_npu/gemma4e_image.cpp new file mode 100644 index 000000000..331fa1850 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_image.cpp @@ -0,0 +1,1853 @@ +#include "gemma4e_image.hpp" + +#include +#include +#include +#ifdef _WIN32 +#include +#endif +#include "utils/debug_utils.hpp" +#include "utils/error_measure.hpp" + +#include "gemma4e_vision_prefill_helper.hpp" + +#include "vision/norm.hpp" +#include "mmRuntimeSequence.hpp" +#include "rot_pos_emb.hpp" +#include "seq_gen.hpp" + +#define DEBUG_PRINT_ENCODE_TIME_DETAIL (DEBUG_LEVEL >= 1) +#define DEBUG_PRINT_ENCODE_ERROR_METRICS (DEBUG_LEVEL >= 1) + +Gemma4e_ImageEncoder::~Gemma4e_ImageEncoder() +{ +} + +Gemma4e_ImageEncoder::Gemma4e_ImageEncoder(LM_Config config, npu_xclbin_manager *npu_instance, gemma4e_npu* parent_npu_ptr) + : config(config), npu(npu_instance), model_path(config.model_path), parent_npu_ptr(parent_npu_ptr) +{ + + //TODO: FIXME:hi + + // load parameters from json file + + { + + MM_tile_M = config.sub("vision_config").value("VISION_MM_TILE_M", -1); + MM_tile_K = config.sub("vision_config").value("VISION_MM_TILE_K", -1); + MM_tile_N = config.sub("vision_config").value("VISION_MM_TILE_N", -1); + + seq_len_pad_requirement_for_MM = MM_ROW_SIZE*MM_tile_M; + assert( MM_tile_K % MM_tile_N == 0); // we need this for the way we generate the sequence, because k and N is interchangable at MLP (gate-down) + DEBUG_BLOCK(1, + std::cout << "MM_tile_M: " << MM_tile_M << ", MM_tile_K: " << MM_tile_K << ", MM_tile_N: " << MM_tile_N << std::endl; + ) + } + + Padded_GEMMA4E_VISION_HIDDEN_SIZE = round_up_to_multiple(this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, MM_tile_K); + Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE = round_up_to_multiple(this->parent_npu_ptr->GEMMA4E_VISION_INTERMEDIATE_SIZE, MM_tile_K); + Padded_GEMMA4E_VISION_OUT_HIDDEN_SIZE = round_up_to_multiple(this->parent_npu_ptr->GEMMA4E_VISION_IMAGE_OUTPUT_SIZE, MM_tile_K); + + // #if DEBUG_PRINT_ENCODE_ERROR_METRICS + // // print all the parameters for debug + // std::cout << "QWEN3_5_VISION_EMBED_DIM: " << QWEN3_5_VISION_EMBED_DIM << std::endl; + // std::cout << "QWEN3_5_VISION_NUM_HEADS: " << QWEN3_5_VISION_NUM_HEADS << std::endl; + // std::cout << "QWEN3_5_VISION_HEAD_DIM: " << QWEN3_5_VISION_HEAD_DIM << std::endl; + // std::cout << "QWEN3_5_VISION_HIDDEN_SIZE: " << _QWEN3_5_VISION_HIDDEN_SIZE << std::endl; + // std::cout << "QWEN3_5_VISION_MLP_INTERMEDIATE_SIZE: " << _QWEN3_5_VISION_MLP_INTERMEDIATE_SIZE << std::endl; + // std::cout << "QWEN3_5_VISION_NUM_POSITION_EMBEDDINGS: " << QWEN3_5_VISION_NUM_POSITION_EMBEDDINGS << std::endl; + // std::cout << "QWEN3_5_VISION_NUM_LAYERS: " << QWEN3_5_VISION_NUM_LAYERS << std::endl; + // std::cout << "QWEN3_5_VISION_LAYER_NORM_EPSILON: " << QWEN3_5_VISION_LAYER_NORM_EPSILON << std::endl; + // std::cout << "QWEN3_5_MERGER_HIDDEN_SIZE: " << _QWEN3_5_MERGER_HIDDEN_SIZE << std::endl; + // std::cout << "QWEN3_5_VISION_OUT_HIDDEN_SIZE: " << _QWEN3_5_VISION_OUT_HIDDEN_SIZE << std::endl; + + // // print the padded parameters for debug + + // #endif + + this->fla = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "vision_attn.xclbin")); + this->proj = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "vision_mm.xclbin")); + this->proj_high_precision = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "vision_mm_high_precision.xclbin")); + + this->flash_attention_app = this->fla->create_app(); + this->patch_embedder_posisiton_embedding_dim_0_app = this->proj->create_app();//this->proj->create_app(); //TODO: this might not needeD? + this->patch_embedder_posisiton_embedding_dim_1_app = this->proj->create_app();//this->proj->create_app(); + this->patch_embedding_app = this->proj_high_precision->create_app();//this->proj->create_app(); + this->q_proj_app = this->proj->create_app(); + this->k_proj_app = this->proj->create_app(); + this->v_proj_app = this->proj->create_app(); + this->o_proj_app = this->proj->create_app(); + this->gate_proj_app = this->proj->create_app(); + this->up_proj_app = this->proj->create_app(); + this->down_proj_app = this->proj->create_app(); + + this->vision_to_language_input_projection_app = this->proj->create_app(); + + // // attempting to read the info that vision model needs from config.sub("vision_config") + + q_proj_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + k_proj_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + v_proj_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + o_proj_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + gate_proj_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + up_proj_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + down_proj_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + q_norm_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + k_norm_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + post_o_norm_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + post_ffn_norm_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + layer_norm_1_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); + layer_norm_2_weight.resize(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS); +} +void Gemma4e_ImageEncoder::init_weights(SafeTensors &q4nx){ + DEBUG_BLOCK(1, + std::cout << "HIT: init_weights of Gemma4e_ImageEncoder is called. Loading weights from " << this->model_path << std::endl; + ); + + { + + buffer temp_buffer; + + q4nx.load_weights(temp_buffer,"model.vision.patch_embedder.position_embedding_table" ); + this->patch_embedder_position_embedding_table = this->patch_embedder_posisiton_embedding_dim_0_app.create_bo_buffer( + temp_buffer.size() + ); + memcpy(this->patch_embedder_position_embedding_table.data(), temp_buffer.data(), temp_buffer.size()*sizeof(bf16)); + } + { + buffer temp_buffer; + q4nx.load_weights(temp_buffer,"model.vision.patch_embd.weight" ); + this->patch_embd_weight = this->patch_embedding_app.create_bo_buffer( + temp_buffer.size() + ); + memcpy(this->patch_embd_weight.data(), temp_buffer.data(), temp_buffer.size()*sizeof(bf16)); + } + { + buffer temp_buffer; + q4nx.load_weights(temp_buffer,"model.vision.embedding_projection.weight" ); + this->vision_to_language_input_projection_weight = this->vision_to_language_input_projection_app.create_bo_buffer( + temp_buffer.size() + ); + memcpy(this->vision_to_language_input_projection_weight.data(), temp_buffer.data(), temp_buffer.size()*sizeof(bf16)); + } + + for(int layer_id=0; layer_id < this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS; layer_id++){ + + this->q_proj_weight[layer_id] = this->q_proj_app.create_bo_buffer( + Padded_GEMMA4E_VISION_HIDDEN_SIZE*Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + q4nx.load_weights(this->q_proj_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".vision_attn.q_proj.weight" + ); + + this->k_proj_weight[layer_id] = this->k_proj_app.create_bo_buffer( + Padded_GEMMA4E_VISION_HIDDEN_SIZE*Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + q4nx.load_weights(this->k_proj_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".vision_attn.k_proj.weight" + ); + this->v_proj_weight[layer_id] = this->v_proj_app.create_bo_buffer( + Padded_GEMMA4E_VISION_HIDDEN_SIZE*Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + q4nx.load_weights(this->v_proj_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".vision_attn.v_proj.weight" + ); + this->o_proj_weight[layer_id] = this->o_proj_app.create_bo_buffer( + Padded_GEMMA4E_VISION_HIDDEN_SIZE*Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + q4nx.load_weights(this->o_proj_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".vision_attn.out_proj.weight" + ); + this->gate_proj_weight[layer_id] = this->gate_proj_app.create_bo_buffer( + Padded_GEMMA4E_VISION_HIDDEN_SIZE*Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE + ); + q4nx.load_weights(this->gate_proj_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".ffn.gate_proj.weight" + ); + this->up_proj_weight[layer_id] = this->up_proj_app.create_bo_buffer( + Padded_GEMMA4E_VISION_HIDDEN_SIZE*Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE + ); + q4nx.load_weights(this->up_proj_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".ffn.up_proj.weight" + ); + this->down_proj_weight[layer_id] = this->down_proj_app.create_bo_buffer( + Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE*Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + q4nx.load_weights(this->down_proj_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".ffn.down_proj.weight" + ); + + // load all norm weights + q4nx.load_weights( + this->q_norm_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".vision_attn.q_norm.weight" + ); + q4nx.load_weights( + this->k_norm_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".vision_attn.k_norm.weight" + ); + q4nx.load_weights( + this->post_o_norm_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".post_attn_norm.weight" + ); + q4nx.load_weights( + this->post_ffn_norm_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".ffn_post_norm.weight" + ); + q4nx.load_weights( + this->layer_norm_1_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".norm1.weight" + ); + q4nx.load_weights( + this->layer_norm_2_weight[layer_id], + "model.vision."+std::to_string(layer_id)+".norm2.weight" + ); + + // now, we load the min max scalar + buffer q_input_min; + q4nx.load_weights(q_input_min, + "model.vision."+std::to_string(layer_id)+".vision_attn.q_input_min" + ); + assert(q_input_min.size() == 1); + this->input_q_min.push_back(q_input_min[0]); + + buffer q_input_max; + q4nx.load_weights(q_input_max, + "model.vision."+std::to_string(layer_id)+".vision_attn.q_input_max" + ); + assert(q_input_max.size() == 1); + this->input_q_max.push_back(q_input_max[0]); + + buffer q_output_min; + q4nx.load_weights(q_output_min, + "model.vision."+std::to_string(layer_id)+".vision_attn.q_output_min" + ); + assert(q_output_min.size() == 1); + this->output_q_min.push_back(q_output_min[0]); + + buffer q_output_max; + q4nx.load_weights(q_output_max, + "model.vision."+std::to_string(layer_id)+".vision_attn.q_output_max" + ); + assert(q_output_max.size() == 1); + this->output_q_max.push_back(q_output_max[0]); + + // k min/max + buffer k_input_min; + q4nx.load_weights(k_input_min, + "model.vision."+std::to_string(layer_id)+".vision_attn.k_input_min" + ); + assert(k_input_min.size() == 1); + this->input_k_min.push_back(k_input_min[0]); + + buffer k_input_max; + q4nx.load_weights(k_input_max, + "model.vision."+std::to_string(layer_id)+".vision_attn.k_input_max" + ); + assert(k_input_max.size() == 1); + this->input_k_max.push_back(k_input_max[0]); + + buffer k_output_min; + q4nx.load_weights(k_output_min, + "model.vision."+std::to_string(layer_id)+".vision_attn.k_output_min" + ); + assert(k_output_min.size() == 1); + this->output_k_min.push_back(k_output_min[0]); + + buffer k_output_max; + q4nx.load_weights(k_output_max, + "model.vision."+std::to_string(layer_id)+".vision_attn.k_output_max" + ); + assert(k_output_max.size() == 1); + this->output_k_max.push_back(k_output_max[0]); + + // v min/max + buffer v_input_min; + q4nx.load_weights(v_input_min, + "model.vision."+std::to_string(layer_id)+".vision_attn.v_input_min" + ); + assert(v_input_min.size() == 1); + this->input_v_min.push_back(v_input_min[0]); + + buffer v_input_max; + q4nx.load_weights(v_input_max, + "model.vision."+std::to_string(layer_id)+".vision_attn.v_input_max" + ); + assert(v_input_max.size() == 1); + this->input_v_max.push_back(v_input_max[0]); + + buffer v_output_min; + q4nx.load_weights(v_output_min, + "model.vision."+std::to_string(layer_id)+".vision_attn.v_output_min" + ); + assert(v_output_min.size() == 1); + this->output_v_min.push_back(v_output_min[0]); + + buffer v_output_max; + q4nx.load_weights(v_output_max, + "model.vision."+std::to_string(layer_id)+".vision_attn.v_output_max" + ); + assert(v_output_max.size() == 1); + this->output_v_max.push_back(v_output_max[0]); + + // o (attn_out) min/max + buffer o_input_min; + q4nx.load_weights(o_input_min, + "model.vision."+std::to_string(layer_id)+".attn_out_input_min" + ); + assert(o_input_min.size() == 1); + this->input_o_min.push_back(o_input_min[0]); + + buffer o_input_max; + q4nx.load_weights(o_input_max, + "model.vision."+std::to_string(layer_id)+".attn_out_input_max" + ); + assert(o_input_max.size() == 1); + this->input_o_max.push_back(o_input_max[0]); + + buffer o_output_min; + q4nx.load_weights(o_output_min, + "model.vision."+std::to_string(layer_id)+".attn_out_output_min" + ); + assert(o_output_min.size() == 1); + this->output_o_min.push_back(o_output_min[0]); + + buffer o_output_max; + q4nx.load_weights(o_output_max, + "model.vision."+std::to_string(layer_id)+".attn_out_output_max" + ); + assert(o_output_max.size() == 1); + this->output_o_max.push_back(o_output_max[0]); + + // gate min/max + buffer gate_input_min; + q4nx.load_weights(gate_input_min, + "model.vision."+std::to_string(layer_id)+".ffn.gate_input_min" + ); + assert(gate_input_min.size() == 1); + this->input_gate_min.push_back(gate_input_min[0]); + + buffer gate_input_max; + q4nx.load_weights(gate_input_max, + "model.vision."+std::to_string(layer_id)+".ffn.gate_input_max" + ); + assert(gate_input_max.size() == 1); + this->input_gate_max.push_back(gate_input_max[0]); + + buffer gate_output_min; + q4nx.load_weights(gate_output_min, + "model.vision."+std::to_string(layer_id)+".ffn.gate_output_min" + ); + assert(gate_output_min.size() == 1); + this->output_gate_min.push_back(gate_output_min[0]); + + buffer gate_output_max; + q4nx.load_weights(gate_output_max, + "model.vision."+std::to_string(layer_id)+".ffn.gate_output_max" + ); + assert(gate_output_max.size() == 1); + this->output_gate_max.push_back(gate_output_max[0]); + + // up min/max + buffer up_input_min; + q4nx.load_weights(up_input_min, + "model.vision."+std::to_string(layer_id)+".ffn.up_input_min" + ); + assert(up_input_min.size() == 1); + this->input_up_min.push_back(up_input_min[0]); + + buffer up_input_max; + q4nx.load_weights(up_input_max, + "model.vision."+std::to_string(layer_id)+".ffn.up_input_max" + ); + assert(up_input_max.size() == 1); + this->input_up_max.push_back(up_input_max[0]); + + buffer up_output_min; + q4nx.load_weights(up_output_min, + "model.vision."+std::to_string(layer_id)+".ffn.up_output_min" + ); + assert(up_output_min.size() == 1); + this->output_up_min.push_back(up_output_min[0]); + + buffer up_output_max; + q4nx.load_weights(up_output_max, + "model.vision."+std::to_string(layer_id)+".ffn.up_output_max" + ); + assert(up_output_max.size() == 1); + this->output_up_max.push_back(up_output_max[0]); + + // down min/max + buffer down_input_min; + q4nx.load_weights(down_input_min, + "model.vision."+std::to_string(layer_id)+".ffn.down_input_min" + ); + assert(down_input_min.size() == 1); + this->input_down_min.push_back(down_input_min[0]); + + buffer down_input_max; + q4nx.load_weights(down_input_max, + "model.vision."+std::to_string(layer_id)+".ffn.down_input_max" + ); + assert(down_input_max.size() == 1); + this->input_down_max.push_back(down_input_max[0]); + + buffer down_output_min; + q4nx.load_weights(down_output_min, + "model.vision."+std::to_string(layer_id)+".ffn.down_output_min" + ); + assert(down_output_min.size() == 1); + this->output_down_min.push_back(down_output_min[0]); + + buffer down_output_max; + q4nx.load_weights(down_output_max, + "model.vision."+std::to_string(layer_id)+".ffn.down_output_max" + ); + assert(down_output_max.size() == 1); + this->output_down_max.push_back(down_output_max[0]); + } + + DEBUG_BLOCK(1, + std::cout << "[DBG] Gemma4e_ImageEncoder::init_weights finished" << std::endl; + ); +} + +std::vector Gemma4e_ImageEncoder::encode( void* image_payload_ptr) +{ + + DEBUG_BLOCK(1, + std::cout << "HIT: Gemma4e_ImageEncoder::encode is called with image_payload_ptr: " << image_payload_ptr << std::endl; + ); + // // + //DEBUG + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + SafeTensors reference_tensor(this->model_path + "/vision_reference_data.safetensors"); + std::cout << "reference_Tensor path is " << this->model_path + "/vision_reference_data.safetensors" << std::endl; + std::cout << "debug Gemma4e_ImageEncoder::encode called with image_payload_ptr: " << image_payload_ptr << std::endl; + #endif + // auto encoder_start_time = std::chrono::high_resolution_clock::now(); + + gemma4e_image_payload_t* image_payload = (gemma4e_image_payload_t*)image_payload_ptr; + + // first, compare the pixel_values with pre_Gemma4VisionPatchEmbedder_pixel_values + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + std::cout << "Error after patch embedding comparison:" << std::endl; + buffer pre_Gemma4VisionPatchEmbedder_pixel_values; + reference_tensor.load_weights( + pre_Gemma4VisionPatchEmbedder_pixel_values, + "pre_Gemma4VisionPatchEmbedder_pixel_values" + ); + std::cout << "Comparing pixel values with reference..." << std::endl; + uint32_t offset = 0; + for(int i = 0; i < image_payload->num_images; i++){ + uint32_t cur_size = image_payload->image_patch__element_per_patch[i].first*image_payload->image_patch__element_per_patch[i].second; + print_error_metrics( + image_payload->pixel_values[i].data(), + pre_Gemma4VisionPatchEmbedder_pixel_values.data() + offset, + 1, + cur_size, 1, + cur_size, 1 + ); + offset += cur_size; + } + + #endif + // std::cout << "Finished comparing pixel values with reference." << std::endl; + + // TODO: lets use avx512 for it later + + // for each (image_payload->pixel_values) pixel_values = 2 * (pixel_values - 0.5) + + // sanity checkst + for(int i = 0; i < image_payload->num_images; i++){ + assert( image_payload->image_patch__element_per_patch[i].second == this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE); + } + + std::vector seq_len_per_image; // unpadded seq_len per image + std::vector start_seq_len_index_per_image; // the start index in the sequence for each image1 + + int seq_len = 0; + int seq_len_of_last_image = 0; + for(auto image_valid_patch: image_payload->valid_patch_size_per_image ){ + seq_len += image_valid_patch; + seq_len_of_last_image = image_valid_patch; + seq_len_per_image.push_back(image_valid_patch); + + start_seq_len_index_per_image.push_back(seq_len - image_valid_patch); // cumulative offset before this image + } + + int seq_len_padded = round_up_to_multiple(seq_len_of_last_image , vision_L_padded_requirement_for_attention) - seq_len_of_last_image; + seq_len_padded += seq_len; + seq_len_padded = round_up_to_multiple(seq_len_padded, seq_len_pad_requirement_for_MM); + + assert(this->parent_npu_ptr->GEMMA4E_VISION_IMAGE_OUTPUT_SIZE % MM_tile_K == 0); + assert(this->parent_npu_ptr->GEMMA4E_VISION_IMAGE_OUTPUT_SIZE % MM_tile_N == 0); + + DEBUG_BLOCK(1, + std::cout <<"seq_len_pad_requirement_for_MM is " << seq_len_pad_requirement_for_MM << std::endl; + ) + buffer patch_emb_input = patch_embedding_app.create_bo_buffer( + seq_len_padded *this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + buffer patch_emb_output = patch_embedding_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + buffer q_projection_input= q_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + buffer k_projection_input = k_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + buffer v_projection_input = v_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + buffer q_projection_output= q_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + buffer k_projection_output = k_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + buffer v_projection_output = v_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + buffer attention_output = this->flash_attention_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + buffer o_projection_output = o_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + buffer gate_input = gate_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + buffer up_input = up_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + buffer gate_output = gate_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE + ); + buffer up_output = up_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE + ); + buffer down_output = down_proj_app.create_bo_buffer( + seq_len_padded * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + //[2, seq_len_padded, GEMMA4E_POSITION_EMBEDDING_SIZE] + buffer one_shot_buffer = patch_embedder_posisiton_embedding_dim_0_app.create_bo_buffer( + seq_len_padded* 2* this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE + ); + //[2, seq_len_padded, PADDED_GEMMA4E_VISION_HIDDEN_SIZE] + buffer position_embedding_table_output = patch_embedder_posisiton_embedding_dim_0_app.create_bo_buffer( + seq_len_padded* 2* Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + // // memset the buffer to zero + + { + + uint32_t ADD_BIAS = false;// no bias at all for all the mm + generate_mm_sequence(*this->patch_embedding_app.seq(), + seq_len_padded, Padded_GEMMA4E_VISION_HIDDEN_SIZE,Padded_GEMMA4E_VISION_HIDDEN_SIZE, + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + ADD_BIAS, 0, + 0, -10000.0, 1000000.0, // do not clamp on output + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + + generate_mm_sequence(*this->patch_embedder_posisiton_embedding_dim_0_app.seq(), + seq_len_padded, this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE ,Padded_GEMMA4E_VISION_HIDDEN_SIZE, + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + ADD_BIAS, 0, + 0, -10000.0, 1000000.0, // do not clamp on output + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + + generate_mm_sequence(*this->patch_embedder_posisiton_embedding_dim_1_app.seq(), + seq_len_padded, this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE ,Padded_GEMMA4E_VISION_HIDDEN_SIZE, + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + + seq_len_padded* this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE, + this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE* Padded_GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_padded* Padded_GEMMA4E_VISION_HIDDEN_SIZE, + + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + ADD_BIAS, 0, + 0, -10000.0, 1000000.0, // do not clamp on output + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + + gen_mha_vision_attention( + + this->flash_attention_app.seq(), + seq_len_per_image, + vision_L_padded_requirement_for_attention, + vision_S_padded_requirement_for_attention, + vision_num_of_columns, + vision_num_of_rows, + vision_CU_mode, + vision_LQ_per_CT, + vision_LK_per_CT, + vision_LQ_internal, + vision_LK_internal, + ENABLE_QKV_REORDER, + this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + this->Padded_GEMMA4E_VISION_HIDDEN_SIZE, + this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM, + this->parent_npu_ptr->GEMMA4E_VISION_NUM_ATTENTION_HEADS + + ); + } + + // now, apply the pixel_values = 2 * (pixel_values - 0.5) to all valye in patch_emb_input + std::vector temp_pixel_values(patch_emb_input.size()); + memset(temp_pixel_values.data(), 0, temp_pixel_values.size() * sizeof(bf16)); + for(int i = 0, seq_len_offset = 0; i < image_payload->num_images; i++){ + + bf16* raw_pixel_values_ptr = image_payload->pixel_values[i].data(); + bf16* patch_emb_input_ptr = (bf16*)temp_pixel_values.data() + (seq_len_offset* Padded_GEMMA4E_VISION_HIDDEN_SIZE); + + for(int l = 0; l < seq_len_per_image[i]; l++){ + for(int d = 0; d < this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; d++){ + float pixel_value = (float)raw_pixel_values_ptr[l*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE + d]; + pixel_value = 2.0f * (pixel_value - 0.5f); + patch_emb_input_ptr[l*Padded_GEMMA4E_VISION_HIDDEN_SIZE + d] = (bf16)pixel_value; + } + } + seq_len_offset += seq_len_per_image[i]; + } + memcpy(patch_emb_input.data(), temp_pixel_values.data(), temp_pixel_values.size()*sizeof(bf16)); + + patch_emb_input.sync_to_device(); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + std::cout << "Error after scaled vision hidden_size :" << std::endl; + buffer scaled_Gemma4VisionPatchEmbedder_pixel_values; + reference_tensor.load_weights( + scaled_Gemma4VisionPatchEmbedder_pixel_values, + "scaled_Gemma4VisionPatchEmbedder_pixel_values" + ); + + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + patch_emb_input.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + scaled_Gemma4VisionPatchEmbedder_pixel_values.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + image_payload->image_patch__element_per_patch[i].first, Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*image_payload->image_patch__element_per_patch[i].second; + } + + // //TODO: FIXME: + + // for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + // // memcpy( + // // patch_emb_input.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + // // scaled_Gemma4VisionPatchEmbedder_pixel_values.data() + ref_offset, + + // // seq_len_per_image[i]* this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16) + + // } + + #endif + DEBUG_BLOCK(1, + std::cout << "model.vision.patch_embd.weight size: " << patch_embd_weight.size() << std::endl; + ) + + #if DEBUG_PRINT_ENCODE_TIME_DETAIL + auto patch_emb_start_time = std::chrono::high_resolution_clock::now(); + #endif + patch_emb_input.sync_to_device(); + patch_embd_weight.sync_to_device(); + patch_embedding_app(patch_emb_input,patch_embd_weight, patch_emb_output ); + patch_emb_output.sync_from_device(); + #if DEBUG_PRINT_ENCODE_TIME_DETAIL + auto patch_emb_end_time = std::chrono::high_resolution_clock::now(); + std::chrono::duration patch_emb_duration = patch_emb_end_time - patch_emb_start_time; + std::cout << "Time taken for patch embedding: " << patch_emb_duration.count() << " ms" << std::endl; + #endif + // #if DEBUG_PRINT_ENCODE_ERROR_METRICS + + // std::cout << "Error for Gemma4VisionPatchEmbedder_hidden_states:" << std::endl; + // buffer Gemma4VisionPatchEmbedder_hidden_states; + // reference_tensor.load_weights( + // Gemma4VisionPatchEmbedder_hidden_states, + // "Gemma4VisionPatchEmbedder_hidden_states" + // ); + + // for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + // print_error_metrics( + // patch_emb_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + // Gemma4VisionPatchEmbedder_hidden_states.data() + ref_offset, + // 1, + // seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + // image_payload->image_patch__element_per_patch[i].first, Padded_GEMMA4E_VISION_HIDDEN_SIZE + + // // //TODO: FIXME: + // // for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + // // memcpy( + // // patch_emb_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + // // Gemma4VisionPatchEmbedder_hidden_states.data() + ref_offset, + + // // seq_len_per_image[i] * this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16) + + // #endif + + { + // now the _position_embeddings + + ///torch.Size([3, 2, 2520, 10240]) + // first, zero out all one_shot_buffer [2, seq_len_padded, GEMMA4E_POSITION_EMBEDDING_SIZE] + memset(one_shot_buffer.data(), 0, one_shot_buffer.size()*sizeof(bf16)); + + bf16* one_shot_buffer_x_base = (bf16*)one_shot_buffer.data(); + bf16* one_shot_buffer_y_base = one_shot_buffer_x_base + seq_len_padded* this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE; + + for(int i = 0; i < image_payload->num_images; i++){ + bf16* x_ptr = one_shot_buffer_x_base + start_seq_len_index_per_image[i] * this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE; + bf16* y_ptr = one_shot_buffer_y_base + start_seq_len_index_per_image[i] * this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE; + + for(int s = 0; s< seq_len_per_image[i]; s++){ + auto x_val = image_payload->image_grid_pairs_per_image[i][s*2]; + auto y_val = image_payload->image_grid_pairs_per_image[i][s*2 + 1]; + + x_ptr[s * this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE + x_val] = 1.0f; + y_ptr[s * this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE + y_val] = 1.0f; + } + } + } + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + + bf16* one_shot_buffer_x_ptr = (bf16*)one_shot_buffer.data(); + bf16* one_shot_buffer_y_ptr = one_shot_buffer_x_ptr + seq_len_padded* this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE; + + std::cout << "Error for Gemma4VisionPatchEmbedder_one_hot_positions:" << std::endl; + buffer Gemma4VisionPatchEmbedder_one_hot_positions; + reference_tensor.load_weights( + Gemma4VisionPatchEmbedder_one_hot_positions, // shape of [num_image, 2, seq_len_per_image, GEMMA4E_POSITION_EMBEDDING_SIZE] + "Gemma4VisionPatchEmbedder_one_hot_positions" + ); + + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + one_shot_buffer_x_ptr + start_seq_len_index_per_image[i]*this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE, + Gemma4VisionPatchEmbedder_one_hot_positions.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE, + image_payload->image_patch__element_per_patch[i].first, this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE + + ); + + print_error_metrics( + one_shot_buffer_y_ptr + start_seq_len_index_per_image[i]*this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE, + Gemma4VisionPatchEmbedder_one_hot_positions.data() + ref_offset + image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE , + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE, + image_payload->image_patch__element_per_patch[i].first, this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE + + ); + + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_POSITION_EMBEDDING_SIZE*2; // times 2 is for x and y; + } + } + + #endif + + // now, we do the + one_shot_buffer.sync_to_device(); + patch_embedder_position_embedding_table.sync_to_device(); + patch_embedder_posisiton_embedding_dim_0_app(one_shot_buffer, patch_embedder_position_embedding_table, position_embedding_table_output); + position_embedding_table_output.sync_from_device(); + + // dim2 + one_shot_buffer.sync_to_device(); + patch_embedder_position_embedding_table.sync_to_device(); + patch_embedder_posisiton_embedding_dim_1_app(one_shot_buffer, patch_embedder_position_embedding_table, position_embedding_table_output); + position_embedding_table_output.sync_from_device(); + + // now, compare with the python reference + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + + bf16* position_embedding_table_output_x_ptr = (bf16*)position_embedding_table_output.data(); + bf16* position_embedding_table_output_y_ptr = position_embedding_table_output_x_ptr + seq_len_padded* Padded_GEMMA4E_VISION_HIDDEN_SIZE; + + std::cout << "Error Gemma4VisionPatchEmbedder_position_embeddings_before_sum:" << std::endl; + buffer Gemma4VisionPatchEmbedder_position_embeddings_before_sum; + reference_tensor.load_weights( + Gemma4VisionPatchEmbedder_position_embeddings_before_sum, // shape of [num_image, 2, seq_len_per_image, GEMMA4E_VISION_HIDDEN_SIZE] (unpadded) + "Gemma4VisionPatchEmbedder_position_embeddings_before_sum" + ); + + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + const int patches = image_payload->image_patch__element_per_patch[i].first; + const int ref_hidden = this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // Python ref is unpadded [seq_len, 768] + + print_error_metrics( + position_embedding_table_output_x_ptr + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + Gemma4VisionPatchEmbedder_position_embeddings_before_sum.data() + ref_offset, + 1, + seq_len_per_image[i], ref_hidden, // cols to compare = ref row stride (unpadded) + patches, Padded_GEMMA4E_VISION_HIDDEN_SIZE // cpp row stride (padded) + ); + + print_error_metrics( + position_embedding_table_output_y_ptr + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + Gemma4VisionPatchEmbedder_position_embeddings_before_sum.data() + ref_offset + patches * ref_hidden, + 1, + seq_len_per_image[i], ref_hidden, // cols to compare = ref row stride (unpadded) + patches, Padded_GEMMA4E_VISION_HIDDEN_SIZE // cpp row stride (padded) + ); + + ref_offset += patches * ref_hidden * 2; // times 2 for x and y, use unpadded ref stride + } + } + #endif + + hidden_state.resize(seq_len_padded * Padded_GEMMA4E_VISION_HIDDEN_SIZE); + memset(hidden_state.data(), 0, hidden_state.size() * sizeof(bf16)); + + // now, we sum the two dim + //TODO: use avx512 for it later + + { + + bf16* position_embedding_table_output_x_ptr = (bf16*)position_embedding_table_output.data(); + bf16* position_embedding_table_output_y_ptr = position_embedding_table_output_x_ptr + seq_len_padded* Padded_GEMMA4E_VISION_HIDDEN_SIZE; + for(int i = 0; i < seq_len* Padded_GEMMA4E_VISION_HIDDEN_SIZE; i++){ + hidden_state[i] = position_embedding_table_output_x_ptr[i] + position_embedding_table_output_y_ptr[i] + patch_emb_output[i]; + } + } + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for vision_inputs_embeds_after_patch_embedder:" << std::endl; + buffer vision_inputs_embeds_after_patch_embedder; + reference_tensor.load_weights( + vision_inputs_embeds_after_patch_embedder, // shape of [seq_len_padded, GEMMA4E_VISION_HIDDEN_SIZE] + "vision_inputs_embeds_after_patch_embedder" + ); + + auto ref_hidden_state = this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // Python ref is unpadded [seq_len, 768] + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + vision_inputs_embeds_after_patch_embedder.data() + ref_offset, + 1, + seq_len_per_image[i],ref_hidden_state, + image_payload->image_patch__element_per_patch[i].first, Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*ref_hidden_state; // use unpadded ref stride + } + + // //TODO: fixme + // //TODO: FIXME: + // for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + // memcpy( + // hidden_state.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + // vision_inputs_embeds_after_patch_embedder.data() + ref_offset, + + // seq_len_per_image[i] * Padded_GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16) + + // ); + // ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + // } + } + + #endif + + // generate the rope, + std::vector cos_emb; + std::vector sin_emb; + + generate_gemma4_vision_rotary_pos_emb( + image_payload->image_grid_pairs_per_image, + seq_len_per_image, + start_seq_len_index_per_image, + seq_len_padded, + (int)this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM, + this->parent_npu_ptr->GEMMA4E_ROPE_THETA, + 1.0f, // attention_scaling = 1.0 + cos_emb, + sin_emb + ); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + const int head_dim = (int)this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM; + + buffer ref_cos, ref_sin; + reference_tensor.load_weights(ref_cos, "Gemma4VisionEncoder_position_embeddings_cos"); + reference_tensor.load_weights(ref_sin, "Gemma4VisionEncoder_position_embeddings_sin"); + + // ref shape is [num_images, ref_seq_per_image, head_dim] — derive per-image stride from total size + const int ref_seq_per_image = (int)ref_cos.size() / ((int)image_payload->num_images * head_dim); + std::cout << "ref_seq_per_image" << ref_seq_per_image << std::endl; + + std::cout << "Error for Gemma4VisionEncoder_position_embeddings_cos:" << std::endl; + for (int i = 0; i < (int)image_payload->num_images; i++) { + print_error_metrics( + cos_emb.data() + start_seq_len_index_per_image[i] * head_dim, + ref_cos.data() + i * ref_seq_per_image * head_dim, + 1, + seq_len_per_image[i], head_dim, // rows/cols to compare; ref row stride = head_dim + seq_len_per_image[i], head_dim // cpp cos_emb row stride = head_dim (no padding) + ); + } + + std::cout << "Error for Gemma4VisionEncoder_position_embeddings_sin:" << std::endl; + for (int i = 0; i < (int)image_payload->num_images; i++) { + print_error_metrics( + sin_emb.data() + start_seq_len_index_per_image[i] * head_dim, + ref_sin.data() + i * ref_seq_per_image * head_dim, + 1, + seq_len_per_image[i], head_dim, + seq_len_per_image[i], head_dim + ); + } + } + #endif + + residual_buffer.resize(seq_len_padded * Padded_GEMMA4E_VISION_HIDDEN_SIZE); + memset(residual_buffer.data(), 0, residual_buffer.size() * sizeof(bf16)); + //TODO: FIXME: + for(int layer_idx = 0; layer_idx < this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS; layer_idx++){ + memcpy(residual_buffer.data(), hidden_state.data(), seq_len* Padded_GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16)); + + // first, comparew with f"Gemma4VisionEncoderLayer_{layer_idx}_initial_hidden_states"] + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionEncoderLayer_" << layer_idx << "_initial_hidden_states:" << std::endl; + buffer ref_initial_hidden_states; + reference_tensor.load_weights( + ref_initial_hidden_states, + "Gemma4VisionEncoderLayer_" + std::to_string(layer_idx) + "_initial_hidden_states" + ); + + // for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + // print_error_metrics( + // hidden_state.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + // ref_initial_hidden_states.data() + ref_offset, + // 1, + // seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + // seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + // ); + // ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + // } + } + #endif + + // do the first RMS norm on hidden_states + + simd_rms_norm( + hidden_state.data(), + this->layer_norm_1_weight[layer_idx].data(), + hidden_state.data(), + seq_len, this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_padded, this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + // compare with "Gemma4VisionEncoderLayer_{layer_idx}_initial_hidden_states" in the reference for the result + // #if DEBUG_PRINT_ENCODE_ERROR_METRICS + // { + // std::cout << "Error for Gemma4VisionEncoderLayer_" << layer_idx << "_post_input_layernorm_hidden_states:" << std::endl; + // buffer ref_post_input_layernorm_hidden_states; + // reference_tensor.load_weights( + // ref_post_input_layernorm_hidden_states, + // "Gemma4VisionEncoderLayer_" + std::to_string(layer_idx) + "_post_input_layernorm_hidden_states" + // ); + // } + // #endif + // #ifdef DEBUG_PRINT_ENCODE_ERROR_METRICS + // { + // std::cout << "Error for Gemma4VisionEncoderLayer_" << layer_idx << "_post_input_layernorm_hidden_states:" << std::endl; + // buffer ref_post_input_layernorm_hidden_states; + // reference_tensor.load_weights( + // ref_post_input_layernorm_hidden_states, + // "Gemma4VisionEncoderLayer_" + std::to_string(layer_idx) + "_post_input_layernorm_hidden_states" + // ); + + // for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + // print_error_metrics( + // hidden_state.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + // ref_post_input_layernorm_hidden_states.data() + ref_offset, + // 1, + // seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + // seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + // ); + // ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + // } + // } + // #endif + + generate_mm_sequence(*this->q_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_VISION_HIDDEN_SIZE ,Padded_GEMMA4E_VISION_HIDDEN_SIZE, + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, + 1, (float)this->output_q_min[layer_idx],(float) this->output_q_max[layer_idx], // clamp output to quantization range for q_proj + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + + simd_clamp( + hidden_state.data(), + q_projection_input.data(), + this->input_q_min[layer_idx], this->input_q_max[layer_idx], + seq_len * Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + memset(q_projection_input.data() + seq_len * Padded_GEMMA4E_VISION_HIDDEN_SIZE, 0, (seq_len_padded - seq_len)* Padded_GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16)); + q_projection_input.sync_to_device(); + + auto q_proj_run = this->q_proj_app.create_run( + q_projection_input, q_proj_weight[layer_idx], q_projection_output + ); + + q_projection_input.sync_to_device(); + this->q_proj_weight[layer_idx].sync_to_device(); + q_proj_run.start(); + + generate_mm_sequence(*this->k_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_VISION_HIDDEN_SIZE ,Padded_GEMMA4E_VISION_HIDDEN_SIZE, + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, + 1, (float)this->output_k_min[layer_idx],(float) this->output_k_max[layer_idx], // clamp output to quantization range for k_proj + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + simd_clamp( + hidden_state.data(), + k_projection_input.data(), + this->input_k_min[layer_idx], this->input_k_max[layer_idx], + seq_len * Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + memset(k_projection_input.data() + seq_len * Padded_GEMMA4E_VISION_HIDDEN_SIZE, 0, (seq_len_padded - seq_len)* Padded_GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16)); + k_projection_input.sync_to_device(); + + q_proj_run.wait(); + q_projection_output.sync_from_device(); + + k_projection_input.sync_to_device(); + this->k_proj_weight[layer_idx].sync_to_device(); + auto k_proj_run = this->k_proj_app.create_run( + k_projection_input, k_proj_weight[layer_idx], k_projection_output + ); + k_proj_run.start(); + + generate_mm_sequence(*this->v_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_VISION_HIDDEN_SIZE ,Padded_GEMMA4E_VISION_HIDDEN_SIZE, + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, + 1, (float)this->output_v_min[layer_idx],(float) this->output_v_max[layer_idx], // clamp output to quantization range for v_proj + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + simd_clamp( + hidden_state.data(), + v_projection_input.data(), + this->input_v_min[layer_idx], this->input_v_max[layer_idx], + seq_len * Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + memset(v_projection_input.data() + seq_len * Padded_GEMMA4E_VISION_HIDDEN_SIZE, 0, (seq_len_padded - seq_len)* Padded_GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16)); + v_projection_input.sync_to_device(); + + k_proj_run.wait(); + k_projection_output.sync_from_device(); + + v_projection_input.sync_to_device(); + this->v_proj_weight[layer_idx].sync_to_device(); + auto v_proj_run = this->v_proj_app.create_run( + v_projection_input, v_proj_weight[layer_idx], v_projection_output + ); + v_proj_run.start(); + // apply norm for q, and k + simd_rms_norm( + q_projection_output.data(), + this->q_norm_weight[layer_idx].data(), + q_projection_output.data(), + seq_len * (this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE / this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM),this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM, + seq_len * (this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE / this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM),this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + apply_multidimensional_rope( + q_projection_output.data(), cos_emb.data(), sin_emb.data(), + seq_len, this->parent_npu_ptr->GEMMA4E_VISION_NUM_ATTENTION_HEADS, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM); + q_projection_output.sync_to_device(); + simd_rms_norm( + k_projection_output.data(), + this->k_norm_weight[layer_idx].data(), + k_projection_output.data(), + seq_len * (this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE / this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM),this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM, + seq_len * (this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE / this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM),this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + + apply_multidimensional_rope( + k_projection_output.data(), cos_emb.data(), sin_emb.data(), + seq_len, this->parent_npu_ptr->GEMMA4E_VISION_NUM_ATTENTION_HEADS, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM); + k_projection_output.sync_to_device(); + + v_proj_run.wait(); + v_projection_output.sync_from_device(); + + // apply norm for v + simd_rms_norm( + v_projection_output.data(), + v_projection_output.data(), + seq_len * (this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE / this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM),this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM, + seq_len * (this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE / this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM),this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + v_projection_output.sync_to_device(); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionAttention_" + std::to_string(layer_idx)+ "_query_states_after_rope:" << std::endl; + buffer ref_q_projection_output; + reference_tensor.load_weights( + ref_q_projection_output, + "Gemma4VisionAttention_" + std::to_string(layer_idx) + "_query_states_after_rope" + ); + + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + q_projection_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + ref_q_projection_output.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + } + std::cout << "Error for Gemma4VisionAttention_" + std::to_string(layer_idx)+ "_key_states_after_rope:" << std::endl; + buffer ref_k_projection_output; + reference_tensor.load_weights( + ref_k_projection_output, + "Gemma4VisionAttention_" + std::to_string(layer_idx) + "_key_states_after_rope" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + k_projection_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + ref_k_projection_output.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + } + + std::cout << "Error for Gemma4VisionAttention_" + std::to_string(layer_idx)+ "_value_states_after_norm:" << std::endl; + buffer ref_v_projection_output; + reference_tensor.load_weights( + ref_v_projection_output, + "Gemma4VisionAttention_" + std::to_string(layer_idx) + "_value_states_after_norm" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + v_projection_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + ref_v_projection_output.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + } + } + #endif + + auto attention_run = this->flash_attention_app.create_run( + attention_output, q_projection_output, k_projection_output, v_projection_output + ); + q_projection_output.sync_to_device(); + k_projection_output.sync_to_device(); + v_projection_output.sync_to_device(); + attention_run.start(); + + generate_mm_sequence(*this->o_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_VISION_HIDDEN_SIZE ,Padded_GEMMA4E_VISION_HIDDEN_SIZE, + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, + 1, (float)this->output_o_min[layer_idx],(float) this->output_o_max[layer_idx], // clamp output to quantization range for v_proj + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + + attention_run.wait(); + attention_output.sync_from_device(); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionAttention_" + std::to_string(layer_idx)+ "_attention_output:" << std::endl; + buffer ref_attention_output; + reference_tensor.load_weights( + ref_attention_output, + "Gemma4VisionAttention_" + std::to_string(layer_idx) +"_attn_output_before_o_proj" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + attention_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + ref_attention_output.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + } + + // //TOOD: FIXME: remove later + // for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + // memcpy( + // attention_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + // ref_attention_output.data() + ref_offset, + // seq_len_per_image[i]* this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16) + + // ); + // ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + // } + } + + #endif + + simd_clamp( + attention_output.data(), + attention_output.data(), + this->input_o_min[layer_idx], this->input_o_max[layer_idx], + seq_len * Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + attention_output.sync_to_device(); + this->o_proj_weight[layer_idx].sync_to_device(); + o_proj_app(attention_output, this->o_proj_weight[layer_idx], o_projection_output ); + o_projection_output.sync_from_device(); + + simd_rms_norm( + o_projection_output.data(), + this->post_o_norm_weight[layer_idx].data(), + o_projection_output.data(), + seq_len, this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_padded, this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionEncoderLayer_" + std::to_string(layer_idx)+ "_post_attention_layernorm_hidden_states:" << std::endl; + buffer ref__post_attention_layernorm_hidden_states; + reference_tensor.load_weights( + ref__post_attention_layernorm_hidden_states, + "Gemma4VisionEncoderLayer_" + std::to_string(layer_idx) + "_post_attention_layernorm_hidden_states" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + o_projection_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + ref__post_attention_layernorm_hidden_states.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + } + } + #endif + + o_projection_output.sync_to_device(); + + simd_add( + o_projection_output.data(), + residual_buffer.data(), + hidden_state.data(), + seq_len * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + // copy to residual buffer + memcpy(residual_buffer.data(), hidden_state.data(), seq_len* Padded_GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16)); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionEncoderLayer_" + std::to_string(layer_idx)+ "_hidden_states_after_attention_residual:" << std::endl; + buffer ref__hidden_states_after_attention_residual; + reference_tensor.load_weights( + ref__hidden_states_after_attention_residual, + "Gemma4VisionEncoderLayer_" + std::to_string(layer_idx) + "_hidden_states_after_attention_residual" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + ref__hidden_states_after_attention_residual.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + } + } + #endif + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for pre mlp norm weight" << std::endl; + //Gemma4VisionEncoderLayer_{layer_idx}_pre_feedforward_layernorm_weights + buffer ref_pre_feedforward_layernorm_weights; + reference_tensor.load_weights( + ref_pre_feedforward_layernorm_weights, + "Gemma4VisionEncoderLayer_" + std::to_string(layer_idx) + "_pre_feedforward_layernorm_weights" + ); + print_error_metrics( + this->layer_norm_2_weight[layer_idx].data(), + ref_pre_feedforward_layernorm_weights.data(), + 1, + 1, this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + 1, this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE + ); + } + #endif + + simd_rms_norm( + hidden_state.data(), + this->layer_norm_2_weight[layer_idx].data(), + hidden_state.data(), + seq_len, this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_padded, this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionEncoderLayer_" + std::to_string(layer_idx)+ "_post_pre_feedforward_layernorm_hidden_states:" << std::endl; + buffer ref_post_pre_feedforward_layernorm_hidden_states; + reference_tensor.load_weights( + ref_post_pre_feedforward_layernorm_hidden_states, + "Gemma4VisionEncoderLayer_" + std::to_string(layer_idx) + "_post_pre_feedforward_layernorm_hidden_states" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + ref_post_pre_feedforward_layernorm_hidden_states.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + } + } + #endif + + simd_clamp( + hidden_state.data(), + gate_input.data(), + this->input_gate_min[layer_idx], this->input_gate_max[layer_idx], + seq_len * Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + gate_input.sync_to_device(); + + generate_mm_sequence(*this->gate_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_VISION_HIDDEN_SIZE ,Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE, + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 1, // gelu for gate + 1, (float)this->output_gate_min[layer_idx],(float) this->output_gate_max[layer_idx], // clamp output to quantization range for v_proj + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + + gate_input.sync_to_device(); + this->gate_proj_weight[layer_idx].sync_to_device(); + auto gate_proj_run = this->gate_proj_app.create_run( + gate_input, gate_proj_weight[layer_idx], gate_output + ); + gate_proj_run.start(); + + simd_clamp( + hidden_state.data(), + up_input.data(), + this->input_up_min[layer_idx], this->input_up_max[layer_idx], + seq_len * Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + up_input.sync_to_device(); + generate_mm_sequence(*this->up_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_VISION_HIDDEN_SIZE ,Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE, + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, + 1, (float)this->output_up_min[layer_idx],(float) this->output_up_max[layer_idx], // clamp output to quantization range for v_proj + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + + gate_proj_run.wait(); + gate_output.sync_from_device(); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionEncoderLayer_" + std::to_string(layer_idx)+ "_gate_proj_act:" << std::endl; + buffer ref_gate_proj_output; + reference_tensor.load_weights( + ref_gate_proj_output, + "Gemma4VisionMLP_layer_" + std::to_string(layer_idx) + "_gate_proj_act" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + gate_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE, + ref_gate_proj_output.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_INTERMEDIATE_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_INTERMEDIATE_SIZE; // use unpadded ref stride + } + } + #endif + + up_input.sync_to_device(); + this->up_proj_weight[layer_idx].sync_to_device(); + auto up_proj_run = this->up_proj_app.create_run( + up_input, up_proj_weight[layer_idx], up_output + ); + up_proj_run.start(); + + generate_mm_sequence(*this->down_proj_app.seq(), + seq_len_padded, Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE, Padded_GEMMA4E_VISION_HIDDEN_SIZE , + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, + 1, (float)this->output_down_min[layer_idx],(float) this->output_down_max[layer_idx], // clamp output to quantization range for v_proj + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + + up_proj_run.wait(); + up_output.sync_from_device(); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionEncoderLayer_" + std::to_string(layer_idx)+ "_up_proj_output:" << std::endl; + buffer ref_up_proj_output; + reference_tensor.load_weights( + ref_up_proj_output, + "Gemma4VisionMLP_layer_" + std::to_string(layer_idx) + "_up_proj" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + up_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE, + ref_up_proj_output.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_INTERMEDIATE_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_INTERMEDIATE_SIZE; // use unpadded ref stride + } + } + #endif + + simd_mul( + gate_output.data(), + up_output.data(), + gate_output.data(), // write back to gate_output buffer to save memory + seq_len * Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE + ); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionEncoderLayer_" + std::to_string(layer_idx)+ "_after_act:" << std::endl; + buffer ref_post_gate_mul_hidden_states; + reference_tensor.load_weights( + ref_post_gate_mul_hidden_states, + "Gemma4VisionMLP_layer_" + std::to_string(layer_idx) + "_after_act" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + gate_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE, + ref_post_gate_mul_hidden_states.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_INTERMEDIATE_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_INTERMEDIATE_SIZE; // use unpadded ref stride + } + } + #endif + + simd_clamp( + gate_output.data(), + gate_output.data(), + this->input_down_min[layer_idx], this->input_down_max[layer_idx], + seq_len * Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + gate_output.sync_to_device(); + this->down_proj_weight[layer_idx].sync_to_device(); + this->down_proj_app(gate_output, this->down_proj_weight[layer_idx], down_output); + down_output.sync_from_device(); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionEncoderLayer_" + std::to_string(layer_idx)+ "_post_mlp_hidden_states:" << std::endl; + buffer ref_post_mlp_hidden_states; + reference_tensor.load_weights( + ref_post_mlp_hidden_states, + "Gemma4VisionEncoderLayer_" + std::to_string(layer_idx) + "_post_mlp_hidden_states" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + down_output.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + ref_post_mlp_hidden_states.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + } + } + #endif + + simd_rms_norm( + down_output.data(), + this->post_ffn_norm_weight[layer_idx].data(), + hidden_state.data(), + seq_len, this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_padded, this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionEncoderLayer_" + std::to_string(layer_idx)+ "_post_feedforward_layernorm_hidden_states:" << std::endl; + buffer ref_post_ffn_layernorm_hidden_states; + reference_tensor.load_weights( + ref_post_ffn_layernorm_hidden_states, + "Gemma4VisionEncoderLayer_" + std::to_string(layer_idx) + "_post_feedforward_layernorm_hidden_states" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + ref_post_ffn_layernorm_hidden_states.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + } + } + #endif + + simd_add( + hidden_state.data(), + residual_buffer.data(), + hidden_state.data(), + seq_len * this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "Error for Gemma4VisionEncoderLayer_" + std::to_string(layer_idx)+ "_final_hidden_states:" << std::endl; + buffer ref__hidden_states_after_ffn_residual; + reference_tensor.load_weights( + ref__hidden_states_after_ffn_residual, + "Gemma4VisionEncoderLayer_" + std::to_string(layer_idx) + "_final_hidden_states" + ); + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + hidden_state.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + ref__hidden_states_after_ffn_residual.data() + ref_offset, + 1, + seq_len_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + seq_len_per_image[i], Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + } + } + #endif + } + + // //TODO: FIXME: DEBUG + // { + + // buffer ref__hidden_states_after_ffn_residual; + // reference_tensor.load_weights( + // ref__hidden_states_after_ffn_residual, + // "Gemma4VisionEncoderLayer_" + std::to_string(this->parent_npu_ptr->GEMMA4E_VISION_NUM_HIDDEN_LAYERS-1) + "_final_hidden_states" + // ); + // for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + // memcpy( + // hidden_state.data() + start_seq_len_index_per_image[i]*Padded_GEMMA4E_VISION_HIDDEN_SIZE, + // ref__hidden_states_after_ffn_residual.data() + ref_offset, + // seq_len_per_image[i] * this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16) + // ); + // ref_offset+= image_payload->image_patch__element_per_patch[i].first*this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; // use unpadded ref stride + // } + + // } + + // // debug, print all the content in seq_len_per_image + + // the vision pooler stage + std::vector k_per_image; + std::vector k_squared_per_image; + for(int i = 0; i < image_payload->num_images; i++){ + + int k = std::sqrt( seq_len_per_image[i]/ image_payload->num_soft_tokens_per_image[i] ); + k_per_image.push_back(k); + k_squared_per_image.push_back(k*k); + } + + std::vector max_x; + + for(int i = 0; i < image_payload->num_images; i++){ + int cur_max_x = -1; + for(int j = 0; j < seq_len_per_image[i]; j++){ + + auto cur_x = image_payload->image_grid_pairs_per_image[i][2*j]; + if(cur_x > cur_max_x){ + cur_max_x = cur_x; + } + } + max_x.push_back(cur_max_x + 1); + } + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + + std::cout << "compare max x" < Gemma4VisionPooler_max_x; + reference_tensor.load_weights( + Gemma4VisionPooler_max_x, + "Gemma4VisionPooler_max_x" + ); + for(int i = 0; i < image_payload->num_images; i++){ + if(max_x[i] != Gemma4VisionPooler_max_x.data()[i]){ + std::cout << "max_x mismatch for image " << i << ": " << max_x[i] << " vs ref " << Gemma4VisionPooler_max_x.data()[i] << std::endl; + } + } + } + #endif + + // now, we divide every image's image_grid_pairs_per_image by k (floor division), matching Python: kernel_idxs = floor(pos / k) + std::vector> kernel_idx(image_payload->num_images); + for(int i = 0; i < image_payload->num_images; i++){ + + for(int j = 0; j < seq_len_per_image[i]; j++){ + image_payload->image_grid_pairs_per_image[i][2*j] /= (float)k_per_image[i]; //NOTE: store back to int, same as round_down in floor mode + image_payload->image_grid_pairs_per_image[i][2*j + 1] /= (float)k_per_image[i]; + + kernel_idx.at(i).push_back( + image_payload->image_grid_pairs_per_image[i][2*j] + (max_x[i] /k_per_image[i] ) * image_payload->image_grid_pairs_per_image[i][2*j+1] + ); + } + } + + // now, compare the error of kernel_idx with "Gemma4VisionPooler_kernel_idxs" + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "compare kernel idx" < Gemma4VisionPooler_kernel_idxs; + reference_tensor.load_weights( + Gemma4VisionPooler_kernel_idxs, + "Gemma4VisionPooler_kernel_idxs" + ); + + int ref_per_image = Gemma4VisionPooler_kernel_idxs.size() / image_payload->num_images; + for(int i = 0, ref_offset=0; i < image_payload->num_images; i++){ + for(int j = 0; j < seq_len_per_image[i]; j++){ + if(kernel_idx[i][j] != Gemma4VisionPooler_kernel_idxs.data()[ref_offset + j]){ + std::cout << "kernel_idx mismatch for image " << i << " token " << j << ": " << kernel_idx[i][j] << " vs ref " << Gemma4VisionPooler_kernel_idxs.data()[ref_offset + j] << std::endl; + } + } + ref_offset+=ref_per_image; + } + } + #endif + + //NOTE: num_soft_token_per_image is in image_payload->num_soft_tokens_per_image[i]; + // L_per_image is in seq_len_per_image[i] + std::vector pooling_output; + //hidden_state is a row-major buffer + // at this point, the hidden_state is in shape of [num_image,seq_len_per_image[], Padded_GEMMA4E_VISION_HIDDEN_SIZE], we will do pooling for each image separately, and the pooling weight will be generated based on kernel_idx and k_squared_per_image (which is the number of tokens in each pooling region), and the pooling weight will be shape of [num_image, seq_len_per_image, num_soft_token_per_image] + //NOTE: seq_len_per_image[] means this varies from image to image, and num_soft_token_per_image is in image_payload->num_soft_tokens_per_image[i] + + //NOTE: the python code generates a matrix that is just too sparse, we use scatter-add instead + // output = weights.transpose(1, 2) @ hidden_states.float() + // equivalently: output[t, h] = (1/k²) * Σ hidden_states[l, h] for all l where kernel_idx[l] == t + + float root_hidden_size = std::sqrt((float)this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE); + for(int i = 0; i < image_payload->num_images; i++){ + const int cur_seq_len = seq_len_per_image[i]; + const int cur_soft_tokens = image_payload->num_soft_tokens_per_image[i]; + const int HIDDEN_SIZE = this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE; + const float scale = 1.0f / (float)k_squared_per_image[i]; + + // accumulate in float for precision + std::vector accum(cur_soft_tokens * HIDDEN_SIZE, 0.0f); + + bf16* hidden_base = hidden_state.data() + start_seq_len_index_per_image[i] * Padded_GEMMA4E_VISION_HIDDEN_SIZE; + for(int l = 0; l < cur_seq_len; l++){ + const int t = kernel_idx[i][l]; + bf16* src = hidden_base + l * Padded_GEMMA4E_VISION_HIDDEN_SIZE; + float* dst = accum.data() + t * HIDDEN_SIZE; + for(int h = 0; h < HIDDEN_SIZE; h++){ + dst[h] += (float)src[h] * scale; + } + } + + // convert back to bf16 + int cur_pooling_output_size = pooling_output.size(); + pooling_output.resize(cur_pooling_output_size + cur_soft_tokens * Padded_GEMMA4E_VISION_HIDDEN_SIZE, bf16(0)); + for(int t = 0; t < cur_soft_tokens; t++){ + for(int h = 0; h < HIDDEN_SIZE; h++){ + // pooling_output[i][t * Padded_GEMMA4E_VISION_HIDDEN_SIZE + h] = bf16(accum[t * HIDDEN_SIZE + h]); + pooling_output[cur_pooling_output_size + t * Padded_GEMMA4E_VISION_HIDDEN_SIZE + h] = bf16(accum[t * HIDDEN_SIZE + h] * root_hidden_size ); // add a scaling factor to prevent overflow, matching the implementation in python code + } + } + } + + size_t vision_output_token_size = pooling_output.size() / this->Padded_GEMMA4E_VISION_HIDDEN_SIZE; + size_t padded_vision_output_token_size = round_up_to_multiple(vision_output_token_size, seq_len_pad_requirement_for_MM); + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "compare Gemma4VisionPooler_hidden_states_after_scaling" << std::endl; + buffer ref_pooling_output; + reference_tensor.load_weights( + ref_pooling_output, + "Gemma4VisionPooler_hidden_states_after_scaling" + ); + + int ref_per_image = ref_pooling_output.size() / image_payload->num_images; + for(int i = 0,pool_offset=0, ref_offset=0; i < image_payload->num_images; i++){ + print_error_metrics( + pooling_output.data() + pool_offset, + ref_pooling_output.data() + ref_offset, + 1, + image_payload->num_soft_tokens_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + image_payload->num_soft_tokens_per_image[i], this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE + ); + ref_offset += ref_per_image; + pool_offset += image_payload->num_soft_tokens_per_image[i] * Padded_GEMMA4E_VISION_HIDDEN_SIZE; + } + } + #endif + + buffer language_embed_input = this->vision_to_language_input_projection_app.create_bo_buffer( + padded_vision_output_token_size* Padded_GEMMA4E_VISION_HIDDEN_SIZE + ); + memset(language_embed_input.data(), 0, padded_vision_output_token_size* Padded_GEMMA4E_VISION_HIDDEN_SIZE * sizeof(bf16)); // zero padding for padded tokens + + buffer language_embed_output = this->vision_to_language_input_projection_app.create_bo_buffer( + padded_vision_output_token_size * this->parent_npu_ptr->GEMMA4E_VISION_IMAGE_OUTPUT_SIZE + ); + + simd_rms_norm( + pooling_output.data(), + language_embed_input.data(), + vision_output_token_size, this->parent_npu_ptr->GEMMA4E_VISION_HIDDEN_SIZE, + padded_vision_output_token_size, this->Padded_GEMMA4E_VISION_HIDDEN_SIZE + + ); + + generate_mm_sequence(*this->vision_to_language_input_projection_app.seq(), + padded_vision_output_token_size, Padded_GEMMA4E_VISION_HIDDEN_SIZE,this->parent_npu_ptr->GEMMA4E_VISION_IMAGE_OUTPUT_SIZE, //TODO: FIXME: + MM_tile_M,MM_tile_K,MM_tile_N, + 8,8,8, + rtp_address, rtp_sync_lock_id, + MM_ROW_SIZE,MM_COL_SIZE, + 0,0,0, + IS_B_ROW_MAJOR, ENABLE_AXI4, true, + false, 0, /// no biase + 0, -10000.0, 1000000.0, // do not clamp on output + ENABLE_QKV_REORDER, this->parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM + ); + + language_embed_input.sync_to_device(); + this->vision_to_language_input_projection_weight.sync_to_device(); + this->vision_to_language_input_projection_app(language_embed_input, this->vision_to_language_input_projection_weight, language_embed_output); + language_embed_output.sync_from_device(); + + #if DEBUG_PRINT_ENCODE_ERROR_METRICS + { + std::cout << "compare vision to language projection output" << std::endl; + buffer ref_vision_to_language_projection_output; + reference_tensor.load_weights( + ref_vision_to_language_projection_output, + "vision_final_embs_after_embed_vision" + ); + std::cout << "finished laoding ref vision to language projection output" << std::endl; + print_error_metrics( + language_embed_output.data(), + ref_vision_to_language_projection_output.data(), + 1, + vision_output_token_size, this->parent_npu_ptr->GEMMA4E_VISION_IMAGE_OUTPUT_SIZE, + padded_vision_output_token_size, this->parent_npu_ptr->GEMMA4E_VISION_IMAGE_OUTPUT_SIZE + + ); + } + #endif + + ///TODO: consider change the output type interface + std::vector final_res(padded_vision_output_token_size *this->parent_npu_ptr->GEMMA4E_VISION_IMAGE_OUTPUT_SIZE ); + memcpy( + final_res.data(), + language_embed_output.data(), + padded_vision_output_token_size * this->parent_npu_ptr->GEMMA4E_VISION_IMAGE_OUTPUT_SIZE * sizeof(bf16) + ); + + return final_res; +} diff --git a/src/detail/gemma4e_npu/gemma4e_image.hpp b/src/detail/gemma4e_npu/gemma4e_image.hpp new file mode 100644 index 000000000..3423a5942 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_image.hpp @@ -0,0 +1,136 @@ +#pragma once +#include "typedef.hpp" +#include +#include +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/gemma4e/gemma4e_npu.hpp" + +#include "vision/norm.hpp" + +class Gemma4e_ImageEncoder{ + + public: + + ~Gemma4e_ImageEncoder(); + void init_weights(SafeTensors &q4nx); + Gemma4e_ImageEncoder(LM_Config config, npu_xclbin_manager *npu_instance, gemma4e_npu* parent_npu_ptr); + + std::vector encode( void* image_payload_ptr); + + LM_Config config; + npu_xclbin_manager *npu; + gemma4e_npu* parent_npu_ptr; + + unsigned int Padded_GEMMA4E_VISION_HIDDEN_SIZE; + unsigned int Padded_GEMMA4E_VISION_MLP_INTERMEDIATE_SIZE; + unsigned int Padded_GEMMA4E_VISION_OUT_HIDDEN_SIZE; + + uint32_t MM_tile_M; + uint32_t MM_tile_K; + uint32_t MM_tile_N; + + uint32_t seq_len_pad_requirement_for_MM; + uint32_t MM_ROW_SIZE = 4; + uint32_t MM_COL_SIZE = 8; + + // parameter for vision attention kernel + uint32_t vision_num_of_columns = 8; + uint32_t vision_num_of_rows = 4; + uint32_t vision_CU_mode = 2; + uint32_t vision_LQ_per_CT = 32; + uint32_t vision_LK_per_CT = 512; + uint32_t vision_LQ_internal=32; + uint32_t vision_LK_internal=32; + uint32_t vision_L_padded_requirement_for_attention = 512; // padding for L_seq in vision attention + uint32_t vision_S_padded_requirement_for_attention = vision_LK_per_CT; // padding for S_seq in vision attention + //uint32_t vision_L_padded_requirement_for_attention = 32; + + // IF true, reorder qkv from L_Seq x (3*QWEN3_5_VISION_NUM_HEADS*QWEN3_VISION_HEAD_DIM) row major -> + // [ 3, QWEN3_5_VISION_NUM_HEADS, L_Seq ,QWEN3_VISION_HEAD_DIM] + bool ENABLE_QKV_REORDER = false; + // for MM runtime sequence + uint32_t rtp_address = 4096; // offset right after stack size + uint32_t rtp_sync_lock_id = 10; // the rtp sync lock + bool ENABLE_AXI4 = true; + bool IS_B_ROW_MAJOR = false; + + // debug ptr for now + std::string model_path; + + npu_app_manager* fla; + npu_app_manager* proj; + npu_app_manager* proj_high_precision; + + // define the necessary bitstream no + npu_app flash_attention_app; + npu_app patch_embedder_posisiton_embedding_dim_0_app; + npu_app patch_embedder_posisiton_embedding_dim_1_app; + npu_app patch_embedding_app; + npu_app q_proj_app; + npu_app k_proj_app; + npu_app v_proj_app; + npu_app o_proj_app; + npu_app gate_proj_app; + npu_app up_proj_app; + npu_app down_proj_app; + + npu_app vision_to_language_input_projection_app; + + // The weights + + buffer patch_embedder_position_embedding_table; + buffer patch_embd_weight; + buffer vision_to_language_input_projection_weight; // project into language hidden space, + + std::vector> q_proj_weight; + std::vector> k_proj_weight; + std::vector> v_proj_weight; + std::vector> o_proj_weight; + std::vector> gate_proj_weight; + std::vector> up_proj_weight; + std::vector> down_proj_weight; + + std::vector hidden_state; + std::vector residual_buffer; + std::vector> q_norm_weight; + std::vector> k_norm_weight; + std::vector> post_o_norm_weight; + std::vector> post_ffn_norm_weight; + std::vector> layer_norm_1_weight; + std::vector> layer_norm_2_weight; + + std::vector input_q_max; + std::vector input_q_min; + std::vector input_k_max; + std::vector input_k_min; + std::vector input_v_max; + std::vector input_v_min; + std::vector input_o_max; + std::vector input_o_min; + std::vector input_gate_max; + std::vector input_gate_min; + std::vector input_up_max; + std::vector input_up_min; + std::vector input_down_max; + std::vector input_down_min; + + std::vector output_q_max; + std::vector output_q_min; + std::vector output_k_max; + std::vector output_k_min; + std::vector output_v_max; + std::vector output_v_min; + std::vector output_o_max; + std::vector output_o_min; + std::vector output_gate_max; + std::vector output_gate_min; + std::vector output_up_max; + std::vector output_up_min; + std::vector output_down_max; + std::vector output_down_min; + + inline int round_up_to_multiple (int x, int multiple) + { + return ((x + multiple - 1) / multiple) * multiple; + }; +}; diff --git a/src/detail/gemma4e_npu/gemma4e_npu.cpp b/src/detail/gemma4e_npu/gemma4e_npu.cpp new file mode 100644 index 000000000..e7cbc06f1 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_npu.cpp @@ -0,0 +1,858 @@ +#include +#include "flm_override.hpp" +#include "gemma4e_npu_detail.hpp" +#include "metrices.hpp" +#include "utils/error_measure.hpp" +#include "mmRuntimeSequence.hpp" + +gemma4e_npu::Impl::Impl(LM_Config config, npu_xclbin_manager *npu_instance, gemma4e_npu* parent_ptr, int MAX_L ) + : config(config), npu(npu_instance), parent_ptr(parent_ptr){ + FLM_OVERRIDE(engine_init, (void)0, this->npu, this->config); + is_vlm = config.get("is_vlm", false); + is_audio = config.get("is_audio", false); + + current_context_length = 0; + + this->MAX_L = std::max(MAX_L, 4096); + // make MAX_L a power of 2 + float log2_MAX_L = std::log2(this->MAX_L); + if (log2_MAX_L != std::floor(log2_MAX_L)){ + this->MAX_L = 1 << ((int)std::floor(log2_MAX_L) + 1); + } + MAX_L = this->MAX_L; // synchronize MAX_L + is_preload_launched = false; + + + try{ + this->desc.build(this->config); + + D = desc.D; + vocab_size = desc.vocab_size; + vocab_size_padded = desc.vocab_size_padded; + DH = desc.DH; + DQ = desc.DQ; + DK = desc.DK; + DV = desc.DV; + SWA_DH = desc.SWA_DH; + SWA_DQ = desc.SWA_DQ; + SWA_DK = desc.SWA_DK; + SWA_DV = desc.SWA_DV; + num_hidden_layers = desc.num_hidden_layers; + SLIDING_LENGTH = desc.SLIDING_LENGTH; + PLI_D = desc.PLI_D; + num_kv_shared_layers = desc.num_kv_shared_layers; + final_logit_softcapping = desc.final_logit_softcapping; + non_skip_layers = desc.non_skip_layers; + INTERMEDIATE_SIZE = desc.INTERMEDIATE_SIZE; + enable_double_wide_mlp = desc.enable_double_wide_mlp; + global_layer_period = desc.global_layer_period; + layer_types = desc.layer_types; + + if (this->is_global_layer_idx(non_skip_layers - 1)){ + last_global_kv_cache_layer_idx = non_skip_layers - 1; + last_swa_kv_cache_layer_idx = non_skip_layers - 2; + } + else { + last_swa_kv_cache_layer_idx = non_skip_layers - 1; + last_global_kv_cache_layer_idx = (last_swa_kv_cache_layer_idx / global_layer_period) * global_layer_period - 1; + } + DEBUG_BLOCK(1, + header_print_g("info", "Gemma4e NPU config:"); + std::cout << "\tD: " << D << std::endl; + std::cout << "\tPLI_D: " << PLI_D << std::endl; + std::cout << "\tDH: " << DH << std::endl; + std::cout << "\tDQ: " << DQ << std::endl; + std::cout << "\tDK: " << DK << std::endl; + std::cout << "\tDV: " << DV << std::endl; + std::cout << "\tSWA_DH: " << SWA_DH << std::endl; + std::cout << "\tSWA_DQ: " << SWA_DQ << std::endl; + std::cout << "\tSWA_DK: " << SWA_DK << std::endl; + std::cout << "\tSWA_DV: " << SWA_DV << std::endl; + std::cout << "\tSLIDING_LENGTH: " << SLIDING_LENGTH << std::endl; + std::cout << "\tINTERMEDIATE_SIZE: " << INTERMEDIATE_SIZE << std::endl; + std::cout << "\tFINAL_LOGIT_SOFTCAPPING: " << final_logit_softcapping << std::endl; + std::cout << "\tvocab_size: " << vocab_size << " (" << vocab_size_padded << " padded)" << std::endl; + std::cout << "\tnum_hidden_layers: " << num_hidden_layers << std::endl; + std::cout << "\tnum_kv_shared_layers: " << num_kv_shared_layers << std::endl; + std::cout << "\tnon_skip_layers: " << non_skip_layers << std::endl; + std::cout << "\tglobal_layer_period: " << global_layer_period << std::endl; + std::cout << "\tlast_swa_kv_cache_layer_idx: " << last_swa_kv_cache_layer_idx << std::endl; + std::cout << "\tlast_global_kv_cache_layer_idx: " << last_global_kv_cache_layer_idx << std::endl; + + ) + DEBUG_BLOCK(2, + std::cout << "\tlayer_types: \n"; + for (int i = 0; i < num_hidden_layers; i++){ + std::string type_str; + switch (layer_types[i]){ + case e_gemma4e_swa_layer: + type_str = "SWA"; + break; + case e_gemma4e_global_layer: + type_str = "Global"; + break; + case e_gemma4e_swa_layer_skip: + type_str = "SWA_Skip"; + break; + case e_gemma4e_global_layer_skip: + type_str = "Global_Skip"; + break; + default: + type_str = "Unknown"; + } + std::cout << "\t\t layer " << i << ": " << type_str << " " << std::endl; + } + std::cout << std::endl; + ) + } + catch (std::exception& e){ + header_print_r("ERROR", "Failed to parse model config: " << e.what()); + throw e; + } + + gemma4e_seq_gen_parameters_t seq_gen_params = { + .D = D, + .DH = DH, + .DQ = DQ, + .DK = DK, + .DV = DV, + .SWA_DH = SWA_DH, + .SWA_DQ = SWA_DQ, + .SWA_DK = SWA_DK, + .SWA_DV = SWA_DV, + .PLI_D = PLI_D, + .INTERMEDIATE_SIZE = INTERMEDIATE_SIZE, + .NUM_ATTENTION_HEADS = (int)config.get("num_attention_heads"), + .NUM_KEY_VALUE_HEADS = (int)config.get("num_key_value_heads"), + .SLIDING_WINDOW_SIZE = SLIDING_LENGTH, + .VOCAB_SIZE_PADDED = vocab_size_padded, + .enable_double_wide_mlp = enable_double_wide_mlp + }; + this->sequence = std::make_unique(seq_gen_params, MAX_L); + // Every sequence below addresses weights through the descriptors, so bind the + // description before generating any of them. + this->sequence->set_desc(&this->desc); + + DEBUG_BLOCK(1, + header_print_g("info", "VLM Enabled: " << (is_vlm ? "Yes" : "No")); + header_print_g("info", "Audio Model Enabled: " << (is_audio ? "Yes" : "No")); + ); + + if (is_vlm){ + this->gemma4e_image_encoder = std::make_unique(config, npu, this->parent_ptr); + } + if(is_audio){ + this->gemma4e_audio_encoder = std::make_unique(config, npu, this->parent_ptr); + } + // NOTE: order matter. The per layer input apps live on the image encoder's + // manager and must be created before layer.xclbin is registered, so this block + // is built here and handed to the prefill context at the end of the ctor. + std::unique_ptr pli_block = + std::make_unique(&this->desc, nullptr, this->gemma4e_image_encoder.get()); + layer_app_manager = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "layer.xclbin")); + + this->global_layer = layer_app_manager->create_app(); + this->swa_layer = layer_app_manager->create_app(); + this->global_skip_layer = layer_app_manager->create_app(); + this->swa_skip_layer = layer_app_manager->create_app(); + this->layer_pre_load = layer_app_manager->create_app(); + + if (!this->npu->is_preemption_enabled()){ + this->layers_run = layer_app_manager->create_runlist(); + } + + lm_head_app_manager = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "lm_head.xclbin")); + this->lm_head = lm_head_app_manager->create_app(); + + // allocate all buffers + rms_weights.resize(num_hidden_layers); + proj_weights.resize(num_hidden_layers); + pli_gate_up_weights.resize(num_hidden_layers); + kv_caches.resize(non_skip_layers); // only allocate kv_cache for non-skip layers + rope_rms_weights.resize(num_hidden_layers); + layer_scale = buffer(num_hidden_layers); + + this->x = layer_app_manager->create_bo_buffer((3 * D + 8191) / 8192 * 8192); // Iterative Hidden States (D) + Final RMS NORM (D) + INITIAL EMBEDDING (D) + + for (int layer_idx = 0; layer_idx < num_hidden_layers; layer_idx++){ + gemma4e_layer_type_t type = layer_types[layer_idx]; + size_t proj_buffer_size = desc.get_proj_weights_byte_size(type); + proj_weights[layer_idx] = layer_app_manager->create_bo_buffer(proj_buffer_size); // allocate 32MB for each layer, which is enough for current model scale. For larger model scale, we may need to dynamically load weights + this->proj_weights[layer_idx].memset((uint8_t)0); + this->proj_weights[layer_idx].sync_to_device(); + this->pli_gate_up_weights[layer_idx] = npu->create_bo_buffer(PLI_D * D * 2); // gate and up projection weights for per layer input, which will be added to the input of each layer after the first layer + DEBUG_BLOCK(2, + header_print("info", "Buffer sizes (in bytes): proj_weights=" + std::to_string(proj_weights[layer_idx].size())); + ) + } + + for (int layer_idx = 0; layer_idx < num_hidden_layers; layer_idx++){ + rms_weights[layer_idx] = layer_app_manager->create_bo_buffer(desc.get_rms_elems(layer_types[layer_idx])); // input_layer_norm (D) + post_attn_layer_norm (D) + pre_ffn_layer_norm (D) + post_ffn_layer_norm (D) for each layer from host. + } + for (int layer_idx = 0; layer_idx < num_hidden_layers; layer_idx++){ + gemma4e_layer_type_t type = layer_types[layer_idx]; + rope_rms_weights[layer_idx] = layer_app_manager->create_bo_buffer(desc.get_rope_rms_elems(type)); // COS/SIN (_DH) + Q_NORM (_DH) + K_NORM (_DH) + PLI_EMBED (PLI_D) + PLI_NORM (PLI_D) + POST_PLI_NORM (D) + layer_scale + } + + for (int layer_idx = 0; layer_idx < non_skip_layers; layer_idx++){ + gemma4e_layer_type_t type = layer_types[layer_idx]; + kv_caches[layer_idx] = layer_app_manager->create_bo_buffer(desc.get_kv_cache_size(type, MAX_L)); + kv_caches[layer_idx].memset((bf16)0); + kv_caches[layer_idx].sync_to_device(); + DEBUG_BLOCK(2, + header_print("info", "Buffer sizes (in bytes): kv_cache for layer " + std::to_string(layer_idx) + " = " + std::to_string(kv_caches[layer_idx].size())); + ) + } + is_checkpoint_valid = false; // checkpoint is not valid until we load kv cache to device after resizing buffers + pli_down_weights = layer_app_manager->create_bo_buffer(num_hidden_layers * PLI_D * D); // down projection weights for per layer input, which will be added to the input of each layer after the first layer + + lm_head_weights = lm_head_app_manager->create_bo_buffer(desc.get_lm_head_w_size()); + logits = lm_head_app_manager->create_bo_buffer(vocab_size_padded); + logits_valid = buffer(logits.data(), vocab_size); + + this->embedding = std::make_unique(vocab_size, D); + this->pli_embedding = std::make_unique(vocab_size, PLI_D * num_hidden_layers); // per layer input embedding, which will be added to the input of each layer after the first layer + + // Everything below this point belongs to the prefill path: its own xclbins, + // sequences, dequantized weights and scratch buffers. + this->prefill_ctx = std::make_unique( + npu, &this->desc, this->config, this->sequence.get(), std::move(pli_block), MAX_L + ); + + this->sequence->gen_lm_head_seq(this->lm_head.seq(), final_logit_softcapping); + this->lm_head_run = FLM_OVERRIDE(lm_head, + this->lm_head.create_run(this->logits, this->lm_head_weights, this->x)); + + // empty seq for preload xclbin + npu_sequence* pre_load_seq = this->layer_pre_load.seq(); + pre_load_seq->clear_cmds(); + pre_load_seq->cmds2seq(); + + // this->sequence->gen_layer_seq(this->linear_layer.seq(), 0, false); +} + +void gemma4e_npu::Impl::_set_rope_rms_weights(int idx){ + + buffer cos_buf(DH / 2); + buffer sin_buf(DH / 2); + buffer swa_cos_buf(SWA_DH / 2); + buffer swa_sin_buf(SWA_DH / 2); + + for (int j = 0; j < DH / 2; j++){ + cos_buf[j] = (bf16)cos(gemma4e_cpu_func::inv_freq_global[j] * idx); + sin_buf[j] = (bf16)sin(gemma4e_cpu_func::inv_freq_global[j] * idx); + } + for (int j = 0; j < SWA_DH / 2; j++){ + swa_cos_buf[j] = (bf16)cos(gemma4e_cpu_func::inv_freq_swa[j] * idx); + swa_sin_buf[j] = (bf16)sin(gemma4e_cpu_func::inv_freq_swa[j] * idx); + } + + for (int i = 0; i < num_hidden_layers; i++){ + bf16* w_rope_ptr = this->rope_rms_weights[i].data(); + if (is_swa_layer(layer_types[i])){ + memcpy(w_rope_ptr, swa_cos_buf.data(), SWA_DH * sizeof(bf16) / 2); + memcpy(w_rope_ptr + SWA_DH / 2, swa_sin_buf.data(), SWA_DH * sizeof(bf16) / 2); + } + else { + memcpy(w_rope_ptr, cos_buf.data(), DH * sizeof(bf16) / 2); + memcpy(w_rope_ptr + DH / 2, sin_buf.data(), DH * sizeof(bf16) / 2); + } + this->rope_rms_weights[i].sync_to_device(); + } +} + +void gemma4e_npu::Impl::set_context_length(int L){ + this->current_context_length = L; + DEBUG_BLOCK(2, + header_print_r("info", "Setting context length to " + std::to_string(L) + " for all layers in the sequence"); + ) + this->sequence->gen_layer_seq(this->global_layer.seq(), L + 1, e_gemma4e_global_layer); + this->sequence->gen_layer_seq(this->swa_layer.seq(), L + 1, e_gemma4e_swa_layer); + this->sequence->gen_layer_seq(this->global_skip_layer.seq(), L + 1, e_gemma4e_global_layer_skip); + this->sequence->gen_layer_seq(this->swa_skip_layer.seq(), L + 1, e_gemma4e_swa_layer_skip); + + if (!this->npu->is_preemption_enabled()){ + this->layers_run.reset(); + + DEBUG_BLOCK(2, + header_print_r("info", "Create runs for all layers in the sequence with the new context length"); + ) + for (uint32_t i = 0; i < num_hidden_layers; i++){ + + DEBUG_BLOCK(2, + header_print_r("info", "Generate run for layer " + std::to_string(i) + " of type " + std::to_string(layer_types[i])); + ) + switch(layer_types[i]){ + case e_gemma4e_global_layer: + this->layers_run.add(FLM_OVERRIDE(global_layer_run, + this->global_layer.create_run(this->x, this->proj_weights[i], this->rms_weights[i], this->rope_rms_weights[i], this->kv_caches[i]), layer_types[i], i, L, this->MAX_L)); + break; + case e_gemma4e_swa_layer: + this->layers_run.add(FLM_OVERRIDE(swa_layer_run, + this->swa_layer.create_run(this->x, this->proj_weights[i], this->rms_weights[i], this->rope_rms_weights[i], this->kv_caches[i]), layer_types[i], i, L, this->MAX_L)); + break; + case e_gemma4e_global_layer_skip: + this->layers_run.add(FLM_OVERRIDE(global_skip_layer_run, + this->global_skip_layer.create_run(this->x, this->proj_weights[i], this->rms_weights[i], this->rope_rms_weights[i], this->kv_caches[last_global_kv_cache_layer_idx]), layer_types[i], i, L, this->MAX_L)); + break; + case e_gemma4e_swa_layer_skip: + this->layers_run.add(FLM_OVERRIDE(swa_skip_layer_run, + this->swa_skip_layer.create_run(this->x, this->proj_weights[i], this->rms_weights[i], this->rope_rms_weights[i], this->kv_caches[last_swa_kv_cache_layer_idx]), layer_types[i], i, L, this->MAX_L)); + break; + } + } + } + + this->_set_rope_rms_weights(L); +} + +buffer gemma4e_npu::Impl::forward(int ids){ + if (is_preload_launched){ + this->pre_load_run.wait(); + is_preload_launched = false; + } + + _process_embedding(ids); + if (!this->npu->is_preemption_enabled()){ + this->layers_run.execute(); + this->layers_run.wait(); + } + else{ + for (uint32_t i = 0; i < num_hidden_layers; i++){ + switch(layer_types[i]){ + case e_gemma4e_global_layer: + FLM_OVERRIDE(global_layer, + this->global_layer(this->x, this->proj_weights[i], this->rms_weights[i], this->rope_rms_weights[i], this->kv_caches[i]), layer_types[i], i, this->current_context_length, this->MAX_L); + break; + case e_gemma4e_swa_layer: + FLM_OVERRIDE(swa_layer, + this->swa_layer(this->x, this->proj_weights[i], this->rms_weights[i], this->rope_rms_weights[i], this->kv_caches[i]), layer_types[i], i, this->current_context_length, this->MAX_L); + break; + case e_gemma4e_global_layer_skip: + FLM_OVERRIDE(global_skip_layer, + this->global_skip_layer(this->x, this->proj_weights[i], this->rms_weights[i], this->rope_rms_weights[i], this->kv_caches[last_global_kv_cache_layer_idx]), layer_types[i], i, this->current_context_length, this->MAX_L); + break; + case e_gemma4e_swa_layer_skip: + FLM_OVERRIDE(swa_skip_layer, + this->swa_skip_layer(this->x, this->proj_weights[i], this->rms_weights[i], this->rope_rms_weights[i], this->kv_caches[last_swa_kv_cache_layer_idx]), layer_types[i], i, this->current_context_length, this->MAX_L); + break; + } + + DEBUG_BLOCK(2, + header_print("info", "Finished layer " + std::to_string(i) + " of type " + std::to_string(layer_types[i]) + " for the current token"); + this->x.sync_from_device(); + ) + } + } + DEBUG_BLOCK(2, + std::cout << std::endl; + header_print("info", "Finished executing all layers for the current token:" + std::to_string(ids)); + this->x.sync_from_device(); + buffer valid_x = buffer(this->x.data(), D); + utils::print_matrix(valid_x, D); + this->x.sync_to_device(); + ) + + this->lm_head_run.start(); + this->set_context_length(this->current_context_length + 1); + this->lm_head_run.wait(); + this->logits.sync_from_device(); + DEBUG_BLOCK(2, + header_print("info", "Finished LM head run for the current token " + std::to_string(current_context_length)); + utils::print_matrix(this->logits_valid, vocab_size); + ) + + this->pre_load_run = this->layer_pre_load.create_run(); // create run for the preload xclbin, which has an empty sequence and will be used to preload the next layer's weights in the background while the current token is being processed + this->pre_load_run.start(); + is_preload_launched = true; + return this->logits_valid; +} + +buffer gemma4e_npu::Impl::prefill(std::vector& ids, void* payload){ + if (is_preload_launched){ + this->pre_load_run.wait(); + is_preload_launched = false; + } + DEBUG_BLOCK(1, + std::cout << "DEBUG: Entering prefill, input length: " << ids.size() << std::endl; + ) +#ifdef MVPREFILL + header_print("warning", "Prefilling with Decoder, Slow!"); + return this->_prefill_with_mv(ids); +#else + return this->_prefill_with_mm(ids, payload); +#endif +} + +buffer gemma4e_npu::Impl::_prefill_with_mv(std::vector& ids, void* payload){ + DEBUG_BLOCK(1, + header_print("info", "Entering prefill with MV, input length: " + std::to_string(ids.size())); + ) + for (int i = 0; i < ids.size(); i++){ + _process_embedding(ids[i]); + for (uint32_t l = 0; l < num_hidden_layers; l++){ + + DEBUG_BLOCK(2, + header_print_r("info", "Running layer " + std::to_string(l) + " of type " + std::to_string(layer_types[l]) + " for token " + std::to_string(ids[i])); + ) + // int test_layer = 15; + switch(layer_types[l]){ + case e_gemma4e_global_layer: + this->global_layer(this->x, this->proj_weights[l], this->rms_weights[l], this->rope_rms_weights[l], this->kv_caches[l]); + break; + case e_gemma4e_swa_layer: + this->swa_layer(this->x, this->proj_weights[l], this->rms_weights[l], this->rope_rms_weights[l], this->kv_caches[l]); + break; + case e_gemma4e_global_layer_skip: + this->global_skip_layer(this->x, this->proj_weights[l], this->rms_weights[l], this->rope_rms_weights[l], this->kv_caches[last_global_kv_cache_layer_idx]); + break; + case e_gemma4e_swa_layer_skip: + this->swa_skip_layer(this->x, this->proj_weights[l], this->rms_weights[l], this->rope_rms_weights[l], this->kv_caches[last_swa_kv_cache_layer_idx]); + break; + } + DEBUG_BLOCK(2, + header_print("info", "Finished layer " + std::to_string(l) + " of type " + std::to_string(layer_types[l]) + " for the current token"); + this->x.sync_from_device(); + this->x.sync_to_device(); + buffer valid_x = buffer(this->x.data(), D); + utils::print_matrix(valid_x, 256); + ) + } + + DEBUG_BLOCK(2, exit(0); ) + this->set_context_length(this->current_context_length + 1); + } + + DEBUG_BLOCK(1, + header_print("info", "Finished all layers"); + this->x.sync_from_device(); + this->x.sync_to_device(); + buffer valid_x = buffer(this->x.data(), D); + utils::print_matrix(valid_x, 256); + ) + this->lm_head_weights.sync_to_device(); + this->lm_head_run.start(); + this->lm_head_run.wait(); + this->logits.sync_from_device(); + + return logits_valid; +} + +buffer gemma4e_npu::Impl::_prefill_with_mm(std::vector& ids, void* payload){ + int L_in = ids.size(); + + std::unique_ptr reference; + buffer input_ids; + DEBUG_BLOCK(2, + std::cout << "DEBUG: Entering prefill with mm, input length: " << ids.size() << std::endl; + std::string reference_path = utils::path_join(config.model_path, "gemma4_ref.safetensors"); + reference = std::make_unique(reference_path); + reference->load_weights(input_ids, "input_ids"); + L_in = input_ids.size(); + ) + + // Sizes the batch and readies every block; sequences and buffers that already + // match this geometry are kept as they are. + const gemma4e_prefill_shape s = this->prefill_ctx->setup(L_in, this->current_context_length); + gemma4e_common_buffers& bufs = this->prefill_ctx->bufs; + + DEBUG_BLOCK(1, + std::cout << "DEBUG: L_begin: " << s.L_begin << ", L_end: " << s.L_end << std::endl; + std::cout << "DEBUG: L_begin_chunked: " << s.L_begin_chunked << ", L_end_chunked: " << s.L_end_chunked << ", L_padded: " << s.L_padded << std::endl; + std::cout << "DEBUG: L_effective: " << s.L_effective << std::endl; + std::cout << "DEBUG: sliding_l_begin: " << s.sliding_l_begin << std::endl; + ) + + for (int i = 0; i < s.L_effective; i++){ + DEBUG_BLOCK(2, + this->pli_embedding->forward(input_ids[i], this->prefill_ctx->pli_embed_row(i)); + this->embedding->forward(input_ids[i], this->prefill_ctx->residual_row(i)); + continue; + ) + if ((ids[i] == image_token_id) || (ids[i] == audio_token_id)){ + this->pli_embedding->forward(0, this->prefill_ctx->pli_embed_row(i)); + continue; + } + else{ + this->embedding->forward(ids[i], this->prefill_ctx->residual_row(i)); + this->pli_embedding->forward(ids[i], this->prefill_ctx->pli_embed_row(i)); + } + } + + if (payload != nullptr){ + std::vector image_embedding; + std::vector audio_embedding; + + gemma4e_multi_modal_payload_t* multi_modal_payload_ptr = (gemma4e_multi_modal_payload_t*)payload; + + if(is_audio && multi_modal_payload_ptr->audio_payload.num_audios > 0){ + audio_embedding = this->gemma4e_audio_encoder->encode(&multi_modal_payload_ptr->audio_payload); + } + if(is_vlm && multi_modal_payload_ptr->image_payload.num_images > 0){ + image_embedding = this->gemma4e_image_encoder->encode(&multi_modal_payload_ptr->image_payload); + } + + bf16* image_token_ptr = image_embedding.data(); + bf16* audio_token_ptr = audio_embedding.data(); + + for (int i = 0; i < ids.size(); i++) { + if (is_vlm && (ids[i] == image_token_id)){ + // copy image embedding to pli_embed_buffer + memcpy(this->prefill_ctx->residual_row(i).data(), image_token_ptr, D * sizeof(bf16)); + image_token_ptr += D; + } + else if (is_audio && (ids[i] == audio_token_id)){ + // copy audio embedding to pli_embed_buffer + memcpy(this->prefill_ctx->residual_row(i).data(), audio_token_ptr, D * sizeof(bf16)); + audio_token_ptr += D; + } + } + } + + DEBUG_BLOCK(2, + header_print("info", "Input embedding for the first " + std::to_string(s.L_effective) + " tokens:"); + buffer embedding_valid = buffer(bufs.pli_embed_buffer.data() + s.L_offset * PLI_D * num_hidden_layers, s.L_effective * PLI_D * num_hidden_layers); + utils::print_matrix(embedding_valid, PLI_D * num_hidden_layers); + ) + + DEBUG_BLOCK(2, + buffer embedding_ref; + reference->load_weights(embedding_ref, "input_embeds"); + buffer embedding_valid = buffer(bufs.residual_buffer.data() + s.L_offset * D, s.L_effective * D); + buffer embedding_valid_ref = buffer(embedding_ref.data() + s.L_offset * D, s.L_effective * D); + print_error_metrics(get_error_metrics(embedding_valid, embedding_valid_ref), "Input Embedding Error: "); + ) + + // fold the token embeddings into the per layer input stream, once for the batch + buffer pli_input_norm(this->rope_rms_weights[0].data() + desc.get_pli_norm_offset(layer_types[0]), PLI_D); + this->prefill_ctx->pli->pre_pass(s, this->pli_down_weights, pli_input_norm); + + for (uint32_t layer_idx = 0; layer_idx < num_hidden_layers; layer_idx++){ + gemma4e_layer_type_t type = layer_types[layer_idx]; + + DEBUG_BLOCK(2, + if (layer_idx > 0){ + buffer residual_overide; + reference->load_weights(residual_overide, "layer_" + std::to_string(layer_idx - 1)); + memcpy(bufs.residual_buffer.data() + s.L_offset * D, residual_overide.data() + s.L_offset * D, s.L_effective * D * sizeof(bf16)); + } + ) + + this->prefill_ctx->forward( + layer_idx, type, s, + this->proj_weights[layer_idx], + this->rms_weights[layer_idx], + this->rope_rms_weights[layer_idx], + this->pli_gate_up_weights[layer_idx], + this->kv_caches[is_skip_layer(type) + ? (is_swa_layer(type) ? last_swa_kv_cache_layer_idx : last_global_kv_cache_layer_idx) + : (int)layer_idx], + this->layer_scale[layer_idx], + reference.get() + ); + } + + DEBUG_BLOCK(2, + reference.reset(); + header_print_r("info", "DEBUG EXIT"); + exit(0); + ) + buffer predict = this->prefill_ctx->residual_row(s.L_effective - 1); + this->lm_head_weights.sync_to_device(); + get_logits(predict); + this->set_context_length(this->current_context_length + s.L_effective); + + this->pre_load_run = this->layer_pre_load.create_run(); + this->pre_load_run.start(); + is_preload_launched = true; + return logits_valid; +} + +buffer gemma4e_npu::Impl::get_k_cache(int layer_idx, int idx){ + this->kv_caches[layer_idx].sync_from_device(); + buffer k_cache(DK); + bf16* k_cache_ptr = this->kv_caches[layer_idx].data(); + uint32_t offset = idx * DK; + memcpy(static_cast(k_cache.data()), static_cast(k_cache_ptr + offset), DK * sizeof(bf16)); + + return k_cache; +} + +buffer gemma4e_npu::Impl::get_v_cache(int layer_idx, int idx){ + this->kv_caches[layer_idx].sync_from_device(); + buffer v_cache(DV); + bf16* v_cache_ptr = this->kv_caches[layer_idx].data(); + uint32_t offset = idx * DV; + memcpy(static_cast(v_cache.data()), static_cast(v_cache_ptr + offset), DV * sizeof(bf16)); + return v_cache; +} + +buffer gemma4e_npu::Impl::get_logits(buffer& predict){ + memcpy(this->x.data(), predict.data(), D * sizeof(bf16)); + this->x.sync_to_device(); + this->lm_head_run.start(); + this->lm_head_run.wait(); + this->logits.sync_from_device(); + return this->logits_valid; +} + +void gemma4e_npu::Impl::clear_context(){ + this->set_context_length(0); + for (int i = 0; i < non_skip_layers; i++){ + this->kv_caches[i].sync_from_device(); + memset(this->kv_caches[i].data(), 0, this->kv_caches[i].size() * sizeof(bf16)); + this->kv_caches[i].sync_to_device(); + } +} + +int gemma4e_npu::Impl::get_current_context_length(){ + return this->current_context_length; +} + +void gemma4e_npu::Impl::update_max_length(uint32_t MAX_L){ + if (MAX_L <= this->MAX_L){ + header_print("FLM", "New length is shorter than the current length, no need to update!"); + return; // no need to update + } + this->MAX_L = MAX_L; + size_t kv_cache_size = MAX_L * (DK + DV); + this->sequence->set_max_length(MAX_L); + + this->prefill_ctx->set_max_length(MAX_L); + + for (uint32_t i = 0; i < non_skip_layers; i++){ + if (!is_swa_layer(layer_types[i])){ + this->kv_caches[i] = buffer(kv_cache_size); + memset(this->kv_caches[i].data(), 0, kv_cache_size * sizeof(bf16)); + this->kv_caches[i].sync_to_device(); + } + else{ + memset(this->kv_caches[i].data(), 0, this->kv_caches[i].size() * sizeof(bf16)); + this->kv_caches[i].sync_to_device(); + } + } + is_checkpoint_valid = false; // force reload checkpoint to clear kv cache on device + this->set_context_length(0); // clear +} + +void gemma4e_npu::Impl::load_weights(Q4NX& q4nx){ + + this->embedding->init_weights(q4nx, "model.embed_tokens"); + this->pli_embedding->init_weights(q4nx, "model.per_layer_token_embd"); + + // Shared by every layer, so it is read once and handed to each of them. + buffer per_layer_norm; + q4nx.load_weights(per_layer_norm, "model.per_layer_proj_norm.weight"); + + for (int layer_idx = 0; layer_idx < num_hidden_layers; layer_idx++){ + this->desc.load_layer_weights(layer_idx, q4nx, + this->proj_weights[layer_idx], + this->rms_weights[layer_idx], + this->rope_rms_weights[layer_idx], + this->pli_gate_up_weights[layer_idx], + per_layer_norm, + this->layer_scale[layer_idx]); + this->pli_gate_up_weights[layer_idx].sync_to_device(); + this->proj_weights[layer_idx].sync_to_device(); + this->rms_weights[layer_idx].sync_to_device(); + this->rope_rms_weights[layer_idx].sync_to_device(); + DEBUG_BLOCK(1, + header_print("info", "Finished loading weights for layer " + std::to_string(layer_idx)); + ) + } + + // model.norm.weight lives in the second D-sized slot of x, behind the hidden state. + this->desc.load_head_weights(q4nx, this->lm_head_weights, this->x.data() + D, this->pli_down_weights); + + this->pli_down_weights.sync_to_device(); + this->lm_head_weights.sync_to_device(); + + DEBUG_BLOCK(1, + header_print("info", "Finished loading all layer weights"); + ) + + DEBUG_BLOCK(1, + header_print("info", "Finished updating sequence buffer offsets based on loaded weights"); + ) + + this->clear_context(); + + DEBUG_BLOCK(1, + header_print("info", "Finished clearing context after loading weights"); + ) + + // read qkv weights + if(is_vlm){ + DEBUG_BLOCK(1, + std::cout << "[DBG] gemma4e_npu::Impl::load_weights: entering VLM branch, path=" + << config.get("vision_model_weight", "") << std::endl; + ) + { + SafeTensors vision_weights(config.get("vision_model_weight", "")); + DEBUG_BLOCK(1, + std::cout << "[DBG] gemma4e_npu::Impl::load_weights: SafeTensors constructed" << std::endl; + ) + this->gemma4e_image_encoder->init_weights(vision_weights); + DEBUG_BLOCK(1, + std::cout << "[DBG] gemma4e_npu::Impl::load_weights: init_weights returned" << std::endl; + ) + } + DEBUG_BLOCK(1, + std::cout << "[DBG] gemma4e_npu::Impl::load_weights: SafeTensors out of scope" << std::endl; + ) + } + if(is_audio){ + SafeTensors audio_weights(config.get("audio_model_weight", "")); + this->gemma4e_audio_encoder->init_weights(audio_weights); + } +} + +int gemma4e_npu::Impl::checkpoint(){ + if (!is_checkpoint_valid){ + _allocate_checkpoint_buffers(); + } // use lazy allocation + header_print_r("FLM", "Creating checkpoint at context length " + std::to_string(this->current_context_length)); + checkpoint_context_length = this->current_context_length; + for (int i = 0; i < non_skip_layers; i++){ + if (!is_global_layer_idx(i)){ + this->kv_caches[i].sync_from_device(); + memcpy(this->kv_checkpoint[i].data(), this->kv_caches[i].data(), this->kv_caches[i].size() * sizeof(bf16)); + this->kv_caches[i].sync_to_device(); + } + } + is_checkpoint_valid = true; + return this->checkpoint_context_length; +} + +int gemma4e_npu::Impl::restore(){ + if (!is_checkpoint_valid){ + header_print("FLM", "No valid checkpoint found, cannot restore context!"); + return -1; + } + header_print_r("FLM", "Restoring checkpoint at context length " + std::to_string(checkpoint_context_length)); + this->current_context_length = checkpoint_context_length; + set_context_length(this->current_context_length); + for (int i = 0; i < non_skip_layers; i++){ + this->kv_caches[i].sync_from_device(); + if (!is_global_layer_idx(i)){ + this->kv_caches[i].sync_from_device(); + memcpy(this->kv_caches[i].data(), this->kv_checkpoint[i].data(), this->kv_caches[i].size() * sizeof(bf16)); + } + else { + int offset = (size_t)current_context_length * (DK); + // first half is K, second half is V, and they are stored contiguously in kv_cache + memset(this->kv_caches[i].data() + offset, 0, (this->kv_caches[i].size() / 2 - offset) * sizeof(bf16)); // zero out the part that is not restored for global layers, since we only restore the sliding part for global layers + memset(this->kv_caches[i].data() + this->kv_caches[i].size() / 2 + offset, 0, (this->kv_caches[i].size() / 2 - offset) * sizeof(bf16)); // zero out the second half for swa + } + this->kv_caches[i].sync_to_device(); + } + return this->current_context_length; +} + +void gemma4e_npu::Impl::_allocate_checkpoint_buffers(){ + // allocate kv cache checkpoint for sliding layers + this->kv_checkpoint.clear(); + this->kv_checkpoint.resize(non_skip_layers); + for (int i = 0; i < non_skip_layers; i++){ + if(!is_global_layer_idx(i)){ + this->kv_checkpoint[i] = buffer(this->kv_caches[i].size()); + } + } + is_checkpoint_valid = false; +} + +///@brief destructor of qwen3vl_npu +gemma4e_npu::Impl::~Impl(){ + // a preload run may still be in flight; it must complete before the + // xrt objects it references are torn down + if (is_preload_launched){ + try{ + this->pre_load_run.wait(); + } + catch (const std::exception& e){ + header_print("FLM", std::string("Failed to wait for the pending preload run: ") + e.what()); + } + is_preload_launched = false; + } +} +/* +qwen3vl_npu::Impl::~Impl(){ + for (int i = 0; i < config.get("num_hidden_layers"); i++){ + this->kv_caches[i].free(); + this->proj_weights[i].free(); + this->rms_weights[i].free(); + } + this->kv_caches.clear(); + this->rms_weights.clear(); + this->proj_weights.clear(); + this->rope_weights.release(); + this->dequantized_qkv_weights.release(); + this->dequantized_o_weights.release(); + this->x.release(); + this->lm_head.reset(); +} +*/ +// =============================================== +// Externals for qwen3vl_npu +// =============================================== +gemma4e_npu::gemma4e_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L){ + + this->load_vision_preprocess_parameters(config); + this->load_audio_preprocess_parameters(config); + this->_impl = new Impl(config, npu_instance, this, MAX_L); +} + +buffer gemma4e_npu::forward(int ids){ + return this->_impl->forward(ids); +} + +buffer gemma4e_npu::prefill(std::vector& ids, void* payload){ + try { + buffer result = this->_impl->prefill(ids, payload); + return result; + } + catch (const std::runtime_error& e) { + header_print("FLM", e.what()); + throw; + } +} + +void gemma4e_npu::set_context_length(int L){ + std::cout << "Setting context length is not supported for qwen3vl_npu!" << std::endl; +} + +void gemma4e_npu::load_weights(Q4NX& q4nx){ + this->_impl->load_weights(q4nx); +} + +void gemma4e_npu::update_max_length(uint32_t MAX_L){ + this->_impl->update_max_length(MAX_L); +} + +void gemma4e_npu::clear_context(){ + this->_impl->clear_context(); +} + +int gemma4e_npu::checkpoint(){ + return this->_impl->checkpoint(); +} + +int gemma4e_npu::restore(){ + return this->_impl->restore(); +} + +buffer gemma4e_npu::get_k_cache(int layer_idx, int idx){ + return this->_impl->get_k_cache(layer_idx, idx); +} + +buffer gemma4e_npu::get_v_cache(int layer_idx, int idx){ + return this->_impl->get_v_cache(layer_idx, idx); +} + +int gemma4e_npu::get_current_context_length(){ + return this->_impl->get_current_context_length(); +} + +gemma4e_npu::~gemma4e_npu(){ + delete this->_impl; +} diff --git a/src/detail/gemma4e_npu/gemma4e_npu_def.hpp b/src/detail/gemma4e_npu/gemma4e_npu_def.hpp new file mode 100644 index 000000000..abd13f8fb --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_npu_def.hpp @@ -0,0 +1,558 @@ +/// \file gemma4e_npu_def.hpp +/// \brief Gemma4e text backbone config + weight descriptors. +/// \author FastFlowLM Team +/// \note Owns everything that describes the model on disk and in device memory: +/// the config-derived dimensions, the per-layer weight descriptors (and +/// therefore the buffer layout the kernels read), and the weight loading +/// itself. Mirrors llama_desc / gemma4_12b_desc. +/// \note Gemma4e is a hybrid model in two independent ways: +/// - attention is sliding-window (SWA) except every `global_layer_period`-th +/// layer, which is full (global) attention; the two differ in head_dim. +/// - the last `num_kv_shared_layers` layers ("skip" layers) reuse the KV +/// cache of an earlier layer, so they carry no k/v projection and (when +/// `use_double_wide_mlp`) a twice-as-wide MLP. +/// That gives four distinct buffer layouts, indexed by gemma4e_layer_type_t: +/// [0]=swa, [1]=global, [2]=swa_skip, [3]=global_skip. +#pragma once +#include +#include +#include +#include "lm_config.hpp" +#include "weight_desc.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/gemma4e/gemma4e_npu.hpp" + +/// minimum tail padding (in bf16 elements) of the rope/rms buffer, kept in sync +/// with gemma4e_npu_sequence.hpp. +#ifndef MIN_BF16_PAD +#define MIN_BF16_PAD 32 +#endif + +/// \brief Descriptors of every weight of one decoder layer. +/// \note The order in which these are handed to the weight_container in +/// gemma4e_desc::_build_layer() IS the on-device layout, and both the +/// sequence generator (gemma4e_npu_sequence::gen_layer_seq) and the +/// dequant sequences depend on it: q/k/v contiguous, then o, then the +/// interleaved up/gate block, then down, then the three bf16 PLI +/// projections. +struct gemma4e_layer_weight_def { + // ---- projections, quantized, live in the per-layer proj buffer ---- + weight_desc_t attn_q; + weight_desc_t attn_k; //!< absent on skip layers, see has_kv + weight_desc_t attn_v; //!< absent on skip layers, see has_kv + weight_desc_t attn_output; + weight_desc_t ffn_up_gate; //!< up and gate interleaved, see UP_GATE_OUTDIM_REORDER + weight_desc_t ffn_down; + + // ---- per-layer-input projections, bf16, same buffer, after the quantized block ---- + weight_desc_t pli_down_proj; + weight_desc_t pli_gate_proj; + weight_desc_t pli_up_proj; + + // ---- rms norms, bf16, live in the per-layer rms buffer ---- + weight_desc_t input_layer_norm; + weight_desc_t post_attention_norm; + weight_desc_t pre_feedforward_norm; + weight_desc_t post_feedforward_norm; + + // ---- rope/rms buffer, bf16: [cos|sin, q_norm, k_norm, pli_embed, pli_norm, post_pli_norm, scale, pad] ---- + weight_desc_t rope_cos_sin; //!< not a checkpoint tensor: host-computed per position + weight_desc_t attn_q_norm; + weight_desc_t attn_k_norm; + weight_desc_t pli_embed; //!< scratch slot the PLI path writes into, not a checkpoint tensor + weight_desc_t pli_norm; //!< model.per_layer_proj_norm.weight, shared by every layer + weight_desc_t post_pli_norm; + weight_desc_t layer_output_scale; + + // ---- fused / aliased views, never registered in a container ---- + // These carry no storage of their own: they name a span that already exists so + // that the code moving it around can be handed a descriptor instead of an + // offset + two dimensions. Their offsets are copied from the real descriptor + // they alias once it has been placed. + weight_desc_t attn_qkv; //!< q(|k|v) as one block, what _move_weights ships + weight_desc_t ffn_up; //!< the up half of ffn_up_gate, what the dequant kernel picks out + weight_desc_t ffn_gate; //!< the gate half of ffn_up_gate + + bool has_kv = true; //!< false on skip layers (they share an earlier layer's cache) + + /// \brief The projection weights, in device-buffer order. + std::vector proj_all() { + std::vector out = {&attn_q}; + if (has_kv) { out.push_back(&attn_k); out.push_back(&attn_v); } + out.push_back(&attn_output); + out.push_back(&ffn_up_gate); + out.push_back(&ffn_down); + out.push_back(&pli_down_proj); + out.push_back(&pli_gate_proj); + out.push_back(&pli_up_proj); + return out; + } + + /// \brief The four layer norms, in device-buffer order. + std::vector rms_all() { + return {&input_layer_norm, &post_attention_norm, &pre_feedforward_norm, &post_feedforward_norm}; + } + + /// \brief The rope/rms buffer entries, in device-buffer order. + std::vector rope_all() { + return {&rope_cos_sin, &attn_q_norm, &attn_k_norm, + &pli_embed, &pli_norm, &post_pli_norm, &layer_output_scale}; + } + + /// \brief Every descriptor of the layer that owns storage. + std::vector all() { + std::vector out = proj_all(); + for (weight_desc_t* w : rms_all()) out.push_back(w); + for (weight_desc_t* w : rope_all()) out.push_back(w); + return out; + } +}; + +/// \brief Gemma4e text backbone description (config.json + model.q4nx). +struct gemma4e_desc { + /// up/gate are interleaved in slices of this many output rows + static constexpr int UP_GATE_OUTDIM_REORDER = 512; + /// \note Switching the body to another 4-bit dtype is a one-line change here: + /// every size, offset and DMA length below is derived from it. + static constexpr flm_dtype_t PROJ_DTYPE = flm_q41; + static constexpr flm_dtype_t LM_HEAD_DTYPE = flm_q41; + static constexpr flm_dtype_t PLI_DTYPE = flm_bf16; + + /// gemma4e moves weights in 32x256 hardware blocks (unlike the 16x256 of gemma4-12b). + static constexpr int QXNX_M = QXNX_ROW_BLOCK_SIZE; + static constexpr int QXNX_K = QXNX_COL_BLOCK_SIZE; + + /// \brief Bytes of one 32x256 hardware block of a quantized dtype. + static size_t block_bytes(flm_dtype_t dtype) { + return get_quantization_byte_size((size_t)QXNX_M * QXNX_K, dtype); + } + + // ---- dimensions (config.json) ---- + uint32_t num_hidden_layers = 0; + uint32_t D = 0; //!< hidden_size + uint32_t INTERMEDIATE_SIZE = 0; + uint32_t PLI_D = 0; //!< hidden_size_per_layer_input + uint32_t num_attention_heads = 0; + uint32_t num_kv_heads = 0; + uint32_t DH = 0; //!< global_head_dim + uint32_t DQ = 0; + uint32_t DK = 0; + uint32_t DV = 0; + uint32_t SWA_DH = 0; //!< head_dim + uint32_t SWA_DQ = 0; + uint32_t SWA_DK = 0; + uint32_t SWA_DV = 0; + uint32_t SLIDING_LENGTH = 0; + uint32_t vocab_size = 0; + uint32_t vocab_size_padded = 0; + uint32_t num_kv_shared_layers = 0; + uint32_t non_skip_layers = 0; + uint32_t global_layer_period = 0; + bool enable_double_wide_mlp = false; + f32 final_logit_softcapping = 0.0f; + + std::vector layer_types; + + // ---- weight descriptors ---- + /// one representative layout per layer kind, indexed by gemma4e_layer_type_t + gemma4e_layer_weight_def layer_defs[4]; + weight_desc_t final_norm; + weight_desc_t lm_head; + + // ---- aggregated buffer byte sizes, indexed by gemma4e_layer_type_t ---- + size_t proj_weights_byte_size[4] = {0, 0, 0, 0}; + size_t rms_weights_byte_size[4] = {0, 0, 0, 0}; + size_t rope_rms_byte_size[4] = {0, 0, 0, 0}; //!< payload only, without MIN_BF16_PAD + + gemma4e_desc() {} + gemma4e_desc(LM_Config config) { build(config); } + + /// \brief Whether layer `layer_idx` uses full (global) attention. + inline bool is_global_layer_idx(int layer_idx) const { + return ((layer_idx + 1) % (int)global_layer_period) == 0; + } + + /// \brief The representative descriptor of a layer kind. + inline gemma4e_layer_weight_def& weight_desc(gemma4e_layer_type_t type) { + return layer_defs[int(type)]; + } + /// \brief The representative descriptor of a layer, by index. + inline gemma4e_layer_weight_def& weight_desc_of(int layer_idx) { + return layer_defs[int(layer_types[layer_idx])]; + } + + // ---- layer-kind-specific dimensions ---- + inline uint32_t get_DH(gemma4e_layer_type_t t) const { return is_global_layer(t) ? DH : SWA_DH; } + inline uint32_t get_DQ(gemma4e_layer_type_t t) const { return is_global_layer(t) ? DQ : SWA_DQ; } + inline uint32_t get_DK(gemma4e_layer_type_t t) const { return is_global_layer(t) ? DK : SWA_DK; } + inline uint32_t get_DV(gemma4e_layer_type_t t) const { return is_global_layer(t) ? DV : SWA_DV; } + /// \brief MLP width of a layer kind: skip layers are twice as wide when enabled. + inline uint32_t get_intermediate_size(gemma4e_layer_type_t t) const { + return (is_skip_layer(t) && enable_double_wide_mlp) ? INTERMEDIATE_SIZE * 2 : INTERMEDIATE_SIZE; + } + /// \brief MLP width of the widest layer kind, which sizes the dequant buffers. + inline uint32_t get_max_intermediate_size() const { + return enable_double_wide_mlp ? INTERMEDIATE_SIZE * 2 : INTERMEDIATE_SIZE; + } + + // ---- buffer sizes ---- + inline size_t get_proj_weights_byte_size(gemma4e_layer_type_t t) const { return proj_weights_byte_size[int(t)]; } + /// \brief rms buffer size, in bf16 elements (input | post attn | pre ffn | post ffn). + inline size_t get_rms_elems(gemma4e_layer_type_t t) const { return rms_weights_byte_size[int(t)] / sizeof(bf16); } + /// \brief rope/rms buffer size, in bf16 elements, padding included. + /// \note The layer scale is the last entry, and MIN_BF16_PAD elements are kept + /// behind it so that the kernel never reads past the buffer. + inline size_t get_rope_rms_elems(gemma4e_layer_type_t t) { + return rope_rms_byte_size[int(t)] / sizeof(bf16) + MIN_BF16_PAD; + } + /// \brief kv cache size of one layer, in bf16 elements. + /// \note Sliding layers only ever keep `sliding_window` positions. + inline size_t get_kv_cache_size(gemma4e_layer_type_t t, uint32_t MAX_L) const { + return is_global_layer(t) ? (size_t)MAX_L * (DK + DV) + : (size_t)SLIDING_LENGTH * (SWA_DK + SWA_DV); + } + inline size_t get_lm_head_w_size() { return lm_head.get_size(); } + + // ---- rope/rms buffer offsets, in bf16 elements ---- + inline uint32_t rope_elem_offset(weight_desc_t& w) const { return (uint32_t)(w.offset / sizeof(bf16)); } + inline uint32_t get_q_norm_offset(gemma4e_layer_type_t t) { return rope_elem_offset(weight_desc(t).attn_q_norm); } + inline uint32_t get_k_norm_offset(gemma4e_layer_type_t t) { return rope_elem_offset(weight_desc(t).attn_k_norm); } + inline uint32_t get_pli_embed_offset(gemma4e_layer_type_t t) { return rope_elem_offset(weight_desc(t).pli_embed); } + inline uint32_t get_pli_norm_offset(gemma4e_layer_type_t t) { return rope_elem_offset(weight_desc(t).pli_norm); } + inline uint32_t get_post_pli_norm_offset(gemma4e_layer_type_t t) { return rope_elem_offset(weight_desc(t).post_pli_norm); } + inline uint32_t get_layer_scale_offset(gemma4e_layer_type_t t){ return rope_elem_offset(weight_desc(t).layer_output_scale); } + + /// \brief Parse config.json and lay out every weight. + inline void build(LM_Config& config) { + const nlohmann::json& jc = config._json_config; + + uint32_t head_dim = 0, global_head_dim = 0; + JSON_GET(num_hidden_layers, jc, "num_hidden_layers", 0, uint32_t); + JSON_GET(D, jc, "hidden_size", 0, uint32_t); + JSON_GET(INTERMEDIATE_SIZE, jc, "intermediate_size", 0, uint32_t); + JSON_GET(num_attention_heads, jc, "num_attention_heads", 0, uint32_t); + JSON_GET(num_kv_heads, jc, "num_key_value_heads", 0, uint32_t); + JSON_GET(head_dim, jc, "head_dim", 0, uint32_t); + JSON_GET(global_head_dim, jc, "global_head_dim", 0, uint32_t); + JSON_GET(SLIDING_LENGTH, jc, "sliding_window", 0, uint32_t); + JSON_GET(PLI_D, jc, "hidden_size_per_layer_input", 0, uint32_t); + JSON_GET(vocab_size, jc, "vocab_size", 0, uint32_t); + JSON_GET(num_kv_shared_layers, jc, "num_kv_shared_layers", 0, uint32_t); + JSON_GET(final_logit_softcapping, jc, "final_logit_softcapping", 0.0f, f32); + JSON_GET(enable_double_wide_mlp, jc, "use_double_wide_mlp", false, bool); + + DH = global_head_dim; + DQ = DH * num_attention_heads; + DK = DH * num_kv_heads; + DV = DK; + SWA_DH = head_dim; + SWA_DQ = SWA_DH * num_attention_heads; + SWA_DK = SWA_DH * num_kv_heads; + SWA_DV = SWA_DK; + vocab_size_padded = (vocab_size + 1024 - 1) / 1024 * 1024; + non_skip_layers = num_hidden_layers - num_kv_shared_layers; + + // The global-layer period is not in config.json; it follows from the model size. + if (D == 1536) global_layer_period = 5; // E2B + else if (D == 2560) global_layer_period = 6; // E4B + else throw std::runtime_error("gemma4e_desc: unsupported hidden size " + std::to_string(D)); + + layer_types.resize(num_hidden_layers); + for (uint32_t i = 0; i < num_hidden_layers; i++) { + int type = 0; + if (is_global_layer_idx(i)) type |= GEMMA4E_IS_GLOBAL_MASK; + if (i >= non_skip_layers) type |= GEMMA4E_IS_SKIP_MASK; + layer_types[i] = static_cast(type); + } + + DEBUG_BLOCK(1, + std::cout << "================ gemma4e_desc ================" << std::endl; + std::cout << std::left << std::setw(28) << "num_hidden_layers" << " = " << num_hidden_layers << std::endl; + std::cout << std::left << std::setw(28) << "D (hidden_size)" << " = " << D << std::endl; + std::cout << std::left << std::setw(28) << "PLI_D" << " = " << PLI_D << std::endl; + std::cout << std::left << std::setw(28) << "INTERMEDIATE_SIZE" << " = " << INTERMEDIATE_SIZE << std::endl; + std::cout << std::left << std::setw(28) << "DH / DQ / DK / DV" << " = " << DH << " / " << DQ << " / " << DK << " / " << DV << std::endl; + std::cout << std::left << std::setw(28) << "SWA DH / DQ / DK / DV" << " = " << SWA_DH << " / " << SWA_DQ << " / " << SWA_DK << " / " << SWA_DV << std::endl; + std::cout << std::left << std::setw(28) << "SLIDING_LENGTH" << " = " << SLIDING_LENGTH << std::endl; + std::cout << std::left << std::setw(28) << "global_layer_period" << " = " << global_layer_period << std::endl; + std::cout << std::left << std::setw(28) << "num_kv_shared_layers" << " = " << num_kv_shared_layers << std::endl; + std::cout << std::left << std::setw(28) << "use_double_wide_mlp" << " = " << enable_double_wide_mlp << std::endl; + std::cout << std::left << std::setw(28) << "final_logit_softcapping" << " = " << final_logit_softcapping << std::endl; + std::cout << std::left << std::setw(28) << "vocab_size" << " = " << vocab_size << " (" << vocab_size_padded << " padded)" << std::endl; + std::cout << "=============================================" << std::endl; + ) + + final_norm = weight_desc_t(flm_bf16, {(int64_t)D}, "model.norm.weight"); + lm_head = weight_desc_t(LM_HEAD_DTYPE, {(int64_t)D, (int64_t)vocab_size_padded}, "lm_head.weight"); + // Neither shares a buffer with anything, so they never go through a + // weight_container and would never get `added` set. Mark them registered + // here; their offset inside their own buffer is 0 by construction. + final_norm.indp(); + lm_head.indp(); + + for (int t = 0; t < 4; t++) _build_layer(static_cast(t)); + } + + /// \brief Copy a quantized weight into the device buffer in NPU block order. + /// \param dst destination inside the per-layer projection buffer + /// \param src the weight as stored in the q4nx file + /// \param col the input dimension of the weight + /// \param dtype the quantized dtype of the weight + /// \param vertical_blocks how many 32-row bands are interleaved + void reorder_cpy(u8* dst, buffer& src, const int col, flm_dtype_t dtype, + const int vertical_blocks = 2) + { + assert(is_quantize(dtype)); + const size_t a_block_size = block_bytes(dtype); + const int blocks_per_row = col / QXNX_K; + const int rows = src.size() / a_block_size / blocks_per_row; + + u8* dst_ptr = dst; + std::vector src_ptr(vertical_blocks); + for (int i = 0; i < vertical_blocks; i++) + src_ptr[i] = src.data() + i * a_block_size * blocks_per_row; + + for (int r = 0; r < rows; r += vertical_blocks) + { + for (int c = 0; c < blocks_per_row; c++) + { + for (int i = 0; i < vertical_blocks; i++) + { + memcpy(dst_ptr, src_ptr[i], a_block_size); + dst_ptr += a_block_size; + src_ptr[i] += a_block_size; + } + } + for (int i = 0; i < vertical_blocks; i++) + { + src_ptr[i] += (vertical_blocks - 1) * a_block_size * blocks_per_row; + if (src_ptr[i] + a_block_size * blocks_per_row > src.end()) + src_ptr[i] = src.data(); // useless padding + } + } + } + + /// \brief Load one decoder layer's weights from the checkpoint into its device buffers. + /// \param layer_idx which layer to load + /// \param q4nx the opened checkpoint + /// \param proj_buffer the layer's projection buffer (quantized block + bf16 PLI block) + /// \param rms_buffer the layer's four layer norms + /// \param rope_rms_buffer the layer's rope/norm/scale buffer + /// \param pli_gate_up_buffer the prefill-shaped copies of the PLI gate/up projections + /// \param per_layer_norm model.per_layer_proj_norm.weight, shared by every layer + /// \param layer_scale_out receives the layer output scale as a float + /// \note Every destination is addressed through its descriptor's offset, so the + /// physical layout lives in _build_layer() alone. + void load_layer_weights(int layer_idx, Q4NX& q4nx, + buffer& proj_buffer, + buffer& rms_buffer, + buffer& rope_rms_buffer, + buffer& pli_gate_up_buffer, + buffer& per_layer_norm, + float& layer_scale_out) + { + const gemma4e_layer_type_t type = layer_types[layer_idx]; + gemma4e_layer_weight_def& L = weight_desc(type); + u8* base = proj_buffer.data(); + + // ---- q (k, v) ---- + { + buffer w; + q4nx.load_weights(w, L.attn_q.format_name(layer_idx)); + reorder_cpy(base + L.attn_q.offset, w, D, L.attn_q.dtype); + L.attn_q.load(); + } + if (L.has_kv) { + for (weight_desc_t* desc : {&L.attn_k, &L.attn_v}) { + buffer w; + q4nx.load_weights(w, desc->format_name(layer_idx)); + reorder_cpy(base + desc->offset, w, D, desc->dtype); + desc->load(); + } + } + // ---- o ---- + { + buffer w; + q4nx.load_weights(w, L.attn_output.format_name(layer_idx)); + reorder_cpy(base + L.attn_output.offset, w, get_DQ(type), L.attn_output.dtype); + L.attn_output.load(); + } + // ---- up / gate, interleaved in slices of UP_GATE_OUTDIM_REORDER output rows ---- + { + buffer w_up, w_gate; + q4nx.load_weights(w_up, L.ffn_up.format_name(layer_idx)); + q4nx.load_weights(w_gate, L.ffn_gate.format_name(layer_idx)); + + const size_t chunk_size = get_quantization_byte_size((size_t)UP_GATE_OUTDIM_REORDER * D, + L.ffn_up_gate.dtype); + const size_t half_size = L.ffn_up_gate.get_size() / 2; + const size_t phases = half_size / chunk_size; + assert(phases * chunk_size == half_size && "up/gate must split into whole slices"); + + tensor_2d up_tensor(w_up, chunk_size, 0); + tensor_2d gate_tensor(w_gate, chunk_size, 0); + u8* w_ptr = base + L.ffn_up_gate.offset; + for (size_t i = 0; i < phases; i++) { + LOG_VERBOSE(1, "Copying up and gate weights to buffer, phase " << i + 1 << "/" << phases); + reorder_cpy(w_ptr, up_tensor[i], D, L.ffn_up_gate.dtype); + w_ptr += chunk_size; + reorder_cpy(w_ptr, gate_tensor[i], D, L.ffn_up_gate.dtype); + w_ptr += chunk_size; + } + L.ffn_up_gate.load(); + L.ffn_up.load(); + L.ffn_gate.load(); + } + // ---- down ---- + { + buffer w; + q4nx.load_weights(w, L.ffn_down.format_name(layer_idx)); + reorder_cpy(base + L.ffn_down.offset, w, get_intermediate_size(type), L.ffn_down.dtype); + L.ffn_down.load(); + } + // ---- per-layer-input projections, plain bf16 ---- + for (weight_desc_t* desc : {&L.pli_down_proj, &L.pli_gate_proj, &L.pli_up_proj}) { + buffer w; + q4nx.load_weights(w, desc->format_name(layer_idx)); + memcpy(base + desc->offset, w.data(), desc->get_size()); + desc->load(); + } + + // ---- the four layer norms ---- + for (weight_desc_t* desc : L.rms_all()) { + buffer w; + q4nx.load_weights(w, desc->format_name(layer_idx)); + memcpy(desc->locate_myself(rms_buffer).data(), w.data(), desc->get_size()); + desc->load(); + } + + // ---- rope/rms buffer ---- + // rope_cos_sin and pli_embed are scratch slots the runtime fills per position, + // so only the four checkpoint-backed entries are read here. + for (weight_desc_t* desc : {&L.attn_q_norm, &L.attn_k_norm, &L.post_pli_norm}) { + buffer w; + q4nx.load_weights(w, desc->format_name(layer_idx)); + memcpy(desc->locate_myself(rope_rms_buffer).data(), w.data(), desc->get_size()); + desc->load(); + } + memcpy(L.pli_norm.locate_myself(rope_rms_buffer).data(), + per_layer_norm.data(), L.pli_norm.get_size()); + L.pli_norm.load(); + { + buffer w; + q4nx.load_weights(w, L.layer_output_scale.format_name(layer_idx)); + layer_scale_out = (float)w[0]; + // The kernel broadcasts the scale, but it still reads a full vector's + // worth behind it, so zero the padding that follows. + bf16* scale_ptr = L.layer_output_scale.locate_myself(rope_rms_buffer).data(); + memset(scale_ptr, 0, (1 + MIN_BF16_PAD) * sizeof(bf16)); + scale_ptr[0] = w[0]; + L.layer_output_scale.load(); + } + + // ---- prefill-shaped copies of the PLI gate/up projections ---- + { + buffer w_gate, w_up; + q4nx.load_weights(w_gate, "model.layers." + std::to_string(layer_idx) + ".inp_gate.weight_prefill"); + q4nx.load_weights(w_up, "model.layers." + std::to_string(layer_idx) + ".per_layer_projection.weight_prefill"); + memcpy(pli_gate_up_buffer.data(), w_gate.data(), (size_t)PLI_D * D * sizeof(bf16)); + memcpy(pli_gate_up_buffer.data() + (size_t)PLI_D * D, w_up.data(), (size_t)PLI_D * D * sizeof(bf16)); + } + } + + /// \brief Load the weights that do not belong to any single layer. + /// \param q4nx the opened checkpoint + /// \param lm_head_buffer destination of the (reordered) lm head + /// \param final_norm_dst destination of model.norm.weight + /// \param pli_down_buffer destination of the prefill-shaped per-layer down projection + void load_head_weights(Q4NX& q4nx, + buffer& lm_head_buffer, + bf16* final_norm_dst, + buffer& pli_down_buffer) + { + { + buffer w; + q4nx.load_weights(w, final_norm.name); + memcpy(final_norm_dst, w.data(), final_norm.get_size()); + final_norm.load(); + } + { + buffer w; + q4nx.load_weights(w, lm_head.name); + // The lm head spans four columns of cores rather than two. + reorder_cpy(lm_head_buffer.data(), w, D, lm_head.dtype, 4); + lm_head.load(); + } + { + buffer w; + q4nx.load_weights(w, "model.per_layer_model_proj.weight_prefill"); + memcpy(pli_down_buffer.data(), w.data(), + (size_t)num_hidden_layers * D * PLI_D * sizeof(bf16)); + } + } + +private: + /// \brief Lay out one layer kind's three buffers and name every tensor. + /// \param type the layer kind whose layout is being built + /// \note The add_weight() call order is the physical layout; see the note on + /// gemma4e_layer_weight_def. + inline void _build_layer(gemma4e_layer_type_t type) { + gemma4e_layer_weight_def& L = layer_defs[int(type)]; + + const int64_t _DH = get_DH(type); + const int64_t _DQ = get_DQ(type); + const int64_t _DK = get_DK(type); + const int64_t _DV = get_DV(type); + const int64_t _IS = get_intermediate_size(type); + const int64_t d = D; + const int64_t pli = PLI_D; + + L.has_kv = !is_skip_layer(type); + + L.attn_q = weight_desc_t(PROJ_DTYPE, {d, _DQ}, "model.layers.%d.self_attn.q_proj.weight"); + L.attn_k = weight_desc_t(PROJ_DTYPE, {d, _DK}, "model.layers.%d.self_attn.k_proj.weight"); + L.attn_v = weight_desc_t(PROJ_DTYPE, {d, _DV}, "model.layers.%d.self_attn.v_proj.weight"); + L.attn_output = weight_desc_t(PROJ_DTYPE, {_DQ, d}, "model.layers.%d.self_attn.o_proj.weight"); + L.ffn_up_gate = weight_desc_t(PROJ_DTYPE, {d, 2 * _IS}, "model.layers.%d.mlp.{up,gate}_proj.weight"); + L.ffn_down = weight_desc_t(PROJ_DTYPE, {_IS, d}, "model.layers.%d.mlp.down_proj.weight"); + L.pli_down_proj = weight_desc_t(PLI_DTYPE, {pli, d}, "model.per_layer_model_proj.weight_layer%d"); + L.pli_gate_proj = weight_desc_t(PLI_DTYPE, {pli, d}, "model.layers.%d.inp_gate.weight"); + L.pli_up_proj = weight_desc_t(PLI_DTYPE, {pli, d}, "model.layers.%d.per_layer_projection.weight"); + + L.input_layer_norm = weight_desc_t(flm_bf16, {d}, "model.layers.%d.input_layernorm.weight"); + L.post_attention_norm = weight_desc_t(flm_bf16, {d}, "model.layers.%d.post_attention_layernorm.weight"); + L.pre_feedforward_norm = weight_desc_t(flm_bf16, {d}, "model.layers.%d.pre_feedforward_layernorm.weight"); + L.post_feedforward_norm = weight_desc_t(flm_bf16, {d}, "model.layers.%d.post_feedforward_layernorm.weight"); + + L.rope_cos_sin = weight_desc_t(flm_bf16, {_DH}, ""); + L.attn_q_norm = weight_desc_t(flm_bf16, {_DH}, "model.layers.%d.self_attn.q_norm.weight"); + L.attn_k_norm = weight_desc_t(flm_bf16, {_DH}, "model.layers.%d.self_attn.k_norm.weight"); + L.pli_embed = weight_desc_t(flm_bf16, {pli}, ""); + L.pli_norm = weight_desc_t(flm_bf16, {pli}, "model.per_layer_proj_norm.weight"); + L.post_pli_norm = weight_desc_t(flm_bf16, {d}, "model.layers.%d.post_layernorm.weight"); + L.layer_output_scale = weight_desc_t(flm_bf16, {1}, "model.layers.%d.layer_output_scale.weight"); + + weight_container proj, rms, rope; + for (weight_desc_t* w : L.proj_all()) proj.add_weight(*w); + for (weight_desc_t* w : L.rms_all()) rms.add_weight(*w); + for (weight_desc_t* w : L.rope_all()) rope.add_weight(*w); + + proj_weights_byte_size[int(type)] = proj.get_size(); + rms_weights_byte_size[int(type)] = rms.get_size(); + rope_rms_byte_size[int(type)] = rope.get_size(); + + // Aliased views over regions that are already placed above. q/k/v are + // contiguous and are shipped and dequantized as one block; up/gate share + // one interleaved region that the dequant kernel splits by output mode. + L.attn_qkv = weight_desc_t(PROJ_DTYPE, {d, L.has_kv ? (_DQ + _DK + _DV) : _DQ}, + ""); + L.attn_qkv.offset = L.attn_q.offset; + L.attn_qkv.indp(); + + L.ffn_up = weight_desc_t(PROJ_DTYPE, {d, _IS}, "model.layers.%d.mlp.up_proj.weight"); + L.ffn_gate = weight_desc_t(PROJ_DTYPE, {d, _IS}, "model.layers.%d.mlp.gate_proj.weight"); + L.ffn_up.offset = L.ffn_gate.offset = L.ffn_up_gate.offset; + L.ffn_up.indp(); + L.ffn_gate.indp(); + } +}; diff --git a/src/detail/gemma4e_npu/gemma4e_npu_detail.hpp b/src/detail/gemma4e_npu/gemma4e_npu_detail.hpp new file mode 100644 index 000000000..9e214c7f8 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_npu_detail.hpp @@ -0,0 +1,181 @@ +#pragma once +#include "models/gemma4e/gemma4e_npu.hpp" +#include "gemma4e_npu_sequence.hpp" +#include "gemma4e_image.hpp" +#include "modules/gemm.hpp" +#include "reorder_cpy.hpp" +#include "avx512_util.hpp" +#include "gemma4e_audio.hpp" +#include "embedding_q8_0.hpp" +#include "gemma4e_prefill.hpp" + +struct gemma4e_npu::Impl{ + /// @brief constexprs + + // some model specific template variables + static constexpr int boi_token_id = 255999; // begin of image token id + static constexpr int image_token_id = 258880; // image token id + static constexpr int eoi_token_id = 258882; // end of image token id + + static constexpr int boa_token_id = 256000; // begin of audio token id + static constexpr int audio_token_id = 258881; // audio token id + static constexpr int eoa_token_id = 258883; // end of audio token id + /// @brief constexprs + //TODO: FIXME: + + static constexpr int LC = 16; + + size_t V_offset; + + int vocab_size; + int vocab_size_padded; + bool is_preload_launched; + + LM_Config config; + npu_xclbin_manager *npu; + + npu_app_manager* layer_app_manager; + npu_app_manager* lm_head_app_manager; + + npu_app global_layer; + npu_app swa_layer; + npu_app global_skip_layer; + npu_app swa_skip_layer; + npu_app layer_pre_load; + flm_rt::run pre_load_run; + + // per layer input + npu_app per_layer_input_down_proj; + npu_app per_layer_input_up_proj; + + npu_app lm_head; + + std::unique_ptr sequence; + std::unique_ptr gemma4e_image_encoder; + std::unique_ptr gemma4e_audio_encoder; + + gemma4e_npu* parent_ptr; + + int D; + int DH; + int DQ; + int DK; + int DV; + int SWA_DH; + int SWA_DQ; + int SWA_DK; + int SWA_DV; + + int MAX_L; + int HIDDEN_SIZE; + int INTERMEDIATE_SIZE; + int PLI_D; + int SLIDING_LENGTH; + + int sliding_layer_interval; + int num_hidden_layers; + int num_kv_shared_layers; + int non_skip_layers; + int global_layer_period; + int last_swa_kv_cache_layer_idx; + int last_global_kv_cache_layer_idx; + float final_logit_softcapping; + + bool enable_double_wide_mlp; + + /// \brief Config-derived dimensions and every weight descriptor of the model. + /// \note Owns the buffer layout: the dequant sequences, the decode sequence + /// generator and load_weights all address weights through it, so the + /// quantization type is switchable from gemma4e_desc::PROJ_DTYPE alone. + gemma4e_desc desc; + std::vector layer_types; + + // size + std::vector> rms_weights; + std::vector> rope_rms_weights; + std::vector> proj_weights; + std::vector> kv_caches; + std::vector> pli_gate_up_weights; + + std::vector> kv_checkpoint; + + buffer pli_down_weights; + buffer layer_scale; + /// Everything the prefill path needs: its own xclbins, sequences and buffers. + std::unique_ptr prefill_ctx; + + buffer x; + std::unique_ptr embedding; + std::unique_ptr pli_embedding; + + buffer logits; + buffer logits_valid; + buffer lm_head_weights; + + flm_rt::runlist layers_run; + flm_rt::run lm_head_run; + flm_rt::runlist dequant_all; + int current_context_length; + + int checkpoint_context_length; + + /// @brief initialize the qwen_npu + /// @param config + /// @param npu_instance + Impl(LM_Config config, npu_xclbin_manager *npu_instance, gemma4e_npu* parent_ptr, int MAX_L = 4096); + ~Impl(); // waits for any in-flight preload run before tearing down + + /// @brief forward the qwen_npu + buffer forward(int ids); + buffer prefill(std::vector& ids, void* payload = nullptr); + buffer _prefill_with_mv(std::vector& ids, void* payload = nullptr); + buffer _prefill_with_mm(std::vector& ids, void* payload = nullptr); + void _load_attn_layer_weights(Q4NX& q4nx, int layer_idx); + void _load_linear_layer_weights(Q4NX& q4nx, int layer_idx); + + void set_context_length(int L); + void load_weights(Q4NX& q4nx); + void update_max_length(uint32_t MAX_L); + void clear_context(); + + bool is_checkpoint_valid; + bool is_vlm; + bool is_audio; + void _allocate_checkpoint_buffers(); + int checkpoint(); + int restore(); + + buffer get_k_cache(int layer_idx, int idx); + buffer get_v_cache(int layer_idx, int idx); + buffer get_logits(buffer& x); + + int get_current_context_length(); + + void _set_rope_rms_weights(int idx); + + inline bool is_global_layer_idx(int layer_idx) {return (layer_idx % global_layer_period) == (global_layer_period - 1); } + inline void _process_embedding(int idx){ + buffer embedding_output = this->embedding->forward(idx); + DEBUG_BLOCK(2, + header_print("debug", "Embedding output for token idx " + std::to_string(idx)); + utils::print_matrix(embedding_output, D); + ) + memcpy(this->x.data(), embedding_output.data(), D * sizeof(bf16)); + memcpy(this->x.data() + D * 2, embedding_output.data(), D * sizeof(bf16)); // copy to the second half for swa + this->x.sync_to_device(); + buffer pli_embedding_output = this->pli_embedding->forward(idx); + + DEBUG_BLOCK(2, + header_print("debug", "PLI Embedding output for token idx " + std::to_string(idx)); + utils::print_matrix(pli_embedding_output, PLI_D); + ) + + bf16* p_pli = pli_embedding_output.data(); + for (int i = 0; i < num_hidden_layers; i++) { + memcpy(this->rope_rms_weights[i].data() + desc.get_pli_embed_offset(layer_types[i]), p_pli, PLI_D * sizeof(bf16)); + p_pli += PLI_D; + this->rope_rms_weights[i].sync_to_device(); + } + } + +}; diff --git a/src/detail/gemma4e_npu/gemma4e_npu_sequence.cpp b/src/detail/gemma4e_npu/gemma4e_npu_sequence.cpp new file mode 100644 index 000000000..b6da27dd4 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_npu_sequence.cpp @@ -0,0 +1,1279 @@ +#include "gemma4e_npu_sequence.hpp" + +gemma4e_npu_sequence::gemma4e_npu_sequence(gemma4e_seq_gen_parameters_t params, uint32_t MAX_L){ + D = params.D; + DH = params.DH; + DQ = params.DQ; + DK = params.DK; + DV = params.DV; + SWA_DH = params.SWA_DH; + SWA_DQ = params.SWA_DQ; + SWA_DK = params.SWA_DK; + SWA_DV = params.SWA_DV; + PLI_D = params.PLI_D; + num_attn_heads = params.NUM_ATTENTION_HEADS; + num_kv_heads = params.NUM_KEY_VALUE_HEADS; + num_kv_per_round = num_kv_heads; + INTERMEDIATE_SIZE = params.INTERMEDIATE_SIZE; + VOCAB_SIZE = params.VOCAB_SIZE_PADDED; + SLIDING_LENGTH = params.SLIDING_WINDOW_SIZE; + enable_double_wide_mlp = params.enable_double_wide_mlp; + + this->MAX_L = MAX_L; + + if (INTERMEDIATE_SIZE == 6144){ // e2b + rtp_addresses = e2b_rtp_addresses; + } + else if (INTERMEDIATE_SIZE == 10240){ + rtp_addresses = e4b_rtp_addresses; + } + else{ + std::cerr << "Unsupported intermediate size: " << INTERMEDIATE_SIZE << std::endl; + throw std::runtime_error("SEQ Unsupported intermediate size"); + } + + DEBUG_BLOCK(1, + header_print_g("info", "Sequence Gen Params: "); + std::cout << "\tD: " << D << std::endl; + std::cout << "\tDH: " << DH << std::endl; + std::cout << "\tDQ: " << DQ << std::endl; + std::cout << "\tDK: " << DK << std::endl; + std::cout << "\tDV: " << DV << std::endl; + std::cout << "\tSWA_DH: " << SWA_DH << std::endl; + std::cout << "\tSWA_DQ: " << SWA_DQ << std::endl; + std::cout << "\tSWA_DK: " << SWA_DK << std::endl; + std::cout << "\tSWA_DV: " << SWA_DV << std::endl; + std::cout << "\tPLI_D: " << PLI_D << std::endl; + std::cout << "\tMAX_L: " << MAX_L << std::endl; + std::cout << "\tINTERMEDIATE_SIZE: " << INTERMEDIATE_SIZE << std::endl; + std::cout << "\tNUM_ATTENTION_HEADS: " << num_attn_heads << std::endl; + std::cout << "\tNUM_KEY_VALUE_HEADS: " << num_kv_heads << std::endl; + std::cout << "\tSLIDING_WINDOW_SIZE: " << SLIDING_LENGTH << std::endl; + std::cout << "\tVOCAB_SIZE_PADDED: " << VOCAB_SIZE << std::endl; + ) + DEBUG_BLOCK(1, + header_print_g("info", "RTP ADDRESS BOOK: "); + std::cout << "\tl_qk_address: " << rtp_addresses.l_qk_address << std::endl; + std::cout << "\tl_kv_address: " << rtp_addresses.l_kv_address << std::endl; + std::cout << "\tswa_l_qk_address: " << rtp_addresses.swa_l_qk_address << std::endl; + std::cout << "\tswa_l_kv_address: " << rtp_addresses.swa_l_kv_address << std::endl; + std::cout << "\tproj_swa_address: " << rtp_addresses.proj_swa_address << std::endl; + std::cout << "\tproj_skip_address: " << rtp_addresses.proj_skip_address << std::endl; + std::cout << "\trms_swa_address: " << rtp_addresses.rms_swa_address << std::endl; + std::cout << "\trms_skip_address: " << rtp_addresses.rms_skip_address << std::endl; + std::cout << "\trope_skip_kv_address: " << rtp_addresses.rope_skip_kv_address << std::endl; + std::cout << "\tswa_rope_skip_kv_address: " << rtp_addresses.swa_rope_skip_kv_address << std::endl; + ) +} + +void gemma4e_npu_sequence::set_max_length(const uint32_t MAX_L){ + this->MAX_L = MAX_L; +} + +void gemma4e_npu_sequence::gen_layer_seq(npu_sequence* seq, const uint32_t L, gemma4e_layer_type_t layer_type){ + constexpr size_t CT_lock_address_base = 0x000001F000; + LAYER_SPECIFIC_DIM(layer_type) + DEBUG_BLOCK(2, + header_print_g("info", "Generating sequence for layer type " + std::to_string(layer_type) + " with L = " + std::to_string(L)); + header_print_g("info", "Layer specific dimensions: "); + std::cout << "\t_ D: " << D << std::endl; + std::cout << "\t_ DH: " << _DH << std::endl; + std::cout << "\t_ DQ: " << _DQ << std::endl; + std::cout << "\t_ DK: " << _DK << std::endl; + std::cout << "\t_ DV: " << _DV << std::endl; + std::cout << "\t_ INTERMEDIATE_SIZE: " << _INTERMEDIATE_SIZE << std::endl; + std::cout << "\t_ QKV_OFFSET: " << weight_elem_offset(layer_weights(layer_type).attn_qkv) << std::endl; + std::cout << "\t_ O_OFFSET: " << weight_elem_offset(layer_weights(layer_type).attn_output) << std::endl; + std::cout << "\t_ UP_OFFSET: " << weight_elem_offset(layer_weights(layer_type).ffn_up_gate) << std::endl; + std::cout << "\t_ DOWN_OFFSET: " << weight_elem_offset(layer_weights(layer_type).ffn_down) << std::endl; + std::cout << "\t_ PLI_DOWN_OFFSET: " << weight_elem_offset(layer_weights(layer_type).pli_down_proj) << std::endl; + std::cout << "\t_ PLI_GATE_OFFSET: " << weight_elem_offset(layer_weights(layer_type).pli_gate_proj) << std::endl; + std::cout << "\t_ PLI_UP_OFFSET: " << weight_elem_offset(layer_weights(layer_type).pli_up_proj) << std::endl; + std::cout << "\t_ IS_SWA: " << is_swa_layer(layer_type) << std::endl; + std::cout << "\t_ IS_SKIP: " << is_skip_layer(layer_type) << std::endl; + ) + int proj_rtp_lock_id = 6; + int rms_rtp_lock_id = 6; + int glu_rtp_lock_id = 6; + seq->clear_cmds(); + int L_local = L; + if (is_swa_layer(layer_type)) { + if (L > SLIDING_LENGTH){ + L_local = SLIDING_LENGTH; + } + } + + for (int i = 0; i < 16; i++){ + seq->rtp_write(proj_tiles[i], rtp_addresses.proj_swa_address, is_swa_layer(layer_type) ? 1 : 0); + seq->rtp_write(proj_tiles[i], rtp_addresses.proj_skip_address, is_skip_layer(layer_type) ? 1 : 0); + seq->rtp_write(proj_tiles[i], CT_lock_address_base + 16 * proj_rtp_lock_id, 1); // set lock to 1 + } + seq->rtp_write(attn_qk_tile, rtp_addresses.l_qk_address, L_local); + seq->rtp_write(attn_kv_tile, rtp_addresses.l_kv_address, L_local); + seq->rtp_write(swa_attn_qk_tile, rtp_addresses.swa_l_qk_address, L_local); + seq->rtp_write(swa_attn_kv_tile, rtp_addresses.swa_l_kv_address, L_local); + seq->rtp_write(rms_tile, rtp_addresses.rms_swa_address, is_swa_layer(layer_type) ? 1 : 0); + seq->rtp_write(rms_tile, rtp_addresses.rms_skip_address, is_skip_layer(layer_type) ? 1 : 0); + seq->rtp_write(rope_ct, rtp_addresses.rope_skip_kv_address, is_skip_layer(layer_type) ? 1 : 0); + seq->rtp_write(swa_rope_ct, rtp_addresses.swa_rope_skip_kv_address, is_skip_layer(layer_type) ? 1 : 0); + seq->rtp_write(glu_tile, rtp_addresses.glu_skip_address, (is_skip_layer(layer_type) && enable_double_wide_mlp) ? 1 : 0); + seq->rtp_write(rms_tile, CT_lock_address_base + 16 * rms_rtp_lock_id, 1); // set lock to 1 + seq->rtp_write(glu_tile, CT_lock_address_base + 16 * glu_rtp_lock_id, 1); // set lock to 1 + + _send_hidden_states(seq); + _send_rms_weights(seq); + _send_rope_rms_weights(seq, layer_type); + + // receive y + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + x_arg_id, + S2MM, + xr_tile, + bd_15, + it_channel_0, + {0, 0, 0, 0}, + {1, 1, 1, (uint32_t)D}, + {0, 0, 0, 1}, + -1, 0, true + ); + gemma4e_layer_weight_def& W = layer_weights(layer_type); + // attn_qkv already spans q alone on skip layers, which carry no k/v projection. + _move_weights(seq, W.attn_qkv); + if (!is_skip_layer(layer_type)){ + _receive_kv_cache(seq, L, layer_type); + } + + _gen_pli_path_seq(seq, layer_type); + _move_kv_cache(seq, L, layer_type); + _move_weights(seq, W.attn_output); + _move_weights(seq, W.ffn_up_gate); + _move_weights(seq, W.ffn_down); + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + proj_arg_id, + MM2S, + IT5, + bd_14, + it_channel_0, + {0, 0, 0, weight_elem_offset(layer_weights(layer_type).pli_gate_proj)}, + {1, 1, 1, (uint32_t)(PLI_D * D)}, + {0, 0, 0, 1}, + -1, 0, true, aggressive_cache + ); + + seq->npu_dma_wait( + IT5, + MM2S, + it_channel_0 + ); + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + proj_arg_id, + MM2S, + IT5, + bd_15, + it_channel_1, + {0, 0, 0, weight_elem_offset(layer_weights(layer_type).pli_up_proj)}, + {1, 1, 1, (uint32_t)(PLI_D * D)}, + {0, 0, 0, 1}, + -1, 0, true, aggressive_cache + ); + + seq->npu_dma_wait( + IT5, + MM2S, + it_channel_1 + ); + // wait for receiving y + seq->npu_dma_wait( + xr_tile, + S2MM, + it_channel_0 + ); + seq->cmds2seq(); +} + +void gemma4e_npu_sequence::gen_lm_head_seq(npu_sequence* seq, float final_scale){ + static constexpr int y_arg_id = 0; + static constexpr int w_arg_id = 1; + static constexpr int x_arg_id = 2; + static constexpr int M_PER_ROUND = 32 * 32; + static constexpr int M = 32; + static constexpr int M_PER_COL = M * 4; + static constexpr int COLS = 8; + static npu_tiles ITs[] = {IT0, IT1, IT2, IT3, IT4, IT5, IT6, IT7}; + const uint32_t final_scale_address = this->rtp_addresses.lm_head_final_tune_address; + assert(VOCAB_SIZE % M_PER_ROUND == 0); + int rounds = VOCAB_SIZE / M_PER_ROUND; + size_t TOTAL_W_SIZE = size_t(D) * size_t(VOCAB_SIZE) * 5 / 8 / 2; + size_t WEIGHTS_PER_IT = TOTAL_W_SIZE / std::size(ITs); + size_t W_PER_COL = (M_PER_COL * D * 5 / 8 / 2); + size_t W_PER_ROUND = W_PER_COL * COLS; + + size_t w_offset = WEIGHTS_PER_IT; + size_t y_offset = VOCAB_SIZE / std::size(ITs); + uint32_t scale_int_view = *((uint32_t*)(&final_scale)); + seq->clear_cmds(); + for (int row = 0; row < 4; row++){ + for (int col = 0; col < 8; col++){ + npu_tiles tile = get_tile(row + 2, col); + seq->rtp_write(tile, final_scale_address, scale_int_view); + } + } + seq->npu_dma_memcpy_nd( + sizeof(bf16), + x_arg_id, + MM2S, + ITs[0], + npu_bd_id(bd_0), + it_channel_0, + {0, 0, 0, 0}, + {1, 1, 1, (uint32_t)D * 2}, + {0, 0, 0, 1}, + -1, 0, false, aggressive_cache + ); + + for (int r = 0; r < rounds; r++) { + int bd_offset = (r % 2) * 8; + npu_bd_id bd_y = npu_bd_id(bd_1 + bd_offset); + npu_bd_id bd_w = npu_bd_id(bd_2 + bd_offset); + for (size_t col = 0; col < std::size(ITs); col++){ + size_t w_col_offset = r * W_PER_ROUND + col * W_PER_COL; + uint32_t y_offset = r * M_PER_ROUND + col * M_PER_COL; + seq->npu_dma_memcpy_nd( + sizeof(bf16), + y_arg_id, + S2MM, + ITs[col], + bd_y, + it_channel_0, + {0, 0, 0, (uint32_t)(y_offset)}, + {1, 1, 1, (uint32_t)M_PER_COL}, + {0, 0, 0, 1}, + -1, 0, true, aggressive_cache + ); + seq->npu_dma_memcpy_nd( + sizeof(bf16), + w_arg_id, + MM2S, + ITs[col], + bd_w, + it_channel_1, + {0, 0, 0, (uint32_t)(w_col_offset)}, + {1, 1, 1, (uint32_t)W_PER_COL}, + {0, 0, 0, 1}, + -1, 0, false, aggressive_cache + ); + if (r > 0){ + seq->npu_dma_wait( + ITs[col], + S2MM, + it_channel_0 + ); + } + } + } + for (size_t col = 0; col < std::size(ITs); col++){ + seq->npu_dma_wait( + ITs[col], + S2MM, + it_channel_0 + ); + } + seq->cmds2seq(); +} + +void gemma4e_npu_sequence::_send_hidden_states(npu_sequence* seq){ + // send x + seq->npu_dma_memcpy_nd( + sizeof(bf16), + x_arg_id, + MM2S, + xr_tile, + bd_0, + it_channel_0, + {0, 0, 0, 0}, + {1, 1, 1, (uint32_t)D}, + {0, 0, 0, 1}, + 0, 0, true, aggressive_cache + ); + seq->npu_dma_wait( + xr_tile, + MM2S, + it_channel_0 + ); +} + +void gemma4e_npu_sequence::_send_rms_weights(npu_sequence* seq){ + seq->npu_dma_memcpy_nd( + sizeof(bf16), + rms_arg_id, + MM2S, + xr_tile, + bd_1, + it_channel_1, + {0, 0, 0, 0}, + {1, 1, 1, (uint32_t)D * 4}, + {0, 0, 0, 1}, + 0, 0, true, aggressive_cache + ); + + seq->npu_dma_wait( + xr_tile, + MM2S, + it_channel_1 + ); +} + +void gemma4e_npu_sequence::_send_rope_rms_weights(npu_sequence* seq, gemma4e_layer_type_t layer_type){ + LAYER_SPECIFIC_DIM(layer_type) + if (is_swa_layer(layer_type)){ + seq->npu_dma_memcpy_nd( + sizeof(bf16), + rope_rms_arg_id, + MM2S, + xr_tile, + bd_4, + it_channel_1, + {0, 0, 0, 0}, + {1, 1, 1, (uint32_t)(_DH * 3)}, + {0, 0, 0, 1}, + 1, 0, true, no_cache + ); + seq->npu_dma_wait( + xr_tile, + MM2S, + it_channel_1 + ); + } + else { + seq->npu_dma_memcpy_nd( + sizeof(bf16), + rope_rms_arg_id, + MM2S, + xr_tile, + bd_4, + it_channel_0, + {0, 0, 0, 0}, + {1, 1, 1, (uint32_t)(_DH * 3)}, + {0, 0, 0, 1}, + 1, 0, true, no_cache + ); + seq->npu_dma_wait( + xr_tile, + MM2S, + it_channel_0 + ); + } +} + +void gemma4e_npu_sequence::_receive_kv_cache(npu_sequence* seq, const int L, gemma4e_layer_type_t layer_type){ + LAYER_SPECIFIC_DIM(layer_type) + DEBUG_BLOCK(2, + header_print_g("info", "Moving KV cache for layer type " + std::to_string(layer_type) + " with L = " + std::to_string(L)); + std::cout << "\t_ L: " << L << std::endl; + std::cout << "\t_ _DK: " << _DK << std::endl; + std::cout << "\t_ _DV: " << _DV << std::endl; + std::cout << "\t_ MAX_L: " << MAX_L << std::endl; + ) + int L_local = L - 1; + if (is_swa_layer(layer_type)) { + L_local = L_local % SLIDING_LENGTH; + } + uint32_t kv_cache_size; + if (is_swa_layer(layer_type)){ + kv_cache_size = SLIDING_LENGTH * (_DK + _DV); + } + else { + kv_cache_size = (_DK + _DV) * MAX_L; + } + uint32_t v_offset = kv_cache_size / 2; + uint32_t L_offset = L_local * _DK; + + npu_it_channel receiving_channel = is_swa_layer(layer_type)? it_channel_1 : it_channel_0; + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + kv_cache_arg_id, + S2MM, + attn_tile, + bd_0, + receiving_channel, + {0, 0, 0, (uint32_t)L_offset}, + {1, 1, 1, (uint32_t)_DK}, + {0, 0, 0, 1}, + -1, 0, true + ); + + seq->npu_dma_wait( + attn_tile, + S2MM, + receiving_channel + ); + seq->npu_dma_memcpy_nd( + sizeof(bf16), + kv_cache_arg_id, + S2MM, + attn_tile, + bd_1, + receiving_channel, + {0, 0, 0, (uint32_t)(L_offset + v_offset)}, + {1, 1, 1, (uint32_t)_DV}, + {0, 0, 0, 1}, + -1, 0, true + ); + seq->npu_dma_wait( + attn_tile, + S2MM, + receiving_channel + ); +} + +void gemma4e_npu_sequence::_move_kv_cache(npu_sequence* seq, const size_t L, gemma4e_layer_type_t layer_type){ + LAYER_SPECIFIC_DIM(layer_type) + DEBUG_BLOCK(2, + header_print_g("info", "Moving KV cache for layer type " + std::to_string(layer_type) + " with L = " + std::to_string(L)); + std::cout << "\t_ L: " << L << std::endl; + std::cout << "\t_ _DK: " << _DK << std::endl; + std::cout << "\t_ _DV: " << _DV << std::endl; + std::cout << "\t_ MAX_L: " << MAX_L << std::endl; + ) + uint32_t kv_cache_size; + if (is_swa_layer(layer_type)){ + kv_cache_size = SLIDING_LENGTH * (_DK + _DV); + } + else { + kv_cache_size = (_DK + _DV) * MAX_L; + } + uint32_t v_offset = kv_cache_size / 2; + int pkt_id = is_swa_layer(layer_type) ? 13 : 12; + + if (is_swa_layer(layer_type)) { + if (L % SLIDING_LENGTH == 0){ // corner, from 0 to end + // move the oldest block to the new block + uint32_t data2move = SLIDING_LENGTH * _DK; + seq->npu_dma_memcpy_nd( + sizeof(bf16), + kv_cache_arg_id, + MM2S, + attn_tile, + bd_8, + it_channel_0, + {0, 0, 0, 0}, + {1, 1, 1, data2move}, + {0, 0, 0, 1}, + pkt_id, 0, false, aggressive_cache + ); + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + kv_cache_arg_id, + MM2S, + attn_tile, + bd_9, + it_channel_1, + {0, 0, 0, (uint32_t)(v_offset)}, + {1, 1, 1, data2move}, + {0, 0, 0, 1}, + pkt_id, 0, false, aggressive_cache + ); + return ; + } + else if (L > SLIDING_LENGTH) { // dual phase + uint32_t L_begin = L % SLIDING_LENGTH; + uint32_t L_phase_1 = SLIDING_LENGTH - L_begin; + uint32_t L_phase_2 = L_begin; + uint32_t offset = L_begin * _DK; + uint32_t data2move_1 = L_phase_1 * _DK; + uint32_t data2move_2 = L_phase_2 * _DK; + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + kv_cache_arg_id, + MM2S, + attn_tile, + bd_8, + it_channel_0, + {0, 0, 0, offset}, + {1, 1, 1, data2move_1}, + {0, 0, 0, 1}, + pkt_id, 0, false, aggressive_cache + ); + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + kv_cache_arg_id, + MM2S, + attn_tile, + bd_9, + it_channel_1, + {0, 0, 0, (uint32_t)(v_offset + offset)}, + {1, 1, 1, data2move_1}, + {0, 0, 0, 1}, + pkt_id, 0, false, aggressive_cache + ); + // phase 2 + seq->npu_dma_memcpy_nd( + sizeof(bf16), + kv_cache_arg_id, + MM2S, + attn_tile, + bd_10, + it_channel_0, + {0, 0, 0, 0}, + {1, 1, 1, data2move_2}, + {0, 0, 0, 1}, + pkt_id, 0, false, aggressive_cache + ); + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + kv_cache_arg_id, + MM2S, + attn_tile, + bd_11, + it_channel_1, + {0, 0, 0, (uint32_t)(v_offset)}, + {1, 1, 1, data2move_2}, + {0, 0, 0, 1}, + pkt_id, 0, false, aggressive_cache + ); + return; + } + } + DEBUG_BLOCK(2, + std::cout << "Single phase move for layer type (Fall back path) " << layer_type << std::endl; + ) + // fallback path, single phase, from zero to L + const int L_padded = (L + L_CHUNK - 1) / L_CHUNK * L_CHUNK; + const uint32_t data2move = L_padded * _DK; + seq->npu_dma_memcpy_nd( + sizeof(bf16), + kv_cache_arg_id, + MM2S, + attn_tile, + bd_8, + it_channel_0, + {0, 0, 0, 0}, + {1, 1, 1, data2move}, + {0, 0, 0, 1}, + pkt_id, 0, true, aggressive_cache + ); + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + kv_cache_arg_id, + MM2S, + attn_tile, + bd_9, + it_channel_1, + {0, 0, 0, (uint32_t)(v_offset)}, + {1, 1, 1, data2move}, + {0, 0, 0, 1}, + pkt_id, 0, true, aggressive_cache + ); + + seq->npu_dma_wait( + attn_tile, + MM2S, + it_channel_0 + ); + seq->npu_dma_wait( + attn_tile, + MM2S, + it_channel_1 + ); +} + +/// \brief Stream one quantized weight from DDR into the mvm cores. +/// \param seq the sequence to append the DMAs to +/// \param weight the weight to move; its shape is {input dim, output dim} and its +/// offset is a byte offset into the layer's projection buffer +/// \note The proj port addresses DDR in bf16 elements, so both the descriptor's +/// byte offset and the block size are halved here. Nothing about this +/// function is tied to a particular 4-bit layout any more: switch +/// gemma4e_desc::PROJ_DTYPE and the block size follows. +void gemma4e_npu_sequence::_move_weights(npu_sequence* seq, weight_desc_t& weight){ + assert(weight.added && "weight must be placed in its buffer before it can be moved"); + assert(is_quantize(weight.dtype) && "_move_weights moves quantized projections only"); + const int m = QXNX_ROW_BLOCK_SIZE; + const int k = QXNX_COL_BLOCK_SIZE; + const uint32_t columns = 4; + const size_t Din = (size_t)weight.shape[0]; + const size_t Dout = (size_t)weight.shape[1]; + const uint32_t a_block_size = (uint32_t)(get_quantization_byte_size((size_t)m * k, weight.dtype) / sizeof(bf16)); + const uint32_t w_offset = weight_elem_offset(weight); + const uint32_t blocks_per_row = Din / k; + const uint32_t cores = columns * 4; + assert(Dout / m / cores > 0); + assert(Dout % (m * cores) == 0); + for (size_t round = 0; round < Dout / m / cores; round++){ + uint32_t bd_offset = (round % 2) * 8; + for (uint32_t col = 0; col < columns; col++){ + seq->npu_dma_memcpy_nd( + sizeof(bf16), + proj_arg_id, + MM2S, + mvm_tiles[col], + npu_bd_id(bd_1 + bd_offset), + it_channel_0, + {0, 0, 0, (uint32_t)((round * cores + col * 4) * a_block_size * blocks_per_row + (uint32_t)w_offset)}, + {1, 1, 1, 2 * blocks_per_row * a_block_size}, + {0, 0, 0, 1}, + -1, 0, true, aggressive_cache + ); + seq->npu_dma_memcpy_nd( + sizeof(bf16), + proj_arg_id, + MM2S, + mvm_tiles[col], + npu_bd_id(bd_2 + bd_offset), + it_channel_1, + {0, 0, 0, (uint32_t)((round * cores + col * 4 + 2) * a_block_size * blocks_per_row + (uint32_t)w_offset)}, + {1, 1, 1, 2 * blocks_per_row * a_block_size}, + {0, 0, 0, 1}, + -1, 0, true, aggressive_cache + ); + } + if (round > 0){ + for (uint32_t col = 0; col < columns; col++){ + seq->npu_dma_wait( + mvm_tiles[col], + MM2S, + it_channel_0 + ); + seq->npu_dma_wait( + mvm_tiles[col], + MM2S, + it_channel_1 + ); + } + } + } + for (uint32_t col = 0; col < columns; col++){ + seq->npu_dma_wait( + mvm_tiles[col], + MM2S, + it_channel_0 + ); + seq->npu_dma_wait( + mvm_tiles[col], + MM2S, + it_channel_1 + ); + } +} + +void gemma4e_npu_sequence::_gen_pli_path_seq(npu_sequence* seq, gemma4e_layer_type_t layer_type){ + LAYER_SPECIFIC_DIM(layer_type) + DEBUG_BLOCK(2, + header_print_g("info", "Generating PLI path sequence for layer type " + std::to_string(layer_type)); + std::cout << "\t_ D: " << D << std::endl; + std::cout << "\t_ DH: " << _DH << std::endl; + std::cout << "\t_ PLI_D: " << PLI_D << std::endl; + std::cout << "\t_ pli_down_proj_offset: " << weight_elem_offset(layer_weights(layer_type).pli_down_proj) << std::endl; + std::cout << "\t_ pli_gate_proj_offset: " << weight_elem_offset(layer_weights(layer_type).pli_gate_proj) << std::endl; + std::cout << "\t_ pli_up_proj_offset: " << weight_elem_offset(layer_weights(layer_type).pli_up_proj) << std::endl; + ) + // move per layer input weights + // send x, x, w + seq->npu_dma_memcpy_nd( + sizeof(bf16), + rope_rms_arg_id, + MM2S, + IT4, + bd_11, + it_channel_0, + {0, 0, 0, (uint32_t)(_DH * 3)}, + {1, 1, 1, (uint32_t)(PLI_D * 2 + D + MIN_BF16_PAD)}, + {0, 0, 0, 1}, + -1, 0, true, no_cache + ); + seq->npu_dma_wait( + IT4, + MM2S, + it_channel_0 + ); + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + x_arg_id, + MM2S, + IT4, + bd_12, + it_channel_0, + {0, 0, 0, (uint32_t)D * 2}, + {1, 1, 1, (uint32_t)D}, + {0, 0, 0, 1}, + -1, 0, false, aggressive_cache + ); + + seq->npu_dma_memcpy_nd( + sizeof(bf16), + proj_arg_id, + MM2S, + IT4, + bd_13, + it_channel_1, + {0, 0, 0, weight_elem_offset(layer_weights(layer_type).pli_down_proj)}, + {1, 1, 1, (uint32_t)(PLI_D * D)}, + {0, 0, 0, 1}, + -1, 0, true, aggressive_cache + ); + + seq->npu_dma_wait( + IT4, + MM2S, + it_channel_1 + ); +} + +/// \brief Build the sequence that dequantizes one weight into a bf16 buffer. +/// \param seq_ptr the sequence to fill +/// \param weight the weight to dequantize; shape is {input dim, output dim} and +/// offset is a byte offset into the layer's projection buffer +/// \param output_mode which half of an interleaved up/gate region to emit, or +/// NORMAL_DEQUANT for a weight that is not interleaved +void gemma4e_npu_sequence::generate_dequant_seq(npu_sequence* seq_ptr, weight_desc_t& weight, dequant_output_mode_t output_mode){ + assert(weight.added && "weight must be placed in its buffer before it can be dequantized"); + assert(is_quantize(weight.dtype) && "only quantized weights need dequantizing"); + const u32 D_in = (u32)weight.shape[0]; + const u32 D_out = (u32)weight.shape[1]; + const u32 weight_offset = (u32)weight.offset; + static constexpr npu_tiles IT[] = {IT0, IT1, IT2, IT3, IT4, IT5, IT6, IT7}; + + static constexpr u32 total_cols = 8; + static constexpr u32 total_rows = 4; + + static constexpr int w_out_arg_idx = 0; + static constexpr int qw_in_arg_idx = 1; + + static constexpr int m_tile_q4 = 32; + static constexpr int k_tile_q4 = 256; + + const uint32_t block_size_in_byte_q4 = (uint32_t)get_quantization_byte_size((size_t)m_tile_q4 * k_tile_q4, weight.dtype); + + static constexpr int m_tile_q8 = 32; + static constexpr int k_tile_q8 = 128; + static constexpr uint32_t block_size_in_byte_q8 = m_tile_q8 * k_tile_q8 * (10)/8; + + static constexpr int quant_block_col_stride = 2; + + static constexpr int desired_k_dequant = 512; + static constexpr int desired_m_dequant = 128; + + static constexpr int glu_slice = 1024; + static constexpr int gate_up_m_interleave_size = glu_slice / 2; + if (D_in % k_tile_q4 != 0) { + std::cerr << "D_in % k_tile_q4 != 0" << std::endl; + exit(1); + } + int bd_wait_counter[8] = {0, 0, 0, 0, 0, 0, 0, 0}; + // although each data block is in mxk block, but the data block could be reorder in col-stride on block view + /* + For example, quant_block_col_stride = 2 means + + //This is the logical view of the data block, each block of m_tile_q4 x k_tile_q4 + [block0, block1, ...... blockD, + blockD+1, blockD+2, ...... + ] + + But in memory order, the data block is arrange as block0, blockD+1, block1, blockD+2 .... + + */ + + if(D_in % desired_k_dequant != 0){ + std::cerr << "D_in % desired_k_dequant != 0" << std::endl; + exit(1); + } + + const uint32_t blocks_per_row = D_in / k_tile_q4; + + if(D_out % desired_m_dequant != 0 ){ + std::cerr << "D_out % desired_m_dequant != 0" << std::endl; + exit(1); + } + + const int quant_in_per_column = (desired_m_dequant / m_tile_q4) * blocks_per_row * block_size_in_byte_q4; + const int total_column_rounds = D_out / (desired_m_dequant); + + const int row_per_round = desired_m_dequant * total_cols; + // down rounds, go though D_out + const int down_rounds = (D_out + row_per_round - 1) / row_per_round; + + npu_sequence& seq = *seq_ptr; + seq.clear_cmds(); + + for(int row = 0; row < 4; row++){ + for (int col = 0; col < 8; col++){ + npu_tiles tile = get_tile(row + 2, col); + seq_ptr->rtp_write(tile, dequant_rtp_address, 0); + } + } + uint32_t input_offset = weight_offset; + + if(output_mode == dequant_output_mode_t::GATE_MATRIX){ + input_offset += (gate_up_m_interleave_size / m_tile_q4) * blocks_per_row * block_size_in_byte_q4; + } + uint32_t gate_up_interleave_counter= 0; + + // first, the dequant of down + for(int i = 0; i < down_rounds; i++){ + for(int col = 0; col < 8; col++){ + uint32_t bd_offset = (i % 2) * 8; + uint32_t round_offset = i * 8 + col; + if(round_offset < total_column_rounds){ + seq.npu_dma_memcpy_nd( + sizeof(char), + qw_in_arg_idx, + MM2S, + IT[col], + (npu_bd_id)(0+bd_offset), + it_channel_0, + {0, 0, 0, input_offset}, + //NOTE: this for now only work if desired_m_dequant == quant_block_col_stride*m_tile_q4 + { + blocks_per_row, + (desired_m_dequant / m_tile_q4) / quant_block_col_stride, + quant_block_col_stride * block_size_in_byte_q4 / 512, + 512 + }, + { + quant_block_col_stride * block_size_in_byte_q4, + quant_block_col_stride * block_size_in_byte_q4 * blocks_per_row, + 512, + 1 + }, + -1 ,0, false + ); + + if(output_mode == dequant_output_mode_t::NORMAL_DEQUANT){ + input_offset += quant_in_per_column; + } + else{ + gate_up_interleave_counter++; + input_offset += quant_in_per_column; + if(gate_up_interleave_counter == (gate_up_m_interleave_size / desired_m_dequant) ){ + gate_up_interleave_counter = 0; + input_offset += (gate_up_m_interleave_size / m_tile_q4) * blocks_per_row * block_size_in_byte_q4; + } + } + + // Each port receive 2*Q4NX_ROWx D_Q4NX_BLOCK_PER_ROW*Q4NX_COL + uint32_t output_offset_0 = round_offset * desired_m_dequant * D_in; + + seq.npu_dma_memcpy_nd( + sizeof(uint16_t),//bf16 outpout + w_out_arg_idx, + S2MM, + IT[col], + (npu_bd_id)(1+bd_offset), + it_channel_0, + {0, 0, 0, output_offset_0}, + { + (uint32_t)D_in/desired_k_dequant, + desired_k_dequant/k_tile_q4, + desired_m_dequant, + k_tile_q4 + }, + { + desired_m_dequant * desired_k_dequant, + k_tile_q4, + desired_k_dequant, + 1 + }, + -1, 0, true, + aggressive_cache + ); + bd_wait_counter[col]++; + } + } + // note: for now + for(int col = 0; col < 8; col++){ + if(bd_wait_counter[col] == 2){ + seq.npu_dma_wait(IT[col], S2MM, it_channel_0); + bd_wait_counter[col]--; + } + } + } + + for(int col = 0; col < 8; col++){ + while(bd_wait_counter[col] != 0){ + seq.npu_dma_wait(IT[col], S2MM, it_channel_0); + bd_wait_counter[col]--; + } + } + seq.cmds2seq(); +} + +void gemma4e_npu_sequence::gen_mha_engine_seq( + npu_sequence* seq, + const uint32_t L_begin, + const uint32_t L_end +){ + const int lc = 8; // local chunk size, each CU works on lc*8 of l at a time. This is determined by the hardware design. + + npu_tiles IT[2][4] = {{IT0, IT1, IT2, IT3}, {IT4, IT5, IT6, IT7}}; + assert(L_begin % (lc * 16) == 0); + assert(L_end % (lc * 16) == 0); + int using_window_size = L_end; + const int Heads = num_attn_heads; + const int GQA = num_attn_heads / num_kv_heads; + const int Num_of_d_in_KV_cache = num_kv_heads; + const int DH = this->DH; + const uint32_t KV_CACHE_SIZE = (DK + DV) * MAX_L; + seq->clear_cmds(); + + const int l_begin_mha_address = 59520; + const int l_end_mha_address = 11520; + const int window_size_address = 59552; + for (int row = 2; row < 6; row++){ + for (int col = 0; col < 8; col++){ + npu_tiles tile = get_tile(row, col); + seq->rtp_write(tile, l_begin_mha_address, L_begin); + seq->rtp_write(tile, l_end_mha_address, L_end); + seq->rtp_write(tile, window_size_address, using_window_size); + } + } + int all_data_size = L_end - L_begin; + const int data_per_round = lc * 16; + const int down_rounds = (all_data_size + data_per_round - 1) / data_per_round; + int num_cu = 2; + for (int head = 0; head < Heads / num_cu; head++){ + for (int round = 0; round < down_rounds; round++){ + int Lq_current = L_begin + round * data_per_round; + int kv_begin = ((Lq_current - using_window_size) > 0) ? (Lq_current - using_window_size) : 0; + int kv_length = Lq_current - kv_begin + data_per_round; + + int bd_offset = (round % 2) * 8; + for (int cu = 0; cu < num_cu; cu++){ + int head_offset = head * num_cu + cu; + for (int col = 0; col < 4; col++){ + // receive y + size_t y_offset = head_offset * DH + (round * data_per_round + col * lc * 4) * DH * Heads; + seq->npu_dma_memcpy_nd( + 2, 0, + S2MM, IT[cu][col], + (npu_bd_id)(bd_offset + 0), it_channel_0, + {0, 0, 0, (uint32_t)y_offset}, + {1, 1, (uint32_t)4 * lc, (uint32_t)(DH)}, + {0, 0, (uint32_t)DH * Heads, 1}, + -1, 0, true + ); + } + // send q + size_t q_offset = head_offset * DH + round * data_per_round * DH * Heads; + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[cu][0], + (npu_bd_id)(bd_offset + 1), it_channel_0, + {0, 0, 0, (uint32_t)q_offset}, + {1, 1, (uint32_t)lc * 4, (uint32_t)(DH)}, + {0, 0, (uint32_t)DH * Heads, 1}, + -1, 0, false + ); + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[cu][0], + (npu_bd_id)(bd_offset + 2), it_channel_1, + {0, 0, 0, (uint32_t)(q_offset + lc * 4 * DH * Heads)}, + {1, 1, (uint32_t)lc * 4, (uint32_t)(DH)}, + {0, 0, (uint32_t)DH * Heads, 1}, + -1, 0, false + ); + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[cu][3], + (npu_bd_id)(bd_offset + 3), it_channel_0, + {0, 0, 0, (uint32_t)(q_offset + lc * 8 * DH * Heads)}, + {1, 1, (uint32_t)lc * 4, (uint32_t)(DH)}, + {0, 0, (uint32_t)DH * Heads, 1}, + -1, 0, false + ); + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[cu][3], + (npu_bd_id)(bd_offset + 4), it_channel_1, + {0, 0, 0, (uint32_t)(q_offset + lc * 12 * DH * Heads)}, + {1, 1, (uint32_t)lc * 4, (uint32_t)(DH)}, + {0, 0, (uint32_t)DH * Heads, 1}, + -1, 0, false + ); + } // cu + int kv_head_offset = head / (GQA / num_cu); + int kv_chunk_offset = kv_head_offset / Num_of_d_in_KV_cache; + int kv_head_offset_in_chunk = kv_head_offset % Num_of_d_in_KV_cache; + size_t k_offset; + size_t v_offset; + if (kv_length <= 128 * 1024){ + k_offset = kv_chunk_offset * MAX_L * DH * Num_of_d_in_KV_cache + kv_head_offset_in_chunk * DH + kv_begin * DH * Num_of_d_in_KV_cache; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[0][2], + (npu_bd_id)(bd_offset + 5), it_channel_0, + {0, 0, 0, (uint32_t)k_offset}, + {1, (uint32_t)kv_length / 128, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + v_offset = k_offset + KV_CACHE_SIZE / 2; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[0][2], + (npu_bd_id)(bd_offset + 6), it_channel_1, + {0, 0, 0, (uint32_t)v_offset}, + {1, (uint32_t)kv_length / 128, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + } + else{ + k_offset = kv_chunk_offset * MAX_L * DH * Num_of_d_in_KV_cache + kv_head_offset_in_chunk * DH + kv_begin * DH * Num_of_d_in_KV_cache; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[0][2], + (npu_bd_id)(bd_offset + 5), it_channel_0, + {0, 0, 0, (uint32_t)k_offset}, + {1, (uint32_t)1024, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + v_offset = k_offset + KV_CACHE_SIZE / 2; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[0][2], + (npu_bd_id)(bd_offset + 6), it_channel_1, + {0, 0, 0, (uint32_t)v_offset}, + {1, (uint32_t)1024, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + int remaining_kv_length = kv_length - 1024 * 128; + k_offset = kv_chunk_offset * MAX_L * DH * Num_of_d_in_KV_cache + kv_head_offset_in_chunk * DH + kv_begin * DH * Num_of_d_in_KV_cache + 1024 * 128 * DH * Num_of_d_in_KV_cache; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[0][2], + (npu_bd_id)(bd_offset + 5), it_channel_0, + {0, 0, 0, (uint32_t)k_offset}, + {1, (uint32_t)remaining_kv_length / 128, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + v_offset = k_offset + KV_CACHE_SIZE / 2 + 1024 * 128 * DH * Num_of_d_in_KV_cache; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[0][2], + (npu_bd_id)(bd_offset + 6), it_channel_1, + {0, 0, 0, (uint32_t)v_offset}, + {1, (uint32_t)remaining_kv_length / 128, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + } + if (round > 0){ + for (int cu = 0; cu < num_cu; cu++){ + for (int col = 0; col < 4; col++){ + seq->npu_dma_wait( + IT[cu][col], + S2MM, + it_channel_0 + ); + } // col + } // cu + }// round > 0 + }// round + for (int cu = 0; cu < num_cu; cu++){ + for (int col = 0; col < 4; col++){ + seq->npu_dma_wait( + IT[cu][col], + S2MM, + it_channel_0 + ); + } // col + } // cu + } // head + seq->cmds2seq(); +} + +void gemma4e_npu_sequence::gen_swa_engine_seq( + npu_sequence* seq, + const uint32_t L_begin, + const uint32_t L_end +){ + const int lc = 16; // local chunk size, each CU works on lc*8 of l at a time. This is determined by the hardware design. + npu_tiles IT[4][2] = {{IT0, IT1}, {IT2, IT3}, {IT4, IT5}, {IT6, IT7}}; + assert(L_begin % (lc * 8) == 0); + assert(L_end % (lc * 8) == 0); + int using_window_size = SLIDING_LENGTH; + const int Heads = num_attn_heads; + const int GQA = num_attn_heads / num_kv_heads; + const int Num_of_d_in_KV_cache = num_kv_heads; + const int DH = this->SWA_DH; + const uint32_t KV_CACHE_SIZE = (SWA_DK + SWA_DV) * MAX_L; + seq->clear_cmds(); + + const int l_begin_mha_address = 61568; + const int l_end_mha_address = 11904; + const int window_size_address = 61600; + for (int row = 2; row < 6; row++){ + for (int col = 0; col < 8; col++){ + npu_tiles tile = get_tile(row, col); + seq->rtp_write(tile, l_begin_mha_address, L_begin); + seq->rtp_write(tile, l_end_mha_address, L_end); + seq->rtp_write(tile, window_size_address, using_window_size); + } + } + // each CU has 8 CTs and works on 1 head. + int all_data_size = L_end - L_begin; + // each round, each CU works on 1 head and lc * 8 of l, total_cols / 2 is corresponding to the number of CUs + const int data_per_round = lc * 8; + // zero padding the last round if necessary + const int down_rounds = (all_data_size + data_per_round - 1) / data_per_round; + + int num_cu = 4; + for (int head = 0; head < Heads / num_cu; head++){ + for (int round = 0; round < down_rounds; round++){ + int Lq_current = L_begin + round * data_per_round; + int kv_begin = ((Lq_current - using_window_size) > 0) ? (Lq_current - using_window_size) : 0; + int kv_length = Lq_current - kv_begin + data_per_round; + + int bd_offset = (round % 2) * 8; + + for (int cu = 0; cu < num_cu; cu++){ + int head_offset = head * num_cu + cu; + + for (int col = 0; col < 2; col++){ + // receive y + size_t y_offset = head_offset * DH + (round * data_per_round + col * lc * 4) * DH * Heads; + seq->npu_dma_memcpy_nd( + 2, 0, + S2MM, IT[cu][col], + (npu_bd_id)(bd_offset + 0), it_channel_0, + {0, 0, 0, (uint32_t)y_offset}, + {1, 1, (uint32_t)4 * lc, (uint32_t)(DH)}, + {0, 0, (uint32_t)DH * Heads, 1}, + -1, 0, true + ); + } + // send q + size_t q_offset = head_offset * DH + round * data_per_round * DH * Heads; + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[cu][0], + (npu_bd_id)(bd_offset + 1), it_channel_0, + {0, 0, 0, (uint32_t)q_offset}, + {1, 1, (uint32_t)lc * 4, (uint32_t)(DH)}, + {0, 0, (uint32_t)DH * Heads, 1}, + -1, 0, false + ); + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[cu][0], + (npu_bd_id)(bd_offset + 2), it_channel_1, + {0, 0, 0, (uint32_t)(q_offset + lc * 4 * DH * Heads)}, + {1, 1, (uint32_t)lc * 4, (uint32_t)(DH)}, + {0, 0, (uint32_t)DH * Heads, 1}, + -1, 0, false + ); + } // cu + + int kv_head_offset = head / (GQA / num_cu); + int kv_chunk_offset = kv_head_offset / Num_of_d_in_KV_cache; + int kv_head_offset_in_chunk = kv_head_offset % Num_of_d_in_KV_cache; + + size_t k_offset; + size_t v_offset; + if (kv_length <= 128 * 1024){ + k_offset = kv_chunk_offset * MAX_L * DH * Num_of_d_in_KV_cache + kv_head_offset_in_chunk * DH + kv_begin * DH * Num_of_d_in_KV_cache; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[1][1], + (npu_bd_id)(bd_offset + 3), it_channel_0, + {0, 0, 0, (uint32_t)k_offset}, + {1, (uint32_t)kv_length / 128, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + v_offset = kv_chunk_offset * MAX_L * DH * Num_of_d_in_KV_cache + kv_head_offset_in_chunk * DH + kv_begin * DH * Num_of_d_in_KV_cache + KV_CACHE_SIZE / 2; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[1][1], + (npu_bd_id)(bd_offset + 4), it_channel_1, + {0, 0, 0, (uint32_t)v_offset}, + {1, (uint32_t)kv_length / 128, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + } else{ + k_offset = kv_chunk_offset * MAX_L * DH * Num_of_d_in_KV_cache + kv_head_offset_in_chunk * DH + kv_begin * DH * Num_of_d_in_KV_cache; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[1][1], + (npu_bd_id)(bd_offset + 3), it_channel_0, + {0, 0, 0, (uint32_t)k_offset}, + {1, (uint32_t)1024, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + v_offset = kv_chunk_offset * MAX_L * DH * Num_of_d_in_KV_cache + kv_head_offset_in_chunk * DH + kv_begin * DH * Num_of_d_in_KV_cache + KV_CACHE_SIZE / 2; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[1][1], + (npu_bd_id)(bd_offset + 4), it_channel_1, + {0, 0, 0, (uint32_t)v_offset}, + {1, (uint32_t)1024, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + int remaining_kv_length = kv_length - 1024 * 128; + k_offset = kv_chunk_offset * MAX_L * DH * Num_of_d_in_KV_cache + kv_head_offset_in_chunk * DH + kv_begin * DH * Num_of_d_in_KV_cache + 1024 * 128 * DH * Num_of_d_in_KV_cache; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[1][1], + (npu_bd_id)(bd_offset + 5), it_channel_0, + {0, 0, 0, (uint32_t)k_offset}, + {1, (uint32_t)remaining_kv_length / 128, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + v_offset = kv_chunk_offset * MAX_L * DH * Num_of_d_in_KV_cache + kv_head_offset_in_chunk * DH + kv_begin * DH * Num_of_d_in_KV_cache + KV_CACHE_SIZE / 2 + 1024 * 128 * DH * Num_of_d_in_KV_cache; + seq->npu_dma_memcpy_nd( + 2, 2, + MM2S, IT[1][1], + (npu_bd_id)(bd_offset + 6), it_channel_1, + {0, 0, 0, (uint32_t)v_offset}, + {1, (uint32_t)remaining_kv_length / 128, (uint32_t)128, (uint32_t)(DH)}, + {0, (uint32_t)128 * DH * Num_of_d_in_KV_cache, (uint32_t)DH * Num_of_d_in_KV_cache, 1}, + -1, 0, false + ); + } + if (round > 0){ + for (int cu = 0; cu < num_cu; cu++){ + for (int col = 0; col < 2; col++){ + seq->npu_dma_wait( + IT[cu][col], + S2MM, + it_channel_0 + ); + } // col + } // cu + }// round > 0 + }// round + for (int cu = 0; cu < num_cu; cu++){ + for (int col = 0; col < 2; col++){ + seq->npu_dma_wait( + IT[cu][col], + S2MM, + it_channel_0 + ); + } // col + } // cu + } // head + seq->cmds2seq(); +} + +gemma4e_npu_sequence::~gemma4e_npu_sequence() = default; diff --git a/src/detail/gemma4e_npu/gemma4e_npu_sequence.hpp b/src/detail/gemma4e_npu/gemma4e_npu_sequence.hpp new file mode 100644 index 000000000..a09606da0 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_npu_sequence.hpp @@ -0,0 +1,216 @@ +#pragma once + +#include "npu_utils/npu_instr_utils.hpp" +#include "lm_config.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/gemma4e/gemma4e_npu.hpp" +#include "weight_desc.hpp" +#include "gemma4e_npu_def.hpp" + +#define MIN_BF16_PAD 32 // minimum padding in number of bf16 elements to avoid NPU OOM when processing long sequences, which is determined empirically + +#define LAYER_SPECIFIC_DIM(type) \ + int _DH = is_global_layer(type) ? DH : SWA_DH; \ + int _DQ = is_global_layer(type) ? DQ : SWA_DQ; \ + int _DK = is_global_layer(type) ? DK : SWA_DK; \ + int _DV = is_global_layer(type) ? DV : SWA_DV; \ + int _INTERMEDIATE_SIZE = (is_skip_layer(type) && enable_double_wide_mlp) ? INTERMEDIATE_SIZE * 2 : INTERMEDIATE_SIZE; + +typedef struct { + int D; + int DH; + int DQ; + int DK; + int DV; + int SWA_DH; + int SWA_DQ; + int SWA_DK; + int SWA_DV; + int PLI_D; + int INTERMEDIATE_SIZE; + int NUM_ATTENTION_HEADS; + int NUM_KEY_VALUE_HEADS; + int SLIDING_WINDOW_SIZE; + int VOCAB_SIZE_PADDED; + bool enable_double_wide_mlp; +} gemma4e_seq_gen_parameters_t; + +struct gemma4e_npu_sequence{ + typedef enum: int{ + NORMAL_DEQUANT = 0, + UP_MATRIX = 1, + GATE_MATRIX = 2 + } dequant_output_mode_t; + + typedef struct { + // for decoding layer + uint32_t l_qk_address; + uint32_t l_kv_address; + uint32_t swa_l_qk_address; + uint32_t swa_l_kv_address; + uint32_t proj_swa_address; + uint32_t proj_skip_address; + uint32_t rms_swa_address; + uint32_t rms_skip_address; + uint32_t rope_skip_kv_address; + uint32_t swa_rope_skip_kv_address; + uint32_t glu_skip_address; + uint32_t lm_head_final_tune_address; + // for others + uint32_t place_holder_0; + } rtp_address_book_t; + + static constexpr rtp_address_book_t e2b_rtp_addresses = { + .l_qk_address = 57344, + .l_kv_address = 14976, + .swa_l_qk_address = 9216, + .swa_l_kv_address = 40960, + .proj_swa_address = 33280, + .proj_skip_address = 49664, + .rms_swa_address = 52224, + .rms_skip_address = 25600, + .rope_skip_kv_address = 33792, + .swa_rope_skip_kv_address = 33280, + .glu_skip_address = 34816, + .lm_head_final_tune_address = 49152, + .place_holder_0 = 0x20050625 + }; + + static constexpr rtp_address_book_t e4b_rtp_addresses = { + .l_qk_address = 57344, + .l_kv_address = 14976, + .swa_l_qk_address = 53248, + .swa_l_kv_address = 57664, + .proj_swa_address = 33280, + .proj_skip_address = 49664, + .rms_swa_address = 55328, + .rms_skip_address = 55392, + .rope_skip_kv_address = 34816, + .swa_rope_skip_kv_address = 33792, + .glu_skip_address = 30720, + .lm_head_final_tune_address = 10240, + .place_holder_0 = 0x20050625 + }; + /// @brief constexprs + static constexpr npu_tiles xr_tile = IT3; + static constexpr npu_tiles out_tile = IT4; + static constexpr npu_tiles attn_tile = IT2; + static constexpr npu_tiles mvm_tiles[4] = {IT0, IT1, IT6, IT7}; + static constexpr npu_tiles rms_tile = CT03; + static constexpr npu_tiles glu_tile = CT13; + static constexpr npu_tiles rope_ct = CT23; + static constexpr npu_tiles swa_rope_ct = CT33; + static constexpr npu_tiles attn_qk_tile = CT02; + static constexpr npu_tiles attn_kv_tile = CT12; + static constexpr npu_tiles swa_attn_qk_tile = CT22; + static constexpr npu_tiles swa_attn_kv_tile = CT32; + static constexpr npu_tiles proj_tiles[] = {CT00, CT10, CT20, CT30, + CT01, CT11, CT21, CT31, + CT06, CT16, CT26, CT36, + CT07, CT17, CT27, CT37 + }; + static constexpr npu_tiles pli_tile = CT05; + + static constexpr int x_arg_id = 0; + static constexpr int proj_arg_id = 1; + static constexpr int rms_arg_id = 2; + static constexpr int rope_rms_arg_id = 3; + static constexpr int kv_cache_arg_id = 4; + + static constexpr int L_CHUNK = 16; + + /// @brief dequant kernel RTP offset, identical for E2B and E4B. + static constexpr uint32_t dequant_rtp_address = 54272; + + /// \brief The model description; owns every weight descriptor this generator addresses. + /// \note Not owned. Set by set_desc() before any sequence is generated. + gemma4e_desc* desc = nullptr; + rtp_address_book_t rtp_addresses; + + uint32_t D; + uint32_t DH; + uint32_t DQ; + uint32_t DK; + uint32_t DV; + uint32_t SWA_DH; + uint32_t SWA_DQ; + uint32_t SWA_DK; + uint32_t SWA_DV; + + uint32_t MAX_L; + uint32_t HIDDEN_SIZE; + uint32_t INTERMEDIATE_SIZE; + uint32_t PLI_D; + uint32_t SLIDING_LENGTH; + + bool enable_double_wide_mlp; + + int num_attn_heads; + int num_kv_heads; + int num_kv_per_round; + + uint32_t rms_addr; + + int VOCAB_SIZE; + + gemma4e_npu_sequence(){} + gemma4e_npu_sequence(gemma4e_seq_gen_parameters_t params, uint32_t MAX_L); + ~gemma4e_npu_sequence(); + /// \brief Human-readable name of a layer kind, for debug output. + static const char* layer_type_name(gemma4e_layer_type_t t) { + switch (t) { + case e_gemma4e_swa_layer: return "swa"; + case e_gemma4e_global_layer: return "global"; + case e_gemma4e_swa_layer_skip: return "swa_skip"; + case e_gemma4e_global_layer_skip: return "global_skip"; + default: return "unknown"; + } + } + + /// \brief Point the generator at the model description. + /// \param desc the description whose weight descriptors name every DMA source + void set_desc(gemma4e_desc* desc) { + this->desc = desc; + DEBUG_BLOCK(2, + header_print("info", "Sequence generator bound to gemma4e_desc; proj layout (bytes):"); + for (int t = 0; t < 4; t++){ + gemma4e_layer_weight_def& L = desc->weight_desc(static_cast(t)); + std::cout << " " << layer_type_name(static_cast(t)) + << ": qkv=" << L.attn_qkv.offset + << ", o=" << L.attn_output.offset + << ", upgate=" << L.ffn_up_gate.offset + << ", down=" << L.ffn_down.offset + << ", pli_down=" << L.pli_down_proj.offset + << ", pli_gate=" << L.pli_gate_proj.offset + << ", pli_up=" << L.pli_up_proj.offset << std::endl; + } + ) + } + + /// \brief The weight layout of a layer kind. + gemma4e_layer_weight_def& layer_weights(gemma4e_layer_type_t layer_type) { + assert(desc != nullptr && "set_desc() must run before any sequence is generated"); + return desc->weight_desc(layer_type); + } + + /// \brief A weight's start, in bf16 elements, which is how the DMA ports address it. + static uint32_t weight_elem_offset(weight_desc_t& weight) { + return (uint32_t)(weight.offset / sizeof(bf16)); + } + + void _send_hidden_states(npu_sequence* seq); + void _send_rms_weights(npu_sequence* seq); + void _send_rope_rms_weights(npu_sequence* seq, gemma4e_layer_type_t layer_type); + void _receive_kv_cache(npu_sequence* seq, const int L, gemma4e_layer_type_t layer_type); + void _move_kv_cache(npu_sequence* seq, const size_t L, gemma4e_layer_type_t layer_type); + void _move_weights(npu_sequence* seq, weight_desc_t& weight); + void _gen_pli_path_seq(npu_sequence* seq_ptr, gemma4e_layer_type_t layer_type); + void gen_lm_head_seq(npu_sequence* seq, float final_scale); + + void gen_layer_seq(npu_sequence* seq, const uint32_t L, gemma4e_layer_type_t layer_type); + void gen_mha_engine_seq(npu_sequence* seq, const uint32_t L_begin, const uint32_t L_end); + void gen_swa_engine_seq(npu_sequence* seq, const uint32_t L_begin, const uint32_t L_end); + void generate_dequant_seq(npu_sequence* seq_ptr, weight_desc_t& weight, dequant_output_mode_t output_mode); + + void set_max_length(const uint32_t MAX_L); +}; diff --git a/src/detail/gemma4e_npu/gemma4e_prefill.cpp b/src/detail/gemma4e_npu/gemma4e_prefill.cpp new file mode 100644 index 000000000..ed83b77c9 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_prefill.cpp @@ -0,0 +1,584 @@ +#include "flm_override.hpp" +#include "gemma4e_prefill.hpp" +#include "metrices.hpp" + +// --------------------------------------------------------------------------- +// attention block +// --------------------------------------------------------------------------- + +void gemma4e_attn_block_prefill_context::_forward_swa( + const gemma4e_prefill_shape& s, + gemma4e_layer_type_t type, + buffer& qkv_weights, + buffer& o_weights, + buffer& kv_cache, + buffer& q_norm, + buffer& k_norm, + SafeTensors* reference +){ + const bool is_skip = is_skip_layer(type); + const int D = desc->D; + const int SWA_DQ = desc->SWA_DQ; + const int SWA_DK = desc->SWA_DK; + const int SWA_DV = desc->SWA_DV; + + FLM_OVERRIDE(q_swa_proj, q_swa_proj(bufs->q_buffer, bufs->hidden_state_buffer, qkv_weights), this->desc, type, s); + bufs->q_buffer.sync_from_device(); + + DEBUG_BLOCK(2, + buffer q_ref; + reference->load_weights(q_ref, "q_proj"); + buffer q_valid = buffer(bufs->q_buffer.data(), s.L_effective * SWA_DQ); + buffer q_valid_ref = buffer(q_ref.data(), s.L_effective * SWA_DQ); + print_error_metrics(get_error_metrics(q_valid, q_valid_ref), "SWA Q Projection Error: "); + ) + if (is_skip){ + // a skip layer carries no k/v projection: it reuses the cache the last + // non-skip layer of its kind filled. + gemma4e_cpu_func::_rope_rms_batch(bufs->q_buffer.data(), SWA_DQ, s.L_offset, s.L_begin, s.L_effective, q_norm.data(), type, desc->get_DH(type)); + bufs->q_buffer.sync_to_device(); + } + else { + auto run_k = FLM_OVERRIDE(k_swa_proj, + k_swa_proj.create_run(bufs->k_buffer, bufs->hidden_state_buffer, qkv_weights), this->desc, type, s); + run_k.start(); + + gemma4e_cpu_func::_rope_rms_batch(bufs->q_buffer.data(), SWA_DQ, s.L_offset, s.L_begin, s.L_effective, q_norm.data(), type, desc->get_DH(type)); + bufs->q_buffer.sync_to_device(); + + run_k.wait(); + bufs->k_buffer.sync_from_device(); + auto run_v = FLM_OVERRIDE(v_swa_proj, + v_swa_proj.create_run(bufs->v_buffer, bufs->hidden_state_buffer, qkv_weights), this->desc, type, s); + run_v.start(); + gemma4e_cpu_func::_rope_rms_batch(bufs->k_buffer.data(), SWA_DK, s.L_offset, s.L_begin, s.L_effective, k_norm.data(), type, desc->get_DH(type)); + bufs->k_buffer.sync_to_device(); + run_v.wait(); + + bufs->v_buffer.sync_from_device(); + gemma4e_cpu_func::_rms_norm_batch(bufs->v_buffer.data(), bufs->v_buffer.data(), nullptr, SWA_DV, SWA_DV, s.L_effective, s.L_offset, s.L_offset); + bufs->v_buffer.sync_to_device(); + } + + DEBUG_BLOCK(2, + buffer q_ref; + reference->load_weights(q_ref, "q_embed"); + buffer q_valid = buffer(bufs->q_buffer.data(), s.L_effective * SWA_DQ); + buffer q_valid_ref = buffer(q_ref.data(), s.L_effective * SWA_DQ); + print_error_metrics(get_error_metrics(q_valid, q_valid_ref), "SWA Q Projection Error: "); + if (!is_skip){ + buffer k_ref; + reference->load_weights(k_ref, "k_embed"); + buffer k_valid = buffer(bufs->k_buffer.data(), s.L_effective * SWA_DK); + buffer k_valid_ref = buffer(k_ref.data(), s.L_effective * SWA_DK); + print_error_metrics(get_error_metrics(k_valid, k_valid_ref), "SWA K Projection Error: "); + + buffer v_ref; + reference->load_weights(v_ref, "v_norm"); + buffer v_valid = buffer(bufs->v_buffer.data(), s.L_effective * SWA_DV); + buffer v_valid_ref = buffer(v_ref.data(), s.L_effective * SWA_DV); + print_error_metrics(get_error_metrics(v_valid, v_valid_ref), "SWA V Projection Error: "); + } + ) + if (!is_skip){ + sync_sliding_kv_cache(bufs->k_buffer, bufs->v_buffer, kv_cache_sliding_prefill, kv_cache, s.L_offset, s.L_begin, s.L_effective, s.sliding_l_begin, s.L_end_chunked, SWA_DK, SWA_DV); + } + else{ + kv_cache_sliding_prefill.sync_from_device(); + kv_cache_sliding_prefill.sync_to_device(); + } + FLM_OVERRIDE(swa_attn_core, + this->swa_engine(bufs->attn_out_buffer, bufs->q_buffer, kv_cache_sliding_prefill), type, s, this->MAX_L); + bufs->attn_out_buffer.sync_from_device(); + DEBUG_BLOCK(2, + buffer attn_out_ref; + reference->load_weights(attn_out_ref, "attention_output"); + buffer attn_out_valid = buffer(bufs->attn_out_buffer.data(), s.L_effective * SWA_DQ); + buffer attn_out_valid_ref = buffer(attn_out_ref.data(), s.L_effective * SWA_DQ); + print_error_metrics(get_error_metrics(attn_out_valid, attn_out_valid_ref), "SWA Attention Output Error: "); + ) + + FLM_OVERRIDE(o_swa_proj, + this->o_swa_proj(bufs->hidden_state_buffer, bufs->attn_out_buffer, o_weights), this->desc, type, s); + bufs->hidden_state_buffer.sync_from_device(); + + DEBUG_BLOCK(2, + buffer o_proj_ref; + reference->load_weights(o_proj_ref, "after_o_proj"); + buffer o_proj_valid = buffer(bufs->hidden_state_buffer.data(), s.L_effective * D); + buffer o_proj_valid_ref = buffer(o_proj_ref.data(), s.L_effective * D); + print_error_metrics(get_error_metrics(o_proj_valid, o_proj_valid_ref), "SWA O Projection Error: "); + ) +} + +void gemma4e_attn_block_prefill_context::_forward_global( + const gemma4e_prefill_shape& s, + gemma4e_layer_type_t type, + buffer& qkv_weights, + buffer& o_weights, + buffer& kv_cache, + buffer& q_norm, + buffer& k_norm, + SafeTensors* reference +){ + const bool is_skip = is_skip_layer(type); + const int D = desc->D; + const int DQ = desc->DQ; + const int DK = desc->DK; + const int DV = desc->DV; + + FLM_OVERRIDE(q_global_proj, q_global_proj(bufs->q_buffer, bufs->hidden_state_buffer, qkv_weights), this->desc, type, s); + bufs->q_buffer.sync_from_device(); + + DEBUG_BLOCK(2, + buffer q_ref; + reference->load_weights(q_ref, "q_proj"); + buffer q_valid = buffer(bufs->q_buffer.data(), s.L_effective * DQ); + buffer q_valid_ref = buffer(q_ref.data(), s.L_effective * DQ); + print_error_metrics(get_error_metrics(q_valid, q_valid_ref), "Q Projection Error: "); + ) + if (is_skip){ + gemma4e_cpu_func::_rope_rms_batch(bufs->q_buffer.data(), DQ, s.L_offset, s.L_begin, s.L_effective, q_norm.data(), type, desc->get_DH(type)); + bufs->q_buffer.sync_to_device(); + } + else { + auto run_k = FLM_OVERRIDE(k_global_proj, + k_global_proj.create_run(bufs->k_buffer, bufs->hidden_state_buffer, qkv_weights), this->desc, type, s); + run_k.start(); + gemma4e_cpu_func::_rope_rms_batch(bufs->q_buffer.data(), DQ, s.L_offset, s.L_begin, s.L_effective, q_norm.data(), type, desc->get_DH(type)); + bufs->q_buffer.sync_to_device(); + run_k.wait(); + bufs->k_buffer.sync_from_device(); + + auto run_v = FLM_OVERRIDE(v_global_proj, + v_global_proj.create_run(bufs->v_buffer, bufs->hidden_state_buffer, qkv_weights), this->desc, type, s); + run_v.start(); + gemma4e_cpu_func::_rope_rms_batch(bufs->k_buffer.data(), DK, s.L_offset, s.L_begin, s.L_effective, k_norm.data(), type, desc->get_DH(type)); + bufs->k_buffer.sync_to_device(); + run_v.wait(); + + bufs->v_buffer.sync_from_device(); + gemma4e_cpu_func::_rms_norm_batch(bufs->v_buffer.data(), bufs->v_buffer.data(), nullptr, DV, DV, s.L_effective, s.L_offset, s.L_offset); + bufs->v_buffer.sync_to_device(); + } + + DEBUG_BLOCK(2, + buffer q_ref; + reference->load_weights(q_ref, "q_embed"); + buffer q_valid = buffer(bufs->q_buffer.data(), s.L_effective * DQ); + buffer q_valid_ref = buffer(q_ref.data(), s.L_effective * DQ); + print_error_metrics(get_error_metrics(q_valid, q_valid_ref), "Q Projection Error: "); + + buffer k_ref; + reference->load_weights(k_ref, "k_embed"); + buffer k_valid = buffer(bufs->k_buffer.data(), s.L_effective * DK); + buffer k_valid_ref = buffer(k_ref.data(), s.L_effective * DK); + print_error_metrics(get_error_metrics(k_valid, k_valid_ref), "K Projection Error: "); + + buffer v_ref; + reference->load_weights(v_ref, "v_norm"); + buffer v_valid = buffer(bufs->v_buffer.data(), s.L_effective * DV); + buffer v_valid_ref = buffer(v_ref.data(), s.L_effective * DV); + print_error_metrics(get_error_metrics(v_valid, v_valid_ref), "V Projection Error: "); + ) + if (!is_skip){ + sync_kv_cache(bufs->k_buffer, bufs->v_buffer, kv_cache_global_prefill, kv_cache, s.L_offset, s.L_begin, s.L_effective, DK, DV); + } + else { + kv_cache_global_prefill.sync_from_device(); + kv_cache_global_prefill.sync_to_device(); + } + FLM_OVERRIDE(global_attn_core, + this->mha_engine(bufs->attn_out_buffer, bufs->q_buffer, kv_cache_global_prefill), type, s, this->MAX_L); + DEBUG_BLOCK(2, + header_print("info", "Attn done!"); + ) + bufs->attn_out_buffer.sync_from_device(); + DEBUG_BLOCK(2, + buffer attn_out_ref; + reference->load_weights(attn_out_ref, "attention_output"); + buffer attn_out_valid = buffer(bufs->attn_out_buffer.data(), s.L_effective * DQ); + buffer attn_out_valid_ref = buffer(attn_out_ref.data(), s.L_effective * DQ); + print_error_metrics(get_error_metrics(attn_out_valid, attn_out_valid_ref), "Attention Output Error: "); + ) + + FLM_OVERRIDE(o_global_proj, + this->o_global_proj(bufs->hidden_state_buffer, bufs->attn_out_buffer, o_weights), this->desc, type, s); + bufs->hidden_state_buffer.sync_from_device(); + + DEBUG_BLOCK(2, + buffer o_proj_ref; + reference->load_weights(o_proj_ref, "after_o_proj"); + buffer o_proj_valid = buffer(bufs->hidden_state_buffer.data(), s.L_effective * D); + buffer o_proj_valid_ref = buffer(o_proj_ref.data(), s.L_effective * D); + print_error_metrics(get_error_metrics(o_proj_valid, o_proj_valid_ref), "O Projection Error: "); + ) +} + +void gemma4e_attn_block_prefill_context::sync_kv_cache( + buffer& buffer_k, + buffer& buffer_v, + buffer& prefill_cache, + buffer& decoding_cache, + int L_offset, + int L_begin, + int L_effective, + int DK, int DV +){ + bf16* prefill_k_cache_ptr = prefill_cache.data(); + bf16* prefill_v_cache_ptr = prefill_cache.data() + (size_t)MAX_L * DK; + bf16* decoding_k_cache_ptr = decoding_cache.data(); + bf16* decoding_v_cache_ptr = decoding_cache.data() + (size_t)MAX_L * DK; + // copy the existing cache up to L_begin from decoding_cache to prefill_cache, and then copy the new k cache from buffer_k and v cache from buffer_v to prefill_cache at the appropriate location, then sync the prefill_cache to device for the attention run + decoding_cache.sync_from_device(); + prefill_cache.sync_from_device(); + memcpy(prefill_k_cache_ptr, decoding_k_cache_ptr, L_begin * DK * sizeof(bf16)); // copy the existing cache up to L_begin + memcpy(prefill_v_cache_ptr, decoding_v_cache_ptr, L_begin * DV * sizeof(bf16)); // copy the existing cache up to L_begin + + prefill_k_cache_ptr += L_begin * DK; + prefill_v_cache_ptr += L_begin * DV; + decoding_k_cache_ptr += L_begin * DK; + decoding_v_cache_ptr += L_begin * DV; + + // copy the new k cache and v cache to the appropriate location in prefill_cache and decoding_cache + bf16* new_k_cache_ptr = buffer_k.data() + L_offset * DK; + bf16* new_v_cache_ptr = buffer_v.data() + L_offset * DV; + + memcpy(prefill_k_cache_ptr, new_k_cache_ptr, L_effective * DK * sizeof(bf16)); + memcpy(prefill_v_cache_ptr, new_v_cache_ptr, L_effective * DV * sizeof(bf16)); + memcpy(decoding_k_cache_ptr, new_k_cache_ptr, L_effective * DK * sizeof(bf16)); + memcpy(decoding_v_cache_ptr, new_v_cache_ptr, L_effective * DV * sizeof(bf16)); + + decoding_cache.sync_to_device(); + prefill_cache.sync_to_device(); +} + +void gemma4e_attn_block_prefill_context::sync_sliding_kv_cache( + buffer& buffer_k, + buffer& buffer_v, + buffer& prefill_cache, + buffer& decoding_cache, + int L_offset, + int L_begin, + int L_effective, + int sliding_l_begin, + int L_end_chunked, + int DK, int DV +){ + const int SLIDING_LENGTH = desc->SLIDING_LENGTH; + auto linear2ring = [sliding = SLIDING_LENGTH](int l) { return l % sliding; }; + int in_memory = (L_begin - SLIDING_LENGTH) > 0 ? L_begin - SLIDING_LENGTH : 0; + int l_idx = in_memory % SLIDING_LENGTH; + int local_offset = in_memory - sliding_l_begin; + int local_length = L_end_chunked - sliding_l_begin; + DEBUG_BLOCK(2, + std::cout << "DEBUG: L_begin: " << L_begin << ", L_end_chunked: " << L_end_chunked << ", sliding_l_begin: " << sliding_l_begin << std::endl; + std::cout << "DEBUG: in_memory: " << in_memory << std::endl; + std::cout << "DEBUG: local_offset: " << local_offset << ", local_length: " << local_length << std::endl; + ) + // | sliding_l_begin -> in_memory | in_memory -> L_begin | L_begin -> L_end | L_end -> L_end_chunked | + // | Part 1 | Part 2 | Part 3 | Part 4 | + + int part_1_length = in_memory - sliding_l_begin; + int part_2_length = L_begin - in_memory; + int part_3_length = L_effective; + int part_4_length = L_end_chunked - L_begin - L_effective; + + bf16* prefill_k_cache_ptr = prefill_cache.data() + sliding_l_begin * DK; + bf16* prefill_v_cache_ptr = prefill_cache.data() + sliding_l_begin * DV + (size_t)MAX_L * DK; + bf16* decoding_k_cache_ptr = decoding_cache.data(); + bf16* decoding_v_cache_ptr = decoding_cache.data() + SLIDING_LENGTH * DK; + + // copy the existing cache up to L_begin from decoding_cache to prefill_cache, and then copy the new k cache from buffer_k and v cache from buffer_v to prefill_cache at the appropriate location, then sync the prefill_cache to device for the attention run + decoding_cache.sync_from_device(); + prefill_cache.sync_from_device(); + + // part 1: padded, no actual data exists + if (part_1_length > 0){ + memset(prefill_k_cache_ptr, 0, part_1_length * DK * sizeof(bf16)); + memset(prefill_v_cache_ptr, 0, part_1_length * DV * sizeof(bf16)); + prefill_k_cache_ptr += part_1_length * DK; + prefill_v_cache_ptr += part_1_length * DV; + } + + // part 2: copy from the existing cache in decoding_cache, but we need to copy in the order of the ring buffer + for (int i = 0; i < part_2_length; i++){ + int ring_idx = linear2ring(in_memory + i); + memcpy(prefill_k_cache_ptr, decoding_k_cache_ptr + ring_idx * DK, DK * sizeof(bf16)); + memcpy(prefill_v_cache_ptr, decoding_v_cache_ptr + ring_idx * DV, DV * sizeof(bf16)); + prefill_k_cache_ptr += DK; + prefill_v_cache_ptr += DV; + } + + // copy the new k cache and v cache to the appropriate location in prefill_cache and decoding_cache + bf16* new_k_cache_ptr = buffer_k.data() + L_offset * DK; + bf16* new_v_cache_ptr = buffer_v.data() + L_offset * DV; + //part 3: copy the new k cache and v cache from buffer_k and buffer_v to prefill_cache and decoding_cache, but we also need to copy in the order of the ring buffer + for (int i = 0; i < part_3_length; i++){ + int ring_idx = linear2ring(L_begin + i); + memcpy(prefill_k_cache_ptr, new_k_cache_ptr + i * DK, DK * sizeof(bf16)); + memcpy(prefill_v_cache_ptr, new_v_cache_ptr + i * DV, DV * sizeof(bf16)); + memcpy(decoding_k_cache_ptr + ring_idx * DK, new_k_cache_ptr + i * DK, DK * sizeof(bf16)); + memcpy(decoding_v_cache_ptr + ring_idx * DV, new_v_cache_ptr + i * DV, DV * sizeof(bf16)); + prefill_k_cache_ptr += DK; + prefill_v_cache_ptr += DV; + } + + // // part 4: padded, no actual data exists + // memset(prefill_k_cache_ptr, 0, part_4_length * DK * sizeof(bf16)); + // memset(prefill_v_cache_ptr, 0, part_4_length * DV * sizeof(bf16)); + decoding_cache.sync_to_device(); + prefill_cache.sync_to_device(); +} + +// --------------------------------------------------------------------------- +// mlp block +// --------------------------------------------------------------------------- + +void gemma4e_mlp_prefill_context::forward( + const gemma4e_prefill_shape& s, + bool double_wide, + buffer& gate_weights, + buffer& up_weights, + buffer& down_weights, + SafeTensors* reference +){ + // a double-wide layer runs the same three gemms over twice the mlp width + npu_app& gate = double_wide ? this->gate_skip_proj : this->gate_proj; + npu_app& up = double_wide ? this->up_skip_proj : this->up_proj; + npu_app& down = double_wide ? this->down_skip_proj : this->down_proj; + const int I = double_wide ? desc->INTERMEDIATE_SIZE * 2 : desc->INTERMEDIATE_SIZE; + + bufs->hidden_state_buffer.sync_to_device(); + FLM_OVERRIDE(gate_proj, gate(bufs->gate_buffer, bufs->hidden_state_buffer, gate_weights), this->desc, double_wide, s); + bufs->gate_buffer.sync_from_device(); + FLM_OVERRIDE(up_proj, up(bufs->up_buffer, bufs->hidden_state_buffer, up_weights), this->desc, double_wide, s); + bufs->up_buffer.sync_from_device(); + + gemma4e_cpu_func::_elementwise_mul_batch(bufs->hid_buffer.data(), bufs->gate_buffer.data(), bufs->up_buffer.data(), I, s.L_effective, s.L_offset, s.L_offset, s.L_offset); + + bufs->hid_buffer.sync_to_device(); + + FLM_OVERRIDE(down_proj, down(bufs->hidden_state_buffer, bufs->hid_buffer, down_weights), this->desc, double_wide, s); + bufs->hidden_state_buffer.sync_from_device(); + + DEBUG_BLOCK(2, + buffer down_ref; + reference->load_weights(down_ref, "mlp_output"); + buffer down_valid = buffer(bufs->hidden_state_buffer.data(), s.L_effective * desc->D); + buffer down_valid_ref = buffer(down_ref.data(), s.L_effective * desc->D); + print_error_metrics(get_error_metrics(down_valid, down_valid_ref), "MLP Error: "); + ) +} + +// --------------------------------------------------------------------------- +// per layer input path +// --------------------------------------------------------------------------- + +void gemma4e_pli_prefill_context::setup(const gemma4e_prefill_shape& s){ + if (L_padded_512_old == s.L_padded_512) { + return; + } + const int D = desc->D; + const int PLI_D = desc->PLI_D; + const int num_hidden_layers = desc->num_hidden_layers; + Gemma4e_ImageEncoder* enc = this->image_encoder; + + generate_mm_sequence(*this->pli_down_proj.seq(), + s.L_padded_512, D, num_hidden_layers * PLI_D, + enc->MM_tile_M, enc->MM_tile_K, enc->MM_tile_N, + 8,8,8, + enc->rtp_address, enc->rtp_sync_lock_id, + enc->MM_ROW_SIZE, enc->MM_COL_SIZE, + 0,0,0, + enc->IS_B_ROW_MAJOR, enc->ENABLE_AXI4, true, + false, 0,// no activation + 0, -10000.0, 1000000.0, // do not clamp on output + false, 0 // no need to reorder it + ); + + generate_mm_sequence(*this->pli_gate_proj.seq(), + s.L_padded_512, D, PLI_D, + enc->MM_tile_M, enc->MM_tile_K, enc->MM_tile_N, + 8,8,8, + enc->rtp_address, enc->rtp_sync_lock_id, + enc->MM_ROW_SIZE, enc->MM_COL_SIZE, + 0,0,0, + enc->IS_B_ROW_MAJOR, enc->ENABLE_AXI4, true, + false, 1,// no activation + 0, -10000.0, 1000000.0, // do not clamp on output + false, 0 // no need to reorder it + ); + + generate_mm_sequence(*this->pli_up_proj.seq(), + s.L_padded_512, PLI_D, D, + enc->MM_tile_M, enc->MM_tile_K, enc->MM_tile_N, + 8,8,8, + enc->rtp_address, enc->rtp_sync_lock_id, + enc->MM_ROW_SIZE, enc->MM_COL_SIZE, + 0,PLI_D * D,0, + enc->IS_B_ROW_MAJOR, enc->ENABLE_AXI4, true, + false, 0,// no activation + 0, -10000.0, 1000000.0, // do not clamp on output + false, 0 // no need to reorder it + ); + L_padded_512_old = s.L_padded_512; +} + +void gemma4e_pli_prefill_context::pre_pass( + const gemma4e_prefill_shape& s, + buffer& pli_down_weights, + buffer& pli_input_norm +){ + const int D = desc->D; + const int PLI_D = desc->PLI_D; + const int num_hidden_layers = desc->num_hidden_layers; + + // the down projection reads the token embeddings, which start life in the residual + memcpy(bufs->hidden_state_buffer.data() + (size_t)s.L_offset * D, bufs->residual_buffer.data() + (size_t)s.L_offset * D, (size_t)s.L_effective * D * sizeof(bf16)); + bufs->hidden_state_buffer.sync_to_device(); + this->pli_down_proj(bufs->hidden_state_buffer, pli_down_weights, bufs->pli_down_buffer); + bufs->pli_down_buffer.sync_from_device(); + + gemma4e_cpu_func::_elementwise_scale_batch(bufs->pli_down_buffer.data(), 1.0 / sqrtf((float)D), PLI_D * num_hidden_layers, s.L_effective, s.L_offset); + gemma4e_cpu_func::_rms_norm_batch(bufs->pli_down_buffer.data(), bufs->pli_down_buffer.data(), pli_input_norm.data(), PLI_D, num_hidden_layers * PLI_D, s.L_effective, s.L_offset, s.L_offset); + gemma4e_cpu_func::_residual_add_batch(bufs->pli_embed_buffer.data(), bufs->pli_down_buffer.data(), bufs->pli_embed_buffer.data(), num_hidden_layers * PLI_D, s.L_effective, s.L_offset, s.L_offset, s.L_offset); + + gemma4e_cpu_func::_elementwise_scale_batch(bufs->pli_embed_buffer.data(), 1.0 / sqrtf((float)2.0), PLI_D * num_hidden_layers, s.L_effective, s.L_offset); +} + +void gemma4e_pli_prefill_context::layer_pass( + const gemma4e_prefill_shape& s, + int layer_idx, + buffer& pli_gate_up_weights, + buffer& pli_final_norm +){ + const int D = desc->D; + const int PLI_D = desc->PLI_D; + const int num_hidden_layers = desc->num_hidden_layers; + + this->pli_gate_proj(bufs->hidden_state_buffer, pli_gate_up_weights, bufs->pli_gate_buffer); + bufs->pli_gate_buffer.sync_from_device(); + + // this layer's slice of the per-layer embeddings gates the projection + gemma4e_cpu_func::_elementwise_mul_batch(bufs->pli_hid_buffer.data(), bufs->pli_gate_buffer.data(), bufs->pli_embed_buffer.data() + (size_t)layer_idx * PLI_D, PLI_D, num_hidden_layers * PLI_D, s.L_effective, s.L_offset, s.L_offset, s.L_offset); + + bufs->pli_hid_buffer.sync_to_device(); + this->pli_up_proj(bufs->pli_hid_buffer, pli_gate_up_weights, bufs->hidden_state_buffer); + bufs->hidden_state_buffer.sync_from_device(); + + gemma4e_cpu_func::_rms_norm_batch(bufs->hidden_state_buffer.data(), bufs->hidden_state_buffer.data(), pli_final_norm.data(), D, D, s.L_effective, s.L_offset, s.L_offset); + bufs->hidden_state_buffer.sync_to_device(); +} + +// --------------------------------------------------------------------------- +// one decoder layer +// --------------------------------------------------------------------------- + +void gemma4e_prefill_context::forward( + int layer_idx, + gemma4e_layer_type_t type, + const gemma4e_prefill_shape& s, + buffer& proj_weights, + buffer& rms_weights, + buffer& rope_rms_weights, + buffer& pli_gate_up_weights, + buffer& kv_cache, + float layer_scale, + SafeTensors* reference +){ + const int D = desc->D; + const bool is_skip = is_skip_layer(type); + + FLM_OVERRIDE(prefill_layer_begin, (void)0, this->desc, layer_idx, type, s); + + this->dequant_block->run(type, proj_weights); + + DEBUG_BLOCK(1, + header_print_r("info", "Running layer " + std::to_string(layer_idx) + " of type " + std::to_string(type)); + ) + + // every norm weight of this layer, located through the descriptor + buffer input_layernorm_weight(rms_weights.data(), D); + buffer post_attention_layernorm_weight(rms_weights.data() + D, D); + buffer pre_feedforward_layernorm_weight(rms_weights.data() + D * 2, D); + buffer post_feedforward_layernorm_weight(rms_weights.data() + D * 3, D); + buffer pli_final_norm(rope_rms_weights.data() + desc->get_post_pli_norm_offset(type), D); + buffer q_norm(rope_rms_weights.data() + desc->get_q_norm_offset(type), desc->get_DH(type)); + buffer k_norm(rope_rms_weights.data() + desc->get_k_norm_offset(type), desc->get_DH(type)); + + // input layernorm + gemma4e_cpu_func::_rms_norm_batch(bufs.hidden_state_buffer.data(), bufs.residual_buffer.data(), input_layernorm_weight.data(), D, D, s.L_effective, s.L_offset, s.L_offset); + bufs.hidden_state_buffer.sync_to_device(); + + DEBUG_BLOCK(2, + buffer norm_ref; + reference->load_weights(norm_ref, "input_layernorm_output"); + buffer norm_valid = buffer(bufs.hidden_state_buffer.data(), s.L_effective * D); + buffer norm_valid_ref = buffer(norm_ref.data(), s.L_effective * D); + print_error_metrics(get_error_metrics(norm_valid, norm_valid_ref), "Input LayerNorm Error: "); + ) + + this->attn_block->forward(s, type, + this->dequant_block->qkv_weights, this->dequant_block->o_weights, + kv_cache, q_norm, k_norm, reference); + + gemma4e_cpu_func::_rms_norm_batch(bufs.hidden_state_buffer.data(), bufs.hidden_state_buffer.data(), post_attention_layernorm_weight.data(), D, D, s.L_effective, s.L_offset, s.L_offset); + bufs.attn_out_buffer.sync_to_device(); + + DEBUG_BLOCK(2, + buffer post_attn_norm_ref; + reference->load_weights(post_attn_norm_ref, "post_attention_norm_output"); + buffer post_attn_norm_valid = buffer(bufs.hidden_state_buffer.data(), s.L_effective * D); + buffer post_attn_norm_valid_ref = buffer(post_attn_norm_ref.data(), s.L_effective * D); + print_error_metrics(get_error_metrics(post_attn_norm_valid, post_attn_norm_valid_ref), "Post Attention LayerNorm Error: "); + ) + + gemma4e_cpu_func::_residual_add_batch(bufs.residual_buffer.data(), bufs.hidden_state_buffer.data(), bufs.residual_buffer.data(), D, s.L_effective, s.L_offset, s.L_offset, s.L_offset); + + gemma4e_cpu_func::_rms_norm_batch(bufs.hidden_state_buffer.data(), bufs.residual_buffer.data(), pre_feedforward_layernorm_weight.data(), D, D, s.L_effective, s.L_offset, s.L_offset); + bufs.hidden_state_buffer.sync_to_device(); + + DEBUG_BLOCK(2, + buffer pre_ffn_norm_ref; + reference->load_weights(pre_ffn_norm_ref, "pre_ffn_norm_output"); + buffer pre_ffn_norm_valid = buffer(bufs.hidden_state_buffer.data(), s.L_effective * D); + buffer pre_ffn_norm_valid_ref = buffer(pre_ffn_norm_ref.data(), s.L_effective * D); + print_error_metrics(get_error_metrics(pre_ffn_norm_valid, pre_ffn_norm_valid_ref), "Pre-FFN LayerNorm Error: "); + ) + + this->mlp->forward(s, is_skip && desc->enable_double_wide_mlp, + this->dequant_block->gate_weights, this->dequant_block->up_weights, + this->dequant_block->down_weights, reference); + + gemma4e_cpu_func::_rms_norm_batch(bufs.hidden_state_buffer.data(), bufs.hidden_state_buffer.data(), post_feedforward_layernorm_weight.data(), D, D, s.L_effective, s.L_offset, s.L_offset); + bufs.hidden_state_buffer.sync_to_device(); + + DEBUG_BLOCK(2, + buffer post_ffn_norm_ref; + reference->load_weights(post_ffn_norm_ref, "post_ffn_norm_output"); + buffer post_ffn_norm_valid = buffer(bufs.hidden_state_buffer.data(), s.L_effective * D); + buffer post_ffn_norm_valid_ref = buffer(post_ffn_norm_ref.data(), s.L_effective * D); + print_error_metrics(get_error_metrics(post_ffn_norm_valid, post_ffn_norm_valid_ref), "Post-FFN LayerNorm Error: "); + ) + + gemma4e_cpu_func::_residual_add_batch(bufs.residual_buffer.data(), bufs.hidden_state_buffer.data(), bufs.residual_buffer.data(), D, s.L_effective, s.L_offset, s.L_offset, s.L_offset); + + // per layer input: the gate reads the layer output, which lives in the residual + memcpy(bufs.hidden_state_buffer.data() + (size_t)s.L_offset * D, bufs.residual_buffer.data() + (size_t)s.L_offset * D, (size_t)s.L_effective * D * sizeof(bf16)); + bufs.hidden_state_buffer.sync_to_device(); + + this->pli->layer_pass(s, layer_idx, pli_gate_up_weights, pli_final_norm); + + gemma4e_cpu_func::_residual_add_batch(bufs.residual_buffer.data(), bufs.hidden_state_buffer.data(), bufs.residual_buffer.data(), D, s.L_effective, s.L_offset, s.L_offset, s.L_offset); + + gemma4e_cpu_func::_elementwise_scale_batch(bufs.residual_buffer.data(), layer_scale, D, s.L_effective, s.L_offset); + + DEBUG_BLOCK(2, + buffer output_ref; + reference->load_weights(output_ref, "layer_" + std::to_string(layer_idx)); + buffer output_valid = buffer(bufs.residual_buffer.data(), s.L_effective * D); + buffer output_valid_ref = buffer(output_ref.data(), s.L_effective * D); + print_error_metrics(get_error_metrics(output_valid, output_valid_ref), "Main path Error: "); + ) +} diff --git a/src/detail/gemma4e_npu/gemma4e_prefill.hpp b/src/detail/gemma4e_npu/gemma4e_prefill.hpp new file mode 100644 index 000000000..9f42f196a --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_prefill.hpp @@ -0,0 +1,521 @@ +#include "flm_override.hpp" +#ifndef __GEMMA4E_PREFILL_HPP__ +#define __GEMMA4E_PREFILL_HPP__ +#include +#include "modules/gemm.hpp" +#include "tensor_2d.hpp" +#include "gemma4e_npu_def.hpp" +#include "gemma4e_npu_sequence.hpp" +#include "gemma4e_cpu_functions.hpp" +#include "gemma4e_image.hpp" +#include "mmRuntimeSequence.hpp" +#include "utils/error_measure.hpp" + +/// @brief Row geometry of one prefill call. +/// +/// The attention kernel consumes whole chunks, so the batch is padded in front +/// (L_offset rows that belong to already-cached tokens) and at the back (up to +/// L_padded rows), and every gemm runs over the full L_padded rows. The per +/// layer input path runs over a coarser 512-row grid, hence L_padded_512. +struct gemma4e_prefill_shape { + int L_in = 0; ///< tokens handed to this call + int L_begin = 0; ///< context length before this call + int L_end = 0; ///< context length after this call + int L_effective = 0; ///< rows that carry a real token, == L_in + int L_offset = 0; ///< rows of chunk padding in front of the first new token + int L_begin_chunked = 0; ///< chunk-aligned start fed to the attention kernel + int L_end_chunked = 0; ///< chunk-aligned end fed to the attention kernel + int L_padded = 0; ///< rows every gemm runs over + int L_padded_512 = 0; ///< L_padded rounded up for the per-layer-input gemms + int sliding_l_begin = 0; ///< first position the sliding window still covers +}; + +/// @brief Buffers shared by every block of a prefill layer. +/// +/// Allocated once per batch geometry and reused across prefill calls; only the +/// residual and the per-layer-input embeddings need clearing, since every other +/// buffer is fully overwritten by the kernel that reads it. +struct gemma4e_common_buffers { + buffer residual_buffer; ///< host: the running residual stream + buffer hidden_state_buffer; ///< device: block input and block output + buffer q_buffer; + buffer k_buffer; + buffer v_buffer; + buffer attn_out_buffer; + buffer gate_buffer; + buffer up_buffer; + buffer hid_buffer; + + // per layer input path + buffer pli_embed_buffer; + buffer pli_down_buffer; + buffer pli_gate_buffer; + buffer pli_hid_buffer; + + tensor_2d tensor_residual; + tensor_2d tensor_pli_embed; + + gemma4e_common_buffers() {} + + /// @brief (Re)allocates every buffer for a batch geometry. + void allocate(npu_xclbin_manager* npu, gemma4e_desc* desc, const gemma4e_prefill_shape& s) { + const size_t D = desc->D; + const size_t I = desc->INTERMEDIATE_SIZE; + const size_t PLI = (size_t)desc->PLI_D * desc->num_hidden_layers; + residual_buffer = buffer((size_t)s.L_padded_512 * D); + hidden_state_buffer = npu->create_bo_buffer((size_t)s.L_padded_512 * D); + q_buffer = npu->create_bo_buffer((size_t)s.L_padded * desc->DQ); + k_buffer = npu->create_bo_buffer((size_t)s.L_padded * desc->DK); + v_buffer = npu->create_bo_buffer((size_t)s.L_padded * desc->DV); + attn_out_buffer = npu->create_bo_buffer((size_t)s.L_padded_512 * desc->DQ); + // the widest mlp a layer can ask for, so skip layers share these buffers + gate_buffer = npu->create_bo_buffer((size_t)s.L_padded * I * 2); + up_buffer = npu->create_bo_buffer((size_t)s.L_padded * I * 2); + hid_buffer = npu->create_bo_buffer((size_t)s.L_padded * I * 2); + pli_embed_buffer = npu->create_bo_buffer((size_t)s.L_padded_512 * PLI); + pli_down_buffer = npu->create_bo_buffer((size_t)s.L_padded_512 * PLI); + pli_gate_buffer = npu->create_bo_buffer((size_t)s.L_padded_512 * desc->PLI_D); + pli_hid_buffer = npu->create_bo_buffer((size_t)s.L_padded_512 * desc->PLI_D); + } + + /// @brief Clears the padding rows and re-points the row views at this batch. + void reset(gemma4e_desc* desc, const gemma4e_prefill_shape& s) { + // rows outside [L_offset, L_offset + L_in) are padding; they still take + // part in every gemm, so leave no stale values in them. + memset(residual_buffer.data(), 0, residual_buffer.size() * sizeof(bf16)); + memset(pli_embed_buffer.data(), 0, pli_embed_buffer.size() * sizeof(bf16)); + tensor_residual.assign(residual_buffer, desc->D, s.L_offset); + tensor_pli_embed.assign(pli_embed_buffer, desc->PLI_D * desc->num_hidden_layers, s.L_offset); + } +}; + +/// @brief Dequantizes one layer's projections into the bf16 buffers the prefill +/// gemms consume. +/// +/// One sequence per (layer kind, weight): which bytes each one reads is entirely +/// the descriptor's business, so changing the quantization type needs no edit +/// here. The five matrices of a layer are launched back to back. +struct gemma4e_dequant_prefill_context { + /// index into the per-layer-kind app table + enum matrix_t { QKV = 0, O = 1, GATE = 2, UP = 3, DOWN = 4, NUM_MATRICES = 5 }; + + gemma4e_desc* desc; + npu_app apps[4][NUM_MATRICES]; ///< [gemma4e_layer_type_t][matrix_t] + + buffer qkv_weights; + buffer o_weights; + buffer gate_weights; + buffer up_weights; + buffer down_weights; + + gemma4e_dequant_prefill_context( + gemma4e_desc* desc, + gemma4e_npu_sequence* seq_gen, + npu_app_manager* dequant_app_manager + ) : desc(desc) + { + const uint32_t D = desc->D; + const uint32_t I = desc->get_max_intermediate_size(); + this->qkv_weights = dequant_app_manager->create_bo_buffer((size_t)D * (desc->DQ + desc->DK + desc->DV)); + this->o_weights = dequant_app_manager->create_bo_buffer((size_t)desc->DQ * D); + this->gate_weights = dequant_app_manager->create_bo_buffer((size_t)D * I); + this->up_weights = dequant_app_manager->create_bo_buffer((size_t)D * I); + this->down_weights = dequant_app_manager->create_bo_buffer((size_t)I * D); + + for (int t = 0; t < 4; t++) { + gemma4e_layer_type_t type = static_cast(t); + gemma4e_layer_weight_def& W = desc->weight_desc(type); + for (int m = 0; m < NUM_MATRICES; m++) { + apps[t][m] = dequant_app_manager->create_app(); + } + // up and gate share one interleaved band and are told apart by the mode + seq_gen->generate_dequant_seq(apps[t][QKV].seq(), W.attn_qkv, gemma4e_npu_sequence::NORMAL_DEQUANT); + seq_gen->generate_dequant_seq(apps[t][O].seq(), W.attn_output, gemma4e_npu_sequence::NORMAL_DEQUANT); + seq_gen->generate_dequant_seq(apps[t][GATE].seq(), W.ffn_gate, gemma4e_npu_sequence::GATE_MATRIX); + seq_gen->generate_dequant_seq(apps[t][UP].seq(), W.ffn_up, gemma4e_npu_sequence::UP_MATRIX); + seq_gen->generate_dequant_seq(apps[t][DOWN].seq(), W.ffn_down, gemma4e_npu_sequence::NORMAL_DEQUANT); + } + } + + /// @brief Dequantizes every projection of one layer, in place. + void run(gemma4e_layer_type_t type, buffer& proj_weights) { + npu_app* a = apps[int(type)]; + FLM_OVERRIDE(dequant_qkv, a[QKV](this->qkv_weights, proj_weights), this->desc, type); + FLM_OVERRIDE(dequant_o, a[O](this->o_weights, proj_weights), this->desc, type); + FLM_OVERRIDE(dequant_up, a[UP](this->up_weights, proj_weights), this->desc, type); + FLM_OVERRIDE(dequant_gate, a[GATE](this->gate_weights, proj_weights), this->desc, type); + FLM_OVERRIDE(dequant_down, a[DOWN](this->down_weights, proj_weights), this->desc, type); + } +}; + +/// @brief Attention half of one prefill layer: q/k/v projections, rope, kv cache +/// fill, attention, output projection. +/// +/// Sliding and global layers run different kernels over differently shaped +/// caches, so both sets of apps live here and forward() dispatches on the layer +/// kind. Skip layers reuse the previous layer's cache and project q only. +struct gemma4e_attn_block_prefill_context { + gemma4e_desc* desc; + Gemm* gemm_seq_gen; + gemma4e_npu_sequence* seq_gen; + gemma4e_common_buffers* bufs; + + int L_padded_old = -1; + int L_begin_chunked_old = -1; + int L_end_chunked_old = -1; + uint32_t MAX_L = 0; + + npu_app q_swa_proj, k_swa_proj, v_swa_proj, o_swa_proj; + npu_app q_global_proj, k_global_proj, v_global_proj, o_global_proj; + npu_app mha_engine; + npu_app swa_engine; + + /// caches the attention kernels read: contiguous, and rebuilt every layer + buffer kv_cache_sliding_prefill; + buffer kv_cache_global_prefill; + + npu_app_manager* mha_app_manager; + npu_app_manager* swa_app_manager; + + gemma4e_attn_block_prefill_context( + gemma4e_desc* desc, + Gemm* gemm_seq_gen, + gemma4e_npu_sequence* seq_gen, + gemma4e_common_buffers* bufs, + npu_app_manager* gemm_app_manager, + npu_app_manager* mha_app_manager, + npu_app_manager* swa_app_manager + ) : desc(desc), gemm_seq_gen(gemm_seq_gen), seq_gen(seq_gen), bufs(bufs), + mha_app_manager(mha_app_manager), swa_app_manager(swa_app_manager) + { + this->q_global_proj = gemm_app_manager->create_app(); + this->k_global_proj = gemm_app_manager->create_app(); + this->v_global_proj = gemm_app_manager->create_app(); + this->o_global_proj = gemm_app_manager->create_app(); + this->q_swa_proj = gemm_app_manager->create_app(); + this->k_swa_proj = gemm_app_manager->create_app(); + this->v_swa_proj = gemm_app_manager->create_app(); + this->o_swa_proj = gemm_app_manager->create_app(); + this->mha_engine = mha_app_manager->create_app(); + this->swa_engine = swa_app_manager->create_app(); + } + + /// @brief Sizes the prefill-side caches, which span the whole context. + void set_max_length(uint32_t MAX_L) { + this->MAX_L = MAX_L; + this->kv_cache_sliding_prefill = swa_app_manager->create_bo_buffer((size_t)MAX_L * (desc->SWA_DK + desc->SWA_DV)); + this->kv_cache_global_prefill = mha_app_manager->create_bo_buffer((size_t)MAX_L * (desc->DK + desc->DV)); + this->kv_cache_sliding_prefill.memset((bf16)0); + this->kv_cache_global_prefill.memset((bf16)0); + this->kv_cache_sliding_prefill.sync_to_device(); + this->kv_cache_global_prefill.sync_to_device(); + // the cached sequences address these buffers, so force a regeneration + this->L_padded_old = -1; + this->L_begin_chunked_old = -1; + this->L_end_chunked_old = -1; + } + + /// @brief Regenerates the projection and attention sequences, if the geometry moved. + void setup(const gemma4e_prefill_shape& s) { + const uint32_t D = desc->D; + if (L_padded_old != s.L_padded) { + gemm_seq_gen->generate_seq(this->q_swa_proj.seq(), s.L_padded, D, desc->SWA_DQ, 0, + false, Gemm::NO_Activation, 0); + gemm_seq_gen->generate_seq(this->k_swa_proj.seq(), s.L_padded, D, desc->SWA_DK, D * desc->SWA_DQ, + false, Gemm::NO_Activation, 0); + gemm_seq_gen->generate_seq(this->v_swa_proj.seq(), s.L_padded, D, desc->SWA_DV, D * (desc->SWA_DQ + desc->SWA_DK), + false, Gemm::NO_Activation, 0); + gemm_seq_gen->generate_seq(this->o_swa_proj.seq(), s.L_padded, desc->SWA_DQ, D, 0, + false, Gemm::NO_Activation, 0); + + gemm_seq_gen->generate_seq(this->q_global_proj.seq(), s.L_padded, D, desc->DQ, 0, + false, Gemm::NO_Activation, 0); + gemm_seq_gen->generate_seq(this->k_global_proj.seq(), s.L_padded, D, desc->DK, D * desc->DQ, + false, Gemm::NO_Activation, 0); + gemm_seq_gen->generate_seq(this->v_global_proj.seq(), s.L_padded, D, desc->DV, D * (desc->DQ + desc->DK), + false, Gemm::NO_Activation, 0); + gemm_seq_gen->generate_seq(this->o_global_proj.seq(), s.L_padded, desc->DQ, D, 0, + false, Gemm::NO_Activation, 0); + L_padded_old = s.L_padded; + } + if (L_begin_chunked_old != s.L_begin_chunked || L_end_chunked_old != s.L_end_chunked) { + seq_gen->gen_mha_engine_seq(this->mha_engine.seq(), s.L_begin_chunked, s.L_end_chunked); + seq_gen->gen_swa_engine_seq(this->swa_engine.seq(), s.L_begin_chunked, s.L_end_chunked); + L_begin_chunked_old = s.L_begin_chunked; + L_end_chunked_old = s.L_end_chunked; + } + } + + /// @brief Runs the attention block of one layer, hidden_state_buffer in and out. + void forward( + const gemma4e_prefill_shape& s, + gemma4e_layer_type_t type, + buffer& qkv_weights, + buffer& o_weights, + buffer& kv_cache, + buffer& q_norm, + buffer& k_norm, + SafeTensors* reference + ) { + if (is_swa_layer(type)) { + _forward_swa(s, type, qkv_weights, o_weights, kv_cache, q_norm, k_norm, reference); + } + else { + _forward_global(s, type, qkv_weights, o_weights, kv_cache, q_norm, k_norm, reference); + } + } + + /// @brief Rebuilds the global prefill cache: everything up to L_begin from the + /// decode cache, then this batch's new rows, written to both caches. + void sync_kv_cache( + buffer& buffer_k, + buffer& buffer_v, + buffer& prefill_cache, + buffer& decoding_cache, + int L_offset, + int L_begin, + int L_effective, + int DK, int DV + ); + + /// @brief Rebuilds the sliding prefill cache, unrolling the decode ring buffer + /// into the linear order the swa kernel expects. + void sync_sliding_kv_cache( + buffer& buffer_k, + buffer& buffer_v, + buffer& prefill_cache, + buffer& decoding_cache, + int L_offset, + int L_begin, + int L_effective, + int sliding_l_begin, + int L_end_chunked, + int DK, int DV + ); + +private: + void _forward_swa(const gemma4e_prefill_shape& s, gemma4e_layer_type_t type, + buffer& qkv_weights, buffer& o_weights, buffer& kv_cache, + buffer& q_norm, buffer& k_norm, SafeTensors* reference); + void _forward_global(const gemma4e_prefill_shape& s, gemma4e_layer_type_t type, + buffer& qkv_weights, buffer& o_weights, buffer& kv_cache, + buffer& q_norm, buffer& k_norm, SafeTensors* reference); +}; + +/// @brief Feed-forward half of one prefill layer. +/// +/// Skip layers run a double-wide mlp when the model enables it, which is a +/// different sequence over the same buffers, so both widths are kept ready. +struct gemma4e_mlp_prefill_context { + gemma4e_desc* desc; + Gemm* gemm_seq_gen; + gemma4e_common_buffers* bufs; + + int L_padded_old = -1; + + npu_app gate_proj, up_proj, down_proj; + npu_app gate_skip_proj, up_skip_proj, down_skip_proj; + + gemma4e_mlp_prefill_context( + gemma4e_desc* desc, + Gemm* gemm_seq_gen, + gemma4e_common_buffers* bufs, + npu_app_manager* gemm_app_manager + ) : desc(desc), gemm_seq_gen(gemm_seq_gen), bufs(bufs) + { + this->gate_proj = gemm_app_manager->create_app(); + this->up_proj = gemm_app_manager->create_app(); + this->down_proj = gemm_app_manager->create_app(); + this->gate_skip_proj = gemm_app_manager->create_app(); + this->up_skip_proj = gemm_app_manager->create_app(); + this->down_skip_proj = gemm_app_manager->create_app(); + } + + void setup(const gemma4e_prefill_shape& s) { + if (L_padded_old == s.L_padded) { + return; + } + const uint32_t D = desc->D; + const uint32_t I = desc->INTERMEDIATE_SIZE; + gemm_seq_gen->generate_seq(this->gate_proj.seq(), s.L_padded, D, I, 0, + false, Gemm::GeLU, 0); + gemm_seq_gen->generate_seq(this->up_proj.seq(), s.L_padded, D, I, 0, + false, Gemm::NO_Activation, 0); + gemm_seq_gen->generate_seq(this->down_proj.seq(), s.L_padded, I, D, 0, + false, Gemm::NO_Activation, 0); + + gemm_seq_gen->generate_seq(this->gate_skip_proj.seq(), s.L_padded, D, I * 2, 0, + false, Gemm::GeLU, 0); + gemm_seq_gen->generate_seq(this->up_skip_proj.seq(), s.L_padded, D, I * 2, 0, + false, Gemm::NO_Activation, 0); + gemm_seq_gen->generate_seq(this->down_skip_proj.seq(), s.L_padded, I * 2, D, 0, + false, Gemm::NO_Activation, 0); + L_padded_old = s.L_padded; + } + + /// @brief Runs the mlp of one layer, hidden_state_buffer in and out. + void forward( + const gemma4e_prefill_shape& s, + bool double_wide, + buffer& gate_weights, + buffer& up_weights, + buffer& down_weights, + SafeTensors* reference + ); +}; + +/// @brief The per-layer-input path: a model-wide down projection run once per +/// batch, and a gate/up pair run at the end of every layer. +struct gemma4e_pli_prefill_context { + gemma4e_desc* desc; + gemma4e_common_buffers* bufs; + Gemma4e_ImageEncoder* image_encoder; ///< owns the mm tiling the pli gemms reuse + + int L_padded_512_old = -1; + + npu_app pli_down_proj, pli_gate_proj, pli_up_proj; + + gemma4e_pli_prefill_context( + gemma4e_desc* desc, + gemma4e_common_buffers* bufs, + Gemma4e_ImageEncoder* image_encoder + ) : desc(desc), bufs(bufs), image_encoder(image_encoder) + { + this->pli_down_proj = image_encoder->proj->create_app(); + this->pli_gate_proj = image_encoder->proj->create_app(); + this->pli_up_proj = image_encoder->proj->create_app(); + } + + void setup(const gemma4e_prefill_shape& s); + + /// @brief Projects the token embeddings down into the per-layer embeddings and + /// folds them into the per-layer-input stream. Runs once per batch. + void pre_pass(const gemma4e_prefill_shape& s, buffer& pli_down_weights, buffer& pli_input_norm); + + /// @brief Gates this layer's per-layer embedding and adds it back to the + /// residual, leaving the result in hidden_state_buffer. + void layer_pass(const gemma4e_prefill_shape& s, int layer_idx, + buffer& pli_gate_up_weights, buffer& pli_final_norm); +}; + +/// @brief One prefill pass over the model, one layer at a time. +/// +/// Owns the prefill xclbins, the sequence generators and every scratch buffer +/// the prefill path needs; the model only hands it weights and a kv cache. The +/// generated sequences are cached on the batch geometry, so a prefill that +/// repeats a shape regenerates nothing. +struct gemma4e_prefill_context { + gemma4e_desc* desc; + npu_xclbin_manager* npu; + gemma4e_common_buffers bufs; + + std::unique_ptr gemm_seq_gen; + std::unique_ptr dequant_block; + std::unique_ptr attn_block; + std::unique_ptr mlp; + std::unique_ptr pli; + + npu_app_manager* gemm_app_manager; + npu_app_manager* dequant_app_manager; + npu_app_manager* mha_app_manager; + npu_app_manager* swa_app_manager; + + /// @brief the attention kernel consumes whole chunks of this many rows + static constexpr int L_chunk = 16 * 8; + /// @brief the attention kernel refuses batches shorter than this + static constexpr int L_MIN = 256; + + /// @note The shared buffers are sized from both L_padded_512 (the residual-side + /// tensors) and L_padded (the q/k/v/gate/up scratch), and the two can move + /// independently, so reallocate when either changes. + int L_padded_512_old = -1; + int L_padded_old = -1; + + gemma4e_prefill_context( + npu_xclbin_manager* npu, + gemma4e_desc* desc, + LM_Config& config, + gemma4e_npu_sequence* seq_gen, + std::unique_ptr pli, + uint32_t MAX_L + ) : desc(desc), npu(npu) + { + this->gemm_app_manager = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "mm.xclbin")); + this->dequant_app_manager = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "dequant.xclbin")); + this->mha_app_manager = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "attn.xclbin")); + this->swa_app_manager = npu->register_xclbin(utils::path_join(config.exec_path, "xclbins", config.model_name, "swa.xclbin")); + + this->gemm_seq_gen = std::make_unique(config); + + this->attn_block = std::make_unique( + desc, gemm_seq_gen.get(), seq_gen, &bufs, + gemm_app_manager, mha_app_manager, swa_app_manager + ); + this->mlp = std::make_unique( + desc, gemm_seq_gen.get(), &bufs, gemm_app_manager + ); + this->dequant_block = std::make_unique( + desc, seq_gen, dequant_app_manager + ); + // the per layer input apps must already exist by now: they are created on + // the image encoder's manager, which has to happen before layer.xclbin is + // registered, so the model builds this block and hands it over. + this->pli = std::move(pli); + this->pli->bufs = &bufs; + + this->attn_block->set_max_length(MAX_L); + } + + void set_max_length(uint32_t MAX_L) { this->attn_block->set_max_length(MAX_L); } + + /// @brief Computes the row geometry of a prefill call and readies every block. + /// @param L_in number of tokens to prefill + /// @param L_begin context length already in the kv cache + gemma4e_prefill_shape setup(int L_in, int L_begin) { + gemma4e_prefill_shape s; + s.L_in = L_in; + s.L_effective = L_in; + s.L_begin = L_begin; + s.L_end = L_begin + L_in; + s.L_begin_chunked = (L_begin / L_chunk) * L_chunk; + s.L_offset = L_begin - s.L_begin_chunked; + s.L_end_chunked = s.L_begin_chunked + (L_in + s.L_offset + L_MIN - 1) / L_MIN * L_MIN; + s.L_padded = s.L_end_chunked - s.L_begin_chunked; + s.L_padded_512 = ((s.L_padded + 511) / 512) * 512; + s.sliding_l_begin = (s.L_begin_chunked - (int)desc->SLIDING_LENGTH) > 0 + ? (s.L_begin_chunked - (int)desc->SLIDING_LENGTH) : 0; + + if (L_padded_512_old != s.L_padded_512 || L_padded_old != s.L_padded) { + bufs.allocate(npu, desc, s); + L_padded_512_old = s.L_padded_512; + L_padded_old = s.L_padded; + } + bufs.reset(desc, s); + + this->attn_block->setup(s); + this->mlp->setup(s); + this->pli->setup(s); + return s; + } + + /// @brief Row `i` of the residual stream, where the token embeddings are written. + buffer& residual_row(int i) { return bufs.tensor_residual[i]; } + /// @brief Row `i` of the per-layer-input embeddings. + buffer& pli_embed_row(int i) { return bufs.tensor_pli_embed[i]; } + + /// @brief Runs one decoder layer over the whole batch, in place on the residual. + void forward( + int layer_idx, + gemma4e_layer_type_t type, + const gemma4e_prefill_shape& s, + buffer& proj_weights, + buffer& rms_weights, + buffer& rope_rms_weights, + buffer& pli_gate_up_weights, + buffer& kv_cache, + float layer_scale, + SafeTensors* reference + ); +}; + +#endif // __GEMMA4E_PREFILL_HPP__ diff --git a/src/detail/gemma4e_npu/gemma4e_vision_prefill_helper.cpp b/src/detail/gemma4e_npu/gemma4e_vision_prefill_helper.cpp new file mode 100644 index 000000000..1008c5246 --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_vision_prefill_helper.cpp @@ -0,0 +1,1320 @@ +#include "gemma4e_vision_prefill_helper.hpp" + +#include "avx512_util.hpp" +#include +#include +#include +#include // std::bit_cast (C++20+) +#include +#include // OpenMP for multi-threading + +void simd_add( + bf16* input1, + bf16* input2, + bf16* output, + size_t size +){ + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 4; // Process 64 elements per iteration + constexpr size_t CHUNK_SIZE = SIMD_WIDTH * UNROLL_FACTOR; // 64 elements + + // Use OpenMP only if size is large enough to benefit from parallelization + // Threshold: 8 chunks (512 elements) per thread minimum to avoid overhead + constexpr size_t min_elements_per_thread = CHUNK_SIZE * 8; + + const bool use_parallel = size >= (min_elements_per_thread * 2); + + // Calculate number of full chunks for parallel processing + const size_t num_chunks = size / CHUNK_SIZE; + const int signed_num_chunks = static_cast(num_chunks); + + // Process full chunks with OpenMP (chunked distribution for cache locality) + #pragma omp parallel for num_threads(max_prefill_threads) if(use_parallel) schedule(static) + for (int chunk_idx = 0; chunk_idx < signed_num_chunks; ++chunk_idx) { + size_t i = static_cast(chunk_idx) * CHUNK_SIZE; + + // Prefetch next cache lines + _mm_prefetch(reinterpret_cast(input1 + i + 64), _MM_HINT_T0); + _mm_prefetch(reinterpret_cast(input2 + i + 64), _MM_HINT_T0); + + // Load all inputs first (better for out-of-order execution) + __m256i a_bh_vec0 = _mm256_loadu_si256(reinterpret_cast(input1 + i)); + __m256i b_bh_vec0 = _mm256_loadu_si256(reinterpret_cast(input2 + i)); + + __m256i a_bh_vec1 = _mm256_loadu_si256(reinterpret_cast(input1 + i + 16)); + __m256i b_bh_vec1 = _mm256_loadu_si256(reinterpret_cast(input2 + i + 16)); + + __m256i a_bh_vec2 = _mm256_loadu_si256(reinterpret_cast(input1 + i + 32)); + __m256i b_bh_vec2 = _mm256_loadu_si256(reinterpret_cast(input2 + i + 32)); + + __m256i a_bh_vec3 = _mm256_loadu_si256(reinterpret_cast(input1 + i + 48)); + __m256i b_bh_vec3 = _mm256_loadu_si256(reinterpret_cast(input2 + i + 48)); + + // Convert to 32-bit and shift (interleaved for better pipeline usage) + __m512i a_shifted0 = _mm512_slli_epi32(_mm512_cvtepu16_epi32(a_bh_vec0), 16); + __m512i b_shifted0 = _mm512_slli_epi32(_mm512_cvtepu16_epi32(b_bh_vec0), 16); + + __m512i a_shifted1 = _mm512_slli_epi32(_mm512_cvtepu16_epi32(a_bh_vec1), 16); + __m512i b_shifted1 = _mm512_slli_epi32(_mm512_cvtepu16_epi32(b_bh_vec1), 16); + + __m512i a_shifted2 = _mm512_slli_epi32(_mm512_cvtepu16_epi32(a_bh_vec2), 16); + __m512i b_shifted2 = _mm512_slli_epi32(_mm512_cvtepu16_epi32(b_bh_vec2), 16); + + __m512i a_shifted3 = _mm512_slli_epi32(_mm512_cvtepu16_epi32(a_bh_vec3), 16); + __m512i b_shifted3 = _mm512_slli_epi32(_mm512_cvtepu16_epi32(b_bh_vec3), 16); + + // Add as floats + __m512 sum0 = _mm512_add_ps(_mm512_castsi512_ps(a_shifted0), _mm512_castsi512_ps(b_shifted0)); + __m512 sum1 = _mm512_add_ps(_mm512_castsi512_ps(a_shifted1), _mm512_castsi512_ps(b_shifted1)); + __m512 sum2 = _mm512_add_ps(_mm512_castsi512_ps(a_shifted2), _mm512_castsi512_ps(b_shifted2)); + __m512 sum3 = _mm512_add_ps(_mm512_castsi512_ps(a_shifted3), _mm512_castsi512_ps(b_shifted3)); + + // Convert back to bfloat16 and store with round-to-nearest-even + store_m512_to_bfloat16_rne(output + i, sum0); + store_m512_to_bfloat16_rne(output + i + 16, sum1); + store_m512_to_bfloat16_rne(output + i + 32, sum2); + store_m512_to_bfloat16_rne(output + i + 48, sum3); + } + + // Process remaining elements (remainder after full chunks) + size_t i = num_chunks * CHUNK_SIZE; + + // Process remaining 16-element chunks + for (; i + SIMD_WIDTH <= size; i += SIMD_WIDTH) { + __m256i a_bh_vec = _mm256_loadu_si256(reinterpret_cast(input1 + i)); + __m256i b_bh_vec = _mm256_loadu_si256(reinterpret_cast(input2 + i)); + + __m512i a_shifted = _mm512_slli_epi32(_mm512_cvtepu16_epi32(a_bh_vec), 16); + __m512i b_shifted = _mm512_slli_epi32(_mm512_cvtepu16_epi32(b_bh_vec), 16); + + __m512 sum = _mm512_add_ps(_mm512_castsi512_ps(a_shifted), _mm512_castsi512_ps(b_shifted)); + + store_m512_to_bfloat16_rne(output + i, sum); + } + + // Scalar tail + for (; i < size; ++i) { + output[i] = static_cast(static_cast(input1[i]) + static_cast(input2[i])); + } +} + +void transpose_2d( + const bf16* input, + bf16* output, + size_t rows, + size_t cols +){ + int signed_rows = static_cast(rows); + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) if(rows >= 8) + for(int r = 0; r < signed_rows; r++){ + for(int c = 0; c < static_cast(cols); c++){ + output[c * rows + r] = input[r * cols + c]; + } + } +} + +void simd_add( + const float* input1, + const bf16* input2, + float* output, + size_t size +) { + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 4; + constexpr size_t CHUNK_SIZE = SIMD_WIDTH * UNROLL_FACTOR; + constexpr size_t min_elements_per_thread = CHUNK_SIZE * 8; + + const bool use_parallel = size >= (min_elements_per_thread * 2); + const size_t num_chunks = size / CHUNK_SIZE; + const int signed_num_chunks = static_cast(num_chunks); + + #pragma omp parallel for num_threads(max_prefill_threads) if(use_parallel) schedule(static) + for (int chunk_idx = 0; chunk_idx < signed_num_chunks; ++chunk_idx) { + size_t base = static_cast(chunk_idx) * CHUNK_SIZE; + + _mm_prefetch(reinterpret_cast(input1 + base + 64), _MM_HINT_T0); + _mm_prefetch(reinterpret_cast(input2 + base + 64), _MM_HINT_T0); + + for (size_t j = 0; j < CHUNK_SIZE; j += SIMD_WIDTH) { + size_t i = base + j; + __m512 a_ps_vec = _mm512_loadu_ps(input1 + i); + __m512 b_ps_vec = load_bfloat16_to_m512(input2 + i); + _mm512_storeu_ps(output + i, _mm512_add_ps(a_ps_vec, b_ps_vec)); + } + } + + // Remainder after full chunks + size_t i = num_chunks * CHUNK_SIZE; + for (; i + SIMD_WIDTH <= size; i += SIMD_WIDTH) { + __m512 a_ps_vec = _mm512_loadu_ps(input1 + i); + __m512 b_ps_vec = load_bfloat16_to_m512(input2 + i); + _mm512_storeu_ps(output + i, _mm512_add_ps(a_ps_vec, b_ps_vec)); + } + + // Scalar tail + for (; i < size; ++i) { + output[i] = input1[i] + static_cast(input2[i]); + } +} + +void simd_add( + const float* input1, + const bf16* input2, + bf16* output, + size_t size +) { + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 4; + constexpr size_t CHUNK_SIZE = SIMD_WIDTH * UNROLL_FACTOR; + constexpr size_t min_elements_per_thread = CHUNK_SIZE * 8; + + const bool use_parallel = size >= (min_elements_per_thread * 2); + const size_t num_chunks = size / CHUNK_SIZE; + const int signed_num_chunks = static_cast(num_chunks); + + #pragma omp parallel for num_threads(max_prefill_threads) if(use_parallel) schedule(static) + for (int chunk_idx = 0; chunk_idx < signed_num_chunks; ++chunk_idx) { + size_t base = static_cast(chunk_idx) * CHUNK_SIZE; + + _mm_prefetch(reinterpret_cast(input1 + base + 64), _MM_HINT_T0); + _mm_prefetch(reinterpret_cast(input2 + base + 64), _MM_HINT_T0); + + __m512 a_vec0 = _mm512_loadu_ps(input1 + base); + __m512 b_vec0 = load_bfloat16_to_m512(input2 + base); + __m512 a_vec1 = _mm512_loadu_ps(input1 + base + 16); + __m512 b_vec1 = load_bfloat16_to_m512(input2 + base + 16); + __m512 a_vec2 = _mm512_loadu_ps(input1 + base + 32); + __m512 b_vec2 = load_bfloat16_to_m512(input2 + base + 32); + __m512 a_vec3 = _mm512_loadu_ps(input1 + base + 48); + __m512 b_vec3 = load_bfloat16_to_m512(input2 + base + 48); + + store_m512_to_bfloat16_rne(output + base, _mm512_add_ps(a_vec0, b_vec0)); + store_m512_to_bfloat16_rne(output + base + 16, _mm512_add_ps(a_vec1, b_vec1)); + store_m512_to_bfloat16_rne(output + base + 32, _mm512_add_ps(a_vec2, b_vec2)); + store_m512_to_bfloat16_rne(output + base + 48, _mm512_add_ps(a_vec3, b_vec3)); + } + + size_t i = num_chunks * CHUNK_SIZE; + for (; i + SIMD_WIDTH <= size; i += SIMD_WIDTH) { + __m512 a_vec = _mm512_loadu_ps(input1 + i); + __m512 b_vec = load_bfloat16_to_m512(input2 + i); + store_m512_to_bfloat16_rne(output + i, _mm512_add_ps(a_vec, b_vec)); + } + + for (; i < size; ++i) { + float sum = input1[i] + static_cast(input2[i]); + output[i] = static_cast(sum); + } +} + +void simd_add( + const bf16* input1, + const bf16* input2, + float* output, + size_t size +) { + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 4; + constexpr size_t CHUNK_SIZE = SIMD_WIDTH * UNROLL_FACTOR; + constexpr size_t min_elements_per_thread = CHUNK_SIZE * 8; + + const bool use_parallel = size >= (min_elements_per_thread * 2); + const size_t num_chunks = size / CHUNK_SIZE; + const int signed_num_chunks = static_cast(num_chunks); + + #pragma omp parallel for num_threads(max_prefill_threads) if(use_parallel) schedule(static) + for (int chunk_idx = 0; chunk_idx < signed_num_chunks; ++chunk_idx) { + size_t base = static_cast(chunk_idx) * CHUNK_SIZE; + + _mm_prefetch(reinterpret_cast(input1 + base + 64), _MM_HINT_T0); + _mm_prefetch(reinterpret_cast(input2 + base + 64), _MM_HINT_T0); + + __m512 a_vec0 = load_bfloat16_to_m512(input1 + base); + __m512 b_vec0 = load_bfloat16_to_m512(input2 + base); + __m512 a_vec1 = load_bfloat16_to_m512(input1 + base + 16); + __m512 b_vec1 = load_bfloat16_to_m512(input2 + base + 16); + __m512 a_vec2 = load_bfloat16_to_m512(input1 + base + 32); + __m512 b_vec2 = load_bfloat16_to_m512(input2 + base + 32); + __m512 a_vec3 = load_bfloat16_to_m512(input1 + base + 48); + __m512 b_vec3 = load_bfloat16_to_m512(input2 + base + 48); + + _mm512_storeu_ps(output + base, _mm512_add_ps(a_vec0, b_vec0)); + _mm512_storeu_ps(output + base + 16, _mm512_add_ps(a_vec1, b_vec1)); + _mm512_storeu_ps(output + base + 32, _mm512_add_ps(a_vec2, b_vec2)); + _mm512_storeu_ps(output + base + 48, _mm512_add_ps(a_vec3, b_vec3)); + } + + size_t i = num_chunks * CHUNK_SIZE; + for (; i + SIMD_WIDTH <= size; i += SIMD_WIDTH) { + __m512 a_vec = load_bfloat16_to_m512(input1 + i); + __m512 b_vec = load_bfloat16_to_m512(input2 + i); + _mm512_storeu_ps(output + i, _mm512_add_ps(a_vec, b_vec)); + } + + for (; i < size; ++i) { + float sum = static_cast(input1[i]) + static_cast(input2[i]); + output[i] = sum; + } +} + +// Optimized AVX-512 version that combines bias addition and GELU activation +// This version processes the entire array in one pass, reducing memory traffic +// Note: bias is broadcast across all sequence positions (hidden_dim sized, repeated for seq_len) +// Optimizations: +// - OpenMP parallelization (4 threads) for sequence positions +// - Prefetching for cache optimization +// - 2x loop unrolling for better ILP (instruction-level parallelism) +void simd_bias_add_gelu( + bf16* input, + const bf16* bias, + bf16* output, + size_t total_size, + size_t hidden_dim +){ + size_t seq_len = total_size / hidden_dim; + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 2; + + // Use signed integer for OpenMP compatibility + int signed_seq_len = static_cast(seq_len); + + // Parallelize over sequence positions with OpenMP (max 4 threads) + // Each thread processes different sequence positions independently + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) + for (int seq = 0; seq < signed_seq_len; ++seq) { + size_t seq_offset = static_cast(seq) * hidden_dim; + size_t i = 0; + + // Process 32 elements at a time (2x unrolled for better ILP) + for (; i + SIMD_WIDTH * UNROLL_FACTOR <= hidden_dim; i += SIMD_WIDTH * UNROLL_FACTOR) { + // Prefetch next cache lines for input, bias, and output + _mm_prefetch(reinterpret_cast(input + seq_offset + i + 64), _MM_HINT_T0); + _mm_prefetch(reinterpret_cast(bias + i + 64), _MM_HINT_T0); + + // First iteration (elements i to i+15) + __m512 input_vec1 = load_bfloat16_to_m512(input + seq_offset + i); + __m512 bias_vec1 = load_bfloat16_to_m512(bias + i); + + // Second iteration (elements i+16 to i+31) - load in parallel to hide latency + __m512 input_vec2 = load_bfloat16_to_m512(input + seq_offset + i + SIMD_WIDTH); + __m512 bias_vec2 = load_bfloat16_to_m512(bias + i + SIMD_WIDTH); + + // Compute first iteration + __m512 sum_vec1 = _mm512_add_ps(input_vec1, bias_vec1); + + // Compute second iteration (parallel to first GELU computation) + __m512 sum_vec2 = _mm512_add_ps(input_vec2, bias_vec2); + + // Apply GELU activation (computationally expensive, so interleave) + __m512 gelu_vec1 = gelu_tanh_avx512_simd(sum_vec1); + __m512 gelu_vec2 = gelu_tanh_avx512_simd(sum_vec2); + + // Store results + store_m512_to_bfloat16_rne(output + seq_offset + i, gelu_vec1); + store_m512_to_bfloat16_rne(output + seq_offset + i + SIMD_WIDTH, gelu_vec2); + } + + // Process remaining 16-element chunks + for (; i + SIMD_WIDTH <= hidden_dim; i += SIMD_WIDTH) { + __m512 input_vec = load_bfloat16_to_m512(input + seq_offset + i); + __m512 bias_vec = load_bfloat16_to_m512(bias + i); + + __m512 sum_vec = _mm512_add_ps(input_vec, bias_vec); + __m512 gelu_vec = gelu_tanh_avx512_simd(sum_vec); + + store_m512_to_bfloat16_rne(output + seq_offset + i, gelu_vec); + } + + // Handle remaining elements with scalar loop + constexpr float sqrt_2_over_pi = 0.7978845608f; // √(2/π) + constexpr float coeff = 0.044715f; + + for (; i < hidden_dim; ++i) { + float x = static_cast(input[seq_offset + i]); + float b = static_cast(bias[i]); + float sum = x + b; + + // GELU approximation (tanh-based) + float x_cubed = sum * sum * sum; + float inner = sqrt_2_over_pi * (sum + coeff * x_cubed); + float tanh_val = std::tanh(inner); + float gelu = 0.5f * sum * (1.0f + tanh_val); + + output[seq_offset + i] = static_cast(gelu); + } + } +} + +void gelu_bfloat16_ref( + const bf16* input, + bf16* output, + size_t size +) { + size_t i = 0; + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 2; // Process 32 elements per iteration + + // Process 32 elements at a time (2x unrolled for GELU's computational intensity) + for (; i + SIMD_WIDTH * UNROLL_FACTOR <= size; i += SIMD_WIDTH * UNROLL_FACTOR) { + // Prefetch next cache line + _mm_prefetch(reinterpret_cast(input + i + 64), _MM_HINT_T0); + + // Load both sets of data + __m512 input_vec1 = load_bfloat16_to_m512(input + i); + __m512 input_vec2 = load_bfloat16_to_m512(input + i + SIMD_WIDTH); + + // Apply GELU activation to both + __m512 gelu_vec1 = gelu_tanh_avx512_simd(input_vec1); + __m512 gelu_vec2 = gelu_tanh_avx512_simd(input_vec2); + + // Store results + store_m512_to_bfloat16_rne(output + i, gelu_vec1); + store_m512_to_bfloat16_rne(output + i + SIMD_WIDTH, gelu_vec2); + } + + // Process remaining 16-element chunks + for (; i + SIMD_WIDTH <= size; i += SIMD_WIDTH) { + __m512 input_vec = load_bfloat16_to_m512(input + i); + __m512 gelu_vec = gelu_tanh_avx512_simd(input_vec); + store_m512_to_bfloat16_rne(output + i, gelu_vec); + } + + // Constants for the GELU approximation (tanh-based) + constexpr double sqrt_2_over_pi = 0.7978845608028654; // √(2/π) + constexpr double coeff = 0.044715; + + // Scalar tail + for (; i < size; ++i) { + double x = static_cast(input[i]); + double x_cubed = x * x * x; + double inner = sqrt_2_over_pi * (x + coeff * x_cubed); + double tanh_val = std::tanh(inner); + double gelu = 0.5 * x * (1.0 + tanh_val); + output[i] = static_cast(static_cast(gelu)); + } +} + +// RMS Norm without scale weights +// Input layout: [seq_len_padded x X_padded], processes seq_len rows, X elements per row +// Stride between rows is X_padded +void simd_rms_norm( + const bf16* input, + bf16* output, + size_t seq_len, + size_t X, + size_t seq_len_padded, + size_t X_padded, + float eps +) { + constexpr size_t SIMD_WIDTH = 16; + const float inv_X = 1.0f / static_cast(X); + + int signed_seq_len = static_cast(seq_len); + + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) if(seq_len >= 8) + for (int row = 0; row < signed_seq_len; ++row) { + const bf16* row_in = input + static_cast(row) * X_padded; + bf16* row_out = output + static_cast(row) * X_padded; + + // === Pass 1: compute sum of squares === + __m512 acc0 = _mm512_setzero_ps(); + __m512 acc1 = _mm512_setzero_ps(); + size_t i = 0; + + // 2x unrolled SIMD loop + for (; i + SIMD_WIDTH * 2 <= X; i += SIMD_WIDTH * 2) { + __m512 v0 = load_bfloat16_to_m512(row_in + i); + __m512 v1 = load_bfloat16_to_m512(row_in + i + SIMD_WIDTH); + acc0 = _mm512_fmadd_ps(v0, v0, acc0); + acc1 = _mm512_fmadd_ps(v1, v1, acc1); + } + for (; i + SIMD_WIDTH <= X; i += SIMD_WIDTH) { + __m512 v = load_bfloat16_to_m512(row_in + i); + acc0 = _mm512_fmadd_ps(v, v, acc0); + } + + float sum_sq = _mm512_reduce_add_ps(_mm512_add_ps(acc0, acc1)); + + // Scalar tail for sum of squares + for (; i < X; ++i) { + float val = static_cast(row_in[i]); + sum_sq += val * val; + } + + // mean_sq = sum_sq / X + eps, then rsqrt + float mean_sq = sum_sq * inv_X + eps; + float scale = 1.0f / std::sqrt(mean_sq); + __m512 scale_vec = _mm512_set1_ps(scale); + + // === Pass 2: multiply input by scale === + i = 0; + for (; i + SIMD_WIDTH * 2 <= X; i += SIMD_WIDTH * 2) { + __m512 v0 = load_bfloat16_to_m512(row_in + i); + __m512 v1 = load_bfloat16_to_m512(row_in + i + SIMD_WIDTH); + store_m512_to_bfloat16_rne(row_out + i, _mm512_mul_ps(v0, scale_vec)); + store_m512_to_bfloat16_rne(row_out + i + SIMD_WIDTH, _mm512_mul_ps(v1, scale_vec)); + } + for (; i + SIMD_WIDTH <= X; i += SIMD_WIDTH) { + __m512 v = load_bfloat16_to_m512(row_in + i); + store_m512_to_bfloat16_rne(row_out + i, _mm512_mul_ps(v, scale_vec)); + } + + // Scalar tail + for (; i < X; ++i) { + float val = static_cast(row_in[i]); + row_out[i] = static_cast(val * scale); + } + } +} + +// RMS Norm with scale weights (Gemma4RMSNorm with with_scale=True) +// Input layout: [seq_len_padded x X_padded], processes seq_len rows, X elements per row +// norm_weight: bf16 pointer of size [X], broadcast across all rows +void simd_rms_norm( + const bf16* input, + const bf16* norm_weight, + bf16* output, + size_t seq_len, + size_t X, + size_t seq_len_padded, + size_t X_padded, + float eps +) { + constexpr size_t SIMD_WIDTH = 16; + const float inv_X = 1.0f / static_cast(X); + + int signed_seq_len = static_cast(seq_len); + + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) if(seq_len >= 8) + for (int row = 0; row < signed_seq_len; ++row) { + const bf16* row_in = input + static_cast(row) * X_padded; + bf16* row_out = output + static_cast(row) * X_padded; + + // === Pass 1: compute sum of squares === + __m512 acc0 = _mm512_setzero_ps(); + __m512 acc1 = _mm512_setzero_ps(); + size_t i = 0; + + // 2x unrolled SIMD loop + for (; i + SIMD_WIDTH * 2 <= X; i += SIMD_WIDTH * 2) { + __m512 v0 = load_bfloat16_to_m512(row_in + i); + __m512 v1 = load_bfloat16_to_m512(row_in + i + SIMD_WIDTH); + acc0 = _mm512_fmadd_ps(v0, v0, acc0); + acc1 = _mm512_fmadd_ps(v1, v1, acc1); + } + for (; i + SIMD_WIDTH <= X; i += SIMD_WIDTH) { + __m512 v = load_bfloat16_to_m512(row_in + i); + acc0 = _mm512_fmadd_ps(v, v, acc0); + } + + float sum_sq = _mm512_reduce_add_ps(_mm512_add_ps(acc0, acc1)); + + // Scalar tail for sum of squares + for (; i < X; ++i) { + float val = static_cast(row_in[i]); + sum_sq += val * val; + } + + // mean_sq = sum_sq / X + eps, then rsqrt + float mean_sq = sum_sq * inv_X + eps; + float scale = 1.0f / std::sqrt(mean_sq); + __m512 scale_vec = _mm512_set1_ps(scale); + + // === Pass 2: multiply input by scale and norm_weight === + i = 0; + for (; i + SIMD_WIDTH * 2 <= X; i += SIMD_WIDTH * 2) { + __m512 v0 = load_bfloat16_to_m512(row_in + i); + __m512 w0 = load_bfloat16_to_m512(norm_weight + i); + __m512 v1 = load_bfloat16_to_m512(row_in + i + SIMD_WIDTH); + __m512 w1 = load_bfloat16_to_m512(norm_weight + i + SIMD_WIDTH); + store_m512_to_bfloat16_rne(row_out + i, _mm512_mul_ps(_mm512_mul_ps(v0, scale_vec), w0)); + store_m512_to_bfloat16_rne(row_out + i + SIMD_WIDTH, _mm512_mul_ps(_mm512_mul_ps(v1, scale_vec), w1)); + } + for (; i + SIMD_WIDTH <= X; i += SIMD_WIDTH) { + __m512 v = load_bfloat16_to_m512(row_in + i); + __m512 w = load_bfloat16_to_m512(norm_weight + i); + store_m512_to_bfloat16_rne(row_out + i, _mm512_mul_ps(_mm512_mul_ps(v, scale_vec), w)); + } + + // Scalar tail + for (; i < X; ++i) { + float val = static_cast(row_in[i]); + row_out[i] = static_cast(val * scale * static_cast(norm_weight[i])); + } + } +} + +// AVX-512 clamp for bfloat16: output[i] = clamp(input[i], min_val, max_val) +// Uses 4x unrolling for better ILP and OpenMP parallelization for large arrays. +void simd_clamp( + const bf16* input, + bf16* output, + bf16 min_val, + bf16 max_val, + size_t size +) { + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 4; + constexpr size_t CHUNK_SIZE = SIMD_WIDTH * UNROLL_FACTOR; // 64 elements + + const __m512 vmin = _mm512_set1_ps(static_cast(min_val)); + const __m512 vmax = _mm512_set1_ps(static_cast(max_val)); + + constexpr size_t min_elements_per_thread = CHUNK_SIZE * 8; + const bool use_parallel = size >= (min_elements_per_thread * 2); + const size_t num_chunks = size / CHUNK_SIZE; + const int signed_num_chunks = static_cast(num_chunks); + + #pragma omp parallel for num_threads(max_prefill_threads) if(use_parallel) schedule(static) + for (int chunk_idx = 0; chunk_idx < signed_num_chunks; ++chunk_idx) { + size_t i = static_cast(chunk_idx) * CHUNK_SIZE; + + _mm_prefetch(reinterpret_cast(input + i + 64), _MM_HINT_T0); + + __m512 v0 = load_bfloat16_to_m512(input + i); + __m512 v1 = load_bfloat16_to_m512(input + i + 16); + __m512 v2 = load_bfloat16_to_m512(input + i + 32); + __m512 v3 = load_bfloat16_to_m512(input + i + 48); + + v0 = _mm512_min_ps(_mm512_max_ps(v0, vmin), vmax); + v1 = _mm512_min_ps(_mm512_max_ps(v1, vmin), vmax); + v2 = _mm512_min_ps(_mm512_max_ps(v2, vmin), vmax); + v3 = _mm512_min_ps(_mm512_max_ps(v3, vmin), vmax); + + store_m512_to_bfloat16_rne(output + i, v0); + store_m512_to_bfloat16_rne(output + i + 16, v1); + store_m512_to_bfloat16_rne(output + i + 32, v2); + store_m512_to_bfloat16_rne(output + i + 48, v3); + } + + // Remaining 16-element chunks + size_t i = num_chunks * CHUNK_SIZE; + for (; i + SIMD_WIDTH <= size; i += SIMD_WIDTH) { + __m512 v = load_bfloat16_to_m512(input + i); + v = _mm512_min_ps(_mm512_max_ps(v, vmin), vmax); + store_m512_to_bfloat16_rne(output + i, v); + } + + // Scalar tail + for (; i < size; ++i) { + float val = static_cast(input[i]); + float fmin = static_cast(min_val); + float fmax = static_cast(max_val); + if (val < fmin) val = fmin; + if (val > fmax) val = fmax; + output[i] = static_cast(val); + } +} + +void simd_mul( + const bf16* input1, + const bf16* input2, + bf16* output, + size_t size +) { + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 4; + constexpr size_t CHUNK_SIZE = SIMD_WIDTH * UNROLL_FACTOR; + constexpr size_t min_elements_per_thread = CHUNK_SIZE * 8; + constexpr int max_threads = max_prefill_threads; + + const bool use_parallel = size >= (min_elements_per_thread * 2); + const size_t num_chunks = size / CHUNK_SIZE; + const int signed_num_chunks = static_cast(num_chunks); + + #pragma omp parallel for num_threads(max_threads) if(use_parallel) schedule(static) + for (int chunk_idx = 0; chunk_idx < signed_num_chunks; ++chunk_idx) { + const size_t i = static_cast(chunk_idx) * CHUNK_SIZE; + + _mm_prefetch(reinterpret_cast(input1 + i + 64), _MM_HINT_T0); + _mm_prefetch(reinterpret_cast(input2 + i + 64), _MM_HINT_T0); + + const __m512 a_vec0 = load_bfloat16_to_m512(input1 + i); + const __m512 b_vec0 = load_bfloat16_to_m512(input2 + i); + const __m512 a_vec1 = load_bfloat16_to_m512(input1 + i + 16); + const __m512 b_vec1 = load_bfloat16_to_m512(input2 + i + 16); + const __m512 a_vec2 = load_bfloat16_to_m512(input1 + i + 32); + const __m512 b_vec2 = load_bfloat16_to_m512(input2 + i + 32); + const __m512 a_vec3 = load_bfloat16_to_m512(input1 + i + 48); + const __m512 b_vec3 = load_bfloat16_to_m512(input2 + i + 48); + + store_m512_to_bfloat16_rne(output + i, _mm512_mul_ps(a_vec0, b_vec0)); + store_m512_to_bfloat16_rne(output + i + 16, _mm512_mul_ps(a_vec1, b_vec1)); + store_m512_to_bfloat16_rne(output + i + 32, _mm512_mul_ps(a_vec2, b_vec2)); + store_m512_to_bfloat16_rne(output + i + 48, _mm512_mul_ps(a_vec3, b_vec3)); + } + + size_t i = num_chunks * CHUNK_SIZE; + + for (; i + SIMD_WIDTH <= size; i += SIMD_WIDTH) { + const __m512 a_vec = load_bfloat16_to_m512(input1 + i); + const __m512 b_vec = load_bfloat16_to_m512(input2 + i); + store_m512_to_bfloat16_rne(output + i, _mm512_mul_ps(a_vec, b_vec)); + } + + for (; i < size; ++i) { + output[i] = static_cast(static_cast(input1[i]) * static_cast(input2[i])); + } +} + +void simd_mul( + const bf16* input1, + bf16 input2_scalar, + bf16* output, + size_t size +) { + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 4; + constexpr size_t CHUNK_SIZE = SIMD_WIDTH * UNROLL_FACTOR; + constexpr size_t min_elements_per_thread = CHUNK_SIZE * 8; + constexpr int max_threads = max_prefill_threads; + + // Broadcast scalar to all 16 lanes once + const __m512 scalar_vec = _mm512_set1_ps(static_cast(input2_scalar)); + + const bool use_parallel = size >= (min_elements_per_thread * 2); + const size_t num_chunks = size / CHUNK_SIZE; + const int signed_num_chunks = static_cast(num_chunks); + + #pragma omp parallel for num_threads(max_threads) if(use_parallel) schedule(static) + for (int chunk_idx = 0; chunk_idx < signed_num_chunks; ++chunk_idx) { + const size_t i = static_cast(chunk_idx) * CHUNK_SIZE; + + _mm_prefetch(reinterpret_cast(input1 + i + 64), _MM_HINT_T0); + + const __m512 a_vec0 = load_bfloat16_to_m512(input1 + i); + const __m512 a_vec1 = load_bfloat16_to_m512(input1 + i + 16); + const __m512 a_vec2 = load_bfloat16_to_m512(input1 + i + 32); + const __m512 a_vec3 = load_bfloat16_to_m512(input1 + i + 48); + + store_m512_to_bfloat16_rne(output + i, _mm512_mul_ps(a_vec0, scalar_vec)); + store_m512_to_bfloat16_rne(output + i + 16, _mm512_mul_ps(a_vec1, scalar_vec)); + store_m512_to_bfloat16_rne(output + i + 32, _mm512_mul_ps(a_vec2, scalar_vec)); + store_m512_to_bfloat16_rne(output + i + 48, _mm512_mul_ps(a_vec3, scalar_vec)); + } + + size_t i = num_chunks * CHUNK_SIZE; + + for (; i + SIMD_WIDTH <= size; i += SIMD_WIDTH) { + const __m512 a_vec = load_bfloat16_to_m512(input1 + i); + store_m512_to_bfloat16_rne(output + i, _mm512_mul_ps(a_vec, scalar_vec)); + } + + const float scalar_f = static_cast(input2_scalar); + for (; i < size; ++i) { + output[i] = static_cast(static_cast(input1[i]) * scalar_f); + } +} + +// ============================================================================ +// simd_relu: AVX-512 ReLU for bfloat16 +// output[i] = max(input[i], 0) +// ============================================================================ +void simd_relu( + const bf16* input, + bf16* output, + size_t size +) { + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 4; + constexpr size_t CHUNK_SIZE = SIMD_WIDTH * UNROLL_FACTOR; // 64 elements + + const __m512 zero = _mm512_setzero_ps(); + + constexpr size_t min_elements_per_thread = CHUNK_SIZE * 8; + const bool use_parallel = size >= (min_elements_per_thread * 2); + const size_t num_chunks = size / CHUNK_SIZE; + const int signed_num_chunks = static_cast(num_chunks); + + #pragma omp parallel for num_threads(max_prefill_threads) if(use_parallel) schedule(static) + for (int chunk_idx = 0; chunk_idx < signed_num_chunks; ++chunk_idx) { + size_t i = static_cast(chunk_idx) * CHUNK_SIZE; + + _mm_prefetch(reinterpret_cast(input + i + 64), _MM_HINT_T0); + + __m512 v0 = load_bfloat16_to_m512(input + i); + __m512 v1 = load_bfloat16_to_m512(input + i + 16); + __m512 v2 = load_bfloat16_to_m512(input + i + 32); + __m512 v3 = load_bfloat16_to_m512(input + i + 48); + + store_m512_to_bfloat16_rne(output + i, _mm512_max_ps(v0, zero)); + store_m512_to_bfloat16_rne(output + i + 16, _mm512_max_ps(v1, zero)); + store_m512_to_bfloat16_rne(output + i + 32, _mm512_max_ps(v2, zero)); + store_m512_to_bfloat16_rne(output + i + 48, _mm512_max_ps(v3, zero)); + } + + size_t i = num_chunks * CHUNK_SIZE; + for (; i + SIMD_WIDTH <= size; i += SIMD_WIDTH) { + __m512 v = load_bfloat16_to_m512(input + i); + store_m512_to_bfloat16_rne(output + i, _mm512_max_ps(v, zero)); + } + + // Scalar tail + for (; i < size; ++i) { + float val = static_cast(input[i]); + output[i] = static_cast(val > 0.0f ? val : 0.0f); + } +} + +// ============================================================================ +// simd_silu: AVX-512 SiLU (Swish) for bfloat16 +// output[i] = input[i] * sigmoid(input[i]) +// ============================================================================ +void simd_silu( + const bf16* input, + bf16* output, + size_t size +) { + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 4; + constexpr size_t CHUNK_SIZE = SIMD_WIDTH * UNROLL_FACTOR; // 64 elements + + constexpr size_t min_elements_per_thread = CHUNK_SIZE * 8; + const bool use_parallel = size >= (min_elements_per_thread * 2); + const size_t num_chunks = size / CHUNK_SIZE; + const int signed_num_chunks = static_cast(num_chunks); + + #pragma omp parallel for num_threads(max_prefill_threads) if(use_parallel) schedule(static) + for (int chunk_idx = 0; chunk_idx < signed_num_chunks; ++chunk_idx) { + size_t i = static_cast(chunk_idx) * CHUNK_SIZE; + + _mm_prefetch(reinterpret_cast(input + i + 64), _MM_HINT_T0); + + __m512 v0 = load_bfloat16_to_m512(input + i); + __m512 v1 = load_bfloat16_to_m512(input + i + 16); + __m512 v2 = load_bfloat16_to_m512(input + i + 32); + __m512 v3 = load_bfloat16_to_m512(input + i + 48); + + store_m512_to_bfloat16_rne(output + i, silu_avx512(v0)); + store_m512_to_bfloat16_rne(output + i + 16, silu_avx512(v1)); + store_m512_to_bfloat16_rne(output + i + 32, silu_avx512(v2)); + store_m512_to_bfloat16_rne(output + i + 48, silu_avx512(v3)); + } + + size_t i = num_chunks * CHUNK_SIZE; + for (; i + SIMD_WIDTH <= size; i += SIMD_WIDTH) { + __m512 v = load_bfloat16_to_m512(input + i); + store_m512_to_bfloat16_rne(output + i, silu_avx512(v)); + } + + // Scalar tail + for (; i < size; ++i) { + float val = static_cast(input[i]); + float sig = 1.0f / (1.0f + std::exp(-val)); + output[i] = static_cast(val * sig); + } +} + +// ============================================================================ +// simd_layernorm: AVX-512 Layer Normalization for bfloat16 +// input shape: [seq_len, D_padded] in row major, only D columns valid +// output shape: [seq_len, D_padded] +// weights: [D] (per-element scale, applied after normalization) +// Formula per row: output = ((input - mean) / sqrt(var + eps)) * weights +// ============================================================================ +void simd_layernorm( + bf16* input, + bf16* output, + const bf16* weights, + int D, + int D_padded, + int seq_len, + float eps +) { + constexpr size_t SIMD_WIDTH = 16; + const float inv_D = 1.0f / static_cast(D); + + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) if(seq_len >= 8) + for (int row = 0; row < seq_len; ++row) { + bf16* row_in = input + static_cast(row) * D_padded; + bf16* row_out = output + static_cast(row) * D_padded; + + // === Pass 1: compute mean === + __m512 sum_acc0 = _mm512_setzero_ps(); + __m512 sum_acc1 = _mm512_setzero_ps(); + size_t i = 0; + + for (; i + SIMD_WIDTH * 2 <= static_cast(D); i += SIMD_WIDTH * 2) { + sum_acc0 = _mm512_add_ps(sum_acc0, load_bfloat16_to_m512(row_in + i)); + sum_acc1 = _mm512_add_ps(sum_acc1, load_bfloat16_to_m512(row_in + i + SIMD_WIDTH)); + } + for (; i + SIMD_WIDTH <= static_cast(D); i += SIMD_WIDTH) { + sum_acc0 = _mm512_add_ps(sum_acc0, load_bfloat16_to_m512(row_in + i)); + } + + float sum = _mm512_reduce_add_ps(_mm512_add_ps(sum_acc0, sum_acc1)); + for (; i < static_cast(D); ++i) { + sum += static_cast(row_in[i]); + } + + float mean = sum * inv_D; + __m512 mean_vec = _mm512_set1_ps(mean); + + // === Pass 2: compute variance === + __m512 var_acc0 = _mm512_setzero_ps(); + __m512 var_acc1 = _mm512_setzero_ps(); + i = 0; + + for (; i + SIMD_WIDTH * 2 <= static_cast(D); i += SIMD_WIDTH * 2) { + __m512 diff0 = _mm512_sub_ps(load_bfloat16_to_m512(row_in + i), mean_vec); + __m512 diff1 = _mm512_sub_ps(load_bfloat16_to_m512(row_in + i + SIMD_WIDTH), mean_vec); + var_acc0 = _mm512_fmadd_ps(diff0, diff0, var_acc0); + var_acc1 = _mm512_fmadd_ps(diff1, diff1, var_acc1); + } + for (; i + SIMD_WIDTH <= static_cast(D); i += SIMD_WIDTH) { + __m512 diff = _mm512_sub_ps(load_bfloat16_to_m512(row_in + i), mean_vec); + var_acc0 = _mm512_fmadd_ps(diff, diff, var_acc0); + } + + float var_sum = _mm512_reduce_add_ps(_mm512_add_ps(var_acc0, var_acc1)); + for (; i < static_cast(D); ++i) { + float diff = static_cast(row_in[i]) - mean; + var_sum += diff * diff; + } + + float variance = var_sum * inv_D; + float inv_std = 1.0f / std::sqrt(variance + eps); + __m512 inv_std_vec = _mm512_set1_ps(inv_std); + + // === Pass 3: normalize and apply weights === + i = 0; + for (; i + SIMD_WIDTH * 2 <= static_cast(D); i += SIMD_WIDTH * 2) { + __m512 x0 = load_bfloat16_to_m512(row_in + i); + __m512 x1 = load_bfloat16_to_m512(row_in + i + SIMD_WIDTH); + __m512 w0 = load_bfloat16_to_m512(weights + i); + __m512 w1 = load_bfloat16_to_m512(weights + i + SIMD_WIDTH); + + __m512 norm0 = _mm512_mul_ps(_mm512_sub_ps(x0, mean_vec), inv_std_vec); + __m512 norm1 = _mm512_mul_ps(_mm512_sub_ps(x1, mean_vec), inv_std_vec); + + store_m512_to_bfloat16_rne(row_out + i, _mm512_mul_ps(norm0, w0)); + store_m512_to_bfloat16_rne(row_out + i + SIMD_WIDTH, _mm512_mul_ps(norm1, w1)); + } + for (; i + SIMD_WIDTH <= static_cast(D); i += SIMD_WIDTH) { + __m512 x = load_bfloat16_to_m512(row_in + i); + __m512 w = load_bfloat16_to_m512(weights + i); + __m512 norm = _mm512_mul_ps(_mm512_sub_ps(x, mean_vec), inv_std_vec); + store_m512_to_bfloat16_rne(row_out + i, _mm512_mul_ps(norm, w)); + } + + // Scalar tail + for (; i < static_cast(D); ++i) { + float val = (static_cast(row_in[i]) - mean) * inv_std; + row_out[i] = static_cast(val * static_cast(weights[i])); + } + } +} + +// ============================================================================ +// scalar_conv2d: Reference scalar 2D convolution for bfloat16 (verification) +// Same interface and semantics as simd_conv2d but purely scalar. +// Accumulates in float32 for precision. +// ============================================================================ +void scalar_conv2d( + const bf16* input, + const bf16* kernel, + bf16* output, + int C_in, int H_in, int W_in, + int C_out, int K, int stride, int padding +) { + const int H_out = (H_in + 2 * padding - K) / stride + 1; + const int W_out = (W_in + 2 * padding - K) / stride + 1; + const int kernel_size = C_in * K * K; + + for (int oc = 0; oc < C_out; ++oc) { + const bf16* oc_kernel = kernel + static_cast(oc) * kernel_size; + bf16* oc_output = output + static_cast(oc) * H_out * W_out; + + for (int oh = 0; oh < H_out; ++oh) { + for (int ow = 0; ow < W_out; ++ow) { + float acc = 0.0f; + int k_idx = 0; + + for (int ic = 0; ic < C_in; ++ic) { + const bf16* ic_input = input + static_cast(ic) * H_in * W_in; + + for (int kh = 0; kh < K; ++kh) { + int ih = oh * stride - padding + kh; + + if (ih < 0 || ih >= H_in) { + k_idx += K; + continue; + } + + const bf16* input_row = ic_input + ih * W_in; + + for (int kw = 0; kw < K; ++kw, ++k_idx) { + int iw = ow * stride - padding + kw; + + if (iw < 0 || iw >= W_in) { + continue; + } + + acc += static_cast(input_row[iw]) * static_cast(oc_kernel[k_idx]); + } + } + } + + oc_output[oh * W_out + ow] = static_cast(acc); + } + } + } +} + +// ============================================================================ +// simd_conv2d: AVX-512 2D convolution for bfloat16 with OpenMP +// input: [C_in, H_in, W_in] in CHW layout +// kernel: [C_out, C_in, K, K] +// output: [C_out, H_out, W_out] +// H_out = (H_in + 2*padding - K) / stride + 1 +// W_out = (W_in + 2*padding - K) / stride + 1 +// +// Strategy: im2col-style accumulation with cache-friendly access patterns. +// - Outer loop over output channels (parallelized with OpenMP) +// - For each output position, accumulate over C_in * K * K with SIMD +// ============================================================================ +void simd_conv2d( + bf16* input, + const bf16* kernel, + bf16* output, + int C_in, int H_in, int W_in, + int C_out, int K, int stride, int padding +) { + const int H_out = (H_in + 2 * padding - K) / stride + 1; + const int W_out = (W_in + 2 * padding - K) / stride + 1; + const int kernel_size = C_in * K * K; // elements per output channel kernel + + constexpr size_t SIMD_WIDTH = 16; + + // Parallelize over output channels — each thread works on independent output channels + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) if(C_out >= max_prefill_threads) + for (int oc = 0; oc < C_out; ++oc) { + const bf16* oc_kernel = kernel + static_cast(oc) * kernel_size; + bf16* oc_output = output + static_cast(oc) * H_out * W_out; + + for (int oh = 0; oh < H_out; ++oh) { + // --- SIMD path: vectorize across output width (ow) --- + // For each (ic, kh, kw), broadcast the kernel weight and accumulate + // over SIMD_WIDTH output positions simultaneously. + int ow = 0; + for (; ow + static_cast(SIMD_WIDTH) <= W_out; ow += static_cast(SIMD_WIDTH)) { + __m512 acc_vec = _mm512_setzero_ps(); + int k_idx = 0; + + for (int ic = 0; ic < C_in; ++ic) { + const bf16* ic_input = input + static_cast(ic) * H_in * W_in; + + for (int kh = 0; kh < K; ++kh) { + int ih = oh * stride - padding + kh; + + if (ih < 0 || ih >= H_in) { + k_idx += K; + continue; + } + + const bf16* input_row = ic_input + ih * W_in; + + for (int kw = 0; kw < K; ++kw, ++k_idx) { + __m512 k_vec = _mm512_set1_ps(static_cast(oc_kernel[k_idx])); + + // Gather SIMD_WIDTH input values for consecutive ow positions + // iw = (ow + lane) * stride - padding + kw + int iw_base = ow * stride - padding + kw; + + if (stride == 1) { + // Contiguous access: load directly from input_row + if (iw_base >= 0 && iw_base + static_cast(SIMD_WIDTH) <= W_in) { + // All elements in bounds — fast path + __m512 in_vec = load_bfloat16_to_m512(input_row + iw_base); + acc_vec = _mm512_fmadd_ps(in_vec, k_vec, acc_vec); + } else { + // Some elements may be out of bounds — zero-padded + alignas(64) float tmp[16] = {0}; + for (int lane = 0; lane < static_cast(SIMD_WIDTH); ++lane) { + int iw = iw_base + lane; + if (iw >= 0 && iw < W_in) { + tmp[lane] = static_cast(input_row[iw]); + } + } + __m512 in_vec = _mm512_load_ps(tmp); + acc_vec = _mm512_fmadd_ps(in_vec, k_vec, acc_vec); + } + } else { + // Strided access — gather element by element + alignas(64) float tmp[16] = {0}; + for (int lane = 0; lane < static_cast(SIMD_WIDTH); ++lane) { + int iw = (ow + lane) * stride - padding + kw; + if (iw >= 0 && iw < W_in) { + tmp[lane] = static_cast(input_row[iw]); + } + } + __m512 in_vec = _mm512_load_ps(tmp); + acc_vec = _mm512_fmadd_ps(in_vec, k_vec, acc_vec); + } + } + } + } + + // Store SIMD_WIDTH output values + store_m512_to_bfloat16_rne(oc_output + oh * W_out + ow, acc_vec); + } + + // --- Scalar tail for remaining ow positions --- + for (; ow < W_out; ++ow) { + float acc_scalar = 0.0f; + int k_idx = 0; + + for (int ic = 0; ic < C_in; ++ic) { + const bf16* ic_input = input + static_cast(ic) * H_in * W_in; + + for (int kh = 0; kh < K; ++kh) { + int ih = oh * stride - padding + kh; + + if (ih < 0 || ih >= H_in) { + k_idx += K; + continue; + } + + const bf16* input_row = ic_input + ih * W_in; + + for (int kw = 0; kw < K; ++kw, ++k_idx) { + int iw = ow * stride - padding + kw; + + if (iw < 0 || iw >= W_in) { + continue; + } + + float in_val = static_cast(input_row[iw]); + float k_val = static_cast(oc_kernel[k_idx]); + acc_scalar += in_val * k_val; + } + } + } + + oc_output[oh * W_out + ow] = static_cast(acc_scalar); + } + } + } +} + +// ============================================================================ +// layernorm_relu_nchw: Applies LayerNorm over C dimension + ReLU activation +// in an NCHW layout. Strides over channels by HW. +// Uses float32 accumulation to match PyTorch's LayerNorm internal precision. +// ============================================================================ +void layernorm_relu_nchw(bf16* data, const bf16* norm_weight, int C, int HW, float eps) { + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) + for(int hw = 0; hw < HW; hw++){ + float mean = 0.0f; + for(int c = 0; c < C; c++){ + mean += static_cast(data[c * HW + hw]); + } + mean /= C; + + float var = 0.0f; + for(int c = 0; c < C; c++){ + float diff = static_cast(data[c * HW + hw]) - mean; + var += diff * diff; + } + var /= C; + + float inv_std = 1.0f / std::sqrt(var + eps); + + for(int c = 0; c < C; c++){ + float val = static_cast(data[c * HW + hw]); + float normed = (val - mean) * inv_std * static_cast(norm_weight[c]); + data[c * HW + hw] = static_cast(std::max(0.0f, normed)); + } + } +} + +// ============================================================================ +// layernorm_gelu_nchw: Applies LayerNorm over C dimension + GELU (tanh approx) +// ============================================================================ +void layernorm_gelu_nchw(bf16* data, const bf16* norm_weight, int C, int HW, float eps) { + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) + for(int hw = 0; hw < HW; hw++){ + double mean = 0.0; + for(int c = 0; c < C; c++){ + mean += static_cast(data[c * HW + hw]); + } + mean /= C; + + double var = 0.0; + for(int c = 0; c < C; c++){ + double diff = static_cast(data[c * HW + hw]) - mean; + var += diff * diff; + } + var /= C; + + float inv_std = 1.0f / std::sqrt(static_cast(var) + eps); + + for(int c = 0; c < C; c++){ + float val = static_cast(data[c * HW + hw]); + float normed = (val - static_cast(mean)) * inv_std * static_cast(norm_weight[c]); + + // GELU tanh approx + float x_cubed = normed * normed * normed; + float inner = 0.7978845608f * (normed + 0.044715f * x_cubed); + float gelu_val = 0.5f * normed * (1.0f + std::tanh(inner)); + + data[c * HW + hw] = static_cast(gelu_val); + } + } +} + +// ============================================================================ +// rmsnorm_gelu_nchw: Applies RMSNorm over C dimension + GELU (tanh approx) +// ============================================================================ +void rmsnorm_gelu_nchw(bf16* data, const bf16* norm_weight, int C, int HW, float eps) { + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) + for(int hw = 0; hw < HW; hw++){ + double sq_sum = 0.0; + for(int c = 0; c < C; c++){ + float val = static_cast(data[c * HW + hw]); + sq_sum += val * val; + } + float var = sq_sum / C; + float inv_std = 1.0f / std::sqrt(var + eps); + + for(int c = 0; c < C; c++){ + float val = static_cast(data[c * HW + hw]); + float normed = val * inv_std * static_cast(norm_weight[c]); + + // GELU tanh approx + float x_cubed = normed * normed * normed; + float inner = 0.7978845608f * (normed + 0.044715f * x_cubed); + float gelu_val = 0.5f * normed * (1.0f + std::tanh(inner)); + + data[c * HW + hw] = static_cast(gelu_val); + } + } +} + +void simd_gemm_abt_bf16( + const bf16* A, const bf16* B, bf16* C, + int M, int N, int K, + int lda, int ldb, int ldc +) { + // C[i][j] = sum_k A[i*lda + k] * B[j*ldb + k] + constexpr size_t SIMD_WIDTH = 16; + + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) if(M * N > 64) + for (int i = 0; i < M; i++) { + const bf16* a_row = A + (size_t)i * lda; + for (int j = 0; j < N; j++) { + const bf16* b_row = B + (size_t)j * ldb; + + __m512 acc0 = _mm512_setzero_ps(); + __m512 acc1 = _mm512_setzero_ps(); + int k = 0; + for (; k + 2 * (int)SIMD_WIDTH <= K; k += 2 * SIMD_WIDTH) { + acc0 = _mm512_fmadd_ps(load_bfloat16_to_m512(a_row + k), + load_bfloat16_to_m512(b_row + k), acc0); + acc1 = _mm512_fmadd_ps(load_bfloat16_to_m512(a_row + k + SIMD_WIDTH), + load_bfloat16_to_m512(b_row + k + SIMD_WIDTH), acc1); + } + __m512 acc = _mm512_add_ps(acc0, acc1); + for (; k + (int)SIMD_WIDTH <= K; k += SIMD_WIDTH) { + acc = _mm512_fmadd_ps(load_bfloat16_to_m512(a_row + k), + load_bfloat16_to_m512(b_row + k), acc); + } + float sum = _mm512_reduce_add_ps(acc); + for (; k < K; k++) { + sum += (float)a_row[k] * (float)b_row[k]; + } + C[(size_t)i * ldc + j] = (bf16)sum; + } + } +} + +// ============================================================================ +// simd_glu: Gated Linear Unit for bfloat16 +// input: [seq_len, 2, hidden_dim] — first half is value, second half is gate +// output: [seq_len, hidden_dim] +// Formula: output[s, :] = input[s, 0, :] * sigmoid(input[s, 1, :]) +// Fused single-pass: load both halves, sigmoid the gate, multiply, store. +// ============================================================================ +void simd_glu( + const bf16* input, + bf16* output, + size_t seq_len, + size_t hidden_dim +) { + constexpr size_t SIMD_WIDTH = 16; + constexpr size_t UNROLL_FACTOR = 4; + constexpr size_t CHUNK_SIZE = SIMD_WIDTH * UNROLL_FACTOR; // 64 elements + const size_t stride = 2 * hidden_dim; // row stride in input + + int signed_seq_len = static_cast(seq_len); + + #pragma omp parallel for num_threads(max_prefill_threads) schedule(static) if(seq_len >= 4) + for (int s = 0; s < signed_seq_len; ++s) { + const bf16* val_ptr = input + (size_t)s * stride; // first half + const bf16* gate_ptr = input + (size_t)s * stride + hidden_dim; // second half + bf16* out_ptr = output + (size_t)s * hidden_dim; + + size_t d = 0; + + // Main loop: 4x unrolled SIMD + for (; d + CHUNK_SIZE <= hidden_dim; d += CHUNK_SIZE) { + _mm_prefetch(reinterpret_cast(val_ptr + d + 64), _MM_HINT_T0); + _mm_prefetch(reinterpret_cast(gate_ptr + d + 64), _MM_HINT_T0); + + __m512 v0 = load_bfloat16_to_m512(val_ptr + d); + __m512 g0 = sigmoid_avx512(load_bfloat16_to_m512(gate_ptr + d)); + + __m512 v1 = load_bfloat16_to_m512(val_ptr + d + 16); + __m512 g1 = sigmoid_avx512(load_bfloat16_to_m512(gate_ptr + d + 16)); + + __m512 v2 = load_bfloat16_to_m512(val_ptr + d + 32); + __m512 g2 = sigmoid_avx512(load_bfloat16_to_m512(gate_ptr + d + 32)); + + __m512 v3 = load_bfloat16_to_m512(val_ptr + d + 48); + __m512 g3 = sigmoid_avx512(load_bfloat16_to_m512(gate_ptr + d + 48)); + + store_m512_to_bfloat16_rne(out_ptr + d, _mm512_mul_ps(v0, g0)); + store_m512_to_bfloat16_rne(out_ptr + d + 16, _mm512_mul_ps(v1, g1)); + store_m512_to_bfloat16_rne(out_ptr + d + 32, _mm512_mul_ps(v2, g2)); + store_m512_to_bfloat16_rne(out_ptr + d + 48, _mm512_mul_ps(v3, g3)); + } + + // Remaining 16-element chunks + for (; d + SIMD_WIDTH <= hidden_dim; d += SIMD_WIDTH) { + __m512 v = load_bfloat16_to_m512(val_ptr + d); + __m512 g = sigmoid_avx512(load_bfloat16_to_m512(gate_ptr + d)); + store_m512_to_bfloat16_rne(out_ptr + d, _mm512_mul_ps(v, g)); + } + + // Scalar tail + for (; d < hidden_dim; ++d) { + float v = static_cast(val_ptr[d]); + float g = static_cast(gate_ptr[d]); + float sig = 1.0f / (1.0f + std::exp(-g)); + out_ptr[d] = static_cast(v * sig); + } + } +} + +void scalar_conv1d( + int conv_kernel_size, + int conv_stride, + const bf16* input, // [seq_len + (conv_kernel_size - conv_stride), hidden_dim] + const bf16* kernel_weights, // [conv_kernel_size, hidden_dim] + bf16* output, // [seq_len, hidden_dim] + int seq_len, + int hidden_dim +){ + for (int o = 0; o < seq_len; o++) { + bf16* out_ptr = output + o * hidden_dim; + for (int d = 0; d < hidden_dim; d++) { + float acc = 0.0f; + for (int k = 0; k < conv_kernel_size; k++) { + int in_idx = (o * conv_stride + k) * hidden_dim + d; + int w_idx = k * hidden_dim + d; + acc += static_cast(input[in_idx]) * static_cast(kernel_weights[w_idx]); + } + out_ptr[d] = static_cast(acc); + } + } +} diff --git a/src/detail/gemma4e_npu/gemma4e_vision_prefill_helper.hpp b/src/detail/gemma4e_npu/gemma4e_vision_prefill_helper.hpp new file mode 100644 index 000000000..29f441f4f --- /dev/null +++ b/src/detail/gemma4e_npu/gemma4e_vision_prefill_helper.hpp @@ -0,0 +1,173 @@ +#include +#include +#include +#include +#include +#include +#include "typedef.hpp" +#pragma once + +void simd_conv2d( + bf16* input, + const bf16* kernel, + bf16* output, + int C_in, int H_in, int W_in, + int C_out, int K, int stride, int padding +); +void scalar_conv2d( + const bf16* input, + const bf16* kernel, + bf16* output, + int C_in, int H_in, int W_in, + int C_out, int K, int stride, int padding +); + +void simd_layernorm( + bf16* input, // shape of [seq_len, D_padded] in row major ordr, but only D is valud + bf16* output, + const bf16* weights, + int D, + int D_padded, + int seq_len, + float eps = 1e-6f +); + +// NCHW channel-wise operations (acting across C dimension, strided by HW) +void layernorm_relu_nchw(bf16* data, const bf16* norm_weight, int C, int HW, float eps = 1e-6f); +void layernorm_gelu_nchw(bf16* data, const bf16* norm_weight, int C, int HW, float eps = 1e-6f); +void rmsnorm_gelu_nchw(bf16* data, const bf16* norm_weight, int C, int HW, float eps = 1e-6f); + +void simd_relu( + const bf16* input, + bf16* output, + size_t size +); + +void simd_silu( + const bf16* input, + bf16* output, + size_t size +); + +void simd_add( + bf16* input1, + bf16* input2, + bf16* output, + size_t size +); + +// Optimized AVX-512 version that combines bias addition and GELU activation +// bias is broadcast across sequence positions (size = hidden_dim, not total_size) +void simd_bias_add_gelu( + bf16* input, + const bf16* bias, + bf16* output, + size_t total_size, + size_t hidden_dim +); + +void gelu_bfloat16_ref( + const bf16* input, + bf16* output, + size_t size +); + +// GELU activation function with tanh approximation (same as PytorchGELUTanh) +inline float gelu_tanh(float x) { + return 0.5f * x * (1 + std::tanh(std::sqrt(2.0f / 3) * (x + 0.044715f * x * x * x))); +} + +// RMS Norm without scale weights +// Input layout: [seq_len_padded x X_padded], only processes seq_len rows and X cols per row +// Formula: output = input * rsqrt(mean(input^2) + eps) +void simd_rms_norm( + const bf16* input, + bf16* output, + size_t seq_len, + size_t X, + size_t seq_len_padded, + size_t X_padded, + float eps = 1e-6f +); + +// RMS Norm with scale weights (Gemma4RMSNorm with with_scale=True) +// Input layout: [seq_len_padded x X_padded], only processes seq_len rows and X cols per row +// norm_weight: bf16 pointer of size [X] +// Formula: output = (input * rsqrt(mean(input^2) + eps)) * weight +void simd_rms_norm( + const bf16* input, + const bf16* norm_weight, + bf16* output, + size_t seq_len, + size_t X, + size_t seq_len_padded, + size_t X_padded, + float eps = 1e-6f +); + +void simd_clamp( + const bf16* input, + bf16* output, + bf16 min_val, + bf16 max_val, + size_t size +); + +void simd_mul( + const bf16* input1, + const bf16* input2, + bf16* output, + size_t size +); + +void simd_mul( + const bf16* input1, + bf16 input2_scalar, + bf16* output, + size_t size +); + +void transpose_2d( + const bf16* input, + bf16* output, + size_t rows, + size_t cols +); + +// General bf16 matmul: C = A @ B^T with float32 accumulation +// A: [M, K] row-major with stride lda +// B: [N, K] row-major with stride ldb (B^T gives [K, N]) +// C: [M, N] row-major with stride ldc +void simd_gemm_abt_bf16( + const bf16* A, const bf16* B, bf16* C, + int M, int N, int K, + int lda, int ldb, int ldc +); + +void simd_glu( + const bf16* input, //[seq_len, 2,hidden_dim] + bf16* output, // [seq_len, hidden_dim] + size_t seq_len, + size_t hidden_dim +); + +void conv1d( + int conv_kernel_size, + int conv_stride, + const bf16* input, /// [seq_len+ (conv_kernel_size-conv_stride), hidden_dim] + const bf16* kernel_weights, // [conv_kernel_size, hidden_dim] + bf16* output, // [seq_len, hidden_dim] + int seq_len, + int hidden_dim + +); + +void scalar_conv1d( + int conv_kernel_size, + int conv_stride, + const bf16* input, /// [seq_len+ (conv_kernel_size-conv_stride), hidden_dim] + const bf16* kernel_weights, // [conv_kernel_size, hidden_dim] + bf16* output, // [seq_len, hidden_dim] + int seq_len, + int hidden_dim +); diff --git a/src/detail/gemma4e_npu/mmRuntimeSequence.hpp b/src/detail/gemma4e_npu/mmRuntimeSequence.hpp new file mode 100644 index 000000000..8be70830c --- /dev/null +++ b/src/detail/gemma4e_npu/mmRuntimeSequence.hpp @@ -0,0 +1,550 @@ +#ifndef __MM_SEQUENCE_HPP__ +#define __MM_SEQUENCE_HPP__ +#include +#include // Required for std::max +#include +#include "npu_utils/npu_instr_utils.hpp" + +template +void generate_shimtile_sequence_per_k_block( + uint32_t shim_index, uint32_t total_npu_row, uint32_t total_npu_col, + uint32_t mega_block_row_idx, uint32_t mega_block_col_idx, + uint32_t M_size, uint32_t K_size, uint32_t N_size, + uint32_t m, uint32_t k, uint32_t n, + uint32_t Arg_A, uint32_t Arg_B, uint32_t Arg_C, + uint32_t A_const_offset, uint32_t B_const_offset, uint32_t C_const_offset, + + std::vector &list_A_shim_queue, std::vector &list_B_shim_queue, std::vector &list_C_shim_queue, + std::vector &list_A_bd_pingpong_flag, std::vector &list_B_bd_pingpong_flag, std::vector &list_C_bd_pingpong_flag, + + bool IS_B_ROW_MAJOR, bool ENABLE_AXI4, + bool B_in_K_N_block_col_major_order, + bool VALID_COLUMN, + bool ADD_BIAS, bool SEND_BIAS, + std::map& valid_A_MT_shimtile_index, + npu_sequence& seq, + std::vector & shimtile_list, + + bool REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH, + uint32_t DH + +){ + + if(B_in_K_N_block_col_major_order){ + assert(IS_B_ROW_MAJOR== false); // on valid for B in col major order + } + // When B_in_K_N_block_col_major_order is set to true, it mean + // B is col-major order && + // B is rearrange into kxn blocks, where blocks are in col-major. Moreover, the data in each blocks is + // also in col-major order. + + // Basically,B as a col-major matrix goes through + // stride: [N_size/n,K_size/k ,n, k] + // offset: [K_size*n,k ,K_SIZE, 1] + + auto AXI_FLAG = aggressive_cache; + ; + uint32_t K_div_k = K_size/k; + + npu_tiles cur_shimtile = shimtile_list.at(shim_index); + + if (valid_A_MT_shimtile_index.contains(shim_index) && valid_A_MT_shimtile_index[shim_index] < total_npu_row){ + + if (list_A_shim_queue.at(shim_index) == 2) { + seq.npu_dma_wait( + cur_shimtile, MM2S, it_channel_0 + ); + list_A_shim_queue.at(shim_index)--; + } + + uint32_t A_offset = mega_block_row_idx *(total_npu_row*m) *K_size; + A_offset += valid_A_MT_shimtile_index[shim_index]*(m*K_size); + npu_bd_id A_bd_id; + if (list_A_bd_pingpong_flag.at(shim_index) ==0){ + A_bd_id = bd_0; + list_A_bd_pingpong_flag.at(shim_index) =1; + }else{ + A_bd_id = bd_1; + list_A_bd_pingpong_flag.at(shim_index) =0; + } + + seq.npu_dma_memcpy_nd( + sizeof(T_in), // bfloat16 + Arg_A, + MM2S, + cur_shimtile, + A_bd_id, + it_channel_0, + {0,0,0,A_offset+ A_const_offset}, + {1, K_div_k, m,k}, + {0, k, K_size, 1}, + -1, 0, true, + // ENABLE_AXI4 ? AXI_FLAG: normal_cache + aggressive_cache + ); + list_A_shim_queue.at(shim_index)++; + } + + if(shim_index < total_npu_col && VALID_COLUMN){ + + if(SEND_BIAS){ + if (list_B_shim_queue[shim_index] == 2){ + + seq.npu_dma_wait( + cur_shimtile, MM2S, it_channel_1 + ); + list_B_shim_queue[shim_index] -= 1; + } + uint32_t _BIAS_DATA_OFFSET = mega_block_col_idx * (total_npu_col*n) + shim_index *n; + seq.npu_dma_memcpy_nd( + sizeof(T_in), + Arg_B, + MM2S, + cur_shimtile, + npu_bd_id(bd_6), //reserved for sending bias + it_channel_1, + {0,0,0,_BIAS_DATA_OFFSET}, + {1, 1,1, k*n}, + {0, 0, 0, 1}, + -1, 0, true, + // ENABLE_AXI4 ? AXI_FLAG: normal_cache + aggressive_cache + ); + list_B_shim_queue[shim_index]++; + } + + uint32_t BIAS_OFFSET = 0; + if (ADD_BIAS){ + BIAS_OFFSET = N_size; + } + + npu_bd_id b_bd_id; + if (list_B_shim_queue[shim_index] == 2){ + + seq.npu_dma_wait( + cur_shimtile, MM2S, it_channel_1 + ); + list_B_shim_queue[shim_index] -= 1; + } + + if (list_B_bd_pingpong_flag[shim_index] == 0) { + b_bd_id = bd_2; + list_B_bd_pingpong_flag[shim_index] = 1; + } else { + b_bd_id = bd_3; + list_B_bd_pingpong_flag[shim_index] = 0; + } + + if (IS_B_ROW_MAJOR){ + uint32_t B_offset = mega_block_col_idx* (total_npu_col) * n; + B_offset += shim_index * n; + seq.npu_dma_memcpy_nd( + sizeof(T_in), + Arg_B, + MM2S, + cur_shimtile, + b_bd_id, + it_channel_1, + {0,0,0,B_offset+ B_const_offset + BIAS_OFFSET}, + {1, K_div_k, k, n}, + {0, k*N_size, N_size, 1}, + -1, 0, true, + // ENABLE_AXI4 ? AXI_FLAG: normal_cache + aggressive_cache + ); + }else{ + uint32_t B_offset = mega_block_col_idx*(total_npu_col*n)*K_size; + B_offset += shim_index * n*K_size; + if(B_in_K_N_block_col_major_order){ + seq.npu_dma_memcpy_nd( + sizeof(T_in), + Arg_B, + MM2S, + cur_shimtile, + b_bd_id, + it_channel_1, + {0,0,0,B_offset+ B_const_offset + BIAS_OFFSET}, + {1, 1,1, K_div_k* n*k}, + {0, 0, 0, 1}, + -1, 0, true, + // ENABLE_AXI4 ? AXI_FLAG: normal_cache + aggressive_cache + ); + }else{ + seq.npu_dma_memcpy_nd( + sizeof(T_in), + Arg_B, + MM2S, + cur_shimtile, + b_bd_id, + it_channel_1, + {0,0,0,B_offset+ B_const_offset + BIAS_OFFSET}, + {1, K_div_k, n, k}, + {0, k, K_size, 1}, + -1, 0, true, + // ENABLE_AXI4 ? AXI_FLAG: normal_cache + aggressive_cache + ); + } + } + list_B_shim_queue[shim_index]++; + } + + if (shim_index < total_npu_col && VALID_COLUMN){ + + if (list_C_shim_queue.at(shim_index) == 2) { + seq.npu_dma_wait( + cur_shimtile, S2MM, it_channel_0 + ); + list_C_shim_queue.at(shim_index)--; + } + + npu_bd_id c_bd_id; + if(list_C_bd_pingpong_flag.at(shim_index) ==0){ + c_bd_id = bd_14; + list_C_bd_pingpong_flag.at(shim_index) =1; + }else{ + c_bd_id = bd_15; + list_C_bd_pingpong_flag.at(shim_index) =0; + } + + if(REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH){ + // Reorder from [M, N] row-major to [N/DH, M, DH] row-major + // where M = L_Seq and N = 3*NUM_HEADS*DH + /// debug + //std::cerr << "DH is " << DH << std::endl; + if( N_size % DH != 0 ){ + std::cerr << "MM: N_size % DH != 0 " << std::endl; + exit(-1); + } + + if( (total_npu_col * n)%DH != 0){ // for now + std::cerr << "MM: (total_npu_col * n)%DH != 0" < +void generate_runtime_sequence( + uint32_t Arg_A, uint32_t Arg_B, uint32_t Arg_C, + uint32_t A_const_offset, uint32_t B_const_offset, uint32_t C_const_offset, + uint32_t M_size, uint32_t N_size, uint32_t K_size, + uint32_t m, uint32_t n, uint32_t k, + uint32_t total_npu_row, uint32_t total_npu_col, + std::vector &list_A_shim_queue, + std::vector &list_B_shim_queue, + std::vector &list_C_shim_queue, + + bool IS_B_ROW_MAJOR, bool ENABLE_AXI4, bool B_in_K_N_block_col_major_order, + bool ADD_BIAS, + std::map& valid_A_MT_shimtile_index, + npu_sequence& seq, + std::vector &shim_tiles, + bool REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH, + uint32_t DH +){ + + uint32_t M_div_num_row_m = M_size/(m*total_npu_row); + uint32_t N_div_num_col_n = N_size/(n*total_npu_col); + + uint32_t N_div_num_col_n_remainder_blocks = (N_size % (n*total_npu_col))/ n; + + std::vector list_A_BD_pingpong_flag(std::max(total_npu_row, total_npu_col), 0); + std::vector list_B_BD_pingpong_flag(std::max(total_npu_row, total_npu_col), 0); + std::vector list_C_BD_pingpong_flag(std::max(total_npu_row, total_npu_col), 0); + + uint32_t col_block_range = N_div_num_col_n; + if (N_div_num_col_n_remainder_blocks!= 0){ + col_block_range += 1; + } + + for(uint32_t mega_block_col_idx=0; mega_block_col_idx( + shim_index, + total_npu_row, total_npu_col, + mega_block_row_idx, mega_block_col_idx, + M_size, K_size, N_size, + m, k, n, + Arg_A, Arg_B, Arg_C, + A_const_offset, B_const_offset, C_const_offset, + list_A_shim_queue, list_B_shim_queue, list_C_shim_queue, + list_A_BD_pingpong_flag, list_B_BD_pingpong_flag, list_C_BD_pingpong_flag, + IS_B_ROW_MAJOR, ENABLE_AXI4, + B_in_K_N_block_col_major_order, + true, + ADD_BIAS, SEND_ADD_BIAS, + valid_A_MT_shimtile_index, + seq, shim_tiles, + REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH, + DH + + ); + } + + else{ + generate_shimtile_sequence_per_k_block( + shim_index, + total_npu_row, total_npu_col, + mega_block_row_idx, mega_block_col_idx, + M_size, K_size, N_size, + m, k, n, + Arg_A, Arg_B, Arg_C, + A_const_offset, B_const_offset, C_const_offset, + list_A_shim_queue, list_B_shim_queue, list_C_shim_queue, + list_A_BD_pingpong_flag, list_B_BD_pingpong_flag, list_C_BD_pingpong_flag, + IS_B_ROW_MAJOR, ENABLE_AXI4, + B_in_K_N_block_col_major_order, + false, + ADD_BIAS, SEND_ADD_BIAS, + valid_A_MT_shimtile_index, + seq, shim_tiles, + REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH, + DH + ); + } + } + else{ + generate_shimtile_sequence_per_k_block( + shim_index, + total_npu_row, total_npu_col, + mega_block_row_idx, mega_block_col_idx, + M_size, K_size, N_size, + m, k, n, + Arg_A, Arg_B, Arg_C, + A_const_offset, B_const_offset, C_const_offset, + list_A_shim_queue, list_B_shim_queue, list_C_shim_queue, + list_A_BD_pingpong_flag, list_B_BD_pingpong_flag, list_C_BD_pingpong_flag, + IS_B_ROW_MAJOR, ENABLE_AXI4, + B_in_K_N_block_col_major_order, + true, + ADD_BIAS, SEND_ADD_BIAS, + valid_A_MT_shimtile_index, + seq, shim_tiles, + REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH, + DH + ); + } + } + } + } +} + +template +void generate_mm_sequence(npu_sequence &seq, uint32_t M, uint32_t K, uint32_t N, + uint32_t m, uint32_t k, uint32_t n, + uint32_t r, uint32_t s, uint32_t t, + uint32_t CT_rtp_address, uint32_t CT_rtp_sync_lock_id, + uint32_t total_row, uint32_t total_col, + uint32_t A_const_offset, uint32_t B_const_offset, uint32_t C_const_offset, + bool IS_B_ROW_MAJOR, bool ENABLE_AXI4, bool B_in_K_N_block_col_major_order, + bool ADD_BIAS, int OUTPUT_MODE, + int OUTPUT_CLAMP, + float output_clamp_min, float output_clamp_max, + bool REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH, // if false, output is in MXN row major + // if true, output is in NUM_DH x M x DH row major + uint32_t DH +){ + + constexpr int CT_lock_address_base = 0x000001F000; + const int Arg_A = 0; + const int Arg_B = 1; + const int Arg_C = 2; + + const int K_div_k = K/k; + + if( M%(m*total_row) != 0){ + std::cerr << "Error: M size not multiple of m * total_row"<< std::endl; + exit(1); + } + if( K%k != 0){ + std::cerr << "Error: K size not multiple of k"<< std::endl; + exit(1); + } + if( N%n != 0){ + std::cerr << "Error: N size not multiple of n"<< std::endl; + exit(1); + } + if(total_row != 4){ + std::cerr << "Error: total_row greater than 4 not supported"<< std::endl; + exit(1); + } + if(total_col !=8){ + std::cerr << "Error: total_col greater than 8 not supported"<< std::endl; + exit(1); + } + if(REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH){ + if( (N % DH) !=0 || DH%n !=0 || DH index offset + std::map valid_A_MT_shimtile_index; + valid_A_MT_shimtile_index[0] = 0; + valid_A_MT_shimtile_index[2] = 1; + valid_A_MT_shimtile_index[4] = 2; + valid_A_MT_shimtile_index[6] = 3; + + //create list of tiles + uint32_t shimtile_size = std::max(total_col, total_row); + std::vector shim_tiles; + + std::vector list_C_shim_queue; // int counter of how many DMA_Wait for C + std::vector list_A_shim_queue; // int counter of how many DMA_Wait for A + std::vector list_B_shim_queue; // int counter of how many DMA_Wait for B + for(size_t i = 0; i < shimtile_size; i++){ + shim_tiles.push_back((get_tile(0, i))); + list_C_shim_queue.push_back(0); + list_A_shim_queue.push_back(0); + list_B_shim_queue.push_back(0); + } + + // first, setup the rtp buffer and the rtp locks + + for(size_t row_idx = 0; row_idx < total_row; row_idx++){ + for(size_t col_idx = 0; col_idx< total_col; col_idx++){ + auto CT_tile = get_tile(row_idx+2, col_idx); + // set RTP value + seq.rtp_write( CT_tile, CT_rtp_address, K_div_k ); + seq.rtp_write( CT_tile, CT_rtp_address+4, M ); + seq.rtp_write( CT_tile, CT_rtp_address+8, N ); + if(ADD_BIAS){ + seq.rtp_write( CT_tile, CT_rtp_address+12, 1 ); + }else{ + seq.rtp_write( CT_tile, CT_rtp_address+12, 0 ); + } + seq.rtp_write( CT_tile, CT_rtp_address+16, OUTPUT_MODE ); // OUTPUT MODE + seq.rtp_write( CT_tile, CT_rtp_address+20, OUTPUT_CLAMP); // OUTPUT CLAMP 0 means no clamp, 1 means clamp + + int32_t output_min_int, output_max_int; + std::memcpy(&output_min_int, &output_clamp_min, sizeof(int32_t)); + std::memcpy(&output_max_int, &output_clamp_max, sizeof(int32_t)); + + seq.rtp_write( CT_tile, CT_rtp_address+24, output_min_int ); // output clamp min value in float32 + seq.rtp_write( CT_tile, CT_rtp_address+28, output_max_int ); // output clamp max value in float32 + // set RTP lock + seq.rtp_write(CT_tile, CT_lock_address_base+16*(CT_rtp_sync_lock_id), 1); // set lock to 1 + } + } + + generate_runtime_sequence( + Arg_A, Arg_B, Arg_C, + A_const_offset, B_const_offset, C_const_offset, + M, N, K, m,n,k, + total_row, total_col, + list_A_shim_queue, list_B_shim_queue, + list_C_shim_queue, + IS_B_ROW_MAJOR, ENABLE_AXI4, B_in_K_N_block_col_major_order, + ADD_BIAS, + valid_A_MT_shimtile_index, seq, shim_tiles, + REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH, + DH + ); + + int max_C_remain = 0; + for( auto li: list_C_shim_queue){ + + max_C_remain = std::max(max_C_remain, li); + } + + for(size_t k = 0; k< max_C_remain; k++){ + for(size_t shim_index = 0; shim_index < total_col; shim_index++){ + + if(list_A_shim_queue.at(shim_index) > 0){ + seq.npu_dma_wait( + shim_tiles.at(shim_index), + MM2S, + it_channel_0 + ); + list_A_shim_queue.at(shim_index)--; + } + if(list_B_shim_queue.at(shim_index) > 0){ + seq.npu_dma_wait( + shim_tiles.at(shim_index), + MM2S, + it_channel_1 + ); + list_B_shim_queue.at(shim_index)--; + } + if(list_C_shim_queue.at(shim_index) > 0){ + seq.npu_dma_wait( + shim_tiles.at(shim_index), + S2MM, + it_channel_0 + + ); + list_C_shim_queue.at(shim_index)--; + } + } + } + + seq.cmds2seq(); +} + +#endif diff --git a/src/detail/gemma4e_npu/reorder_cpy.hpp b/src/detail/gemma4e_npu/reorder_cpy.hpp new file mode 100644 index 000000000..f1456bedb --- /dev/null +++ b/src/detail/gemma4e_npu/reorder_cpy.hpp @@ -0,0 +1,37 @@ +#pragma once +#include "typedef.hpp" + +inline void reorder_cpy(u8 *dst, buffer &src, const int col, const int vertical_blocks = 2) +{ + const int a_block_size = 32 * 256 * 5 / 8; + const int blocks_per_row = col / 256; + + const int rows = src.size() / a_block_size / blocks_per_row; + + u8 *dst_ptr = dst; + std::vector src_ptr(vertical_blocks); + for (int i = 0; i < vertical_blocks; i++) + { + src_ptr[i] = src.data() + i * a_block_size * blocks_per_row; + } + for (int r = 0; r < rows; r += vertical_blocks) + { + for (int c = 0; c < blocks_per_row; c++) + { + for (int i = 0; i < vertical_blocks; i++) + { + memcpy(dst_ptr, src_ptr[i], a_block_size); + dst_ptr += a_block_size; + src_ptr[i] += a_block_size; + } + } + for (int i = 0; i < vertical_blocks; i++) + { + src_ptr[i] += (vertical_blocks - 1) * a_block_size * blocks_per_row; + if (src_ptr[i] + a_block_size * blocks_per_row > src.end()) + { + src_ptr[i] = src.data(); // useless padding + } + } + } +} diff --git a/src/detail/gemma4e_npu/rot_pos_emb.cpp b/src/detail/gemma4e_npu/rot_pos_emb.cpp new file mode 100644 index 000000000..4045b08ae --- /dev/null +++ b/src/detail/gemma4e_npu/rot_pos_emb.cpp @@ -0,0 +1,219 @@ +#include "rot_pos_emb.hpp" +#include +#include +#include +#include +#include +#include +#include +#include "avx512_util.hpp" +#include +#include + +void generate_gemma4_audio_rotary_pos_emb( + int hidden_size, + int attention_chunk_size, + int attention_context_left, + int attention_context_right, + + std::vector& position_embedding // shape of [13, hidden_size] +){ + + int context_size = attention_chunk_size + (attention_context_left-1) + attention_context_right; + + float min_timescale = 1.0; + float max_timescale = 10000.0; + + int num_timescales = hidden_size / 2; + float log_timescale_increment = std::log( + + max_timescale/min_timescale + + )/ std::max( num_timescales - 1, 1) ; + + std::vector inv_timescales(num_timescales); + + for(int i = 0; i < num_timescales; i++){ + inv_timescales[i] = min_timescale * std::exp(i * -log_timescale_increment); + } + + for(int i = 0; i <= 12; i++){ + int pos_val = 12 - i; + + for(int j = 0; j < num_timescales; j++){ + float angle = pos_val * inv_timescales[j]; + + position_embedding[i *hidden_size + j] = bf16(std::sin(angle)); + position_embedding[i *hidden_size + j + num_timescales] = bf16(std::cos(angle)); + } + } +} + +/** + * Apply rotary position embeddings to Q and K in-place within mm_res buffer. + * New Layout: [3 * num_heads, seq_len, head_dim] + * * @param mm_res_ptr Pointer to buffer of shape [3 * num_heads, seq_len, head_dim] + * Structure: [Block Q (Heads 0..H-1)] | [Block K (Heads 0..H-1)] | [Block V...] + * @param cos_emb_ptr Pointer to cosine embeddings of shape [seq_len, head_dim] (float) + * @param sin_emb_ptr Pointer to sine embeddings of shape [seq_len, head_dim] (float) + * @param seq_len The L_Seq dimension size + * @param hidden_size Hidden dimension size (total across all heads) + * @param num_heads Number of attention heads + */ +void generate_gemma4_vision_rotary_pos_emb( + const std::vector>& grid_pairs_per_image, + const std::vector& seq_len_per_image, + const std::vector& start_seq_len_index_per_image, + int seq_len_padded, + int head_dim, + float theta, + float scale, + std::vector& cos_emb, + std::vector& sin_emb +) { + const int spatial_dim = head_dim / 2; // 32 for head_dim=64 + const int inv_freq_len = spatial_dim / 2; // 16 + + // inv_freq[j] = 1.0 / (theta ^ (j*2 / spatial_dim)) + std::vector inv_freq(inv_freq_len); + for (int j = 0; j < inv_freq_len; j++) { + inv_freq[j] = 1.0f / std::pow(theta, (float)(j * 2) / (float)spatial_dim); + } + + cos_emb.assign(seq_len_padded * head_dim, bf16(0.0f)); + sin_emb.assign(seq_len_padded * head_dim, bf16(0.0f)); + + for (int img = 0; img < (int)grid_pairs_per_image.size(); img++) { + const int num_patches = seq_len_per_image[img]; + const int start = start_seq_len_index_per_image[img]; + const auto& grid_pairs = grid_pairs_per_image[img]; + int compact_patch_idx = 0; + + for (int pair_idx = 0; pair_idx + 1 < (int)grid_pairs.size() && compact_patch_idx < num_patches; pair_idx += 2) { + const int x_val = grid_pairs[pair_idx]; + const int y_val = grid_pairs[pair_idx + 1]; + + // Gemma4 position ids are padded with (-1, -1). The encoder state is compacted to + // valid patches only, so rotary embeddings must compact the valid coordinates too. + if (x_val < 0 || y_val < 0) { + continue; + } + + const int global_s = start + compact_patch_idx; + + bf16* cos_row = cos_emb.data() + global_s * head_dim; + bf16* sin_row = sin_emb.data() + global_s * head_dim; + + for (int j = 0; j < inv_freq_len; j++) { + const float x_freq = x_val * inv_freq[j]; + const float y_freq = y_val * inv_freq[j]; + + // x-axis: channels [0..inv_freq_len-1] and [inv_freq_len..spatial_dim-1] (duplication) + cos_row[j] = bf16(std::cos(x_freq) * scale); + cos_row[j + inv_freq_len]= bf16(std::cos(x_freq) * scale); + sin_row[j] = bf16(std::sin(x_freq) * scale); + sin_row[j + inv_freq_len]= bf16(std::sin(x_freq) * scale); + + // y-axis: channels [spatial_dim..spatial_dim+inv_freq_len-1] and [spatial_dim+inv_freq_len..head_dim-1] + cos_row[spatial_dim + j] = bf16(std::cos(y_freq) * scale); + cos_row[spatial_dim + j + inv_freq_len]= bf16(std::cos(y_freq) * scale); + sin_row[spatial_dim + j] = bf16(std::sin(y_freq) * scale); + sin_row[spatial_dim + j + inv_freq_len]= bf16(std::sin(y_freq) * scale); + } + + compact_patch_idx++; + } + + assert(compact_patch_idx == num_patches); + } +} + +void apply_multidimensional_rope( + bf16* qkv, + const bf16* cos_emb, + const bf16* sin_emb, + int seq_len, + int num_head, + int head_dim, + int ndim +) { + // Python: num_rotated_channels_per_dim = 2 * (head_dim // (2 * ndim)) + const int spatial_dim = 2 * (head_dim / (2 * ndim)); // channels per spatial dim (32 for head_dim=64, ndim=2) + const int quarter_dim = spatial_dim / 2; // rotate_half midpoint (16 for spatial_dim=32) + +#ifdef __AVX512F__ + // Fast path: AVX-512, processes 16 bf16 values per register. + // Requires quarter_dim == 16 (head_dim == 64). + if (quarter_dim == 16) { + for (int s = 0; s < seq_len; s++) { + const bf16* cos_row = cos_emb + s * head_dim; + const bf16* sin_row = sin_emb + s * head_dim; + + // Load cos/sin for x-spatial dim (cos[0..15], duplicated at [16..31]) + const __m512 cos_x = load_bfloat16_to_m512(cos_row); + const __m512 sin_x = load_bfloat16_to_m512(sin_row); + // Load cos/sin for y-spatial dim (cos[32..47], duplicated at [48..63]) + const __m512 cos_y = load_bfloat16_to_m512(cos_row + spatial_dim); + const __m512 sin_y = load_bfloat16_to_m512(sin_row + spatial_dim); + + bf16* row = qkv + s * num_head * head_dim; + for (int h = 0; h < num_head; h++) { + bf16* x = row + h * head_dim; + + // --- x-spatial half: channels [0..spatial_dim-1] --- + const __m512 x_lo = load_bfloat16_to_m512(x); + const __m512 x_hi = load_bfloat16_to_m512(x + quarter_dim); + // out_lo = x_lo * cos_x - x_hi * sin_x + const __m512 out_lo = _mm512_fmsub_ps(x_lo, cos_x, _mm512_mul_ps(x_hi, sin_x)); + // out_hi = x_hi * cos_x + x_lo * sin_x + const __m512 out_hi = _mm512_fmadd_ps(x_hi, cos_x, _mm512_mul_ps(x_lo, sin_x)); + store_m512_to_bfloat16_rne(x, out_lo); + store_m512_to_bfloat16_rne(x + quarter_dim, out_hi); + + // --- y-spatial half: channels [spatial_dim..head_dim-1] --- + bf16* y = x + spatial_dim; + const __m512 y_lo = load_bfloat16_to_m512(y); + const __m512 y_hi = load_bfloat16_to_m512(y + quarter_dim); + // out_lo = y_lo * cos_y - y_hi * sin_y + const __m512 y_out_lo = _mm512_fmsub_ps(y_lo, cos_y, _mm512_mul_ps(y_hi, sin_y)); + // out_hi = y_hi * cos_y + y_lo * sin_y + const __m512 y_out_hi = _mm512_fmadd_ps(y_hi, cos_y, _mm512_mul_ps(y_lo, sin_y)); + store_m512_to_bfloat16_rne(y, y_out_lo); + store_m512_to_bfloat16_rne(y + quarter_dim, y_out_hi); + } + } + return; + } +#endif // __AVX512F__ + + // Scalar fallback: works for any head_dim divisible by 4. + for (int s = 0; s < seq_len; s++) { + const bf16* cos_row = cos_emb + s * head_dim; + const bf16* sin_row = sin_emb + s * head_dim; + + bf16* row = qkv + s * num_head * head_dim; + for (int h = 0; h < num_head; h++) { + bf16* x = row + h * head_dim; + + // x-spatial half [0..spatial_dim-1] + for (int j = 0; j < quarter_dim; j++) { + const float x_lo = float(x[j]); + const float x_hi = float(x[j + quarter_dim]); + const float c = float(cos_row[j]); + const float s_ = float(sin_row[j]); + x[j] = bf16(x_lo * c - x_hi * s_); + x[j + quarter_dim]= bf16(x_hi * c + x_lo * s_); + } + + // y-spatial half [spatial_dim..head_dim-1] + for (int j = 0; j < quarter_dim; j++) { + const float y_lo = float(x[spatial_dim + j]); + const float y_hi = float(x[spatial_dim + j + quarter_dim]); + const float c = float(cos_row[spatial_dim + j]); + const float s_ = float(sin_row[spatial_dim + j]); + x[spatial_dim + j] = bf16(y_lo * c - y_hi * s_); + x[spatial_dim + j + quarter_dim]= bf16(y_hi * c + y_lo * s_); + } + } + } +} diff --git a/src/detail/gemma4e_npu/rot_pos_emb.hpp b/src/detail/gemma4e_npu/rot_pos_emb.hpp new file mode 100644 index 000000000..7acc4c69d --- /dev/null +++ b/src/detail/gemma4e_npu/rot_pos_emb.hpp @@ -0,0 +1,82 @@ +#pragma once +#include +#include +#include "typedef.hpp" +#include +#include + +void generate_gemma4_audio_rotary_pos_emb( + int hidden_size, + int attention_chunk_size, + int attention_context_left, + int attention_context_right, + + std::vector& position_embedding // shape of [13, hidden_size] +); + +/** + * Generate Gemma4 vision rotary position embeddings (cos and sin). + * + * Python equivalent: Gemma4VisionRotaryEmbedding.forward(hidden_states, pixel_position_ids) + * + * For each patch s with (x_val, y_val): + * spatial_dim = head_dim / 2 + * inv_freq[j] = 1 / (theta ^ (j*2 / spatial_dim)) for j in [0, spatial_dim/2) + * emb_x[j] = x_val * inv_freq[j] (duplicated: positions [0..spatial_dim-1]) + * emb_y[j] = y_val * inv_freq[j] (duplicated: positions [spatial_dim..head_dim-1]) + * cos_row = [cos(emb_x)*scale, cos(emb_y)*scale] of length head_dim + * + * Output shape: [seq_len_padded, head_dim] (padded rows are zero) + * + * @param grid_pairs_per_image per-image flat (x,y) pairs [img][s*2] + * @param seq_len_per_image unpadded patch count per image + * @param start_seq_len_index_per_image cumulative start offset per image + * @param seq_len_padded padded total sequence length + * @param head_dim GEMMA4E_VISION_HEAD_DIM (e.g. 64) + * @param theta GEMMA4E_ROPE_THETA (e.g. 100.0) + * @param scale attention_scaling (typically 1.0) + * @param cos_emb output [seq_len_padded * head_dim] (resized, bf16) + * @param sin_emb output [seq_len_padded * head_dim] (resized, bf16) + */ +void generate_gemma4_vision_rotary_pos_emb( + const std::vector>& grid_pairs_per_image, + const std::vector& seq_len_per_image, + const std::vector& start_seq_len_index_per_image, + int seq_len_padded, + int head_dim, + float theta, + float scale, + std::vector& cos_emb, + std::vector& sin_emb +); + +/** + * Apply multidimensional rotary position embeddings (RoPE) in-place. + * + * C++ equivalent of Gemma4's apply_multidimensional_rope with ndim=2, followed + * by apply_rotary_pos_emb (rotate_half variant). + * + * The head_dim is split into two spatial halves: + * x-spatial: channels [0 .. spatial_dim-1] (spatial_dim = head_dim/2) + * y-spatial: channels [spatial_dim .. head_dim-1] + * Standard RoPE is applied independently to each half: + * out_lo = x_lo * cos - x_hi * sin + * out_hi = x_hi * cos + x_lo * sin + * where _lo/_hi refer to the lower and upper quarter of each spatial half. + * + * @param qkv Pointer to buffer of shape [seq_len, num_head, head_dim] (bf16, in-place) + * @param cos_emb Pointer to cosine embeddings of shape [seq_len_padded, head_dim] (bf16) + * @param sin_emb Pointer to sine embeddings of shape [seq_len_padded, head_dim] (bf16) + * @param seq_len Number of valid (non-padded) sequence positions to process + * @param num_head Number of attention heads + * @param head_dim Head dimension; must be divisible by 4 (64 for Gemma4e vision) + */ +void apply_multidimensional_rope( + bf16* qkv, + const bf16* cos_emb, + const bf16* sin_emb, + int seq_len, + int num_head, + int head_dim, + int ndim = 2 +); diff --git a/src/detail/gemma4e_npu/seq_gen.hpp b/src/detail/gemma4e_npu/seq_gen.hpp new file mode 100644 index 000000000..18929a437 --- /dev/null +++ b/src/detail/gemma4e_npu/seq_gen.hpp @@ -0,0 +1,327 @@ +#pragma once +#include "npu_utils/npu_instr_utils.hpp" +#include "npu_sequences/image_attention_sequence.hpp" + +void _gen_mha_engine_seq(npu_sequence* seq, const uint32_t L_padded, const uint32_t L, + const int QKV_buffer_offset_bf, + const int O_buffer_offset_bf +){ + constexpr int DQ = 64 * 16; + constexpr int DK = 64 * 16; + constexpr int DV = DK; + constexpr int DH = 64; + int K_OFFSET = DQ; + int V_OFFSET = DQ + DK; + int D_TOTAL = DQ + DK + DV; + constexpr int total_cols = 8; + constexpr int total_rows = 4; + constexpr npu_tiles IT[8] = {IT0, IT1, IT2, IT3, IT4, IT5, IT6, IT7}; + + // each function call init a new list + int k_ping_pong_flag[8] = {0,0,0,0,0,0,0,0}; + int v_ping_pong_flag[8] = {0,0,0,0,0,0,0,0}; + int k_bd_queue[8] = {0,0,0,0,0,0,0,0}; + int v_bd_queue[8] = {0,0,0,0,0,0,0,0}; + + constexpr int lc = 32; + assert(L_padded % (32) == 0); + constexpr int l_qk_mha_address = 21760; + constexpr int l_kv_mha_address = 28672; + // seq->clear_cmds(); + + for (int row = 2; row < total_rows + 2; row++){ + for (int col = 0; col < 4; col++){ + npu_tiles tile_qk = get_tile(row, col * 2); + npu_tiles tile_kv = get_tile(row, col * 2 + 1); + seq->rtp_write(tile_qk, l_qk_mha_address, L); // 32 is lk + seq->rtp_write(tile_kv, l_kv_mha_address, L); // 32 is lk + } + } + + for (int round = 0; round < L_padded / 32; round++){ + int bd_offset = (round % 2) * 8; + for (int col = 0; col < 4; col++){ + // receive y + size_t y_offset = round * lc * DQ + col * 4 * DH + O_buffer_offset_bf; // y does not have history + seq->npu_dma_memcpy_nd( + 2, 0, + S2MM, IT[col * 2 + 1], + (npu_bd_id)(bd_offset + 0), it_channel_0, + {0, 0, 0, (uint32_t)y_offset}, + {1, 1, (uint32_t)lc, (uint32_t)DH * 4}, + {0, 0, (uint32_t)DQ, 1}, + -1, 0, true + ); + // send q + size_t q_offset = round * lc * D_TOTAL + col * 4 * DH +QKV_buffer_offset_bf; // q does not have history + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[col * 2], + (npu_bd_id)(bd_offset + 1), it_channel_0, + {0, 0, 0, (uint32_t)q_offset}, + {1, 1, (uint32_t)lc, (uint32_t)DH * 4}, + {0, 0, (uint32_t)D_TOTAL, 1}, + -1, 0, false + ); + + uint32_t L_padded_div_32 = L_padded / 32; + uint32_t max_L_per_chunk = 512; + // send k + size_t k_offset = col * 4 * DH + K_OFFSET + QKV_buffer_offset_bf; + size_t v_offset = col * 4 * DH + V_OFFSET + QKV_buffer_offset_bf; + for(int chunk_idx = 0; chunk_idx < L_padded_div_32; chunk_idx += max_L_per_chunk){ + + uint32_t cur_chunk_seqlen = std::min(max_L_per_chunk, L_padded_div_32 - chunk_idx); + int k_shim_id = col*2; + int k_bd_offset ; + if(k_ping_pong_flag[k_shim_id] == 0){ + k_bd_offset = 0; + k_ping_pong_flag[k_shim_id] = 1; + }else{ + k_bd_offset = 8; + k_ping_pong_flag[k_shim_id] = 0; + } + if(k_bd_queue[k_shim_id] == 2){ + // wait k + seq->npu_dma_wait( + IT[k_shim_id], + MM2S, + it_channel_1 + ); + k_bd_queue[k_shim_id]--; + } + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[k_shim_id], + (npu_bd_id)(k_bd_offset + 2), it_channel_1, + {0, 0, 0, (uint32_t)k_offset + chunk_idx * D_TOTAL * 32}, //row major of q, k, v per row + {1, cur_chunk_seqlen, (uint32_t)32, (uint32_t)DH * 4}, + {0, (uint32_t)32 * D_TOTAL, (uint32_t)D_TOTAL, 1}, + -1, 0, true, + aggressive_cache + ); + k_bd_queue[k_shim_id]++; + + int v_shim_id = col * 2 + 1; + int v_bd_offset ; + if(v_ping_pong_flag[v_shim_id] == 0){ + v_bd_offset = 0; + v_ping_pong_flag[v_shim_id] = 1; + }else{ + v_bd_offset = 8; + v_ping_pong_flag[v_shim_id] = 0; + } + if(v_bd_queue[v_shim_id] == 2){ + // wait v + seq->npu_dma_wait( + IT[v_shim_id], + MM2S, + it_channel_1 + ); + v_bd_queue[v_shim_id]--; + } + seq->npu_dma_memcpy_nd( + 2, 1, + MM2S, IT[v_shim_id], + (npu_bd_id)(v_bd_offset + 3), it_channel_1, + {0, 0, 0, (uint32_t)v_offset + chunk_idx * D_TOTAL * 32}, + {1, cur_chunk_seqlen, (uint32_t)32, (uint32_t)DH * 4}, + {0, (uint32_t)32 * D_TOTAL, (uint32_t)D_TOTAL, 1}, + -1, 0, true, + aggressive_cache + ); + v_bd_queue[v_shim_id]++; + } + } // col loop + + if (round > 0){ + for (int col = 0; col < 4; col++){ + seq->npu_dma_wait( + IT[col * 2 + 1], + S2MM, + it_channel_0 + ); + } + } + } + + int max_k_queue_remaing = -1; //should be same with v + for(int col = 0; col < 8; col++){ + if( k_bd_queue[col] > max_k_queue_remaing){ + max_k_queue_remaing = k_bd_queue[col]; + } + } + + for(int queue_size = 0; queue_size 0){ + seq->npu_dma_wait( + IT[col], + MM2S, + it_channel_1 + ); + k_bd_queue[col]--; + } + if(v_bd_queue[col] > 0){ + seq->npu_dma_wait( + IT[col], + MM2S, + it_channel_1 + ); + v_bd_queue[col]--; + } + } + } + + for (int col = 0; col < 4; col++){ + seq->npu_dma_wait( + IT[col * 2 + 1], + S2MM, + it_channel_0 + ); + } + // seq->cmds2seq(); +} + +//deprecated +void gen_mha_main( + + npu_sequence* seq, + std::vector> image_grid_thw, + int pad_requirement_for_attention, + int QWEN3_5_VISION_HIDDEN_SIZE + +){ + + // support of multiple batch mha + + seq->clear_cmds(); + seq->npu_preemption(0); + + auto round_up_to_multiple_lambda = [](int x, int multiple) -> int { + return ((x + multiple - 1) / multiple) * multiple; + }; + + int cur_seq_len = 0; + for(int b = 0; b < image_grid_thw.size(); b++){ + + int start_seq_len = cur_seq_len; + int end_seq_len = 1; + for(int v: image_grid_thw[b]){ + end_seq_len *= v; + } + + _gen_mha_engine_seq( + seq, round_up_to_multiple_lambda(end_seq_len, pad_requirement_for_attention), + end_seq_len, + start_seq_len * (3*QWEN3_5_VISION_HIDDEN_SIZE), //q,k,v offser + start_seq_len* QWEN3_5_VISION_HIDDEN_SIZE// offset + + ); + cur_seq_len += end_seq_len; + } + + seq->cmds2seq(); +} + +void gen_mha_vision_attention( + + npu_sequence* seq, + std::vector &seq_len_per_image, + int vision_L_padded_requirement_for_attention, + int vision_S_padded_requirement_for_attention, + uint32_t vision_num_of_columns, + uint32_t vision_num_of_rows, + uint32_t vision_CU_mode, + uint32_t vision_LQ_per_CT, + uint32_t vision_LK_per_CT, + uint32_t vision_LQ_internal, + uint32_t vision_LK_internal, + bool REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH, + int VISION_HIDDEN_SIZE, + int Padded_VISION_HIDDEN_SIZE, + int VISION_HEAD_DIM, + int VISION_NUM_ATTENTION_HEADS +){ + + if(vision_S_padded_requirement_for_attention >vision_L_padded_requirement_for_attention ){ + std::cerr << "vision S padded requirement greater than L padded requirement"<< std::endl; + exit(1); + } + + // support of multiple batch mha + + auto round_up_to_multiple_lambda = [](int x, int multiple) -> int { + return ((x + multiple - 1) / multiple) * multiple; + }; + + std::vector L_seq_list; + std::vector S_seq_list; + std::vector S_seq_padded_list; + + std::vector Q_batch_offset_list; + std::vector K_batch_offset_list; + std::vector V_batch_offset_list; + std::vector O_batch_offset_list; + + int cur_seq_len = 0; + for(int b = 0; b < seq_len_per_image.size(); b++){ + + int start_seq_len = cur_seq_len; + int end_seq_len = seq_len_per_image[b]; + + uint32_t l_seq_padded = round_up_to_multiple_lambda(end_seq_len, vision_L_padded_requirement_for_attention);// same for L, and S + uint32_t s_seq_padded = round_up_to_multiple_lambda(end_seq_len, vision_S_padded_requirement_for_attention); + L_seq_list.push_back(l_seq_padded); + S_seq_list.push_back(end_seq_len); + S_seq_padded_list.push_back(s_seq_padded); + + // but since + if(!REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH){ + Q_batch_offset_list.push_back(start_seq_len * (Padded_VISION_HIDDEN_SIZE) ); + K_batch_offset_list.push_back(start_seq_len * Padded_VISION_HIDDEN_SIZE); + V_batch_offset_list.push_back(start_seq_len * Padded_VISION_HIDDEN_SIZE); + O_batch_offset_list.push_back(start_seq_len * Padded_VISION_HIDDEN_SIZE); + }else{ + std::cout << "Currently the code only support REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH = false, please set it to false" << std::endl; + exit(-1); + // // The buffer is now in [3*QWEN3_VISION_NUM_HEAD, batch, L_Seq_per_batch, parent_npu_ptr->GEMMA4E_VISION_HEAD_DIM] + } + + cur_seq_len += end_seq_len; + } + + // because q, k, v buffer all in on buffer after MM + constexpr int ARG_Q = 1; + constexpr int ARG_K = 2; + constexpr int ARG_V = 3; + constexpr int ARG_O = 0; + + AttentionConfig config = { + (uint32_t)VISION_HEAD_DIM, (uint32_t)VISION_NUM_ATTENTION_HEADS, + vision_LQ_per_CT, vision_LK_per_CT, + vision_LQ_internal, vision_LK_internal, + vision_num_of_columns, vision_num_of_rows, + vision_CU_mode, + (uint32_t)seq_len_per_image.size(), + + ARG_Q, ARG_K, ARG_V, ARG_O, + + Padded_VISION_HIDDEN_SIZE, Padded_VISION_HIDDEN_SIZE, Padded_VISION_HIDDEN_SIZE, + Padded_VISION_HIDDEN_SIZE, + REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH, + 0,0,0 // since this only vailid if REORDER_OUTPUT_FROM_M_N_TO_NUM_DH_M_DH == true, and currently we only support false, so set it to -1 to avoid misuse + }; + + BatchMetadata batch_data = { + L_seq_list, S_seq_list, S_seq_padded_list, + Q_batch_offset_list, K_batch_offset_list, V_batch_offset_list, O_batch_offset_list + }; + + setup_SHM_configuration( + *seq, + config, + batch_data, + 1.0f// NOTE: very special for gemma4e + ); +} diff --git a/src/detail/include/aiebu/aiebu.h b/src/detail/include/aiebu/aiebu.h new file mode 100644 index 000000000..9301ba998 --- /dev/null +++ b/src/detail/include/aiebu/aiebu.h @@ -0,0 +1,88 @@ +// SPDX-License-Identifier: MIT +// Copyright (C) 2024-2025, Advanced Micro Devices, Inc. All rights reserved. +#ifndef AIEBU_H_ +#define AIEBU_H_ + +#include +#include + +#ifdef __cplusplus +extern "C" { +#endif + +enum aiebu_error_code { + aiebu_invalid_asm = 1, + aiebu_invalid_patch_schema, + aiebu_invalid_batch_buffer_type, + aiebu_invalid_buffer_type, + aiebu_invalid_offset, + aiebu_invalid_internal_error, + aiebu_invalid_input, + aiebu_invalid_elf, + aiebu_invalid_opcode +}; + +enum aiebu_assembler_buffer_type { + aiebu_assembler_buffer_type_blob_instr_dpu, + aiebu_assembler_buffer_type_blob_instr_prepost, + aiebu_assembler_buffer_type_blob_instr_transaction, + aiebu_assembler_buffer_type_blob_control_packet, + aiebu_assembler_buffer_type_asm_aie2ps, + aiebu_assembler_buffer_type_asm_aie2, + aiebu_assembler_buffer_type_asm_aie4, + aiebu_assembler_buffer_type_aie2_config, + aiebu_assembler_buffer_type_aie2ps_config, + aiebu_assembler_buffer_type_aie4_config +}; + +struct pm_ctrlpkt { + uint32_t pm_id; + const char* pm_buffer; + size_t pm_buffer_size; +}; + +/* + * This API takes buffer type, 2 buffers, their sizes and external_buffer_id json + * it also allocate elf_buf and It fill elf content in it. + * return, on success return return elf size, else posix error(negative). + * User may pass any combination like + * 1. type as aiebu_assembler_buffer_type_blob_instr_transaction, buffer1 as instruction buffer + * and buffer2 as control_packet: in this case it will package buffers in text and data + * section of elf respectively. + * 2. type as aiebu_assembler_buffer_type_blob_instr_transaction, buffer1 as instruction buffer + * and buffer2 as null: in this case it will package buffer in text section. + * 3. type as aiebu_assembler_buffer_type_asm_aie2ps, buffer1 as asm buffer and buffer2 + * as null: in this case it will assemble the asm code and package in elf. + * + * @type buffer type + * @instr_buf first buffer + * @instr_buf_size first buffer size + * @control_buf second buffer + * @control_buf_size second buffer size + * @elf_buf elf buffer + * @patch_json external_buffer_id_json buffer. + * @patch_json_size patch_json array size + * @libs libs to be included, ";" separated. + * @libpaths paths to search for libs, ";" separated. + * @ctrlpkt array of pm_ctrlpkt holding pm buffer and id + * @ctrlpkt_size size of ctrlpkt array + */ +int +aiebu_assembler_get_elf(enum aiebu_assembler_buffer_type type, + const char* buffer1, + size_t buffer1_size, + const char* buffer2, + size_t buffer2_size, + void** elf_buf, + const char* patch_json, + size_t patch_json_size, + const char* libs, + const char* libpaths, + struct pm_ctrlpkt* pm_ctrlpkts, + size_t pm_ctrlpkt_size); + +#ifdef __cplusplus +} +#endif + +#endif diff --git a/src/detail/include/aiebu/aiebu_assembler.h b/src/detail/include/aiebu/aiebu_assembler.h new file mode 100644 index 000000000..9315d2362 --- /dev/null +++ b/src/detail/include/aiebu/aiebu_assembler.h @@ -0,0 +1,123 @@ +// SPDX-License-Identifier: MIT +// Copyright (C) 2024-2025, Advanced Micro Devices, Inc. All rights reserved. +#ifndef AIEBU_ASSEMBLER_H_ +#define AIEBU_ASSEMBLER_H_ +#include +#include +#include +#include +#include + +namespace aiebu { + +// Assembler Class + +class aiebu_assembler +{ + std::vector elf_data; + + public: + enum class buffer_type { + blob_instr_dpu, + blob_instr_prepost, + blob_instr_transaction, + blob_control_packet, + asm_aie2ps, + asm_aie2, + asm_aie4, + aie2_config, + aie2ps_config, + aie4_config, + elf_aie2, + elf_aie2ps, + pdi_aie2, + pdi_aie2ps, + blob_control_packet_aie2, + elf_aie2_config, + elf_aie2ps_config, + elf_aie4, + elf_aie4_config, + unspecified, + }; + + private: + buffer_type m_type; + buffer_type m_output_type; + + public: + /* + * Constructor takes buffer type , 2 buffer and a vector of symbols with + * external_buffer_id json as argument. + * its throws aiebu::error object. + * User may pass any combination like + * 1. type as blob_instr_transaction, buffer1 as instruction buffer + * and buffer2 as control_packet and pm_ctrlpkt as map of + * : in this case it will package buffers in text section, data section and + * ctrlpkt_pm_N section of elf respectively. + * 2. type as blob_instr_transaction, buffer1 as instruction buffer + * and buffer2 as empty and and pm_ctrlpkt as map of + * : in this case it will package buffer in text section and ctrlpkt_pm_N section of elf respectively. + * 3. type as asm_aie2ps/asm_aie4, buffer1 as asm buffer and buffer2 + * as empty: in this case it will assemble the asm code and package in elf. + * This api can do fileops for include asm/ctrlpkt. + * + * @type buffer type + * @instr_buf first buffer + * @constrol_buf second buffer + * @patch_json external_buffer_id json + * @libs libs to include in elf + * @libpaths paths to search for libs, paths to search for included asm, ctrlpkt. + only paths provided in this, are used for searching. + * @ctrlpkt map of pm id and pm control packet buffer + */ + aiebu_assembler(buffer_type type, + const std::vector& buffer1, + const std::vector& buffer2, + const std::vector& patch_json, + const std::vector& libs = {}, + const std::vector& libpaths = {}, + const std::map >& pm_ctrlpkt = {}); + + /* + * Constructor takes buffer type, buffer, + * and a vector of symbols with their patching information as argument. + * This api can do fileops for include asm/ctrlpkt. + * its throws aiebu::error object. + * + * @type buffer type + * @instr_buf first buffer + * @libs libs to include in elf + * @libpaths paths to search for libs, paths to search for included asm, ctrlpkt. + only paths provided in this, are used for searching. + * @patch_json external_buffer_id json + */ + aiebu_assembler(buffer_type type, + const std::vector& buffer, + const std::vector& libs = {}, + const std::vector& libpaths = {}, + const std::vector& patch_json = {}); + + /* + * This function return vector with elf content. + * + * Inside elf for IPU, instr_buf will be placed in .text section and control_buf will + * be placed in .data section. There are other dynamic sections in the elf + * containing the relocatable information. With this elf, at runtime, XRT + * will patch the symbols (value or address based on the schema) into their + * instruction buffer and control buffer before sending the buffer to device. + * + * return: vector of char with elf content + */ + std::vector + get_elf() const; + + void + get_report(std::ostream &stream) const; + + void + disassemble(const std::filesystem::path &root) const; +}; + +} //namespace aiebu + +#endif // AIEBU_ASSEMBLER_H_ diff --git a/src/detail/include/aiebu/aiebu_error.h b/src/detail/include/aiebu/aiebu_error.h new file mode 100644 index 000000000..1627ecf8c --- /dev/null +++ b/src/detail/include/aiebu/aiebu_error.h @@ -0,0 +1,45 @@ +// SPDX-License-Identifier: MIT +// Copyright (C) 2024-2025, Advanced Micro Devices, Inc. All rights reserved. +#ifndef AIEBU_ERROR_H_ +#define AIEBU_ERROR_H_ +#include "aiebu/aiebu.h" + +#include +#include + +namespace aiebu { + +class error : public std::system_error +{ +public: + enum class error_code : int { + invalid_asm = aiebu_invalid_asm, + invalid_patch_schema = aiebu_invalid_patch_schema, + invalid_patch_buffer_type = aiebu_invalid_batch_buffer_type, + invalid_buffer_type = aiebu_invalid_buffer_type, + invalid_offset = aiebu_invalid_offset, + internal_error = aiebu_invalid_internal_error, + invalid_input = aiebu_invalid_input, + invalid_elf = aiebu_invalid_elf, + invalid_opcode = aiebu_invalid_opcode + }; + + error(error_code ec, const std::error_category& cat, const std::string& what = ""); + + explicit + error(error_code ec, const std::string& what = ""); + + // Retrive underlying code for return plain error code + int + value() const; + + int + get() const; + + int + get_code() const; +}; + +} + +#endif // AIEBU_ERROR_H_ diff --git a/src/detail/include/base64.hpp b/src/detail/include/base64.hpp new file mode 100644 index 000000000..1ce221e84 --- /dev/null +++ b/src/detail/include/base64.hpp @@ -0,0 +1,697 @@ +#ifndef BASE64_HPP_ +#define BASE64_HPP_ + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#if defined(__cpp_lib_bit_cast) +#include // For std::bit_cast. +#endif + +namespace base64 { + +namespace detail { + +#if defined(__cpp_lib_bit_cast) +using std::bit_cast; +#else +template +std::enable_if_t && + std::is_trivially_copyable_v, + To> +bit_cast(const From& src) noexcept { + static_assert(std::is_trivially_constructible_v, + "This implementation additionally requires " + "destination type to be trivially constructible"); + + To dst; + std::memcpy(&dst, &src, sizeof(To)); + return dst; +} +#endif + +inline constexpr char padding_char{'='}; +inline constexpr uint32_t bad_char{0x01FFFFFF}; + +#if !defined(__LITTLE_ENDIAN__) && !defined(__BIG_ENDIAN__) +#if (defined(__BYTE_ORDER__) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__) || \ + (defined(__BYTE_ORDER) && __BYTE_ORDER == __BIG_ENDIAN) || \ + (defined(_BYTE_ORDER) && _BYTE_ORDER == _BIG_ENDIAN) || \ + (defined(BYTE_ORDER) && BYTE_ORDER == BIG_ENDIAN) || \ + (defined(__sun) && defined(__SVR4) && defined(_BIG_ENDIAN)) || \ + defined(__ARMEB__) || defined(__THUMBEB__) || defined(__AARCH64EB__) || \ + defined(_MIBSEB) || defined(__MIBSEB) || defined(__MIBSEB__) || \ + defined(_M_PPC) +#define __BIG_ENDIAN__ +#elif (defined(__BYTE_ORDER__) && \ + __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__) || /* gcc */ \ + (defined(__BYTE_ORDER) && \ + __BYTE_ORDER == __LITTLE_ENDIAN) /* linux header */ \ + || (defined(_BYTE_ORDER) && _BYTE_ORDER == _LITTLE_ENDIAN) || \ + (defined(BYTE_ORDER) && BYTE_ORDER == LITTLE_ENDIAN) /* mingw header */ || \ + (defined(__sun) && defined(__SVR4) && \ + defined(_LITTLE_ENDIAN)) || /* solaris */ \ + defined(__ARMEL__) || \ + defined(__THUMBEL__) || defined(__AARCH64EL__) || defined(_MIPSEL) || \ + defined(__MIPSEL) || defined(__MIPSEL__) || defined(_M_IX86) || \ + defined(_M_X64) || defined(_M_IA64) || /* msvc for intel processors */ \ + defined(_M_ARM) /* msvc code on arm executes in little endian mode */ +#define __LITTLE_ENDIAN__ +#endif +#endif + +#if !defined(__LITTLE_ENDIAN__) & !defined(__BIG_ENDIAN__) +#error "UNKNOWN Platform / endianness. Configure endianness explicitly." +#endif + +#if defined(__LITTLE_ENDIAN__) +std::array constexpr decode_table_0 = { + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x000000f8, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x000000fc, + 0x000000d0, 0x000000d4, 0x000000d8, 0x000000dc, 0x000000e0, 0x000000e4, + 0x000000e8, 0x000000ec, 0x000000f0, 0x000000f4, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00000000, + 0x00000004, 0x00000008, 0x0000000c, 0x00000010, 0x00000014, 0x00000018, + 0x0000001c, 0x00000020, 0x00000024, 0x00000028, 0x0000002c, 0x00000030, + 0x00000034, 0x00000038, 0x0000003c, 0x00000040, 0x00000044, 0x00000048, + 0x0000004c, 0x00000050, 0x00000054, 0x00000058, 0x0000005c, 0x00000060, + 0x00000064, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x00000068, 0x0000006c, 0x00000070, 0x00000074, 0x00000078, + 0x0000007c, 0x00000080, 0x00000084, 0x00000088, 0x0000008c, 0x00000090, + 0x00000094, 0x00000098, 0x0000009c, 0x000000a0, 0x000000a4, 0x000000a8, + 0x000000ac, 0x000000b0, 0x000000b4, 0x000000b8, 0x000000bc, 0x000000c0, + 0x000000c4, 0x000000c8, 0x000000cc, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff}; + +std::array constexpr decode_table_1 = { + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x0000e003, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x0000f003, + 0x00004003, 0x00005003, 0x00006003, 0x00007003, 0x00008003, 0x00009003, + 0x0000a003, 0x0000b003, 0x0000c003, 0x0000d003, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00000000, + 0x00001000, 0x00002000, 0x00003000, 0x00004000, 0x00005000, 0x00006000, + 0x00007000, 0x00008000, 0x00009000, 0x0000a000, 0x0000b000, 0x0000c000, + 0x0000d000, 0x0000e000, 0x0000f000, 0x00000001, 0x00001001, 0x00002001, + 0x00003001, 0x00004001, 0x00005001, 0x00006001, 0x00007001, 0x00008001, + 0x00009001, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x0000a001, 0x0000b001, 0x0000c001, 0x0000d001, 0x0000e001, + 0x0000f001, 0x00000002, 0x00001002, 0x00002002, 0x00003002, 0x00004002, + 0x00005002, 0x00006002, 0x00007002, 0x00008002, 0x00009002, 0x0000a002, + 0x0000b002, 0x0000c002, 0x0000d002, 0x0000e002, 0x0000f002, 0x00000003, + 0x00001003, 0x00002003, 0x00003003, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff}; + +std::array constexpr decode_table_2 = { + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x00800f00, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00c00f00, + 0x00000d00, 0x00400d00, 0x00800d00, 0x00c00d00, 0x00000e00, 0x00400e00, + 0x00800e00, 0x00c00e00, 0x00000f00, 0x00400f00, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00000000, + 0x00400000, 0x00800000, 0x00c00000, 0x00000100, 0x00400100, 0x00800100, + 0x00c00100, 0x00000200, 0x00400200, 0x00800200, 0x00c00200, 0x00000300, + 0x00400300, 0x00800300, 0x00c00300, 0x00000400, 0x00400400, 0x00800400, + 0x00c00400, 0x00000500, 0x00400500, 0x00800500, 0x00c00500, 0x00000600, + 0x00400600, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x00800600, 0x00c00600, 0x00000700, 0x00400700, 0x00800700, + 0x00c00700, 0x00000800, 0x00400800, 0x00800800, 0x00c00800, 0x00000900, + 0x00400900, 0x00800900, 0x00c00900, 0x00000a00, 0x00400a00, 0x00800a00, + 0x00c00a00, 0x00000b00, 0x00400b00, 0x00800b00, 0x00c00b00, 0x00000c00, + 0x00400c00, 0x00800c00, 0x00c00c00, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff}; + +std::array constexpr decode_table_3 = { + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x003e0000, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x003f0000, + 0x00340000, 0x00350000, 0x00360000, 0x00370000, 0x00380000, 0x00390000, + 0x003a0000, 0x003b0000, 0x003c0000, 0x003d0000, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00000000, + 0x00010000, 0x00020000, 0x00030000, 0x00040000, 0x00050000, 0x00060000, + 0x00070000, 0x00080000, 0x00090000, 0x000a0000, 0x000b0000, 0x000c0000, + 0x000d0000, 0x000e0000, 0x000f0000, 0x00100000, 0x00110000, 0x00120000, + 0x00130000, 0x00140000, 0x00150000, 0x00160000, 0x00170000, 0x00180000, + 0x00190000, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x001a0000, 0x001b0000, 0x001c0000, 0x001d0000, 0x001e0000, + 0x001f0000, 0x00200000, 0x00210000, 0x00220000, 0x00230000, 0x00240000, + 0x00250000, 0x00260000, 0x00270000, 0x00280000, 0x00290000, 0x002a0000, + 0x002b0000, 0x002c0000, 0x002d0000, 0x002e0000, 0x002f0000, 0x00300000, + 0x00310000, 0x00320000, 0x00330000, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff}; + +// TODO fix decoding tables to avoid the need for different indices in big +// endian? +inline constexpr size_t decidx0{0}; +inline constexpr size_t decidx1{1}; +inline constexpr size_t decidx2{2}; + +#elif defined(__BIG_ENDIAN__) + +std::array constexpr decode_table_0 = { + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x00f80000, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00fc0000, + 0x00d00000, 0x00d40000, 0x00d80000, 0x00dc0000, 0x00e00000, 0x00e40000, + 0x00e80000, 0x00ec0000, 0x00f00000, 0x00f40000, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00000000, + 0x00040000, 0x00080000, 0x000c0000, 0x00100000, 0x00140000, 0x00180000, + 0x001c0000, 0x00200000, 0x00240000, 0x00280000, 0x002c0000, 0x00300000, + 0x00340000, 0x00380000, 0x003c0000, 0x00400000, 0x00440000, 0x00480000, + 0x004c0000, 0x00500000, 0x00540000, 0x00580000, 0x005c0000, 0x00600000, + 0x00640000, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x00680000, 0x006c0000, 0x00700000, 0x00740000, 0x00780000, + 0x007c0000, 0x00800000, 0x00840000, 0x00880000, 0x008c0000, 0x00900000, + 0x00940000, 0x00980000, 0x009c0000, 0x00a00000, 0x00a40000, 0x00a80000, + 0x00ac0000, 0x00b00000, 0x00b40000, 0x00b80000, 0x00bc0000, 0x00c00000, + 0x00c40000, 0x00c80000, 0x00cc0000, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff}; + +std::array constexpr decode_table_1 = { + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x0003e000, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x0003f000, + 0x00034000, 0x00035000, 0x00036000, 0x00037000, 0x00038000, 0x00039000, + 0x0003a000, 0x0003b000, 0x0003c000, 0x0003d000, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00000000, + 0x00001000, 0x00002000, 0x00003000, 0x00004000, 0x00005000, 0x00006000, + 0x00007000, 0x00008000, 0x00009000, 0x0000a000, 0x0000b000, 0x0000c000, + 0x0000d000, 0x0000e000, 0x0000f000, 0x00010000, 0x00011000, 0x00012000, + 0x00013000, 0x00014000, 0x00015000, 0x00016000, 0x00017000, 0x00018000, + 0x00019000, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x0001a000, 0x0001b000, 0x0001c000, 0x0001d000, 0x0001e000, + 0x0001f000, 0x00020000, 0x00021000, 0x00022000, 0x00023000, 0x00024000, + 0x00025000, 0x00026000, 0x00027000, 0x00028000, 0x00029000, 0x0002a000, + 0x0002b000, 0x0002c000, 0x0002d000, 0x0002e000, 0x0002f000, 0x00030000, + 0x00031000, 0x00032000, 0x00033000, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff}; + +std::array constexpr decode_table_2 = { + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x00000f80, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00000fc0, + 0x00000d00, 0x00000d40, 0x00000d80, 0x00000dc0, 0x00000e00, 0x00000e40, + 0x00000e80, 0x00000ec0, 0x00000f00, 0x00000f40, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00000000, + 0x00000040, 0x00000080, 0x000000c0, 0x00000100, 0x00000140, 0x00000180, + 0x000001c0, 0x00000200, 0x00000240, 0x00000280, 0x000002c0, 0x00000300, + 0x00000340, 0x00000380, 0x000003c0, 0x00000400, 0x00000440, 0x00000480, + 0x000004c0, 0x00000500, 0x00000540, 0x00000580, 0x000005c0, 0x00000600, + 0x00000640, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x00000680, 0x000006c0, 0x00000700, 0x00000740, 0x00000780, + 0x000007c0, 0x00000800, 0x00000840, 0x00000880, 0x000008c0, 0x00000900, + 0x00000940, 0x00000980, 0x000009c0, 0x00000a00, 0x00000a40, 0x00000a80, + 0x00000ac0, 0x00000b00, 0x00000b40, 0x00000b80, 0x00000bc0, 0x00000c00, + 0x00000c40, 0x00000c80, 0x00000cc0, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff}; + +std::array constexpr decode_table_3 = { + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x0000003e, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x0000003f, + 0x00000034, 0x00000035, 0x00000036, 0x00000037, 0x00000038, 0x00000039, + 0x0000003a, 0x0000003b, 0x0000003c, 0x0000003d, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x00000000, + 0x00000001, 0x00000002, 0x00000003, 0x00000004, 0x00000005, 0x00000006, + 0x00000007, 0x00000008, 0x00000009, 0x0000000a, 0x0000000b, 0x0000000c, + 0x0000000d, 0x0000000e, 0x0000000f, 0x00000010, 0x00000011, 0x00000012, + 0x00000013, 0x00000014, 0x00000015, 0x00000016, 0x00000017, 0x00000018, + 0x00000019, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x0000001a, 0x0000001b, 0x0000001c, 0x0000001d, 0x0000001e, + 0x0000001f, 0x00000020, 0x00000021, 0x00000022, 0x00000023, 0x00000024, + 0x00000025, 0x00000026, 0x00000027, 0x00000028, 0x00000029, 0x0000002a, + 0x0000002b, 0x0000002c, 0x0000002d, 0x0000002e, 0x0000002f, 0x00000030, + 0x00000031, 0x00000032, 0x00000033, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff, + 0x01ffffff, 0x01ffffff, 0x01ffffff, 0x01ffffff}; + +// TODO fix decoding tables to avoid the need for different indices in big +// endian? +inline constexpr size_t decidx0{1}; +inline constexpr size_t decidx1{2}; +inline constexpr size_t decidx2{3}; + +#endif + +std::array constexpr encode_table_0 = { + 'A', 'A', 'A', 'A', 'B', 'B', 'B', 'B', 'C', 'C', 'C', 'C', 'D', 'D', 'D', + 'D', 'E', 'E', 'E', 'E', 'F', 'F', 'F', 'F', 'G', 'G', 'G', 'G', 'H', 'H', + 'H', 'H', 'I', 'I', 'I', 'I', 'J', 'J', 'J', 'J', 'K', 'K', 'K', 'K', 'L', + 'L', 'L', 'L', 'M', 'M', 'M', 'M', 'N', 'N', 'N', 'N', 'O', 'O', 'O', 'O', + 'P', 'P', 'P', 'P', 'Q', 'Q', 'Q', 'Q', 'R', 'R', 'R', 'R', 'S', 'S', 'S', + 'S', 'T', 'T', 'T', 'T', 'U', 'U', 'U', 'U', 'V', 'V', 'V', 'V', 'W', 'W', + 'W', 'W', 'X', 'X', 'X', 'X', 'Y', 'Y', 'Y', 'Y', 'Z', 'Z', 'Z', 'Z', 'a', + 'a', 'a', 'a', 'b', 'b', 'b', 'b', 'c', 'c', 'c', 'c', 'd', 'd', 'd', 'd', + 'e', 'e', 'e', 'e', 'f', 'f', 'f', 'f', 'g', 'g', 'g', 'g', 'h', 'h', 'h', + 'h', 'i', 'i', 'i', 'i', 'j', 'j', 'j', 'j', 'k', 'k', 'k', 'k', 'l', 'l', + 'l', 'l', 'm', 'm', 'm', 'm', 'n', 'n', 'n', 'n', 'o', 'o', 'o', 'o', 'p', + 'p', 'p', 'p', 'q', 'q', 'q', 'q', 'r', 'r', 'r', 'r', 's', 's', 's', 's', + 't', 't', 't', 't', 'u', 'u', 'u', 'u', 'v', 'v', 'v', 'v', 'w', 'w', 'w', + 'w', 'x', 'x', 'x', 'x', 'y', 'y', 'y', 'y', 'z', 'z', 'z', 'z', '0', '0', + '0', '0', '1', '1', '1', '1', '2', '2', '2', '2', '3', '3', '3', '3', '4', + '4', '4', '4', '5', '5', '5', '5', '6', '6', '6', '6', '7', '7', '7', '7', + '8', '8', '8', '8', '9', '9', '9', '9', '+', '+', '+', '+', '/', '/', '/', + '/'}; + +std::array constexpr encode_table_1 = { + 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', + 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z', 'a', 'b', 'c', 'd', + 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', + 't', 'u', 'v', 'w', 'x', 'y', 'z', '0', '1', '2', '3', '4', '5', '6', '7', + '8', '9', '+', '/', 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', + 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z', + 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm', 'n', 'o', + 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', '0', '1', '2', '3', + '4', '5', '6', '7', '8', '9', '+', '/', 'A', 'B', 'C', 'D', 'E', 'F', 'G', + 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', + 'W', 'X', 'Y', 'Z', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', + 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', + '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '+', '/', 'A', 'B', 'C', + 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', + 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z', 'a', 'b', 'c', 'd', 'e', 'f', 'g', + 'h', 'i', 'j', 'k', 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', + 'w', 'x', 'y', 'z', '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '+', + '/'}; + +} // namespace detail + +template +inline OutputBuffer encode_into(InputIterator begin, InputIterator end) { + typedef std::decay_t input_value_type; + static_assert(std::is_same_v || + std::is_same_v || + std::is_same_v || + std::is_same_v); + typedef typename OutputBuffer::value_type output_value_type; + static_assert(std::is_same_v || + std::is_same_v || + std::is_same_v || + std::is_same_v); + const size_t binarytextsize = end - begin; + const size_t encodedsize = (binarytextsize / 3 + (binarytextsize % 3 > 0)) + << 2; + OutputBuffer encoded(encodedsize, detail::padding_char); + + const uint8_t* bytes = reinterpret_cast(&*begin); + char* currEncoding = reinterpret_cast(&encoded[0]); + + for (size_t i = binarytextsize / 3; i; --i) { + const uint8_t t1 = *bytes++; + const uint8_t t2 = *bytes++; + const uint8_t t3 = *bytes++; + *currEncoding++ = detail::encode_table_0[t1]; + *currEncoding++ = + detail::encode_table_1[((t1 & 0x03) << 4) | ((t2 >> 4) & 0x0F)]; + *currEncoding++ = + detail::encode_table_1[((t2 & 0x0F) << 2) | ((t3 >> 6) & 0x03)]; + *currEncoding++ = detail::encode_table_1[t3]; + } + + switch (binarytextsize % 3) { + case 0: { + break; + } + case 1: { + const uint8_t t1 = bytes[0]; + *currEncoding++ = detail::encode_table_0[t1]; + *currEncoding++ = detail::encode_table_1[(t1 & 0x03) << 4]; + // *currEncoding++ = detail::padding_char; + // *currEncoding++ = detail::padding_char; + break; + } + case 2: { + const uint8_t t1 = bytes[0]; + const uint8_t t2 = bytes[1]; + *currEncoding++ = detail::encode_table_0[t1]; + *currEncoding++ = + detail::encode_table_1[((t1 & 0x03) << 4) | ((t2 >> 4) & 0x0F)]; + *currEncoding++ = detail::encode_table_1[(t2 & 0x0F) << 2]; + // *currEncoding++ = detail::padding_char; + break; + } + default: { + throw std::runtime_error{"Invalid base64 encoded data"}; + } + } + + return encoded; +} + +template +inline OutputBuffer encode_into(std::string_view data) { + return encode_into(std::begin(data), std::end(data)); +} + +inline std::string to_base64(std::string_view data) { + return encode_into(std::begin(data), std::end(data)); +} + +template +inline OutputBuffer decode_into(std::string_view base64Text) { + typedef typename OutputBuffer::value_type output_value_type; + static_assert(std::is_same_v || + std::is_same_v || + std::is_same_v || + std::is_same_v); + if (base64Text.empty()) { + return OutputBuffer(); + } + + if ((base64Text.size() & 3) != 0) { + throw std::runtime_error{ + "Invalid base64 encoded data - Size not divisible by 4"}; + } + + const size_t numPadding = + std::count(base64Text.rbegin(), base64Text.rbegin() + 4, '='); + if (numPadding > 2) { + throw std::runtime_error{ + "Invalid base64 encoded data - Found more than 2 padding signs"}; + } + + const size_t decodedsize = (base64Text.size() * 3 >> 2) - numPadding; + OutputBuffer decoded(decodedsize, '.'); + + const uint8_t* bytes = reinterpret_cast(&base64Text[0]); + char* currDecoding = reinterpret_cast(&decoded[0]); + + for (size_t i = (base64Text.size() >> 2) - (numPadding != 0); i; --i) { + const uint8_t t1 = *bytes++; + const uint8_t t2 = *bytes++; + const uint8_t t3 = *bytes++; + const uint8_t t4 = *bytes++; + + const uint32_t d1 = detail::decode_table_0[t1]; + const uint32_t d2 = detail::decode_table_1[t2]; + const uint32_t d3 = detail::decode_table_2[t3]; + const uint32_t d4 = detail::decode_table_3[t4]; + + const uint32_t temp = d1 | d2 | d3 | d4; + + if (temp >= detail::bad_char) { + throw std::runtime_error{ + "Invalid base64 encoded data - Invalid character"}; + } + + // Use bit_cast instead of union and type punning to avoid + // undefined behaviour risk: + // https://en.wikipedia.org/wiki/Type_punning#Use_of_union + const std::array tempBytes = + detail::bit_cast, uint32_t>(temp); + + *currDecoding++ = tempBytes[detail::decidx0]; + *currDecoding++ = tempBytes[detail::decidx1]; + *currDecoding++ = tempBytes[detail::decidx2]; + } + + switch (numPadding) { + case 0: { + break; + } + case 1: { + const uint8_t t1 = *bytes++; + const uint8_t t2 = *bytes++; + const uint8_t t3 = *bytes++; + + const uint32_t d1 = detail::decode_table_0[t1]; + const uint32_t d2 = detail::decode_table_1[t2]; + const uint32_t d3 = detail::decode_table_2[t3]; + + const uint32_t temp = d1 | d2 | d3; + + if (temp >= detail::bad_char) { + throw std::runtime_error{ + "Invalid base64 encoded data - Invalid character"}; + } + + // Use bit_cast instead of union and type punning to avoid + // undefined behaviour risk: + // https://en.wikipedia.org/wiki/Type_punning#Use_of_union + const std::array tempBytes = + detail::bit_cast, uint32_t>(temp); + *currDecoding++ = tempBytes[detail::decidx0]; + *currDecoding++ = tempBytes[detail::decidx1]; + break; + } + case 2: { + const uint8_t t1 = *bytes++; + const uint8_t t2 = *bytes++; + + const uint32_t d1 = detail::decode_table_0[t1]; + const uint32_t d2 = detail::decode_table_1[t2]; + + const uint32_t temp = d1 | d2; + + if (temp >= detail::bad_char) { + throw std::runtime_error{ + "Invalid base64 encoded data - Invalid character"}; + } + + const std::array tempBytes = + detail::bit_cast, uint32_t>(temp); + *currDecoding++ = tempBytes[detail::decidx0]; + break; + } + default: { + throw std::runtime_error{ + "Invalid base64 encoded data - Invalid padding number"}; + } + } + + return decoded; +} + +template +inline OutputBuffer decode_into(InputIterator begin, InputIterator end) { + typedef std::decay_t input_value_type; + static_assert(std::is_same_v || + std::is_same_v || + std::is_same_v || + std::is_same_v); + std::string_view data(reinterpret_cast(&*begin), end - begin); + return decode_into(data); +} + +inline std::string from_base64(std::string_view data) { + return decode_into(data); +} + +} // namespace base64 + +#endif // BASE64_HPP_ \ No newline at end of file diff --git a/src/detail/include/biovault_bfloat16.h b/src/detail/include/biovault_bfloat16.h new file mode 100644 index 000000000..6235ccd77 --- /dev/null +++ b/src/detail/include/biovault_bfloat16.h @@ -0,0 +1,210 @@ +#ifndef BIOVAULT_BFLOAT16_H_INCLUDE_GUARD +#define BIOVAULT_BFLOAT16_H_INCLUDE_GUARD + +/******************************************************************************* +* Copyright 2020 LKEB, Leiden University Medical Center +* +* Licensed under the Apache License, Version 2.0 (the "License"); +* you may not use this file except in compliance with the License. +* You may obtain a copy of the License at +* +* http://www.apache.org/licenses/LICENSE-2.0 +* +* Unless required by applicable law or agreed to in writing, software +* distributed under the License is distributed on an "AS IS" BASIS, +* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +* See the License for the specific language governing permissions and +* limitations under the License. +*******************************************************************************/ + +// Adapted from the original dnnl::impl::bfloat16_t implementation of +// oneAPI Deep Neural Network Library (oneDNN), by Intel Corporation, +// which is licensed under the Apache License, Version 2.0: +// https://github.com/oneapi-src/oneDNN/blob/v1.7/LICENSE + +//NOTE: direct copy from https://github.com/biovault/biovault_bfloat16/blob/main/biovault_bfloat16.h +// commit: 5cdca6a3f14a575aef55278241deb47697edca76 + + +#include +#include // For uint16_t and uint32_t. +#include +#include +#include +#include // For enable_if, is_integral, is_trivially_copyable, and is_pod. + + +// For a tagged version of the biovault_bfloat16 repository, having tag name +// "v" "." "." , the following macro defines should +// match that tag: +#define BIOVAULT_BFLOAT16_MAJOR_VERSION 1 +#define BIOVAULT_BFLOAT16_MINOR_VERSION 0 +#define BIOVAULT_BFLOAT16_PATCH_VERSION 2 + + +#ifdef _MSC_VER +# if _MSC_VER < 1900 +// Before Visual Studio 2015, Visual C++ did not yet support constexpr +# define BIOVAULT_BFLOAT16_CONSTEXPR +# endif +#endif + +#ifndef BIOVAULT_BFLOAT16_CONSTEXPR +#define BIOVAULT_BFLOAT16_CONSTEXPR constexpr +#endif + +namespace biovault { + + class bfloat16_t { + + private: + // Ensure that the following integer types can be used without "std::" prefix, + // just like in the original implementation (dnnl::impl::bfloat16_t). + using uint16_t = std::uint16_t; + using uint32_t = std::uint32_t; + + uint16_t raw_bits_; + + // bit_cast implementation originally from oneDNN: + // https://github.com/oneapi-src/oneDNN/blob/v1.7/src/common/bit_cast.hpp + // + // Returns a value of type T by reinterpretting the representation of the input + // value (part of C++20). + // + // Provides a safe implementation of type punning. + // + // Constraints: + // - U and T must have the same size + // - U and T must be trivially copyable + template + static T bit_cast(const U& u) { + static_assert(sizeof(T) == sizeof(U), "Bit-casting must preserve size."); +#if __cplusplus >= 202002L + // C++20 (and later) code + static_assert(std::is_trivially_copyable::value, "T must be trivially copyable."); + static_assert(std::is_trivially_copyable::value, "U must be trivially copyable."); +#else + // Use std::is_pod as older GNU versions do not support + // std::is_trivially_copyable. + static_assert(std::is_pod::value, "T must be trivially copyable."); + static_assert(std::is_pod::value, "U must be trivially copyable."); +#endif + + T t; + std::memcpy(&t, &u, sizeof(U)); + return t; + } + + // Converts the 32 bits of a normal float or zero to the bits of a bfloat16. + static BIOVAULT_BFLOAT16_CONSTEXPR uint16_t convert_bits_of_normal_or_zero( + const uint32_t bits) { + return uint32_t{ + bits + uint32_t {0x7FFFU + (uint32_t {bits >> 16} &1U)} } + >> 16; + } + + + public: + bfloat16_t() = default; + + // Allows specifying a bfloat16 by its raw bits. Equivalent to C++20 + // std::bit_cast(r) (which is more generic, of course.) + // Originally from oneDNN: + // https://github.com/oneapi-src/oneDNN/blob/v1.7/src/common/bfloat16.hpp#L34 + BIOVAULT_BFLOAT16_CONSTEXPR bfloat16_t(const uint16_t r, bool) : raw_bits_(r) {} + + // Supports narrowing (lossy) conversion from 32-bit float to bfloat16. + // Note: This constructor is "explicit" by default, but can be adjusted + // to allow implicit conversion to bfloat16_t by defining the macro + // BIOVAULT_BFLOAT16_CONVERTING_CONSTRUCTORS. (The oneDNN library does allow + // implicit conversion from float to bfloat_t.) +#ifndef BIOVAULT_BFLOAT16_CONVERTING_CONSTRUCTORS + explicit +#endif + bfloat16_t(const float f) { + // Implementation originally from oneDNN: + // https://github.com/oneapi-src/oneDNN/blob/v1.7/src/cpu/bfloat16.cpp#L47-L69 + auto iraw = bit_cast>(f); + switch (std::fpclassify(f)) { + case FP_SUBNORMAL: + case FP_ZERO: + // sign preserving zero (denormal go to zero) + raw_bits_ = iraw[1]; + raw_bits_ &= 0x8000; + break; + case FP_INFINITE: raw_bits_ = iraw[1]; break; + case FP_NAN: + // truncate and set MSB of the mantissa force QNAN + raw_bits_ = iraw[1]; + raw_bits_ |= 1 << 6; + break; + case FP_NORMAL: + // round to nearest even and truncate + const uint32_t rounding_bias = 0x00007FFF + (iraw[1] & 0x1); + const uint32_t int_raw + = bit_cast(f) + rounding_bias; + iraw = bit_cast>(int_raw); + raw_bits_ = iraw[1]; + break; + } + } + + + // Supports possibly narrowing (lossy) conversion from any integer type. + // Equivalent to bfloat16_t{static_cast(i)}, but significantly faster. + // Note: This constructor is "explicit" by default, but can be adjusted + // to allow implicit conversion to bfloat16_t by defining the macro + // BIOVAULT_BFLOAT16_CONVERTING_CONSTRUCTORS. + template ::value>::type> +#ifndef BIOVAULT_BFLOAT16_CONVERTING_CONSTRUCTORS + explicit +#endif + bfloat16_t(const IntegerType i) + : raw_bits_{ convert_bits_of_normal_or_zero( + bit_cast(static_cast(i))) } + { + } + + bfloat16_t& operator=(const float f) { + return (*this) = bfloat16_t{ f }; + } + + template ::value>::type> + bfloat16_t& operator=(const IntegerType i) { + // Call the converting constructor that is optimized for integer types, + // followed by the fast defaulted move-assignment operator. + return (*this) = bfloat16_t{ i }; + } + + // NOLINTNEXTLINE Allow implicit conversion to float, because it is lossless. + operator float() const { + // Implementation originally from: + // https://github.com/oneapi-src/oneDNN/blob/v1.7/src/cpu/bfloat16.cpp#L75-L76 + std::array iraw = { {0, raw_bits_} }; + return bit_cast(iraw); + } + + bfloat16_t& operator+=(const float a) { + (*this) = bfloat16_t{ float{*this} + a }; + return *this; + } + + friend BIOVAULT_BFLOAT16_CONSTEXPR uint16_t get_raw_bits(const bfloat16_t&); + }; + + // Allows retrieving the raw bits of a bfloat16. Equivalent to C++20 + // std::bit_cast(bf16) (which is more generic, of course.) + inline BIOVAULT_BFLOAT16_CONSTEXPR std::uint16_t get_raw_bits(const bfloat16_t& bf16) + { + return bf16.raw_bits_; + } + + static_assert(sizeof(bfloat16_t) == 2, "bfloat16_t must be 2 bytes"); + +} + +#endif \ No newline at end of file diff --git a/src/detail/include/buffer.hpp b/src/detail/include/buffer.hpp new file mode 100644 index 000000000..a8861ac62 --- /dev/null +++ b/src/detail/include/buffer.hpp @@ -0,0 +1,1075 @@ +#if defined(FLM_USE_HRX) +/// \file buffer.hpp +/// \brief Buffer and bytes class for memory management +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This class is used to manage the memory. +#pragma once +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#define FLM_DEVICE_BUFFER + +#ifdef FLM_DEVICE_BUFFER +#include "hrx_cpp/hrx_cpp.hpp" +#endif + +#include "utils/debug_utils.hpp" + +/// \brief bytes class +/// \note This is a buffer wrapper that maps to a bo_buffer or other memory without performing a deep copy. +/// \note A copy (or mapping) does not duplicate the underlying memory; it only maps the pointer. +class bytes { +protected: + std::shared_ptr owned_data_; + uint8_t* data_; + size_t size_; + bool is_owner_; +#ifdef FLM_DEVICE_BUFFER + bool is_bo_owner_; + hrx::bo* bo_; + std::shared_ptr owned_bo_; +#endif + +public: + /// \brief constructor + /// \note This is a buffer wrapper that maps to a bo_buffer or other memory without performing a deep copy. + /// \note A copy (or mapping) does not duplicate the underlying memory; it only maps the pointer. + bytes() : data_(nullptr), size_(0), is_owner_(false) +#ifdef FLM_DEVICE_BUFFER + , is_bo_owner_(false), bo_(nullptr), owned_bo_(nullptr) +#endif + {} + + /// \brief copy constructor + /// \param other the other bytes + /// \note Still a shallow copy of the payload, but the copy shares ownership + /// of the allocation via shared_ptr. hrx::bo deletes its copy ctor, so + /// a unique_ptr copy that dropped ownership left data_ pointing at an + /// unmapped VA after ~bo() — a hard SIGSEGV rather than stale heap. + bytes(const bytes& other) : owned_data_(other.owned_data_), data_(other.data_), size_(other.size_), is_owner_(other.is_owner_) +#ifdef FLM_DEVICE_BUFFER + , is_bo_owner_(other.is_bo_owner_), bo_(other.bo_), owned_bo_(other.owned_bo_) +#endif + {} + + /// \brief move constructor + /// \param other the other bytes + bytes(bytes&& other) noexcept + : owned_data_(std::move(other.owned_data_)), data_(other.data_), size_(other.size_), is_owner_(other.is_owner_) +#ifdef FLM_DEVICE_BUFFER + , is_bo_owner_(other.is_bo_owner_), bo_(other.bo_), owned_bo_(std::move(other.owned_bo_)) +#endif + { + other.data_ = nullptr; + other.size_ = 0; + other.is_owner_ = false; +#ifdef FLM_DEVICE_BUFFER + other.is_bo_owner_ = false; + other.bo_ = nullptr; + other.owned_bo_ = nullptr; +#endif + } + + /// \brief constructor + /// \param size the size + bytes(size_t size) + : size_(size), is_owner_(true) +#ifdef FLM_DEVICE_BUFFER + , is_bo_owner_(false), bo_(nullptr), owned_bo_(nullptr) +#endif + { + if (size > 0 && size < 8ull * 1024 * 1024 * 1024){ + try { + owned_data_ = std::shared_ptr(new uint8_t[size]()); + } + catch (const std::bad_alloc& e) { + throw std::runtime_error(std::string("Failed to allocate bytes of size ") + std::to_string(size) + ": " + e.what()); + } + data_ = owned_data_.get(); + } + else{ + throw std::runtime_error("Invalid size for bytes allocation"); + } + } + + /// \brief constructor + /// \param data the data + /// \param size the size + bytes(uint8_t* data, size_t size) + : owned_data_(nullptr), data_(data), size_(size), is_owner_(false) +#ifdef FLM_DEVICE_BUFFER + , is_bo_owner_(false), bo_(nullptr), owned_bo_(nullptr) +#endif + {} + +#ifdef FLM_DEVICE_BUFFER + /// \brief constructor + /// \param bo the bo + bytes(hrx::bo& bo) + : owned_data_(nullptr), data_(bo.map()), size_(bo.size()), is_owner_(false), is_bo_owner_(false), bo_(&bo), owned_bo_(nullptr) + {} + + /// \brief constructor + /// \param size the size + /// \param device the device + /// \param kernel the kernel + /// \param group_id the group id + /// \param flags the flags + bytes(hrx::device& device, size_t size) + : owned_data_(nullptr), size_(size), is_owner_(false), is_bo_owner_(true) + { + if (size > 3ull * 1024 * 1024 * 1024 || size == 0){ + throw std::runtime_error("Invalid size for bytes allocation"); + } + size_t alignment = 1024 * 1024; + size_t padded_size = (size + alignment - 1) / alignment * alignment; // 1MB alignment + + try { + owned_bo_ = std::make_shared(device, padded_size); + } + catch (const std::exception& e) { + throw std::runtime_error(std::string("Failed to allocate hrx::ext::bo: ") + e.what()); + } + + // uint64_t bo_address = reinterpret_cast(owned_bo_->map()); + // while ( ((bo_address & 0xF0000000) == 0x60000000) || + // ((bo_address & 0xF0000000) == 0x70000000) ) { + + // owned_bo_ = std::make_unique(device, padded_size); + // //header_print("info", "Re-allocating proj_weights for layer " + std::to_string(i) + " to avoid address in 0x60000000 - 0x7FFFFFFF, new address: " + std::to_string(reinterpret_cast(proj_weights[i].data()))); + // bo_address = reinterpret_cast(owned_bo_->map()); + // } + + data_ = owned_bo_->map(); + bo_ = owned_bo_.get(); + } +#endif + + /// \brief destructor + virtual ~bytes() { + if (is_owner_) { + owned_data_.reset(); + } + data_ = nullptr; +#ifdef FLM_DEVICE_BUFFER + if (is_bo_owner_) { + owned_bo_.reset(); + } + bo_ = nullptr; +#endif + } + + /// \brief copy assignment operator + /// \param other the other bytes + bytes& operator=(const bytes& other) { + if (this != &other) { + // Shares ownership rather than dropping it; see the copy constructor note. + owned_data_ = other.owned_data_; + data_ = other.data_; + size_ = other.size_; + is_owner_ = other.is_owner_; +#ifdef FLM_DEVICE_BUFFER + owned_bo_ = other.owned_bo_; + is_bo_owner_ = other.is_bo_owner_; + bo_ = other.bo_; +#endif + } + return *this; + } + + /// \brief move assignment operator + /// \param other the other bytes + bytes& operator=(bytes&& other) noexcept { + if (this != &other) { + if (is_owner_){ + owned_data_.reset(); + } + owned_data_ = std::move(other.owned_data_); + data_ = other.data_; + size_ = other.size_; + is_owner_ = other.is_owner_; +#ifdef FLM_DEVICE_BUFFER + if (is_bo_owner_){ + owned_bo_.reset(); + } + is_bo_owner_ = other.is_bo_owner_; + owned_bo_ = std::move(other.owned_bo_); + bo_ = other.bo_; + other.bo_ = nullptr; + other.is_bo_owner_ = false; +#endif + other.data_ = nullptr; + other.size_ = 0; + other.is_owner_ = false; + } + return *this; + } + + /// \brief operator [] + /// \param index the index + /// \return the value + uint8_t& operator[](size_t index) { + assert(data_ && index < size_); + return data_[index]; + } + + /// \brief operator [] + /// \param index the index + /// \return the value + const uint8_t& operator[](size_t index) const { + assert(data_ && index < size_); + return data_[index]; + } + + size_t size() const { return size_; } + uint8_t* data() const { return data_; } + uint8_t* bdata() const { return data_; } + uint8_t* begin() const { return data_; } + uint8_t* end() const { return data_ + size_; } + + /// \brief copy from + /// \param src the source + /// \param size the size + void copy_from(const uint8_t* src, size_t size) { + assert(size <= size_); + std::memcpy(data_, src, size); + } + + /// \brief resize + /// \param new_size the new size + /// \note Zero-fills, matching bytes(size_t). Callers treat a resized buffer + /// the same as a constructed one and only write the rows they use. + void resize(size_t new_size) { +#ifdef FLM_DEVICE_BUFFER + assert(!is_bo_owner_); +#endif + if (data_ != nullptr && !is_owner_) { + throw std::runtime_error("Cannot resize a non-owner buffer"); + } + if (new_size == 0) { + throw std::runtime_error("Cannot resize to zero size"); + } + try { + owned_data_.reset(new uint8_t[new_size]()); + } + catch (const std::bad_alloc& e) { + throw std::runtime_error(std::string("Failed to allocate bytes of size ") + std::to_string(new_size) + ": " + e.what()); + } + data_ = owned_data_.get(); + size_ = new_size; + is_owner_ = true; + } + + /// \brief free, release the memory or the bo + void free() { +#ifdef FLM_DEVICE_BUFFER + assert(!is_bo_owner_); +#endif + if (is_owner_){ + owned_data_.reset(); + } + data_ = nullptr; + size_ = 0; + is_owner_ = false; +#ifdef FLM_DEVICE_BUFFER + if (is_bo_owner_){ + owned_bo_.reset(); + } + is_bo_owner_ = false; + bo_ = nullptr; +#endif + } + + /// \brief reserve + /// \param size the size + void reserve(size_t size) { resize(size); } + + /// \brief release + void release() { free(); } + + /// \brief is owner + /// \return the is owner + bool is_owner() const { return is_owner_; } +#ifdef FLM_DEVICE_BUFFER + /// \brief is bo owner + /// \return the is bo owner + bool is_bo_owner() const { return is_bo_owner_; } + + /// \brief does this buffer have a device bo attached at all + /// \return true if bo() / sync_*_device() are safe to call + /// \note is_bo_owner() only tells whether *this* object will free the bo; a view + /// onto someone else's bo is still device backed. + bool has_bo() const { return bo_ != nullptr; } + + /// \brief sync to device (host writes -> device) + void sync_to_device() { assert(bo_); bo_->flush(); } + + /// \brief sync from device (device writes -> host) + void sync_from_device() { assert(bo_); bo_->invalidate(); } + + /// \brief bo + /// \return the bo + hrx::bo& bo() { assert(bo_); return *bo_; } +#endif + + /// \brief from file + /// \param filename the filename + /// \param offset the offset + /// \param size the size + void from_file(const std::string& filename, size_t offset = 0, size_t size = 0) { + std::ifstream file(filename, std::ios::binary); + if (!file.is_open()) { + throw std::runtime_error("Failed to open file: " + filename); + } + file.seekg(0, std::ios::end); + size_t file_size = file.tellg(); + file.seekg(0, std::ios::beg); + if (size == 0) size = file_size; + assert(size <= file_size); + assert(offset + size <= size_); + file.read(reinterpret_cast(data_) + offset, size); + file.close(); + } +}; + +/// \brief buffer class +/// \note This class wraps a data type T over the underlying byte buffer. +template +class buffer : public bytes { +public: + /// \brief constructor + buffer() : bytes() {} + + /// \brief constructor + /// \param count the count + buffer(size_t count) : bytes(count * sizeof(T)) {} + + /// \brief constructor + /// \param data the data + /// \param count the count + buffer(T* data, size_t count) + : bytes(reinterpret_cast(data), count * sizeof(T)) {} + + /// \brief shallow copy constructor + /// \param other the other buffer + buffer(const buffer& other) : bytes(other) {} + + /// \brief move constructor + /// \param other the other buffer + /// \note Transfers ownership (owned_data_/owned_bo_) so a returned buffer does not dangle. + buffer(buffer&& other) noexcept : bytes(std::move(other)) {} + +#ifdef FLM_DEVICE_BUFFER + /// \brief constructor + /// \param bo the bo + buffer(hrx::bo& bo) : bytes(bo) {} + + /// \brief constructor + /// \param count the count + /// \param device the device + /// \param kernel the kernel + /// \param group_id the group id + /// \param flags the flags + buffer(hrx::device& device, size_t count) + : bytes(device, count * sizeof(T)) {} +#endif + + /// \brief constructor + /// \param vec the vector + buffer(const std::vector& vec) + : bytes(reinterpret_cast(const_cast(vec.data())), vec.size() * sizeof(T)) + { + } + + /// \brief constructor + /// \param vec the vector + /// \warning This also creates a shallow mapping. + /// \warning The caller must ensure that the vector is not used (and remains valid) + /// \warning after constructing this buffer. + buffer(std::vector&& vec) + : bytes(reinterpret_cast(vec.data()), vec.size() * sizeof(T)) + { + } + + /// \brief copy from + /// \param vec the vector + void copy_from(const std::vector& vec) { + if (vec.size() * sizeof(T) != this->size_) { + throw std::runtime_error("Size mismatch in copy_from(vector)"); + } + std::memcpy(data_, vec.data(), size_); + } + + /// \brief cast to another type + /// \tparam U the type + /// \return the buffer + template + buffer cast_to() { + size_t newCount = size_ / sizeof(U); + return buffer(reinterpret_cast(data_), newCount); + } + + /// \brief as bytes + /// \return the bytes + const bytes as_bytes() const { + return *this; + } + + /// \brief move assignment operator + /// \param other the other buffer + /// \return the buffer + buffer& operator=(buffer&& other) noexcept { + bytes::operator=(std::move(other)); + return *this; + } + + /// \brief copy assignment operator + /// \param other the other buffer + /// \return the buffer + buffer& operator=(const buffer& other) { + bytes::operator=(other); + return *this; + } + + /// \brief operator [] + /// \param index the index + /// \return the value + T& operator[](size_t index) { + assert(data_ != nullptr); + assert(index < size()); + assert(index >= 0); + return reinterpret_cast(data_)[index]; + } + + /// \brief operator [] + /// \param index the index + /// \return the value + const T& operator[](size_t index) const { + assert(data_ != nullptr); + assert(index < size()); + assert(index >= 0); + return reinterpret_cast(data_)[index]; + } + + /// \brief size, number of elements + /// \return the size + size_t size() const { return size_ / sizeof(T); } + + /// \brief data + /// \return the data + T* data() const { return reinterpret_cast(data_); } + + /// \brief begin + /// \return the pointer to the first element + T* begin() const { return reinterpret_cast(data_); } + + /// \brief end + /// \return the pointer to the last element + T* end() const { return reinterpret_cast(data_) + size(); } + + /// \brief resize + /// \param count: the number of elements + void resize(size_t count) { + bytes::resize(count * sizeof(T)); + } + + /// \brief reserve + /// \param count: the number of elements + void reserve(size_t count) { + bytes::reserve(count * sizeof(T)); + } + + /// \brief memset + /// \param value the value + void memset(T value) { + T* ptr = data(); + for (size_t i = 0; i < size(); i++) { + ptr[i] = value; + } + } + + /// \brief copy from + /// \param other the other bytes + void copy_from(const bytes& other) { + if (size_ != other.size()) { + throw std::runtime_error("Size mismatch in copy_from(bytes)"); + } + memcpy(data_, other.data(), size_); // size_ is already in bytes. + } + + /// \brief copy from + /// \param other the other buffer + void copy_from(const buffer& other) { + if (size() != other.size()) { + throw std::runtime_error("Size mismatch in copy_from(buffer)"); + } + memcpy(data_, other.bdata(), size_); // size_ is already in bytes. + } + + /// \brief copy from + /// \param data the data + /// \param size the number of elements + void copy_from(T* data, size_t size) { + if (size > this->size()) { + throw std::runtime_error("Size mismatch in copy_from(pointer)"); + } + memcpy(data_, data, size * sizeof(T)); + } + + /// \brief as bytes + /// \return the bytes + bytes& as_bytes() { + return *this; + } + + /// \brief from file + /// \param filename the filename + /// \param offset the offset + /// \param size the number of elements + void from_file(const std::string& filename, size_t offset = 0, size_t size = 0) { + bytes::from_file(filename, offset * sizeof(T), size * sizeof(T)); + } +}; +#else +/// \file buffer.hpp +/// \brief Buffer and bytes class for memory management +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This class is used to manage the memory. +#pragma once +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#define __XRT__ + +#ifdef __XRT__ +#include "xrt/xrt_bo.h" +#include "xrt/xrt_kernel.h" +#include "xrt/xrt_device.h" +#include "xrt/experimental/xrt_ext.h" +#endif + +#include "utils/debug_utils.hpp" + +/// \brief bytes class +/// \note This is a buffer wrapper that maps to a bo_buffer or other memory without performing a deep copy. +/// \note A copy (or mapping) does not duplicate the underlying memory; it only maps the pointer. +class bytes { +protected: + std::unique_ptr owned_data_; + uint8_t* data_; + size_t size_; + bool is_owner_; +#ifdef __XRT__ + bool is_bo_owner_; + xrt::bo* bo_; + std::unique_ptr owned_bo_; +#endif + +public: + /// \brief constructor + /// \note This is a buffer wrapper that maps to a bo_buffer or other memory without performing a deep copy. + /// \note A copy (or mapping) does not duplicate the underlying memory; it only maps the pointer. + bytes() : data_(nullptr), size_(0), is_owner_(false) +#ifdef __XRT__ + , is_bo_owner_(false), bo_(nullptr), owned_bo_(nullptr) +#endif + {} + + /// \brief copy constructor + /// \param other the other bytes + bytes(const bytes& other) : owned_data_(nullptr), data_(other.data_), size_(other.size_), is_owner_(false) +#ifdef __XRT__ + , is_bo_owner_(false), bo_(other.bo_), owned_bo_(nullptr) +#endif + {} + + /// \brief move constructor + /// \param other the other bytes + bytes(bytes&& other) noexcept + : owned_data_(std::move(other.owned_data_)), data_(other.data_), size_(other.size_), is_owner_(other.is_owner_) +#ifdef __XRT__ + , is_bo_owner_(other.is_bo_owner_), bo_(other.bo_), owned_bo_(std::move(other.owned_bo_)) +#endif + { + other.data_ = nullptr; + other.size_ = 0; + other.is_owner_ = false; +#ifdef __XRT__ + other.is_bo_owner_ = false; + other.bo_ = nullptr; + other.owned_bo_ = nullptr; +#endif + } + + /// \brief constructor + /// \param size the size + bytes(size_t size) + : size_(size), is_owner_(true) +#ifdef __XRT__ + , is_bo_owner_(false), bo_(nullptr), owned_bo_(nullptr) +#endif + { + if (size > 0 && size < 8ull * 1024 * 1024 * 1024){ + try { + owned_data_ = std::make_unique(size); + } + catch (const std::bad_alloc& e) { + throw std::runtime_error(std::string("Failed to allocate bytes of size ") + std::to_string(size) + ": " + e.what()); + } + data_ = owned_data_.get(); + } + else{ + throw std::runtime_error("Invalid size for bytes allocation"); + } + } + + /// \brief constructor + /// \param data the data + /// \param size the size + bytes(uint8_t* data, size_t size) + : owned_data_(nullptr), data_(data), size_(size), is_owner_(false) +#ifdef __XRT__ + , is_bo_owner_(false), bo_(nullptr), owned_bo_(nullptr) +#endif + {} + +#ifdef __XRT__ + /// \brief constructor + /// \param bo the bo + bytes(xrt::bo& bo) + : owned_data_(nullptr), data_(bo.map()), size_(bo.size()), is_owner_(false), is_bo_owner_(false), bo_(&bo), owned_bo_(nullptr) + {} + + /// \brief constructor + /// \param size the size + /// \param device the device + /// \param kernel the kernel + /// \param group_id the group id + /// \param flags the flags + bytes(xrt::device& device, size_t size) + : owned_data_(nullptr), size_(size), is_owner_(false), is_bo_owner_(true) + { + if (size > 3ull * 1024 * 1024 * 1024 || size == 0){ + throw std::runtime_error("Invalid size for bytes allocation"); + } + size_t alignment = 1024 * 1024; + int padded_size = (size + alignment - 1) / alignment * alignment; // 4KB alignment, , (xrt::ext::bo::access_mode)(xrt::ext::bo::access_mode::read_write | xrt::ext::bo::access_mode::process) + + try { + owned_bo_ = std::make_unique(device, padded_size); + } + catch (const std::exception& e) { + throw std::runtime_error(std::string("Failed to allocate xrt::ext::bo: ") + e.what()); + } + + data_ = owned_bo_->map(); + bo_ = owned_bo_.get(); + } +#endif + + /// \brief destructor + virtual ~bytes() { + if (is_owner_) { + owned_data_.reset(); + } + data_ = nullptr; +#ifdef __XRT__ + if (is_bo_owner_) { + owned_bo_.reset(); + } + bo_ = nullptr; +#endif + } + + /// \brief copy assignment operator + /// \param other the other bytes + bytes& operator=(const bytes& other) { + if (this != &other) { + if (is_owner_){ + owned_data_.reset(); + } + data_ = other.data_; + size_ = other.size_; + is_owner_ = false; +#ifdef __XRT__ + if (is_bo_owner_){ + owned_bo_.reset(); + } + is_bo_owner_ = false; + bo_ = other.bo_; +#endif + } + return *this; + } + + /// \brief move assignment operator + /// \param other the other bytes + bytes& operator=(bytes&& other) noexcept { + if (this != &other) { + if (is_owner_){ + owned_data_.reset(); + } + owned_data_ = std::move(other.owned_data_); + data_ = other.data_; + size_ = other.size_; + is_owner_ = other.is_owner_; +#ifdef __XRT__ + if (is_bo_owner_){ + owned_bo_.reset(); + } + is_bo_owner_ = other.is_bo_owner_; + owned_bo_ = std::move(other.owned_bo_); + bo_ = other.bo_; + other.bo_ = nullptr; + other.is_bo_owner_ = false; +#endif + other.data_ = nullptr; + other.size_ = 0; + other.is_owner_ = false; + } + return *this; + } + + /// \brief operator [] + /// \param index the index + /// \return the value + uint8_t& operator[](size_t index) { + assert(data_ && index < size_); + return data_[index]; + } + + /// \brief operator [] + /// \param index the index + /// \return the value + const uint8_t& operator[](size_t index) const { + assert(data_ && index < size_); + return data_[index]; + } + + size_t size() const { return size_; } + uint8_t* data() const { return data_; } + uint8_t* bdata() const { return data_; } + uint8_t* begin() const { return data_; } + uint8_t* end() const { return data_ + size_; } + + /// \brief copy from + /// \param src the source + /// \param size the size + void copy_from(const uint8_t* src, size_t size) { + assert(size <= size_); + std::memcpy(data_, src, size); + } + + /// \brief resize + /// \param new_size the new size + void resize(size_t new_size) { +#ifdef __XRT__ + assert(!is_bo_owner_); +#endif + if (data_ != nullptr && !is_owner_) { + throw std::runtime_error("Cannot resize a non-owner buffer"); + } + if (new_size == 0) { + throw std::runtime_error("Cannot resize to zero size"); + } + try { + owned_data_.reset(new uint8_t[new_size]); + } + catch (const std::bad_alloc& e) { + throw std::runtime_error(std::string("Failed to allocate bytes of size ") + std::to_string(new_size) + ": " + e.what()); + } + data_ = owned_data_.get(); + size_ = new_size; + is_owner_ = true; + } + + /// \brief free, release the memory or the bo + void free() { +#ifdef __XRT__ + assert(!is_bo_owner_); +#endif + if (is_owner_){ + owned_data_.reset(); + } + data_ = nullptr; + size_ = 0; + is_owner_ = false; +#ifdef __XRT__ + if (is_bo_owner_){ + owned_bo_.reset(); + } + is_bo_owner_ = false; + bo_ = nullptr; +#endif + } + + /// \brief reserve + /// \param size the size + void reserve(size_t size) { resize(size); } + + /// \brief release + void release() { free(); } + + /// \brief is owner + /// \return the is owner + bool is_owner() const { return is_owner_; } +#ifdef __XRT__ + /// \brief is bo owner + /// \return the is bo owner + bool is_bo_owner() const { return is_bo_owner_; } + + /// \brief does this buffer have a device bo attached at all + /// \return true if bo() / sync_*_device() are safe to call + /// \note is_bo_owner() only tells whether *this* object will free the bo, a view + /// onto someone else's bo is still device backed. Self tests want this one. + bool has_bo() const { return bo_ != nullptr; } + + /// \brief sync to device + void sync_to_device() { assert(bo_); bo_->sync(XCL_BO_SYNC_BO_TO_DEVICE); } + + /// \brief sync from device + void sync_from_device() { assert(bo_); bo_->sync(XCL_BO_SYNC_BO_FROM_DEVICE); } + + /// \brief bo + /// \return the bo + xrt::bo& bo() { assert(bo_); return *bo_; } +#endif + + /// \brief from file + /// \param filename the filename + /// \param offset the offset + /// \param size the size + void from_file(const std::string& filename, size_t offset = 0, size_t size = 0) { + std::ifstream file(filename, std::ios::binary); + if (!file.is_open()) { + throw std::runtime_error("Failed to open file: " + filename); + } + file.seekg(0, std::ios::end); + size_t file_size = file.tellg(); + file.seekg(0, std::ios::beg); + if (size == 0) size = file_size; + assert(size <= file_size); + assert(offset + size <= size_); + file.read(reinterpret_cast(data_) + offset, size); + file.close(); + } +}; + +/// \brief buffer class +/// \note This class wraps a data type T over the underlying byte buffer. +template +class buffer : public bytes { +public: + /// \brief constructor + buffer() : bytes() {} + + /// \brief constructor + /// \param count the count + buffer(size_t count) : bytes(count * sizeof(T)) {} + + /// \brief constructor + /// \param data the data + /// \param count the count + buffer(T* data, size_t count) + : bytes(reinterpret_cast(data), count * sizeof(T)) {} + + /// \brief shallow copy constructor + /// \param other the other buffer + buffer(const buffer& other) : bytes(other) {} + + /// \brief move constructor + /// \param other the other buffer + /// \note Transfers ownership (owned_data_/owned_bo_) so a returned buffer does not dangle. + buffer(buffer&& other) noexcept : bytes(std::move(other)) {} + +#ifdef __XRT__ + /// \brief constructor + /// \param bo the bo + buffer(xrt::bo& bo) : bytes(bo) {} + + /// \brief constructor + /// \param count the count + /// \param device the device + /// \param kernel the kernel + /// \param group_id the group id + /// \param flags the flags + buffer(xrt::device& device, size_t count) + : bytes(device, count * sizeof(T)) {} +#endif + + /// \brief constructor + /// \param vec the vector + buffer(const std::vector& vec) + : bytes(reinterpret_cast(const_cast(vec.data())), vec.size() * sizeof(T)) + { + } + + /// \brief constructor + /// \param vec the vector + /// \warning This also creates a shallow mapping. + /// \warning The caller must ensure that the vector is not used (and remains valid) + /// \warning after constructing this buffer. + buffer(std::vector&& vec) + : bytes(reinterpret_cast(vec.data()), vec.size() * sizeof(T)) + { + } + + /// \brief copy from + /// \param vec the vector + void copy_from(const std::vector& vec) { + if (vec.size() * sizeof(T) != this->size_) { + throw std::runtime_error("Size mismatch in copy_from(vector)"); + } + std::memcpy(data_, vec.data(), size_); + } + + /// \brief cast to another type + /// \tparam U the type + /// \return the buffer + template + buffer cast_to() { + size_t newCount = size_ / sizeof(U); + return buffer(reinterpret_cast(data_), newCount); + } + + /// \brief as bytes + /// \return the bytes + const bytes as_bytes() const { + return *this; + } + + /// \brief move assignment operator + /// \param other the other buffer + /// \return the buffer + buffer& operator=(buffer&& other) noexcept { + bytes::operator=(std::move(other)); + return *this; + } + + /// \brief copy assignment operator + /// \param other the other buffer + /// \return the buffer + buffer& operator=(const buffer& other) { + bytes::operator=(other); + return *this; + } + + /// \brief operator [] + /// \param index the index + /// \return the value + T& operator[](size_t index) { + assert(data_ != nullptr); + assert(index < size()); + assert(index >= 0); + return reinterpret_cast(data_)[index]; + } + + /// \brief operator [] + /// \param index the index + /// \return the value + const T& operator[](size_t index) const { + assert(data_ != nullptr); + assert(index < size()); + assert(index >= 0); + return reinterpret_cast(data_)[index]; + } + + /// \brief size, number of elements + /// \return the size + size_t size() const { return size_ / sizeof(T); } + + /// \brief data + /// \return the data + T* data() const { return reinterpret_cast(data_); } + + /// \brief begin + /// \return the pointer to the first element + T* begin() const { return reinterpret_cast(data_); } + + /// \brief end + /// \return the pointer to the last element + T* end() const { return reinterpret_cast(data_) + size(); } + + /// \brief resize + /// \param count: the number of elements + void resize(size_t count) { + bytes::resize(count * sizeof(T)); + } + + /// \brief reserve + /// \param count: the number of elements + void reserve(size_t count) { + bytes::reserve(count * sizeof(T)); + } + + /// \brief memset + /// \param value the value + void memset(T value) { + T* ptr = data(); + for (size_t i = 0; i < size(); i++) { + ptr[i] = value; + } + } + + /// \brief copy from + /// \param other the other bytes + void copy_from(const bytes& other) { + if (size_ != other.size()) { + throw std::runtime_error("Size mismatch in copy_from(bytes)"); + } + memcpy(data_, other.data(), size_); // size_ is already in bytes. + } + + /// \brief copy from + /// \param other the other buffer + void copy_from(const buffer& other) { + if (size() != other.size()) { + throw std::runtime_error("Size mismatch in copy_from(buffer)"); + } + memcpy(data_, other.bdata(), size_); // size_ is already in bytes. + } + + /// \brief copy from + /// \param data the data + /// \param size the number of elements + void copy_from(T* data, size_t size) { + if (size > this->size()) { + throw std::runtime_error("Size mismatch in copy_from(pointer)"); + } + memcpy(data_, data, size * sizeof(T)); + } + + /// \brief as bytes + /// \return the bytes + bytes& as_bytes() { + return *this; + } + + /// \brief from file + /// \param filename the filename + /// \param offset the offset + /// \param size the number of elements + void from_file(const std::string& filename, size_t offset = 0, size_t size = 0) { + bytes::from_file(filename, offset * sizeof(T), size * sizeof(T)); + } +}; +#endif diff --git a/src/detail/include/causal_lm.hpp b/src/detail/include/causal_lm.hpp new file mode 100644 index 000000000..d4682a95a --- /dev/null +++ b/src/detail/include/causal_lm.hpp @@ -0,0 +1,64 @@ +/// \file causal_lm.hpp +/// \brief causal_lm class +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This class is a virtual class for causal language models +/// \note All other models should inherit from this class so that they can be used in the same way. +#pragma once +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "buffer.hpp" + +/// \brief causal_lm class +class causal_lm { +public: + causal_lm(){} + virtual ~causal_lm(){} + + /// \brief forward the causal_lm + /// \param ids the ids + /// \return the output + virtual buffer forward(int ids) = 0; + + /// \brief prefill the causal_lm + /// \param ids the ids + /// \return the output + virtual buffer prefill(std::vector& ids, void* payload = nullptr) = 0; + + /// \brief set the context length + /// \param L the context length + virtual void set_context_length(int L) = 0; + + /// \brief load the weights + /// \param q4nx the q4nx + virtual void load_weights(Q4NX& q4nx) = 0; + + /// \brief update the max length + /// \param MAX_L the max length + virtual void update_max_length(uint32_t MAX_L) = 0; + + /// \brief clear the context + virtual void clear_context() = 0; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + virtual buffer get_k_cache(int layer_idx, int idx) = 0; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + virtual buffer get_v_cache(int layer_idx, int idx) = 0; + + /// \brief get the current context length + /// \return the current context length + virtual int get_current_context_length() = 0; + + virtual int checkpoint() = 0; + + virtual int restore() = 0; +}; \ No newline at end of file diff --git a/src/detail/include/embedding_model.hpp b/src/detail/include/embedding_model.hpp new file mode 100644 index 000000000..127919c90 --- /dev/null +++ b/src/detail/include/embedding_model.hpp @@ -0,0 +1,29 @@ +/// \file causal_lm.hpp +/// \brief causal_lm class +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This class is a virtual class for causal language models +/// \note All other models should inherit from this class so that they can be used in the same way. +#pragma once +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "buffer.hpp" + + + +/// \brief causal_lm class +class embedding_model { +public: + embedding_model(){} + virtual ~embedding_model(){} + + /// \brief load the weights + /// \param q4nx the q4nx + virtual void load_weights(Q4NX& q4nx) = 0; + /// \brief embed the embedding_model + /// \param x the input + /// \return the output + virtual buffer embed(std::vector& tokens) = 0; +}; \ No newline at end of file diff --git a/src/detail/include/flm_override.hpp b/src/detail/include/flm_override.hpp new file mode 100644 index 000000000..2b0dc4dfe --- /dev/null +++ b/src/detail/include/flm_override.hpp @@ -0,0 +1,30 @@ +#ifndef FLM_OVERRIDE_HPP +#define FLM_OVERRIDE_HPP + +/// \file flm_override.hpp +/// \brief Compile-time override points for a model's operator dispatch. +/// +/// `FLM_OVERRIDE(name, expr, ...)` expands to `expr`. A build that points +/// `FLM_OVERRIDES` at a header may redefine it to dispatch on `name`; every +/// other build emits the code it would have emitted without the annotation. +/// +/// g++ -DFLM_OVERRIDES='"iron_overrides.h"' ... +/// +/// `name` is a token, not a string, so it costs nothing. The trailing +/// arguments carry values an override needs and the call itself does not. The +/// default expansion discards them unevaluated, so they must be free of side +/// effects, and a value computed only to be passed here belongs inside the +/// annotation too, or `-Wall` reports it unused. +/// +/// A statement-position hook that only hands an override some context writes +/// `(void)0` as its expression. + +#ifdef FLM_OVERRIDES +#include FLM_OVERRIDES +#endif + +#ifndef FLM_OVERRIDE +#define FLM_OVERRIDE(name, expr, ...) (expr) +#endif + +#endif // FLM_OVERRIDE_HPP diff --git a/src/detail/include/hrx_cpp/hrx_cpp.hpp b/src/detail/include/hrx_cpp/hrx_cpp.hpp new file mode 100644 index 000000000..a5b494800 --- /dev/null +++ b/src/detail/include/hrx_cpp/hrx_cpp.hpp @@ -0,0 +1,528 @@ +/// \file hrx_cpp.hpp +/// \brief Minimal C++ `namespace hrx` providing the device/buffer/kernel/run +/// API that FastFlowLM uses, implemented directly on top of libhrx +/// (hrx_runtime.h). +/// +/// NPU control code goes straight from npu_sequence::dump() into an HRX XADX +/// "direct executable"; there is no separate assembler step. +/// +/// Coherence model: this amdxdna device is a single HOST_ONLY heap (cached +/// host DRAM the NPU can snoop). Buffers are allocated HOST_VISIBLE | +/// HOST_CACHED | DEVICE_VISIBLE (0x1A), mapped once (persistent). The heap +/// is not HOST_COHERENT, so FastFlowLM's sync_to_device()/sync_from_device() +/// map to hrx_buffer_flush_range()/hrx_buffer_invalidate_range(). Dispatch is +/// hrx_stream_dispatch() + hrx_stream_synchronize(). +#pragma once + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "hrx_amdxdna.h" +#include "hrx_runtime.h" + +// ---- ert_cmd_state: command states that FLM's npu_utils returns/maps. +#ifndef FLM_ERT_CMD_STATE_DEFINED +#define FLM_ERT_CMD_STATE_DEFINED +enum ert_cmd_state { + ERT_CMD_STATE_NEW = 1, + ERT_CMD_STATE_QUEUED = 2, + ERT_CMD_STATE_RUNNING = 3, + ERT_CMD_STATE_COMPLETED = 4, + ERT_CMD_STATE_ERROR = 5, + ERT_CMD_STATE_ABORT = 6, + ERT_CMD_STATE_SUBMITTED = 7, + ERT_CMD_STATE_TIMEOUT = 8, + ERT_CMD_STATE_NORESPONSE = 9, + ERT_CMD_STATE_SKERROR = 10, + ERT_CMD_STATE_SKCRASHED = 11, + ERT_CMD_STATE_MAX = 12, +}; +#endif + +namespace hrx { + +// --------------------------------------------------------------------------- +// Process-wide HRX runtime (one device + one stream), lazily initialized. +// --------------------------------------------------------------------------- +class Runtime { +public: + static Runtime& get() { + static Runtime r; + return r; + } + hrx_device_t dev = nullptr; + hrx_stream_t stream = nullptr; + bool ok = false; + + void ensure() { + if (ok) return; + // Assign device/stream only after both succeed. A failed stream_create + // used to leave `dev` non-null with `ok == false`, and `if (dev) return` + // then wedged every later call. + hrx_device_t d = nullptr; + hrx_stream_t s = nullptr; + hrx_status_t init_status = hrx_gpu_initialize(0); + const bool initialized = + hrx_status_is_ok(init_status) || + hrx_status_code(init_status) == HRX_STATUS_ALREADY_EXISTS; + hrx_status_ignore(init_status); + if (initialized && + hrx_status_is_ok(hrx_gpu_device_get(0, &d)) && + hrx_status_is_ok(hrx_stream_create(d, 0, &s))) { + dev = d; + stream = s; + ok = true; + if (std::getenv("HRX_DEBUG")) + std::fprintf(stderr, "[hrx] device+stream initialized (Runtime@%p)\n", + (void*)this); + } else { + std::fprintf(stderr, "[hrx] device init FAILED\n"); + ok = false; + dev = nullptr; + stream = nullptr; + } + } + +private: + Runtime() = default; +}; + +inline Runtime& rt() { + Runtime& r = Runtime::get(); + r.ensure(); + return r; +} + +// Report (do not swallow) an HRX error. Returns true if status was an error. +// FLM dispatch silently ignored synchronize/dispatch failures, which turns a +// failed ERT_CMD_CHAIN (e.g. a missing host patch table) into silent no-op +// dispatches -> garbage output at full speed. Always surface these. +inline bool hrx_report(hrx_status_t s, const char* where) { + if (hrx_status_is_ok(s)) return false; + char* m = nullptr; + size_t mn = 0; + hrx_status_to_string(s, &m, &mn); + std::fprintf(stderr, "[hrx][ERROR] %s: %s\n", where, m ? m : "?"); + hrx_status_free_message(m); + hrx_status_ignore(s); + return true; +} + +// --------------------------------------------------------------------------- +// Executable cache: build one HRX XADX executable per distinct executable +// identity (xclbin + control program + host patch table) and resolve its export +// ordinal once. +// --------------------------------------------------------------------------- +struct CachedExe { + hrx_executable_t exe = nullptr; + uint32_t ord = 0; +}; + +inline void append_key_bytes(std::string& key, const void* data, size_t byte_count) { + const uint64_t length = static_cast(byte_count); + key.append(reinterpret_cast(&length), sizeof(length)); + if (data && byte_count) { + key.append(reinterpret_cast(data), byte_count); + } +} + +inline hrx_executable_t build_or_get_executable( + const std::vector& xclbin_bytes, const uint32_t* cc, size_t n, + uint32_t* ord_out) { + static std::mutex mu; + static std::unordered_map cache; + std::string key; + key.reserve(xclbin_bytes.size() + n * sizeof(uint32_t) + + 2 * sizeof(uint64_t)); + append_key_bytes(key, xclbin_bytes.data(), xclbin_bytes.size()); + append_key_bytes(key, cc, n * sizeof(uint32_t)); + static bool no_cache = std::getenv("HRX_NOCACHE") != nullptr; + std::lock_guard lk(mu); + if (!no_cache) { + auto it = cache.find(key); + if (it != cache.end()) { + if (ord_out) *ord_out = it->second.ord; + return it->second.exe; + } + } + hrx_const_byte_span_t xclbin = {xclbin_bytes.data(), xclbin_bytes.size()}; + hrx_amdxdna_executable_run_t run = {}; + run.record_length = sizeof(run); + run.abi_version = HRX_AMDXDNA_EXECUTABLE_RUN_ABI_VERSION_0; + run.transaction = {reinterpret_cast(cc), + n * sizeof(uint32_t)}; + hrx_amdxdna_executable_entry_point_t entry_point = {}; + entry_point.record_length = sizeof(entry_point); + entry_point.abi_version = HRX_AMDXDNA_EXECUTABLE_ENTRY_POINT_ABI_VERSION_0; + entry_point.name = {"MLIR_AIE", std::strlen("MLIR_AIE")}; + entry_point.context_mode = HRX_AMDXDNA_CONTEXT_MODE_CREATE; + entry_point.runs = &run; + entry_point.run_count = 1; + hrx_amdxdna_executable_create_params_t params = {}; + params.record_length = sizeof(params); + params.abi_version = HRX_AMDXDNA_EXECUTABLE_CREATE_PARAMS_ABI_VERSION_0; + params.xclbins = &xclbin; + params.xclbin_count = 1; + params.entry_points = &entry_point; + params.entry_point_count = 1; + hrx_executable_t exe = nullptr; + uint32_t ord = 0; + hrx_status_t create_status = hrx_amdxdna_executable_create( + rt().dev, ¶ms, &exe); + if (hrx_report(create_status, "hrx_amdxdna_executable_create")) { + exe = nullptr; + } + if (exe && hrx_report(hrx_executable_lookup_export_by_name( + exe, "MLIR_AIE", &ord), + "hrx_executable_lookup_export_by_name")) { + hrx_executable_release(exe); + exe = nullptr; + } + // Never cache a failed create: a null entry would make every later call + // return without dispatching (silent garbage). HRX_NOCACHE still skips the + // map so each call rebuilds; those executables live until process exit. + if (exe && !no_cache) cache.emplace(std::move(key), CachedExe{exe, ord}); + if (ord_out) *ord_out = ord; + return exe; +} + +// --------------------------------------------------------------------------- +// uuid / xclbin / device / hw_context +// --------------------------------------------------------------------------- +class uuid { +public: + unsigned char m_uuid[16] = {0}; +}; + +class xclbin { +public: + std::shared_ptr> bytes_ = + std::make_shared>(); + + xclbin() = default; + explicit xclbin(const std::string& path) { + std::FILE* f = std::fopen(path.c_str(), "rb"); + if (!f) throw std::runtime_error("hrx::xclbin: cannot open " + path); + std::fseek(f, 0, SEEK_END); + long n = std::ftell(f); + std::fseek(f, 0, SEEK_SET); + if (n > 0) { + bytes_->resize(static_cast(n)); + size_t rd = std::fread(bytes_->data(), 1, bytes_->size(), f); + (void)rd; + } + std::fclose(f); + } + + // FLM searches kernels for one whose name starts with "MLIR_AIE"; the HRX + // dispatch path always uses the "MLIR_AIE" export, so a single placeholder + // kernel is sufficient (matches the proven interposer behavior). + class kernel { + public: + std::string name = "MLIR_AIE"; + std::string get_name() const { return name; } + }; + std::vector get_kernels() const { return {kernel{}}; } + uuid get_uuid() const { return uuid{}; } + const std::vector& bytes() const { return *bytes_; } + std::shared_ptr> bytes_shared() const { return bytes_; } +}; + +namespace info { +// Argument to device::get_info<>(). FLM only queries the human-readable +// device name for diagnostics. +enum class device { name, architecture }; +} + +class device { +public: + device() = default; + explicit device(unsigned int /*index*/) { rt(); } + uuid register_xclbin(const xclbin& /*xc*/) { return uuid{}; } + void reset() {} + + // Returns a human-readable device identity string for diagnostics: the + // IREE HAL device name (e.g. "amdxdna") via hrx_device_get_property(). + template + std::string get_info() const { + char buf[128] = {0}; + hrx_device_get_property( + rt().dev, + P == info::device::architecture ? HRX_DEVICE_PROPERTY_ARCHITECTURE + : HRX_DEVICE_PROPERTY_NAME, + buf, sizeof(buf)); + return std::string(buf); + } +}; + +class hw_context { +public: + std::shared_ptr> xclbin_bytes_; + + hw_context() = default; + hw_context(const device& /*dev*/, const xclbin& xc) + : xclbin_bytes_(xc.bytes_shared()) {} + // The (device, uuid) form is accepted for source compatibility; it carries + // no xclbin bytes, so prefer the (device, xclbin) form. + hw_context(const device& /*dev*/, const uuid& /*id*/) {} + + const std::vector& xclbin_bytes() const { + static const std::vector empty; + return xclbin_bytes_ ? *xclbin_bytes_ : empty; + } +}; + +// --------------------------------------------------------------------------- +// Per-buffer host<->device coherence state (dirty tracking). +// +// This path uses ONE persistent host-mapped, device-visible buffer per BO; +// coherence is kept purely with clflush-style range ops. The correctness +// comes entirely from GATING those ops per buffer: +// +// host_dirty : the host wrote bytes that are not yet on the device. Must be +// flushed (h2d) before any dispatch reads the buffer, and only +// then. Cleared by the flush. +// dev_dirty : a dispatch wrote bytes that are not yet visible to the host. +// Must be invalidated (d2h) before the host reads, and only +// then. Cleared by the invalidate. +// +// Why gating matters (this is the multi-turn bug): +// * Unconditional flush-before-dispatch pushes a STALE host copy over a +// device-resident buffer (e.g. kv_caches the previous layer just wrote) -> +// corrupts the KV cache. +// * Unconditional invalidate-on-read DROPS the host's own not-yet-flushed +// writes -> the device never sees them on the next turn. +// This path avoids both by only acting when the matching dirty bit is set; on a +// single turn the bits happen to line up either way, but turn>1 reuses buffers +// whose bits diverge, which is why only later turns broke. +struct BufCoh { + bool host_dirty = true; // freshly-allocated content must reach the device once + bool dev_dirty = false; +}; +inline std::mutex& buf_coh_mu() { + static std::mutex m; + return m; +} +inline std::unordered_map& buf_coh() { + static std::unordered_map m; + return m; +} + +// --------------------------------------------------------------------------- +// Buffers +// --------------------------------------------------------------------------- +class bo { +public: + hrx_buffer_t hbuf_ = nullptr; + void* mapped_ = nullptr; + size_t size_ = 0; + bool owns_ = false; + + bo() = default; + virtual ~bo() { + if (owns_ && hbuf_) { + { + std::lock_guard lk(buf_coh_mu()); + buf_coh().erase(hbuf_); + } + hrx_buffer_release(hbuf_); + } + hbuf_ = nullptr; + mapped_ = nullptr; + } + bo(const bo&) = delete; + bo& operator=(const bo&) = delete; + + template + T map() { + return reinterpret_cast(mapped_); + } + size_t size() const { return size_; } + hrx_buffer_t handle() const { return hbuf_; } + + // sync_to_device(): the host has new data. Record it as host_dirty and defer + // the actual h2d flush to the dispatch that consumes the buffer (matching the + // shim). This is NOT an unconditional flush: a buffer the device owns is never + // marked host_dirty here, so it can never be clobbered. + void flush() { + if (!hbuf_) return; + std::lock_guard lk(buf_coh_mu()); + BufCoh& st = buf_coh()[hbuf_]; + st.host_dirty = true; + st.dev_dirty = false; // host is now the source of truth + } + // sync_from_device(): the host wants to read. Only invalidate if a dispatch + // actually produced new device data (dev_dirty); otherwise this would drop + // the host's own writes. + void invalidate() { + if (!hbuf_) return; + std::lock_guard lk(buf_coh_mu()); + BufCoh& st = buf_coh()[hbuf_]; + if (st.dev_dirty) { + hrx_report(hrx_buffer_invalidate_range(hbuf_, 0, size_), + "hrx_buffer_invalidate_range"); + st.dev_dirty = false; + } + } +}; + +namespace ext { +class bo : public hrx::bo { +public: + bo(const device& /*dev*/, size_t sz) { + Runtime& r = rt(); + size_ = sz; + owns_ = true; + if (!r.ok) throw std::runtime_error("hrx::ext::bo: HRX device unavailable"); + // HOST_VISIBLE | HOST_CACHED | DEVICE_VISIBLE (0x1A). This device has + // one HOST_ONLY heap; HOST_LOCAL / HOST_COHERENT / DEVICE_LOCAL are + // rejected. Coherence is flush after host writes and invalidate after + // device writes. Persistent map via hrx_buffer_map_with_mode. + hrx_status_t s = hrx_buffer_allocate( + r.stream, sz, + HRX_MEMORY_TYPE_HOST_VISIBLE | + HRX_MEMORY_TYPE_HOST_CACHED | + HRX_MEMORY_TYPE_DEVICE_VISIBLE, + HRX_BUFFER_USAGE_DEFAULT | HRX_BUFFER_USAGE_MAPPING_PERSISTENT, + &hbuf_); + if (!hrx_status_is_ok(s) || !hbuf_) { + hrx_status_ignore(s); + throw std::runtime_error("hrx::ext::bo: hrx_buffer_allocate failed"); + } + void* p = nullptr; + s = hrx_buffer_map_with_mode(hbuf_, HRX_MAPPING_MODE_PERSISTENT, + HRX_MAP_READ | HRX_MAP_WRITE, 0, sz, &p); + if (!hrx_status_is_ok(s) || !p) { + hrx_status_ignore(s); + hrx_buffer_release(hbuf_); + hbuf_ = nullptr; + throw std::runtime_error("hrx::ext::bo: map_persistent failed"); + } + mapped_ = p; + std::memset(p, 0, sz); + // The zeroed contents are host-side only until the first dispatch; mark + // host_dirty so the gated h2d flushes them to the device exactly once. + { + std::lock_guard lk(buf_coh_mu()); + buf_coh()[hbuf_] = BufCoh{/*host_dirty=*/true, /*dev_dirty=*/false}; + } + } +}; +} // namespace ext + +// --------------------------------------------------------------------------- +// run / runlist +// --------------------------------------------------------------------------- +// Gated host->device flush for a dispatch's bindings: flush only the buffers the +// host has dirtied (host_dirty), exactly once, right before the device reads +// them. Device-owned buffers (host_dirty == false) are left untouched so they +// are never clobbered. Mirrors the shim's "h2d only dirty inputs" phase. +inline void hrx_h2d_bindings(const std::vector& binds) { + std::lock_guard lk(buf_coh_mu()); + auto& coh = buf_coh(); + for (const auto& b : binds) { + if (!b.buffer) continue; + BufCoh& st = coh[b.buffer]; + if (st.host_dirty) { + hrx_report(hrx_buffer_flush_range(b.buffer, b.offset, b.length), + "hrx_buffer_flush_range"); + st.host_dirty = false; + } + } +} + +// After a dispatch completes, the device holds the freshest copy of every bound +// buffer: clear host_dirty and mark dev_dirty so the next host read invalidates +// (lazily, via sync_from_device). Mirrors the shim's post-dispatch readback +// bookkeeping. We over-approximate by treating every binding as a potential +// output; that is safe because an extra dev_dirty only triggers a redundant +// invalidate that re-reads identical bytes. +inline void hrx_mark_dispatched(const std::vector& binds) { + std::lock_guard lk(buf_coh_mu()); + auto& coh = buf_coh(); + for (const auto& b : binds) { + if (!b.buffer) continue; + BufCoh& st = coh[b.buffer]; + st.host_dirty = false; + st.dev_dirty = true; + } +} + +class run { +public: + hrx_executable_t exe_ = nullptr; + uint32_t ord_ = 0; + std::vector binds_; + + run() = default; + explicit run(hrx_executable_t exe, uint32_t ord) : exe_(exe), ord_(ord) {} + + void add_binding(hrx_buffer_t b, size_t size) { + binds_.push_back({b, 0, size}); + } + + void record() { + if (!exe_) { + std::fprintf(stderr, "[hrx][ERROR] run::record with null executable\n"); + return; + } + // Zero bindings is valid: RTP/control-only runs such as + // set_layer_rtp.create_run() and gemma4e layer_pre_load.create_run(). + // Opcode scalars live in the TXN stream, not as BO bindings. + hrx_h2d_bindings(binds_); + hrx_dispatch_config_t cfg = {{1, 1, 1}, {1, 1, 1}, 0}; + hrx_status_t s = hrx_stream_dispatch(rt().stream, exe_, ord_, &cfg, + nullptr, 0, binds_.data(), + binds_.size(), HRX_DISPATCH_FLAG_NONE); + hrx_report(s, "run::record hrx_stream_dispatch"); + } + + void start() { + record(); + hrx_status_t s = hrx_stream_flush(rt().stream); + hrx_report(s, "run::start hrx_stream_flush"); + } + + ert_cmd_state wait() { + hrx_status_t s = hrx_stream_wait(rt().stream); + bool err = hrx_report(s, "run::wait hrx_stream_wait"); + hrx_mark_dispatched(binds_); + return err ? ERT_CMD_STATE_ERROR : ERT_CMD_STATE_COMPLETED; + } +}; + +class runlist { +public: + std::vector runs_; + + runlist() = default; + explicit runlist(const hw_context& /*ctx*/) {} + + void add(const run& r) { runs_.push_back(r); } + void add(run&& r) { runs_.push_back(std::move(r)); } + void reset() { runs_.clear(); } + + void execute() { + for (auto& r : runs_) r.record(); + hrx_status_t s = hrx_stream_flush(rt().stream); + hrx_report(s, "runlist::execute hrx_stream_flush"); + } + ert_cmd_state wait() { + hrx_status_t s = hrx_stream_wait(rt().stream); + bool err = hrx_report(s, "runlist::wait hrx_stream_wait"); + for (auto& r : runs_) hrx_mark_dispatched(r.binds_); + return err ? ERT_CMD_STATE_ERROR : ERT_CMD_STATE_COMPLETED; + } +}; + +} // namespace hrx diff --git a/src/detail/include/lm_config.hpp b/src/detail/include/lm_config.hpp new file mode 100644 index 000000000..40b589280 --- /dev/null +++ b/src/detail/include/lm_config.hpp @@ -0,0 +1,89 @@ +/// \file lm_config.hpp +/// \brief lm_config class +/// \author FastFlowLM Team +/// \date 2025-08-05 +/// \version 0.9.10 +/// \note This class is used to store the model configuration. +#pragma once + +#include "typedef.hpp" +#include "utils/utils.hpp" +#include "nlohmann/json.hpp" +#include + +/// \brief read one parameter out of a config json +/// \note Same semantics as JSON_GET (missing OR null falls back to the default), +/// in expression form. Plain nlohmann .value() throws on a null, and real +/// configs do carry nulls (e.g. "sliding_window": null on qwen3). +template +inline T cfg_get(const nlohmann::json& jc, const char* key, T default_value){ + if (jc.contains(key) && !jc[key].is_null()){ + return T(jc[key]); + } + return default_value; +} + +/// \brief read a nested object out of a config json +/// \note Reference-returning counterpart of cfg_get, for walking nested configs +/// (thinker_config.text_config, ...). Returns a shared empty object when +/// the key is missing or null, so the JSON_GET below it still defaults. +inline const nlohmann::json& cfg_sub(const nlohmann::json& jc, const char* key){ + static const nlohmann::json empty = nlohmann::json::object(); + if (jc.contains(key) && !jc[key].is_null()){ + return jc[key]; + } + return empty; +} + +/// \brief LM_Config class +/// \note Model parameters are NOT cached as members. Everything comes from the +/// model's config.json, so every consumer reads what it needs straight out +/// of _json_config with JSON_GET. from_pretrained() only locates the file +/// and normalizes it, so that all readers share one canonical key set. +/// \note This class is passed by value across the FLM_DLL boundary. Its layout +/// must stay identical to FastFlowLM/src/include/lm_config.hpp. +class LM_Config{ + public: + std::string model_path; + std::string model_name; + std::string exec_path; + std::string flm_version; + + nlohmann::json _json_config; + + /// \brief read one model parameter out of config.json + /// \note Defaults to u32 because most parameters are dimensions: + /// config.get("head_dim"), config.get("rms_norm_eps", 0.0f), + /// config.get("vision_model_weight", ""). + template + T get(const char* key, T default_value = T(0)) const { + return cfg_get(this->_json_config, key, default_value); + } + + /// \brief read a nested config object (vision_config, audio_config, ...) + /// \return the sub-object, or an empty object when absent/null + const nlohmann::json& sub(const char* key) const { + return cfg_sub(this->_json_config, key); + } + + /// \brief from pretrained + /// \param model_name the model name + void from_pretrained(std::string model_name); + std::string _str(); + LM_Config(){} + + protected: + void _resolve_paths(const std::string& model_name); + void _load_json(); + void _normalize_multi_modal(); + std::string _str_from(const nlohmann::json& jc); +}; + +class Whisper_Config : public LM_Config{ +public: + /// \brief from pretrained + /// \param model_name the model name + void from_pretrained(std::string model_name); + std::string _str(); + Whisper_Config(){} +}; diff --git a/src/detail/include/metrices.hpp b/src/detail/include/metrices.hpp new file mode 100644 index 000000000..3480c1c17 --- /dev/null +++ b/src/detail/include/metrices.hpp @@ -0,0 +1,149 @@ +/// \file metrices.hpp +/// \brief metrices class +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This class is used to calculate the numerical error metrics. +#pragma once + +#include "typedef.hpp" + +typedef struct { + /// \brief cosine similarity + float CosineSimilarity; + /// \brief relative L2 + float RelativeL2; +} error_metrics; + +/// \brief AVX2 Absolute value +/// \param x the input value, 256-bits, 8xfloat32 +/// \return the absolute value, 256-bits, 8xfloat32 +inline __m256 _mm256_abs_ps(__m256 x){ + return _mm256_and_ps(x, _mm256_set1_ps(0x7fffffff)); +} + +/// \brief get the error metrics +/// \param y the output value, 256-bits, 8xbf16 +/// \param y_ref the reference value, 256-bits, 8xbf16 +/// \return the error metrics +inline error_metrics get_error_metrics(buffer& y, buffer& y_ref){ + assert(y.size() == y_ref.size()); + error_metrics metrics; + const int simd_width = 8; + + __m256 dot_product = _mm256_setzero_ps(); + __m256 y_square_sum = _mm256_setzero_ps(); + __m256 y_ref_square_sum = _mm256_setzero_ps(); + __m256 error_square_sum = _mm256_setzero_ps(); + __m256 abs_error_sum = _mm256_setzero_ps(); + __m256 abs_y_ref_sum = _mm256_setzero_ps(); + for (int i = 0; i < y.size(); i += simd_width){ + __m128i y_bf16 = _mm_loadu_si128((__m128i*)(y.data() + i)); + __m128i y_ref_bf16 = _mm_loadu_si128((__m128i*)(y_ref.data() + i)); + __m256 y_vec = bf16o_fp32(y_bf16); + __m256 y_ref_vec = bf16o_fp32(y_ref_bf16); + __m256 error_vec = _mm256_sub_ps(y_vec, y_ref_vec); + y_square_sum = _mm256_add_ps(y_square_sum, _mm256_mul_ps(y_vec, y_vec)); + y_ref_square_sum = _mm256_add_ps(y_ref_square_sum, _mm256_mul_ps(y_ref_vec, y_ref_vec)); + dot_product = _mm256_add_ps(dot_product, _mm256_mul_ps(y_vec, y_ref_vec)); + error_square_sum = _mm256_add_ps(error_square_sum, _mm256_mul_ps(error_vec, error_vec)); + abs_error_sum = _mm256_add_ps(abs_error_sum, _mm256_abs_ps(error_vec)); + abs_y_ref_sum = _mm256_add_ps(abs_y_ref_sum, _mm256_abs_ps(y_ref_vec)); + } + f32 temp[simd_width]; + _mm256_storeu_ps(temp, dot_product); + f32 dot_product_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + _mm256_storeu_ps(temp, y_square_sum); + f32 y_square_sum_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + _mm256_storeu_ps(temp, y_ref_square_sum); + f32 y_ref_square_sum_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + _mm256_storeu_ps(temp, error_square_sum); + f32 error_square_sum_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + _mm256_storeu_ps(temp, abs_error_sum); + f32 abs_error_sum_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + _mm256_storeu_ps(temp, abs_y_ref_sum); + f32 abs_y_ref_sum_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + + // Cosine Similarity + float cosine_similarity = dot_product_sum / (sqrt(y_square_sum_sum) * sqrt(y_ref_square_sum_sum)); + // Relative L2 + float relative_l2 = sqrt(error_square_sum_sum / y_ref_square_sum_sum); + metrics.CosineSimilarity = cosine_similarity; + metrics.RelativeL2 = relative_l2; + return metrics; +} + +/// \brief get the error metrics +/// \param y the output value, 256-bits, 8xfloat32 +/// \param y_ref the reference value, 256-bits, 8xfloat32 +/// \return the error metrics +inline error_metrics get_error_metrics(buffer& y, buffer& y_ref){ + assert(y.size() == y_ref.size()); + error_metrics metrics; + const int simd_width = 8; + + __m256 dot_product = _mm256_setzero_ps(); + __m256 y_square_sum = _mm256_setzero_ps(); + __m256 y_ref_square_sum = _mm256_setzero_ps(); + __m256 error_square_sum = _mm256_setzero_ps(); + __m256 abs_error_sum = _mm256_setzero_ps(); + __m256 abs_y_ref_sum = _mm256_setzero_ps(); + for (int i = 0; i < y.size(); i += simd_width){ + __m256 y_vec = _mm256_loadu_ps(y.data() + i); + __m256 y_ref_vec = _mm256_loadu_ps(y_ref.data() + i); + __m256 error_vec = _mm256_sub_ps(y_vec, y_ref_vec); + y_square_sum = _mm256_add_ps(y_square_sum, _mm256_mul_ps(y_vec, y_vec)); + y_ref_square_sum = _mm256_add_ps(y_ref_square_sum, _mm256_mul_ps(y_ref_vec, y_ref_vec)); + dot_product = _mm256_add_ps(dot_product, _mm256_mul_ps(y_vec, y_ref_vec)); + error_square_sum = _mm256_add_ps(error_square_sum, _mm256_mul_ps(error_vec, error_vec)); + abs_error_sum = _mm256_add_ps(abs_error_sum, _mm256_abs_ps(error_vec)); + abs_y_ref_sum = _mm256_add_ps(abs_y_ref_sum, _mm256_abs_ps(y_ref_vec)); + } + f32 temp[simd_width]; + _mm256_storeu_ps(temp, dot_product); + f32 dot_product_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + _mm256_storeu_ps(temp, y_square_sum); + f32 y_square_sum_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + _mm256_storeu_ps(temp, y_ref_square_sum); + f32 y_ref_square_sum_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + _mm256_storeu_ps(temp, error_square_sum); + f32 error_square_sum_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + _mm256_storeu_ps(temp, abs_error_sum); + f32 abs_error_sum_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + _mm256_storeu_ps(temp, abs_y_ref_sum); + f32 abs_y_ref_sum_sum = temp[0] + temp[1] + temp[2] + temp[3] + + temp[4] + temp[5] + temp[6] + temp[7]; + + // Cosine Similarity + float cosine_similarity = dot_product_sum / (sqrt(y_square_sum_sum) * sqrt(y_ref_square_sum_sum)); + // Relative L1 + float relative_l1 = abs_error_sum_sum / abs_y_ref_sum_sum; + // RMSE + float rmse = sqrt(error_square_sum_sum / y.size()); + // Relative L2 + float relative_l2 = sqrt(error_square_sum_sum / y_ref_square_sum_sum); + metrics.CosineSimilarity = cosine_similarity; + metrics.RelativeL2 = relative_l2; + return metrics; +} + +/// \brief print the error metrics +/// \param metrics the error metrics +inline void print_error_metrics(error_metrics metrics, std::string name = "Error Metrics"){ + header_print("info", name); + header_print("info", "\tCosine Similarity: " << metrics.CosineSimilarity); + // header_print("info", "\tRelative L1 : " << metrics.RelativeL1); + // header_print("info", "\tRMSE : " << metrics.RMSE); + header_print("info", "\tRelative L2 : " << metrics.RelativeL2); +} diff --git a/src/detail/include/model_list.hpp b/src/detail/include/model_list.hpp new file mode 100644 index 000000000..6773c56ee --- /dev/null +++ b/src/detail/include/model_list.hpp @@ -0,0 +1,191 @@ +/// \file model_list.hpp +/// \brief model_list class +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This class is used to manage the model list. +#pragma once +#include "nlohmann/json.hpp" +#include +#include +#include +#include +#include "utils/utils.hpp" + +#define __FLM_VERSION__ "0.9.6" + +/// \note This class is used to manage the model list. +class model_list { + public: + /// \brief files required for the model + static constexpr const char* model_files[] = { + "config.json", + "tokenizer.json", + "attn.xclbin", + "mm.xclbin", + "dequant.xclbin", + "layer.xclbin", + "lm_head.xclbin", + "model.q4nx", + "tokenizer_config.json" + }; + static constexpr const char* vision_model_files[] = { + "vision_attn.xclbin", + "vision_mm.xclbin", + "vision_weight.q4nx" + }; + /// \brief number of model files + static constexpr int model_files_count = sizeof(model_files) / sizeof(model_files[0]); + + /// \brief number of vision model files + static constexpr int vision_model_files_count = sizeof(vision_model_files) / sizeof(vision_model_files[0]); + + /// \brief constructor + model_list(){} + + /// \brief constructor + /// \param list_path the path to the model list + /// \param exe_dir the executable directory for resolving relative paths + model_list(std::string& list_path, std::string& exe_dir){ + this->list_path = list_path; + std::ifstream config_file(list_path); + if (!config_file.is_open()) { + std::cerr << "Failed to open config file: " << list_path << std::endl; + exit(1); + } + this->config = nlohmann::json::parse(config_file); + // Resolve model_root_path relative to executable directory + std::string relative_model_path = this->config["model_path"]; + #ifdef _WIN32 + this->model_root_path = exe_dir + "\\" + relative_model_path; + #else + this->model_root_path = exe_dir + "/" + relative_model_path; + #endif + config_file.close(); + } + + /// \brief get the model info + /// \param tag the tag of the model + /// \return the model info + nlohmann::json get_model_info(const std::string& tag){ + static std::string last_error_tag = ""; + std::string new_tag = this->cut_tag(tag); + bool model_found = false; + // get model type, the string before ':' in the tag + std::string model_type; + std::string model_size; + + if (new_tag.find(':') != std::string::npos) { + model_type = new_tag.substr(0, new_tag.find(':')); + model_size = new_tag.substr(new_tag.find(':') + 1); + } + else { + model_type = new_tag; + model_size = ""; + } + + // find the model subset first, compare with the key of the model + bool model_subset_found = false; + for (const auto& [key, model] : this->config["models"].items()) { + if (key == model_type) { + model_subset_found = true; + break; + } + } + if (model_subset_found) { + // find the model in the subset + // check if a size is specified in the tag + // if not use the first model in the subset + if (model_size.empty()) { + model_size = this->config["models"][model_type].begin().key(); + return this->config["models"][model_type][model_size]; + } + bool model_found = false; + for (const auto& [key, model] : this->config["models"][model_type].items()) { + if (key == model_size) { // if the size is found, return the model + model_found = true; + return model; + } + } + if (!model_found) { + if (last_error_tag != new_tag) { + last_error_tag = new_tag; + header_print("ERROR", "Model not found: " + model_size + " in subset " + model_type); + header_print("ERROR", "Using default model: llama3.2-1B"); + } + return this->config["models"]["llama3.2"]["1b"]; + } + } + else{ + if (last_error_tag != new_tag) { + last_error_tag = new_tag; + header_print("ERROR", "Model subset not found: " << model_type << "; using default model: llama3.2-1B"); + } + return this->config["models"]["llama3.2"]["1b"]; + } + return this->config["models"]["llama3.2"]["1b"]; + } + + /// \brief get the model root path + /// \return the model root path, string + std::string get_model_root_path(){ + return this->model_root_path; + } + + /// \brief get all the models + /// \return all the models in json + nlohmann::json get_all_models(){ + nlohmann::json response = { + {"models", nlohmann::json::array()} + }; + + for (const auto& [model_type, model_subset] : this->config["models"].items()) { + for (const auto& [size, model_info] : model_subset.items()) { + nlohmann::json model_entry = { + {"name", model_type + ":" + size}, + {"model", model_type + ":" + size}, + {"modified_at", "2024-03-28T00:00:00Z"}, + {"details", { + {"format", "gguf"}, + {"family", model_type}, + {"parameter_size", size}, + {"quantization_level", "Q4_0"} + }} + }; + response["models"].push_back(model_entry); + } + } + return response; + } + + /// \brief get the model path + /// \param tag the tag of the model + /// \return the model path, string + std::string get_model_path(const std::string& tag){ + std::string new_tag = this->cut_tag(tag); + std::string model_name = this->get_model_info(new_tag)["name"]; + #ifdef _WIN32 + std::string model_path = this->model_root_path + "\\" + model_name; + #else + std::string model_path = this->model_root_path + "/" + model_name; + #endif + return model_path; + } + + /// \brief cut the tag, some program adds a prefix to the tag, like "Ollama/llama3.2-1B", we need to cut the prefix + /// \param tag the tag of the model + /// \return the model type, string + std::string cut_tag(const std::string& tag){ + std::string new_tag = tag; + if (tag.find('/') != std::string::npos) { + new_tag = tag.substr(tag.find('/') + 1); + } + return new_tag; + } + + private: + std::string list_path; + nlohmann::json config; + std::string model_root_path; + +}; \ No newline at end of file diff --git a/src/detail/include/models/gemma/gemma_npu.hpp b/src/detail/include/models/gemma/gemma_npu.hpp new file mode 100644 index 000000000..8af3c64cc --- /dev/null +++ b/src/detail/include/models/gemma/gemma_npu.hpp @@ -0,0 +1,74 @@ +/// \file gemma_npu.hpp +/// \brief gemma_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the gemma_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/gemma/gemma_npu_sequence.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class gemma_npu : public causal_lm{ +public: + /// \brief initialize the gemma_npu + /// \param config the configuration + /// \param npu_instance the npu instance + gemma_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~gemma_npu(); + + /// \brief forward the gemma_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/gemma/gemma_npu_sequence.hpp b/src/detail/include/models/gemma/gemma_npu_sequence.hpp new file mode 100644 index 000000000..57ef32f93 --- /dev/null +++ b/src/detail/include/models/gemma/gemma_npu_sequence.hpp @@ -0,0 +1,56 @@ +/// \file gemma_npu_sequence.hpp +/// \brief gemma_npu_sequence class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the gemma_npu_sequence class +#pragma once +#include "npu_utils/npu_instr_utils.hpp" +#include "lm_config.hpp" + +/// \brief gemma_npu_sequence class +/// \note This is a class for the gemma_npu_sequence +class gemma_npu_sequence{ +public: + gemma_npu_sequence(){} + + /// \brief Constructor + /// \param config the configuration + /// \param MAX_L the max length + gemma_npu_sequence(LM_Config config, uint32_t MAX_L); + ~gemma_npu_sequence(); + + /// \brief Generate the rtp sequence + /// \param seq the sequence + /// \param L the length + void gen_rtp_seq(npu_sequence* seq, const uint32_t L, bool is_sliding_window); + + /// \brief Generate the layer sequence + /// \param seq the sequence + /// \param L the length + void gen_layer_seq(npu_sequence* seq, const uint32_t L, bool is_sliding_window); + + /// \brief Set the max length + /// \param MAX_L the max length + void set_max_length(const uint32_t MAX_L); + + /// \brief Generate the mha engine sequence + /// \param seq the sequence + /// \param L_begin the begin length + /// \param L_end the end length + void gen_mha_engine_seq(npu_sequence* seq, const uint32_t L_begin, const uint32_t L_end, bool is_sliding_window, int buffer_length); + + /// \brief Generate the lm head sequence + /// \param seq the sequence + void gen_lm_head_seq(npu_sequence* seq); + + /// \brief Get the k03 offset + size_t get_k01_offset() const; + size_t get_k23_offset() const; + size_t get_v01_offset() const; + size_t get_v23_offset() const; + +private: + struct Impl; + Impl* _impl; +}; diff --git a/src/detail/include/models/gemma4_12b/gemma4_12b_npu.hpp b/src/detail/include/models/gemma4_12b/gemma4_12b_npu.hpp new file mode 100644 index 000000000..1aee2fa55 --- /dev/null +++ b/src/detail/include/models/gemma4_12b/gemma4_12b_npu.hpp @@ -0,0 +1,216 @@ +/// \file qwen3vl_npu.hpp +/// \brief qwen3vl_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the qwen3vl_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + +// some helper functions for convenience +constexpr int GEMMA4_12B_IS_GLOBAL_MASK = 0x00000001; + +typedef enum :int { + gemma4_12b_swa_layer = 0, + gemma4_12b_global_layer = GEMMA4_12B_IS_GLOBAL_MASK, + gemma4_12b_total_layer_types = 2 +} gemma4_12b_layer_type_t; + +inline bool is_swa_layer(gemma4_12b_layer_type_t layer) { + return (layer & GEMMA4_12B_IS_GLOBAL_MASK) == 0; +} + +inline bool is_global_layer(gemma4_12b_layer_type_t layer) { + return (layer & GEMMA4_12B_IS_GLOBAL_MASK) != 0; +} + +typedef struct { + int height; + int width; + int height_resized; // assigned by image preprocessing + int width_resized; + bytes _data; +} gemma4_12b_image_t; + +typedef struct { + std::vector> image_patch__element_per_patch; // [num_of_image][width, height] + std::vector valid_patch_size_per_image; // [num_of_image], the unpadded size per image + // gemma4-12b is encoder free: the preprocessor already pools the image into model + // patches, so this is [num_of_image][max_soft_tokens * patch_dim] (280 x 6912 on the + // released checkpoint), zero padded past valid_patch_size_per_image, NOT resized pixels. + std::vector> pixel_values; + std::vector< std::vector> image_grid_pairs_per_image; // [num_of_image][num_of_position_id][x, y] + std::vector num_soft_tokens_per_image; // [num_of_image] + unsigned int num_images; +}gemma4_12b_image_payload_t; + +struct gemma4_12b_audio_payload_t { + // Historical names: this model has no audio encoder and takes no mel spectrogram. + // Each "frame" is 640 raw PCM samples (40 ms at 16 kHz) and each "bin" is one sample, + // so this is [num_audios][frames * 640] of waveform, row-major, all rows valid. + std::vector> mel_spectrograms; + std::vector mel_spectrogram_frames_per_audio; // [num_audios] + std::vector mel_spectrogram_bins_per_audio; // [num_audios], == 640 + unsigned int num_audios = 0; + std::vector num_soft_tokens_per_audio; // [num_audios] +}; + +typedef struct { + gemma4_12b_image_payload_t image_payload; + gemma4_12b_audio_payload_t audio_payload; +} gemma4_12b_multi_modal_payload_t; + +class gemma4_12b_npu : public causal_lm{ +public: + /// \brief initialize the qwen3vl_npu + /// \param config the configuration + /// \param npu_instance the npu instance + gemma4_12b_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~gemma4_12b_npu(); + + /// \brief forward the qwen3vl_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; + // ---- preprocessing parameters ------------------------------------------- + // These describe the image/audio front end, not the network, so they come from + // the checkpoint's processor_config.json (image_processor / feature_extractor) + // rather than config.json. config.json is still honoured as a fallback for + // checkpoints packaged before processor_config.json was shipped, and the + // literals below are the released Gemma4-12B values so a checkpoint carrying + // neither still preprocesses correctly. + unsigned int GEMMA4_12B_vision_pooling_kernel_size; + unsigned int GEMMA4_12B_vision_patch_size; + unsigned int GEMMA4_12B_vision_max_soft_tokens; + float GEMMA4_12B_vision_rescale_factor; + float GEMMA4_12B_vision_image_mean; + float GEMMA4_12B_vision_image_std; + // parameters for audio preprocessing + unsigned int GEMMA4_12B_audio_embed_dim; + unsigned int GEMMA4_12B_audio_samples_per_token; + unsigned int GEMMA4_12B_audio_max_soft_tokens; + unsigned int GEMMA4_12B_audio_sampling_rate; + + /// \brief the "image_processor" / "video_processor" / "feature_extractor" block + /// of processor_config.json, or an empty object when absent + inline const nlohmann::json& _processor_sub(LM_Config& config, const char* key){ + return cfg_sub(cfg_sub(config._json_config, "processor_config"), key); + } + + /// \brief first of `processor[key]`, `fallback[key]`, `default_value` that exists + template + inline T _preprocess_value(const nlohmann::json& processor, const nlohmann::json& fallback, + const char* key, T default_value){ + return cfg_get(processor, key, cfg_get(fallback, key, default_value)); + } + + /// \brief same, for the per-channel means/stds, which the processor config + /// writes as a 3-element array and config.json as a scalar + /// \note Gemma4 leaves normalization off (mean 0, std 1) and the host kernel + /// takes one scalar per statistic, so the channels are required to agree. + inline float _preprocess_channel_value(const nlohmann::json& processor, const nlohmann::json& fallback, + const char* key, float default_value){ + for (const nlohmann::json* jc : {&processor, &fallback}){ + if (!jc->contains(key) || (*jc)[key].is_null()){ + continue; + } + const nlohmann::json& v = (*jc)[key]; + if (!v.is_array()){ + return float(v); + } + if (v.empty()){ + continue; + } + const float first = float(v[0]); + for (const auto& c : v){ + if (float(c) != first){ + header_print("warning", std::string("processor config ") + key + + " is not uniform across channels, using " + std::to_string(first)); + break; + } + } + return first; + } + return default_value; + } + + inline void load_vision_preprocess_parameters(LM_Config& config){ + // Note: this should be called by Impl:: constructor + const nlohmann::json& pc = this->_processor_sub(config, "image_processor"); + const nlohmann::json& vc = config.sub("vision_config"); + GEMMA4_12B_vision_pooling_kernel_size = _preprocess_value(pc, vc, "pooling_kernel_size", 3); + GEMMA4_12B_vision_patch_size = _preprocess_value(pc, vc, "patch_size", 16); + GEMMA4_12B_vision_max_soft_tokens = _preprocess_value(pc, vc, "max_soft_tokens", 280); + // mean/std make the normalize step the identity, which is why do_normalize is off. + GEMMA4_12B_vision_rescale_factor = _preprocess_value(pc, vc, "rescale_factor", 1.0f / 255.0f); + GEMMA4_12B_vision_image_mean = _preprocess_channel_value(pc, vc, "image_mean", 0.0f); + GEMMA4_12B_vision_image_std = _preprocess_channel_value(pc, vc, "image_std", 1.0f); + } + + inline void load_audio_preprocess_parameters(LM_Config& config){ + const nlohmann::json& pc = this->_processor_sub(config, "feature_extractor"); + const nlohmann::json& ac = config.sub("audio_config"); + // One soft token is 640 samples at 16 kHz = 40 ms, and the model takes 750 + // of them (30 s). feature_size is the width of one row handed to the audio + // embedder, which for this encoder free model is the raw sample count. + GEMMA4_12B_audio_embed_dim = _preprocess_value(pc, ac, "feature_size", 640); + GEMMA4_12B_audio_samples_per_token = _preprocess_value(pc, ac, "audio_samples_per_token", 640); + GEMMA4_12B_audio_sampling_rate = _preprocess_value(pc, ac, "sampling_rate", 16000); + // audio_seq_length sits at the top level of processor_config.json, next to + // image_seq_length, rather than inside the feature extractor block. + GEMMA4_12B_audio_max_soft_tokens = cfg_get( + cfg_sub(config._json_config, "processor_config"), "audio_seq_length", + cfg_get(ac, "audio_seq_length", 750)); + } + +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/gemma4e/gemma4e_npu.hpp b/src/detail/include/models/gemma4e/gemma4e_npu.hpp new file mode 100644 index 000000000..fe8620c73 --- /dev/null +++ b/src/detail/include/models/gemma4e/gemma4e_npu.hpp @@ -0,0 +1,228 @@ +/// \file qwen3vl_npu.hpp +/// \brief qwen3vl_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the qwen3vl_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + +// some helper functions for convenience +constexpr int GEMMA4E_IS_GLOBAL_MASK = 0x00000001; +constexpr int GEMMA4E_IS_SKIP_MASK = 0x00000002; + +typedef enum :int { + e_gemma4e_swa_layer = 0, + e_gemma4e_global_layer = GEMMA4E_IS_GLOBAL_MASK, + e_gemma4e_swa_layer_skip = GEMMA4E_IS_SKIP_MASK, + e_gemma4e_global_layer_skip = GEMMA4E_IS_GLOBAL_MASK | GEMMA4E_IS_SKIP_MASK, + e_gemma4e_total_layer_types = 4 +} gemma4e_layer_type_t; + +inline bool is_swa_layer(gemma4e_layer_type_t layer) { + return (layer & GEMMA4E_IS_GLOBAL_MASK) == 0; +} + +inline bool is_global_layer(gemma4e_layer_type_t layer) { + return (layer & GEMMA4E_IS_GLOBAL_MASK) != 0; +} + +inline bool is_skip_layer(gemma4e_layer_type_t layer) { + return (layer & GEMMA4E_IS_SKIP_MASK) != 0; +} + +typedef struct { + int height; + int width; + int height_resized; // assigned by image preprocessing + int width_resized; + // int grid_h; + // int grid_w; + + bytes _data; + +} gemma4e_image_t; + + + + + +typedef struct { + // original raw data + std::vector> image_patch__element_per_patch; // [num_of_image][width, height] + std::vector valid_patch_size_per_image; // [num_of_image], the unpadded size per image + std::vector> pixel_values; // [num_of_image][image_size], where image_size = height_resized * width_resized * 3 + std::vector< std::vector> image_grid_pairs_per_image; // [num_of_image][num_of_position_id][x, y] + std::vector num_soft_tokens_per_image; // [num_of_image] + unsigned int num_images; + + + +}gemma4e_image_payload_t; + + +struct gemma4e_audio_payload_t { + // per-audio mel spectrogram data + std::vector> mel_spectrograms; // [num_audios][frames * bins], row-major + std::vector mel_spectrogram_frames_per_audio; // [num_audios] + std::vector mel_spectrogram_bins_per_audio; // [num_audios] + unsigned int num_audios = 0; + std::vector num_soft_tokens_per_audio; // [num_audios] + +}; + + +typedef struct { + gemma4e_image_payload_t image_payload; + gemma4e_audio_payload_t audio_payload; +} gemma4e_multi_modal_payload_t; + +class gemma4e_npu : public causal_lm{ +public: + /// \brief initialize the qwen3vl_npu + /// \param config the configuration + /// \param npu_instance the npu instance + gemma4e_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~gemma4e_npu(); + + /// \brief forward the qwen3vl_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; + // parameters for vision preprocessing in Gemma4e + unsigned int GEMMA4E_VISION_MAX_POSITION_EMBEDDINGS; + unsigned int GEMMA4E_VISION_NUM_HIDDEN_LAYERS; + unsigned int GEMMA4E_VISION_NUM_ATTENTION_HEADS; + unsigned int GEMMA4E_VISION_HIDDEN_SIZE; + unsigned int GEMMA4E_VISION_INTERMEDIATE_SIZE; + unsigned int GEMMA4E_VISION_HEAD_DIM; + unsigned int GEMMA4E_VISION_PATCH_SIZE; + float GEMMA4E_ROPE_THETA; + unsigned int GEMMA4E_POOLING_KERNEL_SIZE; + unsigned int GEMMA4E_POSITION_EMBEDDING_SIZE; + unsigned int GEMMA4E_VISION_IMAGE_OUTPUT_SIZE; + float GEMMA4E_VISION_RESCALE_FACTOR; + float GEMMA4E_VISION_IMAGE_MEAN; + float GEMMA4E_VISION_IMAGE_STD; + + // parameters for audio preprocessing in Gemma4e + unsigned int Audio_MM_TILE_M; + unsigned int Audio_MM_TILE_K; + unsigned int Audio_MM_TILE_N; + int Gemma4E_Audio_resample_rate; + float Gemma4E_Audio_gradient_clipping; + unsigned int Gemma4E_Audio_Multimodal_Output_SIZE; + unsigned int Gemma4E_Audio_language_projection_output_size; + unsigned int Gemma4E_Audio_HIDDEN_SIZE; + unsigned int Gemma4E_Audio_INTERMEDIATE_SIZE; + unsigned int Gemma4E_Audio_attention_chunk_size; + unsigned int Gemma4E_Audio_attention_context_left; + unsigned int Gemma4E_Audio_attention_context_right; + unsigned int Gemma4E_Audio_num_attention_heads; + unsigned int Gemma4E_Audio_num_attention_layers; + unsigned int Gemma4E_Audio_conv1d_kernel_size; + unsigned int Gemma4E_Audio_conv1d_stride; + unsigned int Gemma4E_Audio_conv2d_kernel_size; + unsigned int Gemma4E_Audio_conv2d_Stride; + unsigned int Gemma4e_Audio_conv2d_Padding; + unsigned int Gemma4E_Audio_subsampling_conv_channels_0; + unsigned int Gemma4E_Audio_subsampling_conv_channels_1; + float Gemma4E_Audio_attention_softcap; + + + inline void load_vision_preprocess_parameters(LM_Config& config){ + // Note: this should be called by Impl:: constructor + const nlohmann::json& vc = config.sub("vision_config"); + GEMMA4E_VISION_MAX_POSITION_EMBEDDINGS = vc.value("GEMMA4E_VISION_MAX_POSITION_EMBEDDINGS", -1); + GEMMA4E_VISION_NUM_HIDDEN_LAYERS = vc.value("GEMMA4E_VISION_NUM_HIDDEN_LAYERS", -1); + GEMMA4E_VISION_NUM_ATTENTION_HEADS = vc.value("GEMMA4E_VISION_NUM_ATTENTION_HEADS", -1); + GEMMA4E_VISION_HIDDEN_SIZE = vc.value("GEMMA4E_VISION_HIDDEN_SIZE", -1); + GEMMA4E_VISION_INTERMEDIATE_SIZE = vc.value("GEMMA4E_VISION_INTERMEDIATE_SIZE", -1); + GEMMA4E_VISION_HEAD_DIM = vc.value("GEMMA4E_VISION_HEAD_DIM", -1); + GEMMA4E_VISION_PATCH_SIZE = vc.value("GEMMA4E_VISION_PATCH_SIZE", -1); + GEMMA4E_ROPE_THETA = vc.value("GEMMA4E_ROPE_THETA", -1.0f); + GEMMA4E_POOLING_KERNEL_SIZE = vc.value("GEMMA4E_POOLING_KERNEL_SIZE", -1); + GEMMA4E_POSITION_EMBEDDING_SIZE = vc.value("GEMMA4E_POSITION_EMBEDDING_SIZE", -1); + GEMMA4E_VISION_IMAGE_OUTPUT_SIZE = vc.value("GEMMA4E_VISION_IMAGE_OUTPUT_SIZE", -1); + GEMMA4E_VISION_RESCALE_FACTOR = vc.value("GEMMA4E_VISION_RESCALE_FACTOR", -1.0f); + GEMMA4E_VISION_IMAGE_MEAN = vc.value("GEMMA4E_VISION_IMAGE_MEAN", -1.0f); + GEMMA4E_VISION_IMAGE_STD = vc.value("GEMMA4E_VISION_IMAGE_STD", -1.0f); + } + + inline void load_audio_preprocess_parameters(LM_Config& config){ + const nlohmann::json& ac = config.sub("audio_config"); + Audio_MM_TILE_M = ac.value("Audio_MM_TILE_M", 128); + Audio_MM_TILE_K = ac.value("Audio_MM_TILE_K", 512); + Audio_MM_TILE_N = ac.value("Audio_MM_TILE_N", 64); + Gemma4E_Audio_resample_rate = ac.value("Gemma4E_Audio_audio_resample_rate", -1); + Gemma4E_Audio_gradient_clipping = ac.value("Gemma4E_Audio_gradient_clipping", -1.0f); + Gemma4E_Audio_Multimodal_Output_SIZE = ac.value("Gemma4E_Audio_Multimodal_Output_SIZE", -1); + Gemma4E_Audio_language_projection_output_size = ac.value("Gemma4E_Audio_language_projection_output_size", -1); + Gemma4E_Audio_HIDDEN_SIZE = ac.value("Gemma4E_Audio_HIDDEN_SIZE", -1); + Gemma4E_Audio_INTERMEDIATE_SIZE = ac.value("Gemma4E_Audio_INTERMEDIATE_SIZE", -1); + Gemma4E_Audio_attention_chunk_size = ac.value("Gemma4E_Audio_attention_chunk_size", -1); + Gemma4E_Audio_attention_context_left = ac.value("Gemma4E_Audio_attention_context_left", -1); + Gemma4E_Audio_attention_context_right = ac.value("Gemma4E_Audio_attention_context_right", -1); + Gemma4E_Audio_num_attention_heads = ac.value("Gemma4E_Audio_num_attention_heads", -1); + Gemma4E_Audio_num_attention_layers = ac.value("Gemma4E_Audio_num_attention_layers", -1); + Gemma4E_Audio_conv1d_kernel_size = ac.value("Gemma4E_Audio_conv1d_kernel_size", -1); + Gemma4E_Audio_conv1d_stride = ac.value("Gemma4E_Audio_conv1d_stride", -1); + Gemma4E_Audio_conv2d_kernel_size = ac.value("Gemma4E_conv2d_kernel_size", -1); + Gemma4E_Audio_conv2d_Stride = ac.value("Gemma4E_conv2d_Stride", -1); + Gemma4e_Audio_conv2d_Padding = ac.value("Gemma4e_conv2d_Padding", -1); + Gemma4E_Audio_subsampling_conv_channels_0 = ac.value("Gemma4E_Audio_subsampling_conv_channels_0", -1); + Gemma4E_Audio_subsampling_conv_channels_1 = ac.value("Gemma4E_Audio_subsampling_conv_channels_1", -1); + Gemma4E_Audio_attention_softcap = ac.value("Gemma4E_Audio_attention_softcap", -1.0f); + } + +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/gemma4e_flash/gemma4e_flash.hpp b/src/detail/include/models/gemma4e_flash/gemma4e_flash.hpp new file mode 100644 index 000000000..fe4bd47d0 --- /dev/null +++ b/src/detail/include/models/gemma4e_flash/gemma4e_flash.hpp @@ -0,0 +1,156 @@ +/// \file gemma4e_flash.hpp +/// \brief gemma4e_flash class +/// \author FastFlowLM Team +/// \date 2026-09-11 +/// \version 0.9.28 +/// \note This is a header file for the gemma4e_flash class +/// +/// gemma4e_flash is a second engine for the same Gemma4-E checkpoint as +/// gemma4e_npu. It starts life as a byte-for-byte copy of that engine and is +/// the place to tune prefill for short input prompts; gemma4e_npu remains the +/// general-purpose engine. +/// +/// The layer-type enum, the image/audio payload types and the GEMMA4E_* +/// preprocessing constants are shared with gemma4e_npu, so this header pulls +/// them in rather than redefining them: the application builds one +/// gemma4e_image_payload_t and hands it to either engine. +#pragma once +#include "models/gemma4e/gemma4e_npu.hpp" + +class gemma4e_flash : public causal_lm{ +public: + /// \brief initialize the gemma4e_flash + /// \param config the configuration + /// \param npu_instance the npu instance + gemma4e_flash(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~gemma4e_flash(); + + /// \brief forward the gemma4e_flash + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; + // parameters for vision preprocessing in Gemma4e + unsigned int GEMMA4E_VISION_MAX_POSITION_EMBEDDINGS; + unsigned int GEMMA4E_VISION_NUM_HIDDEN_LAYERS; + unsigned int GEMMA4E_VISION_NUM_ATTENTION_HEADS; + unsigned int GEMMA4E_VISION_HIDDEN_SIZE; + unsigned int GEMMA4E_VISION_INTERMEDIATE_SIZE; + unsigned int GEMMA4E_VISION_HEAD_DIM; + unsigned int GEMMA4E_VISION_PATCH_SIZE; + float GEMMA4E_ROPE_THETA; + unsigned int GEMMA4E_POOLING_KERNEL_SIZE; + unsigned int GEMMA4E_POSITION_EMBEDDING_SIZE; + unsigned int GEMMA4E_VISION_IMAGE_OUTPUT_SIZE; + float GEMMA4E_VISION_RESCALE_FACTOR; + float GEMMA4E_VISION_IMAGE_MEAN; + float GEMMA4E_VISION_IMAGE_STD; + + // parameters for audio preprocessing in Gemma4e + unsigned int Audio_MM_TILE_M; + unsigned int Audio_MM_TILE_K; + unsigned int Audio_MM_TILE_N; + int Gemma4E_Audio_resample_rate; + float Gemma4E_Audio_gradient_clipping; + unsigned int Gemma4E_Audio_Multimodal_Output_SIZE; + unsigned int Gemma4E_Audio_language_projection_output_size; + unsigned int Gemma4E_Audio_HIDDEN_SIZE; + unsigned int Gemma4E_Audio_INTERMEDIATE_SIZE; + unsigned int Gemma4E_Audio_attention_chunk_size; + unsigned int Gemma4E_Audio_attention_context_left; + unsigned int Gemma4E_Audio_attention_context_right; + unsigned int Gemma4E_Audio_num_attention_heads; + unsigned int Gemma4E_Audio_num_attention_layers; + unsigned int Gemma4E_Audio_conv1d_kernel_size; + unsigned int Gemma4E_Audio_conv1d_stride; + unsigned int Gemma4E_Audio_conv2d_kernel_size; + unsigned int Gemma4E_Audio_conv2d_Stride; + unsigned int Gemma4e_Audio_conv2d_Padding; + unsigned int Gemma4E_Audio_subsampling_conv_channels_0; + unsigned int Gemma4E_Audio_subsampling_conv_channels_1; + float Gemma4E_Audio_attention_softcap; + + + inline void load_vision_preprocess_parameters(LM_Config& config){ + // Note: this should be called by Impl:: constructor + const nlohmann::json& vc = config.sub("vision_config"); + GEMMA4E_VISION_MAX_POSITION_EMBEDDINGS = vc.value("GEMMA4E_VISION_MAX_POSITION_EMBEDDINGS", -1); + GEMMA4E_VISION_NUM_HIDDEN_LAYERS = vc.value("GEMMA4E_VISION_NUM_HIDDEN_LAYERS", -1); + GEMMA4E_VISION_NUM_ATTENTION_HEADS = vc.value("GEMMA4E_VISION_NUM_ATTENTION_HEADS", -1); + GEMMA4E_VISION_HIDDEN_SIZE = vc.value("GEMMA4E_VISION_HIDDEN_SIZE", -1); + GEMMA4E_VISION_INTERMEDIATE_SIZE = vc.value("GEMMA4E_VISION_INTERMEDIATE_SIZE", -1); + GEMMA4E_VISION_HEAD_DIM = vc.value("GEMMA4E_VISION_HEAD_DIM", -1); + GEMMA4E_VISION_PATCH_SIZE = vc.value("GEMMA4E_VISION_PATCH_SIZE", -1); + GEMMA4E_ROPE_THETA = vc.value("GEMMA4E_ROPE_THETA", -1.0f); + GEMMA4E_POOLING_KERNEL_SIZE = vc.value("GEMMA4E_POOLING_KERNEL_SIZE", -1); + GEMMA4E_POSITION_EMBEDDING_SIZE = vc.value("GEMMA4E_POSITION_EMBEDDING_SIZE", -1); + GEMMA4E_VISION_IMAGE_OUTPUT_SIZE = vc.value("GEMMA4E_VISION_IMAGE_OUTPUT_SIZE", -1); + GEMMA4E_VISION_RESCALE_FACTOR = vc.value("GEMMA4E_VISION_RESCALE_FACTOR", -1.0f); + GEMMA4E_VISION_IMAGE_MEAN = vc.value("GEMMA4E_VISION_IMAGE_MEAN", -1.0f); + GEMMA4E_VISION_IMAGE_STD = vc.value("GEMMA4E_VISION_IMAGE_STD", -1.0f); + } + + inline void load_audio_preprocess_parameters(LM_Config& config){ + const nlohmann::json& ac = config.sub("audio_config"); + Audio_MM_TILE_M = ac.value("Audio_MM_TILE_M", 128); + Audio_MM_TILE_K = ac.value("Audio_MM_TILE_K", 512); + Audio_MM_TILE_N = ac.value("Audio_MM_TILE_N", 64); + Gemma4E_Audio_resample_rate = ac.value("Gemma4E_Audio_audio_resample_rate", -1); + Gemma4E_Audio_gradient_clipping = ac.value("Gemma4E_Audio_gradient_clipping", -1.0f); + Gemma4E_Audio_Multimodal_Output_SIZE = ac.value("Gemma4E_Audio_Multimodal_Output_SIZE", -1); + Gemma4E_Audio_language_projection_output_size = ac.value("Gemma4E_Audio_language_projection_output_size", -1); + Gemma4E_Audio_HIDDEN_SIZE = ac.value("Gemma4E_Audio_HIDDEN_SIZE", -1); + Gemma4E_Audio_INTERMEDIATE_SIZE = ac.value("Gemma4E_Audio_INTERMEDIATE_SIZE", -1); + Gemma4E_Audio_attention_chunk_size = ac.value("Gemma4E_Audio_attention_chunk_size", -1); + Gemma4E_Audio_attention_context_left = ac.value("Gemma4E_Audio_attention_context_left", -1); + Gemma4E_Audio_attention_context_right = ac.value("Gemma4E_Audio_attention_context_right", -1); + Gemma4E_Audio_num_attention_heads = ac.value("Gemma4E_Audio_num_attention_heads", -1); + Gemma4E_Audio_num_attention_layers = ac.value("Gemma4E_Audio_num_attention_layers", -1); + Gemma4E_Audio_conv1d_kernel_size = ac.value("Gemma4E_Audio_conv1d_kernel_size", -1); + Gemma4E_Audio_conv1d_stride = ac.value("Gemma4E_Audio_conv1d_stride", -1); + Gemma4E_Audio_conv2d_kernel_size = ac.value("Gemma4E_conv2d_kernel_size", -1); + Gemma4E_Audio_conv2d_Stride = ac.value("Gemma4E_conv2d_Stride", -1); + Gemma4e_Audio_conv2d_Padding = ac.value("Gemma4e_conv2d_Padding", -1); + Gemma4E_Audio_subsampling_conv_channels_0 = ac.value("Gemma4E_Audio_subsampling_conv_channels_0", -1); + Gemma4E_Audio_subsampling_conv_channels_1 = ac.value("Gemma4E_Audio_subsampling_conv_channels_1", -1); + Gemma4E_Audio_attention_softcap = ac.value("Gemma4E_Audio_attention_softcap", -1.0f); + } + +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/gemma_embedding/gemma_embedding.hpp b/src/detail/include/models/gemma_embedding/gemma_embedding.hpp new file mode 100644 index 000000000..68b8fe7e7 --- /dev/null +++ b/src/detail/include/models/gemma_embedding/gemma_embedding.hpp @@ -0,0 +1,44 @@ +/// \file gemma_embedding.hpp +/// \brief gemma_embedding class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the gemma_embedding class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "embedding_model.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class gemma_embedding : public embedding_model{ +public: + /// \brief initialize the gemma_embedding + /// \param config the configuration + /// \param npu_instance the npu instance + gemma_embedding(LM_Config config, npu_xclbin_manager *npu_instance); + ~gemma_embedding(); + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief embed the gemma_embedding + /// \param tokens the tokens + /// \return the output tensor + buffer embed(std::vector& tokens) override; + +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/gemma_text/gemma_text_dequant.hpp b/src/detail/include/models/gemma_text/gemma_text_dequant.hpp new file mode 100644 index 000000000..559c37c8c --- /dev/null +++ b/src/detail/include/models/gemma_text/gemma_text_dequant.hpp @@ -0,0 +1,43 @@ +/// \file dequant.hpp +/// \brief dequant class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the dequant class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" + +/// \brief dequant class +/// \note This is a class for the dequant layer +class GemmaTextDequant{ +public: + GemmaTextDequant(){} + + /// \brief Constructor + /// \param config the configuration + /// \param xclbin_name the xclbin name + /// \param npu the npu manager + GemmaTextDequant(LM_Config& config); + ~GemmaTextDequant(); + + /// \brief Generate the sequence + /// \param seq the sequence + /// \param D_in the input dimension + /// \param D_out the output dimension + /// \param weight_offset the weight offset + void generate_seq(npu_sequence* seq, const uint32_t D_in, const uint32_t D_out, const uint32_t weight_offset); + + /// \brief get q_group_id + /// \return the q group id + int q_id(); + /// \brief get qw_group_id + /// \return the qw group id + int qw_id(); + +private: + struct Impl; + Impl* _impl; + +}; + diff --git a/src/detail/include/models/gemma_text/gemma_text_gemm.hpp b/src/detail/include/models/gemma_text/gemma_text_gemm.hpp new file mode 100644 index 000000000..fbb27c7e9 --- /dev/null +++ b/src/detail/include/models/gemma_text/gemma_text_gemm.hpp @@ -0,0 +1,42 @@ +/// \file gemm.hpp +/// \brief gemm class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the gemm class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" + +/// \brief gemm class +/// \note This is a class for the gemm layer +class GemmaTextGemm{ +public: + GemmaTextGemm(){} + + /// \brief Constructor + /// \param config the configuration + /// \param xclbin_name the xclbin name + /// \param npu the npu manager + GemmaTextGemm(LM_Config& config); + ~GemmaTextGemm(); + + /// \brief Generate the sequence + void generate_seq(npu_sequence* seq, const uint32_t M, const uint32_t K, const uint32_t N, const uint32_t weight_offset); + + /// \brief get y_group_id + /// \return the y group id + int y_id(); + /// \brief get x_group_id + /// \return the x group id + int x_id(); + /// \brief get w_group_id + /// \return the w group id + int w_id(); + +private: + struct Impl; + Impl* _impl; + +}; + diff --git a/src/detail/include/models/gemma_text/gemma_text_lm_head.hpp b/src/detail/include/models/gemma_text/gemma_text_lm_head.hpp new file mode 100644 index 000000000..bf776561a --- /dev/null +++ b/src/detail/include/models/gemma_text/gemma_text_lm_head.hpp @@ -0,0 +1,44 @@ +/// \file lm_head.hpp +/// \brief lm_head class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the lm_head class +#pragma once +#include "lm_config.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "npu_utils/npu_utils.hpp" + +/// \brief lm_head class +/// \note This is a class for the lm_head layer +class GemmaTextLMHead{ +public: + GemmaTextLMHead(){} + + /// \brief Constructor + /// \param config the configuration + /// \param xclbin_name the xclbin name + /// \param npu the npu manager + GemmaTextLMHead(LM_Config config, npu_xclbin_manager *npu); + ~GemmaTextLMHead(); + + /// \brief Load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx); + + /// \brief Execute the lm_head + void execute(); + + /// \brief Wait for the lm_head + /// \return the output tensor + buffer wait(); + + /// \brief Get the output tensor + /// \return the output tensor + buffer x_exposed(); + +private: + struct Impl; + Impl* _impl; + +}; diff --git a/src/detail/include/models/gemma_text/gemma_text_npu.hpp b/src/detail/include/models/gemma_text/gemma_text_npu.hpp new file mode 100644 index 000000000..8be19f791 --- /dev/null +++ b/src/detail/include/models/gemma_text/gemma_text_npu.hpp @@ -0,0 +1,74 @@ +/// \file gemma_npu.hpp +/// \brief gemma_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the gemma_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/gemma_text/gemma_text_npu_sequence.hpp" +#include "models/gemma_text/gemma_text_lm_head.hpp" +#include "modules/embedding.hpp" +#include "models/gemma_text/gemma_text_gemm.hpp" +#include "models/gemma_text/gemma_text_dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class gemma_text_npu : public causal_lm{ +public: + /// \brief initialize the gemma_text_npu + /// \param config the configuration + /// \param npu_instance the npu instance + gemma_text_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~gemma_text_npu(); + + /// \brief forward the gemma_text_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/gemma_text/gemma_text_npu_sequence.hpp b/src/detail/include/models/gemma_text/gemma_text_npu_sequence.hpp new file mode 100644 index 000000000..c46d03c12 --- /dev/null +++ b/src/detail/include/models/gemma_text/gemma_text_npu_sequence.hpp @@ -0,0 +1,50 @@ +/// \file gemma_npu_sequence.hpp +/// \brief gemma_npu_sequence class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the gemma_npu_sequence class +#pragma once +#include "npu_utils/npu_instr_utils.hpp" +#include "lm_config.hpp" + +/// \brief gemma_npu_sequence class +/// \note This is a class for the gemma_npu_sequence +class gemma_text_npu_sequence{ +public: + gemma_text_npu_sequence(){} + + /// \brief Constructor + /// \param config the configuration + /// \param MAX_L the max length + gemma_text_npu_sequence(LM_Config config, uint32_t MAX_L); + ~gemma_text_npu_sequence(); + + /// \brief Generate the rtp sequence + /// \param seq the sequence + /// \param L the length + void gen_rtp_seq(npu_sequence* seq, const uint32_t L, bool is_sliding_window); + + /// \brief Generate the layer sequence + /// \param seq the sequence + /// \param L the length + void gen_layer_seq(npu_sequence* seq, const uint32_t L, bool is_sliding_window); + + /// \brief Set the max length + /// \param MAX_L the max length + void set_max_length(const uint32_t MAX_L); + + /// \brief Generate the mha engine sequence + /// \param seq the sequence + /// \param L_begin the begin length + /// \param L_end the end length + void gen_mha_engine_seq(npu_sequence* seq, const uint32_t L_begin, const uint32_t L_end, bool is_sliding_window, int buffer_length); + + /// \brief Get the k offset + size_t get_k_offset() const; + size_t get_v_offset() const; + +private: + struct Impl; + Impl* _impl; +}; diff --git a/src/detail/include/models/gpt_oss/gpt_oss_npu.hpp b/src/detail/include/models/gpt_oss/gpt_oss_npu.hpp new file mode 100644 index 000000000..9d51e6f7d --- /dev/null +++ b/src/detail/include/models/gpt_oss/gpt_oss_npu.hpp @@ -0,0 +1,75 @@ +/// \file gpt_oss_npu.hpp +/// \brief gpt_oss_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the gpt_oss_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/gpt_oss/gpt_oss_npu_sequence.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class gpt_oss_npu : public causal_lm{ +public: + /// \brief initialize the gpt_oss_npu + /// \param config the configuration + /// \param npu_instance the npu instance + gpt_oss_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~gpt_oss_npu(); + + /// \brief forward the gpt_oss_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer forward(buffer& x); + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/gpt_oss/gpt_oss_npu_sequence.hpp b/src/detail/include/models/gpt_oss/gpt_oss_npu_sequence.hpp new file mode 100644 index 000000000..820557963 --- /dev/null +++ b/src/detail/include/models/gpt_oss/gpt_oss_npu_sequence.hpp @@ -0,0 +1,51 @@ +/// \file gpt_oss_npu_sequence.hpp +/// \brief gpt_oss_npu_sequence class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the gpt_oss_npu_sequence class +#pragma once +#include "npu_utils/npu_instr_utils.hpp" +#include "lm_config.hpp" + +/// \brief gpt_oss_npu_sequence class +/// \note This is a class for the gpt_oss_npu_sequence +class gpt_oss_npu_sequence{ +public: + gpt_oss_npu_sequence(){} + + /// \brief Constructor + /// \param config the configuration + /// \param MAX_L the max length + gpt_oss_npu_sequence(LM_Config config, uint32_t MAX_L); + ~gpt_oss_npu_sequence(); + + /// \brief Generate the layer sequence + /// \param seq the sequence + /// \param L the length + void gen_layer_seq(npu_sequence* seq, const int L, bool is_sliding_window); + + /// \brief Set the max length + /// \param MAX_L the max length + void set_max_length(const uint32_t MAX_L); + + /// \brief Generate the mha engine sequence + /// \param seq the sequence + /// \param L_begin the begin length + /// \param L_end the end length + void gen_mha_engine_seq(npu_sequence* seq, const uint32_t L_begin, const uint32_t L_end, bf16* sinks, bool is_sliding_window); + + /// \brief Generate the lm head sequence + /// \param seq the sequence + void generate_lm_head_seq(npu_sequence* seq); + + /// \brief Get the k03 offset + size_t get_k03_offset() const; + size_t get_k47_offset() const; + size_t get_v03_offset() const; + size_t get_v47_offset() const; + +private: + struct Impl; + Impl* _impl; +}; diff --git a/src/detail/include/models/hunyuan/hunyuan_npu.hpp b/src/detail/include/models/hunyuan/hunyuan_npu.hpp new file mode 100644 index 000000000..0b945d675 --- /dev/null +++ b/src/detail/include/models/hunyuan/hunyuan_npu.hpp @@ -0,0 +1,73 @@ +/// \file hunyuan_npu.hpp +/// \brief hunyuan_npu class +/// \author FastFlowLM Team +/// \date 2026-09-01 +/// \note This is a header file for the hunyuan_npu class, the engine for the +/// `hunyuan-dense` architecture (Hy-MT2-1.8B). Q4_0 body, per head q/k +/// RMS norms applied *after* rope. +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class hunyuan_npu : public causal_lm{ +public: + /// \brief initialize the hunyuan_npu + /// \param config the configuration + /// \param npu_instance the npu instance + hunyuan_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~hunyuan_npu(); + + /// \brief forward the hunyuan_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/lfm2/lfm2_npu.hpp b/src/detail/include/models/lfm2/lfm2_npu.hpp new file mode 100644 index 000000000..e41f9e96b --- /dev/null +++ b/src/detail/include/models/lfm2/lfm2_npu.hpp @@ -0,0 +1,74 @@ +/// \file lfm2_npu.hpp +/// \brief lfm2_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the lfm2_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "lfm2_npu.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class lfm2_npu : public causal_lm{ +public: + /// \brief initialize the lfm2_npu + /// \param config the configuration + /// \param npu_instance the npu instance + lfm2_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~lfm2_npu(); + + /// \brief forward the lfm2_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/llama/llama_npu.hpp b/src/detail/include/models/llama/llama_npu.hpp new file mode 100644 index 000000000..e2cb578da --- /dev/null +++ b/src/detail/include/models/llama/llama_npu.hpp @@ -0,0 +1,74 @@ +/// \file llama_npu.hpp +/// \brief llama_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the llama_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/llama/llama_npu_sequence.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class llama_npu : public causal_lm{ +public: + /// \brief initialize the llama_npu + /// \param config the configuration + /// \param npu_instance the npu instance + llama_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~llama_npu(); + + /// \brief forward the llama_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/llama/llama_npu_sequence.hpp b/src/detail/include/models/llama/llama_npu_sequence.hpp new file mode 100644 index 000000000..215aaa8be --- /dev/null +++ b/src/detail/include/models/llama/llama_npu_sequence.hpp @@ -0,0 +1,69 @@ +/// \file llama_npu_sequence.hpp +/// \brief llama_npu_sequence class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the llama_npu_sequence class +#pragma once +#include "npu_utils/npu_instr_utils.hpp" +#include "lm_config.hpp" + +struct llama_desc; + +/// \brief llama_npu_sequence class +/// \note This is a class for the llama_npu_sequence +class llama_npu_sequence{ +public: + llama_npu_sequence(){} + + /// \brief Constructor + /// \param desc the model definition, owned by llama_npu::Impl; the weight + /// descriptors it holds are read directly when moving weights + /// \param config the configuration + /// \param MAX_L the max length + llama_npu_sequence(llama_desc& desc, LM_Config config, uint32_t MAX_L); + ~llama_npu_sequence(); + + /// \brief Generate the decode rtp sequence, run once per token ahead of + /// the per-layer sequences (the RTPs it writes are the same for + /// every layer) + /// \param seq the sequence + /// \param L the length + void gen_layer_rtp_seq(npu_sequence* seq, const uint32_t L); + + /// \brief Generate the layer sequence + /// \param seq the sequence + /// \param L the length + void gen_layer_seq(npu_sequence* seq, const uint32_t L); + + /// \brief Generate the dequant sequence + /// \param seq the sequence + /// \param D_in the input dimension + /// \param D_out the output dimension + /// \param weight_offset the weight offset + void gen_dequant_seq(npu_sequence* seq, const size_t D_in, const size_t D_out, const size_t weight_offset); + + /// \brief Generate the lm head sequence + /// \param seq the sequence + void gen_lm_head_seq(npu_sequence* seq); + + /// \brief Set the max length + /// \param MAX_L the max length + void set_max_length(const uint32_t MAX_L); + + /// \brief Generate the mha engine sequence + /// \param seq the sequence + /// \param L_begin the begin length + /// \param L_end the end length + void gen_mha_engine_seq(npu_sequence* seq, const uint32_t L_begin, const uint32_t L_end); + + /// \brief Get the k03 offset + size_t get_k03_offset() const; + size_t get_k47_offset() const; + size_t get_v03_offset() const; + size_t get_v47_offset() const; + +private: + struct Impl; + Impl* _impl; +}; diff --git a/src/detail/include/models/nanbeige/nanbeige_npu.hpp b/src/detail/include/models/nanbeige/nanbeige_npu.hpp new file mode 100644 index 000000000..20224180b --- /dev/null +++ b/src/detail/include/models/nanbeige/nanbeige_npu.hpp @@ -0,0 +1,74 @@ +/// \file llama_npu.hpp +/// \brief llama_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the llama_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/nanbeige/nanbeige_npu_sequence.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class nanbeige_npu : public causal_lm{ +public: + /// \brief initialize the nanbeige_npu + /// \param config the configuration + /// \param npu_instance the npu instance + nanbeige_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~nanbeige_npu(); + + /// \brief forward the nanbeige_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/nanbeige/nanbeige_npu_sequence.hpp b/src/detail/include/models/nanbeige/nanbeige_npu_sequence.hpp new file mode 100644 index 000000000..b961265ef --- /dev/null +++ b/src/detail/include/models/nanbeige/nanbeige_npu_sequence.hpp @@ -0,0 +1,70 @@ +/// \file nanbeige_npu_sequence.hpp +/// \brief nanbeige_npu_sequence class +/// \author FastFlowLM Team +/// \date 2026-03-24 +/// \version 0.9.36 +/// \note This is a header file for the nanbeige_npu_sequence class +#pragma once +#include "npu_utils/npu_instr_utils.hpp" +#include "lm_config.hpp" + +struct nanbeige_desc; + +/// \brief nanbeige_npu_sequence class +/// \note This is a class for the nanbeige_npu_sequence +class nanbeige_npu_sequence{ +public: + nanbeige_npu_sequence(){} + + /// \brief Constructor + /// \param desc the model definition, owned by nanbeige_npu::Impl; the weight + /// descriptors it holds are read directly when moving weights + /// \param config the configuration + /// \param MAX_L the max length + nanbeige_npu_sequence(nanbeige_desc& desc, LM_Config config, uint32_t MAX_L); + ~nanbeige_npu_sequence(); + + /// \brief Generate the decode rtp sequence, run once per token ahead of + /// the per-layer sequences (the RTPs it writes are the same for + /// every layer) + /// \param seq the sequence + /// \param L the length + void gen_layer_rtp_seq(npu_sequence* seq, const uint32_t L); + + /// \brief Generate the layer sequence + /// \param seq the sequence + /// \param L the length + void gen_layer_seq(npu_sequence* seq, const uint32_t L); + + /// \brief Generate the dequant sequence + /// \param seq the sequence + /// \param D_in the input dimension + /// \param D_out the output dimension + /// \param weight_offset the weight offset + void gen_dequant_seq(npu_sequence* seq, const size_t D_in, const size_t D_out, const size_t weight_offset); + + /// \brief Generate the lm head sequence + /// \param seq the sequence + void gen_lm_head_seq(npu_sequence* seq); + + /// \brief Set the max length + /// \param MAX_L the max length + void set_max_length(const uint32_t MAX_L); + + /// \brief Generate the mha engine sequence + /// \param seq the sequence + /// \param L_begin the begin length + /// \param L_end the end length + /// \param MAX_L the max length + void gen_mha_engine_seq(npu_sequence* seq, const uint32_t L_begin, const uint32_t L_end, const uint32_t MAX_L); + + /// \brief Get the k03 offset + size_t get_k03_offset() const; + size_t get_k47_offset() const; + size_t get_v03_offset() const; + size_t get_v47_offset() const; + +private: + struct Impl; + Impl* _impl; +}; diff --git a/src/detail/include/models/phi4/phi4_npu.hpp b/src/detail/include/models/phi4/phi4_npu.hpp new file mode 100644 index 000000000..25478de4d --- /dev/null +++ b/src/detail/include/models/phi4/phi4_npu.hpp @@ -0,0 +1,74 @@ +/// \file phi4_npu.hpp +/// \brief phi4_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the phi4_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/phi4/phi4_npu_sequence.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class phi4_npu : public causal_lm{ +public: + /// \brief initialize the phi4_npu + /// \param config the configuration + /// \param npu_instance the npu instance + phi4_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~phi4_npu(); + + /// \brief forward the phi4_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/phi4/phi4_npu_sequence.hpp b/src/detail/include/models/phi4/phi4_npu_sequence.hpp new file mode 100644 index 000000000..bde2246b5 --- /dev/null +++ b/src/detail/include/models/phi4/phi4_npu_sequence.hpp @@ -0,0 +1,67 @@ +/// \file phi4_npu_sequence.hpp +/// \brief phi4_npu_sequence class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the phi4_npu_sequence class +#pragma once +#include "npu_utils/npu_instr_utils.hpp" +#include "lm_config.hpp" + +struct phi4_desc; + +/// \brief phi4_npu_sequence class +/// \note This is a class for the phi4_npu_sequence +class phi4_npu_sequence{ +public: + phi4_npu_sequence(){} + + /// \brief Constructor + /// \param desc the model definition, owned by phi4_npu::Impl; the weight + /// descriptors it holds are read directly when moving weights + /// \param config the configuration + /// \param MAX_L the max length + phi4_npu_sequence(phi4_desc& desc, LM_Config config, uint32_t MAX_L); + ~phi4_npu_sequence(); + + /// \brief Generate the rtp sequence + /// \param seq the sequence + /// \param L the length + void gen_layer_rtp_seq(npu_sequence* seq, const uint32_t L); + + /// \brief Generate the layer sequence + /// \param seq the sequence + /// \param L the length + void gen_layer_seq(npu_sequence* seq, const uint32_t L); + + /// \brief Generate the dequant sequence + /// \param seq the sequence + /// \param D_in the input dimension + /// \param D_out the output dimension + /// \param weight_offset the weight offset + void gen_dequant_seq(npu_sequence* seq, const size_t D_in, const size_t D_out, const size_t weight_offset); + + /// \brief Generate the lm head sequence + /// \param seq the sequence + void gen_lm_head_seq(npu_sequence* seq); + + /// \brief Set the max length + /// \param MAX_L the max length + void set_max_length(const uint32_t MAX_L); + + /// \brief Generate the mha engine sequence + /// \param seq the sequence + /// \param L_begin the begin length + /// \param L_end the end length + void gen_mha_engine_seq(npu_sequence* seq, const uint32_t L_begin, const uint32_t L_end); + + /// \brief Get the k03 offset + size_t get_k03_offset() const; + size_t get_k47_offset() const; + size_t get_v03_offset() const; + size_t get_v47_offset() const; + +private: + struct Impl; + Impl* _impl; +}; diff --git a/src/detail/include/models/qwen2/qwen2_npu.hpp b/src/detail/include/models/qwen2/qwen2_npu.hpp new file mode 100644 index 000000000..562949578 --- /dev/null +++ b/src/detail/include/models/qwen2/qwen2_npu.hpp @@ -0,0 +1,73 @@ +/// \file qwen2_npu.hpp +/// \brief qwen2_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the qwen2_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class qwen2_npu : public causal_lm{ +public: + /// \brief initialize the qwen2_npu + /// \param config the configuration + /// \param npu_instance the npu instance + qwen2_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~qwen2_npu(); + + /// \brief forward the qwen2_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/qwen2vl/qwen2vl_npu.hpp b/src/detail/include/models/qwen2vl/qwen2vl_npu.hpp new file mode 100644 index 000000000..ce96c1977 --- /dev/null +++ b/src/detail/include/models/qwen2vl/qwen2vl_npu.hpp @@ -0,0 +1,109 @@ +/// \file qwen2vl_npu.hpp +/// \brief qwen2vl_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the qwen2vl_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + +constexpr unsigned int QWEN2_PATCH_SIZE = 14; +constexpr unsigned int QWEN2_IMAGE_MERGE_SIZE=2; +constexpr unsigned int QWEN2_SPATIAL_MERGE_SIZE=2; +constexpr unsigned int QWEN2_SHORTEST_EDGE = 3136; +constexpr unsigned int QWEN2_LONGEST_EDGE = 12845056; +constexpr float QWEN2_VISION_RESCALE_FACTOR = 0.00392156862745098; +constexpr float QWEN2_VISION_RESCALE_IMAGE_MEAN_R = 0.48145466f; +constexpr float QWEN2_VISION_RESCALE_IMAGE_MEAN_G = 0.4578275f; +constexpr float QWEN2_VISION_RESCALE_IMAGE_MEAN_B = 0.40821073f; +constexpr float QWEN2_VISION_RESCALE_IMAGE_STD_R = 0.26862954f; +constexpr float QWEN2_VISION_RESCALE_IMAGE_STD_G = 0.26130258f; +constexpr float QWEN2_VISION_RESCALE_IMAGE_STD_B = 0.27577711f; +constexpr unsigned int QWEN2_WINDOW_ATTENTION_PIXEL_SIZE = 122; +constexpr unsigned int QWEN2_TEMPORAL_PATCH_SIZE = 2; +constexpr unsigned int QWEN2_MERGE_SIZE = 2; +typedef struct { + int height; + int width; + int height_resized; // assigned by image preprocessing + int width_resized; + int grid_h; + int grid_w; + + bytes _data; + +} qwen2vl_image_t; + + + +typedef struct { + std::vector images; + std::vector _data__processed; + unsigned int num_images; +}qwen2vl_image_payload_t; + + + +class qwen2vl_npu : public causal_lm{ +public: + /// \brief initialize the qwen2vl_npu + /// \param config the configuration + /// \param npu_instance the npu instance + qwen2vl_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~qwen2vl_npu(); + + /// \brief forward the qwen2vl_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/qwen3/qwen3_npu.hpp b/src/detail/include/models/qwen3/qwen3_npu.hpp new file mode 100644 index 000000000..235b5254b --- /dev/null +++ b/src/detail/include/models/qwen3/qwen3_npu.hpp @@ -0,0 +1,74 @@ +/// \file qwen3_npu.hpp +/// \brief qwen3_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the qwen3_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "models/qwen3/qwen3_npu_sequence.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class qwen3_npu : public causal_lm{ +public: + /// \brief initialize the qwen3_npu + /// \param config the configuration + /// \param npu_instance the npu instance + qwen3_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~qwen3_npu(); + + /// \brief forward the qwen3_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/qwen3/qwen3_npu_sequence.hpp b/src/detail/include/models/qwen3/qwen3_npu_sequence.hpp new file mode 100644 index 000000000..b58aacbb8 --- /dev/null +++ b/src/detail/include/models/qwen3/qwen3_npu_sequence.hpp @@ -0,0 +1,69 @@ +/// \file qwen3_npu_sequence.hpp +/// \brief qwen3_npu_sequence class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the qwen3_npu_sequence class +#pragma once +#include "npu_utils/npu_instr_utils.hpp" +#include "lm_config.hpp" + +struct qwen3_desc; + +/// \brief qwen3_npu_sequence class +/// \note This is a class for the qwen3_npu_sequence +class qwen3_npu_sequence{ +public: + qwen3_npu_sequence(){} + + /// \brief Constructor + /// \param desc the model definition, owned by qwen3_npu::Impl; the weight + /// descriptors it holds are read directly when moving weights + /// \param config the configuration + /// \param MAX_L the max length + qwen3_npu_sequence(qwen3_desc& desc, LM_Config config, uint32_t MAX_L); + ~qwen3_npu_sequence(); + + /// \brief Generate the decode rtp sequence, run once per token ahead of + /// the per-layer sequences (the RTPs it writes are the same for + /// every layer) + /// \param seq the sequence + /// \param L the length + void gen_layer_rtp_seq(npu_sequence* seq, const uint32_t L); + + /// \brief Generate the layer sequence + /// \param seq the sequence + /// \param L the length + void gen_layer_seq(npu_sequence* seq, const uint32_t L); + + /// \brief Set the max length + /// \param MAX_L the max length + void set_max_length(const uint32_t MAX_L); + + /// \brief Generate the mha engine sequence + /// \param seq the sequence + /// \param L_begin the begin length + /// \param L_end the end length + void gen_mha_engine_seq(npu_sequence* seq, const uint32_t L_begin, const uint32_t L_end); + + /// \brief Generate the dequant sequence + /// \param seq the sequence + /// \param D_in the input dimension + /// \param D_out the output dimension + /// \param weight_offset the weight offset + void gen_dequant_seq(npu_sequence* seq, const size_t D_in, const size_t D_out, const size_t weight_offset); + + /// \brief Generate the lm head sequence + /// \param seq the sequence + void gen_lm_head_seq(npu_sequence* seq); + + /// \brief Get the k03 offset + size_t get_k03_offset() const; + size_t get_k47_offset() const; + size_t get_v03_offset() const; + size_t get_v47_offset() const; + +private: + struct Impl; + Impl* _impl; +}; diff --git a/src/detail/include/models/qwen3_5_omni/qwen3_5_omni.hpp b/src/detail/include/models/qwen3_5_omni/qwen3_5_omni.hpp new file mode 100644 index 000000000..1351c47a6 --- /dev/null +++ b/src/detail/include/models/qwen3_5_omni/qwen3_5_omni.hpp @@ -0,0 +1,95 @@ +/// \file qwen3_5_omni.hpp +/// \brief qwen3_5_omni class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the qwen3_5_omni class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +typedef struct { + int height; + int width; + int height_resized; // assigned by image preprocessing + int width_resized; + bytes _data; + +} qwen3_5_omni_image_t; + +typedef struct { + std::vector _processed_pixel_values; // [num_of_image][1d array of processed image values] + std::vector< std::vector> image_grid_h_w; //[num_of_images][grid_h, grid_w] + std::vector num_soft_tokens_per_image; // [num_of_image] + unsigned int num_images; +}qwen3_5_omni_image_payload_t; + +struct qwen3_5_omni_audio_payload_t { + // per-audio mel spectrogram data + std::vector> mel_spectrograms; // [num_audios][frames * bins], row-major + std::vector mel_spectrogram_frames_per_audio; // [num_audios] + std::vector mel_spectrogram_bins_per_audio; // [num_audios] + unsigned int num_audios = 0; + std::vector audio_tokens; // [num_audios, string of tokens ] +}; + +typedef struct { + qwen3_5_omni_image_payload_t image_payload; + qwen3_5_omni_audio_payload_t audio_payload; +} qwen3_5_omni_payload_t; + +typedef struct { + buffer logits; + buffer hidden_states; +} qwen3_5_omni_thinker_result_t; + +class qwen3_5_omni{ +public: + /// \brief initialize the qwen3_5_omni + /// \param config the configuration + /// \param npu_instance the npu instance + qwen3_5_omni(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~qwen3_5_omni(); + + /// \brief forward the qwen3_5_omni + /// \param ids the ids + /// \return the output tensor + qwen3_5_omni_thinker_result_t forward(int ids); + buffer say(qwen3_5_omni_thinker_result_t thinker_res); + qwen3_5_omni_thinker_result_t prefill(std::vector& ids, void* payload = nullptr); + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L); + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx); + + /// \brief update the max length + void clear_context(); + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L); + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length(); + int checkpoint(); + int restore(); +private: + struct Impl; + Impl* _impl; +}; diff --git a/src/detail/include/models/qwen3_5vl/qwen3_5vl_npu.hpp b/src/detail/include/models/qwen3_5vl/qwen3_5vl_npu.hpp new file mode 100644 index 000000000..8df1bb346 --- /dev/null +++ b/src/detail/include/models/qwen3_5vl/qwen3_5vl_npu.hpp @@ -0,0 +1,145 @@ +/// \file qwen3vl_npu.hpp +/// \brief qwen3vl_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the qwen3vl_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + +// #define QWEN3_5_VL_4B 1 + +// #ifdef QWEN3_5_VL_4B + + + // constexpr unsigned int QWEN3_5_PATCH_SIZE = 16; + // constexpr unsigned int QWEN3_5_IMAGE_MERGE_SIZE=2; + // constexpr unsigned int QWEN3_5_SPATIAL_MERGE_SIZE=2; + // constexpr unsigned int QWEN3_5_SHORTEST_EDGE = 65536; + // constexpr unsigned int QWEN3_5_LONGEST_EDGE = 16777216; + // constexpr float QWEN3_5_VISION_RESCALE_FACTOR = 0.00392156862745098; + // constexpr float QWEN3_5_VISION_RESCALE_IMAGE_MEAN = 0.5f; + // constexpr float QWEN3_5_VISION_RESCALE_IMAGE_STD = 0.5f; + // constexpr unsigned int QWEN3_5_TEMPORAL_PATCH_SIZE = 2; + // constexpr unsigned int QWEN3_5_MERGE_SIZE = QWEN3_5_IMAGE_MERGE_SIZE; + + + + +// #endif + +typedef struct { + int height; + int width; + int height_resized; // assigned by image preprocessing + int width_resized; + int grid_h; + int grid_w; + + bytes _data; + +} qwen3_5vl_image_t; + + + +typedef struct { + std::vector images; + std::vector _data__processed; + unsigned int num_images; +}qwen3_5vl_image_payload_t; + + + +class qwen3_5vl_npu : public causal_lm{ +public: + /// \brief initialize the qwen3vl_npu + /// \param config the configuration + /// \param npu_instance the npu instance + qwen3_5vl_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~qwen3_5vl_npu(); + + /// \brief forward the qwen3vl_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; + // parameters for vision process in qwen3.5 vl + + unsigned int QWEN3_5_PATCH_SIZE; + unsigned int QWEN3_5_IMAGE_MERGE_SIZE; + unsigned int QWEN3_5_SPATIAL_MERGE_SIZE; + unsigned int QWEN3_5_SHORTEST_EDGE; + unsigned int QWEN3_5_LONGEST_EDGE; + float QWEN3_5_VISION_RESCALE_FACTOR; + float QWEN3_5_VISION_RESCALE_IMAGE_MEAN; + float QWEN3_5_VISION_RESCALE_IMAGE_STD; + unsigned int QWEN3_5_TEMPORAL_PATCH_SIZE; + unsigned int QWEN3_5_MERGE_SIZE; + + + inline void load_vision_preprocess_parameters(LM_Config& config){ + // Note: this should be called by Impl:: constructor + const nlohmann::json& vc = config.sub("vision_config"); + QWEN3_5_PATCH_SIZE = vc.value("QWEN3_5_PATCH_SIZE", -1); + QWEN3_5_IMAGE_MERGE_SIZE = vc.value("QWEN3_5_IMAGE_MERGE_SIZE", -1); + QWEN3_5_SPATIAL_MERGE_SIZE = vc.value("QWEN3_5_SPATIAL_MERGE_SIZE", -1); + QWEN3_5_SHORTEST_EDGE = vc.value("QWEN3_5_SHORTEST_EDGE", -1); + QWEN3_5_LONGEST_EDGE = vc.value("QWEN3_5_LONGEST_EDGE", -1); + QWEN3_5_VISION_RESCALE_FACTOR = vc.value("QWEN3_5_VISION_RESCALE_FACTOR", -1.0f); + QWEN3_5_VISION_RESCALE_IMAGE_MEAN = vc.value("QWEN3_5_VISION_RESCALE_IMAGE_MEAN", -1.0f); + QWEN3_5_VISION_RESCALE_IMAGE_STD = vc.value("QWEN3_5_VISION_RESCALE_IMAGE_STD", -1.0f); + QWEN3_5_TEMPORAL_PATCH_SIZE = vc.value("QWEN3_5_TEMPORAL_PATCH_SIZE", -1); + + QWEN3_5_MERGE_SIZE = QWEN3_5_IMAGE_MERGE_SIZE; + + } +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/qwen3_6_moe/qwen3_6_moe_npu.hpp b/src/detail/include/models/qwen3_6_moe/qwen3_6_moe_npu.hpp new file mode 100644 index 000000000..0e28536f3 --- /dev/null +++ b/src/detail/include/models/qwen3_6_moe/qwen3_6_moe_npu.hpp @@ -0,0 +1,145 @@ +/// \file qwen3.6_moe_npu.hpp +/// \brief qwen3.6_moe_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the qwen3.6_moe_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + +// #define QWEN3_6_MOE_VL_4B 1 + +// #ifdef QWEN3_6_MOE_VL_4B + + + // constexpr unsigned int QWEN3_6_MOE_PATCH_SIZE = 16; + // constexpr unsigned int QWEN3_6_MOE_IMAGE_MERGE_SIZE=2; + // constexpr unsigned int QWEN3_6_MOE_SPATIAL_MERGE_SIZE=2; + // constexpr unsigned int QWEN3_6_MOE_SHORTEST_EDGE = 65536; + // constexpr unsigned int QWEN3_6_MOE_LONGEST_EDGE = 16777216; + // constexpr float QWEN3_6_MOE_VISION_RESCALE_FACTOR = 0.00392156862745098; + // constexpr float QWEN3_6_MOE_VISION_RESCALE_IMAGE_MEAN = 0.5f; + // constexpr float QWEN3_6_MOE_VISION_RESCALE_IMAGE_STD = 0.5f; + // constexpr unsigned int QWEN3_6_MOE_TEMPORAL_PATCH_SIZE = 2; + // constexpr unsigned int QWEN3_6_MOE_MERGE_SIZE = QWEN3_6_MOE_IMAGE_MERGE_SIZE; + + + + +// #endif + +typedef struct { + int height; + int width; + int height_resized; // assigned by image preprocessing + int width_resized; + int grid_h; + int grid_w; + + bytes _data; + +} qwen3_6_moe_image_t; + + + +typedef struct { + std::vector images; + std::vector _data__processed; + unsigned int num_images; +}qwen3_6_moe_image_payload_t; + + + +class qwen3_6_moe_npu : public causal_lm{ +public: + /// \brief initialize the qwen3vl_npu + /// \param config the configuration + /// \param npu_instance the npu instance + qwen3_6_moe_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~qwen3_6_moe_npu(); + + /// \brief forward the qwen3vl_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; + // parameters for vision process in qwen3.5 vl + + unsigned int QWEN3_6_MOE_PATCH_SIZE; + unsigned int QWEN3_6_MOE_IMAGE_MERGE_SIZE; + unsigned int QWEN3_6_MOE_SPATIAL_MERGE_SIZE; + unsigned int QWEN3_6_MOE_SHORTEST_EDGE; + unsigned int QWEN3_6_MOE_LONGEST_EDGE; + float QWEN3_6_MOE_VISION_RESCALE_FACTOR; + float QWEN3_6_MOE_VISION_RESCALE_IMAGE_MEAN; + float QWEN3_6_MOE_VISION_RESCALE_IMAGE_STD; + unsigned int QWEN3_6_MOE_TEMPORAL_PATCH_SIZE; + unsigned int QWEN3_6_MOE_MERGE_SIZE; + + + inline void load_vision_preprocess_parameters(LM_Config& config){ + // Note: this should be called by Impl:: constructor + const nlohmann::json& vc = config.sub("vision_config"); + QWEN3_6_MOE_PATCH_SIZE = vc.value("QWEN3_6_MOE_PATCH_SIZE", -1); + QWEN3_6_MOE_IMAGE_MERGE_SIZE = vc.value("QWEN3_6_MOE_IMAGE_MERGE_SIZE", -1); + QWEN3_6_MOE_SPATIAL_MERGE_SIZE = vc.value("QWEN3_6_MOE_SPATIAL_MERGE_SIZE", -1); + QWEN3_6_MOE_SHORTEST_EDGE = vc.value("QWEN3_6_MOE_SHORTEST_EDGE", -1); + QWEN3_6_MOE_LONGEST_EDGE = vc.value("QWEN3_6_MOE_LONGEST_EDGE", -1); + QWEN3_6_MOE_VISION_RESCALE_FACTOR = vc.value("QWEN3_6_MOE_VISION_RESCALE_FACTOR", -1.0f); + QWEN3_6_MOE_VISION_RESCALE_IMAGE_MEAN = vc.value("QWEN3_6_MOE_VISION_RESCALE_IMAGE_MEAN", -1.0f); + QWEN3_6_MOE_VISION_RESCALE_IMAGE_STD = vc.value("QWEN3_6_MOE_VISION_RESCALE_IMAGE_STD", -1.0f); + QWEN3_6_MOE_TEMPORAL_PATCH_SIZE = vc.value("QWEN3_6_MOE_TEMPORAL_PATCH_SIZE", -1); + + QWEN3_6_MOE_MERGE_SIZE = QWEN3_6_MOE_IMAGE_MERGE_SIZE; + + } +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/qwen3vl/qwen3vl_npu.hpp b/src/detail/include/models/qwen3vl/qwen3vl_npu.hpp new file mode 100644 index 000000000..5a6b17326 --- /dev/null +++ b/src/detail/include/models/qwen3vl/qwen3vl_npu.hpp @@ -0,0 +1,108 @@ +/// \file qwen3vl_npu.hpp +/// \brief qwen3vl_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the qwen3vl_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +//Parameter for QWEN3IMAGE +constexpr unsigned int QWEN3_PATCH_SIZE = 16; +constexpr unsigned int QWEN3_IMAGE_MERGE_SIZE=2; +constexpr unsigned int QWEN3_SPATIAL_MERGE_SIZE=2; +constexpr unsigned int QWEN3_SHORTEST_EDGE = 65536; +constexpr unsigned int QWEN3_LONGEST_EDGE = 16777216; +constexpr float QWEN3_VISION_RESCALE_FACTOR = 0.00392156862745098; +constexpr float QWEN3_VISION_RESCALE_IMAGE_MEAN = 0.5f; +constexpr float QWEN3_VISION_RESCALE_IMAGE_STD = 0.5f; +constexpr unsigned int QWEN3_TEMPORAL_PATCH_SIZE = 2; +constexpr unsigned int QWEN3_MERGE_SIZE = 2; + + +typedef struct { + int height; + int width; + int height_resized; // assigned by image preprocessing + int width_resized; + int grid_h; + int grid_w; + + bytes _data; + +} qwen3vl_image_t; + + + +typedef struct { + std::vector images; + std::vector _data__processed; + unsigned int num_images; +}qwen3vl_image_payload_t; + + + +class qwen3vl_npu : public causal_lm{ +public: + /// \brief initialize the qwen3vl_npu + /// \param config the configuration + /// \param npu_instance the npu instance + qwen3vl_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~qwen3vl_npu(); + + /// \brief forward the qwen3vl_npu + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/models/qwen3vl_flash/qwen3vl_flash.hpp b/src/detail/include/models/qwen3vl_flash/qwen3vl_flash.hpp new file mode 100644 index 000000000..75dce8ed0 --- /dev/null +++ b/src/detail/include/models/qwen3vl_flash/qwen3vl_flash.hpp @@ -0,0 +1,72 @@ +/// \file qwen3vl_flash.hpp +/// \brief qwen3vl_flash class +/// \author FastFlowLM Team +/// \date 2026-09-10 +/// \version 0.9.28 +/// \note This is a header file for the qwen3vl_flash class +/// +/// qwen3vl_flash is a second engine for the same Qwen3-VL checkpoint as +/// qwen3vl_npu. It runs prefill on a single fused overlay -- columns 0-5 a +/// dequant+mm array, columns 6-7 one attention CU (mha_d128_q4_1cu) -- instead +/// of reconfiguring the array between mm.xclbin and attn.xclbin twice per +/// layer, and overlaps the vision encoder with the text prefill setup. It is +/// tuned for short contexts; qwen3vl_npu remains the general-purpose engine. +/// +/// The image payload types and the QWEN3_* preprocessing constants are shared +/// with qwen3vl_npu, so this header pulls them in rather than redefining them: +/// the application builds one qwen3vl_image_payload_t and hands it to either +/// engine. +#pragma once +#include "models/qwen3vl/qwen3vl_npu.hpp" + + +class qwen3vl_flash : public causal_lm{ +public: + /// \brief initialize the qwen3vl_flash + /// \param config the configuration + /// \param npu_instance the npu instance + qwen3vl_flash(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~qwen3vl_flash(); + + /// \brief forward the qwen3vl_flash + /// \param ids the ids + /// \return the output tensor + buffer forward(int ids) override; + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief update the max length + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief update the max length + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; +private: + struct Impl; + Impl* _impl; +}; diff --git a/src/detail/include/models/whisper/whisper_npu.hpp b/src/detail/include/models/whisper/whisper_npu.hpp new file mode 100644 index 000000000..002f46227 --- /dev/null +++ b/src/detail/include/models/whisper/whisper_npu.hpp @@ -0,0 +1,50 @@ +/// \file gemma_npu.hpp +/// \brief gemma_npu class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.28 +/// \note This is a header file for the gemma_npu class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + + +class whisper_npu{ +public: + /// \brief initialize the gemma_npu + /// \param config the configuration + /// \param npu_instance the npu instance + whisper_npu(Whisper_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 448); + ~whisper_npu(); + + /// \brief forward the whisper_npu + /// \param ids the ids + /// \return the output tensor + buffer encode_audio(buffer& mel_feature); + + buffer decode_audio(int last_ids); + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx); + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length(); + + void clear_context(); + +private: + struct Impl; + Impl* _impl; +}; + diff --git a/src/detail/include/modules/dequant.hpp b/src/detail/include/modules/dequant.hpp new file mode 100644 index 000000000..ba1d2982b --- /dev/null +++ b/src/detail/include/modules/dequant.hpp @@ -0,0 +1,45 @@ +/// \file dequant.hpp +/// \brief dequant class +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This is a header file for the dequant class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_instr_utils.hpp" + +/// \brief dequant class +/// \note This is a class for the dequant layer +class Dequant{ +public: + Dequant(){} + + /// \brief Constructor + /// \param config the configuration + /// \param xclbin_name the xclbin name + /// \param npu the npu manager + Dequant(LM_Config& config); + ~Dequant(); + + typedef enum: int { + Q4_1 = 0, + Q8_0 = 1, + Q4_0 = 2 + } quant_block_t; + + void reorder_cpy(u8 *dst, buffer &src, + quant_block_t quant_block_type, + const int quant_matrix_row, + const int quant_matrix_col, + const int vertical_blocks=2, + const int vetrical_block_interleave_byte_size=-1); + void generate_dequant_q4_1_seq(npu_sequence* seq, const uint32_t D_in, const uint32_t D_out, const uint32_t weight_offset, int mode); + /// \brief same sequence as q4_1, for the 4.5 bpw (4608 byte) block + void generate_dequant_q4_0_seq(npu_sequence* seq, const uint32_t D_in, const uint32_t D_out, const uint32_t weight_offset, int mode); + void generate_dequant_q80_packed_in_q4nx_seq(npu_sequence* seq, const uint32_t D_in, const uint32_t D_out, const uint32_t weight_offset, int mode); +private: + struct Impl; + Impl* _impl; + +}; + diff --git a/src/detail/include/modules/embedding.hpp b/src/detail/include/modules/embedding.hpp new file mode 100644 index 000000000..9949cd5e1 --- /dev/null +++ b/src/detail/include/modules/embedding.hpp @@ -0,0 +1,57 @@ +/// \file embedding.hpp +/// \brief embedding class +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This is a header file for the embedding class. + +#pragma once + +#include "typedef.hpp" +#include "tensor_utils/safe_tensors.hpp" + +/// \brief embedding class +/// \note This is a class for the embedding layer +class Embedding{ +public: + vdtype w; + vdtype y; + int vocab_size; + int d_model; + int buffer_id; + + /// \brief Constructor + /// \param vocab_size the vocabulary size + /// \param d_model the model dimension + Embedding(){}; + Embedding(int vocab_size, int d_model){ + this->vocab_size = vocab_size; + this->d_model = d_model; + this->w.resize((size_t)vocab_size * d_model); + this->y.resize(d_model); + } + + /// \brief Forward pass + /// \param x the input tensor + /// \return the output tensor + vdtype forward(int x){ + this->y.copy_from(this->w.begin() + (size_t)x * this->d_model, this->d_model); + return this->y; + } + + /// \brief Forward pass + /// \param x the input tensor + /// \param y the output tensor + /// \return the output tensor + vdtype forward(int x, vdtype& y){ + y.copy_from(this->w.begin() + (size_t)x * this->d_model, this->d_model); + return y; + } + + /// \brief Initialize the weights + /// \param safe_tensors the safe tensors + /// \param weight_name the weight name, which is the name of the weight file + void init_weights(SafeTensors* safe_tensors, std::string weight_name){ + safe_tensors->load_weights(this->w, weight_name + ".weight"); + } +}; \ No newline at end of file diff --git a/src/detail/include/modules/gemm.hpp b/src/detail/include/modules/gemm.hpp new file mode 100644 index 000000000..5ab7a1efe --- /dev/null +++ b/src/detail/include/modules/gemm.hpp @@ -0,0 +1,52 @@ +/// \file gemm.hpp +/// \brief gemm class +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This is a header file for the gemm class +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_instr_utils.hpp" + +/// \brief gemm class +/// \note This is a class for the gemm layer +class Gemm{ +public: + + typedef enum: int { + NO_Activation = 0, + GeLU = 1, + SiLU = 2 + } Activation_Type_t; + Gemm(){} + + /// \brief Constructor + /// \param config the configuration + /// \param xclbin_name the xclbin name + /// \param npu the npu manager + Gemm(LM_Config& config); + ~Gemm(); + + /// \brief Generate the sequence + /// \param seq the npu sequence + /// \param M the M dimension + /// \param K the K dimension + /// \param N the N dimension + /// \param weight_offset the weight offset + /// \param ADD_BIAS whether to add bias + /// \param OUTPUT_MODE the output activation mode + /// \param bias_offset the bias offset + void generate_seq(npu_sequence* seq, const uint32_t M, const uint32_t K, const uint32_t N, const uint32_t weight_offset, bool ADD_BIAS, Activation_Type_t OUTPUT_MODE, const uint32_t bias_offset); + void generate_seq(npu_sequence* seq, const uint32_t M, const uint32_t K, const uint32_t N, const uint32_t weight_offset, bool ADD_BIAS, Activation_Type_t OUTPUT_MODE, const uint32_t bias_offset, + const uint32_t output_offset + ); + uint32_t get_m() const; + uint32_t get_k() const; + uint32_t get_n() const; + +private: + struct Impl; + Impl* _impl; + +}; + diff --git a/src/detail/include/modules/lm_head.hpp b/src/detail/include/modules/lm_head.hpp new file mode 100644 index 000000000..bdf88314f --- /dev/null +++ b/src/detail/include/modules/lm_head.hpp @@ -0,0 +1,44 @@ +/// \file lm_head.hpp +/// \brief lm_head class +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This is a header file for the lm_head class +#pragma once +#include "lm_config.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "npu_utils/npu_utils.hpp" + +/// \brief lm_head class +/// \note This is a class for the lm_head layer +class LMHead{ +public: + LMHead(){} + + /// \brief Constructor + /// \param config the configuration + /// \param xclbin_name the xclbin name + /// \param npu the npu manager + LMHead(LM_Config config, npu_xclbin_manager *npu); + ~LMHead(); + + /// \brief Load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx); + + /// \brief Execute the lm_head + void execute(); + + /// \brief Wait for the lm_head + /// \return the output tensor + buffer wait(); + + /// \brief Get the output tensor + /// \return the output tensor + buffer x_exposed(); + +private: + struct Impl; + Impl* _impl; + +}; diff --git a/src/detail/include/modules/mha.hpp b/src/detail/include/modules/mha.hpp new file mode 100644 index 000000000..dbd9b5a64 --- /dev/null +++ b/src/detail/include/modules/mha.hpp @@ -0,0 +1,47 @@ +/// \file lm_head.hpp +/// \brief lm_head class +/// \author FastFlowLM Team +/// \date 2026-01-23 +/// \version 0.9.29 +/// \note This is a header file for the prefill stage mha +#pragma once +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "npu_utils/npu_utils.hpp" + +typedef enum : uint32_t { + mha_d64_q4, // dh = 64, 4 q head to 1 kv head + mha_d128_q2, // dh = 128, 2 q head to 1 kv head + mha_d128_q3, // dh = 128, 3 q head to 1 kv head + mha_d128_q4, // dh = 128, 4 q head to 1 kv head + mha_d256_q2, // dh = 256, 2 q head to 1 kv head + mha_d256_q4, // dh = 256, 4 q head to 1 kv head + /// dh = 128, 4 q head to 1 kv head, but only one 2-column compute unit, on + /// physical columns 6-7. This is the attention half of a fused prefill + /// overlay whose other six columns are a dequant+mm array; it trades 4x the + /// head rounds for never having to swap the array between the two. + mha_d128_q4_1cu, + mha_not_supported +} mha_type_t; + + +/// \brief MHA class +/// \note This is a class for the mha layer +class MHA{ +public: + MHA(){} + + /// \brief Constructor + /// \param mha_ty the type of mha + /// \param num_kv_heads the number of key-value heads + MHA(mha_type_t& mha_ty, int num_kv_heads); + ~MHA(); + + void generate_mha_sequence(npu_sequence* seq, const uint32_t L_begin, const uint32_t L_end, const uint32_t MAX_L, const bool is_sliding_window, const int WINDOW_SIZE = -1); + + int get_chunk_size(); + +private: + struct Impl; + Impl* _impl; + +}; diff --git a/src/detail/include/modules/sampler.hpp b/src/detail/include/modules/sampler.hpp new file mode 100644 index 000000000..2960c5916 --- /dev/null +++ b/src/detail/include/modules/sampler.hpp @@ -0,0 +1,69 @@ +/// \file sampler.hpp +/// \brief sampler class +/// \author FastFlowLM Team +/// \date 2025-06-24 +/// \version 0.9.10 +/// \note This class is used to sample the tokens. +#pragma once + +#include "typedef.hpp" +#include + +/// \brief sampler config +/// \param temperature the temperature +/// \param top_k the top k +/// \param top_p the top p +/// \param rep_penalty the rep penalty +/// \param freq_penalty the freq penalty +/// \param rep_penalty_window the rep penalty window +typedef struct sampler_config_{ + float temperature = 1.0f; + int top_k = 5; + float top_p = 0.9f; + float rep_penalty = 1.0f; + float freq_penalty = 1.0f; + int rep_penalty_window = 1024; + int freq_penalty_window = 1024; // Window size for frequency penalty +} sampler_config; + +typedef std::pair logits_t; +typedef std::vector logits_list_t; + +class Sampler{ +public: + std::vector logits; + int in_features; + std::vector counters; + logits_list_t top_k_logits; + float rep_penalty; + float freq_penalty; + float temperature; + int top_k; + float top_p; + int total_tokens; + std::vector token_positions; + + // Ring buffer for frequency tracking + std::deque token_history; + size_t freq_penalty_window; + size_t rep_penalty_window; + + /// \brief Constructor + /// \param in_features the input features + /// \param config the configuration + Sampler(){}; + Sampler(int in_features, sampler_config& config); + + /// \brief Reset the penalties + /// \note The function will reset the penalties + /// \note The function will reset the token positions + /// \note The function will reset the token history + /// \note The function will reset the total tokens + /// \note The function will reset the token positions + void reset_penalties(); + + /// \brief Sample the token + /// \param x the input buffer + /// \return the sampled token + int sample(buffer& x); +}; \ No newline at end of file diff --git a/src/detail/include/nlohmann/adl_serializer.hpp b/src/detail/include/nlohmann/adl_serializer.hpp new file mode 100644 index 000000000..c1ddd52b5 --- /dev/null +++ b/src/detail/include/nlohmann/adl_serializer.hpp @@ -0,0 +1,55 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include + +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN + +/// @sa https://json.nlohmann.me/api/adl_serializer/ +template +struct adl_serializer +{ + /// @brief convert a JSON value to any value type + /// @sa https://json.nlohmann.me/api/adl_serializer/from_json/ + template + static auto from_json(BasicJsonType && j, TargetType& val) noexcept( + noexcept(::nlohmann::from_json(std::forward(j), val))) + -> decltype(::nlohmann::from_json(std::forward(j), val), void()) + { + ::nlohmann::from_json(std::forward(j), val); + } + + /// @brief convert a JSON value to any value type + /// @sa https://json.nlohmann.me/api/adl_serializer/from_json/ + template + static auto from_json(BasicJsonType && j) noexcept( + noexcept(::nlohmann::from_json(std::forward(j), detail::identity_tag {}))) + -> decltype(::nlohmann::from_json(std::forward(j), detail::identity_tag {})) + { + return ::nlohmann::from_json(std::forward(j), detail::identity_tag {}); + } + + /// @brief convert any value type to a JSON value + /// @sa https://json.nlohmann.me/api/adl_serializer/to_json/ + template + static auto to_json(BasicJsonType& j, TargetType && val) noexcept( + noexcept(::nlohmann::to_json(j, std::forward(val)))) + -> decltype(::nlohmann::to_json(j, std::forward(val)), void()) + { + ::nlohmann::to_json(j, std::forward(val)); + } +}; + +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/byte_container_with_subtype.hpp b/src/detail/include/nlohmann/byte_container_with_subtype.hpp new file mode 100644 index 000000000..19b84c319 --- /dev/null +++ b/src/detail/include/nlohmann/byte_container_with_subtype.hpp @@ -0,0 +1,103 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // uint8_t, uint64_t +#include // tie +#include // move + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN + +/// @brief an internal type for a backed binary type +/// @sa https://json.nlohmann.me/api/byte_container_with_subtype/ +template +class byte_container_with_subtype : public BinaryType +{ + public: + using container_type = BinaryType; + using subtype_type = std::uint64_t; + + /// @sa https://json.nlohmann.me/api/byte_container_with_subtype/byte_container_with_subtype/ + byte_container_with_subtype() noexcept(noexcept(container_type())) + : container_type() + {} + + /// @sa https://json.nlohmann.me/api/byte_container_with_subtype/byte_container_with_subtype/ + byte_container_with_subtype(const container_type& b) noexcept(noexcept(container_type(b))) + : container_type(b) + {} + + /// @sa https://json.nlohmann.me/api/byte_container_with_subtype/byte_container_with_subtype/ + byte_container_with_subtype(container_type&& b) noexcept(noexcept(container_type(std::move(b)))) + : container_type(std::move(b)) + {} + + /// @sa https://json.nlohmann.me/api/byte_container_with_subtype/byte_container_with_subtype/ + byte_container_with_subtype(const container_type& b, subtype_type subtype_) noexcept(noexcept(container_type(b))) + : container_type(b) + , m_subtype(subtype_) + , m_has_subtype(true) + {} + + /// @sa https://json.nlohmann.me/api/byte_container_with_subtype/byte_container_with_subtype/ + byte_container_with_subtype(container_type&& b, subtype_type subtype_) noexcept(noexcept(container_type(std::move(b)))) + : container_type(std::move(b)) + , m_subtype(subtype_) + , m_has_subtype(true) + {} + + bool operator==(const byte_container_with_subtype& rhs) const + { + return std::tie(static_cast(*this), m_subtype, m_has_subtype) == + std::tie(static_cast(rhs), rhs.m_subtype, rhs.m_has_subtype); + } + + bool operator!=(const byte_container_with_subtype& rhs) const + { + return !(rhs == *this); + } + + /// @brief sets the binary subtype + /// @sa https://json.nlohmann.me/api/byte_container_with_subtype/set_subtype/ + void set_subtype(subtype_type subtype_) noexcept + { + m_subtype = subtype_; + m_has_subtype = true; + } + + /// @brief return the binary subtype + /// @sa https://json.nlohmann.me/api/byte_container_with_subtype/subtype/ + constexpr subtype_type subtype() const noexcept + { + return m_has_subtype ? m_subtype : static_cast(-1); + } + + /// @brief return whether the value has a subtype + /// @sa https://json.nlohmann.me/api/byte_container_with_subtype/has_subtype/ + constexpr bool has_subtype() const noexcept + { + return m_has_subtype; + } + + /// @brief clears the binary subtype + /// @sa https://json.nlohmann.me/api/byte_container_with_subtype/clear_subtype/ + void clear_subtype() noexcept + { + m_subtype = 0; + m_has_subtype = false; + } + + private: + subtype_type m_subtype = 0; + bool m_has_subtype = false; +}; + +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/abi_macros.hpp b/src/detail/include/nlohmann/detail/abi_macros.hpp new file mode 100644 index 000000000..44a1ba569 --- /dev/null +++ b/src/detail/include/nlohmann/detail/abi_macros.hpp @@ -0,0 +1,111 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +// This file contains all macro definitions affecting or depending on the ABI + +#ifndef JSON_SKIP_LIBRARY_VERSION_CHECK + #if defined(NLOHMANN_JSON_VERSION_MAJOR) && defined(NLOHMANN_JSON_VERSION_MINOR) && defined(NLOHMANN_JSON_VERSION_PATCH) + #if NLOHMANN_JSON_VERSION_MAJOR != 3 || NLOHMANN_JSON_VERSION_MINOR != 12 || NLOHMANN_JSON_VERSION_PATCH != 0 + #warning "Already included a different version of the library!" + #endif + #endif +#endif + +#define NLOHMANN_JSON_VERSION_MAJOR 3 // NOLINT(modernize-macro-to-enum) +#define NLOHMANN_JSON_VERSION_MINOR 12 // NOLINT(modernize-macro-to-enum) +#define NLOHMANN_JSON_VERSION_PATCH 0 // NOLINT(modernize-macro-to-enum) + +#ifndef JSON_DIAGNOSTICS + #define JSON_DIAGNOSTICS 0 +#endif + +#ifndef JSON_DIAGNOSTIC_POSITIONS + #define JSON_DIAGNOSTIC_POSITIONS 0 +#endif + +#ifndef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + #define JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON 0 +#endif + +#if JSON_DIAGNOSTICS + #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS _diag +#else + #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS +#endif + +#if JSON_DIAGNOSTIC_POSITIONS + #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS _dp +#else + #define NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS +#endif + +#if JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON _ldvcmp +#else + #define NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON +#endif + +#ifndef NLOHMANN_JSON_NAMESPACE_NO_VERSION + #define NLOHMANN_JSON_NAMESPACE_NO_VERSION 0 +#endif + +// Construct the namespace ABI tags component +#define NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) json_abi ## a ## b ## c +#define NLOHMANN_JSON_ABI_TAGS_CONCAT(a, b, c) \ + NLOHMANN_JSON_ABI_TAGS_CONCAT_EX(a, b, c) + +#define NLOHMANN_JSON_ABI_TAGS \ + NLOHMANN_JSON_ABI_TAGS_CONCAT( \ + NLOHMANN_JSON_ABI_TAG_DIAGNOSTICS, \ + NLOHMANN_JSON_ABI_TAG_LEGACY_DISCARDED_VALUE_COMPARISON, \ + NLOHMANN_JSON_ABI_TAG_DIAGNOSTIC_POSITIONS) + +// Construct the namespace version component +#define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) \ + _v ## major ## _ ## minor ## _ ## patch +#define NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT(major, minor, patch) \ + NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT_EX(major, minor, patch) + +#if NLOHMANN_JSON_NAMESPACE_NO_VERSION +#define NLOHMANN_JSON_NAMESPACE_VERSION +#else +#define NLOHMANN_JSON_NAMESPACE_VERSION \ + NLOHMANN_JSON_NAMESPACE_VERSION_CONCAT(NLOHMANN_JSON_VERSION_MAJOR, \ + NLOHMANN_JSON_VERSION_MINOR, \ + NLOHMANN_JSON_VERSION_PATCH) +#endif + +// Combine namespace components +#define NLOHMANN_JSON_NAMESPACE_CONCAT_EX(a, b) a ## b +#define NLOHMANN_JSON_NAMESPACE_CONCAT(a, b) \ + NLOHMANN_JSON_NAMESPACE_CONCAT_EX(a, b) + +#ifndef NLOHMANN_JSON_NAMESPACE +#define NLOHMANN_JSON_NAMESPACE \ + nlohmann::NLOHMANN_JSON_NAMESPACE_CONCAT( \ + NLOHMANN_JSON_ABI_TAGS, \ + NLOHMANN_JSON_NAMESPACE_VERSION) +#endif + +#ifndef NLOHMANN_JSON_NAMESPACE_BEGIN +#define NLOHMANN_JSON_NAMESPACE_BEGIN \ + namespace nlohmann \ + { \ + inline namespace NLOHMANN_JSON_NAMESPACE_CONCAT( \ + NLOHMANN_JSON_ABI_TAGS, \ + NLOHMANN_JSON_NAMESPACE_VERSION) \ + { +#endif + +#ifndef NLOHMANN_JSON_NAMESPACE_END +#define NLOHMANN_JSON_NAMESPACE_END \ + } /* namespace (inline namespace) NOLINT(readability/namespace) */ \ + } // namespace nlohmann +#endif diff --git a/src/detail/include/nlohmann/detail/conversions/from_json.hpp b/src/detail/include/nlohmann/detail/conversions/from_json.hpp new file mode 100644 index 000000000..824235d1d --- /dev/null +++ b/src/detail/include/nlohmann/detail/conversions/from_json.hpp @@ -0,0 +1,583 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // transform +#include // array +#include // forward_list +#include // inserter, front_inserter, end +#include // map +#include // string +#include // tuple, make_tuple +#include // is_arithmetic, is_same, is_enum, underlying_type, is_convertible +#include // unordered_map +#include // pair, declval +#include // valarray + +#include +#include +#include +#include +#include +#include +#include +#include + +// include after macro_scope.hpp +#ifdef JSON_HAS_CPP_17 + #include // optional +#endif + +#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM + #include // u8string_view +#endif + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +template +inline void from_json(const BasicJsonType& j, typename std::nullptr_t& n) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_null())) + { + JSON_THROW(type_error::create(302, concat("type must be null, but is ", j.type_name()), &j)); + } + n = nullptr; +} + +#ifdef JSON_HAS_CPP_17 +template +void from_json(const BasicJsonType& j, std::optional& opt) +{ + if (j.is_null()) + { + opt = std::nullopt; + } + else + { + opt.emplace(j.template get()); + } +} +#endif // JSON_HAS_CPP_17 + +// overloads for basic_json template parameters +template < typename BasicJsonType, typename ArithmeticType, + enable_if_t < std::is_arithmetic::value&& + !std::is_same::value, + int > = 0 > +void get_arithmetic_value(const BasicJsonType& j, ArithmeticType& val) +{ + switch (static_cast(j)) + { + case value_t::number_unsigned: + { + val = static_cast(*j.template get_ptr()); + break; + } + case value_t::number_integer: + { + val = static_cast(*j.template get_ptr()); + break; + } + case value_t::number_float: + { + val = static_cast(*j.template get_ptr()); + break; + } + + case value_t::null: + case value_t::object: + case value_t::array: + case value_t::string: + case value_t::boolean: + case value_t::binary: + case value_t::discarded: + default: + JSON_THROW(type_error::create(302, concat("type must be number, but is ", j.type_name()), &j)); + } +} + +template +inline void from_json(const BasicJsonType& j, typename BasicJsonType::boolean_t& b) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_boolean())) + { + JSON_THROW(type_error::create(302, concat("type must be boolean, but is ", j.type_name()), &j)); + } + b = *j.template get_ptr(); +} + +template +inline void from_json(const BasicJsonType& j, typename BasicJsonType::string_t& s) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_string())) + { + JSON_THROW(type_error::create(302, concat("type must be string, but is ", j.type_name()), &j)); + } + s = *j.template get_ptr(); +} + +template < + typename BasicJsonType, typename StringType, + enable_if_t < + std::is_assignable::value + && is_detected_exact::value + && !std::is_same::value + && !is_json_ref::value, int > = 0 > +inline void from_json(const BasicJsonType& j, StringType& s) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_string())) + { + JSON_THROW(type_error::create(302, concat("type must be string, but is ", j.type_name()), &j)); + } + + s = *j.template get_ptr(); +} + +template +inline void from_json(const BasicJsonType& j, typename BasicJsonType::number_float_t& val) +{ + get_arithmetic_value(j, val); +} + +template +inline void from_json(const BasicJsonType& j, typename BasicJsonType::number_unsigned_t& val) +{ + get_arithmetic_value(j, val); +} + +template +inline void from_json(const BasicJsonType& j, typename BasicJsonType::number_integer_t& val) +{ + get_arithmetic_value(j, val); +} + +#if !JSON_DISABLE_ENUM_SERIALIZATION +template::value, int> = 0> +inline void from_json(const BasicJsonType& j, EnumType& e) +{ + typename std::underlying_type::type val; + get_arithmetic_value(j, val); + e = static_cast(val); +} +#endif // JSON_DISABLE_ENUM_SERIALIZATION + +// forward_list doesn't have an insert method +template::value, int> = 0> +inline void from_json(const BasicJsonType& j, std::forward_list& l) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); + } + l.clear(); + std::transform(j.rbegin(), j.rend(), + std::front_inserter(l), [](const BasicJsonType & i) + { + return i.template get(); + }); +} + +// valarray doesn't have an insert method +template::value, int> = 0> +inline void from_json(const BasicJsonType& j, std::valarray& l) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); + } + l.resize(j.size()); + std::transform(j.begin(), j.end(), std::begin(l), + [](const BasicJsonType & elem) + { + return elem.template get(); + }); +} + +template +auto from_json(const BasicJsonType& j, T (&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) +-> decltype(j.template get(), void()) +{ + for (std::size_t i = 0; i < N; ++i) + { + arr[i] = j.at(i).template get(); + } +} + +template +auto from_json(const BasicJsonType& j, T (&arr)[N1][N2]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) +-> decltype(j.template get(), void()) +{ + for (std::size_t i1 = 0; i1 < N1; ++i1) + { + for (std::size_t i2 = 0; i2 < N2; ++i2) + { + arr[i1][i2] = j.at(i1).at(i2).template get(); + } + } +} + +template +auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) +-> decltype(j.template get(), void()) +{ + for (std::size_t i1 = 0; i1 < N1; ++i1) + { + for (std::size_t i2 = 0; i2 < N2; ++i2) + { + for (std::size_t i3 = 0; i3 < N3; ++i3) + { + arr[i1][i2][i3] = j.at(i1).at(i2).at(i3).template get(); + } + } + } +} + +template +auto from_json(const BasicJsonType& j, T (&arr)[N1][N2][N3][N4]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) +-> decltype(j.template get(), void()) +{ + for (std::size_t i1 = 0; i1 < N1; ++i1) + { + for (std::size_t i2 = 0; i2 < N2; ++i2) + { + for (std::size_t i3 = 0; i3 < N3; ++i3) + { + for (std::size_t i4 = 0; i4 < N4; ++i4) + { + arr[i1][i2][i3][i4] = j.at(i1).at(i2).at(i3).at(i4).template get(); + } + } + } + } +} + +template +inline void from_json_array_impl(const BasicJsonType& j, typename BasicJsonType::array_t& arr, priority_tag<3> /*unused*/) +{ + arr = *j.template get_ptr(); +} + +template +auto from_json_array_impl(const BasicJsonType& j, std::array& arr, + priority_tag<2> /*unused*/) +-> decltype(j.template get(), void()) +{ + for (std::size_t i = 0; i < N; ++i) + { + arr[i] = j.at(i).template get(); + } +} + +template::value, + int> = 0> +auto from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, priority_tag<1> /*unused*/) +-> decltype( + arr.reserve(std::declval()), + j.template get(), + void()) +{ + using std::end; + + ConstructibleArrayType ret; + ret.reserve(j.size()); + std::transform(j.begin(), j.end(), + std::inserter(ret, end(ret)), [](const BasicJsonType & i) + { + // get() returns *this, this won't call a from_json + // method when value_type is BasicJsonType + return i.template get(); + }); + arr = std::move(ret); +} + +template::value, + int> = 0> +inline void from_json_array_impl(const BasicJsonType& j, ConstructibleArrayType& arr, + priority_tag<0> /*unused*/) +{ + using std::end; + + ConstructibleArrayType ret; + std::transform( + j.begin(), j.end(), std::inserter(ret, end(ret)), + [](const BasicJsonType & i) + { + // get() returns *this, this won't call a from_json + // method when value_type is BasicJsonType + return i.template get(); + }); + arr = std::move(ret); +} + +template < typename BasicJsonType, typename ConstructibleArrayType, + enable_if_t < + is_constructible_array_type::value&& + !is_constructible_object_type::value&& + !is_constructible_string_type::value&& + !std::is_same::value&& + !is_basic_json::value, + int > = 0 > +auto from_json(const BasicJsonType& j, ConstructibleArrayType& arr) +-> decltype(from_json_array_impl(j, arr, priority_tag<3> {}), +j.template get(), +void()) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); + } + + from_json_array_impl(j, arr, priority_tag<3> {}); +} + +template < typename BasicJsonType, typename T, std::size_t... Idx > +std::array from_json_inplace_array_impl(BasicJsonType&& j, + identity_tag> /*unused*/, index_sequence /*unused*/) +{ + return { { std::forward(j).at(Idx).template get()... } }; +} + +template < typename BasicJsonType, typename T, std::size_t N > +auto from_json(BasicJsonType&& j, identity_tag> tag) +-> decltype(from_json_inplace_array_impl(std::forward(j), tag, make_index_sequence {})) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); + } + + return from_json_inplace_array_impl(std::forward(j), tag, make_index_sequence {}); +} + +template +inline void from_json(const BasicJsonType& j, typename BasicJsonType::binary_t& bin) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_binary())) + { + JSON_THROW(type_error::create(302, concat("type must be binary, but is ", j.type_name()), &j)); + } + + bin = *j.template get_ptr(); +} + +template::value, int> = 0> +inline void from_json(const BasicJsonType& j, ConstructibleObjectType& obj) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_object())) + { + JSON_THROW(type_error::create(302, concat("type must be object, but is ", j.type_name()), &j)); + } + + ConstructibleObjectType ret; + const auto* inner_object = j.template get_ptr(); + using value_type = typename ConstructibleObjectType::value_type; + std::transform( + inner_object->begin(), inner_object->end(), + std::inserter(ret, ret.begin()), + [](typename BasicJsonType::object_t::value_type const & p) + { + return value_type(p.first, p.second.template get()); + }); + obj = std::move(ret); +} + +// overload for arithmetic types, not chosen for basic_json template arguments +// (BooleanType, etc.); note: Is it really necessary to provide explicit +// overloads for boolean_t etc. in case of a custom BooleanType which is not +// an arithmetic type? +template < typename BasicJsonType, typename ArithmeticType, + enable_if_t < + std::is_arithmetic::value&& + !std::is_same::value&& + !std::is_same::value&& + !std::is_same::value&& + !std::is_same::value, + int > = 0 > +inline void from_json(const BasicJsonType& j, ArithmeticType& val) +{ + switch (static_cast(j)) + { + case value_t::number_unsigned: + { + val = static_cast(*j.template get_ptr()); + break; + } + case value_t::number_integer: + { + val = static_cast(*j.template get_ptr()); + break; + } + case value_t::number_float: + { + val = static_cast(*j.template get_ptr()); + break; + } + case value_t::boolean: + { + val = static_cast(*j.template get_ptr()); + break; + } + + case value_t::null: + case value_t::object: + case value_t::array: + case value_t::string: + case value_t::binary: + case value_t::discarded: + default: + JSON_THROW(type_error::create(302, concat("type must be number, but is ", j.type_name()), &j)); + } +} + +template +std::tuple from_json_tuple_impl_base(BasicJsonType&& j, index_sequence /*unused*/) +{ + return std::make_tuple(std::forward(j).at(Idx).template get()...); +} + +template +std::tuple<> from_json_tuple_impl_base(BasicJsonType& /*unused*/, index_sequence<> /*unused*/) +{ + return {}; +} + +template < typename BasicJsonType, class A1, class A2 > +std::pair from_json_tuple_impl(BasicJsonType&& j, identity_tag> /*unused*/, priority_tag<0> /*unused*/) +{ + return {std::forward(j).at(0).template get(), + std::forward(j).at(1).template get()}; +} + +template +inline void from_json_tuple_impl(BasicJsonType&& j, std::pair& p, priority_tag<1> /*unused*/) +{ + p = from_json_tuple_impl(std::forward(j), identity_tag> {}, priority_tag<0> {}); +} + +template +std::tuple from_json_tuple_impl(BasicJsonType&& j, identity_tag> /*unused*/, priority_tag<2> /*unused*/) +{ + return from_json_tuple_impl_base(std::forward(j), index_sequence_for {}); +} + +template +inline void from_json_tuple_impl(BasicJsonType&& j, std::tuple& t, priority_tag<3> /*unused*/) +{ + t = from_json_tuple_impl_base(std::forward(j), index_sequence_for {}); +} + +template +auto from_json(BasicJsonType&& j, TupleRelated&& t) +-> decltype(from_json_tuple_impl(std::forward(j), std::forward(t), priority_tag<3> {})) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); + } + + return from_json_tuple_impl(std::forward(j), std::forward(t), priority_tag<3> {}); +} + +template < typename BasicJsonType, typename Key, typename Value, typename Compare, typename Allocator, + typename = enable_if_t < !std::is_constructible < + typename BasicJsonType::string_t, Key >::value >> +inline void from_json(const BasicJsonType& j, std::map& m) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); + } + m.clear(); + for (const auto& p : j) + { + if (JSON_HEDLEY_UNLIKELY(!p.is_array())) + { + JSON_THROW(type_error::create(302, concat("type must be array, but is ", p.type_name()), &j)); + } + m.emplace(p.at(0).template get(), p.at(1).template get()); + } +} + +template < typename BasicJsonType, typename Key, typename Value, typename Hash, typename KeyEqual, typename Allocator, + typename = enable_if_t < !std::is_constructible < + typename BasicJsonType::string_t, Key >::value >> +inline void from_json(const BasicJsonType& j, std::unordered_map& m) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_array())) + { + JSON_THROW(type_error::create(302, concat("type must be array, but is ", j.type_name()), &j)); + } + m.clear(); + for (const auto& p : j) + { + if (JSON_HEDLEY_UNLIKELY(!p.is_array())) + { + JSON_THROW(type_error::create(302, concat("type must be array, but is ", p.type_name()), &j)); + } + m.emplace(p.at(0).template get(), p.at(1).template get()); + } +} + +#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM +template +inline void from_json(const BasicJsonType& j, std_fs::path& p) +{ + if (JSON_HEDLEY_UNLIKELY(!j.is_string())) + { + JSON_THROW(type_error::create(302, concat("type must be string, but is ", j.type_name()), &j)); + } + const auto& s = *j.template get_ptr(); + // Checking for C++20 standard or later can be insufficient in case the + // library support for char8_t is either incomplete or was disabled + // altogether. Use the __cpp_lib_char8_t feature test instead. +#if defined(__cpp_lib_char8_t) && (__cpp_lib_char8_t >= 201907L) + p = std_fs::path(std::u8string_view(reinterpret_cast(s.data()), s.size())); +#else + p = std_fs::u8path(s); // accepts UTF-8 encoded std::string in C++17, deprecated in C++20 +#endif +} +#endif + +struct from_json_fn +{ + template + auto operator()(const BasicJsonType& j, T&& val) const + noexcept(noexcept(from_json(j, std::forward(val)))) + -> decltype(from_json(j, std::forward(val))) + { + return from_json(j, std::forward(val)); + } +}; + +} // namespace detail + +#ifndef JSON_HAS_CPP_17 +/// namespace to hold default `from_json` function +/// to see why this is required: +/// http://www.open-std.org/jtc1/sc22/wg21/docs/papers/2015/n4381.html +namespace // NOLINT(cert-dcl59-cpp,fuchsia-header-anon-namespaces,google-build-namespaces) +{ +#endif +JSON_INLINE_VARIABLE constexpr const auto& from_json = // NOLINT(misc-definitions-in-headers) + detail::static_const::value; +#ifndef JSON_HAS_CPP_17 +} // namespace +#endif + +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/conversions/to_chars.hpp b/src/detail/include/nlohmann/detail/conversions/to_chars.hpp new file mode 100644 index 000000000..b01f78649 --- /dev/null +++ b/src/detail/include/nlohmann/detail/conversions/to_chars.hpp @@ -0,0 +1,1118 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2009 Florian Loitsch +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // signbit, isfinite +#include // intN_t, uintN_t +#include // memcpy, memmove +#include // numeric_limits +#include // conditional + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief implements the Grisu2 algorithm for binary to decimal floating-point +conversion. + +This implementation is a slightly modified version of the reference +implementation which may be obtained from +http://florian.loitsch.com/publications (bench.tar.gz). + +The code is distributed under the MIT license, Copyright (c) 2009 Florian Loitsch. + +For a detailed description of the algorithm see: + +[1] Loitsch, "Printing Floating-Point Numbers Quickly and Accurately with + Integers", Proceedings of the ACM SIGPLAN 2010 Conference on Programming + Language Design and Implementation, PLDI 2010 +[2] Burger, Dybvig, "Printing Floating-Point Numbers Quickly and Accurately", + Proceedings of the ACM SIGPLAN 1996 Conference on Programming Language + Design and Implementation, PLDI 1996 +*/ +namespace dtoa_impl +{ + +template +Target reinterpret_bits(const Source source) +{ + static_assert(sizeof(Target) == sizeof(Source), "size mismatch"); + + Target target; + std::memcpy(&target, &source, sizeof(Source)); + return target; +} + +struct diyfp // f * 2^e +{ + static constexpr int kPrecision = 64; // = q + + std::uint64_t f = 0; + int e = 0; + + constexpr diyfp(std::uint64_t f_, int e_) noexcept : f(f_), e(e_) {} + + /*! + @brief returns x - y + @pre x.e == y.e and x.f >= y.f + */ + static diyfp sub(const diyfp& x, const diyfp& y) noexcept + { + JSON_ASSERT(x.e == y.e); + JSON_ASSERT(x.f >= y.f); + + return {x.f - y.f, x.e}; + } + + /*! + @brief returns x * y + @note The result is rounded. (Only the upper q bits are returned.) + */ + static diyfp mul(const diyfp& x, const diyfp& y) noexcept + { + static_assert(kPrecision == 64, "internal error"); + + // Computes: + // f = round((x.f * y.f) / 2^q) + // e = x.e + y.e + q + + // Emulate the 64-bit * 64-bit multiplication: + // + // p = u * v + // = (u_lo + 2^32 u_hi) (v_lo + 2^32 v_hi) + // = (u_lo v_lo ) + 2^32 ((u_lo v_hi ) + (u_hi v_lo )) + 2^64 (u_hi v_hi ) + // = (p0 ) + 2^32 ((p1 ) + (p2 )) + 2^64 (p3 ) + // = (p0_lo + 2^32 p0_hi) + 2^32 ((p1_lo + 2^32 p1_hi) + (p2_lo + 2^32 p2_hi)) + 2^64 (p3 ) + // = (p0_lo ) + 2^32 (p0_hi + p1_lo + p2_lo ) + 2^64 (p1_hi + p2_hi + p3) + // = (p0_lo ) + 2^32 (Q ) + 2^64 (H ) + // = (p0_lo ) + 2^32 (Q_lo + 2^32 Q_hi ) + 2^64 (H ) + // + // (Since Q might be larger than 2^32 - 1) + // + // = (p0_lo + 2^32 Q_lo) + 2^64 (Q_hi + H) + // + // (Q_hi + H does not overflow a 64-bit int) + // + // = p_lo + 2^64 p_hi + + const std::uint64_t u_lo = x.f & 0xFFFFFFFFu; + const std::uint64_t u_hi = x.f >> 32u; + const std::uint64_t v_lo = y.f & 0xFFFFFFFFu; + const std::uint64_t v_hi = y.f >> 32u; + + const std::uint64_t p0 = u_lo * v_lo; + const std::uint64_t p1 = u_lo * v_hi; + const std::uint64_t p2 = u_hi * v_lo; + const std::uint64_t p3 = u_hi * v_hi; + + const std::uint64_t p0_hi = p0 >> 32u; + const std::uint64_t p1_lo = p1 & 0xFFFFFFFFu; + const std::uint64_t p1_hi = p1 >> 32u; + const std::uint64_t p2_lo = p2 & 0xFFFFFFFFu; + const std::uint64_t p2_hi = p2 >> 32u; + + std::uint64_t Q = p0_hi + p1_lo + p2_lo; + + // The full product might now be computed as + // + // p_hi = p3 + p2_hi + p1_hi + (Q >> 32) + // p_lo = p0_lo + (Q << 32) + // + // But in this particular case here, the full p_lo is not required. + // Effectively, we only need to add the highest bit in p_lo to p_hi (and + // Q_hi + 1 does not overflow). + + Q += std::uint64_t{1} << (64u - 32u - 1u); // round, ties up + + const std::uint64_t h = p3 + p2_hi + p1_hi + (Q >> 32u); + + return {h, x.e + y.e + 64}; + } + + /*! + @brief normalize x such that the significand is >= 2^(q-1) + @pre x.f != 0 + */ + static diyfp normalize(diyfp x) noexcept + { + JSON_ASSERT(x.f != 0); + + while ((x.f >> 63u) == 0) + { + x.f <<= 1u; + x.e--; + } + + return x; + } + + /*! + @brief normalize x such that the result has the exponent E + @pre e >= x.e and the upper e - x.e bits of x.f must be zero. + */ + static diyfp normalize_to(const diyfp& x, const int target_exponent) noexcept + { + const int delta = x.e - target_exponent; + + JSON_ASSERT(delta >= 0); + JSON_ASSERT(((x.f << delta) >> delta) == x.f); + + return {x.f << delta, target_exponent}; + } +}; + +struct boundaries +{ + diyfp w; + diyfp minus; + diyfp plus; +}; + +/*! +Compute the (normalized) diyfp representing the input number 'value' and its +boundaries. + +@pre value must be finite and positive +*/ +template +boundaries compute_boundaries(FloatType value) +{ + JSON_ASSERT(std::isfinite(value)); + JSON_ASSERT(value > 0); + + // Convert the IEEE representation into a diyfp. + // + // If v is denormal: + // value = 0.F * 2^(1 - bias) = ( F) * 2^(1 - bias - (p-1)) + // If v is normalized: + // value = 1.F * 2^(E - bias) = (2^(p-1) + F) * 2^(E - bias - (p-1)) + + static_assert(std::numeric_limits::is_iec559, + "internal error: dtoa_short requires an IEEE-754 floating-point implementation"); + + constexpr int kPrecision = std::numeric_limits::digits; // = p (includes the hidden bit) + constexpr int kBias = std::numeric_limits::max_exponent - 1 + (kPrecision - 1); + constexpr int kMinExp = 1 - kBias; + constexpr std::uint64_t kHiddenBit = std::uint64_t{1} << (kPrecision - 1); // = 2^(p-1) + + using bits_type = typename std::conditional::type; + + const auto bits = static_cast(reinterpret_bits(value)); + const std::uint64_t E = bits >> (kPrecision - 1); + const std::uint64_t F = bits & (kHiddenBit - 1); + + const bool is_denormal = E == 0; + const diyfp v = is_denormal + ? diyfp(F, kMinExp) + : diyfp(F + kHiddenBit, static_cast(E) - kBias); + + // Compute the boundaries m- and m+ of the floating-point value + // v = f * 2^e. + // + // Determine v- and v+, the floating-point predecessor and successor of v, + // respectively. + // + // v- = v - 2^e if f != 2^(p-1) or e == e_min (A) + // = v - 2^(e-1) if f == 2^(p-1) and e > e_min (B) + // + // v+ = v + 2^e + // + // Let m- = (v- + v) / 2 and m+ = (v + v+) / 2. All real numbers _strictly_ + // between m- and m+ round to v, regardless of how the input rounding + // algorithm breaks ties. + // + // ---+-------------+-------------+-------------+-------------+--- (A) + // v- m- v m+ v+ + // + // -----------------+------+------+-------------+-------------+--- (B) + // v- m- v m+ v+ + + const bool lower_boundary_is_closer = F == 0 && E > 1; + const diyfp m_plus = diyfp((2 * v.f) + 1, v.e - 1); + const diyfp m_minus = lower_boundary_is_closer + ? diyfp((4 * v.f) - 1, v.e - 2) // (B) + : diyfp((2 * v.f) - 1, v.e - 1); // (A) + + // Determine the normalized w+ = m+. + const diyfp w_plus = diyfp::normalize(m_plus); + + // Determine w- = m- such that e_(w-) = e_(w+). + const diyfp w_minus = diyfp::normalize_to(m_minus, w_plus.e); + + return {diyfp::normalize(v), w_minus, w_plus}; +} + +// Given normalized diyfp w, Grisu needs to find a (normalized) cached +// power-of-ten c, such that the exponent of the product c * w = f * 2^e lies +// within a certain range [alpha, gamma] (Definition 3.2 from [1]) +// +// alpha <= e = e_c + e_w + q <= gamma +// +// or +// +// f_c * f_w * 2^alpha <= f_c 2^(e_c) * f_w 2^(e_w) * 2^q +// <= f_c * f_w * 2^gamma +// +// Since c and w are normalized, i.e. 2^(q-1) <= f < 2^q, this implies +// +// 2^(q-1) * 2^(q-1) * 2^alpha <= c * w * 2^q < 2^q * 2^q * 2^gamma +// +// or +// +// 2^(q - 2 + alpha) <= c * w < 2^(q + gamma) +// +// The choice of (alpha,gamma) determines the size of the table and the form of +// the digit generation procedure. Using (alpha,gamma)=(-60,-32) works out well +// in practice: +// +// The idea is to cut the number c * w = f * 2^e into two parts, which can be +// processed independently: An integral part p1, and a fractional part p2: +// +// f * 2^e = ( (f div 2^-e) * 2^-e + (f mod 2^-e) ) * 2^e +// = (f div 2^-e) + (f mod 2^-e) * 2^e +// = p1 + p2 * 2^e +// +// The conversion of p1 into decimal form requires a series of divisions and +// modulos by (a power of) 10. These operations are faster for 32-bit than for +// 64-bit integers, so p1 should ideally fit into a 32-bit integer. This can be +// achieved by choosing +// +// -e >= 32 or e <= -32 := gamma +// +// In order to convert the fractional part +// +// p2 * 2^e = p2 / 2^-e = d[-1] / 10^1 + d[-2] / 10^2 + ... +// +// into decimal form, the fraction is repeatedly multiplied by 10 and the digits +// d[-i] are extracted in order: +// +// (10 * p2) div 2^-e = d[-1] +// (10 * p2) mod 2^-e = d[-2] / 10^1 + ... +// +// The multiplication by 10 must not overflow. It is sufficient to choose +// +// 10 * p2 < 16 * p2 = 2^4 * p2 <= 2^64. +// +// Since p2 = f mod 2^-e < 2^-e, +// +// -e <= 60 or e >= -60 := alpha + +constexpr int kAlpha = -60; +constexpr int kGamma = -32; + +struct cached_power // c = f * 2^e ~= 10^k +{ + std::uint64_t f; + int e; + int k; +}; + +/*! +For a normalized diyfp w = f * 2^e, this function returns a (normalized) cached +power-of-ten c = f_c * 2^e_c, such that the exponent of the product w * c +satisfies (Definition 3.2 from [1]) + + alpha <= e_c + e + q <= gamma. +*/ +inline cached_power get_cached_power_for_binary_exponent(int e) +{ + // Now + // + // alpha <= e_c + e + q <= gamma (1) + // ==> f_c * 2^alpha <= c * 2^e * 2^q + // + // and since the c's are normalized, 2^(q-1) <= f_c, + // + // ==> 2^(q - 1 + alpha) <= c * 2^(e + q) + // ==> 2^(alpha - e - 1) <= c + // + // If c were an exact power of ten, i.e. c = 10^k, one may determine k as + // + // k = ceil( log_10( 2^(alpha - e - 1) ) ) + // = ceil( (alpha - e - 1) * log_10(2) ) + // + // From the paper: + // "In theory the result of the procedure could be wrong since c is rounded, + // and the computation itself is approximated [...]. In practice, however, + // this simple function is sufficient." + // + // For IEEE double precision floating-point numbers converted into + // normalized diyfp's w = f * 2^e, with q = 64, + // + // e >= -1022 (min IEEE exponent) + // -52 (p - 1) + // -52 (p - 1, possibly normalize denormal IEEE numbers) + // -11 (normalize the diyfp) + // = -1137 + // + // and + // + // e <= +1023 (max IEEE exponent) + // -52 (p - 1) + // -11 (normalize the diyfp) + // = 960 + // + // This binary exponent range [-1137,960] results in a decimal exponent + // range [-307,324]. One does not need to store a cached power for each + // k in this range. For each such k it suffices to find a cached power + // such that the exponent of the product lies in [alpha,gamma]. + // This implies that the difference of the decimal exponents of adjacent + // table entries must be less than or equal to + // + // floor( (gamma - alpha) * log_10(2) ) = 8. + // + // (A smaller distance gamma-alpha would require a larger table.) + + // NB: + // Actually, this function returns c, such that -60 <= e_c + e + 64 <= -34. + + constexpr int kCachedPowersMinDecExp = -300; + constexpr int kCachedPowersDecStep = 8; + + static constexpr std::array kCachedPowers = + { + { + { 0xAB70FE17C79AC6CA, -1060, -300 }, + { 0xFF77B1FCBEBCDC4F, -1034, -292 }, + { 0xBE5691EF416BD60C, -1007, -284 }, + { 0x8DD01FAD907FFC3C, -980, -276 }, + { 0xD3515C2831559A83, -954, -268 }, + { 0x9D71AC8FADA6C9B5, -927, -260 }, + { 0xEA9C227723EE8BCB, -901, -252 }, + { 0xAECC49914078536D, -874, -244 }, + { 0x823C12795DB6CE57, -847, -236 }, + { 0xC21094364DFB5637, -821, -228 }, + { 0x9096EA6F3848984F, -794, -220 }, + { 0xD77485CB25823AC7, -768, -212 }, + { 0xA086CFCD97BF97F4, -741, -204 }, + { 0xEF340A98172AACE5, -715, -196 }, + { 0xB23867FB2A35B28E, -688, -188 }, + { 0x84C8D4DFD2C63F3B, -661, -180 }, + { 0xC5DD44271AD3CDBA, -635, -172 }, + { 0x936B9FCEBB25C996, -608, -164 }, + { 0xDBAC6C247D62A584, -582, -156 }, + { 0xA3AB66580D5FDAF6, -555, -148 }, + { 0xF3E2F893DEC3F126, -529, -140 }, + { 0xB5B5ADA8AAFF80B8, -502, -132 }, + { 0x87625F056C7C4A8B, -475, -124 }, + { 0xC9BCFF6034C13053, -449, -116 }, + { 0x964E858C91BA2655, -422, -108 }, + { 0xDFF9772470297EBD, -396, -100 }, + { 0xA6DFBD9FB8E5B88F, -369, -92 }, + { 0xF8A95FCF88747D94, -343, -84 }, + { 0xB94470938FA89BCF, -316, -76 }, + { 0x8A08F0F8BF0F156B, -289, -68 }, + { 0xCDB02555653131B6, -263, -60 }, + { 0x993FE2C6D07B7FAC, -236, -52 }, + { 0xE45C10C42A2B3B06, -210, -44 }, + { 0xAA242499697392D3, -183, -36 }, + { 0xFD87B5F28300CA0E, -157, -28 }, + { 0xBCE5086492111AEB, -130, -20 }, + { 0x8CBCCC096F5088CC, -103, -12 }, + { 0xD1B71758E219652C, -77, -4 }, + { 0x9C40000000000000, -50, 4 }, + { 0xE8D4A51000000000, -24, 12 }, + { 0xAD78EBC5AC620000, 3, 20 }, + { 0x813F3978F8940984, 30, 28 }, + { 0xC097CE7BC90715B3, 56, 36 }, + { 0x8F7E32CE7BEA5C70, 83, 44 }, + { 0xD5D238A4ABE98068, 109, 52 }, + { 0x9F4F2726179A2245, 136, 60 }, + { 0xED63A231D4C4FB27, 162, 68 }, + { 0xB0DE65388CC8ADA8, 189, 76 }, + { 0x83C7088E1AAB65DB, 216, 84 }, + { 0xC45D1DF942711D9A, 242, 92 }, + { 0x924D692CA61BE758, 269, 100 }, + { 0xDA01EE641A708DEA, 295, 108 }, + { 0xA26DA3999AEF774A, 322, 116 }, + { 0xF209787BB47D6B85, 348, 124 }, + { 0xB454E4A179DD1877, 375, 132 }, + { 0x865B86925B9BC5C2, 402, 140 }, + { 0xC83553C5C8965D3D, 428, 148 }, + { 0x952AB45CFA97A0B3, 455, 156 }, + { 0xDE469FBD99A05FE3, 481, 164 }, + { 0xA59BC234DB398C25, 508, 172 }, + { 0xF6C69A72A3989F5C, 534, 180 }, + { 0xB7DCBF5354E9BECE, 561, 188 }, + { 0x88FCF317F22241E2, 588, 196 }, + { 0xCC20CE9BD35C78A5, 614, 204 }, + { 0x98165AF37B2153DF, 641, 212 }, + { 0xE2A0B5DC971F303A, 667, 220 }, + { 0xA8D9D1535CE3B396, 694, 228 }, + { 0xFB9B7CD9A4A7443C, 720, 236 }, + { 0xBB764C4CA7A44410, 747, 244 }, + { 0x8BAB8EEFB6409C1A, 774, 252 }, + { 0xD01FEF10A657842C, 800, 260 }, + { 0x9B10A4E5E9913129, 827, 268 }, + { 0xE7109BFBA19C0C9D, 853, 276 }, + { 0xAC2820D9623BF429, 880, 284 }, + { 0x80444B5E7AA7CF85, 907, 292 }, + { 0xBF21E44003ACDD2D, 933, 300 }, + { 0x8E679C2F5E44FF8F, 960, 308 }, + { 0xD433179D9C8CB841, 986, 316 }, + { 0x9E19DB92B4E31BA9, 1013, 324 }, + } + }; + + // This computation gives exactly the same results for k as + // k = ceil((kAlpha - e - 1) * 0.30102999566398114) + // for |e| <= 1500, but doesn't require floating-point operations. + // NB: log_10(2) ~= 78913 / 2^18 + JSON_ASSERT(e >= -1500); + JSON_ASSERT(e <= 1500); + const int f = kAlpha - e - 1; + const int k = ((f * 78913) / (1 << 18)) + static_cast(f > 0); + + const int index = (-kCachedPowersMinDecExp + k + (kCachedPowersDecStep - 1)) / kCachedPowersDecStep; + JSON_ASSERT(index >= 0); + JSON_ASSERT(static_cast(index) < kCachedPowers.size()); + + const cached_power cached = kCachedPowers[static_cast(index)]; + JSON_ASSERT(kAlpha <= cached.e + e + 64); + JSON_ASSERT(kGamma >= cached.e + e + 64); + + return cached; +} + +/*! +For n != 0, returns k, such that pow10 := 10^(k-1) <= n < 10^k. +For n == 0, returns 1 and sets pow10 := 1. +*/ +inline int find_largest_pow10(const std::uint32_t n, std::uint32_t& pow10) +{ + // LCOV_EXCL_START + if (n >= 1000000000) + { + pow10 = 1000000000; + return 10; + } + // LCOV_EXCL_STOP + if (n >= 100000000) + { + pow10 = 100000000; + return 9; + } + if (n >= 10000000) + { + pow10 = 10000000; + return 8; + } + if (n >= 1000000) + { + pow10 = 1000000; + return 7; + } + if (n >= 100000) + { + pow10 = 100000; + return 6; + } + if (n >= 10000) + { + pow10 = 10000; + return 5; + } + if (n >= 1000) + { + pow10 = 1000; + return 4; + } + if (n >= 100) + { + pow10 = 100; + return 3; + } + if (n >= 10) + { + pow10 = 10; + return 2; + } + + pow10 = 1; + return 1; +} + +inline void grisu2_round(char* buf, int len, std::uint64_t dist, std::uint64_t delta, + std::uint64_t rest, std::uint64_t ten_k) +{ + JSON_ASSERT(len >= 1); + JSON_ASSERT(dist <= delta); + JSON_ASSERT(rest <= delta); + JSON_ASSERT(ten_k > 0); + + // <--------------------------- delta ----> + // <---- dist ---------> + // --------------[------------------+-------------------]-------------- + // M- w M+ + // + // ten_k + // <------> + // <---- rest ----> + // --------------[------------------+----+--------------]-------------- + // w V + // = buf * 10^k + // + // ten_k represents a unit-in-the-last-place in the decimal representation + // stored in buf. + // Decrement buf by ten_k while this takes buf closer to w. + + // The tests are written in this order to avoid overflow in unsigned + // integer arithmetic. + + while (rest < dist + && delta - rest >= ten_k + && (rest + ten_k < dist || dist - rest > rest + ten_k - dist)) + { + JSON_ASSERT(buf[len - 1] != '0'); + buf[len - 1]--; + rest += ten_k; + } +} + +/*! +Generates V = buffer * 10^decimal_exponent, such that M- <= V <= M+. +M- and M+ must be normalized and share the same exponent -60 <= e <= -32. +*/ +inline void grisu2_digit_gen(char* buffer, int& length, int& decimal_exponent, + diyfp M_minus, diyfp w, diyfp M_plus) +{ + static_assert(kAlpha >= -60, "internal error"); + static_assert(kGamma <= -32, "internal error"); + + // Generates the digits (and the exponent) of a decimal floating-point + // number V = buffer * 10^decimal_exponent in the range [M-, M+]. The diyfp's + // w, M- and M+ share the same exponent e, which satisfies alpha <= e <= gamma. + // + // <--------------------------- delta ----> + // <---- dist ---------> + // --------------[------------------+-------------------]-------------- + // M- w M+ + // + // Grisu2 generates the digits of M+ from left to right and stops as soon as + // V is in [M-,M+]. + + JSON_ASSERT(M_plus.e >= kAlpha); + JSON_ASSERT(M_plus.e <= kGamma); + + std::uint64_t delta = diyfp::sub(M_plus, M_minus).f; // (significand of (M+ - M-), implicit exponent is e) + std::uint64_t dist = diyfp::sub(M_plus, w ).f; // (significand of (M+ - w ), implicit exponent is e) + + // Split M+ = f * 2^e into two parts p1 and p2 (note: e < 0): + // + // M+ = f * 2^e + // = ((f div 2^-e) * 2^-e + (f mod 2^-e)) * 2^e + // = ((p1 ) * 2^-e + (p2 )) * 2^e + // = p1 + p2 * 2^e + + const diyfp one(std::uint64_t{1} << -M_plus.e, M_plus.e); + + auto p1 = static_cast(M_plus.f >> -one.e); // p1 = f div 2^-e (Since -e >= 32, p1 fits into a 32-bit int.) + std::uint64_t p2 = M_plus.f & (one.f - 1); // p2 = f mod 2^-e + + // 1) + // + // Generate the digits of the integral part p1 = d[n-1]...d[1]d[0] + + JSON_ASSERT(p1 > 0); + + std::uint32_t pow10{}; + const int k = find_largest_pow10(p1, pow10); + + // 10^(k-1) <= p1 < 10^k, pow10 = 10^(k-1) + // + // p1 = (p1 div 10^(k-1)) * 10^(k-1) + (p1 mod 10^(k-1)) + // = (d[k-1] ) * 10^(k-1) + (p1 mod 10^(k-1)) + // + // M+ = p1 + p2 * 2^e + // = d[k-1] * 10^(k-1) + (p1 mod 10^(k-1)) + p2 * 2^e + // = d[k-1] * 10^(k-1) + ((p1 mod 10^(k-1)) * 2^-e + p2) * 2^e + // = d[k-1] * 10^(k-1) + ( rest) * 2^e + // + // Now generate the digits d[n] of p1 from left to right (n = k-1,...,0) + // + // p1 = d[k-1]...d[n] * 10^n + d[n-1]...d[0] + // + // but stop as soon as + // + // rest * 2^e = (d[n-1]...d[0] * 2^-e + p2) * 2^e <= delta * 2^e + + int n = k; + while (n > 0) + { + // Invariants: + // M+ = buffer * 10^n + (p1 + p2 * 2^e) (buffer = 0 for n = k) + // pow10 = 10^(n-1) <= p1 < 10^n + // + const std::uint32_t d = p1 / pow10; // d = p1 div 10^(n-1) + const std::uint32_t r = p1 % pow10; // r = p1 mod 10^(n-1) + // + // M+ = buffer * 10^n + (d * 10^(n-1) + r) + p2 * 2^e + // = (buffer * 10 + d) * 10^(n-1) + (r + p2 * 2^e) + // + JSON_ASSERT(d <= 9); + buffer[length++] = static_cast('0' + d); // buffer := buffer * 10 + d + // + // M+ = buffer * 10^(n-1) + (r + p2 * 2^e) + // + p1 = r; + n--; + // + // M+ = buffer * 10^n + (p1 + p2 * 2^e) + // pow10 = 10^n + // + + // Now check if enough digits have been generated. + // Compute + // + // p1 + p2 * 2^e = (p1 * 2^-e + p2) * 2^e = rest * 2^e + // + // Note: + // Since rest and delta share the same exponent e, it suffices to + // compare the significands. + const std::uint64_t rest = (std::uint64_t{p1} << -one.e) + p2; + if (rest <= delta) + { + // V = buffer * 10^n, with M- <= V <= M+. + + decimal_exponent += n; + + // We may now just stop. But instead, it looks as if the buffer + // could be decremented to bring V closer to w. + // + // pow10 = 10^n is now 1 ulp in the decimal representation V. + // The rounding procedure works with diyfp's with an implicit + // exponent of e. + // + // 10^n = (10^n * 2^-e) * 2^e = ulp * 2^e + // + const std::uint64_t ten_n = std::uint64_t{pow10} << -one.e; + grisu2_round(buffer, length, dist, delta, rest, ten_n); + + return; + } + + pow10 /= 10; + // + // pow10 = 10^(n-1) <= p1 < 10^n + // Invariants restored. + } + + // 2) + // + // The digits of the integral part have been generated: + // + // M+ = d[k-1]...d[1]d[0] + p2 * 2^e + // = buffer + p2 * 2^e + // + // Now generate the digits of the fractional part p2 * 2^e. + // + // Note: + // No decimal point is generated: the exponent is adjusted instead. + // + // p2 actually represents the fraction + // + // p2 * 2^e + // = p2 / 2^-e + // = d[-1] / 10^1 + d[-2] / 10^2 + ... + // + // Now generate the digits d[-m] of p1 from left to right (m = 1,2,...) + // + // p2 * 2^e = d[-1]d[-2]...d[-m] * 10^-m + // + 10^-m * (d[-m-1] / 10^1 + d[-m-2] / 10^2 + ...) + // + // using + // + // 10^m * p2 = ((10^m * p2) div 2^-e) * 2^-e + ((10^m * p2) mod 2^-e) + // = ( d) * 2^-e + ( r) + // + // or + // 10^m * p2 * 2^e = d + r * 2^e + // + // i.e. + // + // M+ = buffer + p2 * 2^e + // = buffer + 10^-m * (d + r * 2^e) + // = (buffer * 10^m + d) * 10^-m + 10^-m * r * 2^e + // + // and stop as soon as 10^-m * r * 2^e <= delta * 2^e + + JSON_ASSERT(p2 > delta); + + int m = 0; + for (;;) + { + // Invariant: + // M+ = buffer * 10^-m + 10^-m * (d[-m-1] / 10 + d[-m-2] / 10^2 + ...) * 2^e + // = buffer * 10^-m + 10^-m * (p2 ) * 2^e + // = buffer * 10^-m + 10^-m * (1/10 * (10 * p2) ) * 2^e + // = buffer * 10^-m + 10^-m * (1/10 * ((10*p2 div 2^-e) * 2^-e + (10*p2 mod 2^-e)) * 2^e + // + JSON_ASSERT(p2 <= (std::numeric_limits::max)() / 10); + p2 *= 10; + const std::uint64_t d = p2 >> -one.e; // d = (10 * p2) div 2^-e + const std::uint64_t r = p2 & (one.f - 1); // r = (10 * p2) mod 2^-e + // + // M+ = buffer * 10^-m + 10^-m * (1/10 * (d * 2^-e + r) * 2^e + // = buffer * 10^-m + 10^-m * (1/10 * (d + r * 2^e)) + // = (buffer * 10 + d) * 10^(-m-1) + 10^(-m-1) * r * 2^e + // + JSON_ASSERT(d <= 9); + buffer[length++] = static_cast('0' + d); // buffer := buffer * 10 + d + // + // M+ = buffer * 10^(-m-1) + 10^(-m-1) * r * 2^e + // + p2 = r; + m++; + // + // M+ = buffer * 10^-m + 10^-m * p2 * 2^e + // Invariant restored. + + // Check if enough digits have been generated. + // + // 10^-m * p2 * 2^e <= delta * 2^e + // p2 * 2^e <= 10^m * delta * 2^e + // p2 <= 10^m * delta + delta *= 10; + dist *= 10; + if (p2 <= delta) + { + break; + } + } + + // V = buffer * 10^-m, with M- <= V <= M+. + + decimal_exponent -= m; + + // 1 ulp in the decimal representation is now 10^-m. + // Since delta and dist are now scaled by 10^m, we need to do the + // same with ulp in order to keep the units in sync. + // + // 10^m * 10^-m = 1 = 2^-e * 2^e = ten_m * 2^e + // + const std::uint64_t ten_m = one.f; + grisu2_round(buffer, length, dist, delta, p2, ten_m); + + // By construction this algorithm generates the shortest possible decimal + // number (Loitsch, Theorem 6.2) which rounds back to w. + // For an input number of precision p, at least + // + // N = 1 + ceil(p * log_10(2)) + // + // decimal digits are sufficient to identify all binary floating-point + // numbers (Matula, "In-and-Out conversions"). + // This implies that the algorithm does not produce more than N decimal + // digits. + // + // N = 17 for p = 53 (IEEE double precision) + // N = 9 for p = 24 (IEEE single precision) +} + +/*! +v = buf * 10^decimal_exponent +len is the length of the buffer (number of decimal digits) +The buffer must be large enough, i.e. >= max_digits10. +*/ +JSON_HEDLEY_NON_NULL(1) +inline void grisu2(char* buf, int& len, int& decimal_exponent, + diyfp m_minus, diyfp v, diyfp m_plus) +{ + JSON_ASSERT(m_plus.e == m_minus.e); + JSON_ASSERT(m_plus.e == v.e); + + // --------(-----------------------+-----------------------)-------- (A) + // m- v m+ + // + // --------------------(-----------+-----------------------)-------- (B) + // m- v m+ + // + // First scale v (and m- and m+) such that the exponent is in the range + // [alpha, gamma]. + + const cached_power cached = get_cached_power_for_binary_exponent(m_plus.e); + + const diyfp c_minus_k(cached.f, cached.e); // = c ~= 10^-k + + // The exponent of the products is = v.e + c_minus_k.e + q and is in the range [alpha,gamma] + const diyfp w = diyfp::mul(v, c_minus_k); + const diyfp w_minus = diyfp::mul(m_minus, c_minus_k); + const diyfp w_plus = diyfp::mul(m_plus, c_minus_k); + + // ----(---+---)---------------(---+---)---------------(---+---)---- + // w- w w+ + // = c*m- = c*v = c*m+ + // + // diyfp::mul rounds its result and c_minus_k is approximated too. w, w- and + // w+ are now off by a small amount. + // In fact: + // + // w - v * 10^k < 1 ulp + // + // To account for this inaccuracy, add resp. subtract 1 ulp. + // + // --------+---[---------------(---+---)---------------]---+-------- + // w- M- w M+ w+ + // + // Now any number in [M-, M+] (bounds included) will round to w when input, + // regardless of how the input rounding algorithm breaks ties. + // + // And digit_gen generates the shortest possible such number in [M-, M+]. + // Note that this does not mean that Grisu2 always generates the shortest + // possible number in the interval (m-, m+). + const diyfp M_minus(w_minus.f + 1, w_minus.e); + const diyfp M_plus (w_plus.f - 1, w_plus.e ); + + decimal_exponent = -cached.k; // = -(-k) = k + + grisu2_digit_gen(buf, len, decimal_exponent, M_minus, w, M_plus); +} + +/*! +v = buf * 10^decimal_exponent +len is the length of the buffer (number of decimal digits) +The buffer must be large enough, i.e. >= max_digits10. +*/ +template +JSON_HEDLEY_NON_NULL(1) +void grisu2(char* buf, int& len, int& decimal_exponent, FloatType value) +{ + static_assert(diyfp::kPrecision >= std::numeric_limits::digits + 3, + "internal error: not enough precision"); + + JSON_ASSERT(std::isfinite(value)); + JSON_ASSERT(value > 0); + + // If the neighbors (and boundaries) of 'value' are always computed for double-precision + // numbers, all float's can be recovered using strtod (and strtof). However, the resulting + // decimal representations are not exactly "short". + // + // The documentation for 'std::to_chars' (https://en.cppreference.com/w/cpp/utility/to_chars) + // says "value is converted to a string as if by std::sprintf in the default ("C") locale" + // and since sprintf promotes floats to doubles, I think this is exactly what 'std::to_chars' + // does. + // On the other hand, the documentation for 'std::to_chars' requires that "parsing the + // representation using the corresponding std::from_chars function recovers value exactly". That + // indicates that single precision floating-point numbers should be recovered using + // 'std::strtof'. + // + // NB: If the neighbors are computed for single-precision numbers, there is a single float + // (7.0385307e-26f) which can't be recovered using strtod. The resulting double precision + // value is off by 1 ulp. +#if 0 // NOLINT(readability-avoid-unconditional-preprocessor-if) + const boundaries w = compute_boundaries(static_cast(value)); +#else + const boundaries w = compute_boundaries(value); +#endif + + grisu2(buf, len, decimal_exponent, w.minus, w.w, w.plus); +} + +/*! +@brief appends a decimal representation of e to buf +@return a pointer to the element following the exponent. +@pre -1000 < e < 1000 +*/ +JSON_HEDLEY_NON_NULL(1) +JSON_HEDLEY_RETURNS_NON_NULL +inline char* append_exponent(char* buf, int e) +{ + JSON_ASSERT(e > -1000); + JSON_ASSERT(e < 1000); + + if (e < 0) + { + e = -e; + *buf++ = '-'; + } + else + { + *buf++ = '+'; + } + + auto k = static_cast(e); + if (k < 10) + { + // Always print at least two digits in the exponent. + // This is for compatibility with printf("%g"). + *buf++ = '0'; + *buf++ = static_cast('0' + k); + } + else if (k < 100) + { + *buf++ = static_cast('0' + (k / 10)); + k %= 10; + *buf++ = static_cast('0' + k); + } + else + { + *buf++ = static_cast('0' + (k / 100)); + k %= 100; + *buf++ = static_cast('0' + (k / 10)); + k %= 10; + *buf++ = static_cast('0' + k); + } + + return buf; +} + +/*! +@brief prettify v = buf * 10^decimal_exponent + +If v is in the range [10^min_exp, 10^max_exp) it will be printed in fixed-point +notation. Otherwise it will be printed in exponential notation. + +@pre min_exp < 0 +@pre max_exp > 0 +*/ +JSON_HEDLEY_NON_NULL(1) +JSON_HEDLEY_RETURNS_NON_NULL +inline char* format_buffer(char* buf, int len, int decimal_exponent, + int min_exp, int max_exp) +{ + JSON_ASSERT(min_exp < 0); + JSON_ASSERT(max_exp > 0); + + const int k = len; + const int n = len + decimal_exponent; + + // v = buf * 10^(n-k) + // k is the length of the buffer (number of decimal digits) + // n is the position of the decimal point relative to the start of the buffer. + + if (k <= n && n <= max_exp) + { + // digits[000] + // len <= max_exp + 2 + + std::memset(buf + k, '0', static_cast(n) - static_cast(k)); + // Make it look like a floating-point number (#362, #378) + buf[n + 0] = '.'; + buf[n + 1] = '0'; + return buf + (static_cast(n) + 2); + } + + if (0 < n && n <= max_exp) + { + // dig.its + // len <= max_digits10 + 1 + + JSON_ASSERT(k > n); + + std::memmove(buf + (static_cast(n) + 1), buf + n, static_cast(k) - static_cast(n)); + buf[n] = '.'; + return buf + (static_cast(k) + 1U); + } + + if (min_exp < n && n <= 0) + { + // 0.[000]digits + // len <= 2 + (-min_exp - 1) + max_digits10 + + std::memmove(buf + (2 + static_cast(-n)), buf, static_cast(k)); + buf[0] = '0'; + buf[1] = '.'; + std::memset(buf + 2, '0', static_cast(-n)); + return buf + (2U + static_cast(-n) + static_cast(k)); + } + + if (k == 1) + { + // dE+123 + // len <= 1 + 5 + + buf += 1; + } + else + { + // d.igitsE+123 + // len <= max_digits10 + 1 + 5 + + std::memmove(buf + 2, buf + 1, static_cast(k) - 1); + buf[1] = '.'; + buf += 1 + static_cast(k); + } + + *buf++ = 'e'; + return append_exponent(buf, n - 1); +} + +} // namespace dtoa_impl + +/*! +@brief generates a decimal representation of the floating-point number value in [first, last). + +The format of the resulting decimal representation is similar to printf's %g +format. Returns an iterator pointing past-the-end of the decimal representation. + +@note The input number must be finite, i.e. NaN's and Inf's are not supported. +@note The buffer must be large enough. +@note The result is NOT null-terminated. +*/ +template +JSON_HEDLEY_NON_NULL(1, 2) +JSON_HEDLEY_RETURNS_NON_NULL +char* to_chars(char* first, const char* last, FloatType value) +{ + static_cast(last); // maybe unused - fix warning + JSON_ASSERT(std::isfinite(value)); + + // Use signbit(value) instead of (value < 0) since signbit works for -0. + if (std::signbit(value)) + { + value = -value; + *first++ = '-'; + } + +#ifdef __GNUC__ +#pragma GCC diagnostic push +#pragma GCC diagnostic ignored "-Wfloat-equal" +#endif + if (value == 0) // +-0 + { + *first++ = '0'; + // Make it look like a floating-point number (#362, #378) + *first++ = '.'; + *first++ = '0'; + return first; + } +#ifdef __GNUC__ +#pragma GCC diagnostic pop +#endif + + JSON_ASSERT(last - first >= std::numeric_limits::max_digits10); + + // Compute v = buffer * 10^decimal_exponent. + // The decimal digits are stored in the buffer, which needs to be interpreted + // as an unsigned decimal integer. + // len is the length of the buffer, i.e., the number of decimal digits. + int len = 0; + int decimal_exponent = 0; + dtoa_impl::grisu2(first, len, decimal_exponent, value); + + JSON_ASSERT(len <= std::numeric_limits::max_digits10); + + // Format the buffer like printf("%.*g", prec, value) + constexpr int kMinExp = -4; + // Use digits10 here to increase compatibility with version 2. + constexpr int kMaxExp = std::numeric_limits::digits10; + + JSON_ASSERT(last - first >= kMaxExp + 2); + JSON_ASSERT(last - first >= 2 + (-kMinExp - 1) + std::numeric_limits::max_digits10); + JSON_ASSERT(last - first >= std::numeric_limits::max_digits10 + 6); + + return dtoa_impl::format_buffer(first, len, decimal_exponent, kMinExp, kMaxExp); +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/conversions/to_json.hpp b/src/detail/include/nlohmann/detail/conversions/to_json.hpp new file mode 100644 index 000000000..05daf2f85 --- /dev/null +++ b/src/detail/include/nlohmann/detail/conversions/to_json.hpp @@ -0,0 +1,486 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // JSON_HAS_CPP_17 +#ifdef JSON_HAS_CPP_17 + #include // optional +#endif + +#include // copy +#include // begin, end +#include // allocator_traits +#include // basic_string, char_traits +#include // tuple, get +#include // is_same, is_constructible, is_floating_point, is_enum, underlying_type +#include // move, forward, declval, pair +#include // valarray +#include // vector + +#include +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +////////////////// +// constructors // +////////////////// + +/* + * Note all external_constructor<>::construct functions need to call + * j.m_data.m_value.destroy(j.m_data.m_type) to avoid a memory leak in case j contains an + * allocated value (e.g., a string). See bug issue + * https://github.com/nlohmann/json/issues/2865 for more information. + */ + +template struct external_constructor; + +template<> +struct external_constructor +{ + template + static void construct(BasicJsonType& j, typename BasicJsonType::boolean_t b) noexcept + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::boolean; + j.m_data.m_value = b; + j.assert_invariant(); + } +}; + +template<> +struct external_constructor +{ + template + static void construct(BasicJsonType& j, const typename BasicJsonType::string_t& s) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::string; + j.m_data.m_value = s; + j.assert_invariant(); + } + + template + static void construct(BasicJsonType& j, typename BasicJsonType::string_t&& s) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::string; + j.m_data.m_value = std::move(s); + j.assert_invariant(); + } + + template < typename BasicJsonType, typename CompatibleStringType, + enable_if_t < !std::is_same::value, + int > = 0 > + static void construct(BasicJsonType& j, const CompatibleStringType& str) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::string; + j.m_data.m_value.string = j.template create(str); + j.assert_invariant(); + } +}; + +template<> +struct external_constructor +{ + template + static void construct(BasicJsonType& j, const typename BasicJsonType::binary_t& b) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::binary; + j.m_data.m_value = typename BasicJsonType::binary_t(b); + j.assert_invariant(); + } + + template + static void construct(BasicJsonType& j, typename BasicJsonType::binary_t&& b) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::binary; + j.m_data.m_value = typename BasicJsonType::binary_t(std::move(b)); + j.assert_invariant(); + } +}; + +template<> +struct external_constructor +{ + template + static void construct(BasicJsonType& j, typename BasicJsonType::number_float_t val) noexcept + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::number_float; + j.m_data.m_value = val; + j.assert_invariant(); + } +}; + +template<> +struct external_constructor +{ + template + static void construct(BasicJsonType& j, typename BasicJsonType::number_unsigned_t val) noexcept + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::number_unsigned; + j.m_data.m_value = val; + j.assert_invariant(); + } +}; + +template<> +struct external_constructor +{ + template + static void construct(BasicJsonType& j, typename BasicJsonType::number_integer_t val) noexcept + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::number_integer; + j.m_data.m_value = val; + j.assert_invariant(); + } +}; + +template<> +struct external_constructor +{ + template + static void construct(BasicJsonType& j, const typename BasicJsonType::array_t& arr) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::array; + j.m_data.m_value = arr; + j.set_parents(); + j.assert_invariant(); + } + + template + static void construct(BasicJsonType& j, typename BasicJsonType::array_t&& arr) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::array; + j.m_data.m_value = std::move(arr); + j.set_parents(); + j.assert_invariant(); + } + + template < typename BasicJsonType, typename CompatibleArrayType, + enable_if_t < !std::is_same::value, + int > = 0 > + static void construct(BasicJsonType& j, const CompatibleArrayType& arr) + { + using std::begin; + using std::end; + + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::array; + j.m_data.m_value.array = j.template create(begin(arr), end(arr)); + j.set_parents(); + j.assert_invariant(); + } + + template + static void construct(BasicJsonType& j, const std::vector& arr) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::array; + j.m_data.m_value = value_t::array; + j.m_data.m_value.array->reserve(arr.size()); + for (const bool x : arr) + { + j.m_data.m_value.array->push_back(x); + j.set_parent(j.m_data.m_value.array->back()); + } + j.assert_invariant(); + } + + template::value, int> = 0> + static void construct(BasicJsonType& j, const std::valarray& arr) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::array; + j.m_data.m_value = value_t::array; + j.m_data.m_value.array->resize(arr.size()); + if (arr.size() > 0) + { + std::copy(std::begin(arr), std::end(arr), j.m_data.m_value.array->begin()); + } + j.set_parents(); + j.assert_invariant(); + } +}; + +template<> +struct external_constructor +{ + template + static void construct(BasicJsonType& j, const typename BasicJsonType::object_t& obj) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::object; + j.m_data.m_value = obj; + j.set_parents(); + j.assert_invariant(); + } + + template + static void construct(BasicJsonType& j, typename BasicJsonType::object_t&& obj) + { + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::object; + j.m_data.m_value = std::move(obj); + j.set_parents(); + j.assert_invariant(); + } + + template < typename BasicJsonType, typename CompatibleObjectType, + enable_if_t < !std::is_same::value, int > = 0 > + static void construct(BasicJsonType& j, const CompatibleObjectType& obj) + { + using std::begin; + using std::end; + + j.m_data.m_value.destroy(j.m_data.m_type); + j.m_data.m_type = value_t::object; + j.m_data.m_value.object = j.template create(begin(obj), end(obj)); + j.set_parents(); + j.assert_invariant(); + } +}; + +///////////// +// to_json // +///////////// + +#ifdef JSON_HAS_CPP_17 +template::value, int> = 0> +void to_json(BasicJsonType& j, const std::optional& opt) noexcept +{ + if (opt.has_value()) + { + j = *opt; + } + else + { + j = nullptr; + } +} +#endif + +template::value, int> = 0> +inline void to_json(BasicJsonType& j, T b) noexcept +{ + external_constructor::construct(j, b); +} + +template < typename BasicJsonType, typename BoolRef, + enable_if_t < + ((std::is_same::reference, BoolRef>::value + && !std::is_same ::reference, typename BasicJsonType::boolean_t&>::value) + || (std::is_same::const_reference, BoolRef>::value + && !std::is_same ::const_reference>, + typename BasicJsonType::boolean_t >::value)) + && std::is_convertible::value, int > = 0 > +inline void to_json(BasicJsonType& j, const BoolRef& b) noexcept +{ + external_constructor::construct(j, static_cast(b)); +} + +template::value, int> = 0> +inline void to_json(BasicJsonType& j, const CompatibleString& s) +{ + external_constructor::construct(j, s); +} + +template +inline void to_json(BasicJsonType& j, typename BasicJsonType::string_t&& s) +{ + external_constructor::construct(j, std::move(s)); +} + +template::value, int> = 0> +inline void to_json(BasicJsonType& j, FloatType val) noexcept +{ + external_constructor::construct(j, static_cast(val)); +} + +template::value, int> = 0> +inline void to_json(BasicJsonType& j, CompatibleNumberUnsignedType val) noexcept +{ + external_constructor::construct(j, static_cast(val)); +} + +template::value, int> = 0> +inline void to_json(BasicJsonType& j, CompatibleNumberIntegerType val) noexcept +{ + external_constructor::construct(j, static_cast(val)); +} + +#if !JSON_DISABLE_ENUM_SERIALIZATION +template::value, int> = 0> +inline void to_json(BasicJsonType& j, EnumType e) noexcept +{ + using underlying_type = typename std::underlying_type::type; + static constexpr value_t integral_value_t = std::is_unsigned::value ? value_t::number_unsigned : value_t::number_integer; + external_constructor::construct(j, static_cast(e)); +} +#endif // JSON_DISABLE_ENUM_SERIALIZATION + +template +inline void to_json(BasicJsonType& j, const std::vector& e) +{ + external_constructor::construct(j, e); +} + +template < typename BasicJsonType, typename CompatibleArrayType, + enable_if_t < is_compatible_array_type::value&& + !is_compatible_object_type::value&& + !is_compatible_string_type::value&& + !std::is_same::value&& + !is_basic_json::value, + int > = 0 > +inline void to_json(BasicJsonType& j, const CompatibleArrayType& arr) +{ + external_constructor::construct(j, arr); +} + +template +inline void to_json(BasicJsonType& j, const typename BasicJsonType::binary_t& bin) +{ + external_constructor::construct(j, bin); +} + +template::value, int> = 0> +inline void to_json(BasicJsonType& j, const std::valarray& arr) +{ + external_constructor::construct(j, std::move(arr)); +} + +template +inline void to_json(BasicJsonType& j, typename BasicJsonType::array_t&& arr) +{ + external_constructor::construct(j, std::move(arr)); +} + +template < typename BasicJsonType, typename CompatibleObjectType, + enable_if_t < is_compatible_object_type::value&& !is_basic_json::value, int > = 0 > +inline void to_json(BasicJsonType& j, const CompatibleObjectType& obj) +{ + external_constructor::construct(j, obj); +} + +template +inline void to_json(BasicJsonType& j, typename BasicJsonType::object_t&& obj) +{ + external_constructor::construct(j, std::move(obj)); +} + +template < + typename BasicJsonType, typename T, std::size_t N, + enable_if_t < !std::is_constructible::value, // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + int > = 0 > +inline void to_json(BasicJsonType& j, const T(&arr)[N]) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) +{ + external_constructor::construct(j, arr); +} + +template < typename BasicJsonType, typename T1, typename T2, enable_if_t < std::is_constructible::value&& std::is_constructible::value, int > = 0 > +inline void to_json(BasicJsonType& j, const std::pair& p) +{ + j = { p.first, p.second }; +} + +// for https://github.com/nlohmann/json/pull/1134 +template>::value, int> = 0> +inline void to_json(BasicJsonType& j, const T& b) +{ + j = { {b.key(), b.value()} }; +} + +template +inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& t, index_sequence /*unused*/) +{ + j = { std::get(t)... }; +} + +template +inline void to_json_tuple_impl(BasicJsonType& j, const Tuple& /*unused*/, index_sequence<> /*unused*/) +{ + using array_t = typename BasicJsonType::array_t; + j = array_t(); +} + +template::value, int > = 0> +inline void to_json(BasicJsonType& j, const T& t) +{ + to_json_tuple_impl(j, t, make_index_sequence::value> {}); +} + +#if JSON_HAS_FILESYSTEM || JSON_HAS_EXPERIMENTAL_FILESYSTEM +#if defined(__cpp_lib_char8_t) +template +inline void to_json(BasicJsonType& j, const std::basic_string& s) +{ + using OtherAllocator = typename std::allocator_traits::template rebind_alloc; + j = std::basic_string, OtherAllocator>(s.begin(), s.end(), s.get_allocator()); +} +#endif + +template +inline void to_json(BasicJsonType& j, const std_fs::path& p) +{ + // Returns either a std::string or a std::u8string depending whether library + // support for char8_t is enabled. + j = p.u8string(); +} +#endif + +struct to_json_fn +{ + template + auto operator()(BasicJsonType& j, T&& val) const noexcept(noexcept(to_json(j, std::forward(val)))) + -> decltype(to_json(j, std::forward(val)), void()) + { + return to_json(j, std::forward(val)); + } +}; +} // namespace detail + +#ifndef JSON_HAS_CPP_17 +/// namespace to hold default `to_json` function +/// to see why this is required: +/// http://www.open-std.org/jtc1/sc22/wg21/docs/papers/2015/n4381.html +namespace // NOLINT(cert-dcl59-cpp,fuchsia-header-anon-namespaces,google-build-namespaces) +{ +#endif +JSON_INLINE_VARIABLE constexpr const auto& to_json = // NOLINT(misc-definitions-in-headers) + detail::static_const::value; +#ifndef JSON_HAS_CPP_17 +} // namespace +#endif + +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/exceptions.hpp b/src/detail/include/nlohmann/detail/exceptions.hpp new file mode 100644 index 000000000..df78d6fc1 --- /dev/null +++ b/src/detail/include/nlohmann/detail/exceptions.hpp @@ -0,0 +1,291 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // nullptr_t +#include // exception +#if JSON_DIAGNOSTICS + #include // accumulate +#endif +#include // runtime_error +#include // to_string +#include // vector + +#include +#include +#include +#include +#include +#include +#include + +// With -Wweak-vtables, Clang will complain about the exception classes as they +// have no out-of-line virtual method definitions and their vtable will be +// emitted in every translation unit. This issue cannot be fixed with a +// header-only library as there is no implementation file to move these +// functions to. As a result, we suppress this warning here to avoid client +// code stumbling over this. See https://github.com/nlohmann/json/issues/4087 +// for a discussion. +#if defined(__clang__) + #pragma clang diagnostic push + #pragma clang diagnostic ignored "-Wweak-vtables" +#endif + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +//////////////// +// exceptions // +//////////////// + +/// @brief general exception of the @ref basic_json class +/// @sa https://json.nlohmann.me/api/basic_json/exception/ +class exception : public std::exception +{ + public: + /// returns the explanatory string + const char* what() const noexcept override + { + return m.what(); + } + + /// the id of the exception + const int id; // NOLINT(cppcoreguidelines-non-private-member-variables-in-classes) + + protected: + JSON_HEDLEY_NON_NULL(3) + exception(int id_, const char* what_arg) : id(id_), m(what_arg) {} // NOLINT(bugprone-throw-keyword-missing) + + static std::string name(const std::string& ename, int id_) + { + return concat("[json.exception.", ename, '.', std::to_string(id_), "] "); + } + + static std::string diagnostics(std::nullptr_t /*leaf_element*/) + { + return ""; + } + + template + static std::string diagnostics(const BasicJsonType* leaf_element) + { +#if JSON_DIAGNOSTICS + std::vector tokens; + for (const auto* current = leaf_element; current != nullptr && current->m_parent != nullptr; current = current->m_parent) + { + switch (current->m_parent->type()) + { + case value_t::array: + { + for (std::size_t i = 0; i < current->m_parent->m_data.m_value.array->size(); ++i) + { + if (¤t->m_parent->m_data.m_value.array->operator[](i) == current) + { + tokens.emplace_back(std::to_string(i)); + break; + } + } + break; + } + + case value_t::object: + { + for (const auto& element : *current->m_parent->m_data.m_value.object) + { + if (&element.second == current) + { + tokens.emplace_back(element.first.c_str()); + break; + } + } + break; + } + + case value_t::null: // LCOV_EXCL_LINE + case value_t::string: // LCOV_EXCL_LINE + case value_t::boolean: // LCOV_EXCL_LINE + case value_t::number_integer: // LCOV_EXCL_LINE + case value_t::number_unsigned: // LCOV_EXCL_LINE + case value_t::number_float: // LCOV_EXCL_LINE + case value_t::binary: // LCOV_EXCL_LINE + case value_t::discarded: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + break; // LCOV_EXCL_LINE + } + } + + if (tokens.empty()) + { + return ""; + } + + auto str = std::accumulate(tokens.rbegin(), tokens.rend(), std::string{}, + [](const std::string & a, const std::string & b) + { + return concat(a, '/', detail::escape(b)); + }); + + return concat('(', str, ") ", get_byte_positions(leaf_element)); +#else + return get_byte_positions(leaf_element); +#endif + } + + private: + /// an exception object as storage for error messages + std::runtime_error m; +#if JSON_DIAGNOSTIC_POSITIONS + template + static std::string get_byte_positions(const BasicJsonType* leaf_element) + { + if ((leaf_element->start_pos() != std::string::npos) && (leaf_element->end_pos() != std::string::npos)) + { + return concat("(bytes ", std::to_string(leaf_element->start_pos()), "-", std::to_string(leaf_element->end_pos()), ") "); + } + return ""; + } +#else + template + static std::string get_byte_positions(const BasicJsonType* leaf_element) + { + static_cast(leaf_element); + return ""; + } +#endif +}; + +/// @brief exception indicating a parse error +/// @sa https://json.nlohmann.me/api/basic_json/parse_error/ +class parse_error : public exception +{ + public: + /*! + @brief create a parse error exception + @param[in] id_ the id of the exception + @param[in] pos the position where the error occurred (or with + chars_read_total=0 if the position cannot be + determined) + @param[in] what_arg the explanatory string + @return parse_error object + */ + template::value, int> = 0> + static parse_error create(int id_, const position_t& pos, const std::string& what_arg, BasicJsonContext context) + { + const std::string w = concat(exception::name("parse_error", id_), "parse error", + position_string(pos), ": ", exception::diagnostics(context), what_arg); + return {id_, pos.chars_read_total, w.c_str()}; + } + + template::value, int> = 0> + static parse_error create(int id_, std::size_t byte_, const std::string& what_arg, BasicJsonContext context) + { + const std::string w = concat(exception::name("parse_error", id_), "parse error", + (byte_ != 0 ? (concat(" at byte ", std::to_string(byte_))) : ""), + ": ", exception::diagnostics(context), what_arg); + return {id_, byte_, w.c_str()}; + } + + /*! + @brief byte index of the parse error + + The byte index of the last read character in the input file. + + @note For an input with n bytes, 1 is the index of the first character and + n+1 is the index of the terminating null byte or the end of file. + This also holds true when reading a byte vector (CBOR or MessagePack). + */ + const std::size_t byte; + + private: + parse_error(int id_, std::size_t byte_, const char* what_arg) + : exception(id_, what_arg), byte(byte_) {} + + static std::string position_string(const position_t& pos) + { + return concat(" at line ", std::to_string(pos.lines_read + 1), + ", column ", std::to_string(pos.chars_read_current_line)); + } +}; + +/// @brief exception indicating errors with iterators +/// @sa https://json.nlohmann.me/api/basic_json/invalid_iterator/ +class invalid_iterator : public exception +{ + public: + template::value, int> = 0> + static invalid_iterator create(int id_, const std::string& what_arg, BasicJsonContext context) + { + const std::string w = concat(exception::name("invalid_iterator", id_), exception::diagnostics(context), what_arg); + return {id_, w.c_str()}; + } + + private: + JSON_HEDLEY_NON_NULL(3) + invalid_iterator(int id_, const char* what_arg) + : exception(id_, what_arg) {} +}; + +/// @brief exception indicating executing a member function with a wrong type +/// @sa https://json.nlohmann.me/api/basic_json/type_error/ +class type_error : public exception +{ + public: + template::value, int> = 0> + static type_error create(int id_, const std::string& what_arg, BasicJsonContext context) + { + const std::string w = concat(exception::name("type_error", id_), exception::diagnostics(context), what_arg); + return {id_, w.c_str()}; + } + + private: + JSON_HEDLEY_NON_NULL(3) + type_error(int id_, const char* what_arg) : exception(id_, what_arg) {} +}; + +/// @brief exception indicating access out of the defined range +/// @sa https://json.nlohmann.me/api/basic_json/out_of_range/ +class out_of_range : public exception +{ + public: + template::value, int> = 0> + static out_of_range create(int id_, const std::string& what_arg, BasicJsonContext context) + { + const std::string w = concat(exception::name("out_of_range", id_), exception::diagnostics(context), what_arg); + return {id_, w.c_str()}; + } + + private: + JSON_HEDLEY_NON_NULL(3) + out_of_range(int id_, const char* what_arg) : exception(id_, what_arg) {} +}; + +/// @brief exception indicating other library errors +/// @sa https://json.nlohmann.me/api/basic_json/other_error/ +class other_error : public exception +{ + public: + template::value, int> = 0> + static other_error create(int id_, const std::string& what_arg, BasicJsonContext context) + { + const std::string w = concat(exception::name("other_error", id_), exception::diagnostics(context), what_arg); + return {id_, w.c_str()}; + } + + private: + JSON_HEDLEY_NON_NULL(3) + other_error(int id_, const char* what_arg) : exception(id_, what_arg) {} +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + +#if defined(__clang__) + #pragma clang diagnostic pop +#endif diff --git a/src/detail/include/nlohmann/detail/hash.hpp b/src/detail/include/nlohmann/detail/hash.hpp new file mode 100644 index 000000000..c764362fe --- /dev/null +++ b/src/detail/include/nlohmann/detail/hash.hpp @@ -0,0 +1,129 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // uint8_t +#include // size_t +#include // hash + +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +// boost::hash_combine +inline std::size_t combine(std::size_t seed, std::size_t h) noexcept +{ + seed ^= h + 0x9e3779b9 + (seed << 6U) + (seed >> 2U); + return seed; +} + +/*! +@brief hash a JSON value + +The hash function tries to rely on std::hash where possible. Furthermore, the +type of the JSON value is taken into account to have different hash values for +null, 0, 0U, and false, etc. + +@tparam BasicJsonType basic_json specialization +@param j JSON value to hash +@return hash value of j +*/ +template +std::size_t hash(const BasicJsonType& j) +{ + using string_t = typename BasicJsonType::string_t; + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + + const auto type = static_cast(j.type()); + switch (j.type()) + { + case BasicJsonType::value_t::null: + case BasicJsonType::value_t::discarded: + { + return combine(type, 0); + } + + case BasicJsonType::value_t::object: + { + auto seed = combine(type, j.size()); + for (const auto& element : j.items()) + { + const auto h = std::hash {}(element.key()); + seed = combine(seed, h); + seed = combine(seed, hash(element.value())); + } + return seed; + } + + case BasicJsonType::value_t::array: + { + auto seed = combine(type, j.size()); + for (const auto& element : j) + { + seed = combine(seed, hash(element)); + } + return seed; + } + + case BasicJsonType::value_t::string: + { + const auto h = std::hash {}(j.template get_ref()); + return combine(type, h); + } + + case BasicJsonType::value_t::boolean: + { + const auto h = std::hash {}(j.template get()); + return combine(type, h); + } + + case BasicJsonType::value_t::number_integer: + { + const auto h = std::hash {}(j.template get()); + return combine(type, h); + } + + case BasicJsonType::value_t::number_unsigned: + { + const auto h = std::hash {}(j.template get()); + return combine(type, h); + } + + case BasicJsonType::value_t::number_float: + { + const auto h = std::hash {}(j.template get()); + return combine(type, h); + } + + case BasicJsonType::value_t::binary: + { + auto seed = combine(type, j.get_binary().size()); + const auto h = std::hash {}(j.get_binary().has_subtype()); + seed = combine(seed, h); + seed = combine(seed, static_cast(j.get_binary().subtype())); + for (const auto byte : j.get_binary()) + { + seed = combine(seed, std::hash {}(byte)); + } + return seed; + } + + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + return 0; // LCOV_EXCL_LINE + } +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/input/binary_reader.hpp b/src/detail/include/nlohmann/detail/input/binary_reader.hpp new file mode 100644 index 000000000..1535183eb --- /dev/null +++ b/src/detail/include/nlohmann/detail/input/binary_reader.hpp @@ -0,0 +1,3081 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // generate_n +#include // array +#include // ldexp +#include // size_t +#include // uint8_t, uint16_t, uint32_t, uint64_t +#include // snprintf +#include // memcpy +#include // back_inserter +#include // numeric_limits +#include // char_traits, string +#include // make_pair, move +#include // vector +#ifdef __cpp_lib_byteswap + #include //byteswap +#endif + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// how to treat CBOR tags +enum class cbor_tag_handler_t +{ + error, ///< throw a parse_error exception in case of a tag + ignore, ///< ignore tags + store ///< store tags as binary type +}; + +/*! +@brief determine system byte order + +@return true if and only if system's byte order is little endian + +@note from https://stackoverflow.com/a/1001328/266378 +*/ +inline bool little_endianness(int num = 1) noexcept +{ + return *reinterpret_cast(&num) == 1; +} + +/////////////////// +// binary reader // +/////////////////// + +/*! +@brief deserialization of CBOR, MessagePack, and UBJSON values +*/ +template> +class binary_reader +{ + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using string_t = typename BasicJsonType::string_t; + using binary_t = typename BasicJsonType::binary_t; + using json_sax_t = SAX; + using char_type = typename InputAdapterType::char_type; + using char_int_type = typename char_traits::int_type; + + public: + /*! + @brief create a binary reader + + @param[in] adapter input adapter to read from + */ + explicit binary_reader(InputAdapterType&& adapter, const input_format_t format = input_format_t::json) noexcept : ia(std::move(adapter)), input_format(format) + { + (void)detail::is_sax_static_asserts {}; + } + + // make class move-only + binary_reader(const binary_reader&) = delete; + binary_reader(binary_reader&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor) + binary_reader& operator=(const binary_reader&) = delete; + binary_reader& operator=(binary_reader&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor) + ~binary_reader() = default; + + /*! + @param[in] format the binary format to parse + @param[in] sax_ a SAX event processor + @param[in] strict whether to expect the input to be consumed completed + @param[in] tag_handler how to treat CBOR tags + + @return whether parsing was successful + */ + JSON_HEDLEY_NON_NULL(3) + bool sax_parse(const input_format_t format, + json_sax_t* sax_, + const bool strict = true, + const cbor_tag_handler_t tag_handler = cbor_tag_handler_t::error) + { + sax = sax_; + bool result = false; + + switch (format) + { + case input_format_t::bson: + result = parse_bson_internal(); + break; + + case input_format_t::cbor: + result = parse_cbor_internal(true, tag_handler); + break; + + case input_format_t::msgpack: + result = parse_msgpack_internal(); + break; + + case input_format_t::ubjson: + case input_format_t::bjdata: + result = parse_ubjson_internal(); + break; + + case input_format_t::json: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + } + + // strict mode: next byte must be EOF + if (result && strict) + { + if (input_format == input_format_t::ubjson || input_format == input_format_t::bjdata) + { + get_ignore_noop(); + } + else + { + get(); + } + + if (JSON_HEDLEY_UNLIKELY(current != char_traits::eof())) + { + return sax->parse_error(chars_read, get_token_string(), parse_error::create(110, chars_read, + exception_message(input_format, concat("expected end of input; last byte: 0x", get_token_string()), "value"), nullptr)); + } + } + + return result; + } + + private: + ////////// + // BSON // + ////////// + + /*! + @brief Reads in a BSON-object and passes it to the SAX-parser. + @return whether a valid BSON-value was passed to the SAX parser + */ + bool parse_bson_internal() + { + std::int32_t document_size{}; + get_number(input_format_t::bson, document_size); + + if (JSON_HEDLEY_UNLIKELY(!sax->start_object(detail::unknown_size()))) + { + return false; + } + + if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_list(/*is_array*/false))) + { + return false; + } + + return sax->end_object(); + } + + /*! + @brief Parses a C-style string from the BSON input. + @param[in,out] result A reference to the string variable where the read + string is to be stored. + @return `true` if the \x00-byte indicating the end of the string was + encountered before the EOF; false` indicates an unexpected EOF. + */ + bool get_bson_cstr(string_t& result) + { + auto out = std::back_inserter(result); + while (true) + { + get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::bson, "cstring"))) + { + return false; + } + if (current == 0x00) + { + return true; + } + *out++ = static_cast(current); + } + } + + /*! + @brief Parses a zero-terminated string of length @a len from the BSON + input. + @param[in] len The length (including the zero-byte at the end) of the + string to be read. + @param[in,out] result A reference to the string variable where the read + string is to be stored. + @tparam NumberType The type of the length @a len + @pre len >= 1 + @return `true` if the string was successfully parsed + */ + template + bool get_bson_string(const NumberType len, string_t& result) + { + if (JSON_HEDLEY_UNLIKELY(len < 1)) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bson, concat("string length must be at least 1, is ", std::to_string(len)), "string"), nullptr)); + } + + return get_string(input_format_t::bson, len - static_cast(1), result) && get() != char_traits::eof(); + } + + /*! + @brief Parses a byte array input of length @a len from the BSON input. + @param[in] len The length of the byte array to be read. + @param[in,out] result A reference to the binary variable where the read + array is to be stored. + @tparam NumberType The type of the length @a len + @pre len >= 0 + @return `true` if the byte array was successfully parsed + */ + template + bool get_bson_binary(const NumberType len, binary_t& result) + { + if (JSON_HEDLEY_UNLIKELY(len < 0)) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::bson, concat("byte array length cannot be negative, is ", std::to_string(len)), "binary"), nullptr)); + } + + // All BSON binary values have a subtype + std::uint8_t subtype{}; + get_number(input_format_t::bson, subtype); + result.set_subtype(subtype); + + return get_binary(input_format_t::bson, len, result); + } + + /*! + @brief Read a BSON document element of the given @a element_type. + @param[in] element_type The BSON element type, c.f. http://bsonspec.org/spec.html + @param[in] element_type_parse_position The position in the input stream, + where the `element_type` was read. + @warning Not all BSON element types are supported yet. An unsupported + @a element_type will give rise to a parse_error.114: + Unsupported BSON record type 0x... + @return whether a valid BSON-object/array was passed to the SAX parser + */ + bool parse_bson_element_internal(const char_int_type element_type, + const std::size_t element_type_parse_position) + { + switch (element_type) + { + case 0x01: // double + { + double number{}; + return get_number(input_format_t::bson, number) && sax->number_float(static_cast(number), ""); + } + + case 0x02: // string + { + std::int32_t len{}; + string_t value; + return get_number(input_format_t::bson, len) && get_bson_string(len, value) && sax->string(value); + } + + case 0x03: // object + { + return parse_bson_internal(); + } + + case 0x04: // array + { + return parse_bson_array(); + } + + case 0x05: // binary + { + std::int32_t len{}; + binary_t value; + return get_number(input_format_t::bson, len) && get_bson_binary(len, value) && sax->binary(value); + } + + case 0x08: // boolean + { + return sax->boolean(get() != 0); + } + + case 0x0A: // null + { + return sax->null(); + } + + case 0x10: // int32 + { + std::int32_t value{}; + return get_number(input_format_t::bson, value) && sax->number_integer(value); + } + + case 0x12: // int64 + { + std::int64_t value{}; + return get_number(input_format_t::bson, value) && sax->number_integer(value); + } + + case 0x11: // uint64 + { + std::uint64_t value{}; + return get_number(input_format_t::bson, value) && sax->number_unsigned(value); + } + + default: // anything else is not supported (yet) + { + std::array cr{{}}; + static_cast((std::snprintf)(cr.data(), cr.size(), "%.2hhX", static_cast(element_type))); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) + const std::string cr_str{cr.data()}; + return sax->parse_error(element_type_parse_position, cr_str, + parse_error::create(114, element_type_parse_position, concat("Unsupported BSON record type 0x", cr_str), nullptr)); + } + } + } + + /*! + @brief Read a BSON element list (as specified in the BSON-spec) + + The same binary layout is used for objects and arrays, hence it must be + indicated with the argument @a is_array which one is expected + (true --> array, false --> object). + + @param[in] is_array Determines if the element list being read is to be + treated as an object (@a is_array == false), or as an + array (@a is_array == true). + @return whether a valid BSON-object/array was passed to the SAX parser + */ + bool parse_bson_element_list(const bool is_array) + { + string_t key; + + while (auto element_type = get()) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::bson, "element list"))) + { + return false; + } + + const std::size_t element_type_parse_position = chars_read; + if (JSON_HEDLEY_UNLIKELY(!get_bson_cstr(key))) + { + return false; + } + + if (!is_array && !sax->key(key)) + { + return false; + } + + if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_internal(element_type, element_type_parse_position))) + { + return false; + } + + // get_bson_cstr only appends + key.clear(); + } + + return true; + } + + /*! + @brief Reads an array from the BSON input and passes it to the SAX-parser. + @return whether a valid BSON-array was passed to the SAX parser + */ + bool parse_bson_array() + { + std::int32_t document_size{}; + get_number(input_format_t::bson, document_size); + + if (JSON_HEDLEY_UNLIKELY(!sax->start_array(detail::unknown_size()))) + { + return false; + } + + if (JSON_HEDLEY_UNLIKELY(!parse_bson_element_list(/*is_array*/true))) + { + return false; + } + + return sax->end_array(); + } + + ////////// + // CBOR // + ////////// + + /*! + @param[in] get_char whether a new character should be retrieved from the + input (true) or whether the last read character should + be considered instead (false) + @param[in] tag_handler how CBOR tags should be treated + + @return whether a valid CBOR value was passed to the SAX parser + */ + bool parse_cbor_internal(const bool get_char, + const cbor_tag_handler_t tag_handler) + { + switch (get_char ? get() : current) + { + // EOF + case char_traits::eof(): + return unexpect_eof(input_format_t::cbor, "value"); + + // Integer 0x00..0x17 (0..23) + case 0x00: + case 0x01: + case 0x02: + case 0x03: + case 0x04: + case 0x05: + case 0x06: + case 0x07: + case 0x08: + case 0x09: + case 0x0A: + case 0x0B: + case 0x0C: + case 0x0D: + case 0x0E: + case 0x0F: + case 0x10: + case 0x11: + case 0x12: + case 0x13: + case 0x14: + case 0x15: + case 0x16: + case 0x17: + return sax->number_unsigned(static_cast(current)); + + case 0x18: // Unsigned integer (one-byte uint8_t follows) + { + std::uint8_t number{}; + return get_number(input_format_t::cbor, number) && sax->number_unsigned(number); + } + + case 0x19: // Unsigned integer (two-byte uint16_t follows) + { + std::uint16_t number{}; + return get_number(input_format_t::cbor, number) && sax->number_unsigned(number); + } + + case 0x1A: // Unsigned integer (four-byte uint32_t follows) + { + std::uint32_t number{}; + return get_number(input_format_t::cbor, number) && sax->number_unsigned(number); + } + + case 0x1B: // Unsigned integer (eight-byte uint64_t follows) + { + std::uint64_t number{}; + return get_number(input_format_t::cbor, number) && sax->number_unsigned(number); + } + + // Negative integer -1-0x00..-1-0x17 (-1..-24) + case 0x20: + case 0x21: + case 0x22: + case 0x23: + case 0x24: + case 0x25: + case 0x26: + case 0x27: + case 0x28: + case 0x29: + case 0x2A: + case 0x2B: + case 0x2C: + case 0x2D: + case 0x2E: + case 0x2F: + case 0x30: + case 0x31: + case 0x32: + case 0x33: + case 0x34: + case 0x35: + case 0x36: + case 0x37: + return sax->number_integer(static_cast(0x20 - 1 - current)); + + case 0x38: // Negative integer (one-byte uint8_t follows) + { + std::uint8_t number{}; + return get_number(input_format_t::cbor, number) && sax->number_integer(static_cast(-1) - number); + } + + case 0x39: // Negative integer -1-n (two-byte uint16_t follows) + { + std::uint16_t number{}; + return get_number(input_format_t::cbor, number) && sax->number_integer(static_cast(-1) - number); + } + + case 0x3A: // Negative integer -1-n (four-byte uint32_t follows) + { + std::uint32_t number{}; + return get_number(input_format_t::cbor, number) && sax->number_integer(static_cast(-1) - number); + } + + case 0x3B: // Negative integer -1-n (eight-byte uint64_t follows) + { + std::uint64_t number{}; + return get_number(input_format_t::cbor, number) && sax->number_integer(static_cast(-1) + - static_cast(number)); + } + + // Binary data (0x00..0x17 bytes follow) + case 0x40: + case 0x41: + case 0x42: + case 0x43: + case 0x44: + case 0x45: + case 0x46: + case 0x47: + case 0x48: + case 0x49: + case 0x4A: + case 0x4B: + case 0x4C: + case 0x4D: + case 0x4E: + case 0x4F: + case 0x50: + case 0x51: + case 0x52: + case 0x53: + case 0x54: + case 0x55: + case 0x56: + case 0x57: + case 0x58: // Binary data (one-byte uint8_t for n follows) + case 0x59: // Binary data (two-byte uint16_t for n follow) + case 0x5A: // Binary data (four-byte uint32_t for n follow) + case 0x5B: // Binary data (eight-byte uint64_t for n follow) + case 0x5F: // Binary data (indefinite length) + { + binary_t b; + return get_cbor_binary(b) && sax->binary(b); + } + + // UTF-8 string (0x00..0x17 bytes follow) + case 0x60: + case 0x61: + case 0x62: + case 0x63: + case 0x64: + case 0x65: + case 0x66: + case 0x67: + case 0x68: + case 0x69: + case 0x6A: + case 0x6B: + case 0x6C: + case 0x6D: + case 0x6E: + case 0x6F: + case 0x70: + case 0x71: + case 0x72: + case 0x73: + case 0x74: + case 0x75: + case 0x76: + case 0x77: + case 0x78: // UTF-8 string (one-byte uint8_t for n follows) + case 0x79: // UTF-8 string (two-byte uint16_t for n follow) + case 0x7A: // UTF-8 string (four-byte uint32_t for n follow) + case 0x7B: // UTF-8 string (eight-byte uint64_t for n follow) + case 0x7F: // UTF-8 string (indefinite length) + { + string_t s; + return get_cbor_string(s) && sax->string(s); + } + + // array (0x00..0x17 data items follow) + case 0x80: + case 0x81: + case 0x82: + case 0x83: + case 0x84: + case 0x85: + case 0x86: + case 0x87: + case 0x88: + case 0x89: + case 0x8A: + case 0x8B: + case 0x8C: + case 0x8D: + case 0x8E: + case 0x8F: + case 0x90: + case 0x91: + case 0x92: + case 0x93: + case 0x94: + case 0x95: + case 0x96: + case 0x97: + return get_cbor_array( + conditional_static_cast(static_cast(current) & 0x1Fu), tag_handler); + + case 0x98: // array (one-byte uint8_t for n follows) + { + std::uint8_t len{}; + return get_number(input_format_t::cbor, len) && get_cbor_array(static_cast(len), tag_handler); + } + + case 0x99: // array (two-byte uint16_t for n follow) + { + std::uint16_t len{}; + return get_number(input_format_t::cbor, len) && get_cbor_array(static_cast(len), tag_handler); + } + + case 0x9A: // array (four-byte uint32_t for n follow) + { + std::uint32_t len{}; + return get_number(input_format_t::cbor, len) && get_cbor_array(conditional_static_cast(len), tag_handler); + } + + case 0x9B: // array (eight-byte uint64_t for n follow) + { + std::uint64_t len{}; + return get_number(input_format_t::cbor, len) && get_cbor_array(conditional_static_cast(len), tag_handler); + } + + case 0x9F: // array (indefinite length) + return get_cbor_array(detail::unknown_size(), tag_handler); + + // map (0x00..0x17 pairs of data items follow) + case 0xA0: + case 0xA1: + case 0xA2: + case 0xA3: + case 0xA4: + case 0xA5: + case 0xA6: + case 0xA7: + case 0xA8: + case 0xA9: + case 0xAA: + case 0xAB: + case 0xAC: + case 0xAD: + case 0xAE: + case 0xAF: + case 0xB0: + case 0xB1: + case 0xB2: + case 0xB3: + case 0xB4: + case 0xB5: + case 0xB6: + case 0xB7: + return get_cbor_object(conditional_static_cast(static_cast(current) & 0x1Fu), tag_handler); + + case 0xB8: // map (one-byte uint8_t for n follows) + { + std::uint8_t len{}; + return get_number(input_format_t::cbor, len) && get_cbor_object(static_cast(len), tag_handler); + } + + case 0xB9: // map (two-byte uint16_t for n follow) + { + std::uint16_t len{}; + return get_number(input_format_t::cbor, len) && get_cbor_object(static_cast(len), tag_handler); + } + + case 0xBA: // map (four-byte uint32_t for n follow) + { + std::uint32_t len{}; + return get_number(input_format_t::cbor, len) && get_cbor_object(conditional_static_cast(len), tag_handler); + } + + case 0xBB: // map (eight-byte uint64_t for n follow) + { + std::uint64_t len{}; + return get_number(input_format_t::cbor, len) && get_cbor_object(conditional_static_cast(len), tag_handler); + } + + case 0xBF: // map (indefinite length) + return get_cbor_object(detail::unknown_size(), tag_handler); + + case 0xC6: // tagged item + case 0xC7: + case 0xC8: + case 0xC9: + case 0xCA: + case 0xCB: + case 0xCC: + case 0xCD: + case 0xCE: + case 0xCF: + case 0xD0: + case 0xD1: + case 0xD2: + case 0xD3: + case 0xD4: + case 0xD8: // tagged item (1 byte follows) + case 0xD9: // tagged item (2 bytes follow) + case 0xDA: // tagged item (4 bytes follow) + case 0xDB: // tagged item (8 bytes follow) + { + switch (tag_handler) + { + case cbor_tag_handler_t::error: + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)); + } + + case cbor_tag_handler_t::ignore: + { + // ignore binary subtype + switch (current) + { + case 0xD8: + { + std::uint8_t subtype_to_ignore{}; + get_number(input_format_t::cbor, subtype_to_ignore); + break; + } + case 0xD9: + { + std::uint16_t subtype_to_ignore{}; + get_number(input_format_t::cbor, subtype_to_ignore); + break; + } + case 0xDA: + { + std::uint32_t subtype_to_ignore{}; + get_number(input_format_t::cbor, subtype_to_ignore); + break; + } + case 0xDB: + { + std::uint64_t subtype_to_ignore{}; + get_number(input_format_t::cbor, subtype_to_ignore); + break; + } + default: + break; + } + return parse_cbor_internal(true, tag_handler); + } + + case cbor_tag_handler_t::store: + { + binary_t b; + // use binary subtype and store in a binary container + switch (current) + { + case 0xD8: + { + std::uint8_t subtype{}; + get_number(input_format_t::cbor, subtype); + b.set_subtype(detail::conditional_static_cast(subtype)); + break; + } + case 0xD9: + { + std::uint16_t subtype{}; + get_number(input_format_t::cbor, subtype); + b.set_subtype(detail::conditional_static_cast(subtype)); + break; + } + case 0xDA: + { + std::uint32_t subtype{}; + get_number(input_format_t::cbor, subtype); + b.set_subtype(detail::conditional_static_cast(subtype)); + break; + } + case 0xDB: + { + std::uint64_t subtype{}; + get_number(input_format_t::cbor, subtype); + b.set_subtype(detail::conditional_static_cast(subtype)); + break; + } + default: + return parse_cbor_internal(true, tag_handler); + } + get(); + return get_cbor_binary(b) && sax->binary(b); + } + + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + return false; // LCOV_EXCL_LINE + } + } + + case 0xF4: // false + return sax->boolean(false); + + case 0xF5: // true + return sax->boolean(true); + + case 0xF6: // null + return sax->null(); + + case 0xF9: // Half-Precision Float (two-byte IEEE 754) + { + const auto byte1_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number"))) + { + return false; + } + const auto byte2_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "number"))) + { + return false; + } + + const auto byte1 = static_cast(byte1_raw); + const auto byte2 = static_cast(byte2_raw); + + // Code from RFC 7049, Appendix D, Figure 3: + // As half-precision floating-point numbers were only added + // to IEEE 754 in 2008, today's programming platforms often + // still only have limited support for them. It is very + // easy to include at least decoding support for them even + // without such support. An example of a small decoder for + // half-precision floating-point numbers in the C language + // is shown in Fig. 3. + const auto half = static_cast((byte1 << 8u) + byte2); + const double val = [&half] + { + const int exp = (half >> 10u) & 0x1Fu; + const unsigned int mant = half & 0x3FFu; + JSON_ASSERT(0 <= exp&& exp <= 32); + JSON_ASSERT(mant <= 1024); + switch (exp) + { + case 0: + return std::ldexp(mant, -24); + case 31: + return (mant == 0) + ? std::numeric_limits::infinity() + : std::numeric_limits::quiet_NaN(); + default: + return std::ldexp(mant + 1024, exp - 25); + } + }(); + return sax->number_float((half & 0x8000u) != 0 + ? static_cast(-val) + : static_cast(val), ""); + } + + case 0xFA: // Single-Precision Float (four-byte IEEE 754) + { + float number{}; + return get_number(input_format_t::cbor, number) && sax->number_float(static_cast(number), ""); + } + + case 0xFB: // Double-Precision Float (eight-byte IEEE 754) + { + double number{}; + return get_number(input_format_t::cbor, number) && sax->number_float(static_cast(number), ""); + } + + default: // anything else (0xFF is handled inside the other types) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::cbor, concat("invalid byte: 0x", last_token), "value"), nullptr)); + } + } + } + + /*! + @brief reads a CBOR string + + This function first reads starting bytes to determine the expected + string length and then copies this number of bytes into a string. + Additionally, CBOR's strings with indefinite lengths are supported. + + @param[out] result created string + + @return whether string creation completed + */ + bool get_cbor_string(string_t& result) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "string"))) + { + return false; + } + + switch (current) + { + // UTF-8 string (0x00..0x17 bytes follow) + case 0x60: + case 0x61: + case 0x62: + case 0x63: + case 0x64: + case 0x65: + case 0x66: + case 0x67: + case 0x68: + case 0x69: + case 0x6A: + case 0x6B: + case 0x6C: + case 0x6D: + case 0x6E: + case 0x6F: + case 0x70: + case 0x71: + case 0x72: + case 0x73: + case 0x74: + case 0x75: + case 0x76: + case 0x77: + { + return get_string(input_format_t::cbor, static_cast(current) & 0x1Fu, result); + } + + case 0x78: // UTF-8 string (one-byte uint8_t for n follows) + { + std::uint8_t len{}; + return get_number(input_format_t::cbor, len) && get_string(input_format_t::cbor, len, result); + } + + case 0x79: // UTF-8 string (two-byte uint16_t for n follow) + { + std::uint16_t len{}; + return get_number(input_format_t::cbor, len) && get_string(input_format_t::cbor, len, result); + } + + case 0x7A: // UTF-8 string (four-byte uint32_t for n follow) + { + std::uint32_t len{}; + return get_number(input_format_t::cbor, len) && get_string(input_format_t::cbor, len, result); + } + + case 0x7B: // UTF-8 string (eight-byte uint64_t for n follow) + { + std::uint64_t len{}; + return get_number(input_format_t::cbor, len) && get_string(input_format_t::cbor, len, result); + } + + case 0x7F: // UTF-8 string (indefinite length) + { + while (get() != 0xFF) + { + string_t chunk; + if (!get_cbor_string(chunk)) + { + return false; + } + result.append(chunk); + } + return true; + } + + default: + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::cbor, concat("expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x", last_token), "string"), nullptr)); + } + } + } + + /*! + @brief reads a CBOR byte array + + This function first reads starting bytes to determine the expected + byte array length and then copies this number of bytes into the byte array. + Additionally, CBOR's byte arrays with indefinite lengths are supported. + + @param[out] result created byte array + + @return whether byte array creation completed + */ + bool get_cbor_binary(binary_t& result) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::cbor, "binary"))) + { + return false; + } + + switch (current) + { + // Binary data (0x00..0x17 bytes follow) + case 0x40: + case 0x41: + case 0x42: + case 0x43: + case 0x44: + case 0x45: + case 0x46: + case 0x47: + case 0x48: + case 0x49: + case 0x4A: + case 0x4B: + case 0x4C: + case 0x4D: + case 0x4E: + case 0x4F: + case 0x50: + case 0x51: + case 0x52: + case 0x53: + case 0x54: + case 0x55: + case 0x56: + case 0x57: + { + return get_binary(input_format_t::cbor, static_cast(current) & 0x1Fu, result); + } + + case 0x58: // Binary data (one-byte uint8_t for n follows) + { + std::uint8_t len{}; + return get_number(input_format_t::cbor, len) && + get_binary(input_format_t::cbor, len, result); + } + + case 0x59: // Binary data (two-byte uint16_t for n follow) + { + std::uint16_t len{}; + return get_number(input_format_t::cbor, len) && + get_binary(input_format_t::cbor, len, result); + } + + case 0x5A: // Binary data (four-byte uint32_t for n follow) + { + std::uint32_t len{}; + return get_number(input_format_t::cbor, len) && + get_binary(input_format_t::cbor, len, result); + } + + case 0x5B: // Binary data (eight-byte uint64_t for n follow) + { + std::uint64_t len{}; + return get_number(input_format_t::cbor, len) && + get_binary(input_format_t::cbor, len, result); + } + + case 0x5F: // Binary data (indefinite length) + { + while (get() != 0xFF) + { + binary_t chunk; + if (!get_cbor_binary(chunk)) + { + return false; + } + result.insert(result.end(), chunk.begin(), chunk.end()); + } + return true; + } + + default: + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::cbor, concat("expected length specification (0x40-0x5B) or indefinite binary array type (0x5F); last byte: 0x", last_token), "binary"), nullptr)); + } + } + } + + /*! + @param[in] len the length of the array or detail::unknown_size() for an + array of indefinite size + @param[in] tag_handler how CBOR tags should be treated + @return whether array creation completed + */ + bool get_cbor_array(const std::size_t len, + const cbor_tag_handler_t tag_handler) + { + if (JSON_HEDLEY_UNLIKELY(!sax->start_array(len))) + { + return false; + } + + if (len != detail::unknown_size()) + { + for (std::size_t i = 0; i < len; ++i) + { + if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(true, tag_handler))) + { + return false; + } + } + } + else + { + while (get() != 0xFF) + { + if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(false, tag_handler))) + { + return false; + } + } + } + + return sax->end_array(); + } + + /*! + @param[in] len the length of the object or detail::unknown_size() for an + object of indefinite size + @param[in] tag_handler how CBOR tags should be treated + @return whether object creation completed + */ + bool get_cbor_object(const std::size_t len, + const cbor_tag_handler_t tag_handler) + { + if (JSON_HEDLEY_UNLIKELY(!sax->start_object(len))) + { + return false; + } + + if (len != 0) + { + string_t key; + if (len != detail::unknown_size()) + { + for (std::size_t i = 0; i < len; ++i) + { + get(); + if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key))) + { + return false; + } + + if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(true, tag_handler))) + { + return false; + } + key.clear(); + } + } + else + { + while (get() != 0xFF) + { + if (JSON_HEDLEY_UNLIKELY(!get_cbor_string(key) || !sax->key(key))) + { + return false; + } + + if (JSON_HEDLEY_UNLIKELY(!parse_cbor_internal(true, tag_handler))) + { + return false; + } + key.clear(); + } + } + } + + return sax->end_object(); + } + + ///////////// + // MsgPack // + ///////////// + + /*! + @return whether a valid MessagePack value was passed to the SAX parser + */ + bool parse_msgpack_internal() + { + switch (get()) + { + // EOF + case char_traits::eof(): + return unexpect_eof(input_format_t::msgpack, "value"); + + // positive fixint + case 0x00: + case 0x01: + case 0x02: + case 0x03: + case 0x04: + case 0x05: + case 0x06: + case 0x07: + case 0x08: + case 0x09: + case 0x0A: + case 0x0B: + case 0x0C: + case 0x0D: + case 0x0E: + case 0x0F: + case 0x10: + case 0x11: + case 0x12: + case 0x13: + case 0x14: + case 0x15: + case 0x16: + case 0x17: + case 0x18: + case 0x19: + case 0x1A: + case 0x1B: + case 0x1C: + case 0x1D: + case 0x1E: + case 0x1F: + case 0x20: + case 0x21: + case 0x22: + case 0x23: + case 0x24: + case 0x25: + case 0x26: + case 0x27: + case 0x28: + case 0x29: + case 0x2A: + case 0x2B: + case 0x2C: + case 0x2D: + case 0x2E: + case 0x2F: + case 0x30: + case 0x31: + case 0x32: + case 0x33: + case 0x34: + case 0x35: + case 0x36: + case 0x37: + case 0x38: + case 0x39: + case 0x3A: + case 0x3B: + case 0x3C: + case 0x3D: + case 0x3E: + case 0x3F: + case 0x40: + case 0x41: + case 0x42: + case 0x43: + case 0x44: + case 0x45: + case 0x46: + case 0x47: + case 0x48: + case 0x49: + case 0x4A: + case 0x4B: + case 0x4C: + case 0x4D: + case 0x4E: + case 0x4F: + case 0x50: + case 0x51: + case 0x52: + case 0x53: + case 0x54: + case 0x55: + case 0x56: + case 0x57: + case 0x58: + case 0x59: + case 0x5A: + case 0x5B: + case 0x5C: + case 0x5D: + case 0x5E: + case 0x5F: + case 0x60: + case 0x61: + case 0x62: + case 0x63: + case 0x64: + case 0x65: + case 0x66: + case 0x67: + case 0x68: + case 0x69: + case 0x6A: + case 0x6B: + case 0x6C: + case 0x6D: + case 0x6E: + case 0x6F: + case 0x70: + case 0x71: + case 0x72: + case 0x73: + case 0x74: + case 0x75: + case 0x76: + case 0x77: + case 0x78: + case 0x79: + case 0x7A: + case 0x7B: + case 0x7C: + case 0x7D: + case 0x7E: + case 0x7F: + return sax->number_unsigned(static_cast(current)); + + // fixmap + case 0x80: + case 0x81: + case 0x82: + case 0x83: + case 0x84: + case 0x85: + case 0x86: + case 0x87: + case 0x88: + case 0x89: + case 0x8A: + case 0x8B: + case 0x8C: + case 0x8D: + case 0x8E: + case 0x8F: + return get_msgpack_object(conditional_static_cast(static_cast(current) & 0x0Fu)); + + // fixarray + case 0x90: + case 0x91: + case 0x92: + case 0x93: + case 0x94: + case 0x95: + case 0x96: + case 0x97: + case 0x98: + case 0x99: + case 0x9A: + case 0x9B: + case 0x9C: + case 0x9D: + case 0x9E: + case 0x9F: + return get_msgpack_array(conditional_static_cast(static_cast(current) & 0x0Fu)); + + // fixstr + case 0xA0: + case 0xA1: + case 0xA2: + case 0xA3: + case 0xA4: + case 0xA5: + case 0xA6: + case 0xA7: + case 0xA8: + case 0xA9: + case 0xAA: + case 0xAB: + case 0xAC: + case 0xAD: + case 0xAE: + case 0xAF: + case 0xB0: + case 0xB1: + case 0xB2: + case 0xB3: + case 0xB4: + case 0xB5: + case 0xB6: + case 0xB7: + case 0xB8: + case 0xB9: + case 0xBA: + case 0xBB: + case 0xBC: + case 0xBD: + case 0xBE: + case 0xBF: + case 0xD9: // str 8 + case 0xDA: // str 16 + case 0xDB: // str 32 + { + string_t s; + return get_msgpack_string(s) && sax->string(s); + } + + case 0xC0: // nil + return sax->null(); + + case 0xC2: // false + return sax->boolean(false); + + case 0xC3: // true + return sax->boolean(true); + + case 0xC4: // bin 8 + case 0xC5: // bin 16 + case 0xC6: // bin 32 + case 0xC7: // ext 8 + case 0xC8: // ext 16 + case 0xC9: // ext 32 + case 0xD4: // fixext 1 + case 0xD5: // fixext 2 + case 0xD6: // fixext 4 + case 0xD7: // fixext 8 + case 0xD8: // fixext 16 + { + binary_t b; + return get_msgpack_binary(b) && sax->binary(b); + } + + case 0xCA: // float 32 + { + float number{}; + return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast(number), ""); + } + + case 0xCB: // float 64 + { + double number{}; + return get_number(input_format_t::msgpack, number) && sax->number_float(static_cast(number), ""); + } + + case 0xCC: // uint 8 + { + std::uint8_t number{}; + return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number); + } + + case 0xCD: // uint 16 + { + std::uint16_t number{}; + return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number); + } + + case 0xCE: // uint 32 + { + std::uint32_t number{}; + return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number); + } + + case 0xCF: // uint 64 + { + std::uint64_t number{}; + return get_number(input_format_t::msgpack, number) && sax->number_unsigned(number); + } + + case 0xD0: // int 8 + { + std::int8_t number{}; + return get_number(input_format_t::msgpack, number) && sax->number_integer(number); + } + + case 0xD1: // int 16 + { + std::int16_t number{}; + return get_number(input_format_t::msgpack, number) && sax->number_integer(number); + } + + case 0xD2: // int 32 + { + std::int32_t number{}; + return get_number(input_format_t::msgpack, number) && sax->number_integer(number); + } + + case 0xD3: // int 64 + { + std::int64_t number{}; + return get_number(input_format_t::msgpack, number) && sax->number_integer(number); + } + + case 0xDC: // array 16 + { + std::uint16_t len{}; + return get_number(input_format_t::msgpack, len) && get_msgpack_array(static_cast(len)); + } + + case 0xDD: // array 32 + { + std::uint32_t len{}; + return get_number(input_format_t::msgpack, len) && get_msgpack_array(conditional_static_cast(len)); + } + + case 0xDE: // map 16 + { + std::uint16_t len{}; + return get_number(input_format_t::msgpack, len) && get_msgpack_object(static_cast(len)); + } + + case 0xDF: // map 32 + { + std::uint32_t len{}; + return get_number(input_format_t::msgpack, len) && get_msgpack_object(conditional_static_cast(len)); + } + + // negative fixint + case 0xE0: + case 0xE1: + case 0xE2: + case 0xE3: + case 0xE4: + case 0xE5: + case 0xE6: + case 0xE7: + case 0xE8: + case 0xE9: + case 0xEA: + case 0xEB: + case 0xEC: + case 0xED: + case 0xEE: + case 0xEF: + case 0xF0: + case 0xF1: + case 0xF2: + case 0xF3: + case 0xF4: + case 0xF5: + case 0xF6: + case 0xF7: + case 0xF8: + case 0xF9: + case 0xFA: + case 0xFB: + case 0xFC: + case 0xFD: + case 0xFE: + case 0xFF: + return sax->number_integer(static_cast(current)); + + default: // anything else + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format_t::msgpack, concat("invalid byte: 0x", last_token), "value"), nullptr)); + } + } + } + + /*! + @brief reads a MessagePack string + + This function first reads starting bytes to determine the expected + string length and then copies this number of bytes into a string. + + @param[out] result created string + + @return whether string creation completed + */ + bool get_msgpack_string(string_t& result) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format_t::msgpack, "string"))) + { + return false; + } + + switch (current) + { + // fixstr + case 0xA0: + case 0xA1: + case 0xA2: + case 0xA3: + case 0xA4: + case 0xA5: + case 0xA6: + case 0xA7: + case 0xA8: + case 0xA9: + case 0xAA: + case 0xAB: + case 0xAC: + case 0xAD: + case 0xAE: + case 0xAF: + case 0xB0: + case 0xB1: + case 0xB2: + case 0xB3: + case 0xB4: + case 0xB5: + case 0xB6: + case 0xB7: + case 0xB8: + case 0xB9: + case 0xBA: + case 0xBB: + case 0xBC: + case 0xBD: + case 0xBE: + case 0xBF: + { + return get_string(input_format_t::msgpack, static_cast(current) & 0x1Fu, result); + } + + case 0xD9: // str 8 + { + std::uint8_t len{}; + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + } + + case 0xDA: // str 16 + { + std::uint16_t len{}; + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + } + + case 0xDB: // str 32 + { + std::uint32_t len{}; + return get_number(input_format_t::msgpack, len) && get_string(input_format_t::msgpack, len, result); + } + + default: + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format_t::msgpack, concat("expected length specification (0xA0-0xBF, 0xD9-0xDB); last byte: 0x", last_token), "string"), nullptr)); + } + } + } + + /*! + @brief reads a MessagePack byte array + + This function first reads starting bytes to determine the expected + byte array length and then copies this number of bytes into a byte array. + + @param[out] result created byte array + + @return whether byte array creation completed + */ + bool get_msgpack_binary(binary_t& result) + { + // helper function to set the subtype + auto assign_and_return_true = [&result](std::int8_t subtype) + { + result.set_subtype(static_cast(subtype)); + return true; + }; + + switch (current) + { + case 0xC4: // bin 8 + { + std::uint8_t len{}; + return get_number(input_format_t::msgpack, len) && + get_binary(input_format_t::msgpack, len, result); + } + + case 0xC5: // bin 16 + { + std::uint16_t len{}; + return get_number(input_format_t::msgpack, len) && + get_binary(input_format_t::msgpack, len, result); + } + + case 0xC6: // bin 32 + { + std::uint32_t len{}; + return get_number(input_format_t::msgpack, len) && + get_binary(input_format_t::msgpack, len, result); + } + + case 0xC7: // ext 8 + { + std::uint8_t len{}; + std::int8_t subtype{}; + return get_number(input_format_t::msgpack, len) && + get_number(input_format_t::msgpack, subtype) && + get_binary(input_format_t::msgpack, len, result) && + assign_and_return_true(subtype); + } + + case 0xC8: // ext 16 + { + std::uint16_t len{}; + std::int8_t subtype{}; + return get_number(input_format_t::msgpack, len) && + get_number(input_format_t::msgpack, subtype) && + get_binary(input_format_t::msgpack, len, result) && + assign_and_return_true(subtype); + } + + case 0xC9: // ext 32 + { + std::uint32_t len{}; + std::int8_t subtype{}; + return get_number(input_format_t::msgpack, len) && + get_number(input_format_t::msgpack, subtype) && + get_binary(input_format_t::msgpack, len, result) && + assign_and_return_true(subtype); + } + + case 0xD4: // fixext 1 + { + std::int8_t subtype{}; + return get_number(input_format_t::msgpack, subtype) && + get_binary(input_format_t::msgpack, 1, result) && + assign_and_return_true(subtype); + } + + case 0xD5: // fixext 2 + { + std::int8_t subtype{}; + return get_number(input_format_t::msgpack, subtype) && + get_binary(input_format_t::msgpack, 2, result) && + assign_and_return_true(subtype); + } + + case 0xD6: // fixext 4 + { + std::int8_t subtype{}; + return get_number(input_format_t::msgpack, subtype) && + get_binary(input_format_t::msgpack, 4, result) && + assign_and_return_true(subtype); + } + + case 0xD7: // fixext 8 + { + std::int8_t subtype{}; + return get_number(input_format_t::msgpack, subtype) && + get_binary(input_format_t::msgpack, 8, result) && + assign_and_return_true(subtype); + } + + case 0xD8: // fixext 16 + { + std::int8_t subtype{}; + return get_number(input_format_t::msgpack, subtype) && + get_binary(input_format_t::msgpack, 16, result) && + assign_and_return_true(subtype); + } + + default: // LCOV_EXCL_LINE + return false; // LCOV_EXCL_LINE + } + } + + /*! + @param[in] len the length of the array + @return whether array creation completed + */ + bool get_msgpack_array(const std::size_t len) + { + if (JSON_HEDLEY_UNLIKELY(!sax->start_array(len))) + { + return false; + } + + for (std::size_t i = 0; i < len; ++i) + { + if (JSON_HEDLEY_UNLIKELY(!parse_msgpack_internal())) + { + return false; + } + } + + return sax->end_array(); + } + + /*! + @param[in] len the length of the object + @return whether object creation completed + */ + bool get_msgpack_object(const std::size_t len) + { + if (JSON_HEDLEY_UNLIKELY(!sax->start_object(len))) + { + return false; + } + + string_t key; + for (std::size_t i = 0; i < len; ++i) + { + get(); + if (JSON_HEDLEY_UNLIKELY(!get_msgpack_string(key) || !sax->key(key))) + { + return false; + } + + if (JSON_HEDLEY_UNLIKELY(!parse_msgpack_internal())) + { + return false; + } + key.clear(); + } + + return sax->end_object(); + } + + //////////// + // UBJSON // + //////////// + + /*! + @param[in] get_char whether a new character should be retrieved from the + input (true, default) or whether the last read + character should be considered instead + + @return whether a valid UBJSON value was passed to the SAX parser + */ + bool parse_ubjson_internal(const bool get_char = true) + { + return get_ubjson_value(get_char ? get_ignore_noop() : current); + } + + /*! + @brief reads a UBJSON string + + This function is either called after reading the 'S' byte explicitly + indicating a string, or in case of an object key where the 'S' byte can be + left out. + + @param[out] result created string + @param[in] get_char whether a new character should be retrieved from the + input (true, default) or whether the last read + character should be considered instead + + @return whether string creation completed + */ + bool get_ubjson_string(string_t& result, const bool get_char = true) + { + if (get_char) + { + get(); // TODO(niels): may we ignore N here? + } + + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "value"))) + { + return false; + } + + switch (current) + { + case 'U': + { + std::uint8_t len{}; + return get_number(input_format, len) && get_string(input_format, len, result); + } + + case 'i': + { + std::int8_t len{}; + return get_number(input_format, len) && get_string(input_format, len, result); + } + + case 'I': + { + std::int16_t len{}; + return get_number(input_format, len) && get_string(input_format, len, result); + } + + case 'l': + { + std::int32_t len{}; + return get_number(input_format, len) && get_string(input_format, len, result); + } + + case 'L': + { + std::int64_t len{}; + return get_number(input_format, len) && get_string(input_format, len, result); + } + + case 'u': + { + if (input_format != input_format_t::bjdata) + { + break; + } + std::uint16_t len{}; + return get_number(input_format, len) && get_string(input_format, len, result); + } + + case 'm': + { + if (input_format != input_format_t::bjdata) + { + break; + } + std::uint32_t len{}; + return get_number(input_format, len) && get_string(input_format, len, result); + } + + case 'M': + { + if (input_format != input_format_t::bjdata) + { + break; + } + std::uint64_t len{}; + return get_number(input_format, len) && get_string(input_format, len, result); + } + + default: + break; + } + auto last_token = get_token_string(); + std::string message; + + if (input_format != input_format_t::bjdata) + { + message = "expected length type specification (U, i, I, l, L); last byte: 0x" + last_token; + } + else + { + message = "expected length type specification (U, i, u, I, m, l, M, L); last byte: 0x" + last_token; + } + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, exception_message(input_format, message, "string"), nullptr)); + } + + /*! + @param[out] dim an integer vector storing the ND array dimensions + @return whether reading ND array size vector is successful + */ + bool get_ubjson_ndarray_size(std::vector& dim) + { + std::pair size_and_type; + size_t dimlen = 0; + bool no_ndarray = true; + + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_size_type(size_and_type, no_ndarray))) + { + return false; + } + + if (size_and_type.first != npos) + { + if (size_and_type.second != 0) + { + if (size_and_type.second != 'N') + { + for (std::size_t i = 0; i < size_and_type.first; ++i) + { + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_size_value(dimlen, no_ndarray, size_and_type.second))) + { + return false; + } + dim.push_back(dimlen); + } + } + } + else + { + for (std::size_t i = 0; i < size_and_type.first; ++i) + { + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_size_value(dimlen, no_ndarray))) + { + return false; + } + dim.push_back(dimlen); + } + } + } + else + { + while (current != ']') + { + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_size_value(dimlen, no_ndarray, current))) + { + return false; + } + dim.push_back(dimlen); + get_ignore_noop(); + } + } + return true; + } + + /*! + @param[out] result determined size + @param[in,out] is_ndarray for input, `true` means already inside an ndarray vector + or ndarray dimension is not allowed; `false` means ndarray + is allowed; for output, `true` means an ndarray is found; + is_ndarray can only return `true` when its initial value + is `false` + @param[in] prefix type marker if already read, otherwise set to 0 + + @return whether size determination completed + */ + bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0) + { + if (prefix == 0) + { + prefix = get_ignore_noop(); + } + + switch (prefix) + { + case 'U': + { + std::uint8_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + result = static_cast(number); + return true; + } + + case 'i': + { + std::int8_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + if (number < 0) + { + return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, + exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); + } + result = static_cast(number); // NOLINT(bugprone-signed-char-misuse,cert-str34-c): number is not a char + return true; + } + + case 'I': + { + std::int16_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + if (number < 0) + { + return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, + exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); + } + result = static_cast(number); + return true; + } + + case 'l': + { + std::int32_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + if (number < 0) + { + return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, + exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); + } + result = static_cast(number); + return true; + } + + case 'L': + { + std::int64_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + if (number < 0) + { + return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, + exception_message(input_format, "count in an optimized container must be positive", "size"), nullptr)); + } + if (!value_in_range_of(number)) + { + return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, + exception_message(input_format, "integer value overflow", "size"), nullptr)); + } + result = static_cast(number); + return true; + } + + case 'u': + { + if (input_format != input_format_t::bjdata) + { + break; + } + std::uint16_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + result = static_cast(number); + return true; + } + + case 'm': + { + if (input_format != input_format_t::bjdata) + { + break; + } + std::uint32_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + result = conditional_static_cast(number); + return true; + } + + case 'M': + { + if (input_format != input_format_t::bjdata) + { + break; + } + std::uint64_t number{}; + if (JSON_HEDLEY_UNLIKELY(!get_number(input_format, number))) + { + return false; + } + if (!value_in_range_of(number)) + { + return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, + exception_message(input_format, "integer value overflow", "size"), nullptr)); + } + result = detail::conditional_static_cast(number); + return true; + } + + case '[': + { + if (input_format != input_format_t::bjdata) + { + break; + } + if (is_ndarray) // ndarray dimensional vector can only contain integers and cannot embed another array + { + return sax->parse_error(chars_read, get_token_string(), parse_error::create(113, chars_read, exception_message(input_format, "ndarray dimensional vector is not allowed", "size"), nullptr)); + } + std::vector dim; + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_ndarray_size(dim))) + { + return false; + } + if (dim.size() == 1 || (dim.size() == 2 && dim.at(0) == 1)) // return normal array size if 1D row vector + { + result = dim.at(dim.size() - 1); + return true; + } + if (!dim.empty()) // if ndarray, convert to an object in JData annotated array format + { + for (auto i : dim) // test if any dimension in an ndarray is 0, if so, return a 1D empty container + { + if ( i == 0 ) + { + result = 0; + return true; + } + } + + string_t key = "_ArraySize_"; + if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3) || !sax->key(key) || !sax->start_array(dim.size()))) + { + return false; + } + result = 1; + for (auto i : dim) + { + // Pre-multiplication overflow check: if i > 0 and result > SIZE_MAX/i, then result*i would overflow. + // This check must happen before multiplication since overflow detection after the fact is unreliable + // as modular arithmetic can produce any value, not just 0 or SIZE_MAX. + if (JSON_HEDLEY_UNLIKELY(i > 0 && result > (std::numeric_limits::max)() / i)) + { + return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr)); + } + result *= i; + // Additional post-multiplication check to catch any edge cases the pre-check might miss + if (result == 0 || result == npos) + { + return sax->parse_error(chars_read, get_token_string(), out_of_range::create(408, exception_message(input_format, "excessive ndarray size caused overflow", "size"), nullptr)); + } + if (JSON_HEDLEY_UNLIKELY(!sax->number_unsigned(static_cast(i)))) + { + return false; + } + } + is_ndarray = true; + return sax->end_array(); + } + result = 0; + return true; + } + + default: + break; + } + auto last_token = get_token_string(); + std::string message; + + if (input_format != input_format_t::bjdata) + { + message = "expected length type specification (U, i, I, l, L) after '#'; last byte: 0x" + last_token; + } + else + { + message = "expected length type specification (U, i, u, I, m, l, M, L) after '#'; last byte: 0x" + last_token; + } + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, exception_message(input_format, message, "size"), nullptr)); + } + + /*! + @brief determine the type and size for a container + + In the optimized UBJSON format, a type and a size can be provided to allow + for a more compact representation. + + @param[out] result pair of the size and the type + @param[in] inside_ndarray whether the parser is parsing an ND array dimensional vector + + @return whether pair creation completed + */ + bool get_ubjson_size_type(std::pair& result, bool inside_ndarray = false) + { + result.first = npos; // size + result.second = 0; // type + bool is_ndarray = false; + + get_ignore_noop(); + + if (current == '$') + { + result.second = get(); // must not ignore 'N', because 'N' maybe the type + if (input_format == input_format_t::bjdata + && JSON_HEDLEY_UNLIKELY(std::binary_search(bjd_optimized_type_markers.begin(), bjd_optimized_type_markers.end(), result.second))) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format, concat("marker 0x", last_token, " is not a permitted optimized array type"), "type"), nullptr)); + } + + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "type"))) + { + return false; + } + + get_ignore_noop(); + if (JSON_HEDLEY_UNLIKELY(current != '#')) + { + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "value"))) + { + return false; + } + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr)); + } + + const bool is_error = get_ubjson_size_value(result.first, is_ndarray); + if (input_format == input_format_t::bjdata && is_ndarray) + { + if (inside_ndarray) + { + return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read, + exception_message(input_format, "ndarray can not be recursive", "size"), nullptr)); + } + result.second |= (1 << 8); // use bit 8 to indicate ndarray, all UBJSON and BJData markers should be ASCII letters + } + return is_error; + } + + if (current == '#') + { + const bool is_error = get_ubjson_size_value(result.first, is_ndarray); + if (input_format == input_format_t::bjdata && is_ndarray) + { + return sax->parse_error(chars_read, get_token_string(), parse_error::create(112, chars_read, + exception_message(input_format, "ndarray requires both type and size", "size"), nullptr)); + } + return is_error; + } + + return true; + } + + /*! + @param prefix the previously read or set type prefix + @return whether value creation completed + */ + bool get_ubjson_value(const char_int_type prefix) + { + switch (prefix) + { + case char_traits::eof(): // EOF + return unexpect_eof(input_format, "value"); + + case 'T': // true + return sax->boolean(true); + case 'F': // false + return sax->boolean(false); + + case 'Z': // null + return sax->null(); + + case 'B': // byte + { + if (input_format != input_format_t::bjdata) + { + break; + } + std::uint8_t number{}; + return get_number(input_format, number) && sax->number_unsigned(number); + } + + case 'U': + { + std::uint8_t number{}; + return get_number(input_format, number) && sax->number_unsigned(number); + } + + case 'i': + { + std::int8_t number{}; + return get_number(input_format, number) && sax->number_integer(number); + } + + case 'I': + { + std::int16_t number{}; + return get_number(input_format, number) && sax->number_integer(number); + } + + case 'l': + { + std::int32_t number{}; + return get_number(input_format, number) && sax->number_integer(number); + } + + case 'L': + { + std::int64_t number{}; + return get_number(input_format, number) && sax->number_integer(number); + } + + case 'u': + { + if (input_format != input_format_t::bjdata) + { + break; + } + std::uint16_t number{}; + return get_number(input_format, number) && sax->number_unsigned(number); + } + + case 'm': + { + if (input_format != input_format_t::bjdata) + { + break; + } + std::uint32_t number{}; + return get_number(input_format, number) && sax->number_unsigned(number); + } + + case 'M': + { + if (input_format != input_format_t::bjdata) + { + break; + } + std::uint64_t number{}; + return get_number(input_format, number) && sax->number_unsigned(number); + } + + case 'h': + { + if (input_format != input_format_t::bjdata) + { + break; + } + const auto byte1_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) + { + return false; + } + const auto byte2_raw = get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) + { + return false; + } + + const auto byte1 = static_cast(byte1_raw); + const auto byte2 = static_cast(byte2_raw); + + // Code from RFC 7049, Appendix D, Figure 3: + // As half-precision floating-point numbers were only added + // to IEEE 754 in 2008, today's programming platforms often + // still only have limited support for them. It is very + // easy to include at least decoding support for them even + // without such support. An example of a small decoder for + // half-precision floating-point numbers in the C language + // is shown in Fig. 3. + const auto half = static_cast((byte2 << 8u) + byte1); + const double val = [&half] + { + const int exp = (half >> 10u) & 0x1Fu; + const unsigned int mant = half & 0x3FFu; + JSON_ASSERT(0 <= exp&& exp <= 32); + JSON_ASSERT(mant <= 1024); + switch (exp) + { + case 0: + return std::ldexp(mant, -24); + case 31: + return (mant == 0) + ? std::numeric_limits::infinity() + : std::numeric_limits::quiet_NaN(); + default: + return std::ldexp(mant + 1024, exp - 25); + } + }(); + return sax->number_float((half & 0x8000u) != 0 + ? static_cast(-val) + : static_cast(val), ""); + } + + case 'd': + { + float number{}; + return get_number(input_format, number) && sax->number_float(static_cast(number), ""); + } + + case 'D': + { + double number{}; + return get_number(input_format, number) && sax->number_float(static_cast(number), ""); + } + + case 'H': + { + return get_ubjson_high_precision_number(); + } + + case 'C': // char + { + get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "char"))) + { + return false; + } + if (JSON_HEDLEY_UNLIKELY(current > 127)) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(113, chars_read, + exception_message(input_format, concat("byte after 'C' must be in range 0x00..0x7F; last byte: 0x", last_token), "char"), nullptr)); + } + string_t s(1, static_cast(current)); + return sax->string(s); + } + + case 'S': // string + { + string_t s; + return get_ubjson_string(s) && sax->string(s); + } + + case '[': // array + return get_ubjson_array(); + + case '{': // object + return get_ubjson_object(); + + default: // anything else + break; + } + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, exception_message(input_format, "invalid byte: 0x" + last_token, "value"), nullptr)); + } + + /*! + @return whether array creation completed + */ + bool get_ubjson_array() + { + std::pair size_and_type; + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_size_type(size_and_type))) + { + return false; + } + + // if bit-8 of size_and_type.second is set to 1, encode bjdata ndarray as an object in JData annotated array format (https://github.com/NeuroJSON/jdata): + // {"_ArrayType_" : "typeid", "_ArraySize_" : [n1, n2, ...], "_ArrayData_" : [v1, v2, ...]} + + if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0) + { + size_and_type.second &= ~(static_cast(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker + auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t) + { + return p.first < t; + }); + string_t key = "_ArrayType_"; + if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second)) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr)); + } + + string_t type = it->second; // sax->string() takes a reference + if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type))) + { + return false; + } + + if (size_and_type.second == 'C' || size_and_type.second == 'B') + { + size_and_type.second = 'U'; + } + + key = "_ArrayData_"; + if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) )) + { + return false; + } + + for (std::size_t i = 0; i < size_and_type.first; ++i) + { + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(size_and_type.second))) + { + return false; + } + } + + return (sax->end_array() && sax->end_object()); + } + + // If BJData type marker is 'B' decode as binary + if (input_format == input_format_t::bjdata && size_and_type.first != npos && size_and_type.second == 'B') + { + binary_t result; + return get_binary(input_format, size_and_type.first, result) && sax->binary(result); + } + + if (size_and_type.first != npos) + { + if (JSON_HEDLEY_UNLIKELY(!sax->start_array(size_and_type.first))) + { + return false; + } + + if (size_and_type.second != 0) + { + if (size_and_type.second != 'N') + { + for (std::size_t i = 0; i < size_and_type.first; ++i) + { + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(size_and_type.second))) + { + return false; + } + } + } + } + else + { + for (std::size_t i = 0; i < size_and_type.first; ++i) + { + if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal())) + { + return false; + } + } + } + } + else + { + if (JSON_HEDLEY_UNLIKELY(!sax->start_array(detail::unknown_size()))) + { + return false; + } + + while (current != ']') + { + if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal(false))) + { + return false; + } + get_ignore_noop(); + } + } + + return sax->end_array(); + } + + /*! + @return whether object creation completed + */ + bool get_ubjson_object() + { + std::pair size_and_type; + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_size_type(size_and_type))) + { + return false; + } + + // do not accept ND-array size in objects in BJData + if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format, "BJData object does not support ND-array size in optimized format", "object"), nullptr)); + } + + string_t key; + if (size_and_type.first != npos) + { + if (JSON_HEDLEY_UNLIKELY(!sax->start_object(size_and_type.first))) + { + return false; + } + + if (size_and_type.second != 0) + { + for (std::size_t i = 0; i < size_and_type.first; ++i) + { + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) + { + return false; + } + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_value(size_and_type.second))) + { + return false; + } + key.clear(); + } + } + else + { + for (std::size_t i = 0; i < size_and_type.first; ++i) + { + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key) || !sax->key(key))) + { + return false; + } + if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal())) + { + return false; + } + key.clear(); + } + } + } + else + { + if (JSON_HEDLEY_UNLIKELY(!sax->start_object(detail::unknown_size()))) + { + return false; + } + + while (current != '}') + { + if (JSON_HEDLEY_UNLIKELY(!get_ubjson_string(key, false) || !sax->key(key))) + { + return false; + } + if (JSON_HEDLEY_UNLIKELY(!parse_ubjson_internal())) + { + return false; + } + get_ignore_noop(); + key.clear(); + } + } + + return sax->end_object(); + } + + // Note, no reader for UBJSON binary types is implemented because they do + // not exist + + bool get_ubjson_high_precision_number() + { + // get the size of the following number string + std::size_t size{}; + bool no_ndarray = true; + auto res = get_ubjson_size_value(size, no_ndarray); + if (JSON_HEDLEY_UNLIKELY(!res)) + { + return res; + } + + // get number string + std::vector number_vector; + for (std::size_t i = 0; i < size; ++i) + { + get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "number"))) + { + return false; + } + number_vector.push_back(static_cast(current)); + } + + // parse number string + using ia_type = decltype(detail::input_adapter(number_vector)); + auto number_lexer = detail::lexer(detail::input_adapter(number_vector), false); + const auto result_number = number_lexer.scan(); + const auto number_string = number_lexer.get_token_string(); + const auto result_remainder = number_lexer.scan(); + + using token_type = typename detail::lexer_base::token_type; + + if (JSON_HEDLEY_UNLIKELY(result_remainder != token_type::end_of_input)) + { + return sax->parse_error(chars_read, number_string, parse_error::create(115, chars_read, + exception_message(input_format, concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr)); + } + + switch (result_number) + { + case token_type::value_integer: + return sax->number_integer(number_lexer.get_number_integer()); + case token_type::value_unsigned: + return sax->number_unsigned(number_lexer.get_number_unsigned()); + case token_type::value_float: + return sax->number_float(number_lexer.get_number_float(), std::move(number_string)); + case token_type::uninitialized: + case token_type::literal_true: + case token_type::literal_false: + case token_type::literal_null: + case token_type::value_string: + case token_type::begin_array: + case token_type::begin_object: + case token_type::end_array: + case token_type::end_object: + case token_type::name_separator: + case token_type::value_separator: + case token_type::parse_error: + case token_type::end_of_input: + case token_type::literal_or_value: + default: + return sax->parse_error(chars_read, number_string, parse_error::create(115, chars_read, + exception_message(input_format, concat("invalid number text: ", number_lexer.get_token_string()), "high-precision number"), nullptr)); + } + } + + /////////////////////// + // Utility functions // + /////////////////////// + + /*! + @brief get next character from the input + + This function provides the interface to the used input adapter. It does + not throw in case the input reached EOF, but returns a -'ve valued + `char_traits::eof()` in that case. + + @return character read from the input + */ + char_int_type get() + { + ++chars_read; + return current = ia.get_character(); + } + + /*! + @brief get_to read into a primitive type + + This function provides the interface to the used input adapter. It does + not throw in case the input reached EOF, but returns false instead + + @return bool, whether the read was successful + */ + template + bool get_to(T& dest, const input_format_t format, const char* context) + { + auto new_chars_read = ia.get_elements(&dest); + chars_read += new_chars_read; + if (JSON_HEDLEY_UNLIKELY(new_chars_read < sizeof(T))) + { + // in case of failure, advance position by 1 to report the failing location + ++chars_read; + sax->parse_error(chars_read, "", parse_error::create(110, chars_read, exception_message(format, "unexpected end of input", context), nullptr)); + return false; + } + return true; + } + + /*! + @return character read from the input after ignoring all 'N' entries + */ + char_int_type get_ignore_noop() + { + do + { + get(); + } + while (current == 'N'); + + return current; + } + + template + static void byte_swap(NumberType& number) + { + constexpr std::size_t sz = sizeof(number); +#ifdef __cpp_lib_byteswap + if constexpr (sz == 1) + { + return; + } + else if constexpr(std::is_integral_v) + { + number = std::byteswap(number); + return; + } + else + { +#endif + auto* ptr = reinterpret_cast(&number); + for (std::size_t i = 0; i < sz / 2; ++i) + { + std::swap(ptr[i], ptr[sz - i - 1]); + } +#ifdef __cpp_lib_byteswap + } +#endif + } + + /* + @brief read a number from the input + + @tparam NumberType the type of the number + @param[in] format the current format (for diagnostics) + @param[out] result number of type @a NumberType + + @return whether conversion completed + + @note This function needs to respect the system's endianness, because + bytes in CBOR, MessagePack, and UBJSON are stored in network order + (big endian) and therefore need reordering on little endian systems. + On the other hand, BSON and BJData use little endian and should reorder + on big endian systems. + */ + template + bool get_number(const input_format_t format, NumberType& result) + { + // read in the original format + + if (JSON_HEDLEY_UNLIKELY(!get_to(result, format, "number"))) + { + return false; + } + if (is_little_endian != (InputIsLittleEndian || format == input_format_t::bjdata)) + { + byte_swap(result); + } + return true; + } + + /*! + @brief create a string by reading characters from the input + + @tparam NumberType the type of the number + @param[in] format the current format (for diagnostics) + @param[in] len number of characters to read + @param[out] result string created by reading @a len bytes + + @return whether string creation completed + + @note We can not reserve @a len bytes for the result, because @a len + may be too large. Usually, @ref unexpect_eof() detects the end of + the input before we run out of string memory. + */ + template + bool get_string(const input_format_t format, + const NumberType len, + string_t& result) + { + bool success = true; + for (NumberType i = 0; i < len; i++) + { + get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "string"))) + { + success = false; + break; + } + result.push_back(static_cast(current)); + } + return success; + } + + /*! + @brief create a byte array by reading bytes from the input + + @tparam NumberType the type of the number + @param[in] format the current format (for diagnostics) + @param[in] len number of bytes to read + @param[out] result byte array created by reading @a len bytes + + @return whether byte array creation completed + + @note We can not reserve @a len bytes for the result, because @a len + may be too large. Usually, @ref unexpect_eof() detects the end of + the input before we run out of memory. + */ + template + bool get_binary(const input_format_t format, + const NumberType len, + binary_t& result) + { + bool success = true; + for (NumberType i = 0; i < len; i++) + { + get(); + if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(format, "binary"))) + { + success = false; + break; + } + result.push_back(static_cast(current)); + } + return success; + } + + /*! + @param[in] format the current format (for diagnostics) + @param[in] context further context information (for diagnostics) + @return whether the last read character is not EOF + */ + JSON_HEDLEY_NON_NULL(3) + bool unexpect_eof(const input_format_t format, const char* context) const + { + if (JSON_HEDLEY_UNLIKELY(current == char_traits::eof())) + { + return sax->parse_error(chars_read, "", + parse_error::create(110, chars_read, exception_message(format, "unexpected end of input", context), nullptr)); + } + return true; + } + + /*! + @return a string representation of the last read byte + */ + std::string get_token_string() const + { + std::array cr{{}}; + static_cast((std::snprintf)(cr.data(), cr.size(), "%.2hhX", static_cast(current))); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) + return std::string{cr.data()}; + } + + /*! + @param[in] format the current format + @param[in] detail a detailed error message + @param[in] context further context information + @return a message string to use in the parse_error exceptions + */ + std::string exception_message(const input_format_t format, + const std::string& detail, + const std::string& context) const + { + std::string error_msg = "syntax error while parsing "; + + switch (format) + { + case input_format_t::cbor: + error_msg += "CBOR"; + break; + + case input_format_t::msgpack: + error_msg += "MessagePack"; + break; + + case input_format_t::ubjson: + error_msg += "UBJSON"; + break; + + case input_format_t::bson: + error_msg += "BSON"; + break; + + case input_format_t::bjdata: + error_msg += "BJData"; + break; + + case input_format_t::json: // LCOV_EXCL_LINE + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + } + + return concat(error_msg, ' ', context, ": ", detail); + } + + private: + static JSON_INLINE_VARIABLE constexpr std::size_t npos = detail::unknown_size(); + + /// input adapter + InputAdapterType ia; + + /// the current character + char_int_type current = char_traits::eof(); + + /// the number of characters read + std::size_t chars_read = 0; + + /// whether we can assume little endianness + const bool is_little_endian = little_endianness(); + + /// input format + const input_format_t input_format = input_format_t::json; + + /// the SAX parser + json_sax_t* sax = nullptr; + + // excluded markers in bjdata optimized type +#define JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ \ + make_array('F', 'H', 'N', 'S', 'T', 'Z', '[', '{') + +#define JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ \ + make_array( \ + bjd_type{'B', "byte"}, \ + bjd_type{'C', "char"}, \ + bjd_type{'D', "double"}, \ + bjd_type{'I', "int16"}, \ + bjd_type{'L', "int64"}, \ + bjd_type{'M', "uint64"}, \ + bjd_type{'U', "uint8"}, \ + bjd_type{'d', "single"}, \ + bjd_type{'i', "int8"}, \ + bjd_type{'l', "int32"}, \ + bjd_type{'m', "uint32"}, \ + bjd_type{'u', "uint16"}) + + JSON_PRIVATE_UNLESS_TESTED: + // lookup tables + // NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes) + const decltype(JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_) bjd_optimized_type_markers = + JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_; + + using bjd_type = std::pair; + // NOLINTNEXTLINE(cppcoreguidelines-non-private-member-variables-in-classes) + const decltype(JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_) bjd_types_map = + JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_; + +#undef JSON_BINARY_READER_MAKE_BJD_OPTIMIZED_TYPE_MARKERS_ +#undef JSON_BINARY_READER_MAKE_BJD_TYPES_MAP_ +}; + +#ifndef JSON_HAS_CPP_17 + template + constexpr std::size_t binary_reader::npos; +#endif + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/input/input_adapters.hpp b/src/detail/include/nlohmann/detail/input/input_adapters.hpp new file mode 100644 index 000000000..b9ee65e04 --- /dev/null +++ b/src/detail/include/nlohmann/detail/input/input_adapters.hpp @@ -0,0 +1,549 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // size_t +#include // strlen +#include // begin, end, iterator_traits, random_access_iterator_tag, distance, next +#include // shared_ptr, make_shared, addressof +#include // accumulate +#include // string, char_traits +#include // enable_if, is_base_of, is_pointer, is_integral, remove_pointer +#include // pair, declval + +#ifndef JSON_NO_IO + #include // FILE * + #include // istream +#endif // JSON_NO_IO + +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// the supported input formats +enum class input_format_t { json, cbor, msgpack, ubjson, bson, bjdata }; + +//////////////////// +// input adapters // +//////////////////// + +#ifndef JSON_NO_IO +/*! +Input adapter for stdio file access. This adapter read only 1 byte and do not use any + buffer. This adapter is a very low level adapter. +*/ +class file_input_adapter +{ + public: + using char_type = char; + + JSON_HEDLEY_NON_NULL(2) + explicit file_input_adapter(std::FILE* f) noexcept + : m_file(f) + { + JSON_ASSERT(m_file != nullptr); + } + + // make class move-only + file_input_adapter(const file_input_adapter&) = delete; + file_input_adapter(file_input_adapter&&) noexcept = default; + file_input_adapter& operator=(const file_input_adapter&) = delete; + file_input_adapter& operator=(file_input_adapter&&) = delete; + ~file_input_adapter() = default; + + std::char_traits::int_type get_character() noexcept + { + return std::fgetc(m_file); + } + + // returns the number of characters successfully read + template + std::size_t get_elements(T* dest, std::size_t count = 1) + { + return fread(dest, 1, sizeof(T) * count, m_file); + } + + private: + /// the file pointer to read from + std::FILE* m_file; +}; + +/*! +Input adapter for a (caching) istream. Ignores a UFT Byte Order Mark at +beginning of input. Does not support changing the underlying std::streambuf +in mid-input. Maintains underlying std::istream and std::streambuf to support +subsequent use of standard std::istream operations to process any input +characters following those used in parsing the JSON input. Clears the +std::istream flags; any input errors (e.g., EOF) will be detected by the first +subsequent call for input from the std::istream. +*/ +class input_stream_adapter +{ + public: + using char_type = char; + + ~input_stream_adapter() + { + // clear stream flags; we use underlying streambuf I/O, do not + // maintain ifstream flags, except eof + if (is != nullptr) + { + is->clear(is->rdstate() & std::ios::eofbit); + } + } + + explicit input_stream_adapter(std::istream& i) + : is(&i), sb(i.rdbuf()) + {} + + // deleted because of pointer members + input_stream_adapter(const input_stream_adapter&) = delete; + input_stream_adapter& operator=(input_stream_adapter&) = delete; + input_stream_adapter& operator=(input_stream_adapter&&) = delete; + + input_stream_adapter(input_stream_adapter&& rhs) noexcept + : is(rhs.is), sb(rhs.sb) + { + rhs.is = nullptr; + rhs.sb = nullptr; + } + + // std::istream/std::streambuf use std::char_traits::to_int_type, to + // ensure that std::char_traits::eof() and the character 0xFF do not + // end up as the same value, e.g., 0xFFFFFFFF. + std::char_traits::int_type get_character() + { + auto res = sb->sbumpc(); + // set eof manually, as we don't use the istream interface. + if (JSON_HEDLEY_UNLIKELY(res == std::char_traits::eof())) + { + is->clear(is->rdstate() | std::ios::eofbit); + } + return res; + } + + template + std::size_t get_elements(T* dest, std::size_t count = 1) + { + auto res = static_cast(sb->sgetn(reinterpret_cast(dest), static_cast(count * sizeof(T)))); + if (JSON_HEDLEY_UNLIKELY(res < count * sizeof(T))) + { + is->clear(is->rdstate() | std::ios::eofbit); + } + return res; + } + + private: + /// the associated input stream + std::istream* is = nullptr; + std::streambuf* sb = nullptr; +}; +#endif // JSON_NO_IO + +// General-purpose iterator-based adapter. It might not be as fast as +// theoretically possible for some containers, but it is extremely versatile. +template +class iterator_input_adapter +{ + public: + using char_type = typename std::iterator_traits::value_type; + + iterator_input_adapter(IteratorType first, IteratorType last) + : current(std::move(first)), end(std::move(last)) + {} + + typename char_traits::int_type get_character() + { + if (JSON_HEDLEY_LIKELY(current != end)) + { + auto result = char_traits::to_int_type(*current); + std::advance(current, 1); + return result; + } + + return char_traits::eof(); + } + + // for general iterators, we cannot really do something better than falling back to processing the range one-by-one + template + std::size_t get_elements(T* dest, std::size_t count = 1) + { + auto* ptr = reinterpret_cast(dest); + for (std::size_t read_index = 0; read_index < count * sizeof(T); ++read_index) + { + if (JSON_HEDLEY_LIKELY(current != end)) + { + ptr[read_index] = static_cast(*current); + std::advance(current, 1); + } + else + { + return read_index; + } + } + return count * sizeof(T); + } + + private: + IteratorType current; + IteratorType end; + + template + friend struct wide_string_input_helper; + + bool empty() const + { + return current == end; + } +}; + +template +struct wide_string_input_helper; + +template +struct wide_string_input_helper +{ + // UTF-32 + static void fill_buffer(BaseInputAdapter& input, + std::array::int_type, 4>& utf8_bytes, + size_t& utf8_bytes_index, + size_t& utf8_bytes_filled) + { + utf8_bytes_index = 0; + + if (JSON_HEDLEY_UNLIKELY(input.empty())) + { + utf8_bytes[0] = std::char_traits::eof(); + utf8_bytes_filled = 1; + } + else + { + // get the current character + const auto wc = input.get_character(); + + // UTF-32 to UTF-8 encoding + if (wc < 0x80) + { + utf8_bytes[0] = static_cast::int_type>(wc); + utf8_bytes_filled = 1; + } + else if (wc <= 0x7FF) + { + utf8_bytes[0] = static_cast::int_type>(0xC0u | ((static_cast(wc) >> 6u) & 0x1Fu)); + utf8_bytes[1] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); + utf8_bytes_filled = 2; + } + else if (wc <= 0xFFFF) + { + utf8_bytes[0] = static_cast::int_type>(0xE0u | ((static_cast(wc) >> 12u) & 0x0Fu)); + utf8_bytes[1] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 6u) & 0x3Fu)); + utf8_bytes[2] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); + utf8_bytes_filled = 3; + } + else if (wc <= 0x10FFFF) + { + utf8_bytes[0] = static_cast::int_type>(0xF0u | ((static_cast(wc) >> 18u) & 0x07u)); + utf8_bytes[1] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 12u) & 0x3Fu)); + utf8_bytes[2] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 6u) & 0x3Fu)); + utf8_bytes[3] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); + utf8_bytes_filled = 4; + } + else + { + // unknown character + utf8_bytes[0] = static_cast::int_type>(wc); + utf8_bytes_filled = 1; + } + } + } +}; + +template +struct wide_string_input_helper +{ + // UTF-16 + static void fill_buffer(BaseInputAdapter& input, + std::array::int_type, 4>& utf8_bytes, + size_t& utf8_bytes_index, + size_t& utf8_bytes_filled) + { + utf8_bytes_index = 0; + + if (JSON_HEDLEY_UNLIKELY(input.empty())) + { + utf8_bytes[0] = std::char_traits::eof(); + utf8_bytes_filled = 1; + } + else + { + // get the current character + const auto wc = input.get_character(); + + // UTF-16 to UTF-8 encoding + if (wc < 0x80) + { + utf8_bytes[0] = static_cast::int_type>(wc); + utf8_bytes_filled = 1; + } + else if (wc <= 0x7FF) + { + utf8_bytes[0] = static_cast::int_type>(0xC0u | ((static_cast(wc) >> 6u))); + utf8_bytes[1] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); + utf8_bytes_filled = 2; + } + else if (0xD800 > wc || wc >= 0xE000) + { + utf8_bytes[0] = static_cast::int_type>(0xE0u | ((static_cast(wc) >> 12u))); + utf8_bytes[1] = static_cast::int_type>(0x80u | ((static_cast(wc) >> 6u) & 0x3Fu)); + utf8_bytes[2] = static_cast::int_type>(0x80u | (static_cast(wc) & 0x3Fu)); + utf8_bytes_filled = 3; + } + else + { + if (JSON_HEDLEY_UNLIKELY(!input.empty())) + { + const auto wc2 = static_cast(input.get_character()); + const auto charcode = 0x10000u + (((static_cast(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu)); + utf8_bytes[0] = static_cast::int_type>(0xF0u | (charcode >> 18u)); + utf8_bytes[1] = static_cast::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu)); + utf8_bytes[2] = static_cast::int_type>(0x80u | ((charcode >> 6u) & 0x3Fu)); + utf8_bytes[3] = static_cast::int_type>(0x80u | (charcode & 0x3Fu)); + utf8_bytes_filled = 4; + } + else + { + utf8_bytes[0] = static_cast::int_type>(wc); + utf8_bytes_filled = 1; + } + } + } + } +}; + +// Wraps another input adapter to convert wide character types into individual bytes. +template +class wide_string_input_adapter +{ + public: + using char_type = char; + + wide_string_input_adapter(BaseInputAdapter base) + : base_adapter(base) {} + + typename std::char_traits::int_type get_character() noexcept + { + // check if the buffer needs to be filled + if (utf8_bytes_index == utf8_bytes_filled) + { + fill_buffer(); + + JSON_ASSERT(utf8_bytes_filled > 0); + JSON_ASSERT(utf8_bytes_index == 0); + } + + // use buffer + JSON_ASSERT(utf8_bytes_filled > 0); + JSON_ASSERT(utf8_bytes_index < utf8_bytes_filled); + return utf8_bytes[utf8_bytes_index++]; + } + + // parsing binary with wchar doesn't make sense, but since the parsing mode can be runtime, we need something here + template + std::size_t get_elements(T* /*dest*/, std::size_t /*count*/ = 1) + { + JSON_THROW(parse_error::create(112, 1, "wide string type cannot be interpreted as binary data", nullptr)); + } + + private: + BaseInputAdapter base_adapter; + + template + void fill_buffer() + { + wide_string_input_helper::fill_buffer(base_adapter, utf8_bytes, utf8_bytes_index, utf8_bytes_filled); + } + + /// a buffer for UTF-8 bytes + std::array::int_type, 4> utf8_bytes = {{0, 0, 0, 0}}; + + /// index to the utf8_codes array for the next valid byte + std::size_t utf8_bytes_index = 0; + /// number of valid bytes in the utf8_codes array + std::size_t utf8_bytes_filled = 0; +}; + +template +struct iterator_input_adapter_factory +{ + using iterator_type = IteratorType; + using char_type = typename std::iterator_traits::value_type; + using adapter_type = iterator_input_adapter; + + static adapter_type create(IteratorType first, IteratorType last) + { + return adapter_type(std::move(first), std::move(last)); + } +}; + +template +struct is_iterator_of_multibyte +{ + using value_type = typename std::iterator_traits::value_type; + enum + { + value = sizeof(value_type) > 1 + }; +}; + +template +struct iterator_input_adapter_factory::value>> +{ + using iterator_type = IteratorType; + using char_type = typename std::iterator_traits::value_type; + using base_adapter_type = iterator_input_adapter; + using adapter_type = wide_string_input_adapter; + + static adapter_type create(IteratorType first, IteratorType last) + { + return adapter_type(base_adapter_type(std::move(first), std::move(last))); + } +}; + +// General purpose iterator-based input +template +typename iterator_input_adapter_factory::adapter_type input_adapter(IteratorType first, IteratorType last) +{ + using factory_type = iterator_input_adapter_factory; + return factory_type::create(first, last); +} + +// Convenience shorthand from container to iterator +// Enables ADL on begin(container) and end(container) +// Encloses the using declarations in namespace for not to leak them to outside scope + +namespace container_input_adapter_factory_impl +{ + +using std::begin; +using std::end; + +template +struct container_input_adapter_factory {}; + +template +struct container_input_adapter_factory< ContainerType, + void_t()), end(std::declval()))>> + { + using adapter_type = decltype(input_adapter(begin(std::declval()), end(std::declval()))); + + static adapter_type create(const ContainerType& container) +{ + return input_adapter(begin(container), end(container)); +} + }; + +} // namespace container_input_adapter_factory_impl + +template +typename container_input_adapter_factory_impl::container_input_adapter_factory::adapter_type input_adapter(const ContainerType& container) +{ + return container_input_adapter_factory_impl::container_input_adapter_factory::create(container); +} + +// specialization for std::string +using string_input_adapter_type = decltype(input_adapter(std::declval())); + +#ifndef JSON_NO_IO +// Special cases with fast paths +inline file_input_adapter input_adapter(std::FILE* file) +{ + if (file == nullptr) + { + JSON_THROW(parse_error::create(101, 0, "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr)); + } + return file_input_adapter(file); +} + +inline input_stream_adapter input_adapter(std::istream& stream) +{ + return input_stream_adapter(stream); +} + +inline input_stream_adapter input_adapter(std::istream&& stream) +{ + return input_stream_adapter(stream); +} +#endif // JSON_NO_IO + +using contiguous_bytes_input_adapter = decltype(input_adapter(std::declval(), std::declval())); + +// Null-delimited strings, and the like. +template < typename CharT, + typename std::enable_if < + std::is_pointer::value&& + !std::is_array::value&& + std::is_integral::type>::value&& + sizeof(typename std::remove_pointer::type) == 1, + int >::type = 0 > +contiguous_bytes_input_adapter input_adapter(CharT b) +{ + if (b == nullptr) + { + JSON_THROW(parse_error::create(101, 0, "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr)); + } + auto length = std::strlen(reinterpret_cast(b)); + const auto* ptr = reinterpret_cast(b); + return input_adapter(ptr, ptr + length); // cppcheck-suppress[nullPointerArithmeticRedundantCheck] +} + +template +auto input_adapter(T (&array)[N]) -> decltype(input_adapter(array, array + N)) // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) +{ + return input_adapter(array, array + N); +} + +// This class only handles inputs of input_buffer_adapter type. +// It's required so that expressions like {ptr, len} can be implicitly cast +// to the correct adapter. +class span_input_adapter +{ + public: + template < typename CharT, + typename std::enable_if < + std::is_pointer::value&& + std::is_integral::type>::value&& + sizeof(typename std::remove_pointer::type) == 1, + int >::type = 0 > + span_input_adapter(CharT b, std::size_t l) + : ia(reinterpret_cast(b), reinterpret_cast(b) + l) {} + + template::iterator_category, std::random_access_iterator_tag>::value, + int>::type = 0> + span_input_adapter(IteratorType first, IteratorType last) + : ia(input_adapter(first, last)) {} + + contiguous_bytes_input_adapter&& get() + { + return std::move(ia); // NOLINT(hicpp-move-const-arg,performance-move-const-arg) + } + + private: + contiguous_bytes_input_adapter ia; +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/input/json_sax.hpp b/src/detail/include/nlohmann/detail/input/json_sax.hpp new file mode 100644 index 000000000..f27d14b48 --- /dev/null +++ b/src/detail/include/nlohmann/detail/input/json_sax.hpp @@ -0,0 +1,986 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include +#include // string +#include // enable_if_t +#include // move +#include // vector + +#include +#include +#include +#include +NLOHMANN_JSON_NAMESPACE_BEGIN + +/*! +@brief SAX interface + +This class describes the SAX interface used by @ref nlohmann::json::sax_parse. +Each function is called in different situations while the input is parsed. The +boolean return value informs the parser whether to continue processing the +input. +*/ +template +struct json_sax +{ + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using string_t = typename BasicJsonType::string_t; + using binary_t = typename BasicJsonType::binary_t; + + /*! + @brief a null value was read + @return whether parsing should proceed + */ + virtual bool null() = 0; + + /*! + @brief a boolean value was read + @param[in] val boolean value + @return whether parsing should proceed + */ + virtual bool boolean(bool val) = 0; + + /*! + @brief an integer number was read + @param[in] val integer value + @return whether parsing should proceed + */ + virtual bool number_integer(number_integer_t val) = 0; + + /*! + @brief an unsigned integer number was read + @param[in] val unsigned integer value + @return whether parsing should proceed + */ + virtual bool number_unsigned(number_unsigned_t val) = 0; + + /*! + @brief a floating-point number was read + @param[in] val floating-point value + @param[in] s raw token value + @return whether parsing should proceed + */ + virtual bool number_float(number_float_t val, const string_t& s) = 0; + + /*! + @brief a string value was read + @param[in] val string value + @return whether parsing should proceed + @note It is safe to move the passed string value. + */ + virtual bool string(string_t& val) = 0; + + /*! + @brief a binary value was read + @param[in] val binary value + @return whether parsing should proceed + @note It is safe to move the passed binary value. + */ + virtual bool binary(binary_t& val) = 0; + + /*! + @brief the beginning of an object was read + @param[in] elements number of object elements or -1 if unknown + @return whether parsing should proceed + @note binary formats may report the number of elements + */ + virtual bool start_object(std::size_t elements) = 0; + + /*! + @brief an object key was read + @param[in] val object key + @return whether parsing should proceed + @note It is safe to move the passed string. + */ + virtual bool key(string_t& val) = 0; + + /*! + @brief the end of an object was read + @return whether parsing should proceed + */ + virtual bool end_object() = 0; + + /*! + @brief the beginning of an array was read + @param[in] elements number of array elements or -1 if unknown + @return whether parsing should proceed + @note binary formats may report the number of elements + */ + virtual bool start_array(std::size_t elements) = 0; + + /*! + @brief the end of an array was read + @return whether parsing should proceed + */ + virtual bool end_array() = 0; + + /*! + @brief a parse error occurred + @param[in] position the position in the input where the error occurs + @param[in] last_token the last read token + @param[in] ex an exception object describing the error + @return whether parsing should proceed (must return false) + */ + virtual bool parse_error(std::size_t position, + const std::string& last_token, + const detail::exception& ex) = 0; + + json_sax() = default; + json_sax(const json_sax&) = default; + json_sax(json_sax&&) noexcept = default; + json_sax& operator=(const json_sax&) = default; + json_sax& operator=(json_sax&&) noexcept = default; + virtual ~json_sax() = default; +}; + +namespace detail +{ +constexpr std::size_t unknown_size() +{ + return (std::numeric_limits::max)(); +} + +/*! +@brief SAX implementation to create a JSON value from SAX events + +This class implements the @ref json_sax interface and processes the SAX events +to create a JSON value which makes it basically a DOM parser. The structure or +hierarchy of the JSON value is managed by the stack `ref_stack` which contains +a pointer to the respective array or object for each recursion depth. + +After successful parsing, the value that is passed by reference to the +constructor contains the parsed value. + +@tparam BasicJsonType the JSON type +*/ +template +class json_sax_dom_parser +{ + public: + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using string_t = typename BasicJsonType::string_t; + using binary_t = typename BasicJsonType::binary_t; + using lexer_t = lexer; + + /*! + @param[in,out] r reference to a JSON value that is manipulated while + parsing + @param[in] allow_exceptions_ whether parse errors yield exceptions + */ + explicit json_sax_dom_parser(BasicJsonType& r, const bool allow_exceptions_ = true, lexer_t* lexer_ = nullptr) + : root(r), allow_exceptions(allow_exceptions_), m_lexer_ref(lexer_) + {} + + // make class move-only + json_sax_dom_parser(const json_sax_dom_parser&) = delete; + json_sax_dom_parser(json_sax_dom_parser&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor) + json_sax_dom_parser& operator=(const json_sax_dom_parser&) = delete; + json_sax_dom_parser& operator=(json_sax_dom_parser&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor) + ~json_sax_dom_parser() = default; + + bool null() + { + handle_value(nullptr); + return true; + } + + bool boolean(bool val) + { + handle_value(val); + return true; + } + + bool number_integer(number_integer_t val) + { + handle_value(val); + return true; + } + + bool number_unsigned(number_unsigned_t val) + { + handle_value(val); + return true; + } + + bool number_float(number_float_t val, const string_t& /*unused*/) + { + handle_value(val); + return true; + } + + bool string(string_t& val) + { + handle_value(val); + return true; + } + + bool binary(binary_t& val) + { + handle_value(std::move(val)); + return true; + } + + bool start_object(std::size_t len) + { + ref_stack.push_back(handle_value(BasicJsonType::value_t::object)); + +#if JSON_DIAGNOSTIC_POSITIONS + // Manually set the start position of the object here. + // Ensure this is after the call to handle_value to ensure correct start position. + if (m_lexer_ref) + { + // Lexer has read the first character of the object, so + // subtract 1 from the position to get the correct start position. + ref_stack.back()->start_position = m_lexer_ref->get_position() - 1; + } +#endif + + if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) + { + JSON_THROW(out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); + } + + return true; + } + + bool key(string_t& val) + { + JSON_ASSERT(!ref_stack.empty()); + JSON_ASSERT(ref_stack.back()->is_object()); + + // add null at the given key and store the reference for later + object_element = &(ref_stack.back()->m_data.m_value.object->operator[](val)); + return true; + } + + bool end_object() + { + JSON_ASSERT(!ref_stack.empty()); + JSON_ASSERT(ref_stack.back()->is_object()); + +#if JSON_DIAGNOSTIC_POSITIONS + if (m_lexer_ref) + { + // Lexer's position is past the closing brace, so set that as the end position. + ref_stack.back()->end_position = m_lexer_ref->get_position(); + } +#endif + + ref_stack.back()->set_parents(); + ref_stack.pop_back(); + return true; + } + + bool start_array(std::size_t len) + { + ref_stack.push_back(handle_value(BasicJsonType::value_t::array)); + +#if JSON_DIAGNOSTIC_POSITIONS + // Manually set the start position of the array here. + // Ensure this is after the call to handle_value to ensure correct start position. + if (m_lexer_ref) + { + ref_stack.back()->start_position = m_lexer_ref->get_position() - 1; + } +#endif + + if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) + { + JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + } + + return true; + } + + bool end_array() + { + JSON_ASSERT(!ref_stack.empty()); + JSON_ASSERT(ref_stack.back()->is_array()); + +#if JSON_DIAGNOSTIC_POSITIONS + if (m_lexer_ref) + { + // Lexer's position is past the closing bracket, so set that as the end position. + ref_stack.back()->end_position = m_lexer_ref->get_position(); + } +#endif + + ref_stack.back()->set_parents(); + ref_stack.pop_back(); + return true; + } + + template + bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, + const Exception& ex) + { + errored = true; + static_cast(ex); + if (allow_exceptions) + { + JSON_THROW(ex); + } + return false; + } + + constexpr bool is_errored() const + { + return errored; + } + + private: + +#if JSON_DIAGNOSTIC_POSITIONS + void handle_diagnostic_positions_for_json_value(BasicJsonType& v) + { + if (m_lexer_ref) + { + // Lexer has read past the current field value, so set the end position to the current position. + // The start position will be set below based on the length of the string representation + // of the value. + v.end_position = m_lexer_ref->get_position(); + + switch (v.type()) + { + case value_t::boolean: + { + // 4 and 5 are the string length of "true" and "false" + v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5); + break; + } + + case value_t::null: + { + // 4 is the string length of "null" + v.start_position = v.end_position - 4; + break; + } + + case value_t::string: + { + // include the length of the quotes, which is 2 + v.start_position = v.end_position - v.m_data.m_value.string->size() - 2; + break; + } + + // As we handle the start and end positions for values created during parsing, + // we do not expect the following value type to be called. Regardless, set the positions + // in case this is created manually or through a different constructor. Exclude from lcov + // since the exact condition of this switch is esoteric. + // LCOV_EXCL_START + case value_t::discarded: + { + v.end_position = std::string::npos; + v.start_position = v.end_position; + break; + } + // LCOV_EXCL_STOP + case value_t::binary: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + { + v.start_position = v.end_position - m_lexer_ref->get_string().size(); + break; + } + case value_t::object: + case value_t::array: + { + // object and array are handled in start_object() and start_array() handlers + // skip setting the values here. + break; + } + default: // LCOV_EXCL_LINE + // Handle all possible types discretely, default handler should never be reached. + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE + } + } + } +#endif + + /*! + @invariant If the ref stack is empty, then the passed value will be the new + root. + @invariant If the ref stack contains a value, then it is an array or an + object to which we can add elements + */ + template + JSON_HEDLEY_RETURNS_NON_NULL + BasicJsonType* handle_value(Value&& v) + { + if (ref_stack.empty()) + { + root = BasicJsonType(std::forward(v)); + +#if JSON_DIAGNOSTIC_POSITIONS + handle_diagnostic_positions_for_json_value(root); +#endif + + return &root; + } + + JSON_ASSERT(ref_stack.back()->is_array() || ref_stack.back()->is_object()); + + if (ref_stack.back()->is_array()) + { + ref_stack.back()->m_data.m_value.array->emplace_back(std::forward(v)); + +#if JSON_DIAGNOSTIC_POSITIONS + handle_diagnostic_positions_for_json_value(ref_stack.back()->m_data.m_value.array->back()); +#endif + + return &(ref_stack.back()->m_data.m_value.array->back()); + } + + JSON_ASSERT(ref_stack.back()->is_object()); + JSON_ASSERT(object_element); + *object_element = BasicJsonType(std::forward(v)); + +#if JSON_DIAGNOSTIC_POSITIONS + handle_diagnostic_positions_for_json_value(*object_element); +#endif + + return object_element; + } + + /// the parsed JSON value + BasicJsonType& root; + /// stack to model hierarchy of values + std::vector ref_stack {}; + /// helper to hold the reference for the next object element + BasicJsonType* object_element = nullptr; + /// whether a syntax error occurred + bool errored = false; + /// whether to throw exceptions in case of errors + const bool allow_exceptions = true; + /// the lexer reference to obtain the current position + lexer_t* m_lexer_ref = nullptr; +}; + +template +class json_sax_dom_callback_parser +{ + public: + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using string_t = typename BasicJsonType::string_t; + using binary_t = typename BasicJsonType::binary_t; + using parser_callback_t = typename BasicJsonType::parser_callback_t; + using parse_event_t = typename BasicJsonType::parse_event_t; + using lexer_t = lexer; + + json_sax_dom_callback_parser(BasicJsonType& r, + parser_callback_t cb, + const bool allow_exceptions_ = true, + lexer_t* lexer_ = nullptr) + : root(r), callback(std::move(cb)), allow_exceptions(allow_exceptions_), m_lexer_ref(lexer_) + { + keep_stack.push_back(true); + } + + // make class move-only + json_sax_dom_callback_parser(const json_sax_dom_callback_parser&) = delete; + json_sax_dom_callback_parser(json_sax_dom_callback_parser&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor) + json_sax_dom_callback_parser& operator=(const json_sax_dom_callback_parser&) = delete; + json_sax_dom_callback_parser& operator=(json_sax_dom_callback_parser&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor) + ~json_sax_dom_callback_parser() = default; + + bool null() + { + handle_value(nullptr); + return true; + } + + bool boolean(bool val) + { + handle_value(val); + return true; + } + + bool number_integer(number_integer_t val) + { + handle_value(val); + return true; + } + + bool number_unsigned(number_unsigned_t val) + { + handle_value(val); + return true; + } + + bool number_float(number_float_t val, const string_t& /*unused*/) + { + handle_value(val); + return true; + } + + bool string(string_t& val) + { + handle_value(val); + return true; + } + + bool binary(binary_t& val) + { + handle_value(std::move(val)); + return true; + } + + bool start_object(std::size_t len) + { + // check callback for object start + const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::object_start, discarded); + keep_stack.push_back(keep); + + auto val = handle_value(BasicJsonType::value_t::object, true); + ref_stack.push_back(val.second); + + if (ref_stack.back()) + { + +#if JSON_DIAGNOSTIC_POSITIONS + // Manually set the start position of the object here. + // Ensure this is after the call to handle_value to ensure correct start position. + if (m_lexer_ref) + { + // Lexer has read the first character of the object, so + // subtract 1 from the position to get the correct start position. + ref_stack.back()->start_position = m_lexer_ref->get_position() - 1; + } +#endif + + // check object limit + if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) + { + JSON_THROW(out_of_range::create(408, concat("excessive object size: ", std::to_string(len)), ref_stack.back())); + } + } + return true; + } + + bool key(string_t& val) + { + BasicJsonType k = BasicJsonType(val); + + // check callback for the key + const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::key, k); + key_keep_stack.push_back(keep); + + // add discarded value at the given key and store the reference for later + if (keep && ref_stack.back()) + { + object_element = &(ref_stack.back()->m_data.m_value.object->operator[](val) = discarded); + } + + return true; + } + + bool end_object() + { + if (ref_stack.back()) + { + if (!callback(static_cast(ref_stack.size()) - 1, parse_event_t::object_end, *ref_stack.back())) + { + // discard object + *ref_stack.back() = discarded; + +#if JSON_DIAGNOSTIC_POSITIONS + // Set start/end positions for discarded object. + handle_diagnostic_positions_for_json_value(*ref_stack.back()); +#endif + } + else + { + +#if JSON_DIAGNOSTIC_POSITIONS + if (m_lexer_ref) + { + // Lexer's position is past the closing brace, so set that as the end position. + ref_stack.back()->end_position = m_lexer_ref->get_position(); + } +#endif + + ref_stack.back()->set_parents(); + } + } + + JSON_ASSERT(!ref_stack.empty()); + JSON_ASSERT(!keep_stack.empty()); + ref_stack.pop_back(); + keep_stack.pop_back(); + + if (!ref_stack.empty() && ref_stack.back() && ref_stack.back()->is_structured()) + { + // remove discarded value + for (auto it = ref_stack.back()->begin(); it != ref_stack.back()->end(); ++it) + { + if (it->is_discarded()) + { + ref_stack.back()->erase(it); + break; + } + } + } + + return true; + } + + bool start_array(std::size_t len) + { + const bool keep = callback(static_cast(ref_stack.size()), parse_event_t::array_start, discarded); + keep_stack.push_back(keep); + + auto val = handle_value(BasicJsonType::value_t::array, true); + ref_stack.push_back(val.second); + + if (ref_stack.back()) + { + +#if JSON_DIAGNOSTIC_POSITIONS + // Manually set the start position of the array here. + // Ensure this is after the call to handle_value to ensure correct start position. + if (m_lexer_ref) + { + // Lexer has read the first character of the array, so + // subtract 1 from the position to get the correct start position. + ref_stack.back()->start_position = m_lexer_ref->get_position() - 1; + } +#endif + + // check array limit + if (JSON_HEDLEY_UNLIKELY(len != detail::unknown_size() && len > ref_stack.back()->max_size())) + { + JSON_THROW(out_of_range::create(408, concat("excessive array size: ", std::to_string(len)), ref_stack.back())); + } + } + + return true; + } + + bool end_array() + { + bool keep = true; + + if (ref_stack.back()) + { + keep = callback(static_cast(ref_stack.size()) - 1, parse_event_t::array_end, *ref_stack.back()); + if (keep) + { + +#if JSON_DIAGNOSTIC_POSITIONS + if (m_lexer_ref) + { + // Lexer's position is past the closing bracket, so set that as the end position. + ref_stack.back()->end_position = m_lexer_ref->get_position(); + } +#endif + + ref_stack.back()->set_parents(); + } + else + { + // discard array + *ref_stack.back() = discarded; + +#if JSON_DIAGNOSTIC_POSITIONS + // Set start/end positions for discarded array. + handle_diagnostic_positions_for_json_value(*ref_stack.back()); +#endif + } + } + + JSON_ASSERT(!ref_stack.empty()); + JSON_ASSERT(!keep_stack.empty()); + ref_stack.pop_back(); + keep_stack.pop_back(); + + // remove discarded value + if (!keep && !ref_stack.empty() && ref_stack.back()->is_array()) + { + ref_stack.back()->m_data.m_value.array->pop_back(); + } + + return true; + } + + template + bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, + const Exception& ex) + { + errored = true; + static_cast(ex); + if (allow_exceptions) + { + JSON_THROW(ex); + } + return false; + } + + constexpr bool is_errored() const + { + return errored; + } + + private: + +#if JSON_DIAGNOSTIC_POSITIONS + void handle_diagnostic_positions_for_json_value(BasicJsonType& v) + { + if (m_lexer_ref) + { + // Lexer has read past the current field value, so set the end position to the current position. + // The start position will be set below based on the length of the string representation + // of the value. + v.end_position = m_lexer_ref->get_position(); + + switch (v.type()) + { + case value_t::boolean: + { + // 4 and 5 are the string length of "true" and "false" + v.start_position = v.end_position - (v.m_data.m_value.boolean ? 4 : 5); + break; + } + + case value_t::null: + { + // 4 is the string length of "null" + v.start_position = v.end_position - 4; + break; + } + + case value_t::string: + { + // include the length of the quotes, which is 2 + v.start_position = v.end_position - v.m_data.m_value.string->size() - 2; + break; + } + + case value_t::discarded: + { + v.end_position = std::string::npos; + v.start_position = v.end_position; + break; + } + + case value_t::binary: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + { + v.start_position = v.end_position - m_lexer_ref->get_string().size(); + break; + } + + case value_t::object: + case value_t::array: + { + // object and array are handled in start_object() and start_array() handlers + // skip setting the values here. + break; + } + default: // LCOV_EXCL_LINE + // Handle all possible types discretely, default handler should never be reached. + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert,-warnings-as-errors) LCOV_EXCL_LINE + } + } + } +#endif + + /*! + @param[in] v value to add to the JSON value we build during parsing + @param[in] skip_callback whether we should skip calling the callback + function; this is required after start_array() and + start_object() SAX events, because otherwise we would call the + callback function with an empty array or object, respectively. + + @invariant If the ref stack is empty, then the passed value will be the new + root. + @invariant If the ref stack contains a value, then it is an array or an + object to which we can add elements + + @return pair of boolean (whether value should be kept) and pointer (to the + passed value in the ref_stack hierarchy; nullptr if not kept) + */ + template + std::pair handle_value(Value&& v, const bool skip_callback = false) + { + JSON_ASSERT(!keep_stack.empty()); + + // do not handle this value if we know it would be added to a discarded + // container + if (!keep_stack.back()) + { + return {false, nullptr}; + } + + // create value + auto value = BasicJsonType(std::forward(v)); + +#if JSON_DIAGNOSTIC_POSITIONS + handle_diagnostic_positions_for_json_value(value); +#endif + + // check callback + const bool keep = skip_callback || callback(static_cast(ref_stack.size()), parse_event_t::value, value); + + // do not handle this value if we just learnt it shall be discarded + if (!keep) + { + return {false, nullptr}; + } + + if (ref_stack.empty()) + { + root = std::move(value); + return {true, & root}; + } + + // skip this value if we already decided to skip the parent + // (https://github.com/nlohmann/json/issues/971#issuecomment-413678360) + if (!ref_stack.back()) + { + return {false, nullptr}; + } + + // we now only expect arrays and objects + JSON_ASSERT(ref_stack.back()->is_array() || ref_stack.back()->is_object()); + + // array + if (ref_stack.back()->is_array()) + { + ref_stack.back()->m_data.m_value.array->emplace_back(std::move(value)); + return {true, & (ref_stack.back()->m_data.m_value.array->back())}; + } + + // object + JSON_ASSERT(ref_stack.back()->is_object()); + // check if we should store an element for the current key + JSON_ASSERT(!key_keep_stack.empty()); + const bool store_element = key_keep_stack.back(); + key_keep_stack.pop_back(); + + if (!store_element) + { + return {false, nullptr}; + } + + JSON_ASSERT(object_element); + *object_element = std::move(value); + return {true, object_element}; + } + + /// the parsed JSON value + BasicJsonType& root; + /// stack to model hierarchy of values + std::vector ref_stack {}; + /// stack to manage which values to keep + std::vector keep_stack {}; // NOLINT(readability-redundant-member-init) + /// stack to manage which object keys to keep + std::vector key_keep_stack {}; // NOLINT(readability-redundant-member-init) + /// helper to hold the reference for the next object element + BasicJsonType* object_element = nullptr; + /// whether a syntax error occurred + bool errored = false; + /// callback function + const parser_callback_t callback = nullptr; + /// whether to throw exceptions in case of errors + const bool allow_exceptions = true; + /// a discarded value for the callback + BasicJsonType discarded = BasicJsonType::value_t::discarded; + /// the lexer reference to obtain the current position + lexer_t* m_lexer_ref = nullptr; +}; + +template +class json_sax_acceptor +{ + public: + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using string_t = typename BasicJsonType::string_t; + using binary_t = typename BasicJsonType::binary_t; + + bool null() + { + return true; + } + + bool boolean(bool /*unused*/) + { + return true; + } + + bool number_integer(number_integer_t /*unused*/) + { + return true; + } + + bool number_unsigned(number_unsigned_t /*unused*/) + { + return true; + } + + bool number_float(number_float_t /*unused*/, const string_t& /*unused*/) + { + return true; + } + + bool string(string_t& /*unused*/) + { + return true; + } + + bool binary(binary_t& /*unused*/) + { + return true; + } + + bool start_object(std::size_t /*unused*/ = detail::unknown_size()) + { + return true; + } + + bool key(string_t& /*unused*/) + { + return true; + } + + bool end_object() + { + return true; + } + + bool start_array(std::size_t /*unused*/ = detail::unknown_size()) + { + return true; + } + + bool end_array() + { + return true; + } + + bool parse_error(std::size_t /*unused*/, const std::string& /*unused*/, const detail::exception& /*unused*/) + { + return false; + } +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/input/lexer.hpp b/src/detail/include/nlohmann/detail/input/lexer.hpp new file mode 100644 index 000000000..ae93f9595 --- /dev/null +++ b/src/detail/include/nlohmann/detail/input/lexer.hpp @@ -0,0 +1,1643 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // localeconv +#include // size_t +#include // snprintf +#include // strtof, strtod, strtold, strtoll, strtoull +#include // initializer_list +#include // char_traits, string +#include // move +#include // vector + +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/////////// +// lexer // +/////////// + +template +class lexer_base +{ + public: + /// token types for the parser + enum class token_type + { + uninitialized, ///< indicating the scanner is uninitialized + literal_true, ///< the `true` literal + literal_false, ///< the `false` literal + literal_null, ///< the `null` literal + value_string, ///< a string -- use get_string() for actual value + value_unsigned, ///< an unsigned integer -- use get_number_unsigned() for actual value + value_integer, ///< a signed integer -- use get_number_integer() for actual value + value_float, ///< an floating point number -- use get_number_float() for actual value + begin_array, ///< the character for array begin `[` + begin_object, ///< the character for object begin `{` + end_array, ///< the character for array end `]` + end_object, ///< the character for object end `}` + name_separator, ///< the name separator `:` + value_separator, ///< the value separator `,` + parse_error, ///< indicating a parse error + end_of_input, ///< indicating the end of the input buffer + literal_or_value ///< a literal or the begin of a value (only for diagnostics) + }; + + /// return name of values of type token_type (only used for errors) + JSON_HEDLEY_RETURNS_NON_NULL + JSON_HEDLEY_CONST + static const char* token_type_name(const token_type t) noexcept + { + switch (t) + { + case token_type::uninitialized: + return ""; + case token_type::literal_true: + return "true literal"; + case token_type::literal_false: + return "false literal"; + case token_type::literal_null: + return "null literal"; + case token_type::value_string: + return "string literal"; + case token_type::value_unsigned: + case token_type::value_integer: + case token_type::value_float: + return "number literal"; + case token_type::begin_array: + return "'['"; + case token_type::begin_object: + return "'{'"; + case token_type::end_array: + return "']'"; + case token_type::end_object: + return "'}'"; + case token_type::name_separator: + return "':'"; + case token_type::value_separator: + return "','"; + case token_type::parse_error: + return ""; + case token_type::end_of_input: + return "end of input"; + case token_type::literal_or_value: + return "'[', '{', or a literal"; + // LCOV_EXCL_START + default: // catch non-enum values + return "unknown token"; + // LCOV_EXCL_STOP + } + } +}; +/*! +@brief lexical analysis + +This class organizes the lexical analysis during JSON deserialization. +*/ +template +class lexer : public lexer_base +{ + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using string_t = typename BasicJsonType::string_t; + using char_type = typename InputAdapterType::char_type; + using char_int_type = typename char_traits::int_type; + + public: + using token_type = typename lexer_base::token_type; + + explicit lexer(InputAdapterType&& adapter, bool ignore_comments_ = false) noexcept + : ia(std::move(adapter)) + , ignore_comments(ignore_comments_) + , decimal_point_char(static_cast(get_decimal_point())) + {} + + // deleted because of pointer members + lexer(const lexer&) = delete; + lexer(lexer&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor) + lexer& operator=(lexer&) = delete; + lexer& operator=(lexer&&) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor) + ~lexer() = default; + + private: + ///////////////////// + // locales + ///////////////////// + + /// return the locale-dependent decimal point + JSON_HEDLEY_PURE + static char get_decimal_point() noexcept + { + const auto* loc = localeconv(); + JSON_ASSERT(loc != nullptr); + return (loc->decimal_point == nullptr) ? '.' : *(loc->decimal_point); + } + + ///////////////////// + // scan functions + ///////////////////// + + /*! + @brief get codepoint from 4 hex characters following `\u` + + For input "\u c1 c2 c3 c4" the codepoint is: + (c1 * 0x1000) + (c2 * 0x0100) + (c3 * 0x0010) + c4 + = (c1 << 12) + (c2 << 8) + (c3 << 4) + (c4 << 0) + + Furthermore, the possible characters '0'..'9', 'A'..'F', and 'a'..'f' + must be converted to the integers 0x0..0x9, 0xA..0xF, 0xA..0xF, resp. The + conversion is done by subtracting the offset (0x30, 0x37, and 0x57) + between the ASCII value of the character and the desired integer value. + + @return codepoint (0x0000..0xFFFF) or -1 in case of an error (e.g. EOF or + non-hex character) + */ + int get_codepoint() + { + // this function only makes sense after reading `\u` + JSON_ASSERT(current == 'u'); + int codepoint = 0; + + const auto factors = { 12u, 8u, 4u, 0u }; + for (const auto factor : factors) + { + get(); + + if (current >= '0' && current <= '9') + { + codepoint += static_cast((static_cast(current) - 0x30u) << factor); + } + else if (current >= 'A' && current <= 'F') + { + codepoint += static_cast((static_cast(current) - 0x37u) << factor); + } + else if (current >= 'a' && current <= 'f') + { + codepoint += static_cast((static_cast(current) - 0x57u) << factor); + } + else + { + return -1; + } + } + + JSON_ASSERT(0x0000 <= codepoint && codepoint <= 0xFFFF); + return codepoint; + } + + /*! + @brief check if the next byte(s) are inside a given range + + Adds the current byte and, for each passed range, reads a new byte and + checks if it is inside the range. If a violation was detected, set up an + error message and return false. Otherwise, return true. + + @param[in] ranges list of integers; interpreted as list of pairs of + inclusive lower and upper bound, respectively + + @pre The passed list @a ranges must have 2, 4, or 6 elements; that is, + 1, 2, or 3 pairs. This precondition is enforced by an assertion. + + @return true if and only if no range violation was detected + */ + bool next_byte_in_range(std::initializer_list ranges) + { + JSON_ASSERT(ranges.size() == 2 || ranges.size() == 4 || ranges.size() == 6); + add(current); + + for (auto range = ranges.begin(); range != ranges.end(); ++range) + { + get(); + if (JSON_HEDLEY_LIKELY(*range <= current && current <= *(++range))) // NOLINT(bugprone-inc-dec-in-conditions) + { + add(current); + } + else + { + error_message = "invalid string: ill-formed UTF-8 byte"; + return false; + } + } + + return true; + } + + /*! + @brief scan a string literal + + This function scans a string according to Sect. 7 of RFC 8259. While + scanning, bytes are escaped and copied into buffer token_buffer. Then the + function returns successfully, token_buffer is *not* null-terminated (as it + may contain \0 bytes), and token_buffer.size() is the number of bytes in the + string. + + @return token_type::value_string if string could be successfully scanned, + token_type::parse_error otherwise + + @note In case of errors, variable error_message contains a textual + description. + */ + token_type scan_string() + { + // reset token_buffer (ignore opening quote) + reset(); + + // we entered the function by reading an open quote + JSON_ASSERT(current == '\"'); + + while (true) + { + // get the next character + switch (get()) + { + // end of file while parsing the string + case char_traits::eof(): + { + error_message = "invalid string: missing closing quote"; + return token_type::parse_error; + } + + // closing quote + case '\"': + { + return token_type::value_string; + } + + // escapes + case '\\': + { + switch (get()) + { + // quotation mark + case '\"': + add('\"'); + break; + // reverse solidus + case '\\': + add('\\'); + break; + // solidus + case '/': + add('/'); + break; + // backspace + case 'b': + add('\b'); + break; + // form feed + case 'f': + add('\f'); + break; + // line feed + case 'n': + add('\n'); + break; + // carriage return + case 'r': + add('\r'); + break; + // tab + case 't': + add('\t'); + break; + + // unicode escapes + case 'u': + { + const int codepoint1 = get_codepoint(); + int codepoint = codepoint1; // start with codepoint1 + + if (JSON_HEDLEY_UNLIKELY(codepoint1 == -1)) + { + error_message = "invalid string: '\\u' must be followed by 4 hex digits"; + return token_type::parse_error; + } + + // check if code point is a high surrogate + if (0xD800 <= codepoint1 && codepoint1 <= 0xDBFF) + { + // expect next \uxxxx entry + if (JSON_HEDLEY_LIKELY(get() == '\\' && get() == 'u')) + { + const int codepoint2 = get_codepoint(); + + if (JSON_HEDLEY_UNLIKELY(codepoint2 == -1)) + { + error_message = "invalid string: '\\u' must be followed by 4 hex digits"; + return token_type::parse_error; + } + + // check if codepoint2 is a low surrogate + if (JSON_HEDLEY_LIKELY(0xDC00 <= codepoint2 && codepoint2 <= 0xDFFF)) + { + // overwrite codepoint + codepoint = static_cast( + // high surrogate occupies the most significant 22 bits + (static_cast(codepoint1) << 10u) + // low surrogate occupies the least significant 15 bits + + static_cast(codepoint2) + // there is still the 0xD800, 0xDC00, and 0x10000 noise + // in the result, so we have to subtract with: + // (0xD800 << 10) + DC00 - 0x10000 = 0x35FDC00 + - 0x35FDC00u); + } + else + { + error_message = "invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF"; + return token_type::parse_error; + } + } + else + { + error_message = "invalid string: surrogate U+D800..U+DBFF must be followed by U+DC00..U+DFFF"; + return token_type::parse_error; + } + } + else + { + if (JSON_HEDLEY_UNLIKELY(0xDC00 <= codepoint1 && codepoint1 <= 0xDFFF)) + { + error_message = "invalid string: surrogate U+DC00..U+DFFF must follow U+D800..U+DBFF"; + return token_type::parse_error; + } + } + + // the result of the above calculation yields a proper codepoint + JSON_ASSERT(0x00 <= codepoint && codepoint <= 0x10FFFF); + + // translate codepoint into bytes + if (codepoint < 0x80) + { + // 1-byte characters: 0xxxxxxx (ASCII) + add(static_cast(codepoint)); + } + else if (codepoint <= 0x7FF) + { + // 2-byte characters: 110xxxxx 10xxxxxx + add(static_cast(0xC0u | (static_cast(codepoint) >> 6u))); + add(static_cast(0x80u | (static_cast(codepoint) & 0x3Fu))); + } + else if (codepoint <= 0xFFFF) + { + // 3-byte characters: 1110xxxx 10xxxxxx 10xxxxxx + add(static_cast(0xE0u | (static_cast(codepoint) >> 12u))); + add(static_cast(0x80u | ((static_cast(codepoint) >> 6u) & 0x3Fu))); + add(static_cast(0x80u | (static_cast(codepoint) & 0x3Fu))); + } + else + { + // 4-byte characters: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx + add(static_cast(0xF0u | (static_cast(codepoint) >> 18u))); + add(static_cast(0x80u | ((static_cast(codepoint) >> 12u) & 0x3Fu))); + add(static_cast(0x80u | ((static_cast(codepoint) >> 6u) & 0x3Fu))); + add(static_cast(0x80u | (static_cast(codepoint) & 0x3Fu))); + } + + break; + } + + // other characters after escape + default: + error_message = "invalid string: forbidden character after backslash"; + return token_type::parse_error; + } + + break; + } + + // invalid control characters + case 0x00: + { + error_message = "invalid string: control character U+0000 (NUL) must be escaped to \\u0000"; + return token_type::parse_error; + } + + case 0x01: + { + error_message = "invalid string: control character U+0001 (SOH) must be escaped to \\u0001"; + return token_type::parse_error; + } + + case 0x02: + { + error_message = "invalid string: control character U+0002 (STX) must be escaped to \\u0002"; + return token_type::parse_error; + } + + case 0x03: + { + error_message = "invalid string: control character U+0003 (ETX) must be escaped to \\u0003"; + return token_type::parse_error; + } + + case 0x04: + { + error_message = "invalid string: control character U+0004 (EOT) must be escaped to \\u0004"; + return token_type::parse_error; + } + + case 0x05: + { + error_message = "invalid string: control character U+0005 (ENQ) must be escaped to \\u0005"; + return token_type::parse_error; + } + + case 0x06: + { + error_message = "invalid string: control character U+0006 (ACK) must be escaped to \\u0006"; + return token_type::parse_error; + } + + case 0x07: + { + error_message = "invalid string: control character U+0007 (BEL) must be escaped to \\u0007"; + return token_type::parse_error; + } + + case 0x08: + { + error_message = "invalid string: control character U+0008 (BS) must be escaped to \\u0008 or \\b"; + return token_type::parse_error; + } + + case 0x09: + { + error_message = "invalid string: control character U+0009 (HT) must be escaped to \\u0009 or \\t"; + return token_type::parse_error; + } + + case 0x0A: + { + error_message = "invalid string: control character U+000A (LF) must be escaped to \\u000A or \\n"; + return token_type::parse_error; + } + + case 0x0B: + { + error_message = "invalid string: control character U+000B (VT) must be escaped to \\u000B"; + return token_type::parse_error; + } + + case 0x0C: + { + error_message = "invalid string: control character U+000C (FF) must be escaped to \\u000C or \\f"; + return token_type::parse_error; + } + + case 0x0D: + { + error_message = "invalid string: control character U+000D (CR) must be escaped to \\u000D or \\r"; + return token_type::parse_error; + } + + case 0x0E: + { + error_message = "invalid string: control character U+000E (SO) must be escaped to \\u000E"; + return token_type::parse_error; + } + + case 0x0F: + { + error_message = "invalid string: control character U+000F (SI) must be escaped to \\u000F"; + return token_type::parse_error; + } + + case 0x10: + { + error_message = "invalid string: control character U+0010 (DLE) must be escaped to \\u0010"; + return token_type::parse_error; + } + + case 0x11: + { + error_message = "invalid string: control character U+0011 (DC1) must be escaped to \\u0011"; + return token_type::parse_error; + } + + case 0x12: + { + error_message = "invalid string: control character U+0012 (DC2) must be escaped to \\u0012"; + return token_type::parse_error; + } + + case 0x13: + { + error_message = "invalid string: control character U+0013 (DC3) must be escaped to \\u0013"; + return token_type::parse_error; + } + + case 0x14: + { + error_message = "invalid string: control character U+0014 (DC4) must be escaped to \\u0014"; + return token_type::parse_error; + } + + case 0x15: + { + error_message = "invalid string: control character U+0015 (NAK) must be escaped to \\u0015"; + return token_type::parse_error; + } + + case 0x16: + { + error_message = "invalid string: control character U+0016 (SYN) must be escaped to \\u0016"; + return token_type::parse_error; + } + + case 0x17: + { + error_message = "invalid string: control character U+0017 (ETB) must be escaped to \\u0017"; + return token_type::parse_error; + } + + case 0x18: + { + error_message = "invalid string: control character U+0018 (CAN) must be escaped to \\u0018"; + return token_type::parse_error; + } + + case 0x19: + { + error_message = "invalid string: control character U+0019 (EM) must be escaped to \\u0019"; + return token_type::parse_error; + } + + case 0x1A: + { + error_message = "invalid string: control character U+001A (SUB) must be escaped to \\u001A"; + return token_type::parse_error; + } + + case 0x1B: + { + error_message = "invalid string: control character U+001B (ESC) must be escaped to \\u001B"; + return token_type::parse_error; + } + + case 0x1C: + { + error_message = "invalid string: control character U+001C (FS) must be escaped to \\u001C"; + return token_type::parse_error; + } + + case 0x1D: + { + error_message = "invalid string: control character U+001D (GS) must be escaped to \\u001D"; + return token_type::parse_error; + } + + case 0x1E: + { + error_message = "invalid string: control character U+001E (RS) must be escaped to \\u001E"; + return token_type::parse_error; + } + + case 0x1F: + { + error_message = "invalid string: control character U+001F (US) must be escaped to \\u001F"; + return token_type::parse_error; + } + + // U+0020..U+007F (except U+0022 (quote) and U+005C (backspace)) + case 0x20: + case 0x21: + case 0x23: + case 0x24: + case 0x25: + case 0x26: + case 0x27: + case 0x28: + case 0x29: + case 0x2A: + case 0x2B: + case 0x2C: + case 0x2D: + case 0x2E: + case 0x2F: + case 0x30: + case 0x31: + case 0x32: + case 0x33: + case 0x34: + case 0x35: + case 0x36: + case 0x37: + case 0x38: + case 0x39: + case 0x3A: + case 0x3B: + case 0x3C: + case 0x3D: + case 0x3E: + case 0x3F: + case 0x40: + case 0x41: + case 0x42: + case 0x43: + case 0x44: + case 0x45: + case 0x46: + case 0x47: + case 0x48: + case 0x49: + case 0x4A: + case 0x4B: + case 0x4C: + case 0x4D: + case 0x4E: + case 0x4F: + case 0x50: + case 0x51: + case 0x52: + case 0x53: + case 0x54: + case 0x55: + case 0x56: + case 0x57: + case 0x58: + case 0x59: + case 0x5A: + case 0x5B: + case 0x5D: + case 0x5E: + case 0x5F: + case 0x60: + case 0x61: + case 0x62: + case 0x63: + case 0x64: + case 0x65: + case 0x66: + case 0x67: + case 0x68: + case 0x69: + case 0x6A: + case 0x6B: + case 0x6C: + case 0x6D: + case 0x6E: + case 0x6F: + case 0x70: + case 0x71: + case 0x72: + case 0x73: + case 0x74: + case 0x75: + case 0x76: + case 0x77: + case 0x78: + case 0x79: + case 0x7A: + case 0x7B: + case 0x7C: + case 0x7D: + case 0x7E: + case 0x7F: + { + add(current); + break; + } + + // U+0080..U+07FF: bytes C2..DF 80..BF + case 0xC2: + case 0xC3: + case 0xC4: + case 0xC5: + case 0xC6: + case 0xC7: + case 0xC8: + case 0xC9: + case 0xCA: + case 0xCB: + case 0xCC: + case 0xCD: + case 0xCE: + case 0xCF: + case 0xD0: + case 0xD1: + case 0xD2: + case 0xD3: + case 0xD4: + case 0xD5: + case 0xD6: + case 0xD7: + case 0xD8: + case 0xD9: + case 0xDA: + case 0xDB: + case 0xDC: + case 0xDD: + case 0xDE: + case 0xDF: + { + if (JSON_HEDLEY_UNLIKELY(!next_byte_in_range({0x80, 0xBF}))) + { + return token_type::parse_error; + } + break; + } + + // U+0800..U+0FFF: bytes E0 A0..BF 80..BF + case 0xE0: + { + if (JSON_HEDLEY_UNLIKELY(!(next_byte_in_range({0xA0, 0xBF, 0x80, 0xBF})))) + { + return token_type::parse_error; + } + break; + } + + // U+1000..U+CFFF: bytes E1..EC 80..BF 80..BF + // U+E000..U+FFFF: bytes EE..EF 80..BF 80..BF + case 0xE1: + case 0xE2: + case 0xE3: + case 0xE4: + case 0xE5: + case 0xE6: + case 0xE7: + case 0xE8: + case 0xE9: + case 0xEA: + case 0xEB: + case 0xEC: + case 0xEE: + case 0xEF: + { + if (JSON_HEDLEY_UNLIKELY(!(next_byte_in_range({0x80, 0xBF, 0x80, 0xBF})))) + { + return token_type::parse_error; + } + break; + } + + // U+D000..U+D7FF: bytes ED 80..9F 80..BF + case 0xED: + { + if (JSON_HEDLEY_UNLIKELY(!(next_byte_in_range({0x80, 0x9F, 0x80, 0xBF})))) + { + return token_type::parse_error; + } + break; + } + + // U+10000..U+3FFFF F0 90..BF 80..BF 80..BF + case 0xF0: + { + if (JSON_HEDLEY_UNLIKELY(!(next_byte_in_range({0x90, 0xBF, 0x80, 0xBF, 0x80, 0xBF})))) + { + return token_type::parse_error; + } + break; + } + + // U+40000..U+FFFFF F1..F3 80..BF 80..BF 80..BF + case 0xF1: + case 0xF2: + case 0xF3: + { + if (JSON_HEDLEY_UNLIKELY(!(next_byte_in_range({0x80, 0xBF, 0x80, 0xBF, 0x80, 0xBF})))) + { + return token_type::parse_error; + } + break; + } + + // U+100000..U+10FFFF F4 80..8F 80..BF 80..BF + case 0xF4: + { + if (JSON_HEDLEY_UNLIKELY(!(next_byte_in_range({0x80, 0x8F, 0x80, 0xBF, 0x80, 0xBF})))) + { + return token_type::parse_error; + } + break; + } + + // the remaining bytes (80..C1 and F5..FF) are ill-formed + default: + { + error_message = "invalid string: ill-formed UTF-8 byte"; + return token_type::parse_error; + } + } + } + } + + /*! + * @brief scan a comment + * @return whether comment could be scanned successfully + */ + bool scan_comment() + { + switch (get()) + { + // single-line comments skip input until a newline or EOF is read + case '/': + { + while (true) + { + switch (get()) + { + case '\n': + case '\r': + case char_traits::eof(): + case '\0': + return true; + + default: + break; + } + } + } + + // multi-line comments skip input until */ is read + case '*': + { + while (true) + { + switch (get()) + { + case char_traits::eof(): + case '\0': + { + error_message = "invalid comment; missing closing '*/'"; + return false; + } + + case '*': + { + switch (get()) + { + case '/': + return true; + + default: + { + unget(); + continue; + } + } + } + + default: + continue; + } + } + } + + // unexpected character after reading '/' + default: + { + error_message = "invalid comment; expecting '/' or '*' after '/'"; + return false; + } + } + } + + JSON_HEDLEY_NON_NULL(2) + static void strtof(float& f, const char* str, char** endptr) noexcept + { + f = std::strtof(str, endptr); + } + + JSON_HEDLEY_NON_NULL(2) + static void strtof(double& f, const char* str, char** endptr) noexcept + { + f = std::strtod(str, endptr); + } + + JSON_HEDLEY_NON_NULL(2) + static void strtof(long double& f, const char* str, char** endptr) noexcept + { + f = std::strtold(str, endptr); + } + + /*! + @brief scan a number literal + + This function scans a string according to Sect. 6 of RFC 8259. + + The function is realized with a deterministic finite state machine derived + from the grammar described in RFC 8259. Starting in state "init", the + input is read and used to determined the next state. Only state "done" + accepts the number. State "error" is a trap state to model errors. In the + table below, "anything" means any character but the ones listed before. + + state | 0 | 1-9 | e E | + | - | . | anything + ---------|----------|----------|----------|---------|---------|----------|----------- + init | zero | any1 | [error] | [error] | minus | [error] | [error] + minus | zero | any1 | [error] | [error] | [error] | [error] | [error] + zero | done | done | exponent | done | done | decimal1 | done + any1 | any1 | any1 | exponent | done | done | decimal1 | done + decimal1 | decimal2 | decimal2 | [error] | [error] | [error] | [error] | [error] + decimal2 | decimal2 | decimal2 | exponent | done | done | done | done + exponent | any2 | any2 | [error] | sign | sign | [error] | [error] + sign | any2 | any2 | [error] | [error] | [error] | [error] | [error] + any2 | any2 | any2 | done | done | done | done | done + + The state machine is realized with one label per state (prefixed with + "scan_number_") and `goto` statements between them. The state machine + contains cycles, but any cycle can be left when EOF is read. Therefore, + the function is guaranteed to terminate. + + During scanning, the read bytes are stored in token_buffer. This string is + then converted to a signed integer, an unsigned integer, or a + floating-point number. + + @return token_type::value_unsigned, token_type::value_integer, or + token_type::value_float if number could be successfully scanned, + token_type::parse_error otherwise + + @note The scanner is independent of the current locale. Internally, the + locale's decimal point is used instead of `.` to work with the + locale-dependent converters. + */ + token_type scan_number() // lgtm [cpp/use-of-goto] `goto` is used in this function to implement the number-parsing state machine described above. By design, any finite input will eventually reach the "done" state or return token_type::parse_error. In each intermediate state, 1 byte of the input is appended to the token_buffer vector, and only the already initialized variables token_buffer, number_type, and error_message are manipulated. + { + // reset token_buffer to store the number's bytes + reset(); + + // the type of the parsed number; initially set to unsigned; will be + // changed if minus sign, decimal point, or exponent is read + token_type number_type = token_type::value_unsigned; + + // state (init): we just found out we need to scan a number + switch (current) + { + case '-': + { + add(current); + goto scan_number_minus; + } + + case '0': + { + add(current); + goto scan_number_zero; + } + + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + { + add(current); + goto scan_number_any1; + } + + // all other characters are rejected outside scan_number() + default: // LCOV_EXCL_LINE + JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE + } + +scan_number_minus: + // state: we just parsed a leading minus sign + number_type = token_type::value_integer; + switch (get()) + { + case '0': + { + add(current); + goto scan_number_zero; + } + + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + { + add(current); + goto scan_number_any1; + } + + default: + { + error_message = "invalid number; expected digit after '-'"; + return token_type::parse_error; + } + } + +scan_number_zero: + // state: we just parse a zero (maybe with a leading minus sign) + switch (get()) + { + case '.': + { + add(decimal_point_char); + decimal_point_position = token_buffer.size() - 1; + goto scan_number_decimal1; + } + + case 'e': + case 'E': + { + add(current); + goto scan_number_exponent; + } + + default: + goto scan_number_done; + } + +scan_number_any1: + // state: we just parsed a number 0-9 (maybe with a leading minus sign) + switch (get()) + { + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + { + add(current); + goto scan_number_any1; + } + + case '.': + { + add(decimal_point_char); + decimal_point_position = token_buffer.size() - 1; + goto scan_number_decimal1; + } + + case 'e': + case 'E': + { + add(current); + goto scan_number_exponent; + } + + default: + goto scan_number_done; + } + +scan_number_decimal1: + // state: we just parsed a decimal point + number_type = token_type::value_float; + switch (get()) + { + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + { + add(current); + goto scan_number_decimal2; + } + + default: + { + error_message = "invalid number; expected digit after '.'"; + return token_type::parse_error; + } + } + +scan_number_decimal2: + // we just parsed at least one number after a decimal point + switch (get()) + { + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + { + add(current); + goto scan_number_decimal2; + } + + case 'e': + case 'E': + { + add(current); + goto scan_number_exponent; + } + + default: + goto scan_number_done; + } + +scan_number_exponent: + // we just parsed an exponent + number_type = token_type::value_float; + switch (get()) + { + case '+': + case '-': + { + add(current); + goto scan_number_sign; + } + + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + { + add(current); + goto scan_number_any2; + } + + default: + { + error_message = + "invalid number; expected '+', '-', or digit after exponent"; + return token_type::parse_error; + } + } + +scan_number_sign: + // we just parsed an exponent sign + switch (get()) + { + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + { + add(current); + goto scan_number_any2; + } + + default: + { + error_message = "invalid number; expected digit after exponent sign"; + return token_type::parse_error; + } + } + +scan_number_any2: + // we just parsed a number after the exponent or exponent sign + switch (get()) + { + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + { + add(current); + goto scan_number_any2; + } + + default: + goto scan_number_done; + } + +scan_number_done: + // unget the character after the number (we only read it to know that + // we are done scanning a number) + unget(); + + char* endptr = nullptr; // NOLINT(misc-const-correctness,cppcoreguidelines-pro-type-vararg,hicpp-vararg) + errno = 0; + + // try to parse integers first and fall back to floats + if (number_type == token_type::value_unsigned) + { + const auto x = std::strtoull(token_buffer.data(), &endptr, 10); + + // we checked the number format before + JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); + + if (errno != ERANGE) + { + value_unsigned = static_cast(x); + if (value_unsigned == x) + { + return token_type::value_unsigned; + } + } + } + else if (number_type == token_type::value_integer) + { + const auto x = std::strtoll(token_buffer.data(), &endptr, 10); + + // we checked the number format before + JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); + + if (errno != ERANGE) + { + value_integer = static_cast(x); + if (value_integer == x) + { + return token_type::value_integer; + } + } + } + + // this code is reached if we parse a floating-point number or if an + // integer conversion above failed + strtof(value_float, token_buffer.data(), &endptr); + + // we checked the number format before + JSON_ASSERT(endptr == token_buffer.data() + token_buffer.size()); + + return token_type::value_float; + } + + /*! + @param[in] literal_text the literal text to expect + @param[in] length the length of the passed literal text + @param[in] return_type the token type to return on success + */ + JSON_HEDLEY_NON_NULL(2) + token_type scan_literal(const char_type* literal_text, const std::size_t length, + token_type return_type) + { + JSON_ASSERT(char_traits::to_char_type(current) == literal_text[0]); + for (std::size_t i = 1; i < length; ++i) + { + if (JSON_HEDLEY_UNLIKELY(char_traits::to_char_type(get()) != literal_text[i])) + { + error_message = "invalid literal"; + return token_type::parse_error; + } + } + return return_type; + } + + ///////////////////// + // input management + ///////////////////// + + /// reset token_buffer; current character is beginning of token + void reset() noexcept + { + token_buffer.clear(); + token_string.clear(); + decimal_point_position = std::string::npos; + token_string.push_back(char_traits::to_char_type(current)); + } + + /* + @brief get next character from the input + + This function provides the interface to the used input adapter. It does + not throw in case the input reached EOF, but returns a + `char_traits::eof()` in that case. Stores the scanned characters + for use in error messages. + + @return character read from the input + */ + char_int_type get() + { + ++position.chars_read_total; + ++position.chars_read_current_line; + + if (next_unget) + { + // only reset the next_unget variable and work with current + next_unget = false; + } + else + { + current = ia.get_character(); + } + + if (JSON_HEDLEY_LIKELY(current != char_traits::eof())) + { + token_string.push_back(char_traits::to_char_type(current)); + } + + if (current == '\n') + { + ++position.lines_read; + position.chars_read_current_line = 0; + } + + return current; + } + + /*! + @brief unget current character (read it again on next get) + + We implement unget by setting variable next_unget to true. The input is not + changed - we just simulate ungetting by modifying chars_read_total, + chars_read_current_line, and token_string. The next call to get() will + behave as if the unget character is read again. + */ + void unget() + { + next_unget = true; + + --position.chars_read_total; + + // in case we "unget" a newline, we have to also decrement the lines_read + if (position.chars_read_current_line == 0) + { + if (position.lines_read > 0) + { + --position.lines_read; + } + } + else + { + --position.chars_read_current_line; + } + + if (JSON_HEDLEY_LIKELY(current != char_traits::eof())) + { + JSON_ASSERT(!token_string.empty()); + token_string.pop_back(); + } + } + + /// add a character to token_buffer + void add(char_int_type c) + { + token_buffer.push_back(static_cast(c)); + } + + public: + ///////////////////// + // value getters + ///////////////////// + + /// return integer value + constexpr number_integer_t get_number_integer() const noexcept + { + return value_integer; + } + + /// return unsigned integer value + constexpr number_unsigned_t get_number_unsigned() const noexcept + { + return value_unsigned; + } + + /// return floating-point value + constexpr number_float_t get_number_float() const noexcept + { + return value_float; + } + + /// return current string value (implicitly resets the token; useful only once) + string_t& get_string() + { + // translate decimal points from locale back to '.' (#4084) + if (decimal_point_char != '.' && decimal_point_position != std::string::npos) + { + token_buffer[decimal_point_position] = '.'; + } + return token_buffer; + } + + ///////////////////// + // diagnostics + ///////////////////// + + /// return position of last read token + constexpr position_t get_position() const noexcept + { + return position; + } + + /// return the last read token (for errors only). Will never contain EOF + /// (an arbitrary value that is not a valid char value, often -1), because + /// 255 may legitimately occur. May contain NUL, which should be escaped. + std::string get_token_string() const + { + // escape control characters + std::string result; + for (const auto c : token_string) + { + if (static_cast(c) <= '\x1F') + { + // escape control characters + std::array cs{{}}; + static_cast((std::snprintf)(cs.data(), cs.size(), "", static_cast(c))); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg) + result += cs.data(); + } + else + { + // add character as is + result.push_back(static_cast(c)); + } + } + + return result; + } + + /// return syntax error message + JSON_HEDLEY_RETURNS_NON_NULL + constexpr const char* get_error_message() const noexcept + { + return error_message; + } + + ///////////////////// + // actual scanner + ///////////////////// + + /*! + @brief skip the UTF-8 byte order mark + @return true iff there is no BOM or the correct BOM has been skipped + */ + bool skip_bom() + { + if (get() == 0xEF) + { + // check if we completely parse the BOM + return get() == 0xBB && get() == 0xBF; + } + + // the first character is not the beginning of the BOM; unget it to + // process is later + unget(); + return true; + } + + void skip_whitespace() + { + do + { + get(); + } + while (current == ' ' || current == '\t' || current == '\n' || current == '\r'); + } + + token_type scan() + { + // initially, skip the BOM + if (position.chars_read_total == 0 && !skip_bom()) + { + error_message = "invalid BOM; must be 0xEF 0xBB 0xBF if given"; + return token_type::parse_error; + } + + // read the next character and ignore whitespace + skip_whitespace(); + + // ignore comments + while (ignore_comments && current == '/') + { + if (!scan_comment()) + { + return token_type::parse_error; + } + + // skip following whitespace + skip_whitespace(); + } + + switch (current) + { + // structural characters + case '[': + return token_type::begin_array; + case ']': + return token_type::end_array; + case '{': + return token_type::begin_object; + case '}': + return token_type::end_object; + case ':': + return token_type::name_separator; + case ',': + return token_type::value_separator; + + // literals + case 't': + { + std::array true_literal = {{static_cast('t'), static_cast('r'), static_cast('u'), static_cast('e')}}; + return scan_literal(true_literal.data(), true_literal.size(), token_type::literal_true); + } + case 'f': + { + std::array false_literal = {{static_cast('f'), static_cast('a'), static_cast('l'), static_cast('s'), static_cast('e')}}; + return scan_literal(false_literal.data(), false_literal.size(), token_type::literal_false); + } + case 'n': + { + std::array null_literal = {{static_cast('n'), static_cast('u'), static_cast('l'), static_cast('l')}}; + return scan_literal(null_literal.data(), null_literal.size(), token_type::literal_null); + } + + // string + case '\"': + return scan_string(); + + // number + case '-': + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + return scan_number(); + + // end of input (the null byte is needed when parsing from + // string literals) + case '\0': + case char_traits::eof(): + return token_type::end_of_input; + + // error + default: + error_message = "invalid literal"; + return token_type::parse_error; + } + } + + private: + /// input adapter + InputAdapterType ia; + + /// whether comments should be ignored (true) or signaled as errors (false) + const bool ignore_comments = false; + + /// the current character + char_int_type current = char_traits::eof(); + + /// whether the next get() call should just return current + bool next_unget = false; + + /// the start position of the current token + position_t position {}; + + /// raw input token string (for error messages) + std::vector token_string {}; + + /// buffer for variable-length tokens (numbers, strings) + string_t token_buffer {}; + + /// a description of occurred lexer errors + const char* error_message = ""; + + // number values + number_integer_t value_integer = 0; + number_unsigned_t value_unsigned = 0; + number_float_t value_float = 0; + + /// the decimal point + const char_int_type decimal_point_char = '.'; + /// the position of the decimal point in the input + std::size_t decimal_point_position = std::string::npos; +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/input/parser.hpp b/src/detail/include/nlohmann/detail/input/parser.hpp new file mode 100644 index 000000000..c3839ea58 --- /dev/null +++ b/src/detail/include/nlohmann/detail/input/parser.hpp @@ -0,0 +1,519 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // isfinite +#include // uint8_t +#include // function +#include // string +#include // move +#include // vector + +#include +#include +#include +#include +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +//////////// +// parser // +//////////// + +enum class parse_event_t : std::uint8_t +{ + /// the parser read `{` and started to process a JSON object + object_start, + /// the parser read `}` and finished processing a JSON object + object_end, + /// the parser read `[` and started to process a JSON array + array_start, + /// the parser read `]` and finished processing a JSON array + array_end, + /// the parser read a key of a value in an object + key, + /// the parser finished reading a JSON value + value +}; + +template +using parser_callback_t = + std::function; + +/*! +@brief syntax analysis + +This class implements a recursive descent parser. +*/ +template +class parser +{ + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using string_t = typename BasicJsonType::string_t; + using lexer_t = lexer; + using token_type = typename lexer_t::token_type; + + public: + /// a parser reading from an input adapter + explicit parser(InputAdapterType&& adapter, + parser_callback_t cb = nullptr, + const bool allow_exceptions_ = true, + const bool skip_comments = false) + : callback(std::move(cb)) + , m_lexer(std::move(adapter), skip_comments) + , allow_exceptions(allow_exceptions_) + { + // read first token + get_token(); + } + + /*! + @brief public parser interface + + @param[in] strict whether to expect the last token to be EOF + @param[in,out] result parsed JSON value + + @throw parse_error.101 in case of an unexpected token + @throw parse_error.102 if to_unicode fails or surrogate error + @throw parse_error.103 if to_unicode fails + */ + void parse(const bool strict, BasicJsonType& result) + { + if (callback) + { + json_sax_dom_callback_parser sdp(result, callback, allow_exceptions, &m_lexer); + sax_parse_internal(&sdp); + + // in strict mode, input must be completely read + if (strict && (get_token() != token_type::end_of_input)) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), + exception_message(token_type::end_of_input, "value"), nullptr)); + } + + // in case of an error, return a discarded value + if (sdp.is_errored()) + { + result = value_t::discarded; + return; + } + + // set top-level value to null if it was discarded by the callback + // function + if (result.is_discarded()) + { + result = nullptr; + } + } + else + { + json_sax_dom_parser sdp(result, allow_exceptions, &m_lexer); + sax_parse_internal(&sdp); + + // in strict mode, input must be completely read + if (strict && (get_token() != token_type::end_of_input)) + { + sdp.parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + + // in case of an error, return a discarded value + if (sdp.is_errored()) + { + result = value_t::discarded; + return; + } + } + + result.assert_invariant(); + } + + /*! + @brief public accept interface + + @param[in] strict whether to expect the last token to be EOF + @return whether the input is a proper JSON text + */ + bool accept(const bool strict = true) + { + json_sax_acceptor sax_acceptor; + return sax_parse(&sax_acceptor, strict); + } + + template + JSON_HEDLEY_NON_NULL(2) + bool sax_parse(SAX* sax, const bool strict = true) + { + (void)detail::is_sax_static_asserts {}; + const bool result = sax_parse_internal(sax); + + // strict mode: next byte must be EOF + if (result && strict && (get_token() != token_type::end_of_input)) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_of_input, "value"), nullptr)); + } + + return result; + } + + private: + template + JSON_HEDLEY_NON_NULL(2) + bool sax_parse_internal(SAX* sax) + { + // stack to remember the hierarchy of structured values we are parsing + // true = array; false = object + std::vector states; + // value to avoid a goto (see comment where set to true) + bool skip_to_state_evaluation = false; + + while (true) + { + if (!skip_to_state_evaluation) + { + // invariant: get_token() was called before each iteration + switch (last_token) + { + case token_type::begin_object: + { + if (JSON_HEDLEY_UNLIKELY(!sax->start_object(detail::unknown_size()))) + { + return false; + } + + // closing } -> we are done + if (get_token() == token_type::end_object) + { + if (JSON_HEDLEY_UNLIKELY(!sax->end_object())) + { + return false; + } + break; + } + + // parse key + if (JSON_HEDLEY_UNLIKELY(last_token != token_type::value_string)) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr)); + } + if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string()))) + { + return false; + } + + // parse separator (:) + if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator)) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr)); + } + + // remember we are now inside an object + states.push_back(false); + + // parse values + get_token(); + continue; + } + + case token_type::begin_array: + { + if (JSON_HEDLEY_UNLIKELY(!sax->start_array(detail::unknown_size()))) + { + return false; + } + + // closing ] -> we are done + if (get_token() == token_type::end_array) + { + if (JSON_HEDLEY_UNLIKELY(!sax->end_array())) + { + return false; + } + break; + } + + // remember we are now inside an array + states.push_back(true); + + // parse values (no need to call get_token) + continue; + } + + case token_type::value_float: + { + const auto res = m_lexer.get_number_float(); + + if (JSON_HEDLEY_UNLIKELY(!std::isfinite(res))) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + out_of_range::create(406, concat("number overflow parsing '", m_lexer.get_token_string(), '\''), nullptr)); + } + + if (JSON_HEDLEY_UNLIKELY(!sax->number_float(res, m_lexer.get_string()))) + { + return false; + } + + break; + } + + case token_type::literal_false: + { + if (JSON_HEDLEY_UNLIKELY(!sax->boolean(false))) + { + return false; + } + break; + } + + case token_type::literal_null: + { + if (JSON_HEDLEY_UNLIKELY(!sax->null())) + { + return false; + } + break; + } + + case token_type::literal_true: + { + if (JSON_HEDLEY_UNLIKELY(!sax->boolean(true))) + { + return false; + } + break; + } + + case token_type::value_integer: + { + if (JSON_HEDLEY_UNLIKELY(!sax->number_integer(m_lexer.get_number_integer()))) + { + return false; + } + break; + } + + case token_type::value_string: + { + if (JSON_HEDLEY_UNLIKELY(!sax->string(m_lexer.get_string()))) + { + return false; + } + break; + } + + case token_type::value_unsigned: + { + if (JSON_HEDLEY_UNLIKELY(!sax->number_unsigned(m_lexer.get_number_unsigned()))) + { + return false; + } + break; + } + + case token_type::parse_error: + { + // using "uninitialized" to avoid an "expected" message + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::uninitialized, "value"), nullptr)); + } + case token_type::end_of_input: + { + if (JSON_HEDLEY_UNLIKELY(m_lexer.get_position().chars_read_total == 1)) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), + "attempting to parse an empty input; check that your input string or stream contains the expected JSON", nullptr)); + } + + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr)); + } + case token_type::uninitialized: + case token_type::end_array: + case token_type::end_object: + case token_type::name_separator: + case token_type::value_separator: + case token_type::literal_or_value: + default: // the last token was unexpected + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::literal_or_value, "value"), nullptr)); + } + } + } + else + { + skip_to_state_evaluation = false; + } + + // we reached this line after we successfully parsed a value + if (states.empty()) + { + // empty stack: we reached the end of the hierarchy: done + return true; + } + + if (states.back()) // array + { + // comma -> next value + if (get_token() == token_type::value_separator) + { + // parse a new value + get_token(); + continue; + } + + // closing ] + if (JSON_HEDLEY_LIKELY(last_token == token_type::end_array)) + { + if (JSON_HEDLEY_UNLIKELY(!sax->end_array())) + { + return false; + } + + // We are done with this array. Before we can parse a + // new value, we need to evaluate the new state first. + // By setting skip_to_state_evaluation to false, we + // are effectively jumping to the beginning of this if. + JSON_ASSERT(!states.empty()); + states.pop_back(); + skip_to_state_evaluation = true; + continue; + } + + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_array, "array"), nullptr)); + } + + // states.back() is false -> object + + // comma -> next value + if (get_token() == token_type::value_separator) + { + // parse key + if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::value_string)) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::value_string, "object key"), nullptr)); + } + + if (JSON_HEDLEY_UNLIKELY(!sax->key(m_lexer.get_string()))) + { + return false; + } + + // parse separator (:) + if (JSON_HEDLEY_UNLIKELY(get_token() != token_type::name_separator)) + { + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::name_separator, "object separator"), nullptr)); + } + + // parse values + get_token(); + continue; + } + + // closing } + if (JSON_HEDLEY_LIKELY(last_token == token_type::end_object)) + { + if (JSON_HEDLEY_UNLIKELY(!sax->end_object())) + { + return false; + } + + // We are done with this object. Before we can parse a + // new value, we need to evaluate the new state first. + // By setting skip_to_state_evaluation to false, we + // are effectively jumping to the beginning of this if. + JSON_ASSERT(!states.empty()); + states.pop_back(); + skip_to_state_evaluation = true; + continue; + } + + return sax->parse_error(m_lexer.get_position(), + m_lexer.get_token_string(), + parse_error::create(101, m_lexer.get_position(), exception_message(token_type::end_object, "object"), nullptr)); + } + } + + /// get next token from lexer + token_type get_token() + { + return last_token = m_lexer.scan(); + } + + std::string exception_message(const token_type expected, const std::string& context) + { + std::string error_msg = "syntax error "; + + if (!context.empty()) + { + error_msg += concat("while parsing ", context, ' '); + } + + error_msg += "- "; + + if (last_token == token_type::parse_error) + { + error_msg += concat(m_lexer.get_error_message(), "; last read: '", + m_lexer.get_token_string(), '\''); + } + else + { + error_msg += concat("unexpected ", lexer_t::token_type_name(last_token)); + } + + if (expected != token_type::uninitialized) + { + error_msg += concat("; expected ", lexer_t::token_type_name(expected)); + } + + return error_msg; + } + + private: + /// callback function + const parser_callback_t callback = nullptr; + /// the type of the last read token + token_type last_token = token_type::uninitialized; + /// the lexer + lexer_t m_lexer; + /// whether to throw exceptions in case of errors + const bool allow_exceptions = true; +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/input/position_t.hpp b/src/detail/include/nlohmann/detail/input/position_t.hpp new file mode 100644 index 000000000..e02ba24b8 --- /dev/null +++ b/src/detail/include/nlohmann/detail/input/position_t.hpp @@ -0,0 +1,37 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/// struct to capture the start position of the current token +struct position_t +{ + /// the total number of characters read + std::size_t chars_read_total = 0; + /// the number of characters read in the current line + std::size_t chars_read_current_line = 0; + /// the number of lines read + std::size_t lines_read = 0; + + /// conversion to size_t to preserve SAX interface + constexpr operator size_t() const + { + return chars_read_total; + } +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/iterators/internal_iterator.hpp b/src/detail/include/nlohmann/detail/iterators/internal_iterator.hpp new file mode 100644 index 000000000..1da4aeff4 --- /dev/null +++ b/src/detail/include/nlohmann/detail/iterators/internal_iterator.hpp @@ -0,0 +1,35 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief an iterator value + +@note This structure could easily be a union, but MSVC currently does not allow +unions members with complex constructors, see https://github.com/nlohmann/json/pull/105. +*/ +template struct internal_iterator +{ + /// iterator for JSON objects + typename BasicJsonType::object_t::iterator object_iterator {}; + /// iterator for JSON arrays + typename BasicJsonType::array_t::iterator array_iterator {}; + /// generic iterator for all other types + primitive_iterator_t primitive_iterator {}; +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/iterators/iter_impl.hpp b/src/detail/include/nlohmann/detail/iterators/iter_impl.hpp new file mode 100644 index 000000000..3e46e9474 --- /dev/null +++ b/src/detail/include/nlohmann/detail/iterators/iter_impl.hpp @@ -0,0 +1,760 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // iterator, random_access_iterator_tag, bidirectional_iterator_tag, advance, next +#include // conditional, is_const, remove_const + +#include +#include +#include +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +// forward declare to be able to friend it later on +template class iteration_proxy; +template class iteration_proxy_value; + +/*! +@brief a template for a bidirectional iterator for the @ref basic_json class +This class implements a both iterators (iterator and const_iterator) for the +@ref basic_json class. +@note An iterator is called *initialized* when a pointer to a JSON value has + been set (e.g., by a constructor or a copy assignment). If the iterator is + default-constructed, it is *uninitialized* and most methods are undefined. + **The library uses assertions to detect calls on uninitialized iterators.** +@requirement The class satisfies the following concept requirements: +- +[BidirectionalIterator](https://en.cppreference.com/w/cpp/named_req/BidirectionalIterator): + The iterator that can be moved can be moved in both directions (i.e. + incremented and decremented). +@since version 1.0.0, simplified in version 2.0.9, change to bidirectional + iterators in version 3.0.0 (see https://github.com/nlohmann/json/issues/593) +*/ +template +class iter_impl // NOLINT(cppcoreguidelines-special-member-functions,hicpp-special-member-functions) +{ + /// the iterator with BasicJsonType of different const-ness + using other_iter_impl = iter_impl::value, typename std::remove_const::type, const BasicJsonType>::type>; + /// allow basic_json to access private members + friend other_iter_impl; + friend BasicJsonType; + friend iteration_proxy; + friend iteration_proxy_value; + + using object_t = typename BasicJsonType::object_t; + using array_t = typename BasicJsonType::array_t; + // make sure BasicJsonType is basic_json or const basic_json + static_assert(is_basic_json::type>::value, + "iter_impl only accepts (const) basic_json"); + // superficial check for the LegacyBidirectionalIterator named requirement + static_assert(std::is_base_of::value + && std::is_base_of::iterator_category>::value, + "basic_json iterator assumes array and object type iterators satisfy the LegacyBidirectionalIterator named requirement."); + + public: + /// The std::iterator class template (used as a base class to provide typedefs) is deprecated in C++17. + /// The C++ Standard has never required user-defined iterators to derive from std::iterator. + /// A user-defined iterator should provide publicly accessible typedefs named + /// iterator_category, value_type, difference_type, pointer, and reference. + /// Note that value_type is required to be non-const, even for constant iterators. + using iterator_category = std::bidirectional_iterator_tag; + + /// the type of the values when the iterator is dereferenced + using value_type = typename BasicJsonType::value_type; + /// a type to represent differences between iterators + using difference_type = typename BasicJsonType::difference_type; + /// defines a pointer to the type iterated over (value_type) + using pointer = typename std::conditional::value, + typename BasicJsonType::const_pointer, + typename BasicJsonType::pointer>::type; + /// defines a reference to the type iterated over (value_type) + using reference = + typename std::conditional::value, + typename BasicJsonType::const_reference, + typename BasicJsonType::reference>::type; + + iter_impl() = default; + ~iter_impl() = default; + iter_impl(iter_impl&&) noexcept = default; + iter_impl& operator=(iter_impl&&) noexcept = default; + + /*! + @brief constructor for a given JSON instance + @param[in] object pointer to a JSON object for this iterator + @pre object != nullptr + @post The iterator is initialized; i.e. `m_object != nullptr`. + */ + explicit iter_impl(pointer object) noexcept : m_object(object) + { + JSON_ASSERT(m_object != nullptr); + + switch (m_object->m_data.m_type) + { + case value_t::object: + { + m_it.object_iterator = typename object_t::iterator(); + break; + } + + case value_t::array: + { + m_it.array_iterator = typename array_t::iterator(); + break; + } + + case value_t::null: + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + { + m_it.primitive_iterator = primitive_iterator_t(); + break; + } + } + } + + /*! + @note The conventional copy constructor and copy assignment are implicitly + defined. Combined with the following converting constructor and + assignment, they support: (1) copy from iterator to iterator, (2) + copy from const iterator to const iterator, and (3) conversion from + iterator to const iterator. However conversion from const iterator + to iterator is not defined. + */ + + /*! + @brief const copy constructor + @param[in] other const iterator to copy from + @note This copy constructor had to be defined explicitly to circumvent a bug + occurring on msvc v19.0 compiler (VS 2015) debug build. For more + information refer to: https://github.com/nlohmann/json/issues/1608 + */ + iter_impl(const iter_impl& other) noexcept + : m_object(other.m_object), m_it(other.m_it) + {} + + /*! + @brief converting assignment + @param[in] other const iterator to copy from + @return const/non-const iterator + @note It is not checked whether @a other is initialized. + */ + iter_impl& operator=(const iter_impl& other) noexcept + { + if (&other != this) + { + m_object = other.m_object; + m_it = other.m_it; + } + return *this; + } + + /*! + @brief converting constructor + @param[in] other non-const iterator to copy from + @note It is not checked whether @a other is initialized. + */ + iter_impl(const iter_impl::type>& other) noexcept + : m_object(other.m_object), m_it(other.m_it) + {} + + /*! + @brief converting assignment + @param[in] other non-const iterator to copy from + @return const/non-const iterator + @note It is not checked whether @a other is initialized. + */ + iter_impl& operator=(const iter_impl::type>& other) noexcept // NOLINT(cert-oop54-cpp) + { + m_object = other.m_object; + m_it = other.m_it; + return *this; + } + + JSON_PRIVATE_UNLESS_TESTED: + /*! + @brief set the iterator to the first value + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + void set_begin() noexcept + { + JSON_ASSERT(m_object != nullptr); + + switch (m_object->m_data.m_type) + { + case value_t::object: + { + m_it.object_iterator = m_object->m_data.m_value.object->begin(); + break; + } + + case value_t::array: + { + m_it.array_iterator = m_object->m_data.m_value.array->begin(); + break; + } + + case value_t::null: + { + // set to end so begin()==end() is true: null is empty + m_it.primitive_iterator.set_end(); + break; + } + + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + { + m_it.primitive_iterator.set_begin(); + break; + } + } + } + + /*! + @brief set the iterator past the last value + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + void set_end() noexcept + { + JSON_ASSERT(m_object != nullptr); + + switch (m_object->m_data.m_type) + { + case value_t::object: + { + m_it.object_iterator = m_object->m_data.m_value.object->end(); + break; + } + + case value_t::array: + { + m_it.array_iterator = m_object->m_data.m_value.array->end(); + break; + } + + case value_t::null: + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + { + m_it.primitive_iterator.set_end(); + break; + } + } + } + + public: + /*! + @brief return a reference to the value pointed to by the iterator + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + reference operator*() const + { + JSON_ASSERT(m_object != nullptr); + + switch (m_object->m_data.m_type) + { + case value_t::object: + { + JSON_ASSERT(m_it.object_iterator != m_object->m_data.m_value.object->end()); + return m_it.object_iterator->second; + } + + case value_t::array: + { + JSON_ASSERT(m_it.array_iterator != m_object->m_data.m_value.array->end()); + return *m_it.array_iterator; + } + + case value_t::null: + JSON_THROW(invalid_iterator::create(214, "cannot get value", m_object)); + + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + { + if (JSON_HEDLEY_LIKELY(m_it.primitive_iterator.is_begin())) + { + return *m_object; + } + + JSON_THROW(invalid_iterator::create(214, "cannot get value", m_object)); + } + } + } + + /*! + @brief dereference the iterator + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + pointer operator->() const + { + JSON_ASSERT(m_object != nullptr); + + switch (m_object->m_data.m_type) + { + case value_t::object: + { + JSON_ASSERT(m_it.object_iterator != m_object->m_data.m_value.object->end()); + return &(m_it.object_iterator->second); + } + + case value_t::array: + { + JSON_ASSERT(m_it.array_iterator != m_object->m_data.m_value.array->end()); + return &*m_it.array_iterator; + } + + case value_t::null: + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + { + if (JSON_HEDLEY_LIKELY(m_it.primitive_iterator.is_begin())) + { + return m_object; + } + + JSON_THROW(invalid_iterator::create(214, "cannot get value", m_object)); + } + } + } + + /*! + @brief post-increment (it++) + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + iter_impl operator++(int)& // NOLINT(cert-dcl21-cpp) + { + auto result = *this; + ++(*this); + return result; + } + + /*! + @brief pre-increment (++it) + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + iter_impl& operator++() + { + JSON_ASSERT(m_object != nullptr); + + switch (m_object->m_data.m_type) + { + case value_t::object: + { + std::advance(m_it.object_iterator, 1); + break; + } + + case value_t::array: + { + std::advance(m_it.array_iterator, 1); + break; + } + + case value_t::null: + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + { + ++m_it.primitive_iterator; + break; + } + } + + return *this; + } + + /*! + @brief post-decrement (it--) + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + iter_impl operator--(int)& // NOLINT(cert-dcl21-cpp) + { + auto result = *this; + --(*this); + return result; + } + + /*! + @brief pre-decrement (--it) + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + iter_impl& operator--() + { + JSON_ASSERT(m_object != nullptr); + + switch (m_object->m_data.m_type) + { + case value_t::object: + { + std::advance(m_it.object_iterator, -1); + break; + } + + case value_t::array: + { + std::advance(m_it.array_iterator, -1); + break; + } + + case value_t::null: + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + { + --m_it.primitive_iterator; + break; + } + } + + return *this; + } + + /*! + @brief comparison: equal + @pre (1) Both iterators are initialized to point to the same object, or (2) both iterators are value-initialized. + */ + template < typename IterImpl, detail::enable_if_t < (std::is_same::value || std::is_same::value), std::nullptr_t > = nullptr > + bool operator==(const IterImpl& other) const + { + // if objects are not the same, the comparison is undefined + if (JSON_HEDLEY_UNLIKELY(m_object != other.m_object)) + { + JSON_THROW(invalid_iterator::create(212, "cannot compare iterators of different containers", m_object)); + } + + // value-initialized forward iterators can be compared, and must compare equal to other value-initialized iterators of the same type #4493 + if (m_object == nullptr) + { + return true; + } + + switch (m_object->m_data.m_type) + { + case value_t::object: + return (m_it.object_iterator == other.m_it.object_iterator); + + case value_t::array: + return (m_it.array_iterator == other.m_it.array_iterator); + + case value_t::null: + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + return (m_it.primitive_iterator == other.m_it.primitive_iterator); + } + } + + /*! + @brief comparison: not equal + @pre (1) Both iterators are initialized to point to the same object, or (2) both iterators are value-initialized. + */ + template < typename IterImpl, detail::enable_if_t < (std::is_same::value || std::is_same::value), std::nullptr_t > = nullptr > + bool operator!=(const IterImpl& other) const + { + return !operator==(other); + } + + /*! + @brief comparison: smaller + @pre (1) Both iterators are initialized to point to the same object, or (2) both iterators are value-initialized. + */ + bool operator<(const iter_impl& other) const + { + // if objects are not the same, the comparison is undefined + if (JSON_HEDLEY_UNLIKELY(m_object != other.m_object)) + { + JSON_THROW(invalid_iterator::create(212, "cannot compare iterators of different containers", m_object)); + } + + // value-initialized forward iterators can be compared, and must compare equal to other value-initialized iterators of the same type #4493 + if (m_object == nullptr) + { + // the iterators are both value-initialized and are to be considered equal, but this function checks for smaller, so we return false + return false; + } + + switch (m_object->m_data.m_type) + { + case value_t::object: + JSON_THROW(invalid_iterator::create(213, "cannot compare order of object iterators", m_object)); + + case value_t::array: + return (m_it.array_iterator < other.m_it.array_iterator); + + case value_t::null: + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + return (m_it.primitive_iterator < other.m_it.primitive_iterator); + } + } + + /*! + @brief comparison: less than or equal + @pre (1) Both iterators are initialized to point to the same object, or (2) both iterators are value-initialized. + */ + bool operator<=(const iter_impl& other) const + { + return !other.operator < (*this); + } + + /*! + @brief comparison: greater than + @pre (1) Both iterators are initialized to point to the same object, or (2) both iterators are value-initialized. + */ + bool operator>(const iter_impl& other) const + { + return !operator<=(other); + } + + /*! + @brief comparison: greater than or equal + @pre (1) The iterator is initialized; i.e. `m_object != nullptr`, or (2) both iterators are value-initialized. + */ + bool operator>=(const iter_impl& other) const + { + return !operator<(other); + } + + /*! + @brief add to iterator + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + iter_impl& operator+=(difference_type i) + { + JSON_ASSERT(m_object != nullptr); + + switch (m_object->m_data.m_type) + { + case value_t::object: + JSON_THROW(invalid_iterator::create(209, "cannot use offsets with object iterators", m_object)); + + case value_t::array: + { + std::advance(m_it.array_iterator, i); + break; + } + + case value_t::null: + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + { + m_it.primitive_iterator += i; + break; + } + } + + return *this; + } + + /*! + @brief subtract from iterator + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + iter_impl& operator-=(difference_type i) + { + return operator+=(-i); + } + + /*! + @brief add to iterator + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + iter_impl operator+(difference_type i) const + { + auto result = *this; + result += i; + return result; + } + + /*! + @brief addition of distance and iterator + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + friend iter_impl operator+(difference_type i, const iter_impl& it) + { + auto result = it; + result += i; + return result; + } + + /*! + @brief subtract from iterator + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + iter_impl operator-(difference_type i) const + { + auto result = *this; + result -= i; + return result; + } + + /*! + @brief return difference + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + difference_type operator-(const iter_impl& other) const + { + JSON_ASSERT(m_object != nullptr); + + switch (m_object->m_data.m_type) + { + case value_t::object: + JSON_THROW(invalid_iterator::create(209, "cannot use offsets with object iterators", m_object)); + + case value_t::array: + return m_it.array_iterator - other.m_it.array_iterator; + + case value_t::null: + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + return m_it.primitive_iterator - other.m_it.primitive_iterator; + } + } + + /*! + @brief access to successor + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + reference operator[](difference_type n) const + { + JSON_ASSERT(m_object != nullptr); + + switch (m_object->m_data.m_type) + { + case value_t::object: + JSON_THROW(invalid_iterator::create(208, "cannot use operator[] for object iterators", m_object)); + + case value_t::array: + return *std::next(m_it.array_iterator, n); + + case value_t::null: + JSON_THROW(invalid_iterator::create(214, "cannot get value", m_object)); + + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + { + if (JSON_HEDLEY_LIKELY(m_it.primitive_iterator.get_value() == -n)) + { + return *m_object; + } + + JSON_THROW(invalid_iterator::create(214, "cannot get value", m_object)); + } + } + } + + /*! + @brief return the key of an object iterator + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + const typename object_t::key_type& key() const + { + JSON_ASSERT(m_object != nullptr); + + if (JSON_HEDLEY_LIKELY(m_object->is_object())) + { + return m_it.object_iterator->first; + } + + JSON_THROW(invalid_iterator::create(207, "cannot use key() for non-object iterators", m_object)); + } + + /*! + @brief return the value of an iterator + @pre The iterator is initialized; i.e. `m_object != nullptr`. + */ + reference value() const + { + return operator*(); + } + + JSON_PRIVATE_UNLESS_TESTED: + /// associated JSON instance + pointer m_object = nullptr; + /// the actual iterator of the associated instance + internal_iterator::type> m_it {}; +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/iterators/iteration_proxy.hpp b/src/detail/include/nlohmann/detail/iterators/iteration_proxy.hpp new file mode 100644 index 000000000..fa3813e1c --- /dev/null +++ b/src/detail/include/nlohmann/detail/iterators/iteration_proxy.hpp @@ -0,0 +1,235 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t +#include // forward_iterator_tag +#include // tuple_size, get, tuple_element +#include // move + +#if JSON_HAS_RANGES + #include // enable_borrowed_range +#endif + +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +template class iteration_proxy_value +{ + public: + using difference_type = std::ptrdiff_t; + using value_type = iteration_proxy_value; + using pointer = value_type *; + using reference = value_type &; + using iterator_category = std::forward_iterator_tag; + using string_type = typename std::remove_cv< typename std::remove_reference().key() ) >::type >::type; + + private: + /// the iterator + IteratorType anchor{}; + /// an index for arrays (used to create key names) + std::size_t array_index = 0; + /// last stringified array index + mutable std::size_t array_index_last = 0; + /// a string representation of the array index + mutable string_type array_index_str = "0"; + /// an empty string (to return a reference for primitive values) + string_type empty_str{}; + + public: + explicit iteration_proxy_value() = default; + explicit iteration_proxy_value(IteratorType it, std::size_t array_index_ = 0) + noexcept(std::is_nothrow_move_constructible::value + && std::is_nothrow_default_constructible::value) + : anchor(std::move(it)) + , array_index(array_index_) + {} + + iteration_proxy_value(iteration_proxy_value const&) = default; + iteration_proxy_value& operator=(iteration_proxy_value const&) = default; + // older GCCs are a bit fussy and require explicit noexcept specifiers on defaulted functions + iteration_proxy_value(iteration_proxy_value&&) + noexcept(std::is_nothrow_move_constructible::value + && std::is_nothrow_move_constructible::value) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor,cppcoreguidelines-noexcept-move-operations) + iteration_proxy_value& operator=(iteration_proxy_value&&) + noexcept(std::is_nothrow_move_assignable::value + && std::is_nothrow_move_assignable::value) = default; // NOLINT(hicpp-noexcept-move,performance-noexcept-move-constructor,cppcoreguidelines-noexcept-move-operations) + ~iteration_proxy_value() = default; + + /// dereference operator (needed for range-based for) + const iteration_proxy_value& operator*() const + { + return *this; + } + + /// increment operator (needed for range-based for) + iteration_proxy_value& operator++() + { + ++anchor; + ++array_index; + + return *this; + } + + iteration_proxy_value operator++(int)& // NOLINT(cert-dcl21-cpp) + { + auto tmp = iteration_proxy_value(anchor, array_index); + ++anchor; + ++array_index; + return tmp; + } + + /// equality operator (needed for InputIterator) + bool operator==(const iteration_proxy_value& o) const + { + return anchor == o.anchor; + } + + /// inequality operator (needed for range-based for) + bool operator!=(const iteration_proxy_value& o) const + { + return anchor != o.anchor; + } + + /// return key of the iterator + const string_type& key() const + { + JSON_ASSERT(anchor.m_object != nullptr); + + switch (anchor.m_object->type()) + { + // use integer array index as key + case value_t::array: + { + if (array_index != array_index_last) + { + int_to_string( array_index_str, array_index ); + array_index_last = array_index; + } + return array_index_str; + } + + // use key from the object + case value_t::object: + return anchor.key(); + + // use an empty key for all primitive types + case value_t::null: + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + case value_t::discarded: + default: + return empty_str; + } + } + + /// return value of the iterator + typename IteratorType::reference value() const + { + return anchor.value(); + } +}; + +/// proxy class for the items() function +template class iteration_proxy +{ + private: + /// the container to iterate + typename IteratorType::pointer container = nullptr; + + public: + explicit iteration_proxy() = default; + + /// construct iteration proxy from a container + explicit iteration_proxy(typename IteratorType::reference cont) noexcept + : container(&cont) {} + + iteration_proxy(iteration_proxy const&) = default; + iteration_proxy& operator=(iteration_proxy const&) = default; + iteration_proxy(iteration_proxy&&) noexcept = default; + iteration_proxy& operator=(iteration_proxy&&) noexcept = default; + ~iteration_proxy() = default; + + /// return iterator begin (needed for range-based for) + iteration_proxy_value begin() const noexcept + { + return iteration_proxy_value(container->begin()); + } + + /// return iterator end (needed for range-based for) + iteration_proxy_value end() const noexcept + { + return iteration_proxy_value(container->end()); + } +}; + +// Structured Bindings Support +// For further reference see https://blog.tartanllama.xyz/structured-bindings/ +// And see https://github.com/nlohmann/json/pull/1391 +template = 0> +auto get(const nlohmann::detail::iteration_proxy_value& i) -> decltype(i.key()) +{ + return i.key(); +} +// Structured Bindings Support +// For further reference see https://blog.tartanllama.xyz/structured-bindings/ +// And see https://github.com/nlohmann/json/pull/1391 +template = 0> +auto get(const nlohmann::detail::iteration_proxy_value& i) -> decltype(i.value()) +{ + return i.value(); +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + +// The Addition to the STD Namespace is required to add +// Structured Bindings Support to the iteration_proxy_value class +// For further reference see https://blog.tartanllama.xyz/structured-bindings/ +// And see https://github.com/nlohmann/json/pull/1391 +namespace std +{ + +#if defined(__clang__) + // Fix: https://github.com/nlohmann/json/issues/1401 + #pragma clang diagnostic push + #pragma clang diagnostic ignored "-Wmismatched-tags" +#endif +template +class tuple_size<::nlohmann::detail::iteration_proxy_value> // NOLINT(cert-dcl58-cpp) + : public std::integral_constant {}; + +template +class tuple_element> // NOLINT(cert-dcl58-cpp) +{ + public: + using type = decltype( + get(std::declval < + ::nlohmann::detail::iteration_proxy_value> ())); +}; +#if defined(__clang__) + #pragma clang diagnostic pop +#endif + +} // namespace std + +#if JSON_HAS_RANGES + template + inline constexpr bool ::std::ranges::enable_borrowed_range<::nlohmann::detail::iteration_proxy> = true; +#endif diff --git a/src/detail/include/nlohmann/detail/iterators/iterator_traits.hpp b/src/detail/include/nlohmann/detail/iterators/iterator_traits.hpp new file mode 100644 index 000000000..fab29d17e --- /dev/null +++ b/src/detail/include/nlohmann/detail/iterators/iterator_traits.hpp @@ -0,0 +1,61 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // random_access_iterator_tag + +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +template +struct iterator_types {}; + +template +struct iterator_types < + It, + void_t> +{ + using difference_type = typename It::difference_type; + using value_type = typename It::value_type; + using pointer = typename It::pointer; + using reference = typename It::reference; + using iterator_category = typename It::iterator_category; +}; + +// This is required as some compilers implement std::iterator_traits in a way that +// doesn't work with SFINAE. See https://github.com/nlohmann/json/issues/1341. +template +struct iterator_traits +{ +}; + +template +struct iterator_traits < T, enable_if_t < !std::is_pointer::value >> + : iterator_types +{ +}; + +template +struct iterator_traits::value>> +{ + using iterator_category = std::random_access_iterator_tag; + using value_type = T; + using difference_type = ptrdiff_t; + using pointer = T*; + using reference = T&; +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/iterators/json_reverse_iterator.hpp b/src/detail/include/nlohmann/detail/iterators/json_reverse_iterator.hpp new file mode 100644 index 000000000..92ba01d7e --- /dev/null +++ b/src/detail/include/nlohmann/detail/iterators/json_reverse_iterator.hpp @@ -0,0 +1,130 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // ptrdiff_t +#include // reverse_iterator +#include // declval + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +////////////////////// +// reverse_iterator // +////////////////////// + +/*! +@brief a template for a reverse iterator class + +@tparam Base the base iterator type to reverse. Valid types are @ref +iterator (to create @ref reverse_iterator) and @ref const_iterator (to +create @ref const_reverse_iterator). + +@requirement The class satisfies the following concept requirements: +- +[BidirectionalIterator](https://en.cppreference.com/w/cpp/named_req/BidirectionalIterator): + The iterator that can be moved can be moved in both directions (i.e. + incremented and decremented). +- [OutputIterator](https://en.cppreference.com/w/cpp/named_req/OutputIterator): + It is possible to write to the pointed-to element (only if @a Base is + @ref iterator). + +@since version 1.0.0 +*/ +template +class json_reverse_iterator : public std::reverse_iterator +{ + public: + using difference_type = std::ptrdiff_t; + /// shortcut to the reverse iterator adapter + using base_iterator = std::reverse_iterator; + /// the reference type for the pointed-to element + using reference = typename Base::reference; + + /// create reverse iterator from iterator + explicit json_reverse_iterator(const typename base_iterator::iterator_type& it) noexcept + : base_iterator(it) {} + + /// create reverse iterator from base class + explicit json_reverse_iterator(const base_iterator& it) noexcept : base_iterator(it) {} + + /// post-increment (it++) + json_reverse_iterator operator++(int)& // NOLINT(cert-dcl21-cpp) + { + return static_cast(base_iterator::operator++(1)); + } + + /// pre-increment (++it) + json_reverse_iterator& operator++() + { + return static_cast(base_iterator::operator++()); + } + + /// post-decrement (it--) + json_reverse_iterator operator--(int)& // NOLINT(cert-dcl21-cpp) + { + return static_cast(base_iterator::operator--(1)); + } + + /// pre-decrement (--it) + json_reverse_iterator& operator--() + { + return static_cast(base_iterator::operator--()); + } + + /// add to iterator + json_reverse_iterator& operator+=(difference_type i) + { + return static_cast(base_iterator::operator+=(i)); + } + + /// add to iterator + json_reverse_iterator operator+(difference_type i) const + { + return static_cast(base_iterator::operator+(i)); + } + + /// subtract from iterator + json_reverse_iterator operator-(difference_type i) const + { + return static_cast(base_iterator::operator-(i)); + } + + /// return difference + difference_type operator-(const json_reverse_iterator& other) const + { + return base_iterator(*this) - base_iterator(other); + } + + /// access to successor + reference operator[](difference_type n) const + { + return *(this->operator+(n)); + } + + /// return the key of an object iterator + auto key() const -> decltype(std::declval().key()) + { + auto it = --this->base(); + return it.key(); + } + + /// return the value of an iterator + reference value() const + { + auto it = --this->base(); + return it.operator * (); + } +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/iterators/primitive_iterator.hpp b/src/detail/include/nlohmann/detail/iterators/primitive_iterator.hpp new file mode 100644 index 000000000..289224efc --- /dev/null +++ b/src/detail/include/nlohmann/detail/iterators/primitive_iterator.hpp @@ -0,0 +1,132 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // ptrdiff_t +#include // numeric_limits + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/* +@brief an iterator for primitive JSON types + +This class models an iterator for primitive JSON types (boolean, number, +string). Its only purpose is to allow the iterator/const_iterator classes +to "iterate" over primitive values. Internally, the iterator is modeled by +a `difference_type` variable. Value begin_value (`0`) models the begin and +end_value (`1`) models past the end. +*/ +class primitive_iterator_t +{ + private: + using difference_type = std::ptrdiff_t; + static constexpr difference_type begin_value = 0; + static constexpr difference_type end_value = begin_value + 1; + + JSON_PRIVATE_UNLESS_TESTED: + /// iterator as signed integer type + difference_type m_it = (std::numeric_limits::min)(); + + public: + constexpr difference_type get_value() const noexcept + { + return m_it; + } + + /// set iterator to a defined beginning + void set_begin() noexcept + { + m_it = begin_value; + } + + /// set iterator to a defined past the end + void set_end() noexcept + { + m_it = end_value; + } + + /// return whether the iterator can be dereferenced + constexpr bool is_begin() const noexcept + { + return m_it == begin_value; + } + + /// return whether the iterator is at end + constexpr bool is_end() const noexcept + { + return m_it == end_value; + } + + friend constexpr bool operator==(primitive_iterator_t lhs, primitive_iterator_t rhs) noexcept + { + return lhs.m_it == rhs.m_it; + } + + friend constexpr bool operator<(primitive_iterator_t lhs, primitive_iterator_t rhs) noexcept + { + return lhs.m_it < rhs.m_it; + } + + primitive_iterator_t operator+(difference_type n) noexcept + { + auto result = *this; + result += n; + return result; + } + + friend constexpr difference_type operator-(primitive_iterator_t lhs, primitive_iterator_t rhs) noexcept + { + return lhs.m_it - rhs.m_it; + } + + primitive_iterator_t& operator++() noexcept + { + ++m_it; + return *this; + } + + primitive_iterator_t operator++(int)& noexcept // NOLINT(cert-dcl21-cpp) + { + auto result = *this; + ++m_it; + return result; + } + + primitive_iterator_t& operator--() noexcept + { + --m_it; + return *this; + } + + primitive_iterator_t operator--(int)& noexcept // NOLINT(cert-dcl21-cpp) + { + auto result = *this; + --m_it; + return result; + } + + primitive_iterator_t& operator+=(difference_type n) noexcept + { + m_it += n; + return *this; + } + + primitive_iterator_t& operator-=(difference_type n) noexcept + { + m_it -= n; + return *this; + } +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/json_custom_base_class.hpp b/src/detail/include/nlohmann/detail/json_custom_base_class.hpp new file mode 100644 index 000000000..47986b654 --- /dev/null +++ b/src/detail/include/nlohmann/detail/json_custom_base_class.hpp @@ -0,0 +1,39 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // conditional, is_same + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief Default base class of the @ref basic_json class. + +So that the correct implementations of the copy / move ctors / assign operators +of @ref basic_json do not require complex case distinctions +(no base class / custom base class used as customization point), +@ref basic_json always has a base class. +By default, this class is used because it is empty and thus has no effect +on the behavior of @ref basic_json. +*/ +struct json_default_base {}; + +template +using json_base_class = typename std::conditional < + std::is_same::value, + json_default_base, + T + >::type; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/json_pointer.hpp b/src/detail/include/nlohmann/detail/json_pointer.hpp new file mode 100644 index 000000000..098d1c50e --- /dev/null +++ b/src/detail/include/nlohmann/detail/json_pointer.hpp @@ -0,0 +1,988 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // all_of +#include // isdigit +#include // errno, ERANGE +#include // strtoull +#ifndef JSON_NO_IO + #include // ostream +#endif // JSON_NO_IO +#include // max +#include // accumulate +#include // string +#include // move +#include // vector + +#include +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN + +/// @brief JSON Pointer defines a string syntax for identifying a specific value within a JSON document +/// @sa https://json.nlohmann.me/api/json_pointer/ +template +class json_pointer +{ + // allow basic_json to access private members + NLOHMANN_BASIC_JSON_TPL_DECLARATION + friend class basic_json; + + template + friend class json_pointer; + + template + struct string_t_helper + { + using type = T; + }; + + NLOHMANN_BASIC_JSON_TPL_DECLARATION + struct string_t_helper + { + using type = StringType; + }; + + public: + // for backwards compatibility accept BasicJsonType + using string_t = typename string_t_helper::type; + + /// @brief create JSON pointer + /// @sa https://json.nlohmann.me/api/json_pointer/json_pointer/ + explicit json_pointer(const string_t& s = "") + : reference_tokens(split(s)) + {} + + /// @brief return a string representation of the JSON pointer + /// @sa https://json.nlohmann.me/api/json_pointer/to_string/ + string_t to_string() const + { + return std::accumulate(reference_tokens.begin(), reference_tokens.end(), + string_t{}, + [](const string_t& a, const string_t& b) + { + return detail::concat(a, '/', detail::escape(b)); + }); + } + + /// @brief return a string representation of the JSON pointer + /// @sa https://json.nlohmann.me/api/json_pointer/operator_string/ + JSON_HEDLEY_DEPRECATED_FOR(3.11.0, to_string()) + operator string_t() const + { + return to_string(); + } + +#ifndef JSON_NO_IO + /// @brief write string representation of the JSON pointer to stream + /// @sa https://json.nlohmann.me/api/basic_json/operator_ltlt/ + friend std::ostream& operator<<(std::ostream& o, const json_pointer& ptr) + { + o << ptr.to_string(); + return o; + } +#endif + + /// @brief append another JSON pointer at the end of this JSON pointer + /// @sa https://json.nlohmann.me/api/json_pointer/operator_slasheq/ + json_pointer& operator/=(const json_pointer& ptr) + { + reference_tokens.insert(reference_tokens.end(), + ptr.reference_tokens.begin(), + ptr.reference_tokens.end()); + return *this; + } + + /// @brief append an unescaped reference token at the end of this JSON pointer + /// @sa https://json.nlohmann.me/api/json_pointer/operator_slasheq/ + json_pointer& operator/=(string_t token) + { + push_back(std::move(token)); + return *this; + } + + /// @brief append an array index at the end of this JSON pointer + /// @sa https://json.nlohmann.me/api/json_pointer/operator_slasheq/ + json_pointer& operator/=(std::size_t array_idx) + { + return *this /= std::to_string(array_idx); + } + + /// @brief create a new JSON pointer by appending the right JSON pointer at the end of the left JSON pointer + /// @sa https://json.nlohmann.me/api/json_pointer/operator_slash/ + friend json_pointer operator/(const json_pointer& lhs, + const json_pointer& rhs) + { + return json_pointer(lhs) /= rhs; + } + + /// @brief create a new JSON pointer by appending the unescaped token at the end of the JSON pointer + /// @sa https://json.nlohmann.me/api/json_pointer/operator_slash/ + friend json_pointer operator/(const json_pointer& lhs, string_t token) // NOLINT(performance-unnecessary-value-param) + { + return json_pointer(lhs) /= std::move(token); + } + + /// @brief create a new JSON pointer by appending the array-index-token at the end of the JSON pointer + /// @sa https://json.nlohmann.me/api/json_pointer/operator_slash/ + friend json_pointer operator/(const json_pointer& lhs, std::size_t array_idx) + { + return json_pointer(lhs) /= array_idx; + } + + /// @brief returns the parent of this JSON pointer + /// @sa https://json.nlohmann.me/api/json_pointer/parent_pointer/ + json_pointer parent_pointer() const + { + if (empty()) + { + return *this; + } + + json_pointer res = *this; + res.pop_back(); + return res; + } + + /// @brief remove last reference token + /// @sa https://json.nlohmann.me/api/json_pointer/pop_back/ + void pop_back() + { + if (JSON_HEDLEY_UNLIKELY(empty())) + { + JSON_THROW(detail::out_of_range::create(405, "JSON pointer has no parent", nullptr)); + } + + reference_tokens.pop_back(); + } + + /// @brief return last reference token + /// @sa https://json.nlohmann.me/api/json_pointer/back/ + const string_t& back() const + { + if (JSON_HEDLEY_UNLIKELY(empty())) + { + JSON_THROW(detail::out_of_range::create(405, "JSON pointer has no parent", nullptr)); + } + + return reference_tokens.back(); + } + + /// @brief append an unescaped token at the end of the reference pointer + /// @sa https://json.nlohmann.me/api/json_pointer/push_back/ + void push_back(const string_t& token) + { + reference_tokens.push_back(token); + } + + /// @brief append an unescaped token at the end of the reference pointer + /// @sa https://json.nlohmann.me/api/json_pointer/push_back/ + void push_back(string_t&& token) + { + reference_tokens.push_back(std::move(token)); + } + + /// @brief return whether pointer points to the root document + /// @sa https://json.nlohmann.me/api/json_pointer/empty/ + bool empty() const noexcept + { + return reference_tokens.empty(); + } + + private: + /*! + @param[in] s reference token to be converted into an array index + + @return integer representation of @a s + + @throw parse_error.106 if an array index begins with '0' + @throw parse_error.109 if an array index begins not with a digit + @throw out_of_range.404 if string @a s could not be converted to an integer + @throw out_of_range.410 if an array index exceeds size_type + */ + template + static typename BasicJsonType::size_type array_index(const string_t& s) + { + using size_type = typename BasicJsonType::size_type; + + // error condition (cf. RFC 6901, Sect. 4) + if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && s[0] == '0')) + { + JSON_THROW(detail::parse_error::create(106, 0, detail::concat("array index '", s, "' must not begin with '0'"), nullptr)); + } + + // error condition (cf. RFC 6901, Sect. 4) + if (JSON_HEDLEY_UNLIKELY(s.size() > 1 && !(s[0] >= '1' && s[0] <= '9'))) + { + JSON_THROW(detail::parse_error::create(109, 0, detail::concat("array index '", s, "' is not a number"), nullptr)); + } + + const char* p = s.c_str(); + char* p_end = nullptr; // NOLINT(misc-const-correctness) + errno = 0; // strtoull doesn't reset errno + const unsigned long long res = std::strtoull(p, &p_end, 10); // NOLINT(runtime/int) + if (p == p_end // invalid input or empty string + || errno == ERANGE // out of range + || JSON_HEDLEY_UNLIKELY(static_cast(p_end - p) != s.size())) // incomplete read + { + JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", s, "'"), nullptr)); + } + + // only triggered on special platforms (like 32bit), see also + // https://github.com/nlohmann/json/pull/2203 + if (res >= static_cast((std::numeric_limits::max)())) // NOLINT(runtime/int) + { + JSON_THROW(detail::out_of_range::create(410, detail::concat("array index ", s, " exceeds size_type"), nullptr)); // LCOV_EXCL_LINE + } + + return static_cast(res); + } + + JSON_PRIVATE_UNLESS_TESTED: + json_pointer top() const + { + if (JSON_HEDLEY_UNLIKELY(empty())) + { + JSON_THROW(detail::out_of_range::create(405, "JSON pointer has no parent", nullptr)); + } + + json_pointer result = *this; + result.reference_tokens = {reference_tokens[0]}; + return result; + } + + private: + /*! + @brief create and return a reference to the pointed to value + + @complexity Linear in the number of reference tokens. + + @throw parse_error.109 if array index is not a number + @throw type_error.313 if value cannot be unflattened + */ + template + BasicJsonType& get_and_create(BasicJsonType& j) const + { + auto* result = &j; + + // in case no reference tokens exist, return a reference to the JSON value + // j which will be overwritten by a primitive value + for (const auto& reference_token : reference_tokens) + { + switch (result->type()) + { + case detail::value_t::null: + { + if (reference_token == "0") + { + // start a new array if the reference token is 0 + result = &result->operator[](0); + } + else + { + // start a new object otherwise + result = &result->operator[](reference_token); + } + break; + } + + case detail::value_t::object: + { + // create an entry in the object + result = &result->operator[](reference_token); + break; + } + + case detail::value_t::array: + { + // create an entry in the array + result = &result->operator[](array_index(reference_token)); + break; + } + + /* + The following code is only reached if there exists a reference + token _and_ the current value is primitive. In this case, we have + an error situation, because primitive values may only occur as + a single value; that is, with an empty list of reference tokens. + */ + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + JSON_THROW(detail::type_error::create(313, "invalid value to unflatten", &j)); + } + } + + return *result; + } + + /*! + @brief return a reference to the pointed to value + + @note This version does not throw if a value is not present, but tries to + create nested values instead. For instance, calling this function + with pointer `"/this/that"` on a null value is equivalent to calling + `operator[]("this").operator[]("that")` on that value, effectively + changing the null value to an object. + + @param[in] ptr a JSON value + + @return reference to the JSON value pointed to by the JSON pointer + + @complexity Linear in the length of the JSON pointer. + + @throw parse_error.106 if an array index begins with '0' + @throw parse_error.109 if an array index was not a number + @throw out_of_range.404 if the JSON pointer can not be resolved + */ + template + BasicJsonType& get_unchecked(BasicJsonType* ptr) const + { + for (const auto& reference_token : reference_tokens) + { + // convert null values to arrays or objects before continuing + if (ptr->is_null()) + { + // check if the reference token is a number + const bool nums = + std::all_of(reference_token.begin(), reference_token.end(), + [](const unsigned char x) + { + return std::isdigit(x); + }); + + // change value to an array for numbers or "-" or to object otherwise + *ptr = (nums || reference_token == "-") + ? detail::value_t::array + : detail::value_t::object; + } + + switch (ptr->type()) + { + case detail::value_t::object: + { + // use unchecked object access + ptr = &ptr->operator[](reference_token); + break; + } + + case detail::value_t::array: + { + if (reference_token == "-") + { + // explicitly treat "-" as index beyond the end + ptr = &ptr->operator[](ptr->m_data.m_value.array->size()); + } + else + { + // convert array index to number; unchecked access + ptr = &ptr->operator[](array_index(reference_token)); + } + break; + } + + case detail::value_t::null: + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", reference_token, "'"), ptr)); + } + } + + return *ptr; + } + + /*! + @throw parse_error.106 if an array index begins with '0' + @throw parse_error.109 if an array index was not a number + @throw out_of_range.402 if the array index '-' is used + @throw out_of_range.404 if the JSON pointer can not be resolved + */ + template + BasicJsonType& get_checked(BasicJsonType* ptr) const + { + for (const auto& reference_token : reference_tokens) + { + switch (ptr->type()) + { + case detail::value_t::object: + { + // note: at performs range check + ptr = &ptr->at(reference_token); + break; + } + + case detail::value_t::array: + { + if (JSON_HEDLEY_UNLIKELY(reference_token == "-")) + { + // "-" always fails the range check + JSON_THROW(detail::out_of_range::create(402, detail::concat( + "array index '-' (", std::to_string(ptr->m_data.m_value.array->size()), + ") is out of range"), ptr)); + } + + // note: at performs range check + ptr = &ptr->at(array_index(reference_token)); + break; + } + + case detail::value_t::null: + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", reference_token, "'"), ptr)); + } + } + + return *ptr; + } + + /*! + @brief return a const reference to the pointed to value + + @param[in] ptr a JSON value + + @return const reference to the JSON value pointed to by the JSON + pointer + + @throw parse_error.106 if an array index begins with '0' + @throw parse_error.109 if an array index was not a number + @throw out_of_range.402 if the array index '-' is used + @throw out_of_range.404 if the JSON pointer can not be resolved + */ + template + const BasicJsonType& get_unchecked(const BasicJsonType* ptr) const + { + for (const auto& reference_token : reference_tokens) + { + switch (ptr->type()) + { + case detail::value_t::object: + { + // use unchecked object access + ptr = &ptr->operator[](reference_token); + break; + } + + case detail::value_t::array: + { + if (JSON_HEDLEY_UNLIKELY(reference_token == "-")) + { + // "-" cannot be used for const access + JSON_THROW(detail::out_of_range::create(402, detail::concat("array index '-' (", std::to_string(ptr->m_data.m_value.array->size()), ") is out of range"), ptr)); + } + + // use unchecked array access + ptr = &ptr->operator[](array_index(reference_token)); + break; + } + + case detail::value_t::null: + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", reference_token, "'"), ptr)); + } + } + + return *ptr; + } + + /*! + @throw parse_error.106 if an array index begins with '0' + @throw parse_error.109 if an array index was not a number + @throw out_of_range.402 if the array index '-' is used + @throw out_of_range.404 if the JSON pointer can not be resolved + */ + template + const BasicJsonType& get_checked(const BasicJsonType* ptr) const + { + for (const auto& reference_token : reference_tokens) + { + switch (ptr->type()) + { + case detail::value_t::object: + { + // note: at performs range check + ptr = &ptr->at(reference_token); + break; + } + + case detail::value_t::array: + { + if (JSON_HEDLEY_UNLIKELY(reference_token == "-")) + { + // "-" always fails the range check + JSON_THROW(detail::out_of_range::create(402, detail::concat( + "array index '-' (", std::to_string(ptr->m_data.m_value.array->size()), + ") is out of range"), ptr)); + } + + // note: at performs range check + ptr = &ptr->at(array_index(reference_token)); + break; + } + + case detail::value_t::null: + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + JSON_THROW(detail::out_of_range::create(404, detail::concat("unresolved reference token '", reference_token, "'"), ptr)); + } + } + + return *ptr; + } + + /*! + @throw parse_error.106 if an array index begins with '0' + @throw parse_error.109 if an array index was not a number + */ + template + bool contains(const BasicJsonType* ptr) const + { + for (const auto& reference_token : reference_tokens) + { + switch (ptr->type()) + { + case detail::value_t::object: + { + if (!ptr->contains(reference_token)) + { + // we did not find the key in the object + return false; + } + + ptr = &ptr->operator[](reference_token); + break; + } + + case detail::value_t::array: + { + if (JSON_HEDLEY_UNLIKELY(reference_token == "-")) + { + // "-" always fails the range check + return false; + } + if (JSON_HEDLEY_UNLIKELY(reference_token.size() == 1 && !("0" <= reference_token && reference_token <= "9"))) + { + // invalid char + return false; + } + if (JSON_HEDLEY_UNLIKELY(reference_token.size() > 1)) + { + if (JSON_HEDLEY_UNLIKELY(!('1' <= reference_token[0] && reference_token[0] <= '9'))) + { + // the first char should be between '1' and '9' + return false; + } + for (std::size_t i = 1; i < reference_token.size(); i++) + { + if (JSON_HEDLEY_UNLIKELY(!('0' <= reference_token[i] && reference_token[i] <= '9'))) + { + // other char should be between '0' and '9' + return false; + } + } + } + + const auto idx = array_index(reference_token); + if (idx >= ptr->size()) + { + // index out of range + return false; + } + + ptr = &ptr->operator[](idx); + break; + } + + case detail::value_t::null: + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + { + // we do not expect primitive values if there is still a + // reference token to process + return false; + } + } + } + + // no reference token left means we found a primitive value + return true; + } + + /*! + @brief split the string input to reference tokens + + @note This function is only called by the json_pointer constructor. + All exceptions below are documented there. + + @throw parse_error.107 if the pointer is not empty or begins with '/' + @throw parse_error.108 if character '~' is not followed by '0' or '1' + */ + static std::vector split(const string_t& reference_string) + { + std::vector result; + + // special case: empty reference string -> no reference tokens + if (reference_string.empty()) + { + return result; + } + + // check if a nonempty reference string begins with slash + if (JSON_HEDLEY_UNLIKELY(reference_string[0] != '/')) + { + JSON_THROW(detail::parse_error::create(107, 1, detail::concat("JSON pointer must be empty or begin with '/' - was: '", reference_string, "'"), nullptr)); + } + + // extract the reference tokens: + // - slash: position of the last read slash (or end of string) + // - start: position after the previous slash + for ( + // search for the first slash after the first character + std::size_t slash = reference_string.find_first_of('/', 1), + // set the beginning of the first reference token + start = 1; + // we can stop if start == 0 (if slash == string_t::npos) + start != 0; + // set the beginning of the next reference token + // (will eventually be 0 if slash == string_t::npos) + start = (slash == string_t::npos) ? 0 : slash + 1, + // find next slash + slash = reference_string.find_first_of('/', start)) + { + // use the text between the beginning of the reference token + // (start) and the last slash (slash). + auto reference_token = reference_string.substr(start, slash - start); + + // check reference tokens are properly escaped + for (std::size_t pos = reference_token.find_first_of('~'); + pos != string_t::npos; + pos = reference_token.find_first_of('~', pos + 1)) + { + JSON_ASSERT(reference_token[pos] == '~'); + + // ~ must be followed by 0 or 1 + if (JSON_HEDLEY_UNLIKELY(pos == reference_token.size() - 1 || + (reference_token[pos + 1] != '0' && + reference_token[pos + 1] != '1'))) + { + JSON_THROW(detail::parse_error::create(108, 0, "escape character '~' must be followed with '0' or '1'", nullptr)); + } + } + + // finally, store the reference token + detail::unescape(reference_token); + result.push_back(reference_token); + } + + return result; + } + + private: + /*! + @param[in] reference_string the reference string to the current value + @param[in] value the value to consider + @param[in,out] result the result object to insert values to + + @note Empty objects or arrays are flattened to `null`. + */ + template + static void flatten(const string_t& reference_string, + const BasicJsonType& value, + BasicJsonType& result) + { + switch (value.type()) + { + case detail::value_t::array: + { + if (value.m_data.m_value.array->empty()) + { + // flatten empty array as null + result[reference_string] = nullptr; + } + else + { + // iterate array and use index as a reference string + for (std::size_t i = 0; i < value.m_data.m_value.array->size(); ++i) + { + flatten(detail::concat(reference_string, '/', std::to_string(i)), + value.m_data.m_value.array->operator[](i), result); + } + } + break; + } + + case detail::value_t::object: + { + if (value.m_data.m_value.object->empty()) + { + // flatten empty object as null + result[reference_string] = nullptr; + } + else + { + // iterate object and use keys as reference string + for (const auto& element : *value.m_data.m_value.object) + { + flatten(detail::concat(reference_string, '/', detail::escape(element.first)), element.second, result); + } + } + break; + } + + case detail::value_t::null: + case detail::value_t::string: + case detail::value_t::boolean: + case detail::value_t::number_integer: + case detail::value_t::number_unsigned: + case detail::value_t::number_float: + case detail::value_t::binary: + case detail::value_t::discarded: + default: + { + // add a primitive value with its reference string + result[reference_string] = value; + break; + } + } + } + + /*! + @param[in] value flattened JSON + + @return unflattened JSON + + @throw parse_error.109 if array index is not a number + @throw type_error.314 if value is not an object + @throw type_error.315 if object values are not primitive + @throw type_error.313 if value cannot be unflattened + */ + template + static BasicJsonType + unflatten(const BasicJsonType& value) + { + if (JSON_HEDLEY_UNLIKELY(!value.is_object())) + { + JSON_THROW(detail::type_error::create(314, "only objects can be unflattened", &value)); + } + + BasicJsonType result; + + // iterate the JSON object values + for (const auto& element : *value.m_data.m_value.object) + { + if (JSON_HEDLEY_UNLIKELY(!element.second.is_primitive())) + { + JSON_THROW(detail::type_error::create(315, "values in object must be primitive", &element.second)); + } + + // Assign the value to the reference pointed to by JSON pointer. Note + // that if the JSON pointer is "" (i.e., points to the whole value), + // function get_and_create returns a reference to the result itself. + // An assignment will then create a primitive value. + json_pointer(element.first).get_and_create(result) = element.second; + } + + return result; + } + + // can't use the conversion operator because of ambiguity + json_pointer convert() const& + { + json_pointer result; + result.reference_tokens = reference_tokens; + return result; + } + + json_pointer convert()&& + { + json_pointer result; + result.reference_tokens = std::move(reference_tokens); + return result; + } + + public: +#if JSON_HAS_THREE_WAY_COMPARISON + /// @brief compares two JSON pointers for equality + /// @sa https://json.nlohmann.me/api/json_pointer/operator_eq/ + template + bool operator==(const json_pointer& rhs) const noexcept + { + return reference_tokens == rhs.reference_tokens; + } + + /// @brief compares JSON pointer and string for equality + /// @sa https://json.nlohmann.me/api/json_pointer/operator_eq/ + JSON_HEDLEY_DEPRECATED_FOR(3.11.2, operator==(json_pointer)) + bool operator==(const string_t& rhs) const + { + return *this == json_pointer(rhs); + } + + /// @brief 3-way compares two JSON pointers + template + std::strong_ordering operator<=>(const json_pointer& rhs) const noexcept // *NOPAD* + { + return reference_tokens <=> rhs.reference_tokens; // *NOPAD* + } +#else + /// @brief compares two JSON pointers for equality + /// @sa https://json.nlohmann.me/api/json_pointer/operator_eq/ + template + // NOLINTNEXTLINE(readability-redundant-declaration) + friend bool operator==(const json_pointer& lhs, + const json_pointer& rhs) noexcept; + + /// @brief compares JSON pointer and string for equality + /// @sa https://json.nlohmann.me/api/json_pointer/operator_eq/ + template + // NOLINTNEXTLINE(readability-redundant-declaration) + friend bool operator==(const json_pointer& lhs, + const StringType& rhs); + + /// @brief compares string and JSON pointer for equality + /// @sa https://json.nlohmann.me/api/json_pointer/operator_eq/ + template + // NOLINTNEXTLINE(readability-redundant-declaration) + friend bool operator==(const StringType& lhs, + const json_pointer& rhs); + + /// @brief compares two JSON pointers for inequality + /// @sa https://json.nlohmann.me/api/json_pointer/operator_ne/ + template + // NOLINTNEXTLINE(readability-redundant-declaration) + friend bool operator!=(const json_pointer& lhs, + const json_pointer& rhs) noexcept; + + /// @brief compares JSON pointer and string for inequality + /// @sa https://json.nlohmann.me/api/json_pointer/operator_ne/ + template + // NOLINTNEXTLINE(readability-redundant-declaration) + friend bool operator!=(const json_pointer& lhs, + const StringType& rhs); + + /// @brief compares string and JSON pointer for inequality + /// @sa https://json.nlohmann.me/api/json_pointer/operator_ne/ + template + // NOLINTNEXTLINE(readability-redundant-declaration) + friend bool operator!=(const StringType& lhs, + const json_pointer& rhs); + + /// @brief compares two JSON pointer for less-than + template + // NOLINTNEXTLINE(readability-redundant-declaration) + friend bool operator<(const json_pointer& lhs, + const json_pointer& rhs) noexcept; +#endif + + private: + /// the reference tokens + std::vector reference_tokens; +}; + +#if !JSON_HAS_THREE_WAY_COMPARISON +// functions cannot be defined inside the class due to ODR violations +template +inline bool operator==(const json_pointer& lhs, + const json_pointer& rhs) noexcept +{ + return lhs.reference_tokens == rhs.reference_tokens; +} + +template::string_t> +JSON_HEDLEY_DEPRECATED_FOR(3.11.2, operator==(json_pointer, json_pointer)) +inline bool operator==(const json_pointer& lhs, + const StringType& rhs) +{ + return lhs == json_pointer(rhs); +} + +template::string_t> +JSON_HEDLEY_DEPRECATED_FOR(3.11.2, operator==(json_pointer, json_pointer)) +inline bool operator==(const StringType& lhs, + const json_pointer& rhs) +{ + return json_pointer(lhs) == rhs; +} + +template +inline bool operator!=(const json_pointer& lhs, + const json_pointer& rhs) noexcept +{ + return !(lhs == rhs); +} + +template::string_t> +JSON_HEDLEY_DEPRECATED_FOR(3.11.2, operator!=(json_pointer, json_pointer)) +inline bool operator!=(const json_pointer& lhs, + const StringType& rhs) +{ + return !(lhs == rhs); +} + +template::string_t> +JSON_HEDLEY_DEPRECATED_FOR(3.11.2, operator!=(json_pointer, json_pointer)) +inline bool operator!=(const StringType& lhs, + const json_pointer& rhs) +{ + return !(lhs == rhs); +} + +template +inline bool operator<(const json_pointer& lhs, + const json_pointer& rhs) noexcept +{ + return lhs.reference_tokens < rhs.reference_tokens; +} +#endif + +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/json_ref.hpp b/src/detail/include/nlohmann/detail/json_ref.hpp new file mode 100644 index 000000000..65eb125ac --- /dev/null +++ b/src/detail/include/nlohmann/detail/json_ref.hpp @@ -0,0 +1,78 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include +#include + +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +template +class json_ref +{ + public: + using value_type = BasicJsonType; + + json_ref(value_type&& value) + : owned_value(std::move(value)) + {} + + json_ref(const value_type& value) + : value_ref(&value) + {} + + json_ref(std::initializer_list init) + : owned_value(init) + {} + + template < + class... Args, + enable_if_t::value, int> = 0 > + json_ref(Args && ... args) + : owned_value(std::forward(args)...) + {} + + // class should be movable only + json_ref(json_ref&&) noexcept = default; + json_ref(const json_ref&) = delete; + json_ref& operator=(const json_ref&) = delete; + json_ref& operator=(json_ref&&) = delete; + ~json_ref() = default; + + value_type moved_or_copied() const + { + if (value_ref == nullptr) + { + return std::move(owned_value); + } + return *value_ref; + } + + value_type const& operator*() const + { + return value_ref ? *value_ref : owned_value; + } + + value_type const* operator->() const + { + return &** this; + } + + private: + mutable value_type owned_value = nullptr; + value_type const* value_ref = nullptr; +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/macro_scope.hpp b/src/detail/include/nlohmann/detail/macro_scope.hpp new file mode 100644 index 000000000..4bafa697e --- /dev/null +++ b/src/detail/include/nlohmann/detail/macro_scope.hpp @@ -0,0 +1,601 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // declval, pair +#include +#include + +// This file contains all internal macro definitions (except those affecting ABI) +// You MUST include macro_unscope.hpp at the end of json.hpp to undef all of them + +#include + +// exclude unsupported compilers +#if !defined(JSON_SKIP_UNSUPPORTED_COMPILER_CHECK) + #if defined(__clang__) + #if (__clang_major__ * 10000 + __clang_minor__ * 100 + __clang_patchlevel__) < 30400 + #error "unsupported Clang version - see https://github.com/nlohmann/json#supported-compilers" + #endif + #elif defined(__GNUC__) && !(defined(__ICC) || defined(__INTEL_COMPILER)) + #if (__GNUC__ * 10000 + __GNUC_MINOR__ * 100 + __GNUC_PATCHLEVEL__) < 40800 + #error "unsupported GCC version - see https://github.com/nlohmann/json#supported-compilers" + #endif + #endif +#endif + +// C++ language standard detection +// if the user manually specified the used C++ version, this is skipped +#if !defined(JSON_HAS_CPP_26) && !defined(JSON_HAS_CPP_23) && !defined(JSON_HAS_CPP_20) && !defined(JSON_HAS_CPP_17) && !defined(JSON_HAS_CPP_14) && !defined(JSON_HAS_CPP_11) + #if (defined(__cplusplus) && __cplusplus > 202302L) || (defined(_MSVC_LANG) && _MSVC_LANG > 202302L) + #define JSON_HAS_CPP_26 + #define JSON_HAS_CPP_23 + #define JSON_HAS_CPP_20 + #define JSON_HAS_CPP_17 + #define JSON_HAS_CPP_14 + #elif (defined(__cplusplus) && __cplusplus > 202002L) || (defined(_MSVC_LANG) && _MSVC_LANG > 202002L) + #define JSON_HAS_CPP_23 + #define JSON_HAS_CPP_20 + #define JSON_HAS_CPP_17 + #define JSON_HAS_CPP_14 + #elif (defined(__cplusplus) && __cplusplus > 201703L) || (defined(_MSVC_LANG) && _MSVC_LANG > 201703L) + #define JSON_HAS_CPP_20 + #define JSON_HAS_CPP_17 + #define JSON_HAS_CPP_14 + #elif (defined(__cplusplus) && __cplusplus > 201402L) || (defined(_HAS_CXX17) && _HAS_CXX17 == 1) // fix for issue #464 + #define JSON_HAS_CPP_17 + #define JSON_HAS_CPP_14 + #elif (defined(__cplusplus) && __cplusplus > 201103L) || (defined(_HAS_CXX14) && _HAS_CXX14 == 1) + #define JSON_HAS_CPP_14 + #endif + // the cpp 11 flag is always specified because it is the minimal required version + #define JSON_HAS_CPP_11 +#endif + +#ifdef __has_include + #if __has_include() + #include + #endif +#endif + +#if !defined(JSON_HAS_FILESYSTEM) && !defined(JSON_HAS_EXPERIMENTAL_FILESYSTEM) + #ifdef JSON_HAS_CPP_17 + #if defined(__cpp_lib_filesystem) + #define JSON_HAS_FILESYSTEM 1 + #elif defined(__cpp_lib_experimental_filesystem) + #define JSON_HAS_EXPERIMENTAL_FILESYSTEM 1 + #elif !defined(__has_include) + #define JSON_HAS_EXPERIMENTAL_FILESYSTEM 1 + #elif __has_include() + #define JSON_HAS_FILESYSTEM 1 + #elif __has_include() + #define JSON_HAS_EXPERIMENTAL_FILESYSTEM 1 + #endif + + // std::filesystem does not work on MinGW GCC 8: https://sourceforge.net/p/mingw-w64/bugs/737/ + #if defined(__MINGW32__) && defined(__GNUC__) && __GNUC__ == 8 + #undef JSON_HAS_FILESYSTEM + #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM + #endif + + // no filesystem support before GCC 8: https://en.cppreference.com/w/cpp/compiler_support + #if defined(__GNUC__) && !defined(__clang__) && __GNUC__ < 8 + #undef JSON_HAS_FILESYSTEM + #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM + #endif + + // no filesystem support before Clang 7: https://en.cppreference.com/w/cpp/compiler_support + #if defined(__clang_major__) && __clang_major__ < 7 + #undef JSON_HAS_FILESYSTEM + #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM + #endif + + // no filesystem support before MSVC 19.14: https://en.cppreference.com/w/cpp/compiler_support + #if defined(_MSC_VER) && _MSC_VER < 1914 + #undef JSON_HAS_FILESYSTEM + #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM + #endif + + // no filesystem support before iOS 13 + #if defined(__IPHONE_OS_VERSION_MIN_REQUIRED) && __IPHONE_OS_VERSION_MIN_REQUIRED < 130000 + #undef JSON_HAS_FILESYSTEM + #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM + #endif + + // no filesystem support before macOS Catalina + #if defined(__MAC_OS_X_VERSION_MIN_REQUIRED) && __MAC_OS_X_VERSION_MIN_REQUIRED < 101500 + #undef JSON_HAS_FILESYSTEM + #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM + #endif + #endif +#endif + +#ifndef JSON_HAS_EXPERIMENTAL_FILESYSTEM + #define JSON_HAS_EXPERIMENTAL_FILESYSTEM 0 +#endif + +#ifndef JSON_HAS_FILESYSTEM + #define JSON_HAS_FILESYSTEM 0 +#endif + +#ifndef JSON_HAS_THREE_WAY_COMPARISON + #if defined(__cpp_impl_three_way_comparison) && __cpp_impl_three_way_comparison >= 201907L \ + && defined(__cpp_lib_three_way_comparison) && __cpp_lib_three_way_comparison >= 201907L + #define JSON_HAS_THREE_WAY_COMPARISON 1 + #else + #define JSON_HAS_THREE_WAY_COMPARISON 0 + #endif +#endif + +#ifndef JSON_HAS_RANGES + // ranges header shipping in GCC 11.1.0 (released 2021-04-27) has a syntax error + #if defined(__GLIBCXX__) && __GLIBCXX__ == 20210427 + #define JSON_HAS_RANGES 0 + #elif defined(__cpp_lib_ranges) + #define JSON_HAS_RANGES 1 + #else + #define JSON_HAS_RANGES 0 + #endif +#endif + +#ifndef JSON_HAS_STATIC_RTTI + #if !defined(_HAS_STATIC_RTTI) || _HAS_STATIC_RTTI != 0 + #define JSON_HAS_STATIC_RTTI 1 + #else + #define JSON_HAS_STATIC_RTTI 0 + #endif +#endif + +#ifdef JSON_HAS_CPP_17 + #define JSON_INLINE_VARIABLE inline +#else + #define JSON_INLINE_VARIABLE +#endif + +#if JSON_HEDLEY_HAS_ATTRIBUTE(no_unique_address) + #define JSON_NO_UNIQUE_ADDRESS [[no_unique_address]] +#else + #define JSON_NO_UNIQUE_ADDRESS +#endif + +// disable documentation warnings on clang +#if defined(__clang__) + #pragma clang diagnostic push + #pragma clang diagnostic ignored "-Wdocumentation" + #pragma clang diagnostic ignored "-Wdocumentation-unknown-command" +#endif + +// allow disabling exceptions +#if (defined(__cpp_exceptions) || defined(__EXCEPTIONS) || defined(_CPPUNWIND)) && !defined(JSON_NOEXCEPTION) + #define JSON_THROW(exception) throw exception + #define JSON_TRY try + #define JSON_CATCH(exception) catch(exception) + #define JSON_INTERNAL_CATCH(exception) catch(exception) +#else + #include + #define JSON_THROW(exception) std::abort() + #define JSON_TRY if(true) + #define JSON_CATCH(exception) if(false) + #define JSON_INTERNAL_CATCH(exception) if(false) +#endif + +// override exception macros +#if defined(JSON_THROW_USER) + #undef JSON_THROW + #define JSON_THROW JSON_THROW_USER +#endif +#if defined(JSON_TRY_USER) + #undef JSON_TRY + #define JSON_TRY JSON_TRY_USER +#endif +#if defined(JSON_CATCH_USER) + #undef JSON_CATCH + #define JSON_CATCH JSON_CATCH_USER + #undef JSON_INTERNAL_CATCH + #define JSON_INTERNAL_CATCH JSON_CATCH_USER +#endif +#if defined(JSON_INTERNAL_CATCH_USER) + #undef JSON_INTERNAL_CATCH + #define JSON_INTERNAL_CATCH JSON_INTERNAL_CATCH_USER +#endif + +// allow overriding assert +#if !defined(JSON_ASSERT) + #include // assert + #define JSON_ASSERT(x) assert(x) +#endif + +// allow accessing some private functions (needed by the test suite) +#if defined(JSON_TESTS_PRIVATE) + #define JSON_PRIVATE_UNLESS_TESTED public +#else + #define JSON_PRIVATE_UNLESS_TESTED private +#endif + +/*! +@brief macro to briefly define a mapping between an enum and JSON +@def NLOHMANN_JSON_SERIALIZE_ENUM +@since version 3.4.0 +*/ +#define NLOHMANN_JSON_SERIALIZE_ENUM(ENUM_TYPE, ...) \ + template \ + inline void to_json(BasicJsonType& j, const ENUM_TYPE& e) \ + { \ + /* NOLINTNEXTLINE(modernize-type-traits) we use C++11 */ \ + static_assert(std::is_enum::value, #ENUM_TYPE " must be an enum!"); \ + /* NOLINTNEXTLINE(modernize-avoid-c-arrays) we don't want to depend on */ \ + static const std::pair m[] = __VA_ARGS__; \ + auto it = std::find_if(std::begin(m), std::end(m), \ + [e](const std::pair& ej_pair) -> bool \ + { \ + return ej_pair.first == e; \ + }); \ + j = ((it != std::end(m)) ? it : std::begin(m))->second; \ + } \ + template \ + inline void from_json(const BasicJsonType& j, ENUM_TYPE& e) \ + { \ + /* NOLINTNEXTLINE(modernize-type-traits) we use C++11 */ \ + static_assert(std::is_enum::value, #ENUM_TYPE " must be an enum!"); \ + /* NOLINTNEXTLINE(modernize-avoid-c-arrays) we don't want to depend on */ \ + static const std::pair m[] = __VA_ARGS__; \ + auto it = std::find_if(std::begin(m), std::end(m), \ + [&j](const std::pair& ej_pair) -> bool \ + { \ + return ej_pair.second == j; \ + }); \ + e = ((it != std::end(m)) ? it : std::begin(m))->first; \ + } + +// Ugly macros to avoid uglier copy-paste when specializing basic_json. They +// may be removed in the future once the class is split. + +#define NLOHMANN_BASIC_JSON_TPL_DECLARATION \ + template class ObjectType, \ + template class ArrayType, \ + class StringType, class BooleanType, class NumberIntegerType, \ + class NumberUnsignedType, class NumberFloatType, \ + template class AllocatorType, \ + template class JSONSerializer, \ + class BinaryType, \ + class CustomBaseClass> + +#define NLOHMANN_BASIC_JSON_TPL \ + basic_json + +// Macros to simplify conversion from/to types + +#define NLOHMANN_JSON_EXPAND( x ) x +#define NLOHMANN_JSON_GET_MACRO(_1, _2, _3, _4, _5, _6, _7, _8, _9, _10, _11, _12, _13, _14, _15, _16, _17, _18, _19, _20, _21, _22, _23, _24, _25, _26, _27, _28, _29, _30, _31, _32, _33, _34, _35, _36, _37, _38, _39, _40, _41, _42, _43, _44, _45, _46, _47, _48, _49, _50, _51, _52, _53, _54, _55, _56, _57, _58, _59, _60, _61, _62, _63, _64, NAME,...) NAME +#define NLOHMANN_JSON_PASTE(...) NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_GET_MACRO(__VA_ARGS__, \ + NLOHMANN_JSON_PASTE64, \ + NLOHMANN_JSON_PASTE63, \ + NLOHMANN_JSON_PASTE62, \ + NLOHMANN_JSON_PASTE61, \ + NLOHMANN_JSON_PASTE60, \ + NLOHMANN_JSON_PASTE59, \ + NLOHMANN_JSON_PASTE58, \ + NLOHMANN_JSON_PASTE57, \ + NLOHMANN_JSON_PASTE56, \ + NLOHMANN_JSON_PASTE55, \ + NLOHMANN_JSON_PASTE54, \ + NLOHMANN_JSON_PASTE53, \ + NLOHMANN_JSON_PASTE52, \ + NLOHMANN_JSON_PASTE51, \ + NLOHMANN_JSON_PASTE50, \ + NLOHMANN_JSON_PASTE49, \ + NLOHMANN_JSON_PASTE48, \ + NLOHMANN_JSON_PASTE47, \ + NLOHMANN_JSON_PASTE46, \ + NLOHMANN_JSON_PASTE45, \ + NLOHMANN_JSON_PASTE44, \ + NLOHMANN_JSON_PASTE43, \ + NLOHMANN_JSON_PASTE42, \ + NLOHMANN_JSON_PASTE41, \ + NLOHMANN_JSON_PASTE40, \ + NLOHMANN_JSON_PASTE39, \ + NLOHMANN_JSON_PASTE38, \ + NLOHMANN_JSON_PASTE37, \ + NLOHMANN_JSON_PASTE36, \ + NLOHMANN_JSON_PASTE35, \ + NLOHMANN_JSON_PASTE34, \ + NLOHMANN_JSON_PASTE33, \ + NLOHMANN_JSON_PASTE32, \ + NLOHMANN_JSON_PASTE31, \ + NLOHMANN_JSON_PASTE30, \ + NLOHMANN_JSON_PASTE29, \ + NLOHMANN_JSON_PASTE28, \ + NLOHMANN_JSON_PASTE27, \ + NLOHMANN_JSON_PASTE26, \ + NLOHMANN_JSON_PASTE25, \ + NLOHMANN_JSON_PASTE24, \ + NLOHMANN_JSON_PASTE23, \ + NLOHMANN_JSON_PASTE22, \ + NLOHMANN_JSON_PASTE21, \ + NLOHMANN_JSON_PASTE20, \ + NLOHMANN_JSON_PASTE19, \ + NLOHMANN_JSON_PASTE18, \ + NLOHMANN_JSON_PASTE17, \ + NLOHMANN_JSON_PASTE16, \ + NLOHMANN_JSON_PASTE15, \ + NLOHMANN_JSON_PASTE14, \ + NLOHMANN_JSON_PASTE13, \ + NLOHMANN_JSON_PASTE12, \ + NLOHMANN_JSON_PASTE11, \ + NLOHMANN_JSON_PASTE10, \ + NLOHMANN_JSON_PASTE9, \ + NLOHMANN_JSON_PASTE8, \ + NLOHMANN_JSON_PASTE7, \ + NLOHMANN_JSON_PASTE6, \ + NLOHMANN_JSON_PASTE5, \ + NLOHMANN_JSON_PASTE4, \ + NLOHMANN_JSON_PASTE3, \ + NLOHMANN_JSON_PASTE2, \ + NLOHMANN_JSON_PASTE1)(__VA_ARGS__)) +#define NLOHMANN_JSON_PASTE2(func, v1) func(v1) +#define NLOHMANN_JSON_PASTE3(func, v1, v2) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE2(func, v2) +#define NLOHMANN_JSON_PASTE4(func, v1, v2, v3) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE3(func, v2, v3) +#define NLOHMANN_JSON_PASTE5(func, v1, v2, v3, v4) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE4(func, v2, v3, v4) +#define NLOHMANN_JSON_PASTE6(func, v1, v2, v3, v4, v5) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE5(func, v2, v3, v4, v5) +#define NLOHMANN_JSON_PASTE7(func, v1, v2, v3, v4, v5, v6) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE6(func, v2, v3, v4, v5, v6) +#define NLOHMANN_JSON_PASTE8(func, v1, v2, v3, v4, v5, v6, v7) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE7(func, v2, v3, v4, v5, v6, v7) +#define NLOHMANN_JSON_PASTE9(func, v1, v2, v3, v4, v5, v6, v7, v8) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE8(func, v2, v3, v4, v5, v6, v7, v8) +#define NLOHMANN_JSON_PASTE10(func, v1, v2, v3, v4, v5, v6, v7, v8, v9) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE9(func, v2, v3, v4, v5, v6, v7, v8, v9) +#define NLOHMANN_JSON_PASTE11(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE10(func, v2, v3, v4, v5, v6, v7, v8, v9, v10) +#define NLOHMANN_JSON_PASTE12(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE11(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11) +#define NLOHMANN_JSON_PASTE13(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE12(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12) +#define NLOHMANN_JSON_PASTE14(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE13(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13) +#define NLOHMANN_JSON_PASTE15(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE14(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14) +#define NLOHMANN_JSON_PASTE16(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE15(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15) +#define NLOHMANN_JSON_PASTE17(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE16(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16) +#define NLOHMANN_JSON_PASTE18(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE17(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17) +#define NLOHMANN_JSON_PASTE19(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE18(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18) +#define NLOHMANN_JSON_PASTE20(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE19(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19) +#define NLOHMANN_JSON_PASTE21(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE20(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20) +#define NLOHMANN_JSON_PASTE22(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE21(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21) +#define NLOHMANN_JSON_PASTE23(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE22(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22) +#define NLOHMANN_JSON_PASTE24(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE23(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23) +#define NLOHMANN_JSON_PASTE25(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE24(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24) +#define NLOHMANN_JSON_PASTE26(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE25(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25) +#define NLOHMANN_JSON_PASTE27(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE26(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26) +#define NLOHMANN_JSON_PASTE28(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE27(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27) +#define NLOHMANN_JSON_PASTE29(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE28(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28) +#define NLOHMANN_JSON_PASTE30(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE29(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29) +#define NLOHMANN_JSON_PASTE31(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE30(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30) +#define NLOHMANN_JSON_PASTE32(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE31(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31) +#define NLOHMANN_JSON_PASTE33(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE32(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32) +#define NLOHMANN_JSON_PASTE34(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE33(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33) +#define NLOHMANN_JSON_PASTE35(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE34(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34) +#define NLOHMANN_JSON_PASTE36(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE35(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35) +#define NLOHMANN_JSON_PASTE37(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE36(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36) +#define NLOHMANN_JSON_PASTE38(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE37(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37) +#define NLOHMANN_JSON_PASTE39(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE38(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38) +#define NLOHMANN_JSON_PASTE40(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE39(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39) +#define NLOHMANN_JSON_PASTE41(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE40(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40) +#define NLOHMANN_JSON_PASTE42(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE41(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41) +#define NLOHMANN_JSON_PASTE43(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE42(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42) +#define NLOHMANN_JSON_PASTE44(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE43(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43) +#define NLOHMANN_JSON_PASTE45(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE44(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44) +#define NLOHMANN_JSON_PASTE46(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE45(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45) +#define NLOHMANN_JSON_PASTE47(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE46(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46) +#define NLOHMANN_JSON_PASTE48(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE47(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47) +#define NLOHMANN_JSON_PASTE49(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE48(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48) +#define NLOHMANN_JSON_PASTE50(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE49(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49) +#define NLOHMANN_JSON_PASTE51(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE50(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50) +#define NLOHMANN_JSON_PASTE52(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE51(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51) +#define NLOHMANN_JSON_PASTE53(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE52(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52) +#define NLOHMANN_JSON_PASTE54(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE53(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53) +#define NLOHMANN_JSON_PASTE55(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE54(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54) +#define NLOHMANN_JSON_PASTE56(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE55(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55) +#define NLOHMANN_JSON_PASTE57(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE56(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56) +#define NLOHMANN_JSON_PASTE58(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE57(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57) +#define NLOHMANN_JSON_PASTE59(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE58(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58) +#define NLOHMANN_JSON_PASTE60(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58, v59) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE59(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58, v59) +#define NLOHMANN_JSON_PASTE61(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58, v59, v60) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE60(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58, v59, v60) +#define NLOHMANN_JSON_PASTE62(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58, v59, v60, v61) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE61(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58, v59, v60, v61) +#define NLOHMANN_JSON_PASTE63(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58, v59, v60, v61, v62) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE62(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58, v59, v60, v61, v62) +#define NLOHMANN_JSON_PASTE64(func, v1, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58, v59, v60, v61, v62, v63) NLOHMANN_JSON_PASTE2(func, v1) NLOHMANN_JSON_PASTE63(func, v2, v3, v4, v5, v6, v7, v8, v9, v10, v11, v12, v13, v14, v15, v16, v17, v18, v19, v20, v21, v22, v23, v24, v25, v26, v27, v28, v29, v30, v31, v32, v33, v34, v35, v36, v37, v38, v39, v40, v41, v42, v43, v44, v45, v46, v47, v48, v49, v50, v51, v52, v53, v54, v55, v56, v57, v58, v59, v60, v61, v62, v63) + +#define NLOHMANN_JSON_TO(v1) nlohmann_json_j[#v1] = nlohmann_json_t.v1; +#define NLOHMANN_JSON_FROM(v1) nlohmann_json_j.at(#v1).get_to(nlohmann_json_t.v1); +#define NLOHMANN_JSON_FROM_WITH_DEFAULT(v1) nlohmann_json_t.v1 = !nlohmann_json_j.is_null() ? nlohmann_json_j.value(#v1, nlohmann_json_default_obj.v1) : nlohmann_json_default_obj.v1; + +/*! +@brief macro +@def NLOHMANN_DEFINE_TYPE_INTRUSIVE +@since version 3.9.0 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_type_intrusive/ +*/ +#define NLOHMANN_DEFINE_TYPE_INTRUSIVE(Type, ...) \ + template::value, int> = 0> \ + friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } \ + template::value, int> = 0> \ + friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT +@since version 3.11.0 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_type_intrusive/ +*/ +#define NLOHMANN_DEFINE_TYPE_INTRUSIVE_WITH_DEFAULT(Type, ...) \ + template::value, int> = 0> \ + friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } \ + template::value, int> = 0> \ + friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE +@since version 3.11.3 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_type_intrusive/ +*/ +#define NLOHMANN_DEFINE_TYPE_INTRUSIVE_ONLY_SERIALIZE(Type, ...) \ + template::value, int> = 0> \ + friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE +@since version 3.9.0 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_type_non_intrusive/ +*/ +#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE(Type, ...) \ + template::value, int> = 0> \ + void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } \ + template::value, int> = 0> \ + void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT +@since version 3.11.0 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_type_non_intrusive/ +*/ +#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_WITH_DEFAULT(Type, ...) \ + template::value, int> = 0> \ + void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } \ + template::value, int> = 0> \ + void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE +@since version 3.11.3 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_type_non_intrusive/ +*/ +#define NLOHMANN_DEFINE_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(Type, ...) \ + template::value, int> = 0> \ + void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE +@since version 3.12.0 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ +*/ +#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE(Type, BaseType, ...) \ + template::value, int> = 0> \ + friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } \ + template::value, int> = 0> \ + friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT +@since version 3.12.0 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ +*/ +#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_WITH_DEFAULT(Type, BaseType, ...) \ + template::value, int> = 0> \ + friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } \ + template::value, int> = 0> \ + friend void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE +@since version 3.12.0 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ +*/ +#define NLOHMANN_DEFINE_DERIVED_TYPE_INTRUSIVE_ONLY_SERIALIZE(Type, BaseType, ...) \ + template::value, int> = 0> \ + friend void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE +@since version 3.12.0 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ +*/ +#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE(Type, BaseType, ...) \ + template::value, int> = 0> \ + void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } \ + template::value, int> = 0> \ + void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT +@since version 3.12.0 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ +*/ +#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_WITH_DEFAULT(Type, BaseType, ...) \ + template::value, int> = 0> \ + void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } \ + template::value, int> = 0> \ + void from_json(const BasicJsonType& nlohmann_json_j, Type& nlohmann_json_t) { nlohmann::from_json(nlohmann_json_j, static_cast(nlohmann_json_t)); const Type nlohmann_json_default_obj{}; NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_FROM_WITH_DEFAULT, __VA_ARGS__)) } + +/*! +@brief macro +@def NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE +@since version 3.12.0 +@sa https://json.nlohmann.me/api/macros/nlohmann_define_derived_type/ +*/ +#define NLOHMANN_DEFINE_DERIVED_TYPE_NON_INTRUSIVE_ONLY_SERIALIZE(Type, BaseType, ...) \ + template::value, int> = 0> \ + void to_json(BasicJsonType& nlohmann_json_j, const Type& nlohmann_json_t) { nlohmann::to_json(nlohmann_json_j, static_cast(nlohmann_json_t)); NLOHMANN_JSON_EXPAND(NLOHMANN_JSON_PASTE(NLOHMANN_JSON_TO, __VA_ARGS__)) } + +// inspired from https://stackoverflow.com/a/26745591 +// allows calling any std function as if (e.g., with begin): +// using std::begin; begin(x); +// +// it allows using the detected idiom to retrieve the return type +// of such an expression +#define NLOHMANN_CAN_CALL_STD_FUNC_IMPL(std_name) \ + namespace detail { \ + using std::std_name; \ + \ + template \ + using result_of_##std_name = decltype(std_name(std::declval()...)); \ + } \ + \ + namespace detail2 { \ + struct std_name##_tag \ + { \ + }; \ + \ + template \ + std_name##_tag std_name(T&&...); \ + \ + template \ + using result_of_##std_name = decltype(std_name(std::declval()...)); \ + \ + template \ + struct would_call_std_##std_name \ + { \ + static constexpr auto const value = ::nlohmann::detail:: \ + is_detected_exact::value; \ + }; \ + } /* namespace detail2 */ \ + \ + template \ + struct would_call_std_##std_name : detail2::would_call_std_##std_name \ + { \ + } + +#ifndef JSON_USE_IMPLICIT_CONVERSIONS + #define JSON_USE_IMPLICIT_CONVERSIONS 1 +#endif + +#if JSON_USE_IMPLICIT_CONVERSIONS + #define JSON_EXPLICIT +#else + #define JSON_EXPLICIT explicit +#endif + +#ifndef JSON_DISABLE_ENUM_SERIALIZATION + #define JSON_DISABLE_ENUM_SERIALIZATION 0 +#endif + +#ifndef JSON_USE_GLOBAL_UDLS + #define JSON_USE_GLOBAL_UDLS 1 +#endif diff --git a/src/detail/include/nlohmann/detail/macro_unscope.hpp b/src/detail/include/nlohmann/detail/macro_unscope.hpp new file mode 100644 index 000000000..b3d5188c4 --- /dev/null +++ b/src/detail/include/nlohmann/detail/macro_unscope.hpp @@ -0,0 +1,47 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +// restore clang diagnostic settings +#if defined(__clang__) + #pragma clang diagnostic pop +#endif + +// clean up +#undef JSON_ASSERT +#undef JSON_INTERNAL_CATCH +#undef JSON_THROW +#undef JSON_PRIVATE_UNLESS_TESTED +#undef NLOHMANN_BASIC_JSON_TPL_DECLARATION +#undef NLOHMANN_BASIC_JSON_TPL +#undef JSON_EXPLICIT +#undef NLOHMANN_CAN_CALL_STD_FUNC_IMPL +#undef JSON_INLINE_VARIABLE +#undef JSON_NO_UNIQUE_ADDRESS +#undef JSON_DISABLE_ENUM_SERIALIZATION +#undef JSON_USE_GLOBAL_UDLS + +#ifndef JSON_TEST_KEEP_MACROS + #undef JSON_CATCH + #undef JSON_TRY + #undef JSON_HAS_CPP_11 + #undef JSON_HAS_CPP_14 + #undef JSON_HAS_CPP_17 + #undef JSON_HAS_CPP_20 + #undef JSON_HAS_CPP_23 + #undef JSON_HAS_CPP_26 + #undef JSON_HAS_FILESYSTEM + #undef JSON_HAS_EXPERIMENTAL_FILESYSTEM + #undef JSON_HAS_THREE_WAY_COMPARISON + #undef JSON_HAS_RANGES + #undef JSON_HAS_STATIC_RTTI + #undef JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON +#endif + +#include diff --git a/src/detail/include/nlohmann/detail/meta/call_std/begin.hpp b/src/detail/include/nlohmann/detail/meta/call_std/begin.hpp new file mode 100644 index 000000000..ca9b79c10 --- /dev/null +++ b/src/detail/include/nlohmann/detail/meta/call_std/begin.hpp @@ -0,0 +1,17 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN + +NLOHMANN_CAN_CALL_STD_FUNC_IMPL(begin); + +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/meta/call_std/end.hpp b/src/detail/include/nlohmann/detail/meta/call_std/end.hpp new file mode 100644 index 000000000..724749e1b --- /dev/null +++ b/src/detail/include/nlohmann/detail/meta/call_std/end.hpp @@ -0,0 +1,17 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN + +NLOHMANN_CAN_CALL_STD_FUNC_IMPL(end); + +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/meta/cpp_future.hpp b/src/detail/include/nlohmann/detail/meta/cpp_future.hpp new file mode 100644 index 000000000..08ba5f7e9 --- /dev/null +++ b/src/detail/include/nlohmann/detail/meta/cpp_future.hpp @@ -0,0 +1,171 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-FileCopyrightText: 2018 The Abseil Authors +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // size_t +#include // conditional, enable_if, false_type, integral_constant, is_constructible, is_integral, is_same, remove_cv, remove_reference, true_type +#include // index_sequence, make_index_sequence, index_sequence_for + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +template +using uncvref_t = typename std::remove_cv::type>::type; + +#ifdef JSON_HAS_CPP_14 + +// the following utilities are natively available in C++14 +using std::enable_if_t; +using std::index_sequence; +using std::make_index_sequence; +using std::index_sequence_for; + +#else + +// alias templates to reduce boilerplate +template +using enable_if_t = typename std::enable_if::type; + +// The following code is taken from https://github.com/abseil/abseil-cpp/blob/10cb35e459f5ecca5b2ff107635da0bfa41011b4/absl/utility/utility.h +// which is part of Google Abseil (https://github.com/abseil/abseil-cpp), licensed under the Apache License 2.0. + +//// START OF CODE FROM GOOGLE ABSEIL + +// integer_sequence +// +// Class template representing a compile-time integer sequence. An instantiation +// of `integer_sequence` has a sequence of integers encoded in its +// type through its template arguments (which is a common need when +// working with C++11 variadic templates). `absl::integer_sequence` is designed +// to be a drop-in replacement for C++14's `std::integer_sequence`. +// +// Example: +// +// template< class T, T... Ints > +// void user_function(integer_sequence); +// +// int main() +// { +// // user_function's `T` will be deduced to `int` and `Ints...` +// // will be deduced to `0, 1, 2, 3, 4`. +// user_function(make_integer_sequence()); +// } +template +struct integer_sequence +{ + using value_type = T; + static constexpr std::size_t size() noexcept + { + return sizeof...(Ints); + } +}; + +// index_sequence +// +// A helper template for an `integer_sequence` of `size_t`, +// `absl::index_sequence` is designed to be a drop-in replacement for C++14's +// `std::index_sequence`. +template +using index_sequence = integer_sequence; + +namespace utility_internal +{ + +template +struct Extend; + +// Note that SeqSize == sizeof...(Ints). It's passed explicitly for efficiency. +template +struct Extend, SeqSize, 0> +{ + using type = integer_sequence < T, Ints..., (Ints + SeqSize)... >; +}; + +template +struct Extend, SeqSize, 1> +{ + using type = integer_sequence < T, Ints..., (Ints + SeqSize)..., 2 * SeqSize >; +}; + +// Recursion helper for 'make_integer_sequence'. +// 'Gen::type' is an alias for 'integer_sequence'. +template +struct Gen +{ + using type = + typename Extend < typename Gen < T, N / 2 >::type, N / 2, N % 2 >::type; +}; + +template +struct Gen +{ + using type = integer_sequence; +}; + +} // namespace utility_internal + +// Compile-time sequences of integers + +// make_integer_sequence +// +// This template alias is equivalent to +// `integer_sequence`, and is designed to be a drop-in +// replacement for C++14's `std::make_integer_sequence`. +template +using make_integer_sequence = typename utility_internal::Gen::type; + +// make_index_sequence +// +// This template alias is equivalent to `index_sequence<0, 1, ..., N-1>`, +// and is designed to be a drop-in replacement for C++14's +// `std::make_index_sequence`. +template +using make_index_sequence = make_integer_sequence; + +// index_sequence_for +// +// Converts a typename pack into an index sequence of the same length, and +// is designed to be a drop-in replacement for C++14's +// `std::index_sequence_for()` +template +using index_sequence_for = make_index_sequence; + +//// END OF CODE FROM GOOGLE ABSEIL + +#endif + +// dispatch utility (taken from ranges-v3) +template struct priority_tag : priority_tag < N - 1 > {}; +template<> struct priority_tag<0> {}; + +// taken from ranges-v3 +template +struct static_const +{ + static JSON_INLINE_VARIABLE constexpr T value{}; +}; + +#ifndef JSON_HAS_CPP_17 + template + constexpr T static_const::value; +#endif + +template +constexpr std::array make_array(Args&& ... args) +{ + return std::array {{static_cast(std::forward(args))...}}; +} + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/meta/detected.hpp b/src/detail/include/nlohmann/detail/meta/detected.hpp new file mode 100644 index 000000000..1b91ee0ed --- /dev/null +++ b/src/detail/include/nlohmann/detail/meta/detected.hpp @@ -0,0 +1,70 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +// https://en.cppreference.com/w/cpp/experimental/is_detected +struct nonesuch +{ + nonesuch() = delete; + ~nonesuch() = delete; + nonesuch(nonesuch const&) = delete; + nonesuch(nonesuch const&&) = delete; + void operator=(nonesuch const&) = delete; + void operator=(nonesuch&&) = delete; +}; + +template class Op, + class... Args> +struct detector +{ + using value_t = std::false_type; + using type = Default; +}; + +template class Op, class... Args> +struct detector>, Op, Args...> +{ + using value_t = std::true_type; + using type = Op; +}; + +template class Op, class... Args> +using is_detected = typename detector::value_t; + +template class Op, class... Args> +struct is_detected_lazy : is_detected { }; + +template class Op, class... Args> +using detected_t = typename detector::type; + +template class Op, class... Args> +using detected_or = detector; + +template class Op, class... Args> +using detected_or_t = typename detected_or::type; + +template class Op, class... Args> +using is_detected_exact = std::is_same>; + +template class Op, class... Args> +using is_detected_convertible = + std::is_convertible, To>; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/meta/identity_tag.hpp b/src/detail/include/nlohmann/detail/meta/identity_tag.hpp new file mode 100644 index 000000000..254191962 --- /dev/null +++ b/src/detail/include/nlohmann/detail/meta/identity_tag.hpp @@ -0,0 +1,21 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +// dispatching helper struct +template struct identity_tag {}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/meta/is_sax.hpp b/src/detail/include/nlohmann/detail/meta/is_sax.hpp new file mode 100644 index 000000000..21d7af8a4 --- /dev/null +++ b/src/detail/include/nlohmann/detail/meta/is_sax.hpp @@ -0,0 +1,159 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t +#include // declval +#include // string + +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +template +using null_function_t = decltype(std::declval().null()); + +template +using boolean_function_t = + decltype(std::declval().boolean(std::declval())); + +template +using number_integer_function_t = + decltype(std::declval().number_integer(std::declval())); + +template +using number_unsigned_function_t = + decltype(std::declval().number_unsigned(std::declval())); + +template +using number_float_function_t = decltype(std::declval().number_float( + std::declval(), std::declval())); + +template +using string_function_t = + decltype(std::declval().string(std::declval())); + +template +using binary_function_t = + decltype(std::declval().binary(std::declval())); + +template +using start_object_function_t = + decltype(std::declval().start_object(std::declval())); + +template +using key_function_t = + decltype(std::declval().key(std::declval())); + +template +using end_object_function_t = decltype(std::declval().end_object()); + +template +using start_array_function_t = + decltype(std::declval().start_array(std::declval())); + +template +using end_array_function_t = decltype(std::declval().end_array()); + +template +using parse_error_function_t = decltype(std::declval().parse_error( + std::declval(), std::declval(), + std::declval())); + +template +struct is_sax +{ + private: + static_assert(is_basic_json::value, + "BasicJsonType must be of type basic_json<...>"); + + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using string_t = typename BasicJsonType::string_t; + using binary_t = typename BasicJsonType::binary_t; + using exception_t = typename BasicJsonType::exception; + + public: + static constexpr bool value = + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value && + is_detected_exact::value; +}; + +template +struct is_sax_static_asserts +{ + private: + static_assert(is_basic_json::value, + "BasicJsonType must be of type basic_json<...>"); + + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using string_t = typename BasicJsonType::string_t; + using binary_t = typename BasicJsonType::binary_t; + using exception_t = typename BasicJsonType::exception; + + public: + static_assert(is_detected_exact::value, + "Missing/invalid function: bool null()"); + static_assert(is_detected_exact::value, + "Missing/invalid function: bool boolean(bool)"); + static_assert(is_detected_exact::value, + "Missing/invalid function: bool boolean(bool)"); + static_assert( + is_detected_exact::value, + "Missing/invalid function: bool number_integer(number_integer_t)"); + static_assert( + is_detected_exact::value, + "Missing/invalid function: bool number_unsigned(number_unsigned_t)"); + static_assert(is_detected_exact::value, + "Missing/invalid function: bool number_float(number_float_t, const string_t&)"); + static_assert( + is_detected_exact::value, + "Missing/invalid function: bool string(string_t&)"); + static_assert( + is_detected_exact::value, + "Missing/invalid function: bool binary(binary_t&)"); + static_assert(is_detected_exact::value, + "Missing/invalid function: bool start_object(std::size_t)"); + static_assert(is_detected_exact::value, + "Missing/invalid function: bool key(string_t&)"); + static_assert(is_detected_exact::value, + "Missing/invalid function: bool end_object()"); + static_assert(is_detected_exact::value, + "Missing/invalid function: bool start_array(std::size_t)"); + static_assert(is_detected_exact::value, + "Missing/invalid function: bool end_array()"); + static_assert( + is_detected_exact::value, + "Missing/invalid function: bool parse_error(std::size_t, const " + "std::string&, const exception&)"); +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/src/detail/include/nlohmann/detail/meta/std_fs.hpp b/src/detail/include/nlohmann/detail/meta/std_fs.hpp new file mode 100644 index 000000000..ff4d39528 --- /dev/null +++ b/src/detail/include/nlohmann/detail/meta/std_fs.hpp @@ -0,0 +1,29 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include + +#if JSON_HAS_EXPERIMENTAL_FILESYSTEM +#include +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace std_fs = std::experimental::filesystem; +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END +#elif JSON_HAS_FILESYSTEM +#include // NOLINT(build/c++17) +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace std_fs = std::filesystem; +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END +#endif diff --git a/src/detail/include/nlohmann/detail/meta/type_traits.hpp b/src/detail/include/nlohmann/detail/meta/type_traits.hpp new file mode 100644 index 000000000..18e160bbd --- /dev/null +++ b/src/detail/include/nlohmann/detail/meta/type_traits.hpp @@ -0,0 +1,821 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013 - 2025 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // numeric_limits +#include // char_traits +#include // tuple +#include // false_type, is_constructible, is_integral, is_same, true_type +#include // declval +#if defined(__cpp_lib_byte) && __cpp_lib_byte >= 201603L + #include // byte +#endif +#include +#include +#include +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +/*! +@brief detail namespace with internal helper functions + +This namespace collects functions that should not be exposed, +implementations of some @ref basic_json methods, and meta-programming helpers. + +@since version 2.1.0 +*/ +namespace detail +{ + +///////////// +// helpers // +///////////// + +// Note to maintainers: +// +// Every trait in this file expects a non-CV-qualified type. +// The only exceptions are in the 'aliases for detected' section +// (i.e., those of the form: decltype(T::member_function(std::declval()))) +// +// In this case, T has to be properly CV-qualified to constraint the function arguments +// (e.g., to_json(BasicJsonType&, const T&)) + +template struct is_basic_json : std::false_type {}; + +NLOHMANN_BASIC_JSON_TPL_DECLARATION +struct is_basic_json : std::true_type {}; + +// used by exceptions create() member functions +// true_type for the pointer to possibly cv-qualified basic_json or std::nullptr_t +// false_type otherwise +template +struct is_basic_json_context : + std::integral_constant < bool, + is_basic_json::type>::type>::value + || std::is_same::value > +{}; + +////////////////////// +// json_ref helpers // +////////////////////// + +template +class json_ref; + +template +struct is_json_ref : std::false_type {}; + +template +struct is_json_ref> : std::true_type {}; + +////////////////////////// +// aliases for detected // +////////////////////////// + +template +using mapped_type_t = typename T::mapped_type; + +template +using key_type_t = typename T::key_type; + +template +using value_type_t = typename T::value_type; + +template +using difference_type_t = typename T::difference_type; + +template +using pointer_t = typename T::pointer; + +template +using reference_t = typename T::reference; + +template +using iterator_category_t = typename T::iterator_category; + +template +using to_json_function = decltype(T::to_json(std::declval()...)); + +template +using from_json_function = decltype(T::from_json(std::declval()...)); + +template +using get_template_function = decltype(std::declval().template get()); + +// trait checking if JSONSerializer::from_json(json const&, udt&) exists +template +struct has_from_json : std::false_type {}; + +// trait checking if j.get is valid +// use this trait instead of std::is_constructible or std::is_convertible, +// both rely on, or make use of implicit conversions, and thus fail when T +// has several constructors/operator= (see https://github.com/nlohmann/json/issues/958) +template +struct is_getable +{ + static constexpr bool value = is_detected::value; +}; + +template +struct has_from_json < BasicJsonType, T, enable_if_t < !is_basic_json::value >> +{ + using serializer = typename BasicJsonType::template json_serializer; + + static constexpr bool value = + is_detected_exact::value; +}; + +// This trait checks if JSONSerializer::from_json(json const&) exists +// this overload is used for non-default-constructible user-defined-types +template +struct has_non_default_from_json : std::false_type {}; + +template +struct has_non_default_from_json < BasicJsonType, T, enable_if_t < !is_basic_json::value >> +{ + using serializer = typename BasicJsonType::template json_serializer; + + static constexpr bool value = + is_detected_exact::value; +}; + +// This trait checks if BasicJsonType::json_serializer::to_json exists +// Do not evaluate the trait when T is a basic_json type, to avoid template instantiation infinite recursion. +template +struct has_to_json : std::false_type {}; + +template +struct has_to_json < BasicJsonType, T, enable_if_t < !is_basic_json::value >> +{ + using serializer = typename BasicJsonType::template json_serializer; + + static constexpr bool value = + is_detected_exact::value; +}; + +template +using detect_key_compare = typename T::key_compare; + +template +struct has_key_compare : std::integral_constant::value> {}; + +// obtains the actual object key comparator +template +struct actual_object_comparator +{ + using object_t = typename BasicJsonType::object_t; + using object_comparator_t = typename BasicJsonType::default_object_comparator_t; + using type = typename std::conditional < has_key_compare::value, + typename object_t::key_compare, object_comparator_t>::type; +}; + +template +using actual_object_comparator_t = typename actual_object_comparator::type; + +///////////////// +// char_traits // +///////////////// + +// Primary template of char_traits calls std char_traits +template +struct char_traits : std::char_traits +{}; + +// Explicitly define char traits for unsigned char since it is not standard +template<> +struct char_traits : std::char_traits +{ + using char_type = unsigned char; + using int_type = uint64_t; + + // Redefine to_int_type function + static int_type to_int_type(char_type c) noexcept + { + return static_cast(c); + } + + static char_type to_char_type(int_type i) noexcept + { + return static_cast(i); + } + + static constexpr int_type eof() noexcept + { + return static_cast(std::char_traits::eof()); + } +}; + +// Explicitly define char traits for signed char since it is not standard +template<> +struct char_traits : std::char_traits +{ + using char_type = signed char; + using int_type = uint64_t; + + // Redefine to_int_type function + static int_type to_int_type(char_type c) noexcept + { + return static_cast(c); + } + + static char_type to_char_type(int_type i) noexcept + { + return static_cast(i); + } + + static constexpr int_type eof() noexcept + { + return static_cast(std::char_traits::eof()); + } +}; + +#if defined(__cpp_lib_byte) && __cpp_lib_byte >= 201603L +template<> +struct char_traits : std::char_traits +{ + using char_type = std::byte; + using int_type = uint64_t; + + static int_type to_int_type(char_type c) noexcept + { + return static_cast(std::to_integer(c)); + } + + static char_type to_char_type(int_type i) noexcept + { + return std::byte(static_cast(i)); + } + + static constexpr int_type eof() noexcept + { + return static_cast(std::char_traits::eof()); + } +}; +#endif + +/////////////////// +// is_ functions // +/////////////////// + +// https://en.cppreference.com/w/cpp/types/conjunction +template struct conjunction : std::true_type { }; +template struct conjunction : B { }; +template +struct conjunction +: std::conditional(B::value), conjunction, B>::type {}; + +// https://en.cppreference.com/w/cpp/types/negation +template struct negation : std::integral_constant < bool, !B::value > { }; + +// Reimplementation of is_constructible and is_default_constructible, due to them being broken for +// std::pair and std::tuple until LWG 2367 fix (see https://cplusplus.github.io/LWG/lwg-defects.html#2367). +// This causes compile errors in e.g., Clang 3.5 or GCC 4.9. +template +struct is_default_constructible : std::is_default_constructible {}; + +template +struct is_default_constructible> + : conjunction, is_default_constructible> {}; + +template +struct is_default_constructible> + : conjunction, is_default_constructible> {}; + +template +struct is_default_constructible> + : conjunction...> {}; + +template +struct is_default_constructible> + : conjunction...> {}; + +template +struct is_constructible : std::is_constructible {}; + +template +struct is_constructible> : is_default_constructible> {}; + +template +struct is_constructible> : is_default_constructible> {}; + +template +struct is_constructible> : is_default_constructible> {}; + +template +struct is_constructible> : is_default_constructible> {}; + +template +struct is_iterator_traits : std::false_type {}; + +template +struct is_iterator_traits> +{ + private: + using traits = iterator_traits; + + public: + static constexpr auto value = + is_detected::value && + is_detected::value && + is_detected::value && + is_detected::value && + is_detected::value; +}; + +template +struct is_range +{ + private: + using t_ref = typename std::add_lvalue_reference::type; + + using iterator = detected_t; + using sentinel = detected_t; + + // to be 100% correct, it should use https://en.cppreference.com/w/cpp/iterator/input_or_output_iterator + // and https://en.cppreference.com/w/cpp/iterator/sentinel_for + // but reimplementing these would be too much work, as a lot of other concepts are used underneath + static constexpr auto is_iterator_begin = + is_iterator_traits>::value; + + public: + static constexpr bool value = !std::is_same::value && !std::is_same::value && is_iterator_begin; +}; + +template +using iterator_t = enable_if_t::value, result_of_begin())>>; + +template +using range_value_t = value_type_t>>; + +// The following implementation of is_complete_type is taken from +// https://blogs.msdn.microsoft.com/vcblog/2015/12/02/partial-support-for-expression-sfinae-in-vs-2015-update-1/ +// and is written by Xiang Fan who agreed to use it in this library. + +template +struct is_complete_type : std::false_type {}; + +template +struct is_complete_type : std::true_type {}; + +template +struct is_compatible_object_type_impl : std::false_type {}; + +template +struct is_compatible_object_type_impl < + BasicJsonType, CompatibleObjectType, + enable_if_t < is_detected::value&& + is_detected::value >> +{ + using object_t = typename BasicJsonType::object_t; + + // macOS's is_constructible does not play well with nonesuch... + static constexpr bool value = + is_constructible::value && + is_constructible::value; +}; + +template +struct is_compatible_object_type + : is_compatible_object_type_impl {}; + +template +struct is_constructible_object_type_impl : std::false_type {}; + +template +struct is_constructible_object_type_impl < + BasicJsonType, ConstructibleObjectType, + enable_if_t < is_detected::value&& + is_detected::value >> +{ + using object_t = typename BasicJsonType::object_t; + + static constexpr bool value = + (is_default_constructible::value && + (std::is_move_assignable::value || + std::is_copy_assignable::value) && + (is_constructible::value && + std::is_same < + typename object_t::mapped_type, + typename ConstructibleObjectType::mapped_type >::value)) || + (has_from_json::value || + has_non_default_from_json < + BasicJsonType, + typename ConstructibleObjectType::mapped_type >::value); +}; + +template +struct is_constructible_object_type + : is_constructible_object_type_impl {}; + +template +struct is_compatible_string_type +{ + static constexpr auto value = + is_constructible::value; +}; + +template +struct is_constructible_string_type +{ + // launder type through decltype() to fix compilation failure on ICPC +#ifdef __INTEL_COMPILER + using laundered_type = decltype(std::declval()); +#else + using laundered_type = ConstructibleStringType; +#endif + + static constexpr auto value = + conjunction < + is_constructible, + is_detected_exact>::value; +}; + +template +struct is_compatible_array_type_impl : std::false_type {}; + +template +struct is_compatible_array_type_impl < + BasicJsonType, CompatibleArrayType, + enable_if_t < + is_detected::value&& + is_iterator_traits>>::value&& +// special case for types like std::filesystem::path whose iterator's value_type are themselves +// c.f. https://github.com/nlohmann/json/pull/3073 + !std::is_same>::value >> +{ + static constexpr bool value = + is_constructible>::value; +}; + +template +struct is_compatible_array_type + : is_compatible_array_type_impl {}; + +template +struct is_constructible_array_type_impl : std::false_type {}; + +template +struct is_constructible_array_type_impl < + BasicJsonType, ConstructibleArrayType, + enable_if_t::value >> + : std::true_type {}; + +template +struct is_constructible_array_type_impl < + BasicJsonType, ConstructibleArrayType, + enable_if_t < !std::is_same::value&& + !is_compatible_string_type::value&& + is_default_constructible::value&& +(std::is_move_assignable::value || + std::is_copy_assignable::value)&& +is_detected::value&& +is_iterator_traits>>::value&& +is_detected::value&& +// special case for types like std::filesystem::path whose iterator's value_type are themselves +// c.f. https://github.com/nlohmann/json/pull/3073 +!std::is_same>::value&& +is_complete_type < +detected_t>::value >> +{ + using value_type = range_value_t; + + static constexpr bool value = + std::is_same::value || + has_from_json::value || + has_non_default_from_json < + BasicJsonType, + value_type >::value; +}; + +template +struct is_constructible_array_type + : is_constructible_array_type_impl {}; + +template +struct is_compatible_integer_type_impl : std::false_type {}; + +template +struct is_compatible_integer_type_impl < + RealIntegerType, CompatibleNumberIntegerType, + enable_if_t < std::is_integral::value&& + std::is_integral::value&& + !std::is_same::value >> +{ + // is there an assert somewhere on overflows? + using RealLimits = std::numeric_limits; + using CompatibleLimits = std::numeric_limits; + + static constexpr auto value = + is_constructible::value && + CompatibleLimits::is_integer && + RealLimits::is_signed == CompatibleLimits::is_signed; +}; + +template +struct is_compatible_integer_type + : is_compatible_integer_type_impl {}; + +template +struct is_compatible_type_impl: std::false_type {}; + +template +struct is_compatible_type_impl < + BasicJsonType, CompatibleType, + enable_if_t::value >> +{ + static constexpr bool value = + has_to_json::value; +}; + +template +struct is_compatible_type + : is_compatible_type_impl {}; + +template +struct is_constructible_tuple : std::false_type {}; + +template +struct is_constructible_tuple> : conjunction...> {}; + +template +struct is_json_iterator_of : std::false_type {}; + +template +struct is_json_iterator_of : std::true_type {}; + +template +struct is_json_iterator_of : std::true_type +{}; + +// checks if a given type T is a template specialization of Primary +template