#include "ggml-decoder.h"
#include "ggml-impl.h"
#include "ggml-openvino-extra.h"
#include "ggml-openvino.h"
#include "ggml-quants.h"
#include "ggml.h"
#include "utils.h"
#include <algorithm>
#include <cassert>
#include <cstddef>
#include <cstdint>
#include <cstdlib>
#include <fstream>
#include <iomanip>
#include <map>
#include <memory>
#include <mutex>
#include <openvino/core/dimension.hpp>
#include <openvino/core/except.hpp>
#include <openvino/core/node.hpp>
#include <openvino/core/partial_shape.hpp>
#include <openvino/core/type/bfloat16.hpp>
#include <openvino/core/type/element_type.hpp>
#include <openvino/core/type/float16.hpp>
#include <openvino/op/constant.hpp>
#include <openvino/op/convert.hpp>
#include <openvino/runtime/tensor.hpp>
#include <ostream>
#include <set>
#include <stdexcept>
#include <string>
#include <cstring>
#include <unordered_map>
#include <vector>
GgmlOvDecoder::GgmlOvDecoder(ggml_cgraph * cgraph,
ModelParams & model_params,
ComputeParams & compute_params,
std::map<std::string, std::shared_ptr<ov::Node>> & model_weights,
bool is_static,
bool is_stateful,
bool model_is_splitted,
bool is_prefill,
int prefill_chunk_size) :
m_is_static(is_static),
m_is_stateful(is_stateful),
m_is_prefill(is_prefill),
m_naive(false),
m_prefill_chunk_size(prefill_chunk_size),
m_model_is_splitted(model_is_splitted),
m_cgraph(cgraph),
m_model_weights(model_weights),
m_model_params(model_params),
m_compute_params(compute_params) {
static bool printed_address_map = false;
if (!printed_address_map) {
if (ggml_openvino_getenv_int("GGML_OPENVINO_PRINT_CGRAPH_TENSOR_ADDRESS")) {
printed_address_map = true;
print_tensor_address_map(cgraph);
}
}
validate_cgraph();
set_input_output();
compute_node_dynamic_dims();
compute_model_inputs();
compute_model_outputs();
for (int node_n = 0; node_n < cgraph->n_nodes; node_n++) {
m_node_info_list[node_n].node_op_case = compute_op_case(m_node_info_list[node_n].node);
m_node_info_list[node_n].node_op_type = compute_op_type(m_node_info_list[node_n].node);
}
add_extra_inputs();
}
void GgmlOvDecoder::update_io(ggml_cgraph * cgraph) {
m_cgraph = cgraph;
m_model_inputs.clear();
m_model_outputs.clear();
m_node_info_list.clear();
set_input_output();
compute_model_inputs();
compute_model_outputs();
}
GgmlOvDecoder::GgmlOvDecoder(ggml_cgraph * cgraph, std::map<std::string, std::shared_ptr<ov::Node>> & model_weights) {
m_cgraph = cgraph;
m_model_weights = model_weights;
m_naive = true;
set_input_output();
compute_model_inputs();
compute_model_outputs();
for (int node_n = 0; node_n < cgraph->n_nodes; node_n++) {
m_node_info_list[node_n].node_op_case = compute_op_case(m_node_info_list[node_n].node);
m_node_info_list[node_n].node_op_type = compute_op_type(m_node_info_list[node_n].node);
}
}
namespace {
bool is_inplace_op(const ggml_tensor * node) {
return node->op == GGML_OP_SET_ROWS || node->op == GGML_OP_CPY || (node->op == GGML_OP_SCALE && node->view_src);
}
bool is_same_shape(const ggml_tensor * a, const ggml_tensor * b) {
for (int i = 0; i < GGML_MAX_DIMS; i++) {
if (a->ne[i] != b->ne[i]) {
return false;
}
}
return true;
}
bool is_conv_states_all_tensor(const ggml_tensor * tensor) {
return tensor != nullptr && strncmp(tensor->name, "conv_states_all", strlen("conv_states_all")) == 0;
}
}
bool is_moe_expert_sum_add(const ggml_tensor * node) {
std::vector<const ggml_tensor *> leaves;
const ggml_tensor * cur = node;
while (cur->op == GGML_OP_ADD) {
if (cur->src[0] == nullptr || cur->src[1] == nullptr) {
return false;
}
leaves.push_back(cur->src[1]);
cur = cur->src[0];
}
leaves.push_back(cur);
const ggml_tensor * base = nullptr;
std::set<int64_t> plane_indices;
for (const ggml_tensor * leaf : leaves) {
if (leaf->op != GGML_OP_VIEW || leaf->src[0] == nullptr) {
return false;
}
const ggml_tensor * leaf_base = leaf->src[0];
if (base == nullptr) {
base = leaf_base;
} else if (leaf_base != base) {
return false;
}
if (leaf->ne[0] != base->ne[0] || leaf->ne[1] != base->ne[2] || leaf->ne[2] != 1 || leaf->ne[3] != 1 ||
leaf->nb[1] != base->nb[2]) {
return false;
}
if (base->nb[1] == 0 || leaf->view_offs % base->nb[1] != 0) {
return false;
}
int64_t plane = static_cast<int64_t>(leaf->view_offs / base->nb[1]);
if (plane < 0 || plane >= base->ne[1] || !plane_indices.insert(plane).second) {
return false;
}
}
return base != nullptr && base->ne[1] > 1 && plane_indices.size() == static_cast<size_t>(base->ne[1]);
}
std::string GgmlOvDecoder::get_tensor_name(const ggml_cgraph * cgraph, const ggml_tensor * tensor) {
if (tensor == nullptr) {
return "";
}
if ((tensor->flags & GGML_TENSOR_FLAG_COMPUTE) || is_kvcache(tensor, nullptr)) {
const auto * node = std::find(cgraph->nodes, cgraph->nodes + cgraph->n_nodes, tensor);
if (node != cgraph->nodes + cgraph->n_nodes) {
return std::string(tensor->name) + "#n" + std::to_string(node - cgraph->nodes);
}
const auto * leaf = std::find(cgraph->leafs, cgraph->leafs + cgraph->n_leafs, tensor);
if (leaf != cgraph->leafs + cgraph->n_leafs) {
return std::string(tensor->name) + "#l" + std::to_string(leaf - cgraph->leafs);
}
}
return tensor->name;
}
static std::string get_tensor_ov_name(const ggml_cgraph * cgraph, const ggml_tensor * tensor) {
return GgmlOvDecoder::get_tensor_name(cgraph, tensor);
}
static std::string get_tensor_graph_input_ov_name(const GgmlOvDecoder * decoder,
const ggml_cgraph * cgraph,
const ggml_tensor * tensor,
const ggml_tensor * op) {
if (GgmlOvDecoder::is_inp_pos(tensor, op)) {
return "inp_pos";
}
if (GgmlOvDecoder::is_inp_emb(tensor, op)) {
return "embd";
}
if (GgmlOvDecoder::is_inp_mask(tensor, op)) {
const bool is_swa = decoder->is_swa_mask(tensor);
if (decoder->is_stateful()) {
return is_swa ? "self_kq_mask_swa" : "self_kq_mask";
}
if (is_swa) {
return get_tensor_ov_name(cgraph, tensor) + "_swa";
}
}
return get_tensor_ov_name(cgraph, tensor);
}
void GgmlOvDecoder::set_input_output() {
for (int node_n = 0; node_n < m_cgraph->n_nodes; node_n++) {
auto * node = m_cgraph->nodes[node_n];
NodeInfo current_node_info;
auto node_name = get_tensor_ov_name(m_cgraph, node);
current_node_info.node = node;
current_node_info.node_name = node_name;
current_node_info.node_op_case = 0;
current_node_info.data_addr = node->data;
for (int i = 0; i < GGML_MAX_SRC; i++) {
auto * src = node->src[i];
if (src == nullptr) {
continue;
}
auto src_name = get_tensor_ov_name(m_cgraph, src);
if (src->flags & GGML_TENSOR_FLAG_INPUT) {
src_name = get_tensor_graph_input_ov_name(this, m_cgraph, src, node);
}
current_node_info.node_inputs[src_name] = src;
current_node_info.node_inputs_names.push_back(src_name);
if (src->op == GGML_OP_VIEW) {
std::remove_reference_t<decltype(current_node_info.node_inputs_views[src_name])> view_chain;
auto * current = src;
while (current != nullptr) {
auto current_name = get_tensor_ov_name(m_cgraph, current);
if (current->flags & GGML_TENSOR_FLAG_INPUT) {
current_name = get_tensor_graph_input_ov_name(this, m_cgraph, current, node);
}
view_chain.emplace_back(current_name, current);
if (current->src[0] != nullptr && current->src[0]->op == GGML_OP_VIEW) {
current = current->src[0];
} else {
break;
}
}
current_node_info.node_inputs_views[src_name] = view_chain;
}
}
m_node_info_list.push_back(current_node_info);
}
}
int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
int op_case = 0;
switch (node->op) {
case GGML_OP_RESHAPE: {
auto name = std::string(node->name);
auto * src = node->src[0];
if (src->op == GGML_OP_RESHAPE && src->src[0]->ne[0] == node->ne[0] && src->src[0]->ne[1] == node->ne[1]) {
op_case = 4;
} else if (node->ne[0] * node->ne[1] == src->ne[0]) {
op_case = 1;
} else if (src->ne[0] * src->ne[1] == node->ne[0]) {
op_case = 2;
if (src->ne[2] * src->ne[3] == node->ne[1]) {
op_case = 5;
}
} else if (src->ne[0] * src->ne[1] * src->ne[2] == node->ne[1]) {
op_case = 3;
} else if (name.find("linear_attn_qkv_mixed") == 0 || name.find("alpha") == 0) {
op_case = 6;
} else if (name.find("linear_attn_out") == 0) {
op_case = 7;
} else if (name.find("state_predelta") == 0) {
op_case = 8;
}
break;
}
case GGML_OP_PERMUTE: {
if (node->src[0]->op != GGML_OP_VIEW) {
op_case = 1;
} else if (node->src[0]->src[0]->op == GGML_OP_NONE) {
std::string src_name(node->view_src->name);
int layer = extract_layer_from_name(src_name).value();
if (ggml_is_contiguous(node->src[0])) {
if (!is_swa_layer(layer)) {
op_case = 3;
} else {
op_case = 4;
}
} else {
if (!is_swa_layer(layer)) {
op_case = 5;
} else {
op_case = 6;
}
}
} else {
op_case = 2;
}
break;
}
case GGML_OP_MUL_MAT: {
if (node->src[1]->op == GGML_OP_SOFT_MAX) {
op_case = 2;
}
break;
}
case GGML_OP_GET_ROWS: {
if (node->src[1]->op == GGML_OP_VIEW) {
if (node->src[0]->op == GGML_OP_RESHAPE && node->src[0]->src[0] != nullptr &&
is_kvcache(node->src[0]->src[0], nullptr)) {
op_case = node->src[1]->view_offs == 0 ? 1 : 2;
}
}
break;
}
case GGML_OP_ROPE: {
const int mode = node->op_params[2];
switch (mode) {
case GGML_ROPE_TYPE_NEOX: {
op_case = 1;
break;
}
case GGML_ROPE_TYPE_IMROPE: {
op_case = 2;
break;
}
default:
op_case = 0;
break;
}
break;
}
case GGML_OP_VIEW: {
if (m_is_static && node->src[0] != nullptr &&
(node->src[0]->op == GGML_OP_GATED_DELTA_NET || node->src[0]->op == GGML_OP_CONCAT)) {
op_case = 1;
break;
}
if (node->src[0]->op == GGML_OP_VIEW) {
auto * src = node->src[0];
if (ggml_nelements(node) != ggml_nelements(src)) {
}
op_case = 0;
if (m_model_is_splitted && m_model_inputs.find(get_tensor_ov_name(m_cgraph, src)) != m_model_inputs.end()) {
op_case = 0;
}
}
{
auto * src = node->src[0];
if (ggml_nelements(node) != ggml_nelements(src)) {
if (node->ne[0] == src->ne[0] && node->ne[1] == src->ne[2] && node->ne[2] == 1 &&
node->nb[0] == src->nb[0] && node->nb[1] == src->nb[2] && src->ne[1] > 1) {
op_case = 0;
break;
}
int diff_count = 0;
for (int i = 0; i < GGML_MAX_DIMS; i++) {
if (node->ne[i] != src->ne[i]) {
diff_count++;
}
if (node->ne[i] > src->ne[i]) {
return 0;
}
}
if (diff_count >= 1) {
op_case = 0;
}
}
}
break;
}
case GGML_OP_RMS_NORM: {
if (node->src[0]->op == GGML_OP_VIEW) {
if (is_same_shape(node->src[0]->src[0], node->src[0])) {
op_case = 1;
} else if (node->src[0]->src[0]->op == GGML_OP_GATED_DELTA_NET) {
op_case = 2;
}
}
break;
}
case GGML_OP_POOL_2D: {
const ggml_op_pool pool_mode = static_cast<ggml_op_pool>(node->op_params[0]);
switch (pool_mode) {
case GGML_OP_POOL_MAX: {
op_case = 1;
break;
}
case GGML_OP_POOL_AVG: {
op_case = 2;
break;
}
default:
op_case = 0;
break;
}
break;
}
case GGML_OP_CPY: {
if (node->src[0]->op == GGML_OP_VIEW) {
if (node->src[0]->src[0]->op == GGML_OP_GATED_DELTA_NET) {
op_case = 1;
} else if (GgmlOvDecoder::is_conv_state_writeback(node)) {
op_case = 2;
break;
} else if (is_conv_states_all_tensor(node->view_src) && node->src[1] != nullptr &&
node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src == node->view_src) {
op_case = 4;
break;
}
} else if (node->src[0]->op == GGML_OP_GET_ROWS && node->src[1] != nullptr &&
node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src != nullptr &&
is_kvcache(node->src[1]->view_src, nullptr)) {
op_case = 3;
} else if (node->src[1] != nullptr && node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src != nullptr) {
const ggml_tensor * kv_buf = node->src[1]->view_src;
if (kv_buf->ne[1] == 1 && kv_buf->ne[2] == 1 && kv_buf->ne[3] == 1) {
op_case = 6;
for (int i = 0; i < m_cgraph->n_nodes; i++) {
const ggml_tensor * n = m_cgraph->nodes[i];
if (n->op != GGML_OP_FLASH_ATTN_EXT) {
continue;
}
if ((n->src[1] != nullptr && n->src[1]->view_src == kv_buf) ||
(n->src[2] != nullptr && n->src[2]->view_src == kv_buf)) {
if (n->src[3] != nullptr) {
op_case = 5; }
break;
}
}
}
}
break;
}
case GGML_OP_ADD: {
if (is_moe_expert_sum_add(node)) {
op_case = 1;
}
break;
}
case GGML_OP_SCALE: {
if (node->view_src && node->buffer->usage == GGML_BACKEND_BUFFER_USAGE_ANY) {
op_case = 1;
}
break;
}
case GGML_OP_L2_NORM: {
if (std::string(node->name).find("predelta") != std::string::npos) {
op_case = 1;
}
break;
}
case GGML_OP_FLASH_ATTN_EXT: {
if (node->src[1] != nullptr && node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src != nullptr) {
const ggml_tensor * kv_buf = node->src[1]->view_src;
if (kv_buf->ne[1] == 1 && kv_buf->ne[2] == 1 && kv_buf->ne[3] == 1) {
op_case = (node->src[3] != nullptr) ? 1 : 2;
}
}
break;
}
default:
break;
}
return op_case;
}
std::optional<int> extract_layer_from_name(const std::string & name) {
size_t pos1 = name.find("_l");
if (pos1 == std::string::npos) {
return std::nullopt;
}
pos1 += 2;
size_t pos2 = name.find(' ', pos1);
if (pos2 == std::string::npos) {
pos2 = name.length();
}
std::string layer_str = name.substr(pos1, pos2 - pos1);
int layer = std::stoi(layer_str);
return layer;
}
static int get_swa_window_from_mask(const ggml_tensor * mask) {
if (mask->data == nullptr || !ggml_backend_buffer_is_host(mask->buffer)) {
return -1;
}
if (mask->type != GGML_TYPE_F16 && mask->type != GGML_TYPE_F32) {
return -1;
}
const int64_t n_kv = mask->ne[0];
const int64_t n_tokens = mask->ne[1];
int64_t window = 0;
for (int64_t r = 0; r < n_tokens; r++) {
int64_t kept = 0;
for (int64_t c = 0; c < n_kv; c++) {
const size_t i = (size_t) r * n_kv + c;
const float v = mask->type == GGML_TYPE_F16 ? ggml_fp16_to_fp32(((const ggml_fp16_t *) mask->data)[i]) :
((const float *) mask->data)[i];
if (v > -INFINITY) {
kept++;
}
}
window = std::max(window, kept);
}
return window > 0 ? (int) window : -1;
}
std::pair<ModelParams, ComputeParams> GgmlOvDecoder::compute_llm_params(ggml_cgraph * cgraph, bool is_static) {
ModelParams model_params;
ComputeParams compute_params;
auto get_attention_pattern_case = [](const ggml_tensor * node) -> int {
if (node == nullptr) {
return -1;
}
switch (node->op) {
case GGML_OP_FLASH_ATTN_EXT:
if (node->src[0] == nullptr || node->src[1] == nullptr) {
return -1;
}
switch (node->src[1]->op) {
case GGML_OP_PERMUTE:
if (node->src[3] != nullptr && node->src[1]->src[0] != nullptr &&
node->src[1]->src[0]->op == GGML_OP_VIEW) {
return 0;
}
break;
case GGML_OP_CPY:
if (node->src[3] != nullptr && node->src[1]->src[0] != nullptr &&
node->src[1]->src[0]->op == GGML_OP_PERMUTE && node->src[1]->src[0]->src[0] != nullptr &&
node->src[1]->src[0]->src[0]->op == GGML_OP_VIEW) {
return 1;
}
break;
case GGML_OP_VIEW:
if (node->src[1]->view_src != nullptr) {
if (node->src[3] != nullptr) {
return 4; }
return 5; }
break;
default:
break;
}
break;
case GGML_OP_SOFT_MAX:
if (node->src[0] != nullptr && node->src[1] != nullptr && node->src[0]->op == GGML_OP_MUL_MAT &&
node->src[0]->src[0] != nullptr && node->src[0]->src[1] != nullptr &&
node->src[0]->src[0]->op == GGML_OP_PERMUTE && node->src[0]->src[0]->src[0] != nullptr &&
node->src[0]->src[0]->src[0]->op == GGML_OP_VIEW) {
return 2;
}
if (node->src[0]->op == GGML_OP_ADD && node->src[0]->src[0] != nullptr &&
node->src[0]->src[0]->op == GGML_OP_MUL_MAT && node->src[0]->src[0]->src[0] != nullptr &&
node->src[0]->src[0]->src[0]->op == GGML_OP_PERMUTE) {
return 3;
}
break;
default:
break;
}
return -1;
};
auto get_attention_op_mask = [&get_attention_pattern_case](const ggml_tensor * node) -> const ggml_tensor * {
switch (get_attention_pattern_case(node)) {
case 0:
case 1:
return node->src[3];
case 2:
case 3:
return node->src[1];
default:
return nullptr;
}
};
{
std::map<int, int64_t> layer_extent; std::map<int, const ggml_tensor *> layer_mask; int64_t max_extent = 0;
for (int i = 0; i < cgraph->n_nodes; i++) {
const ggml_tensor * mask = get_attention_op_mask(cgraph->nodes[i]);
if (mask == nullptr) {
continue;
}
const ggml_tensor * cache_k_permute = nullptr;
switch (get_attention_pattern_case(cgraph->nodes[i])) {
case 0: cache_k_permute = cgraph->nodes[i]->src[1]; break;
case 1: cache_k_permute = cgraph->nodes[i]->src[1]->src[0]; break;
case 2: cache_k_permute = cgraph->nodes[i]->src[0]->src[0]; break;
default: cache_k_permute = cgraph->nodes[i]->src[0]->src[0]->src[0]; break;
}
const ggml_tensor * cache_k_view = cache_k_permute->src[0];
if (cache_k_view->op != GGML_OP_VIEW) {
continue;
}
const ggml_tensor * leaf = cache_k_view->src[0];
auto layer = extract_layer_from_name(leaf->name);
if (!layer.has_value()) {
continue;
}
layer_extent[layer.value()] = leaf->ne[1];
layer_mask[layer.value()] = mask;
max_extent = std::max(max_extent, leaf->ne[1]);
}
for (const auto & [layer, extent] : layer_extent) {
if (extent < max_extent) {
model_params.swa_layers.push_back(layer);
if (model_params.swa_mask == nullptr) {
model_params.swa_mask = layer_mask[layer];
}
}
}
std::sort(model_params.swa_layers.begin(), model_params.swa_layers.end());
if (ggml_openvino_getenv_int("GGML_OPENVINO_LOG_SWA_LAYERS")) {
std::string per_layer;
for (const auto & [layer, extent] : layer_extent) {
per_layer += " " + std::to_string(layer) + ":" + std::to_string(extent) +
(extent < max_extent ? "(swa)" : "");
}
GGML_LOG_WARN("ov-swa: attn_layers=%zu max_extent=%ld swa_layers=%zu |%s\n", layer_extent.size(),
(long) max_extent, model_params.swa_layers.size(), per_layer.c_str());
}
}
bool rope_seen = false;
for (int i = 0; i < cgraph->n_nodes; i++) {
ggml_tensor * node = cgraph->nodes[i];
const int attention_pattern_case = get_attention_pattern_case(node);
if (attention_pattern_case != -1) {
ggml_tensor * cache_k_permute = nullptr;
ggml_tensor * mask = nullptr;
switch (attention_pattern_case) {
case 0:
cache_k_permute = node->src[1];
mask = node->src[3];
break;
case 1:
cache_k_permute = node->src[1]->src[0];
mask = node->src[3];
break;
case 2:
cache_k_permute = node->src[0]->src[0];
mask = node->src[1];
break;
case 3:
cache_k_permute = node->src[0]->src[0]->src[0];
mask = node->src[1];
break;
case 4:
case 5: {
auto * cache_k_view = node->src[1]; compute_params.token_len_per_seq = node->src[0]->ne[1];
if (attention_pattern_case == 4) {
compute_params.attention_size = cache_k_view->ne[1];
} else {
compute_params.attention_size_static = cache_k_view->ne[1];
}
continue;
}
default:
break;
}
assert(cache_k_permute != nullptr);
model_params.head_size = cache_k_permute->ne[0];
model_params.n_heads_kv = cache_k_permute->ne[2];
compute_params.input_len = node->src[0]->ne[1];
compute_params.token_len_per_seq = node->src[0]->ne[1];
auto * cache_k_view = cache_k_permute->src[0];
if (cache_k_view->op != GGML_OP_VIEW || mask == nullptr) {
continue;
}
ggml_tensor * cache_k = cache_k_view->src[0];
int layer = extract_layer_from_name(cache_k->name).value();
const bool layer_is_swa = std::find(model_params.swa_layers.begin(), model_params.swa_layers.end(),
layer) != model_params.swa_layers.end();
model_params.kv_buffer_ctx_id = ggml_backend_openvino_buffer_get_ctx_id(cache_k->buffer);
model_params.n_heads_kv_per_layer[layer] = cache_k_permute->ne[2];
if (layer_is_swa) {
model_params.ctx_per_seq_swa = cache_k->ne[1];
} else {
model_params.ctx_per_seq = cache_k->ne[1];
model_params.n_seq = cache_k->ne[2];
}
compute_params.n_seq_active = mask->ne[3];
auto seq_size = cache_k->ne[0] * cache_k->ne[1] * ggml_type_size(cache_k->type);
size_t offset;
memcpy(&offset, cache_k_view->op_params, sizeof(size_t));
compute_params.seq_active_start = offset / seq_size;
if (layer_is_swa) {
compute_params.attention_size_swa = mask->ne[0];
compute_params.swa_window = get_swa_window_from_mask(mask);
} else {
compute_params.attention_size = mask->ne[0];
}
if (is_static) {
compute_params.attention_size = model_params.ctx_per_seq;
compute_params.attention_size_swa = model_params.ctx_per_seq_swa;
compute_params.token_len_per_seq = 1;
}
}
if (node->op == GGML_OP_MUL_MAT && node->src[0]->op == GGML_OP_PERMUTE &&
node->src[0]->src[0]->op == GGML_OP_VIEW && is_kvcache(node->src[0]->view_src, node->view_src)) {
if (node->src[1]->op == GGML_OP_PERMUTE && node->src[1]->src[0]->op == GGML_OP_VIEW &&
node->src[1]->src[0]->src[0]->op == GGML_OP_ROPE) {
compute_params.attention_size = node->ne[0];
}
}
if (node->op == GGML_OP_TRANSPOSE && node->src[0]->op == GGML_OP_PERMUTE &&
node->src[0]->src[0]->op == GGML_OP_VIEW) {
compute_params.attention_size = node->ne[0];
if (is_static) {
compute_params.attention_size = model_params.ctx_per_seq;
}
}
if (node->op == GGML_OP_ROPE) {
if (compute_params.token_len_per_seq == -1 && node->src[1] != nullptr) {
compute_params.token_len_per_seq = ggml_nelements(node->src[1]);
}
static_assert(sizeof(model_params.rope_params) == sizeof(int32_t) * 16, "rope_params size");
if (!rope_seen) {
memcpy(model_params.rope_params, node->op_params, sizeof(int32_t) * 16);
rope_seen = true;
} else if (memcmp(model_params.rope_params, node->op_params, sizeof(int32_t) * 16) != 0) {
model_params.mixed_rope_params = true;
}
}
if (node->op == GGML_OP_GATED_DELTA_NET) {
model_params.state_size = node->src[0]->ne[0];
}
if (node->op == GGML_OP_SCALE && node->view_src != nullptr && is_kvcache(node->view_src, nullptr)) {
compute_params.cache_rs_reset_len = ggml_nelements(node) / node->view_src->ne[0];
compute_params.cache_rs_reset_idx = node->src[0]->view_offs / node->view_src->ne[0];
}
if (node->op == GGML_OP_CPY && node->view_src != nullptr && is_kvcache(node->view_src, nullptr) &&
node->src[1] != nullptr && node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src == node->view_src) {
const bool is_conv = is_conv_state_writeback(node);
const bool is_gdn = node->src[0]->op == GGML_OP_VIEW && node->src[0]->src[0] != nullptr &&
node->src[0]->src[0]->op == GGML_OP_GATED_DELTA_NET;
const bool is_extra = node->src[0]->op == GGML_OP_GET_ROWS;
const ggml_tensor * dest_view = node->src[1];
const ggml_tensor * cache = node->view_src;
const size_t row_bytes = cache->ne[0] * ggml_type_size(cache->type);
if (row_bytes > 0 && (is_conv || is_gdn || is_extra)) {
ComputeParams::RsWriteback writeback;
writeback.slot_begin = (int) (dest_view->view_offs / row_bytes);
if (is_conv) {
writeback.src_begin = (int) (node->src[0]->view_offs / node->src[0]->view_src->nb[0]);
} else if (is_gdn) {
writeback.src_begin = (int) (node->src[0]->view_offs / node->src[0]->view_src->nb[1]);
}
compute_params.rs_writebacks[get_tensor_ov_name(cgraph, node)] = writeback;
}
if (is_conv || is_gdn) {
compute_params.s_copy_active_slot_len = (int) dest_view->ne[1];
}
}
}
if (model_params.n_heads_kv == -1) {
for (int i = 0; i < cgraph->n_nodes; i++) {
const auto * node = cgraph->nodes[i];
const ggml_tensor * mask = nullptr;
if (node->op == GGML_OP_SOFT_MAX) {
mask = node->src[1];
} else if (node->op == GGML_OP_FLASH_ATTN_EXT) {
mask = node->src[3];
} else {
continue;
}
if (mask == nullptr || mask->op != GGML_OP_NONE || !(mask->flags & GGML_TENSOR_FLAG_INPUT) ||
node->src[0] == nullptr) {
continue;
}
model_params.is_cacheless_attn = true;
model_params.n_seq = 1;
model_params.ctx_per_seq = mask->ne[0];
compute_params.input_len = node->src[0]->ne[1];
compute_params.token_len_per_seq = compute_params.input_len;
break;
}
}
auto * output_tensor = cgraph->nodes[cgraph->n_nodes - 1];
compute_params.output_len = output_tensor->ne[1];
if (model_params.is_cacheless_attn) {
for (int i = 0; i < cgraph->n_nodes; i++) {
const auto * node = cgraph->nodes[i];
if (node->op == GGML_OP_GET_ROWS && is_output_idx(node->src[1], node)) {
compute_params.output_len = node->src[1]->ne[0];
break;
}
}
}
if (is_static && compute_params.output_len == 0) {
compute_params.output_len = 1;
}
model_params.ctx = model_params.ctx_per_seq * model_params.n_seq;
return {model_params, compute_params};
}
void GgmlOvDecoder::validate_cgraph() const {
if (m_model_params.n_seq > 1 && m_is_static == true) {
throw std::runtime_error("n_seq > 1 is not supported on NPU. Try setting -np 1.");
}
}
ov::PartialShape GgmlOvDecoder::get_graph_input_shape(const ggml_tensor * op,
const ggml_tensor * input,
int dynamic_dim_index) const {
if (m_naive) {
return input != nullptr ? ov::PartialShape{get_shape(input)} : ov::PartialShape{get_shape(op)};
}
ov::PartialShape input_shape;
if (is_inp_tok(input, op) || is_inp_pos(input, op)) {
int len = m_is_static ? (m_is_prefill ? m_prefill_chunk_size : 1) : -1;
if (m_is_static && is_inp_pos(input, op)) {
len *= get_inp_pos_n_planes(op);
}
input_shape = ov::PartialShape{1, 1, 1, len};
} else if (is_output_idx(input, op)) {
input_shape = ov::PartialShape{1, 1, 1, m_is_static ? m_compute_params.output_len : -1};
} else if (is_inp_mean(input, op)) {
input_shape = m_is_static ? ov::PartialShape{1, 1, input->ne[1], m_prefill_chunk_size} :
ov::PartialShape{1, 1, -1, -1};
} else if (is_inp_mask(input, op)) {
if (m_is_static) {
input_shape = ov::PartialShape{1, 1, m_is_prefill ? m_prefill_chunk_size : 1, m_model_params.ctx};
} else if (m_is_stateful) {
input_shape = ov::PartialShape{1, 1, -1, -1};
} else {
input_shape = ov::PartialShape{-1, 1, -1, -1};
}
} else if (is_kvcache(input, op)) {
input_shape = ov::PartialShape{get_shape(input)};
const bool is_flat_kv = (input->ne[1] == 1 && input->ne[2] == 1 && input->ne[3] == 1);
if (!m_is_static && !is_flat_kv) {
input_shape[2] = -1;
}
if (is_stateful() && !is_flat_kv) {
assert(input_shape.size() == 4 && input_shape[0] == 1 && input_shape[1] == 1 &&
input_shape[2].is_dynamic() && input_shape[3].is_static());
const int n_heads_kv = get_n_heads_kv_for_tensor(input);
assert(n_heads_kv > 0 && input_shape[3].get_length() % n_heads_kv == 0);
const int64_t combined_dim = input_shape[3].get_length(); const int64_t head_size = combined_dim / n_heads_kv;
input_shape = {input_shape[0], ov::Dimension::dynamic(), n_heads_kv, head_size};
}
} else if (is_kv_idx(input, op)) {
int len = m_is_static ? (m_is_prefill ? m_prefill_chunk_size : 1) : -1;
input_shape = ov::PartialShape{1, 1, 1, len};
} else if (is_inp_s_copy(input, op) || is_s_copy_leaf(input)) {
input_shape = m_is_static ? ov::PartialShape{get_shape(input)} : ov::PartialShape{1, 1, 1, -1};
} else {
input_shape = ov::PartialShape{get_shape(input)};
}
if (dynamic_dim_index != -1 && m_model_is_splitted) {
input_shape[3 - dynamic_dim_index] = -1;
}
if (op->op == GGML_OP_SOFT_MAX && op->src[1] != nullptr && op->src[1]->op == GGML_OP_NONE &&
op->src[1]->flags & GGML_TENSOR_FLAG_INPUT && op->src[1] == input) {
if (m_is_static) {
const int64_t seq_active = m_is_prefill ? m_prefill_chunk_size : 1;
input_shape[2] = seq_active;
input_shape[3] = seq_active;
} else {
input_shape[2] = -1;
input_shape[3] = -1;
}
}
return input_shape;
}
bool GgmlOvDecoder::is_s_copy_leaf(const ggml_tensor * tensor) const {
if (tensor == nullptr || tensor->op != GGML_OP_NONE || m_cgraph == nullptr) {
return false;
}
for (int i = 0; i < m_cgraph->n_nodes; i++) {
const ggml_tensor * node = m_cgraph->nodes[i];
if (node->op != GGML_OP_GET_ROWS || node->src[0] == nullptr || node->src[1] == nullptr) {
continue;
}
const ggml_tensor * idx = node->src[1];
while (idx != nullptr && idx->op == GGML_OP_VIEW) {
idx = idx->src[0];
}
if (idx != tensor) {
continue;
}
const ggml_tensor * data = node->src[0];
while (data != nullptr && (data->op == GGML_OP_VIEW || data->op == GGML_OP_RESHAPE)) {
data = data->src[0];
}
if (data != nullptr && is_kvcache(data, nullptr)) {
return true;
}
}
return false;
}
void GgmlOvDecoder::add_extra_inputs() {
auto create_1d_input = [this](const std::string & name, int64_t value, bool force_parameter = false) {
m_model_extra_inputs[name] = {ov::element::i64, ov::Shape{1}, value, force_parameter || !m_is_static};
};
if (m_compute_params.attention_size != -1) {
create_1d_input("attention_size", m_compute_params.attention_size);
}
if (m_compute_params.attention_size_static != -1) {
create_1d_input("attention_size_static", m_compute_params.attention_size_static);
}
if (m_compute_params.attention_size_swa != -1) {
create_1d_input("attention_size_swa", m_compute_params.attention_size_swa);
}
if (is_stateful() && m_compute_params.swa_window != -1) {
create_1d_input("swa_window", m_compute_params.swa_window);
}
create_1d_input("n_seq_active", m_compute_params.n_seq_active);
create_1d_input("seq_active_start", m_compute_params.seq_active_start);
create_1d_input("seq_active_end", m_compute_params.seq_active_start + m_compute_params.n_seq_active);
if (m_compute_params.token_len_per_seq != -1) {
create_1d_input("token_len_per_seq", m_compute_params.token_len_per_seq);
}
if (m_compute_params.cache_rs_reset_idx != -1) {
create_1d_input("cache_rs_reset_idx", m_compute_params.cache_rs_reset_idx, true);
create_1d_input("cache_rs_reset_len", m_compute_params.cache_rs_reset_len, true);
}
if (m_compute_params.s_copy_active_slot_len != -1) {
create_1d_input("s_copy_active_slot_len", m_compute_params.s_copy_active_slot_len);
if (m_is_static) {
create_1d_input("chunk_valid_len", get_static_n_tokens(), true);
}
}
for (const auto & [node_name, writeback] : m_compute_params.rs_writebacks) {
create_1d_input("rs_slot_begin_" + node_name, writeback.slot_begin);
if (!m_is_static) {
create_1d_input("rs_src_begin_" + node_name, writeback.src_begin);
}
}
}
bool GgmlOvDecoder::node_is_used_as_src(const int node_idx) {
ggml_tensor * node = m_cgraph->nodes[node_idx];
for (int i = node_idx; i < m_cgraph->n_nodes; i++) {
ggml_tensor * other_node = m_cgraph->nodes[i];
for (int j = 0; j < GGML_MAX_SRC; j++) {
if (other_node->src[j] == node) {
return true;
}
}
}
return false;
}
void GgmlOvDecoder::compute_model_inputs() {
m_model_inputs.clear();
m_inputs.clear();
for (int i = 0; i < m_cgraph->n_nodes; i++) {
ggml_tensor * node = m_cgraph->nodes[i];
if (node->op == GGML_OP_NONE && node_is_used_as_src(i)) {
std::string node_name = get_tensor_ov_name(m_cgraph, node);
if (m_model_weights.find(node_name) == m_model_weights.end()) {
m_inputs[node_name] = node;
m_model_inputs[node_name] = {get_ov_type(node),
get_graph_input_shape(node, nullptr, m_node_dynamic_dims[node])};
}
continue;
}
for (int i = 0; i < GGML_MAX_SRC; i++) {
auto * src = node->src[i];
if (src == nullptr) {
continue;
}
std::string src_name = get_tensor_ov_name(m_cgraph, src);
if (src->flags & GGML_TENSOR_FLAG_INPUT) {
src_name = get_tensor_graph_input_ov_name(this, m_cgraph, src, node);
}
if (m_model_weights.find(src_name) != m_model_weights.end()) {
continue;
}
bool is_intermediate_node = false;
for (const auto & node_info : m_node_info_list) {
if (node_info.node == src) {
is_intermediate_node = true;
break;
}
}
if (is_intermediate_node) {
continue;
}
if (m_model_inputs.find(src_name) != m_model_inputs.end()) {
continue;
}
m_inputs[src_name] = src;
ggml_backend_buffer * buffer = src->buffer;
if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_ANY) {
if (auto it = std::find(m_model_params.kv_names.begin(), m_model_params.kv_names.end(), src_name);
it == m_model_params.kv_names.end()) {
m_model_params.kv_names.push_back(src_name);
}
}
while (src->op == GGML_OP_VIEW && src->src[0] != nullptr) {
src = src->src[0];
src_name = get_tensor_ov_name(m_cgraph, src);
}
m_inputs[src_name] = src;
m_model_inputs[src_name] = {get_ov_type(src),
get_graph_input_shape(node, src, m_node_dynamic_dims[src])};
}
}
}
void GgmlOvDecoder::compute_model_outputs() {
m_model_outputs.clear();
m_model_output_names.clear();
for (int node_n = 0; node_n < m_cgraph->n_nodes; node_n++) {
auto * cur_node = m_cgraph->nodes[node_n];
if (cur_node->op == GGML_OP_NONE || cur_node->op == GGML_OP_VIEW || cur_node->op == GGML_OP_RESHAPE) {
continue;
}
auto cur_node_use_count = m_cgraph->use_counts[ggml_hash_find(&m_cgraph->visited_hash_set, cur_node)];
if (cur_node_use_count == 0) {
if (cur_node != nullptr && ::is_inplace_op(cur_node) && ggml_nbytes(cur_node) > 0) {
cur_node = cur_node->view_src;
}
} else {
int input_use_count = 0;
for (int i = 0; i < m_cgraph->n_nodes; i++) {
ggml_tensor * node = m_cgraph->nodes[i];
for (int j = 0; j < GGML_MAX_SRC; j++) {
if (node->src[j] != NULL && node->src[j] == cur_node) {
input_use_count++;
}
}
}
if (input_use_count == cur_node_use_count) {
cur_node = nullptr;
}
}
if (cur_node != nullptr) {
std::string cur_node_name = get_tensor_ov_name(m_cgraph, cur_node);
m_model_outputs[cur_node_name] = cur_node;
m_model_output_names.insert(cur_node_name);
}
}
}
const ggml_tensor * GgmlOvDecoder::get_tensor_used_op(const ggml_tensor * tensor) const {
if (tensor == nullptr) {
return nullptr;
}
for (int i = 0; i < m_cgraph->n_nodes; i++) {
const auto * node = m_cgraph->nodes[i];
for (int j = 0; j < GGML_MAX_SRC; j++) {
if (node->src[j] == tensor) {
return node;
}
}
}
return nullptr;
}
const ggml_tensor * GgmlOvDecoder::get_tensor_from_name(const std::string & name) const {
for (int i = 0; i < m_cgraph->n_nodes; i++) {
const auto * node = m_cgraph->nodes[i];
for (int j = 0; j < GGML_MAX_SRC; j++) {
const auto * src = node->src[j];
if (src == nullptr) {
break;
}
if (get_tensor_ov_name(m_cgraph, src) == name) {
return src;
}
}
}
return nullptr;
}
std::map<std::string, std::string> GgmlOvDecoder::get_kv_param_res_names() const {
std::map<std::string, std::string> kv_param_res_names;
for (const auto & name : m_model_params.kv_names) {
kv_param_res_names[name] = name;
}
return kv_param_res_names;
}
static bool is_mul_mat_id_expert_weight(const ggml_tensor * node, int src_index) {
return node->op == GGML_OP_MUL_MAT_ID && src_index == 0;
}
std::map<std::string, std::shared_ptr<ov::Node>> GgmlOvDecoder::create_weight_nodes(ggml_cgraph * cgraph, bool naive) {
std::map<std::string, std::shared_ptr<ov::Node>> model_weights;
auto * nodes = cgraph->nodes;
auto n_nodes = cgraph->n_nodes;
for (int node_i = 0; node_i < n_nodes; node_i++) {
auto * node = nodes[node_i];
for (int i = 0; i < GGML_MAX_SRC; i++) {
auto * src = node->src[i];
if (src == nullptr) {
continue;
}
std::string src_name = get_tensor_ov_name(cgraph, src);
if (is_rope_freqs_weight(src, node)) {
src_name = "rope_freqs.weight";
}
if (!src->view_src) {
ggml_backend_buffer * buffer = src->buffer;
if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(src->type) ||
is_mul_mat_id_expert_weight(node, i)) {
if (model_weights.find(src_name) == model_weights.end()) {
auto weight_node = create_weight_node(src, naive);
weight_node->set_friendly_name(src_name);
model_weights[src_name] = weight_node;
}
}
}
}
}
return model_weights;
}
static std::mutex g_nonov_weight_cache_mutex;
static std::unordered_map<const void *, std::shared_ptr<ov::Node>> g_nonov_weight_cache;
std::set<std::string> GgmlOvDecoder::collect_weight_names(ggml_cgraph * cgraph) {
std::set<std::string> names;
for (int node_i = 0; node_i < cgraph->n_nodes; node_i++) {
auto * node = cgraph->nodes[node_i];
for (int i = 0; i < GGML_MAX_SRC; i++) {
auto * src = node->src[i];
if (src == nullptr) {
continue;
}
std::string src_name(src->name);
if (is_rope_freqs_weight(src, node)) {
src_name = "rope_freqs.weight";
}
if (!src->view_src) {
ggml_backend_buffer * buffer = src->buffer;
if (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(src->type)) {
names.insert(src_name);
}
}
}
}
return names;
}
std::shared_ptr<ov::Node> GgmlOvDecoder::create_weight_node(ggml_tensor * tensor, bool naive) {
const bool is_ov_buffer = ggml_backend_buffer_is_openvino(tensor->buffer);
if (tensor->extra) {
OPENVINO_ASSERT(is_ov_buffer, "Unsupported weight tensor: " + std::string(tensor->name) +
" Possibly this is a cpu backend repacked quantized weights");
auto * extra_base = static_cast<ggml_openvino_extra_base *>(tensor->extra);
if (extra_base->type == ggml_openvino_extra_base::Type::WEIGHT) {
auto * weight_extra = static_cast<ggml_openvino_weight_extra *>(tensor->extra);
if (weight_extra->weight_node) {
return weight_extra->weight_node;
}
} else if (extra_base->type == ggml_openvino_extra_base::Type::QUANTIZED_WEIGHT) {
auto * quant_extra = static_cast<ggml_openvino_quantized_weight_extra *>(tensor->extra);
if (quant_extra->weight_node) {
return quant_extra->weight_node;
}
}
}
if (!ggml_is_quantized(tensor->type) && (tensor->ne[2] > 1 || tensor->ne[3] > 1)) {
auto weight_tensor = ov::Tensor(get_ov_type(tensor), get_shape(tensor), tensor->data);
auto weight_node = std::make_shared<ov::op::v0::Constant>(weight_tensor);
weight_node->set_friendly_name(tensor->name);
return weight_node;
}
const bool cacheable_nonov = ggml_openvino_reduce_compile_mem_enabled() && !is_ov_buffer &&
!naive && tensor->data != nullptr;
if (cacheable_nonov) {
std::lock_guard<std::mutex> lock(g_nonov_weight_cache_mutex);
auto it = g_nonov_weight_cache.find(tensor->data);
if (it != g_nonov_weight_cache.end()) {
return it->second;
}
}
static const std::set<ggml_type> weight_types = {GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_BF16, GGML_TYPE_Q8_0,
GGML_TYPE_Q4_0, GGML_TYPE_Q4_1, GGML_TYPE_Q5_1, GGML_TYPE_Q4_K,
GGML_TYPE_Q5_K, GGML_TYPE_Q6_K, GGML_TYPE_MXFP4};
if (weight_types.find(tensor->type) == weight_types.end()) {
throw std::runtime_error("Unexpected weight tensor type: " + std::string(tensor->name) + " with type " +
ggml_type_name(tensor->type));
}
OvWeight ov_weight;
if (ggml_is_quantized(tensor->type)) {
auto use_bias = naive;
if (is_ov_buffer) {
size_t raw_size = ggml_nbytes(tensor);
std::vector<uint8_t> tmp(raw_size);
memcpy(tmp.data(), tensor->data, raw_size);
ov_weight = process_weight_tensor(tensor, tmp.data(), tensor->data, use_bias);
} else {
ov_weight = process_weight_tensor(tensor, tensor->data, nullptr, use_bias);
}
} else {
ov_weight = process_weight_tensor(tensor, tensor->data, tensor->data);
}
ov_weight.weight_node->set_friendly_name(tensor->name);
if (!is_ov_buffer) {
if (cacheable_nonov) {
std::lock_guard<std::mutex> lock(g_nonov_weight_cache_mutex);
auto [it, inserted] = g_nonov_weight_cache.emplace(tensor->data, ov_weight.weight_node);
return it->second;
}
return ov_weight.weight_node;
}
ggml_openvino_extra_base * extra;
if (ov_weight.is_quantized()) {
extra = new ggml_openvino_quantized_weight_extra(std::move(ov_weight.weights), std::move(ov_weight.scales),
std::move(ov_weight.zp), ov_weight.weight_node);
} else {
extra = new ggml_openvino_weight_extra(std::move(ov_weight.weights), ov_weight.weight_node);
}
ggml_openvino_buffer_register_extra(tensor, extra);
return ov_weight.weight_node;
}
void GgmlOvDecoder::dump_cgraph(const ggml_cgraph * cgraph, std::string & filename) {
std::ofstream file(filename);
if (!file.is_open()) {
std::cerr << "Failed to open file" << '\n';
return;
}
file << "=== GRAPH ===\n";
file << "n_nodes = " << cgraph->n_nodes << "\n";
file << " " << std::setw(3) << "nodes"
<< std::setw(15) << "shape"
<< std::setw(20) << "op"
<< std::setw(20) << "name"
<< std::setw(3) << " "
<< std::setw(62) << "stride"
<< std::setw(20) << "buffer_type"
<< "\n";
for (int i = 0; i < cgraph->n_nodes; i++) {
ggml_tensor * node = cgraph->nodes[i];
const char * buf_name = "none";
ggml_backend_buffer_t buf = node->view_src ? node->view_src->buffer : node->buffer;
if (buf) {
buf_name = ggml_backend_buffer_name(buf);
}
file << " - " << std::setw(3) << i << ": [ "
<< std::setw(5) << node->ne[0] << ", "
<< std::setw(5) << node->ne[1] << ", "
<< std::setw(5) << node->ne[2] << ", "
<< std::setw(5) << node->ne[3] << "] "
<< std::left << std::setw(20) << ggml_op_name(node->op) << std::right << " "
<< std::left << std::setw(45) << node->name << std::right
<< std::setw(2) << "[ "
<< std::setw(0) << node->nb[0] << ", "
<< std::setw(5) << node->nb[1] << ", "
<< std::setw(5) << node->nb[2] << ", "
<< std::setw(5) << node->nb[3] << "] "
<< std::right << std::setw(15) << buf_name << std::right
<< "\n";
for (int i = 0; i < GGML_MAX_SRC; i++) {
if (auto* src = node->src[i]) {
const char * src_buf_name = "none";
ggml_backend_buffer_t src_buf = src->view_src ? src->view_src->buffer : src->buffer;
if (src_buf) {
src_buf_name = ggml_backend_buffer_name(src_buf);
}
file << std::setw(10) << " [ "
<< std::setw(5) << src->ne[0] << ", "
<< std::setw(5) << src->ne[1] << ", "
<< std::setw(5) << src->ne[2] << ", "
<< std::setw(5) << src->ne[3] << "] "
<< std::setw(12)
<< i << ": " << std::left << std::setw(12) << ggml_op_name(src->op) << std::right;
file << std::left << std::setw(30) << src->name << std::right
<< std::setw(16) << "[ "
<< std::setw(0) << src->nb[0] << ", "
<< std::setw(5) << src->nb[1] << ", "
<< std::setw(5) << src->nb[2] << ", "
<< std::setw(5) << src->nb[3] << "] "
<< std::right << std::setw(15) << src_buf_name << std::right
<< "\n";
}
}
}
file << "n_leafs = " << cgraph->n_leafs << "\n";
for (int i = 0; i < cgraph->n_leafs; i++) {
ggml_tensor * node = cgraph->leafs[i];
const char * leaf_buf_name = "none";
ggml_backend_buffer_t leaf_buf = node->view_src ? node->view_src->buffer : node->buffer;
if (leaf_buf) {
leaf_buf_name = ggml_backend_buffer_name(leaf_buf);
}
file << " - " << std::setw(3) << i << ": [ "
<< std::setw(5) << node->ne[0] << ", "
<< std::setw(5) << node->ne[1] << "] "
<< std::setw(8) << ggml_op_name(node->op) << " "
<< std::setw(16) << ggml_get_name(node)
<< std::setw(20) << leaf_buf_name << "\n";
}
file << "========================================\n";
file.close();
}
void print_tensor_address_map(const ggml_cgraph * cgraph) {
std::map<void *, std::vector<std::string>> address_map;
for (int node_n = 0; node_n < cgraph->n_nodes; node_n++) {
auto * node = cgraph->nodes[node_n];
if (node->data) {
auto it = address_map.find(node->data);
if (it == address_map.end()) {
address_map[node->data] = std::vector<std::string>();
}
address_map[node->data].push_back(node->name);
}
}
for (const auto & pair : address_map) {
std::cout << "Address: " << pair.first << '\n';
for (const auto & name : pair.second) {
std::cout << name << " ; ";
}
std::cout << "\n\n";
}
}
ov::Shape GgmlOvDecoder::get_shape(const ggml_tensor * tensor) {
std::vector<size_t> shape;
for (int i = GGML_MAX_DIMS - 1; i >= 0; --i) {
shape.push_back(static_cast<size_t>(tensor->ne[i]));
}
return shape;
}
std::vector<size_t> GgmlOvDecoder::get_stride(const ggml_tensor * tensor) {
std::vector<size_t> stride;
for (int i = GGML_MAX_DIMS - 1; i >= 0; --i) {
stride.push_back(static_cast<size_t>(tensor->nb[i]));
}
return stride;
}
ov::element::Type GgmlOvDecoder::get_ov_type(const ggml_tensor * tensor) {
switch (tensor->type) {
case GGML_TYPE_F64:
return ov::element::f64;
case GGML_TYPE_F32:
return ov::element::f32;
case GGML_TYPE_F16:
return ov::element::f16;
case GGML_TYPE_BF16:
return ov::element::bf16;
case GGML_TYPE_I8:
return ov::element::i8;
case GGML_TYPE_I16:
return ov::element::i16;
case GGML_TYPE_I32:
return ov::element::i32;
case GGML_TYPE_I64:
return ov::element::i64;
default:
return ov::element::dynamic;
}
}
ov::PartialShape GgmlOvDecoder::get_input_shape(int node_idx, const std::string & name) const {
return ov::PartialShape(get_shape(m_node_info_list[node_idx].node_inputs.at(name)));
}
std::vector<size_t> GgmlOvDecoder::get_input_stride(int node_idx, const std::string & name) const {
return get_stride(m_node_info_list[node_idx].node_inputs.at(name));
}
size_t GgmlOvDecoder::get_view_input_size(int node_idx, const std::string & name) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
return it->second.size();
}
return 0;
}
size_t GgmlOvDecoder::get_view_input_offset(int node_idx, const std::string & name, size_t view_index) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
if (view_index < it->second.size()) {
return it->second[view_index].second->view_offs;
}
}
return 0;
}
size_t GgmlOvDecoder::get_view_input_src_offset(int node_idx, const std::string & name, size_t view_index) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
if (view_index < it->second.size()) {
auto * view_tensor = it->second[view_index].second;
if (view_tensor && view_tensor->src[0]) {
return view_tensor->src[0]->view_offs;
}
}
}
return 0;
}
std::vector<size_t> GgmlOvDecoder::get_view_input_stride(int node_idx,
const std::string & name,
size_t view_index) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
if (view_index < it->second.size()) {
return get_stride(it->second[view_index].second);
}
}
return {};
}
std::vector<size_t> GgmlOvDecoder::get_view_input_src_stride(int node_idx,
const std::string & name,
size_t view_index) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
if (view_index < it->second.size()) {
auto * view_tensor = it->second[view_index].second;
if (view_tensor && view_tensor->src[0]) {
return get_stride(view_tensor->src[0]);
}
}
}
return {};
}
ov::Shape GgmlOvDecoder::get_view_input_ggml_shape(int node_idx, const std::string & name, size_t view_index) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
if (view_index < it->second.size()) {
return get_shape(it->second[view_index].second);
}
}
return {};
}
ov::Shape GgmlOvDecoder::get_view_input_src_ggml_shape(int node_idx,
const std::string & name,
size_t view_index) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
if (view_index < it->second.size()) {
auto * view_tensor = it->second[view_index].second;
if (view_tensor && view_tensor->src[0]) {
return get_shape(view_tensor->src[0]);
}
}
}
return {};
}
ov::PartialShape GgmlOvDecoder::get_view_input_ov_shape(int node_idx,
const std::string & name,
size_t view_index) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
if (view_index < it->second.size()) {
auto * tensor = it->second[view_index].second;
ov::PartialShape shape = ov::PartialShape{get_shape(tensor)};
auto dynamic_it = m_node_dynamic_dims.find(tensor);
if (dynamic_it != m_node_dynamic_dims.end() && dynamic_it->second != -1) {
int dynamic_dim_index = dynamic_it->second;
shape[3 - dynamic_dim_index] = m_is_static ? get_static_n_tokens() : -1;
}
return shape;
}
}
return {};
}
ov::PartialShape GgmlOvDecoder::get_view_input_src_ov_shape(int node_idx,
const std::string & name,
size_t view_index) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
if (view_index < it->second.size()) {
auto * view_tensor = it->second[view_index].second;
if (view_tensor && view_tensor->src[0]) {
auto * src_tensor = view_tensor->src[0];
ov::PartialShape shape = ov::PartialShape{get_shape(src_tensor)};
auto dynamic_it = m_node_dynamic_dims.find(src_tensor);
if (dynamic_it != m_node_dynamic_dims.end() && dynamic_it->second != -1) {
int dynamic_dim_index = dynamic_it->second;
shape[3 - dynamic_dim_index] = m_is_static ? get_static_n_tokens() : -1;
}
return shape;
}
}
}
return {};
}
std::string GgmlOvDecoder::get_view_input_name(int node_idx, const std::string & name, size_t view_index) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
if (view_index < it->second.size()) {
return it->second[view_index].first;
}
}
return "";
}
std::string GgmlOvDecoder::get_view_input_src_name(int node_idx, const std::string & name, size_t view_index) const {
auto it = m_node_info_list[node_idx].node_inputs_views.find(name);
if (it != m_node_info_list[node_idx].node_inputs_views.end()) {
if (view_index < it->second.size()) {
auto * view_tensor = it->second[view_index].second;
if (view_tensor && view_tensor->src[0]) {
return get_tensor_ov_name(m_cgraph, view_tensor->src[0]);
}
}
}
return "";
}
ov::element::Type GgmlOvDecoder::get_input_type(int node_idx, const std::string & name) const {
return get_ov_type(m_node_info_list[node_idx].node_inputs.at(name));
}
size_t GgmlOvDecoder::get_input_size() const {
return m_model_inputs.size();
}
size_t GgmlOvDecoder::get_input_size(int node_idx) const {
return m_node_info_list[node_idx].node_inputs_names.size();
}
std::vector<std::string> GgmlOvDecoder::get_input_names(int node_idx) const {
return m_node_info_list[node_idx].node_inputs_names;
}
ov::PartialShape GgmlOvDecoder::get_output_shape(int node_idx) const {
auto * ggml_tensor = m_node_info_list[node_idx].node;
return ov::PartialShape(get_shape(ggml_tensor));
}
ov::element::Type GgmlOvDecoder::get_output_type(const int node_idx) const {
return get_ov_type(m_node_info_list[node_idx].node);
}
std::vector<size_t> GgmlOvDecoder::get_output_stride(int node_idx) const {
auto * ggml_tensor = m_node_info_list[node_idx].node;
return get_stride(ggml_tensor);
}
std::vector<std::string> GgmlOvDecoder::get_output_names(int node_idx) const {
return {m_node_info_list[node_idx].node_name};
}
std::string GgmlOvDecoder::get_inplace_op_src(int node_idx) const {
auto * node = m_node_info_list[node_idx].node;
if (!::is_inplace_op(node) || node->view_src == nullptr || ggml_nbytes(node) == 0) {
return "";
}
const int op_case = m_node_info_list[node_idx].node_op_case;
if (node->op == GGML_OP_CPY && (op_case == 1 || op_case == 2 || op_case == 3) &&
m_compute_params.s_copy_active_slot_len == -1) {
return "";
}
return get_tensor_ov_name(m_cgraph, node->view_src);
}
bool GgmlOvDecoder::is_view_like_alias_of(int node_idx, const std::string & view_src_name) const {
auto * node = m_node_info_list[node_idx].node;
if (node->view_src == nullptr || get_tensor_ov_name(m_cgraph, node->view_src) != view_src_name) {
return false;
}
return node->op == GGML_OP_RESHAPE || node->op == GGML_OP_VIEW;
}
const std::string & GgmlOvDecoder::get_op_name() const {
static const std::string unknown_name = "UNKNOWN_OP_NAME";
return unknown_name;
}
int32_t GgmlOvDecoder::get_op_dynamic_dim(int node_idx) const {
auto it = m_node_dynamic_dims.find(m_node_info_list[node_idx].node);
if (it == m_node_dynamic_dims.end()) {
return -1;
}
return it->second;
}
const std::string & GgmlOvDecoder::get_op_name(int node_idx) const {
return m_node_info_list[node_idx].node_name;
}
int32_t * GgmlOvDecoder::get_input_op_params(int node_idx, const std::string & name) const {
return m_node_info_list[node_idx].node_inputs.at(name)->op_params;
}
int32_t * GgmlOvDecoder::get_output_op_params(int node_idx) const {
return m_node_info_list[node_idx].node->op_params;
}
size_t GgmlOvDecoder::get_output_op_offset(int node_idx) const {
return m_node_info_list[node_idx].node->view_offs;
}
void GgmlOvDecoder::visit_subgraph(std::function<void(std::shared_ptr<GgmlDecoder>, int node_idx)> node_visitor) const {
for (int node_idx = 0; node_idx < m_cgraph->n_nodes; node_idx++) {
if (m_cgraph->nodes[node_idx]->op == GGML_OP_NONE) {
continue;
}
node_visitor(std::make_shared<GgmlOvDecoder>(*this), node_idx);
}
}
std::string GgmlOvDecoder::compute_op_type(const ggml_tensor * node) {
switch (node->op) {
case GGML_OP_UNARY:
return std::string("GGML_UNARY_OP_") + ggml_unary_op_name(ggml_get_unary_op(node));
case GGML_OP_GLU:
return std::string("GGML_GLU_OP_") + ggml_glu_op_name(ggml_get_glu_op(node));
default:
return std::string("GGML_OP_") + ggml_op_name(node->op);
}
}
const std::string & GgmlOvDecoder::get_op_type(int node_idx) const {
return m_node_info_list[node_idx].node_op_type;
}
const std::string & GgmlOvDecoder::get_op_type() const {
static const std::string unknown_op = "UNKNOWN_GGML_OP";
return unknown_op;
}
void GgmlOvDecoder::compute_node_dynamic_dims() {
auto visit_node = [&](auto && self, ggml_tensor * node) -> void {
if (!node) {
return;
}
if (node->op == GGML_OP_CPY) {
m_node_dynamic_dims[node] = -1;
}
if (m_node_dynamic_dims.count(node)) {
return;
}
for (int i = 0; i < GGML_MAX_SRC; i++) {
ggml_tensor * src = node->src[i];
if (src == nullptr) {
continue;
}
struct ggml_tensor * root_src = nullptr;
if (root_src) {
if (is_inp_tok(root_src, node) || is_inp_pos(root_src, node) || is_output_idx(root_src, node)) {
m_node_dynamic_dims[root_src] = 0;
m_node_dynamic_dims[src] = m_node_dynamic_dims[root_src];
continue;
}
self(self, root_src);
m_node_dynamic_dims[src] = m_node_dynamic_dims[root_src];
} else {
if (is_inp_tok(src, node) || is_inp_pos(src, node) || is_output_idx(src, node)) {
m_node_dynamic_dims[src] = 0;
continue;
}
if (node->op == GGML_OP_VIEW && src->op == GGML_OP_NONE && !is_stateful() && !m_model_is_splitted) {
m_node_dynamic_dims[src] = 1;
continue;
}
self(self, src);
}
}
switch (node->op) {
case GGML_OP_NONE:
m_node_dynamic_dims[node] = -1;
break;
case GGML_OP_GET_ROWS:
m_node_dynamic_dims[node] = -1;
if (m_node_dynamic_dims[node->src[1]] != -1) {
auto dynamic_dim_idx = m_node_dynamic_dims[node->src[1]];
if (dynamic_dim_idx == 0) {
m_node_dynamic_dims[node] = 1;
} else {
auto dynamic_dim_stride = node->src[1]->nb[dynamic_dim_idx] / ggml_type_size(node->src[1]->type) *
ggml_type_size(node->src[0]->type);
for (int i = 0; i < GGML_MAX_DIMS; i++) {
if (dynamic_dim_stride == node->src[0]->nb[i]) {
m_node_dynamic_dims[node] = i;
break;
}
}
}
}
break;
case GGML_OP_MUL:
case GGML_OP_MUL_MAT:
m_node_dynamic_dims[node] = -1;
if (m_node_dynamic_dims[node->src[0]] != -1) {
m_node_dynamic_dims[node] = m_node_dynamic_dims[node->src[0]];
}
if (m_node_dynamic_dims[node->src[1]] != -1) {
m_node_dynamic_dims[node] = m_node_dynamic_dims[node->src[1]];
}
break;
case GGML_OP_PERMUTE:
m_node_dynamic_dims[node] = -1;
if (m_node_dynamic_dims[node->src[0]] != -1) {
auto dynamic_dim_idx = m_node_dynamic_dims[node->src[0]];
for (int i = 0; i < GGML_MAX_DIMS; i++) {
if (node->op_params[i] == dynamic_dim_idx) {
m_node_dynamic_dims[node] = i;
break;
}
}
}
break;
case GGML_OP_VIEW: {
m_node_dynamic_dims[node] = -1;
if (m_node_dynamic_dims[node->src[0]] != -1) {
if (node->src[0]->op == GGML_OP_NONE) {
m_node_dynamic_dims[node] = m_node_dynamic_dims[node->src[0]];
break;
}
auto dynamic_dim_idx = m_node_dynamic_dims[node->src[0]];
auto dynamic_dim_value = node->src[0]->ne[dynamic_dim_idx];
auto dynamic_dim_stride =
node->src[0]->nb[dynamic_dim_idx] / ggml_type_size(node->src[0]->type) * ggml_type_size(node->type);
for (int i = 0; i < GGML_MAX_DIMS; i++) {
if (node->nb[i] == dynamic_dim_stride) {
m_node_dynamic_dims[node] = i;
break;
}
}
if (m_node_dynamic_dims[node] != -1 && dynamic_dim_value != node->ne[m_node_dynamic_dims[node]]) {
m_node_dynamic_dims[node] = -1;
GGML_LOG_WARN("ggml-openvino: dynamic dim value mismatch for VIEW node '%s', src[0]: '%s'\n",
node->name, node->src[0]->name);
}
}
break;
}
case GGML_OP_TRANSPOSE:
case GGML_OP_RESHAPE: {
if (is_same_shape(node->src[0], node)) {
m_node_dynamic_dims[node] = m_node_dynamic_dims[node->src[0]];
break;
}
m_node_dynamic_dims[node] = -1;
if (m_node_dynamic_dims[node->src[0]] != -1) {
auto dynamic_dim_idx = m_node_dynamic_dims[node->src[0]];
auto dynamic_dim_stride = node->src[0]->nb[dynamic_dim_idx];
for (int i = 0; i < GGML_MAX_DIMS; i++) {
if (node->nb[i] == dynamic_dim_stride && node->ne[i] == node->src[0]->ne[dynamic_dim_idx]) {
m_node_dynamic_dims[node] = i;
break;
}
}
if (m_node_dynamic_dims[node] == -1) {
GGML_LOG_WARN("ggml-openvino: cannot determine dynamic dim for RESHAPE node '%s'\n", node->name);
}
}
break;
}
case GGML_OP_FLASH_ATTN_EXT: {
constexpr int q_to_out[GGML_MAX_DIMS] = {0, 2, 1, 3};
m_node_dynamic_dims[node] = -1;
if (m_node_dynamic_dims[node->src[0]] != -1) {
auto q_dynamic_dim = m_node_dynamic_dims[node->src[0]];
m_node_dynamic_dims[node] = q_to_out[q_dynamic_dim];
}
break;
}
case GGML_OP_CONT:
m_node_dynamic_dims[node] = -1;
if (m_node_dynamic_dims[node->src[0]] != -1) {
auto dynamic_dim_idx = m_node_dynamic_dims[node->src[0]];
if (ggml_are_same_shape(node, node->src[0])) {
m_node_dynamic_dims[node] = dynamic_dim_idx;
} else {
size_t src_logical_nb[GGML_MAX_DIMS];
src_logical_nb[0] = ggml_type_size(node->src[0]->type);
src_logical_nb[1] = src_logical_nb[0] * (node->src[0]->ne[0] / ggml_blck_size(node->src[0]->type));
for (int i = 2; i < GGML_MAX_DIMS; i++) {
src_logical_nb[i] = src_logical_nb[i - 1] * node->src[0]->ne[i - 1];
}
auto dynamic_dim_stride = src_logical_nb[dynamic_dim_idx] / ggml_type_size(node->src[0]->type) *
ggml_type_size(node->type);
int matched_dim_count = 0;
int first_matched_dim = -1;
for (int i = 0; i < GGML_MAX_DIMS; i++) {
if (node->nb[i] == dynamic_dim_stride && node->ne[i] == node->src[0]->ne[dynamic_dim_idx]) {
if (first_matched_dim == -1) {
first_matched_dim = i;
}
m_node_dynamic_dims[node] = i;
matched_dim_count++;
}
}
if (matched_dim_count > 1 && node->src[0]->ne[dynamic_dim_idx] == 1) {
m_node_dynamic_dims[node] = first_matched_dim;
} else if (matched_dim_count != 1) {
m_node_dynamic_dims[node] = -1;
GGML_LOG_WARN("ggml-openvino: cannot determine dynamic dim for CONT node '%s', src[0]: '%s'\n",
node->name, node->src[0]->name);
}
}
}
break;
case GGML_OP_CONCAT:
for (int i = 0; i < GGML_MAX_DIMS; i++) {
if (node->src[0]->ne[i] != node->ne[i]) {
m_node_dynamic_dims[node] = i;
break;
}
}
break;
case GGML_OP_SSM_CONV:
case GGML_OP_GATED_DELTA_NET:
m_node_dynamic_dims[node] = 1;
break;
case GGML_OP_RMS_NORM:
case GGML_OP_L2_NORM:
case GGML_OP_NORM:
case GGML_OP_ADD:
case GGML_OP_SUB:
case GGML_OP_GLU:
case GGML_OP_ROPE:
case GGML_OP_SCALE:
case GGML_OP_SOFT_MAX:
case GGML_OP_ARGSORT:
case GGML_OP_ADD_ID:
case GGML_OP_UNARY:
case GGML_OP_CUMSUM:
case GGML_OP_FILL:
case GGML_OP_SET:
case GGML_OP_DIAG:
case GGML_OP_TRI:
case GGML_OP_REPEAT:
case GGML_OP_DIV:
case GGML_OP_CLAMP:
case GGML_OP_PAD:
m_node_dynamic_dims[node] = m_node_dynamic_dims[node->src[0]];
break;
case GGML_OP_SUM_ROWS:
m_node_dynamic_dims[node] =
(m_node_dynamic_dims[node->src[0]] == 0) ? -1 : m_node_dynamic_dims[node->src[0]];
break;
case GGML_OP_MUL_MAT_ID:
case GGML_OP_SOLVE_TRI:
m_node_dynamic_dims[node] = m_node_dynamic_dims[node->src[1]];
break;
case GGML_OP_CPY:
case GGML_OP_SET_ROWS:
m_node_dynamic_dims[node] = -1;
break;
case GGML_OP_IM2COL: {
m_node_dynamic_dims[node] = -1;
if (m_node_dynamic_dims[node->src[1]] != -1) {
const bool is_2D = node->op_params[6] == 1;
const int src_dyn = m_node_dynamic_dims[node->src[1]];
if (is_2D) {
if (src_dyn == 0) {
m_node_dynamic_dims[node] = 1; } else if (src_dyn == 1) {
m_node_dynamic_dims[node] = 2; } else if (src_dyn == 3) {
m_node_dynamic_dims[node] = 3; }
} else {
if (src_dyn == 0) {
m_node_dynamic_dims[node] = 1; } else if (src_dyn == 2) {
m_node_dynamic_dims[node] = 2; }
}
if (m_node_dynamic_dims[node] != -1) {
OPENVINO_ASSERT(node->src[1]->ne[src_dyn] == node->ne[m_node_dynamic_dims[node]],
"Dynamic dim value mismatch for IM2COL node: " + std::string(node->name) +
" and its src[1]: " + std::string(node->src[1]->name));
}
}
break;
}
default:
GGML_LOG_DEBUG("ggml-openvino: compute_node_dynamic_dims: unhandled op %s for node '%s'\n",
ggml_op_name(node->op), node->name);
break;
}
};
for (int i = 0; i < m_cgraph->n_nodes; i++) {
ggml_tensor * node = m_cgraph->nodes[i];
visit_node(visit_node, node);
}
if (0) {
for (int i = 0; i < m_cgraph->n_nodes; i++) {
ggml_tensor * node = m_cgraph->nodes[i];
int dynamic_dim = m_node_dynamic_dims[node];
std::cout << "[" << i << "] " << "node_name: " << node->name << " op: " << ggml_op_name(node->op)
<< " shape: [";
for (int j = 0; j < 4; j++) {
if (j == dynamic_dim) {
std::cout << "*";
} else {
std::cout << node->ne[j];
}
if (j < 3) {
std::cout << ", ";
}
}
std::cout << "]" << '\n';
for (int j = 0; j < GGML_MAX_SRC; j++) {
ggml_tensor * src = node->src[j];
if (src == nullptr) {
continue;
}
int src_dynamic_dim = m_node_dynamic_dims[src];
std::cout << " [" << j << "] src_name: " << src->name << " [";
for (int k = 0; k < 4; k++) {
if (k == src_dynamic_dim) {
std::cout << "*";
} else {
std::cout << src->ne[k];
}
if (k < 3) {
std::cout << ", ";
}
}
std::cout << "]" << '\n';
}
std::cout << '\n';
}
}
}