-
Notifications
You must be signed in to change notification settings - Fork 24.1k
ggml : remove GGML_KQ_MASK_PAD constant #17910
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from 1 commit
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -385,7 +385,7 @@ bool llm_graph_input_attn_kv::can_reuse(const llm_graph_params & params) { | |
| //res &= self_v_idxs->ne[0] == params.ubatch.n_tokens; // TODO: need to move this to the unified cache and check there | ||
|
|
||
| res &= self_kq_mask->ne[0] == mctx->get_n_kv(); | ||
| res &= self_kq_mask->ne[1] == GGML_PAD(params.ubatch.n_tokens, GGML_KQ_MASK_PAD); | ||
| res &= self_kq_mask->ne[1] == params.ubatch.n_tokens; | ||
|
|
||
| return res; | ||
| } | ||
|
|
@@ -416,10 +416,10 @@ bool llm_graph_input_attn_kv_iswa::can_reuse(const llm_graph_params & params) { | |
| //res &= self_v_idxs_swa->ne[0] == params.ubatch.n_tokens; // TODO: need to move this to the unified cache and check there | ||
|
|
||
| res &= self_kq_mask->ne[0] == mctx->get_base()->get_n_kv(); | ||
| res &= self_kq_mask->ne[1] == GGML_PAD(params.ubatch.n_tokens, GGML_KQ_MASK_PAD); | ||
| res &= self_kq_mask->ne[1] == params.ubatch.n_tokens; | ||
|
|
||
| res &= self_kq_mask_swa->ne[0] == mctx->get_swa()->get_n_kv(); | ||
| res &= self_kq_mask_swa->ne[1] == GGML_PAD(params.ubatch.n_tokens, GGML_KQ_MASK_PAD); | ||
| res &= self_kq_mask_swa->ne[1] == params.ubatch.n_tokens; | ||
|
|
||
| return res; | ||
| } | ||
|
|
@@ -452,7 +452,7 @@ void llm_graph_input_attn_cross::set_input(const llama_ubatch * ubatch) { | |
| } | ||
| } | ||
|
|
||
| for (int i = n_tokens; i < GGML_PAD(n_tokens, GGML_KQ_MASK_PAD); ++i) { | ||
| for (int i = n_tokens; i < n_tokens; ++i) { | ||
|
Collaborator
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. @ggerganov Totally random, but while looking at some of the ModernBert stuff, I (and @hansolosan) came across this line which now seems to do nothing with the removal of the padding. Should this whole set of nested loops be removed?
Collaborator
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. cc @ryan-mangeno in case this has anything to do with your ongoing Modern Bert divergence issues
Collaborator
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. I don't think it related to Modern Bert after all. From what I can tell, this (as well as
Collaborator
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. I don't think this is actually causing any problems, but I made a PR to clean it up for future readers: #18795
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
alright, thanks for letting me know though @gabe-l-hart ! I just got back to school but I have been doing quite a bit of debugging between transformers and llama.cpp in terms of finding where they diverge, if its useful ... I have been inspecting the pooling process much closer and using the llama-eval-callback tool, but there is no pooling flag to be passed, would it make sense to add support for this / expose it. In my case I can hardcode it in llama-embedding.cpp and view the tensors made during build_pooling. |
||
| for (int j = 0; j < n_enc; ++j) { | ||
| data[h*(n_enc*n_tokens) + i*n_enc + j] = -INFINITY; | ||
| } | ||
|
|
@@ -1470,13 +1470,13 @@ llm_graph_input_attn_no_cache * llm_graph_context::build_attn_inp_no_cache() con | |
| auto inp = std::make_unique<llm_graph_input_attn_no_cache>(hparams, cparams); | ||
|
|
||
| // note: there is no KV cache, so the number of KV values is equal to the number of tokens in the batch | ||
| inp->self_kq_mask = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_tokens, GGML_PAD(n_tokens, GGML_KQ_MASK_PAD), 1, 1); | ||
| inp->self_kq_mask = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_tokens, n_tokens, 1, 1); | ||
| ggml_set_input(inp->self_kq_mask); | ||
|
|
||
| inp->self_kq_mask_cnv = cparams.flash_attn ? ggml_cast(ctx0, inp->self_kq_mask, GGML_TYPE_F16) : inp->self_kq_mask; | ||
|
|
||
| if (hparams.swa_type != LLAMA_SWA_TYPE_NONE) { | ||
| inp->self_kq_mask_swa = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_tokens, GGML_PAD(n_tokens, GGML_KQ_MASK_PAD), 1, 1); | ||
| inp->self_kq_mask_swa = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_tokens, n_tokens, 1, 1); | ||
| ggml_set_input(inp->self_kq_mask_swa); | ||
|
|
||
| inp->self_kq_mask_swa_cnv = cparams.flash_attn ? ggml_cast(ctx0, inp->self_kq_mask_swa, GGML_TYPE_F16) : inp->self_kq_mask_swa; | ||
|
|
@@ -1558,7 +1558,7 @@ static std::unique_ptr<llm_graph_input_attn_kv> build_attn_inp_kv_impl( | |
| inp->self_k_idxs = mctx_cur->build_input_k_idxs(ctx0, ubatch); | ||
| inp->self_v_idxs = mctx_cur->build_input_v_idxs(ctx0, ubatch); | ||
|
|
||
| inp->self_kq_mask = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_kv, GGML_PAD(n_tokens/n_stream, GGML_KQ_MASK_PAD), 1, n_stream); | ||
| inp->self_kq_mask = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_kv, n_tokens/n_stream, 1, n_stream); | ||
| ggml_set_input(inp->self_kq_mask); | ||
|
|
||
| inp->self_kq_mask_cnv = cparams.flash_attn ? ggml_cast(ctx0, inp->self_kq_mask, GGML_TYPE_F16) : inp->self_kq_mask; | ||
|
|
@@ -1701,7 +1701,7 @@ llm_graph_input_attn_cross * llm_graph_context::build_attn_inp_cross() const { | |
|
|
||
| const int32_t n_enc = !cross->v_embd.empty() ? cross->n_enc : hparams.n_ctx_train; | ||
|
|
||
| inp->cross_kq_mask = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_enc, GGML_PAD(n_tokens, GGML_KQ_MASK_PAD), 1, 1); | ||
| inp->cross_kq_mask = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_enc, n_tokens, 1, 1); | ||
| ggml_set_input(inp->cross_kq_mask); | ||
|
|
||
| inp->cross_kq_mask_cnv = cparams.flash_attn ? ggml_cast(ctx0, inp->cross_kq_mask, GGML_TYPE_F16) : inp->cross_kq_mask; | ||
|
|
@@ -1767,7 +1767,7 @@ llm_graph_input_attn_kv_iswa * llm_graph_context::build_attn_inp_kv_iswa() const | |
| inp->self_k_idxs = mctx_cur->get_base()->build_input_k_idxs(ctx0, ubatch); | ||
| inp->self_v_idxs = mctx_cur->get_base()->build_input_v_idxs(ctx0, ubatch); | ||
|
|
||
| inp->self_kq_mask = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_kv, GGML_PAD(n_tokens/n_stream, GGML_KQ_MASK_PAD), 1, n_stream); | ||
| inp->self_kq_mask = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_kv, n_tokens/n_stream, 1, n_stream); | ||
| ggml_set_input(inp->self_kq_mask); | ||
|
|
||
| inp->self_kq_mask_cnv = cparams.flash_attn ? ggml_cast(ctx0, inp->self_kq_mask, GGML_TYPE_F16) : inp->self_kq_mask; | ||
|
|
@@ -1781,7 +1781,7 @@ llm_graph_input_attn_kv_iswa * llm_graph_context::build_attn_inp_kv_iswa() const | |
| inp->self_k_idxs_swa = mctx_cur->get_swa()->build_input_k_idxs(ctx0, ubatch); | ||
| inp->self_v_idxs_swa = mctx_cur->get_swa()->build_input_v_idxs(ctx0, ubatch); | ||
|
|
||
| inp->self_kq_mask_swa = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_kv, GGML_PAD(n_tokens/n_stream, GGML_KQ_MASK_PAD), 1, n_stream); | ||
| inp->self_kq_mask_swa = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, n_kv, n_tokens/n_stream, 1, n_stream); | ||
| ggml_set_input(inp->self_kq_mask_swa); | ||
|
|
||
| inp->self_kq_mask_swa_cnv = cparams.flash_attn ? ggml_cast(ctx0, inp->self_kq_mask_swa, GGML_TYPE_F16) : inp->self_kq_mask_swa; | ||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
There's a use of GGML_KQ_MASK_PAD in the comment on the next line.