llama : (mrope) allow using normal 1D position for text token (#13138)

* llama : (mrope) use normal position for text token * rm n_pos_per_embd from llm_graph_input_attn_temp
2025-08-21 07:03:43 -04:00 · 2025-04-28 14:20:56 +02:00
parent 5fa9e63be8
commit d2b2031e5f
3 changed files with 24 additions and 22 deletions
--- a/src/llama-graph.h
+++ b/src/llama-graph.h
@@ -90,29 +90,27 @@ public:

 class llm_graph_input_pos : public llm_graph_input_i {
 public:
-    llm_graph_input_pos(int64_t n_pos_per_token) : n_pos_per_token(n_pos_per_token) {}
+    llm_graph_input_pos(int64_t n_pos_per_embd) : n_pos_per_embd(n_pos_per_embd) {}
    virtual ~llm_graph_input_pos() = default;

    void set_input(const llama_ubatch * ubatch) override;

    ggml_tensor * pos = nullptr; // I32 [n_batch]

-    const int64_t n_pos_per_token = 1;
+    const int64_t n_pos_per_embd = 1;
 };

 // temperature tuning, used by llama4
 class llm_graph_input_attn_temp : public llm_graph_input_i {
 public:
-    llm_graph_input_attn_temp(int64_t n_pos_per_token, uint32_t n_attn_temp_floor_scale, float f_attn_temp_scale)
-        : n_pos_per_token(n_pos_per_token), n_attn_temp_floor_scale(n_attn_temp_floor_scale), f_attn_temp_scale(f_attn_temp_scale) {}
+    llm_graph_input_attn_temp(uint32_t n_attn_temp_floor_scale, float f_attn_temp_scale)
+        : n_attn_temp_floor_scale(n_attn_temp_floor_scale), f_attn_temp_scale(f_attn_temp_scale) {}
    virtual ~llm_graph_input_attn_temp() = default;

    void set_input(const llama_ubatch * ubatch) override;

    ggml_tensor * attn_scale = nullptr; // F32 [n_batch]

-    const int64_t n_pos_per_token = 1;
-
    const uint32_t n_attn_temp_floor_scale;
    const float    f_attn_temp_scale;
 };
@@ -419,7 +417,7 @@ struct llm_graph_context {

    llm_graph_context(const llm_graph_params & params);

-    int64_t n_pos_per_token() const;
+    int64_t n_pos_per_embd() const;

    void cb(ggml_tensor * cur, const char * name, int il) const;