Use runtime profiling to replace manual memory analyzers (#81)

2023-05-19 11:35:44 -06:00
parent 825d8892b5
commit f756799b84
14 changed files with 211 additions and 478 deletions
--- a/cacheflow/model_executor/models/llama.py
+++ b/cacheflow/model_executor/models/llama.py
@ -104,7 +104,8 @@ class LlamaAttention(nn.Module):
            input_is_parallel=True,
            perform_initialization=False,
        )
-        self.attn = GPTNeoXCacheFlowAttention(self.scaling, self.head_dim)
+        self.attn = GPTNeoXCacheFlowAttention(self.num_heads, self.head_dim,
+                                              self.scaling, rotary_dim=self.head_dim)

    def forward(
        self,