add llama config; add llama2 70 support; update readme

daixu · daixu · commit 1794ec46e8ce · 2024-03-27T03:38:40.000Z
diff --git a/configs/llama2_70b_config.json b/configs/llama2_70b_config.json
@@ -0,0 +1,26 @@
+{
+    "source": "https://huggingface.co/meta-llama/Llama-2-70b-hf/blob/main/config.json",
+    "_name_or_path": "meta-llama/Llama-2-70b-hf",
+    "architectures": [
+      "LlamaForCausalLM"
+    ],
+    "bos_token_id": 1,
+    "eos_token_id": 2,
+    "hidden_act": "silu",
+    "hidden_size": 8192,
+    "initializer_range": 0.02,
+    "intermediate_size": 28672,
+    "max_position_embeddings": 4096,
+    "model_type": "llama",
+    "num_attention_heads": 64,
+    "num_hidden_layers": 80,
+    "num_key_value_heads": 8,
+    "pretraining_tp": 1,
+    "rms_norm_eps": 1e-05,
+    "rope_scaling": null,
+    "tie_word_embeddings": false,
+    "torch_dtype": "float16",
+    "transformers_version": "4.32.0.dev0",
+    "use_cache": true,
+    "vocab_size": 32000
+  }
diff --git a/configs/llama2_7b_config.json b/configs/llama2_7b_config.json
@@ -0,0 +1,26 @@
+{
+    "source": "https://huggingface.co/meta-llama/Llama-2-7b-hf/blob/main/config.json",
+    "_name_or_path": "meta-llama/Llama-2-7b-hf",
+    "architectures": [
+      "LlamaForCausalLM"
+    ],
+    "bos_token_id": 1,
+    "eos_token_id": 2,
+    "hidden_act": "silu",
+    "hidden_size": 4096,
+    "initializer_range": 0.02,
+    "intermediate_size": 11008,
+    "max_position_embeddings": 4096,
+    "model_type": "llama",
+    "num_attention_heads": 32,
+    "num_hidden_layers": 32,
+    "num_key_value_heads": 32,
+    "pretraining_tp": 1,
+    "rms_norm_eps": 1e-05,
+    "rope_scaling": null,
+    "tie_word_embeddings": false,
+    "torch_dtype": "float16",
+    "transformers_version": "4.31.0.dev0",
+    "use_cache": true,
+    "vocab_size": 32000
+  }
diff --git a/configuration_llama.py b/configuration_llama.py
@@ -0,0 +1,178 @@
+# coding=utf-8
+# Copyright 2022 EleutherAI and the HuggingFace Inc. team. All rights reserved.
+#
+# This code is based on EleutherAI's GPT-NeoX library and the GPT-NeoX
+# and OPT implementations in this library. It has been modified from its
+# original forms to accommodate minor architectural differences compared
+# to GPT-NeoX and OPT used by the Meta AI team that trained the model.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+""" LLaMA model configuration"""
+
+
+LLAMA_PRETRAINED_CONFIG_ARCHIVE_MAP = {}
+
+
+class LlamaConfig():
+    r"""
+    This is the configuration class to store the configuration of a [`LlamaModel`]. It is used to instantiate an LLaMA
+    model according to the specified arguments, defining the model architecture. Instantiating a configuration with the
+    defaults will yield a similar configuration to that of the LLaMA-7B.
+
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+
+
+    Args:
+        vocab_size (`int`, *optional*, defaults to 32000):
+            Vocabulary size of the LLaMA model. Defines the number of different tokens that can be represented by the
+            `inputs_ids` passed when calling [`LlamaModel`]
+        hidden_size (`int`, *optional*, defaults to 4096):
+            Dimension of the hidden representations.
+        intermediate_size (`int`, *optional*, defaults to 11008):
+            Dimension of the MLP representations.
+        num_hidden_layers (`int`, *optional*, defaults to 32):
+            Number of hidden layers in the Transformer decoder.
+        num_attention_heads (`int`, *optional*, defaults to 32):
+            Number of attention heads for each attention layer in the Transformer decoder.
+        num_key_value_heads (`int`, *optional*):
+            This is the number of key_value heads that should be used to implement Grouped Query Attention. If
+            `num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if
+            `num_key_value_heads=1 the model will use Multi Query Attention (MQA) otherwise GQA is used. When
+            converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
+            by meanpooling all the original heads within that group. For more details checkout [this
+            paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to
+            `num_attention_heads`.
+        hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
+            The non-linear activation function (function or string) in the decoder.
+        max_position_embeddings (`int`, *optional*, defaults to 2048):
+            The maximum sequence length that this model might ever be used with. Llama 1 supports up to 2048 tokens,
+            Llama 2 up to 4096, CodeLlama up to 16384.
+        initializer_range (`float`, *optional*, defaults to 0.02):
+            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
+        rms_norm_eps (`float`, *optional*, defaults to 1e-06):
+            The epsilon used by the rms normalization layers.
+        use_cache (`bool`, *optional*, defaults to `True`):
+            Whether or not the model should return the last key/values attentions (not used by all models). Only
+            relevant if `config.is_decoder=True`.
+        pad_token_id (`int`, *optional*):
+            Padding token id.
+        bos_token_id (`int`, *optional*, defaults to 1):
+            Beginning of stream token id.
+        eos_token_id (`int`, *optional*, defaults to 2):
+            End of stream token id.
+        pretraining_tp (`int`, *optional*, defaults to 1):
+            Experimental feature. Tensor parallelism rank used during pretraining. Please refer to [this
+            document](https://huggingface.co/docs/transformers/parallelism) to understand more about it. This value is
+            necessary to ensure exact reproducibility of the pretraining results. Please refer to [this
+            issue](https://github.com/pytorch/pytorch/issues/76232).
+        tie_word_embeddings (`bool`, *optional*, defaults to `False`):
+            Whether to tie weight embeddings
+        rope_theta (`float`, *optional*, defaults to 10000.0):
+            The base period of the RoPE embeddings.
+        rope_scaling (`Dict`, *optional*):
+            Dictionary containing the scaling configuration for the RoPE embeddings. Currently supports two scaling
+            strategies: linear and dynamic. Their scaling factor must be a float greater than 1. The expected format is
+            `{"type": strategy name, "factor": scaling factor}`. When using this flag, don't update
+            `max_position_embeddings` to the expected new maximum. See the following thread for more information on how
+            these scaling strategies behave:
+            https://www.reddit.com/r/LocalLLaMA/comments/14mrgpr/dynamically_scaled_rope_further_increases/. This is an
+            experimental feature, subject to breaking API changes in future versions.
+        attention_bias (`bool`, defaults to `False`, *optional*, defaults to `False`):
+            Whether to use a bias in the query, key, value and output projection layers during self-attention.
+
+
+    ```python
+    from configuration_llama import LlamaConfig
+
+    # 使用默认的配置
+    configuration = LlamaConfig()
+
+    # 使用配置数据初始化一个LlamaConfig对象
+    import json
+    with open('configs/llama2_7b_config.json', 'r') as f:
+        config_data = json.load(f)
+    configuration = LlamaConfig(**config_data)
+    
+    ```
+    """
+    model_type = "llama"
+    keys_to_ignore_at_inference = ["past_key_values"]
+
+    def __init__(
+        self,
+        vocab_size=32000,
+        hidden_size=4096,
+        intermediate_size=11008,
+        num_hidden_layers=32,
+        num_attention_heads=32,
+        num_key_value_heads=None,
+        hidden_act="silu",
+        max_position_embeddings=2048,
+        initializer_range=0.02,
+        rms_norm_eps=1e-6,
+        use_cache=True,
+        pad_token_id=None,
+        bos_token_id=1,
+        eos_token_id=2,
+        pretraining_tp=1,
+        tie_word_embeddings=False,
+        rope_theta=10000.0,
+        rope_scaling=None,
+        attention_bias=False,
+        weights_dir = 'weights/llama2_7b',
+        **kwargs,
+    ):
+        self.vocab_size = vocab_size
+        self.max_position_embeddings = max_position_embeddings
+        self.hidden_size = hidden_size
+        self.intermediate_size = intermediate_size
+        self.num_hidden_layers = num_hidden_layers
+        self.num_attention_heads = num_attention_heads
+
+        # for backward compatibility
+        if num_key_value_heads is None:
+            num_key_value_heads = num_attention_heads
+
+        self.num_key_value_heads = num_key_value_heads
+        self.hidden_act = hidden_act
+        self.initializer_range = initializer_range
+        self.rms_norm_eps = rms_norm_eps
+        self.pretraining_tp = pretraining_tp
+        self.use_cache = use_cache
+        self.rope_theta = rope_theta
+        self.rope_scaling = rope_scaling
+        self._rope_scaling_validation()
+        self.attention_bias = attention_bias
+        self.weights_dir = weights_dir
+
+    def _rope_scaling_validation(self):
+        """
+        Validate the `rope_scaling` configuration.
+        """
+        if self.rope_scaling is None:
+            return
+
+        if not isinstance(self.rope_scaling, dict) or len(self.rope_scaling) != 2:
+            raise ValueError(
+                "`rope_scaling` must be a dictionary with with two fields, `type` and `factor`, "
+                f"got {self.rope_scaling}"
+            )
+        rope_scaling_type = self.rope_scaling.get("type", None)
+        rope_scaling_factor = self.rope_scaling.get("factor", None)
+        if rope_scaling_type is None or rope_scaling_type not in ["linear", "dynamic"]:
+            raise ValueError(
+                f"`rope_scaling`'s type field must be one of ['linear', 'dynamic'], got {rope_scaling_type}"
+            )
+        if rope_scaling_factor is None or not isinstance(rope_scaling_factor, float) or rope_scaling_factor <= 1.0:
+            raise ValueError(f"`rope_scaling`'s factor field must be a float > 1, got {rope_scaling_factor}")
diff --git a/convert_hf_to_pkl.py b/convert_hf_to_pkl.py
@@ -12,6 +12,11 @@
         'hf_model': 'meta-llama/Llama-2-7b-hf',
         'tokenizer': 'meta-llama/Llama-2-7b-hf',
         'weights_dir': 'weights/llama2_7b/'
+    },
+    'llama2_70b': {
+        'hf_model': 'meta-llama/Llama-2-70b-hf',
+        'tokenizer': 'meta-llama/Llama-2-70b-hf',
+        'weights_dir': 'weights/llama2_70b/'
     }
 }
 
diff --git a/naked_llama2.py b/naked_llama2.py
@@ -0,0 +1,101 @@
+import os.path as osp
+import torch
+import argparse
+from transformers import AutoTokenizer, LlamaForCausalLM
+from utils import npy_to_tensor, load_llama_config, get_attentioin_mask
+from configuration_llama import LlamaConfig
+from layers.norm import RMSNorm
+from layers.rope import init_rope_embeddings
+from layers.embedding import embedding_lookup
+from layers.matmul import LlamaMLP, lm_head
+from layers.transformer_block import llama2_transformer_block
+
+
+def llama2(token_ids: torch.Tensor, config: LlamaConfig):
+    """
+    手动实现llama2 7B/13B/70B的推理计算。
+    
+    参数:
+    - token_ids: token id组成的tensor，形状为 [batch_size, seq_length]
+    """
+    bsz, seq_length = token_ids.shape
+    # embedding 
+    embdding_weights = npy_to_tensor(osp.join(config.weights_dir, 'model.embed_tokens.weight.npy'))
+    input_embeds = embedding_lookup(token_ids, embdding_weights)
+    hidden_states = input_embeds  # shape [batch_size, seq_length, hidden_size], hidden_size=4096
+    
+    # mask
+    mask = get_attentioin_mask(start_pos=0, seq_length=seq_length, ref_tensor=hidden_states)
+
+    # 重复 32次(7B)/ 80次(70B) llama2_transformer_block 的计算
+    for layer_id in range(config.num_hidden_layers):
+        print(f'Naked llama2: Computing Layer {layer_id}')
+        output = llama2_transformer_block(hidden_states, config, layer_id=layer_id, attention_mask=mask)
+        hidden_states = output[0]
+    
+    # 先 RMSNorm，然后head输出
+    norm_weight = npy_to_tensor(osp.join(config.weights_dir, 'model.norm.weight.npy'))
+    hidden_states = RMSNorm(hidden_states, norm_weight, eps=config.rms_norm_eps)
+    
+    lm_head_weight = npy_to_tensor(osp.join(config.weights_dir, 'lm_head.weight.npy'))
+    logits = lm_head(hidden_states, lm_head_weight)
+    return logits
+
+
+if __name__ == '__main__':
+    parser = argparse.ArgumentParser(description='nake.')
+    parser.add_argument('--model_size', type=str, 
+                        help='prammeter size of the llama2 model to use', 
+                        default='7b', 
+                        choices=['7b', '70b']
+                        )
+    args = parser.parse_args()
+    
+    # initial rope embeddings
+    init_rope_embeddings(dim=128)
+    prompt = "Hey, are you conscious? Can you talk to me?"
+    model_dict = {
+        "llama2_7b": {
+            'tokenizer': 'meta-llama/Llama-2-7b-hf',
+            'config_path': 'configs/llama2_7b_config.json',
+            'weights_dir': 'weights/llama2_7b/'
+        },
+        "llama2_70b": {
+            'tokenizer': 'meta-llama/Llama-2-70b-hf',
+            'config_path': 'configs/llama2_70b_config.json',
+            'weights_dir': 'weights/llama2_70b/'
+        }
+    }
+    if args.model_size == '7b':
+        model_name = "llama2_7b"
+    elif args.model_size == '70b':
+        model_name = "llama2_70b"
+        
+    print('Model:', model_name)   
+    
+    # tokenization
+    tokenizer = AutoTokenizer.from_pretrained(model_dict[model_name]['tokenizer'])
+    inputs = tokenizer(prompt, return_tensors="pt")
+    token_ids = inputs.input_ids
+
+    # random input
+    # token_ids = torch.randint(0, 32000, (1, 512))  # (1, 512) shape
+
+    config = load_llama_config(model_dict[model_name]['config_path'])
+    config.weights_dir = model_dict[model_name]['weights_dir']
+    logits = llama2(token_ids, config)
+    
+    print('Naked llama result:')
+    print(logits)
+    
+    # check result
+    model = LlamaForCausalLM.from_pretrained("meta-llama/Llama-2-7b-hf")
+    model.eval()
+    with torch.inference_mode():
+        hf_res = model(input_ids = token_ids)
+        print('Hugging face llama result:')
+        print(hf_res.logits)
+    error = torch.abs(hf_res.logits-logits)
+    print(f"Compare error sum: {torch.sum(error)}") 
+
+    
diff --git a/utils.py b/utils.py
@@ -1,5 +1,7 @@
 import numpy as np
 import torch
+import json
+from configuration_llama import LlamaConfig
 
 
 def npy_to_tensor(npy_name):
@@ -8,3 +10,28 @@ def npy_to_tensor(npy_name):
     loaded_tensor = torch.from_numpy(loaded_numpy_array)
     loaded_tensor = loaded_tensor.to(torch.float32) # 将张量转换为 float32 类型
     return loaded_tensor
+
+
+def load_llama_config(config_file):
+    with open(config_file, "r") as f:
+        config_data = json.load(f)
+    configuration = LlamaConfig(**config_data)
+    return configuration
+
+
+def get_attentioin_mask(start_pos, seq_length, ref_tensor):
+    if seq_length > 1:
+        mask = torch.full((seq_length, seq_length), float("-inf"), device=ref_tensor.device)
+        
+        mask = torch.triu(mask, diagonal=1)
+        # When performing key-value caching, we compute the attention scores
+        # only for the new sequence. Thus, the matrix of scores is of size
+        # (seqlen, cache_len + seqlen), and the only masked entries are (i, j) for
+        # j > cache_len + i, since row i corresponds to token cache_len + i.
+        mask = torch.hstack([
+            torch.zeros((seq_length, start_pos), device=ref_tensor.device),
+            mask
+        ]).type_as(ref_tensor)
+    else:
+        mask = None
+    return mask

Original file line number	Diff line number	Diff line change
`@@ -12,6 +12,11 @@`
`12`	`12`	`'hf_model': 'meta-llama/Llama-2-7b-hf',`
`13`	`13`	`'tokenizer': 'meta-llama/Llama-2-7b-hf',`
`14`	`14`	`'weights_dir': 'weights/llama2_7b/'`
	`15`	`+ },`
	`16`	`+ 'llama2_70b': {`
	`17`	`+ 'hf_model': 'meta-llama/Llama-2-70b-hf',`
	`18`	`+ 'tokenizer': 'meta-llama/Llama-2-70b-hf',`
	`19`	`+ 'weights_dir': 'weights/llama2_70b/'`
`15`	`20`	`}`
`16`	`21`	`}`
`17`	`22`