[Quality] Add code formatter and linter (#326)

2023-07-03 11:31:55 -07:00
parent 0ffded812a
commit d6fa1be3a8
47 changed files with 1547 additions and 617 deletions
--- a/vllm/model_executor/models/gpt_bigcode.py
+++ b/vllm/model_executor/models/gpt_bigcode.py
@@ -1,5 +1,6 @@
 # coding=utf-8
-# Adapted from https://github.com/huggingface/transformers/blob/v4.28.0/src/transformers/models/gpt2/modeling_gpt2.py
+# Adapted from
+# https://github.com/huggingface/transformers/blob/v4.28.0/src/transformers/models/gpt2/modeling_gpt2.py
 # Copyright 2023 The vLLM team.
 # Copyright 2023 CTranslate2, and Michael Feil
 # Copyright 2018 The OpenAI Team Authors and HuggingFace Inc. team.
@@ -49,19 +50,25 @@ class GPTBigCodeAttention(nn.Module):
        super().__init__()
        self.hidden_size = config.hidden_size
        total_num_heads = config.num_attention_heads
-        tensor_model_parallel_world_size = get_tensor_model_parallel_world_size()
+        tensor_model_parallel_world_size = (
+            get_tensor_model_parallel_world_size())
        assert total_num_heads % tensor_model_parallel_world_size == 0
        self.num_heads = total_num_heads // tensor_model_parallel_world_size
        self.head_dim = self.hidden_size // total_num_heads
-        self.scale = self.head_dim ** -0.5
+        self.scale = self.head_dim**-0.5

-        self.c_attn = ColumnParallelLinear(self.hidden_size, 3 * self.hidden_size,
-                                           bias=True, gather_output=False,
+        self.c_attn = ColumnParallelLinear(self.hidden_size,
+                                           3 * self.hidden_size,
+                                           bias=True,
+                                           gather_output=False,
                                           perform_initialization=False)
-        self.c_proj = RowParallelLinear(self.hidden_size, self.hidden_size,
-                                        bias=True, input_is_parallel=True,
+        self.c_proj = RowParallelLinear(self.hidden_size,
+                                        self.hidden_size,
+                                        bias=True,
+                                        input_is_parallel=True,
                                        perform_initialization=False)
-        self.attn = PagedAttention(self.num_heads, self.head_dim,
+        self.attn = PagedAttention(self.num_heads,
+                                   self.head_dim,
                                   scale=self.scale)

    def forward(
@@ -74,8 +81,8 @@ class GPTBigCodeAttention(nn.Module):
        qkv, _ = self.c_attn(hidden_states)
        q, k, v = qkv.chunk(chunks=3, dim=-1)
        key_cache, value_cache = kv_cache
-        attn_output = self.attn(
-            q, k, v, key_cache, value_cache, input_metadata, cache_event)
+        attn_output = self.attn(q, k, v, key_cache, value_cache,
+                                input_metadata, cache_event)
        attn_output, _ = self.c_proj(attn_output)
        return attn_output

@@ -89,11 +96,15 @@ class GPTBigMLP(nn.Module):
    ):
        super().__init__()
        hidden_size = config.hidden_size
-        self.c_fc = ColumnParallelLinear(hidden_size, intermediate_size,
-                                         bias=True, gather_output=False,
+        self.c_fc = ColumnParallelLinear(hidden_size,
+                                         intermediate_size,
+                                         bias=True,
+                                         gather_output=False,
                                         perform_initialization=False)
-        self.c_proj = RowParallelLinear(intermediate_size, hidden_size,
-                                        bias=True, input_is_parallel=True,
+        self.c_proj = RowParallelLinear(intermediate_size,
+                                        hidden_size,
+                                        bias=True,
+                                        input_is_parallel=True,
                                        perform_initialization=False)
        self.act = get_act_fn(config.activation_function)

@@ -109,7 +120,8 @@ class GPTBigCodeBlock(nn.Module):
    def __init__(self, config: GPTBigCodeConfig):
        super().__init__()
        hidden_size = config.hidden_size
-        inner_dim = config.n_inner if config.n_inner is not None else 4 * hidden_size
+        inner_dim = (config.n_inner if config.n_inner is not None else 4 *
+                     hidden_size)

        self.ln_1 = nn.LayerNorm(hidden_size, eps=config.layer_norm_epsilon)
        self.attn = GPTBigCodeAttention(config)
@@ -147,7 +159,7 @@ class GPTBigCodeModel(nn.Module):
    def __init__(self, config: GPTBigCodeConfig):
        super().__init__()
        self.config = config
-        assert config.add_cross_attention == False
+        assert not config.add_cross_attention

        self.embed_dim = config.hidden_size

@@ -181,8 +193,8 @@ class GPTBigCodeModel(nn.Module):
            else:
                cache_event = cache_events[i]
            layer = self.h[i]
-            hidden_states = layer(
-                hidden_states, kv_caches[i], input_metadata, cache_event)
+            hidden_states = layer(hidden_states, kv_caches[i], input_metadata,
+                                  cache_event)

        hidden_states = self.ln_f(hidden_states)
        return hidden_states
@@ -207,24 +219,26 @@ class GPTBigCodeForCausalLM(nn.Module):
        input_metadata: InputMetadata,
        cache_events: Optional[List[torch.cuda.Event]],
    ) -> Dict[int, SequenceOutputs]:
-        hidden_states = self.transformer(
-            input_ids, positions, kv_caches, input_metadata, cache_events)
-        next_tokens = self.sampler(
-            self.lm_head_weight, hidden_states, input_metadata)
+        hidden_states = self.transformer(input_ids, positions, kv_caches,
+                                         input_metadata, cache_events)
+        next_tokens = self.sampler(self.lm_head_weight, hidden_states,
+                                   input_metadata)
        return next_tokens

    _column_parallel_weights = ["wte.weight", "c_fc.weight", "c_fc.bias"]
    _row_parallel_weights = ["c_proj.weight"]

-    def load_weights(self, model_name_or_path: str,
+    def load_weights(self,
+                     model_name_or_path: str,
                     cache_dir: Optional[str] = None,
                     use_np_cache: bool = False):
-        tensor_model_parallel_world_size = get_tensor_model_parallel_world_size()
+        tensor_model_parallel_world_size = (
+            get_tensor_model_parallel_world_size())
        tensor_model_parallel_rank = get_tensor_model_parallel_rank()
        state_dict = self.state_dict()

        for name, loaded_weight in hf_model_weights_iterator(
-            model_name_or_path, cache_dir, use_np_cache):
+                model_name_or_path, cache_dir, use_np_cache):
            if "lm_head.weight" in name:
                # GPT-2 ties the weights of the embedding layer and the final
                # linear layer.
@@ -241,9 +255,11 @@ class GPTBigCodeForCausalLM(nn.Module):

            if name == "transformer.wte.weight":
                # Consider padding in the vocab size.
-                padded_vocab_size = param.shape[0] * tensor_model_parallel_world_size
+                padded_vocab_size = param.shape[
+                    0] * tensor_model_parallel_world_size
                num_extra_rows = padded_vocab_size - self.config.vocab_size
-                extra_rows = torch.empty(num_extra_rows, loaded_weight.shape[1])
+                extra_rows = torch.empty(num_extra_rows,
+                                         loaded_weight.shape[1])
                extra_rows = extra_rows.to(loaded_weight)
                loaded_weight = torch.cat([loaded_weight, extra_rows], dim=0)

@@ -258,25 +274,31 @@ class GPTBigCodeForCausalLM(nn.Module):
                qkv_array = qkv_array.numpy()

                dims_q = n_head * head_dim
-                q, k, v = np.split(qkv_array, (dims_q, dims_q + head_dim), axis=0)
-                # q is fine, but k & v have not replicated shape along the first axis
-                # as long as MQA is not nativly supported, increase memory and replicated
-                # (head_dim, hidden_dim) to (n_heads * head_dim, hidden_dim)
+                # pylint: disable=unbalanced-tuple-unpacking
+                q, k, v = np.split(qkv_array, (dims_q, dims_q + head_dim),
+                                   axis=0)
+                # q is fine, but k & v have not replicated shape along the first
+                # axis as long as MQA is not nativly supported, increase memory
+                # and replicated (head_dim, hidden_dim) to
+                # (n_heads * head_dim, hidden_dim)
                if k.ndim == 2 and v.ndim == 2:
                    replication = (n_head, 1)  # weights
                else:
                    replication = n_head  # biases
                # replicate n_head times for q, v
                k, v = np.tile(k, replication), np.tile(v, replication)
-                # concat q, k, v along the first axis (n_heads * head_dim, hidden_dim)
+                # concat q, k, v along the first axis
+                # (n_heads * head_dim, hidden_dim)
                # to (3 * n_heads * head_dim, hidden_dim)
                qkv_array = np.concatenate((q, k, v), axis=0)
                return torch.from_numpy(qkv_array)

            # For the fused QKV linear layer, manually shard the weights.
            if "c_attn" in name:
-                # GPT-2's fused QKV has the shape of [3 * num_heads * head_size, hidden_size].
-                # When tensor parallelism is used, we shard the weights along the head dimension.
+                # GPT-2's fused QKV has the shape of
+                # [3 * num_heads * head_size, hidden_size].
+                # When tensor parallelism is used, we shard the weights along
+                # the head dimension.
                total_num_heads = self.config.num_attention_heads
                hidden_size = self.config.hidden_size
                head_size = hidden_size // total_num_heads
@@ -285,13 +307,19 @@ class GPTBigCodeForCausalLM(nn.Module):
                head_end = (tensor_model_parallel_rank + 1) * num_heads

                if name.endswith(".weight"):
-                    loaded_weight = _expand_mqa_mha(loaded_weight, n_head=total_num_heads, head_dim=head_size)
-                    loaded_weight = loaded_weight.view(3, total_num_heads, head_size, hidden_size)
+                    loaded_weight = _expand_mqa_mha(loaded_weight,
+                                                    n_head=total_num_heads,
+                                                    head_dim=head_size)
+                    loaded_weight = loaded_weight.view(3, total_num_heads,
+                                                       head_size, hidden_size)
                    loaded_weight = loaded_weight[:, head_start:head_end, :, :]
                    loaded_weight = loaded_weight.reshape(-1, hidden_size)
                elif name.endswith(".bias"):
-                    loaded_weight = _expand_mqa_mha(loaded_weight, n_head=total_num_heads, head_dim=head_size)
-                    loaded_weight = loaded_weight.view(3, total_num_heads, head_size)
+                    loaded_weight = _expand_mqa_mha(loaded_weight,
+                                                    n_head=total_num_heads,
+                                                    head_dim=head_size)
+                    loaded_weight = loaded_weight.view(3, total_num_heads,
+                                                       head_size)
                    loaded_weight = loaded_weight[:, head_start:head_end, :]
                    loaded_weight = loaded_weight.reshape(-1)
                else: