Spaces:

longlian
/

describe-anything

Running on Zero

App Files Files Community

longlian commited on Dec 2, 2024

Commit

c62168d

0 Parent(s):

Initial commit

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

.gitattributes +35 -0
README.md +12 -0
app.py +165 -0
dam/.DS_Store +0 -0
dam/__init__.py +2 -0
dam/__pycache__/__init__.cpython-310.pyc +0 -0
dam/__pycache__/describe_anything_model.cpython-310.pyc +0 -0
dam/describe_anything_model.py +212 -0
dam/model/__init__.py +4 -0
dam/model/__pycache__/__init__.cpython-310.pyc +0 -0
dam/model/__pycache__/configuration_llava.cpython-310.pyc +0 -0
dam/model/__pycache__/constants.cpython-310.pyc +0 -0
dam/model/__pycache__/conversation.cpython-310.pyc +0 -0
dam/model/__pycache__/llava_arch.cpython-310.pyc +0 -0
dam/model/__pycache__/mm_utils.cpython-310.pyc +0 -0
dam/model/__pycache__/model_utils.cpython-310.pyc +0 -0
dam/model/__pycache__/utils.cpython-310.pyc +0 -0
dam/model/builder_ignored.py +260 -0
dam/model/configuration_llava.py +55 -0
dam/model/consolidate.py +29 -0
dam/model/constants.py +32 -0
dam/model/conversation.py +474 -0
dam/model/language_model/__pycache__/builder.cpython-310.pyc +0 -0
dam/model/language_model/__pycache__/llava_llama.cpython-310.pyc +0 -0
dam/model/language_model/__pycache__/llava_mistral.cpython-310.pyc +0 -0
dam/model/language_model/builder.py +111 -0
dam/model/language_model/llava_gemma_ignored.py +161 -0
dam/model/language_model/llava_llama.py +180 -0
dam/model/language_model/llava_mistral_ignored.py +145 -0
dam/model/language_model/llava_mpt_ignored.py +115 -0
dam/model/language_model/mpt_ignored/adapt_tokenizer.py +41 -0
dam/model/language_model/mpt_ignored/attention.py +300 -0
dam/model/language_model/mpt_ignored/blocks.py +41 -0
dam/model/language_model/mpt_ignored/configuration_mpt.py +118 -0
dam/model/language_model/mpt_ignored/custom_embedding.py +11 -0
dam/model/language_model/mpt_ignored/flash_attn_triton.py +484 -0
dam/model/language_model/mpt_ignored/hf_prefixlm_converter.py +415 -0
dam/model/language_model/mpt_ignored/meta_init_context.py +94 -0
dam/model/language_model/mpt_ignored/modeling_mpt.py +331 -0
dam/model/language_model/mpt_ignored/norm.py +56 -0
dam/model/language_model/mpt_ignored/param_init_fns.py +181 -0
dam/model/llava_arch.py +676 -0
dam/model/mm_utils.py +312 -0
dam/model/model_utils.py +268 -0
dam/model/multimodal_encoder/__pycache__/builder.cpython-310.pyc +0 -0
dam/model/multimodal_encoder/__pycache__/context_provider.cpython-310.pyc +0 -0
dam/model/multimodal_encoder/__pycache__/siglip_encoder.cpython-310.pyc +0 -0
dam/model/multimodal_encoder/__pycache__/vision_encoder.cpython-310.pyc +0 -0
dam/model/multimodal_encoder/builder.py +54 -0
dam/model/multimodal_encoder/clip_encoder_ignored.py +19 -0

.gitattributes ADDED Viewed

	@@ -0,0 +1,35 @@

+*.7z filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+*.bz2 filter=lfs diff=lfs merge=lfs -text
+*.ckpt filter=lfs diff=lfs merge=lfs -text
+*.ftz filter=lfs diff=lfs merge=lfs -text
+*.gz filter=lfs diff=lfs merge=lfs -text
+*.h5 filter=lfs diff=lfs merge=lfs -text
+*.joblib filter=lfs diff=lfs merge=lfs -text
+*.lfs.* filter=lfs diff=lfs merge=lfs -text
+*.mlmodel filter=lfs diff=lfs merge=lfs -text
+*.model filter=lfs diff=lfs merge=lfs -text
+*.msgpack filter=lfs diff=lfs merge=lfs -text
+*.npy filter=lfs diff=lfs merge=lfs -text
+*.npz filter=lfs diff=lfs merge=lfs -text
+*.onnx filter=lfs diff=lfs merge=lfs -text
+*.ot filter=lfs diff=lfs merge=lfs -text
+*.parquet filter=lfs diff=lfs merge=lfs -text
+*.pb filter=lfs diff=lfs merge=lfs -text
+*.pickle filter=lfs diff=lfs merge=lfs -text
+*.pkl filter=lfs diff=lfs merge=lfs -text
+*.pt filter=lfs diff=lfs merge=lfs -text
+*.pth filter=lfs diff=lfs merge=lfs -text
+*.rar filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
+*.tar.* filter=lfs diff=lfs merge=lfs -text
+*.tar filter=lfs diff=lfs merge=lfs -text
+*.tflite filter=lfs diff=lfs merge=lfs -text
+*.tgz filter=lfs diff=lfs merge=lfs -text
+*.wasm filter=lfs diff=lfs merge=lfs -text
+*.xz filter=lfs diff=lfs merge=lfs -text
+*.zip filter=lfs diff=lfs merge=lfs -text
+*.zst filter=lfs diff=lfs merge=lfs -text
+*tfevents* filter=lfs diff=lfs merge=lfs -text

README.md ADDED Viewed

	@@ -0,0 +1,12 @@

+---
+title: Describe Anything
+emoji: ⚡
+colorFrom: yellow
+colorTo: purple
+sdk: gradio
+sdk_version: 5.7.1
+app_file: app.py
+pinned: false
+---
+Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference

app.py ADDED Viewed

	@@ -0,0 +1,165 @@

+import os
+os.environ["GRADIO_SSR_MODE"] = "false"
+if not os.path.exists("checkpoints"):
+    os.makedirs("checkpoints")
+    os.system("pip install gdown")
+    os.system("gdown https://drive.google.com/uc?id=1eQe6blJcyI7oy78C8ozwj1IUkbkFEItf; unzip -o dam_3b_v1.zip -d checkpoints")
+from segment_anything import sam_model_registry, SamPredictor
+import gradio as gr
+import numpy as np
+import cv2
+import base64
+import torch
+from PIL import Image
+import io
+import argparse
+from fastapi import FastAPI
+from fastapi.staticfiles import StaticFiles
+from transformers import SamModel, SamProcessor
+from dam import DescribeAnythingModel, disable_torch_init
+try:
+    from spaces import GPU
+except ImportError:
+    print("Spaces not installed, using dummy GPU decorator")
+    GPU = lambda fn: fn
+# Load SAM model
+device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
+sam_model = SamModel.from_pretrained("facebook/sam-vit-huge").to(device)
+sam_processor = SamProcessor.from_pretrained("facebook/sam-vit-huge")
+@GPU(duration=75)
+def image_to_sam_embedding(base64_image):
+    try:
+        # Decode base64 string to bytes
+        image_bytes = base64.b64decode(base64_image)
+        # Convert bytes to PIL Image
+        image = Image.open(io.BytesIO(image_bytes))
+        # Process image with SAM processor
+        inputs = sam_processor(image, return_tensors="pt").to(device)
+        # Get image embedding
+        with torch.no_grad():
+            image_embedding = sam_model.get_image_embeddings(inputs["pixel_values"])
+        # Convert to CPU and numpy
+        image_embedding = image_embedding.cpu().numpy()
+        # Encode the embedding as base64
+        embedding_bytes = image_embedding.tobytes()
+        embedding_base64 = base64.b64encode(embedding_bytes).decode('utf-8')
+        return embedding_base64
+    except Exception as e:
+        print(f"Error processing image: {str(e)}")
+        raise gr.Error(f"Failed to process image: {str(e)}")
+@GPU(duration=75)
+def describe(image_base64: str, mask_base64: str, query: str):
+    # Convert base64 to PIL Image
+    image_bytes = base64.b64decode(image_base64.split(',')[1] if ',' in image_base64 else image_base64)
+    img = Image.open(io.BytesIO(image_bytes))
+    mask_bytes = base64.b64decode(mask_base64.split(',')[1] if ',' in mask_base64 else mask_base64)
+    mask = Image.open(io.BytesIO(mask_bytes))
+    # Process the mask
+    mask = Image.fromarray((np.array(mask.convert('L')) > 0).astype(np.uint8) * 255)
+    # Get description using DAM with streaming
+    description_generator = dam.get_description(img, mask, query, streaming=True)
+    # Stream the tokens
+    text = ""
+    for token in description_generator:
+        text += token
+        yield text
+@GPU(duration=75)
+def describe_without_streaming(image_base64: str, mask_base64: str, query: str):
+    # Convert base64 to PIL Image
+    image_bytes = base64.b64decode(image_base64.split(',')[1] if ',' in image_base64 else image_base64)
+    img = Image.open(io.BytesIO(image_bytes))
+    mask_bytes = base64.b64decode(mask_base64.split(',')[1] if ',' in mask_base64 else mask_base64)
+    mask = Image.open(io.BytesIO(mask_bytes))
+    # Process the mask
+    mask = Image.fromarray((np.array(mask.convert('L')) > 0).astype(np.uint8) * 255)
+    # Get description using DAM
+    description = dam.get_description(img, mask, query)
+    return description
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser(description="Describe Anything gradio demo")
+    parser.add_argument("--model-path", type=str, default="checkpoints/dam_3b_v1", help="Path to the model checkpoint")
+    parser.add_argument("--prompt-mode", type=str, default="full+focal_crop", help="Prompt mode")
+    parser.add_argument("--conv-mode", type=str, default="v1", help="Conversation mode")
+    parser.add_argument("--temperature", type=float, default=0.2, help="Sampling temperature")
+    parser.add_argument("--top_p", type=float, default=0.5, help="Top-p for sampling")
+    args = parser.parse_args()
+    # Initialize DAM model
+    disable_torch_init()
+    dam = DescribeAnythingModel(
+        model_path=args.model_path,
+        conv_mode=args.conv_mode,
+        prompt_mode=args.prompt_mode,
+        temperature=args.temperature,
+        top_p=args.top_p,
+        num_beams=1,
+        max_new_tokens=512,
+    ).to(device)
+    # Create Gradio interface
+    with gr.Blocks() as demo:
+        gr.Interface(
+            fn=image_to_sam_embedding,
+            inputs=gr.Textbox(label="Image Base64"),
+            outputs=gr.Textbox(label="Embedding Base64"),
+            title="Image Embedding Generator",
+            api_name="image_to_sam_embedding"
+        )
+        gr.Interface(
+            fn=describe,
+            inputs=[
+                gr.Textbox(label="Image Base64"),
+                gr.Text(label="Mask Base64"),
+                gr.Text(label="Prompt")
+            ],
+            outputs=[
+                gr.Text(label="Description")
+            ],
+            title="Mask Description Generator",
+            api_name="describe"
+        )
+        gr.Interface(
+            fn=describe_without_streaming,
+            inputs=[
+                gr.Textbox(label="Image Base64"),
+                gr.Text(label="Mask Base64"),
+                gr.Text(label="Prompt")
+            ],
+            outputs=[
+                gr.Text(label="Description")
+            ],
+            title="Mask Description Generator (Non-Streaming)",
+            api_name="describe_without_streaming"
+        )
+    demo._block_thread = demo.block_thread
+    demo.block_thread = lambda: None
+    demo.launch()
+    for route in demo.app.routes:
+        if route.path == "/":
+            # demo.app.routes.remove(route)
+            route.path = "/gradio"
+    demo.app.mount("/", StaticFiles(directory="dist", html=True), name="demo")
+    demo._block_thread()

dam/.DS_Store ADDED Viewed

Binary file (6.15 kB). View file

dam/__init__.py ADDED Viewed

	@@ -0,0 +1,2 @@


1	+ from .describe_anything_model import *
2	+ from .model import *

dam/__pycache__/__init__.cpython-310.pyc ADDED Viewed

Binary file (203 Bytes). View file

dam/__pycache__/describe_anything_model.cpython-310.pyc ADDED Viewed

Binary file (7.27 kB). View file

dam/describe_anything_model.py ADDED Viewed

	@@ -0,0 +1,212 @@

+import torch
+import torch.nn as nn
+import numpy as np
+from PIL import Image
+from .model.constants import DEFAULT_IMAGE_TOKEN, IMAGE_TOKEN_INDEX
+from .model.conversation import SeparatorStyle, conv_templates
+from .model.mm_utils import KeywordsStoppingCriteria, process_image, tokenizer_image_token
+from .model import get_model_name_from_path, load_pretrained_model
+from transformers import TextIteratorStreamer
+from threading import Thread
+class DescribeAnythingModel(nn.Module):
+    def __init__(self, model_path, conv_mode, prompt_mode, temperature, top_p, num_beams, max_new_tokens, **kwargs):
+        super().__init__()
+        self.model_path = model_path
+        self.conv_mode = conv_mode
+        self.prompt_mode = prompt_mode
+        self.temperature = temperature
+        self.top_p = top_p
+        self.num_beams = num_beams
+        self.max_new_tokens = max_new_tokens
+        tokenizer, model, image_processor, context_len = load_pretrained_model(model_path, None, None, **kwargs)
+        model.config.image_processor = image_processor
+        self.tokenizer = tokenizer
+        self.model = model
+        self.context_len = context_len
+        self.model_name = get_model_name_from_path(model_path)
+    def get_prompt(self, qs):
+        if DEFAULT_IMAGE_TOKEN not in qs:
+            raise ValueError("no <image> tag found in input.")
+        conv = conv_templates[self.conv_mode].copy()
+        conv.append_message(conv.roles[0], qs)
+        conv.append_message(conv.roles[1], None)
+        prompt = conv.get_prompt()
+        return prompt, conv
+    @staticmethod
+    def mask_to_box(mask_np):
+        mask_coords = np.argwhere(mask_np)
+        y0, x0 = mask_coords.min(axis=0)
+        y1, x1 = mask_coords.max(axis=0) + 1
+        h = y1 - y0
+        w = x1 - x0
+        return x0, y0, w, h
+    @classmethod
+    def crop_image(cls, pil_img, mask_np, crop_mode, min_box_w=48, min_box_h=48):
+        if crop_mode == "full":
+            # no crop
+            info = dict(mask_np=mask_np)
+            return pil_img, info
+        if crop_mode == "crop":
+            # crop image and mask
+            x0, y0, w, h = cls.mask_to_box(mask_np)
+            img_np = np.asarray(pil_img)
+            assert img_np.shape[:2] == mask_np.shape, f"image shape mismatches with mask shape: {img_np.shape}, {mask_np.shape}"
+            cropped_mask_np = mask_np[y0:y0+h, x0:x0+w]
+            cropped_img_np = img_np[y0:y0+h, x0:x0+w]
+            cropped_pil_img = Image.fromarray(cropped_img_np)
+        elif crop_mode == "context_crop":
+            # crop image and mask
+            x0, y0, w, h = cls.mask_to_box(mask_np)
+            img_np = np.asarray(pil_img)
+            assert img_np.shape[:2] == mask_np.shape, f"image shape mismatches with mask shape: {img_np.shape}, {mask_np.shape}"
+            img_h, img_w = img_np.shape[:2]
+            cropped_mask_np = mask_np[max(y0-h, 0):min(y0+2*h, img_h), max(x0-w, 0):min(x0+2*w, img_w)]
+            cropped_img_np = img_np[max(y0-h, 0):min(y0+2*h, img_h), max(x0-w, 0):min(x0+2*w, img_w)]
+            cropped_pil_img = Image.fromarray(cropped_img_np)
+        elif crop_mode == "focal_crop":
+            # crop image and mask
+            x0, y0, w, h = cls.mask_to_box(mask_np)
+            img_np = np.asarray(pil_img)
+            assert img_np.shape[:2] == mask_np.shape, f"image shape mismatches with mask shape: {img_np.shape}, {mask_np.shape}"
+            img_h, img_w = img_np.shape[:2]
+            xc, yc = x0 + w/2, y0 + h/2
+            # focal_crop: need to have at least min_box_w and min_box_h pixels, otherwise resizing to (384, 384) leads to artifacts that may be OOD
+            w, h = max(w, min_box_w), max(h, min_box_h)
+            x0, y0 = int(xc - w / 2), int(yc - h / 2)
+            cropped_mask_np = mask_np[max(y0-h, 0):min(y0+2*h, img_h), max(x0-w, 0):min(x0+2*w, img_w)]
+            cropped_img_np = img_np[max(y0-h, 0):min(y0+2*h, img_h), max(x0-w, 0):min(x0+2*w, img_w)]
+            cropped_pil_img = Image.fromarray(cropped_img_np)
+        elif crop_mode == "crop_mask":
+            # crop image and mask
+            x0, y0, w, h = cls.mask_to_box(mask_np)
+            img_np = np.asarray(pil_img)
+            assert img_np.shape[:2] == mask_np.shape, f"image shape mismatches with mask shape: {img_np.shape}, {mask_np.shape}"
+            cropped_mask_np = mask_np[y0:y0+h, x0:x0+w]
+            cropped_img_np = img_np[y0:y0+h, x0:x0+w]
+            # Mask the image
+            cropped_img_np = cropped_img_np * cropped_mask_np[..., None]
+            cropped_pil_img = Image.fromarray(cropped_img_np)
+        else:
+            raise ValueError(f"Unsupported crop_mode: {crop_mode}")
+        info = dict(mask_np=cropped_mask_np)
+        return cropped_pil_img, info
+    def get_description(self, image_pil, mask_pil, query, streaming=False):
+        prompt, conv = self.get_prompt(query)
+        if not isinstance(image_pil, (list, tuple)):
+            assert not isinstance(mask_pil, (list, tuple)), "image_pil and mask_pil must be both list or tuple or not list or tuple."
+            image_pils = [image_pil]
+            mask_pils = [mask_pil]
+        else:
+            image_pils = image_pil
+            mask_pils = mask_pil
+        description = self.get_description_from_prompt(image_pils, mask_pils, prompt, conv, streaming=streaming)
+        return description
+    def get_image_tensor(self, image_pil, mask_pil, crop_mode, crop_mode2):
+        # the pil has True/False (if the value is non-zero, then we treat it as True)
+        mask_np = (np.asarray(mask_pil) > 0).astype(np.uint8)
+        images_tensor, image_info = process_image(image_pil, self.model.config, None, pil_preprocess_fn=lambda pil_img: self.crop_image(image_pil, mask_np=mask_np, crop_mode=crop_mode))
+        images_tensor = images_tensor[None].to(self.model.device, dtype=torch.float16)
+        mask_np = image_info["mask_np"]
+        mask_pil = Image.fromarray(mask_np * 255)
+        masks_tensor = process_image(mask_pil, self.model.config, None)
+        masks_tensor = masks_tensor[None].to(self.model.device, dtype=torch.float16)
+        images_tensor = torch.cat((images_tensor, masks_tensor[:, :1, ...]), dim=1)
+        if crop_mode2 is not None:
+            images_tensor2, image_info2 = process_image(image_pil, self.model.config, None, pil_preprocess_fn=lambda pil_img: self.crop_image(pil_img, mask_np=mask_np, crop_mode=crop_mode2))
+            images_tensor2 = images_tensor2[None].to(self.model.device, dtype=torch.float16)
+            mask_np2 = image_info2["mask_np"]
+            mask_pil2 = Image.fromarray(mask_np2 * 255)
+            masks_tensor2 = process_image(mask_pil2, self.model.config, None)
+            masks_tensor2 = masks_tensor2[None].to(self.model.device, dtype=torch.float16)
+            images_tensor2 = torch.cat((images_tensor2, masks_tensor2[:, :1, ...]), dim=1)
+        else:
+            images_tensor2 = None
+        return torch.cat((images_tensor, images_tensor2), dim=1) if images_tensor2 is not None else images_tensor
+    def get_description_from_prompt(self, image_pils, mask_pils, prompt, conv, streaming=False):
+        if streaming:
+            return self.get_description_from_prompt_iterator(image_pils, mask_pils, prompt, conv, streaming=True)
+        else:
+            # If streaming is False, there will be only one output
+            output = self.get_description_from_prompt_iterator(image_pils, mask_pils, prompt, conv, streaming=False)
+            return next(output)
+    def get_description_from_prompt_iterator(self, image_pils, mask_pils, prompt, conv, streaming=False):
+        crop_mode, crop_mode2 = self.prompt_mode.split("+")
+        assert crop_mode == "full", "Current prompt only supports first crop as full (non-cropped). If you need other specifications, please update the prompt."
+        assert len(image_pils) == len(mask_pils), f"image_pils and mask_pils must have the same length. Got {len(image_pils)} and {len(mask_pils)}."
+        image_tensors = [self.get_image_tensor(image_pil, mask_pil, crop_mode=crop_mode, crop_mode2=crop_mode2) for image_pil, mask_pil in zip(image_pils, mask_pils)]
+        input_ids = tokenizer_image_token(prompt, self.tokenizer, IMAGE_TOKEN_INDEX, return_tensors="pt").unsqueeze(0).cuda()
+        stop_str = conv.sep if conv.sep_style != SeparatorStyle.TWO else conv.sep2
+        keywords = [stop_str]
+        stopping_criteria = KeywordsStoppingCriteria(keywords, self.tokenizer, input_ids)
+        streamer = TextIteratorStreamer(self.tokenizer, skip_prompt=True, skip_special_tokens=True) if streaming else None
+        generation_kwargs = dict(
+            input_ids=input_ids,
+            images=image_tensors,
+            do_sample=True if self.temperature > 0 else False,
+            temperature=self.temperature,
+            top_p=self.top_p,
+            num_beams=self.num_beams,
+            max_new_tokens=self.max_new_tokens,
+            use_cache=True,
+            stopping_criteria=[stopping_criteria],
+            streamer=streamer
+        )
+        if streaming:
+            thread = Thread(target=self.model.generate, kwargs=generation_kwargs)
+            thread.start()
+            generated_text = ""
+            for new_text in streamer:
+                generated_text += new_text
+                if stop_str in generated_text:
+                    generated_text = generated_text[:generated_text.find(stop_str)]
+                    break
+                yield new_text
+            thread.join()
+        else:
+            with torch.inference_mode():
+                output_ids = self.model.generate(**generation_kwargs)
+            outputs = self.tokenizer.batch_decode(output_ids, skip_special_tokens=True)[0]
+            outputs = outputs.strip()
+            if outputs.endswith(stop_str):
+                outputs = outputs[: -len(stop_str)]
+            outputs = outputs.strip()
+            yield outputs

dam/model/__init__.py ADDED Viewed

	@@ -0,0 +1,4 @@

+from .constants import *
+from .conversation import *
+from .mm_utils import *
+from .model_utils import *

dam/model/__pycache__/__init__.cpython-310.pyc ADDED Viewed

Binary file (241 Bytes). View file

dam/model/__pycache__/configuration_llava.cpython-310.pyc ADDED Viewed

Binary file (1.33 kB). View file

dam/model/__pycache__/constants.cpython-310.pyc ADDED Viewed

Binary file (536 Bytes). View file

dam/model/__pycache__/conversation.cpython-310.pyc ADDED Viewed

Binary file (11.5 kB). View file

dam/model/__pycache__/llava_arch.cpython-310.pyc ADDED Viewed

Binary file (16.2 kB). View file

dam/model/__pycache__/mm_utils.cpython-310.pyc ADDED Viewed

Binary file (8.51 kB). View file

dam/model/__pycache__/model_utils.cpython-310.pyc ADDED Viewed

Binary file (6.34 kB). View file

dam/model/__pycache__/utils.cpython-310.pyc ADDED Viewed

Binary file (2.53 kB). View file

dam/model/builder_ignored.py ADDED Viewed

	@@ -0,0 +1,260 @@

+# This file is modified from https://github.com/haotian-liu/LLaVA/
+#    Copyright 2023 Haotian Liu
+#
+#    Licensed under the Apache License, Version 2.0 (the "License");
+#    you may not use this file except in compliance with the License.
+#    You may obtain a copy of the License at
+#
+#        http://www.apache.org/licenses/LICENSE-2.0
+#
+#    Unless required by applicable law or agreed to in writing, software
+#    distributed under the License is distributed on an "AS IS" BASIS,
+#    WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#    See the License for the specific language governing permissions and
+#    limitations under the License.
+import os
+import warnings
+import shutil
+from transformers import (
+    AutoTokenizer,
+    AutoModelForCausalLM,
+    AutoConfig,
+    BitsAndBytesConfig,
+    PretrainedConfig,
+    PreTrainedModel,
+)
+import torch
+from llava.model import *
+from llava.model.utils import is_mm_model
+from llava.model.language_model.llava_llama import LlavaConfig
+from llava.constants import (
+    DEFAULT_IMAGE_PATCH_TOKEN,
+    DEFAULT_IM_START_TOKEN,
+    DEFAULT_IM_END_TOKEN,
+)
+def load_pretrained_model(
+    model_path,
+    model_name,
+    model_base=None,
+    load_8bit=False,
+    load_4bit=False,
+    device_map="auto",
+    device="cuda",
+    **kwargs,
+):
+    kwargs = {"device_map": device_map, **kwargs}
+    if device != "cuda":
+        kwargs["device_map"] = {"": device}
+    if load_8bit:
+        kwargs["load_in_8bit"] = True
+    elif load_4bit:
+        kwargs["load_in_4bit"] = True
+        kwargs["quantization_config"] = BitsAndBytesConfig(
+            load_in_4bit=True,
+            bnb_4bit_compute_dtype=torch.float16,
+            bnb_4bit_use_double_quant=True,
+            bnb_4bit_quant_type="nf4",
+        )
+    else:
+        kwargs["torch_dtype"] = torch.float16
+    if is_mm_model(model_path):
+        # Load LLaVA model
+        ## TODO @yunhao: mind fixing lora
+        if "lora" in model_name.lower() and model_base is None:
+            warnings.warn(
+                "There is `lora` in model name but no `model_base` is provided. If you are loading a LoRA model, please provide the `model_base` argument. Detailed instruction: https://github.com/haotian-liu/LLaVA#launch-a-model-worker-lora-weights-unmerged."
+            )
+        if "lora" in model_name.lower() and model_base is not None:
+            lora_cfg_pretrained = AutoConfig.from_pretrained(model_path)
+            tokenizer = AutoTokenizer.from_pretrained(
+                model_base, use_fast=False, legacy=False
+            )
+            print("Loading LLaVA from base model...")
+            model = LlavaLlamaForCausalLM.from_pretrained(
+                model_base, low_cpu_mem_usage=True, config=lora_cfg_pretrained, **kwargs
+            )
+            token_num, tokem_dim = model.lm_head.out_features, model.lm_head.in_features
+            if model.lm_head.weight.shape[0] != token_num:
+                model.lm_head.weight = torch.nn.Parameter(
+                    torch.empty(
+                        token_num, tokem_dim, device=model.device, dtype=model.dtype
+                    )
+                )
+                model.model.embed_tokens.weight = torch.nn.Parameter(
+                    torch.empty(
+                        token_num, tokem_dim, device=model.device, dtype=model.dtype
+                    )
+                )
+            print("Loading additional LLaVA weights...")
+            if os.path.exists(os.path.join(model_path, "non_lora_trainables.bin")):
+                non_lora_trainables = torch.load(
+                    os.path.join(model_path, "non_lora_trainables.bin"),
+                    map_location="cpu",
+                )
+            else:
+                # this is probably from HF Hub
+                from huggingface_hub import hf_hub_download
+                def load_from_hf(repo_id, filename, subfolder=None):
+                    cache_file = hf_hub_download(
+                        repo_id=repo_id, filename=filename, subfolder=subfolder
+                    )
+                    return torch.load(cache_file, map_location="cpu")
+                non_lora_trainables = load_from_hf(
+                    model_path, "non_lora_trainables.bin"
+                )
+            non_lora_trainables = {
+                (k[11:] if k.startswith("base_model.") else k): v
+                for k, v in non_lora_trainables.items()
+            }
+            if any(k.startswith("model.model.") for k in non_lora_trainables):
+                non_lora_trainables = {
+                    (k[6:] if k.startswith("model.") else k): v
+                    for k, v in non_lora_trainables.items()
+                }
+            model.load_state_dict(non_lora_trainables, strict=False)
+            from peft import PeftModel
+            print("Loading LoRA weights...")
+            model = PeftModel.from_pretrained(model, model_path)
+            print("Merging LoRA weights...")
+            model = model.merge_and_unload()
+            print("Model is loaded...")
+        ## TODO @yunhao: mind fixing this
+        elif model_base is not None:
+            # this may be mm projector only
+            print("Loading LLaVA from base model...")
+            cfg_pretrained = AutoConfig.from_pretrained(
+                model_path, trust_remote_code=True
+            )
+            mm_config_wrapper(config, kwargs)
+            if "mpt" in model_name.lower():
+                if not os.path.isfile(os.path.join(model_path, "configuration_mpt.py")):
+                    shutil.copyfile(
+                        os.path.join(model_base, "configuration_mpt.py"),
+                        os.path.join(model_path, "configuration_mpt.py"),
+                    )
+                tokenizer = AutoTokenizer.from_pretrained(model_base, use_fast=True)
+                model = LlavaMPTForCausalLM.from_pretrained(
+                    model_base, low_cpu_mem_usage=True, config=cfg_pretrained, **kwargs
+                )
+            else:
+                tokenizer = AutoTokenizer.from_pretrained(
+                    model_base, use_fast=False, legacy=False
+                )
+                model = LlavaLlamaForCausalLM.from_pretrained(
+                    model_base, low_cpu_mem_usage=True, config=cfg_pretrained, **kwargs
+                )
+        else:
+            config = AutoConfig.from_pretrained(model_path)
+            config.resume_path = model_path
+            prepare_config_for_eval(config, kwargs)
+            if "mpt" in model_name.lower():
+                model = LlavaMPTForCausalLM.from_pretrained(
+                    model_path, config=config, low_cpu_mem_usage=True, **kwargs
+                )
+            elif "mistral" in model_name.lower() or "mixtral" in model_name.lower():
+                model = LlavaMistralForCausalLM.from_pretrained(
+                    model_path, config=config, low_cpu_mem_usage=True, **kwargs
+                )
+            elif "gemma" in model_name.lower():
+                model = LlavaGemmaForCausalLM.from_pretrained(
+                    model_path, config=config, low_cpu_mem_usage=True, **kwargs
+                )
+            else:
+                # kentang-mit@: llama-2 model
+                # config._attn_implementation = "flash_attention_2"
+                model = LlavaLlamaModel(
+                    config=config,
+                    low_cpu_mem_usage=True,
+                    **kwargs
+                )
+            tokenizer = model.tokenizer
+    else:
+        # Load language model
+        if model_base is not None:
+            # PEFT model
+            from peft import PeftModel
+            tokenizer = AutoTokenizer.from_pretrained(model_base, use_fast=False)
+            model = AutoModelForCausalLM.from_pretrained(
+                model_base, low_cpu_mem_usage=True, **kwargs
+            )
+            print(f"Loading LoRA weights from {model_path}")
+            model = PeftModel.from_pretrained(model, model_path)
+            print(f"Merging weights")
+            model = model.merge_and_unload()
+            print("Convert to FP16...")
+            model.to(torch.float16)
+        else:
+            if "mpt" in model_name.lower():
+                tokenizer = AutoTokenizer.from_pretrained(model_path, use_fast=True)
+                model = AutoModelForCausalLM.from_pretrained(
+                    model_path, low_cpu_mem_usage=True, trust_remote_code=True, **kwargs
+                )
+            else:
+                tokenizer = AutoTokenizer.from_pretrained(
+                    model_path, use_fast=False, legacy=False
+                )
+                model = AutoModelForCausalLM.from_pretrained(
+                    model_path, low_cpu_mem_usage=True, **kwargs
+                )
+    model.eval()
+    image_processor = None
+    if is_mm_model(model_path):
+        mm_use_im_start_end = getattr(model.config, "mm_use_im_start_end", False)
+        mm_use_im_patch_token = getattr(model.config, "mm_use_im_patch_token", True)
+        if mm_use_im_patch_token:
+            tokenizer.add_tokens([DEFAULT_IMAGE_PATCH_TOKEN], special_tokens=True)
+        if mm_use_im_start_end:
+            tokenizer.add_tokens(
+                [DEFAULT_IM_START_TOKEN, DEFAULT_IM_END_TOKEN], special_tokens=True
+            )
+        model.resize_token_embeddings(len(tokenizer))
+        vision_tower = model.get_vision_tower()
+        vision_tower.to(device=device, dtype=torch.float16)
+        mm_projector = model.get_mm_projector()
+        mm_projector.to(device=device, dtype=torch.float16)
+        image_processor = vision_tower.image_processor
+    if hasattr(model.llm.config, "max_sequence_length"):
+        context_len = model.config.max_sequence_length
+    else:
+        context_len = 2048
+    return tokenizer, model, image_processor, context_len
+def parse_model_name_or_path(config: PretrainedConfig, model_name="llm", suffix="_cfg"):
+    target_model = f"{model_name}{suffix}"
+    target_cfg = getattr(config, target_model, None)
+    if isinstance(target_cfg, str):
+        return target_cfg
+    elif isinstance(target_cfg, dict):
+        return target_cfg["architectures"][0]
+    else:
+        raise ValueError(f"Invalid {target_model} configuration!")
+def prepare_config_for_eval(config: PretrainedConfig, kwargs: dict):
+    try:
+        # compatible with deprecated config convention
+        if getattr(config, "vision_tower_cfg", None) is None:
+            config.vision_tower_cfg = config.mm_vision_tower
+    except AttributeError:
+        raise ValueError(f"Invalid configuration! Cannot find vision_tower in config:\n{config}")
+    config.model_dtype = kwargs.pop("torch_dtype").__str__()
+    # siglip does not support device_map = "auto"
+    vision_tower_name = parse_model_name_or_path(config, "vision_tower")
+    if "siglip" in vision_tower_name.lower():
+        kwargs["device_map"] = "cuda"

dam/model/configuration_llava.py ADDED Viewed

	@@ -0,0 +1,55 @@

+from transformers import PretrainedConfig
+class LlavaConfig(PretrainedConfig):
+    model_type = "llava"
+    def __init__(
+        self,
+        llm_cfg=None,
+        vision_tower_cfg=None,
+        mm_projector_cfg=None,
+        mask_encoder_cfg=None,
+        context_provider_cfg=None,
+        architectures=None,
+        resume_path=None,
+        hidden_size=None,
+        mm_hidden_size=None,
+        image_aspect_ratio=None,
+        num_video_frames=None,
+        mm_vision_select_layer=None,
+        mm_vision_select_feature=None,
+        mm_use_im_start_end=False,
+        mm_use_im_patch_token=True,
+        mm_projector_lr=None,
+        vision_resolution=None,
+        interpolate_mode=None,
+        s2=None,
+        s2_scales=None,
+        s2_max_split_size=None,
+        **kwargs
+    ):
+        super().__init__()
+        self.architectures = architectures
+        self.llm_cfg = llm_cfg
+        self.vision_tower_cfg = vision_tower_cfg
+        self.mm_projector_cfg = mm_projector_cfg
+        self.mask_encoder_cfg = mask_encoder_cfg
+        self.context_provider_cfg = context_provider_cfg
+        self.resume_path = resume_path
+        self.hidden_size = hidden_size
+        self.mm_hidden_size = mm_hidden_size
+        self.image_aspect_ratio = image_aspect_ratio
+        self.num_video_frames = num_video_frames
+        self.mm_vision_select_layer = mm_vision_select_layer
+        self.mm_vision_select_feature = mm_vision_select_feature
+        self.mm_use_im_start_end = mm_use_im_start_end
+        self.mm_use_im_start_end = mm_use_im_start_end
+        self.mm_use_im_patch_token = mm_use_im_patch_token
+        self.mm_projector_lr = mm_projector_lr
+        self.vision_resolution = vision_resolution
+        self.interpolate_mode = interpolate_mode
+        self.s2 = s2
+        self.s2_scales = s2_scales
+        self.s2_max_split_size = s2_max_split_size

dam/model/consolidate.py ADDED Viewed

	@@ -0,0 +1,29 @@

+"""
+Usage:
+python3 -m llava.model.consolidate --src ~/model_weights/llava-7b --dst ~/model_weights/llava-7b_consolidate
+"""
+import argparse
+import torch
+from transformers import AutoTokenizer, AutoModelForCausalLM
+from llava.model import *
+from llava.model.utils import auto_upgrade
+def consolidate_ckpt(src_path, dst_path):
+    print("Loading model")
+    auto_upgrade(src_path)
+    src_model = AutoModelForCausalLM.from_pretrained(src_path, torch_dtype=torch.float16, low_cpu_mem_usage=True)
+    src_tokenizer = AutoTokenizer.from_pretrained(src_path, use_fast=False)
+    src_model.save_pretrained(dst_path)
+    src_tokenizer.save_pretrained(dst_path)
+if __name__ == "__main__":
+    parser = argparse.ArgumentParser()
+    parser.add_argument("--src", type=str, required=True)
+    parser.add_argument("--dst", type=str, required=True)
+    args = parser.parse_args()
+    consolidate_ckpt(args.src, args.dst)

dam/model/constants.py ADDED Viewed

	@@ -0,0 +1,32 @@

+# Copyright 2024 NVIDIA CORPORATION & AFFILIATES
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+# SPDX-License-Identifier: Apache-2.0
+# This file is modified from https://github.com/haotian-liu/LLaVA/
+CONTROLLER_HEART_BEAT_EXPIRATION = 30
+WORKER_HEART_BEAT_INTERVAL = 15
+LOGDIR = "."
+# Model Constants
+IGNORE_INDEX = -100
+IMAGE_TOKEN_INDEX = -200
+MASK_TOKEN_INDEX = -300
+DEFAULT_IMAGE_TOKEN = "<image>"
+DEFAULT_IMAGE_PATCH_TOKEN = "<im_patch>"
+DEFAULT_IM_START_TOKEN = "<im_start>"
+DEFAULT_IM_END_TOKEN = "<im_end>"
+IMAGE_PLACEHOLDER = "<image-placeholder>"

dam/model/conversation.py ADDED Viewed

	@@ -0,0 +1,474 @@

+# Copyright 2024 NVIDIA CORPORATION & AFFILIATES
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+# SPDX-License-Identifier: Apache-2.0
+# This file is modified from https://github.com/haotian-liu/LLaVA/
+import dataclasses
+from enum import auto, Enum
+from typing import List, Tuple
+class SeparatorStyle(Enum):
+    """Different separator style."""
+    SINGLE = auto()
+    TWO = auto()
+    MPT = auto()
+    PLAIN = auto()
+    LLAMA_2 = auto()
+    MISTRAL = auto()
+    LLAMA_3 = auto()
+@dataclasses.dataclass
+class Conversation:
+    """A class that keeps all conversation history."""
+    system: str
+    roles: List[str]
+    messages: List[List[str]]
+    offset: int
+    sep_style: SeparatorStyle = SeparatorStyle.SINGLE
+    sep: str = "###"
+    sep2: str = None
+    version: str = "Unknown"
+    skip_next: bool = False
+    def get_prompt(self):
+        messages = self.messages
+        if len(messages) > 0 and type(messages[0][1]) is tuple:
+            messages = self.messages.copy()
+            init_role, init_msg = messages[0].copy()
+            init_msg = init_msg[0].replace("<image>", "").strip()
+            if 'mmtag' in self.version:
+                messages[0] = (init_role, init_msg)
+                messages.insert(0, (self.roles[0], "<Image><image></Image>"))
+                messages.insert(1, (self.roles[1], "Received."))
+            else:
+                messages[0] = (init_role, "<image>\n" + init_msg)
+        if self.sep_style == SeparatorStyle.SINGLE:
+            ret = self.system + self.sep
+            for role, message in messages:
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    ret += role + ": " + message + self.sep
+                else:
+                    ret += role + ":"
+        elif self.sep_style == SeparatorStyle.TWO:
+            seps = [self.sep, self.sep2]
+            ret = self.system + seps[0]
+            for i, (role, message) in enumerate(messages):
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    ret += role + ": " + message + seps[i % 2]
+                else:
+                    ret += role + ":"
+        elif self.sep_style == SeparatorStyle.LLAMA_3:
+            ret = self.system + self.sep
+            for role, message in messages:
+                if message:
+                    if type(message) is tuple:
+                        message = message[0]
+                    ret += role + message + self.sep
+                else:
+                    ret += role
+        elif self.sep_style == SeparatorStyle.MPT:
+            ret = self.system + self.sep
+            for role, message in messages:
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    ret += role + message + self.sep
+                else:
+                    ret += role
+        elif self.sep_style == SeparatorStyle.LLAMA_2 or self.sep_style == SeparatorStyle.MISTRAL:
+            if self.sep_style == SeparatorStyle.LLAMA_2:
+                wrap_sys = lambda msg: f"<<SYS>>\n{msg}\n<</SYS>>\n\n"
+            else:
+                wrap_sys = lambda msg: f"{msg}" + ("\n" if msg else "")
+            wrap_inst = lambda msg: f"[INST] {msg} [/INST]"
+            ret = ""
+            if self.sep_style == SeparatorStyle.MISTRAL:
+                ret += "<s>"
+            for i, (role, message) in enumerate(messages):
+                if i == 0:
+                    assert message, "first message should not be none"
+                    assert role == self.roles[0], "first message should come from user"
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    if i == 0: message = wrap_sys(self.system) + message
+                    if i % 2 == 0:
+                        message = wrap_inst(message)
+                        ret += self.sep + message
+                    else:
+                        if self.sep_style == SeparatorStyle.LLAMA_2:
+                            ret += " " + message + " " + self.sep2
+                        else:
+                            ret += message + self.sep2
+                else:
+                    ret += ""
+            ret = ret.lstrip(self.sep)
+        elif self.sep_style == SeparatorStyle.PLAIN:
+            seps = [self.sep, self.sep2]
+            ret = self.system
+            for i, (role, message) in enumerate(messages):
+                if message:
+                    if type(message) is tuple:
+                        message, _, _ = message
+                    ret += message + seps[i % 2]
+                else:
+                    ret += ""
+        else:
+            raise ValueError(f"Invalid style: {self.sep_style}")
+        return ret
+    def append_message(self, role, message):
+        self.messages.append([role, message])
+    def get_images(self, return_pil=False):
+        images = []
+        for i, (role, msg) in enumerate(self.messages[self.offset:]):
+            if i % 2 == 0:
+                if type(msg) is tuple:
+                    import base64
+                    from io import BytesIO
+                    from PIL import Image
+                    msg, image, image_process_mode = msg
+                    if image_process_mode == "Pad":
+                        def expand2square(pil_img, background_color=(122, 116, 104)):
+                            width, height = pil_img.size
+                            if width == height:
+                                return pil_img
+                            elif width > height:
+                                result = Image.new(pil_img.mode, (width, width), background_color)
+                                result.paste(pil_img, (0, (width - height) // 2))
+                                return result
+                            else:
+                                result = Image.new(pil_img.mode, (height, height), background_color)
+                                result.paste(pil_img, ((height - width) // 2, 0))
+                                return result
+                        image = expand2square(image)
+                    elif image_process_mode in ["Default", "Crop"]:
+                        pass
+                    elif image_process_mode == "Resize":
+                        image = image.resize((336, 336))
+                    else:
+                        raise ValueError(f"Invalid image_process_mode: {image_process_mode}")
+                    max_hw, min_hw = max(image.size), min(image.size)
+                    aspect_ratio = max_hw / min_hw
+                    max_len, min_len = 800, 400
+                    shortest_edge = int(min(max_len / aspect_ratio, min_len, min_hw))
+                    longest_edge = int(shortest_edge * aspect_ratio)
+                    W, H = image.size
+                    if longest_edge != max(image.size):
+                        if H > W:
+                            H, W = longest_edge, shortest_edge
+                        else:
+                            H, W = shortest_edge, longest_edge
+                        image = image.resize((W, H))
+                    if return_pil:
+                        images.append(image)
+                    else:
+                        buffered = BytesIO()
+                        image.save(buffered, format="PNG")
+                        img_b64_str = base64.b64encode(buffered.getvalue()).decode()
+                        images.append(img_b64_str)
+        return images
+    def to_gradio_chatbot(self):
+        ret = []
+        for i, (role, msg) in enumerate(self.messages[self.offset:]):
+            if i % 2 == 0:
+                if type(msg) is tuple:
+                    import base64
+                    from io import BytesIO
+                    msg, image, image_process_mode = msg
+                    max_hw, min_hw = max(image.size), min(image.size)
+                    aspect_ratio = max_hw / min_hw
+                    max_len, min_len = 800, 400
+                    shortest_edge = int(min(max_len / aspect_ratio, min_len, min_hw))
+                    longest_edge = int(shortest_edge * aspect_ratio)
+                    W, H = image.size
+                    if H > W:
+                        H, W = longest_edge, shortest_edge
+                    else:
+                        H, W = shortest_edge, longest_edge
+                    image = image.resize((W, H))
+                    buffered = BytesIO()
+                    image.save(buffered, format="JPEG")
+                    img_b64_str = base64.b64encode(buffered.getvalue()).decode()
+                    img_str = f'<img src="data:image/png;base64,{img_b64_str}" alt="user upload image" />'
+                    msg = img_str + msg.replace('<image>', '').strip()
+                    ret.append([msg, None])
+                else:
+                    ret.append([msg, None])
+            else:
+                ret[-1][-1] = msg
+        return ret
+    def copy(self):
+        return Conversation(
+            system=self.system,
+            roles=self.roles,
+            messages=[[x, y] for x, y in self.messages],
+            offset=self.offset,
+            sep_style=self.sep_style,
+            sep=self.sep,
+            sep2=self.sep2,
+            version=self.version)
+    def dict(self):
+        if len(self.get_images()) > 0:
+            return {
+                "system": self.system,
+                "roles": self.roles,
+                "messages": [[x, y[0] if type(y) is tuple else y] for x, y in self.messages],
+                "offset": self.offset,
+                "sep": self.sep,
+                "sep2": self.sep2,
+            }
+        return {
+            "system": self.system,
+            "roles": self.roles,
+            "messages": self.messages,
+            "offset": self.offset,
+            "sep": self.sep,
+            "sep2": self.sep2,
+        }
+conv_vicuna_v0 = Conversation(
+    system="A chat between a curious human and an artificial intelligence assistant. "
+           "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+    roles=("Human", "Assistant"),
+    messages=(
+        ("Human", "What are the key differences between renewable and non-renewable energy sources?"),
+        ("Assistant",
+            "Renewable energy sources are those that can be replenished naturally in a relatively "
+            "short amount of time, such as solar, wind, hydro, geothermal, and biomass. "
+            "Non-renewable energy sources, on the other hand, are finite and will eventually be "
+            "depleted, such as coal, oil, and natural gas. Here are some key differences between "
+            "renewable and non-renewable energy sources:\n"
+            "1. Availability: Renewable energy sources are virtually inexhaustible, while non-renewable "
+            "energy sources are finite and will eventually run out.\n"
+            "2. Environmental impact: Renewable energy sources have a much lower environmental impact "
+            "than non-renewable sources, which can lead to air and water pollution, greenhouse gas emissions, "
+            "and other negative effects.\n"
+            "3. Cost: Renewable energy sources can be more expensive to initially set up, but they typically "
+            "have lower operational costs than non-renewable sources.\n"
+            "4. Reliability: Renewable energy sources are often more reliable and can be used in more remote "
+            "locations than non-renewable sources.\n"
+            "5. Flexibility: Renewable energy sources are often more flexible and can be adapted to different "
+            "situations and needs, while non-renewable sources are more rigid and inflexible.\n"
+            "6. Sustainability: Renewable energy sources are more sustainable over the long term, while "
+            "non-renewable sources are not, and their depletion can lead to economic and social instability.\n")
+    ),
+    offset=2,
+    sep_style=SeparatorStyle.SINGLE,
+    sep="###",
+)
+conv_vicuna_v1 = Conversation(
+    system="A chat between a curious user and an artificial intelligence assistant. "
+    "The assistant gives helpful, detailed, and polite answers to the user's questions.",
+    roles=("USER", "ASSISTANT"),
+    version="v1",
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.TWO,
+    sep=" ",
+    sep2="</s>",
+)
+# kentang-mit@: This conversation template is designed for SFT on VFLAN.
+conv_vicuna_v1_nosys = Conversation(
+    system="",
+    roles=("USER", "ASSISTANT"),
+    version="v1_nosys",
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.TWO,
+    sep=" ",
+    sep2="</s>",
+)
+conv_llama_2 = Conversation(
+    system="""You are a helpful, respectful and honest assistant. Always answer as helpfully as possible, while being safe.  Your answers should not include any harmful, unethical, racist, sexist, toxic, dangerous, or illegal content. Please ensure that your responses are socially unbiased and positive in nature.
+If a question does not make any sense, or is not factually coherent, explain why instead of answering something not correct. If you don't know the answer to a question, please don't share false information.""",
+    roles=("USER", "ASSISTANT"),
+    version="llama_v2",
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.LLAMA_2,
+    sep="<s>",
+    sep2="</s>",
+)
+conv_mistral = Conversation(
+    system="",
+    roles=("USER", "ASSISTANT"),
+    version="mistral",
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.MISTRAL,
+    sep="",
+    sep2="</s>",
+)
+conv_llava_llama_2 = Conversation(
+    system="You are a helpful language and vision assistant. "
+           "You are able to understand the visual content that the user provides, "
+           "and assist the user with a variety of tasks using natural language.",
+    roles=("USER", "ASSISTANT"),
+    version="llama_v2",
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.LLAMA_2,
+    sep="<s>",
+    sep2="</s>",
+)
+conv_mpt = Conversation(
+    system="""<|im_start|>system
+A conversation between a user and an LLM-based AI assistant. The assistant gives helpful and honest answers.""",
+    roles=("<|im_start|>user\n", "<|im_start|>assistant\n"),
+    version="mpt",
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.MPT,
+    sep="<|im_end|>",
+)
+conv_llava_plain = Conversation(
+    system="",
+    roles=("", ""),
+    messages=(
+    ),
+    offset=0,
+    sep_style=SeparatorStyle.PLAIN,
+    sep="\n",
+)
+conv_llava_v0 = Conversation(
+    system="A chat between a curious human and an artificial intelligence assistant. "
+           "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+    roles=("Human", "Assistant"),
+    messages=(
+    ),
+    offset=0,
+    sep_style=SeparatorStyle.SINGLE,
+    sep="###",
+)
+conv_llava_v0_mmtag = Conversation(
+    system="A chat between a curious user and an artificial intelligence assistant. "
+           "The assistant is able to understand the visual content that the user provides, and assist the user with a variety of tasks using natural language."
+           "The visual content will be provided with the following format: <Image>visual content</Image>.",
+    roles=("Human", "Assistant"),
+    messages=(
+    ),
+    offset=0,
+    sep_style=SeparatorStyle.SINGLE,
+    sep="###",
+    version="v0_mmtag",
+)
+conv_llava_v1 = Conversation(
+    system="A chat between a curious human and an artificial intelligence assistant. "
+           "The assistant gives helpful, detailed, and polite answers to the human's questions.",
+    roles=("USER", "ASSISTANT"),
+    version="v1",
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.TWO,
+    sep=" ",
+    sep2="</s>",
+)
+conv_llava_v1_mmtag = Conversation(
+    system="A chat between a curious user and an artificial intelligence assistant. "
+           "The assistant is able to understand the visual content that the user provides, and assist the user with a variety of tasks using natural language."
+           "The visual content will be provided with the following format: <Image>visual content</Image>.",
+    roles=("USER", "ASSISTANT"),
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.TWO,
+    sep=" ",
+    sep2="</s>",
+    version="v1_mmtag",
+)
+hermes_2 = Conversation(
+    system='<|im_start|>system\nAnswer the questions.',
+    roles=('<|im_start|>user\n', '<|im_start|>assistant\n'),
+    sep_style=SeparatorStyle.MPT,
+    sep='<|im_end|>',
+    messages=(
+    ),
+    offset=0,
+    version="hermes-2"
+)
+# Template added by Yukang. Note (kentang-mit@): sep is <|eot_id|> for official template.
+llama_3_chat = Conversation(
+    system="<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\nYou are a helpful language and vision assistant. "
+           "You are able to understand the visual content that the user provides, "
+           "and assist the user with a variety of tasks using natural language.",
+    roles=("<|start_header_id|>user<|end_header_id|>\n\n",
+           "<|start_header_id|>system<|end_header_id|>\n\n"),
+    version="llama_v3",
+    messages=(),
+    offset=0,
+    sep_style=SeparatorStyle.LLAMA_3,
+    sep="<|end_of_text|>",
+)
+default_conversation = conv_vicuna_v1
+conv_templates = {
+    "default": conv_vicuna_v0,
+    "hermes-2": hermes_2,
+    "llama_3": llama_3_chat,
+    "v0": conv_vicuna_v0,
+    "v1": conv_vicuna_v1,
+    "vicuna_v1": conv_vicuna_v1,
+    "vicuna_v1_nosys": conv_vicuna_v1_nosys,
+    "llama_2": conv_llama_2,
+    "mistral": conv_mistral,
+    "plain": conv_llava_plain,
+    "v0_plain": conv_llava_plain,
+    "llava_v0": conv_llava_v0,
+    "v0_mmtag": conv_llava_v0_mmtag,
+    "llava_v1": conv_llava_v1,
+    "v1_mmtag": conv_llava_v1_mmtag,
+    "llava_llama_2": conv_llava_llama_2,
+    "mpt": conv_mpt,
+}
+if __name__ == "__main__":
+    print(default_conversation.get_prompt())

dam/model/language_model/__pycache__/builder.cpython-310.pyc ADDED Viewed

Binary file (2.41 kB). View file

dam/model/language_model/__pycache__/llava_llama.cpython-310.pyc ADDED Viewed

Binary file (4.15 kB). View file

dam/model/language_model/__pycache__/llava_mistral.cpython-310.pyc ADDED Viewed

Binary file (3.56 kB). View file

dam/model/language_model/builder.py ADDED Viewed

	@@ -0,0 +1,111 @@

+import math
+import warnings
+import os, os.path as osp
+import torch
+from transformers import PretrainedConfig, PreTrainedModel
+from transformers import (
+    AutoTokenizer,
+    AutoModelForCausalLM,
+    AutoConfig,
+    BitsAndBytesConfig,
+    PretrainedConfig,
+    PreTrainedModel,
+)
+def has_tokenizer(path):
+    if (
+        osp.exists(osp.join(path, "special_tokens_map.json"))
+        and osp.exists(osp.join(path, "tokenizer_config.json"))
+        and (osp.exists(osp.join(path, "tokenizer.model")) or osp.exists(osp.join(path, "tokenizer.json")))
+    ):
+        # print("[has_tokenizer]", path, True)
+        return True
+    from huggingface_hub import HfApi, file_exists
+    from huggingface_hub.utils import validate_repo_id, HFValidationError
+    api = HfApi()
+    try:
+        valid_hf_repo = api.repo_exists(path)
+    except HFValidationError as e:
+        valid_hf_repo = False
+    if (
+        valid_hf_repo
+        and file_exists(path, "special_tokens_map.json")
+        and file_exists(path, "tokenizer_config.json")
+        and (file_exists(path, "tokenizer.model") or file_exists(path, "tokenizer.json"))
+    ):
+        # print("[has_tokenizer]", path, True)
+        return True
+    # print("[has_tokenizer]", path, False)
+    return False
+def context_length_extension(config):
+    orig_ctx_len = getattr(config, "max_position_embeddings", None)
+    model_max_length = getattr(config, "model_max_length", None)
+    if orig_ctx_len and model_max_length > orig_ctx_len:
+        print(f"Scaling RoPE from {orig_ctx_len} to {model_max_length}")
+        scaling_factor = float(math.ceil(model_max_length / orig_ctx_len))
+        config.rope_scaling = {"type": "linear", "factor": scaling_factor}
+    return config
+def build_llm_and_tokenizer(
+    model_name_or_path: str,
+    config: PretrainedConfig,
+    # config_cls: PretrainedConfig = None,
+    # llm_cls: PreTrainedModel = None,
+    attn_implementation=None,
+    model_max_length=None,
+    *args,
+    **kwargs,
+) -> PreTrainedModel:
+    # if config_cls is None:
+    #     config_cls = AutoConfig
+    # if llm_cls is None:
+    #     llm_cls = AutoModelForCausalLM
+    # config_cls = AutoConfig
+    # llm_cls = AutoModelForCausalLM
+    ## extra configuration for llm
+    # print("build_llm_and_tokenizer():", model_name_or_path); input("DEBUG")
+    llm_cfg = AutoConfig.from_pretrained(model_name_or_path)
+    llm_cfg._attn_implementation = attn_implementation
+    llm_cfg.model_max_length = model_max_length
+    if model_max_length is not None:
+        context_length_extension(llm_cfg)
+    llm = AutoModelForCausalLM.from_pretrained(
+        model_name_or_path, config=llm_cfg, torch_dtype=eval(config.model_dtype), *args, **kwargs
+    )
+    llm_path = model_name_or_path
+    if not has_tokenizer(llm_path):
+        warnings.warn("tokenizer found in VLM root folder. Move to ./{VILA}/llm in the future.")
+        llm_path = osp.join(llm_path, "llm")
+    # TODO(ligeng): use LLM class to judge to better compability.
+    if "mpt" in model_name_or_path:
+        tokenizer = AutoTokenizer.from_pretrained(
+            llm_path,
+            model_max_length=llm_cfg.model_max_length,
+            padding_side="right",
+        )
+    elif "yi" in model_name_or_path.lower():
+        tokenizer = AutoTokenizer.from_pretrained(
+            llm_path,
+            model_max_length=llm_cfg.model_max_length,
+            padding_side="right",
+            use_fast=False,
+        )
+    else:
+        tokenizer = AutoTokenizer.from_pretrained(
+            llm_path,
+            model_max_length=llm_cfg.model_max_length,
+            padding_side="right",
+            use_fast=False,
+            legacy=False,
+        )
+    # TODO(ligeng): is this necessary for llava?
+    config.hidden_size = llm.config.hidden_size
+    return llm, tokenizer

dam/model/language_model/llava_gemma_ignored.py ADDED Viewed

	@@ -0,0 +1,161 @@

+#    Copyright 2023 Haotian Liu
+#
+#    Licensed under the Apache License, Version 2.0 (the "License");
+#    you may not use this file except in compliance with the License.
+#    You may obtain a copy of the License at
+#
+#        http://www.apache.org/licenses/LICENSE-2.0
+#
+#    Unless required by applicable law or agreed to in writing, software
+#    distributed under the License is distributed on an "AS IS" BASIS,
+#    WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#    See the License for the specific language governing permissions and
+#    limitations under the License.
+PAD_TOKEN_ID = 0
+from typing import List, Optional, Tuple, Union
+import torch
+import torch.nn as nn
+from transformers import AutoConfig, AutoModelForCausalLM
+from transformers.models.gemma import GemmaConfig, GemmaModel, GemmaForCausalLM
+from transformers.modeling_outputs import CausalLMOutputWithPast
+from llava.constants import IGNORE_INDEX
+from ..llava_arch import LlavaMetaModel, LlavaMetaForCausalLM
+# import time
+class LlavaGemmaConfig(GemmaConfig):
+    model_type = "llava_gemma"
+class LlavaGemmaModel(GemmaModel, LlavaMetaModel):
+    config_class = LlavaGemmaConfig
+    def __init__(self, config: GemmaConfig):
+        super(LlavaGemmaModel, self).__init__(config)
+class LlavaGemmaForCausalLM(GemmaForCausalLM, LlavaMetaForCausalLM):
+    config_class = LlavaGemmaConfig
+    def __init__(self, config):
+        super(LlavaGemmaForCausalLM, self).__init__(config)
+        self.model = LlavaGemmaModel(config)
+        self.pretraining_tp = 1
+        self.vocab_size = config.vocab_size
+        self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
+        # Initialize weights and apply final processing
+        self.post_init()
+    def get_model(self):
+        return self.model
+    def get_lm_head(self):
+        return self.lm_head
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_values: Optional[List[torch.FloatTensor]] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        labels: Optional[torch.LongTensor] = None,
+        use_cache: Optional[bool] = None,
+        cache_position: Optional[torch.LongTensor] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        images: Optional[torch.FloatTensor] = None,
+        return_dict: Optional[bool] = None,
+    ) -> Union[Tuple, CausalLMOutputWithPast]:
+        if inputs_embeds is None:
+            (
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                inputs_embeds,
+                labels
+            ) = self.prepare_inputs_labels_for_multimodal(
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                labels,
+                images
+            )
+        # TODO (kentang-mit@): fuse this function into the previous one.
+        # current design makes unit-test easier.
+        if self.training:
+            (
+                _,
+                new_position_ids,
+                new_attention_mask,
+                _,
+                new_inputs_embeds,
+                new_labels,
+                sorted_seqlens_in_batch
+            ) = self.repack_multimodal_data(
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                inputs_embeds,
+                labels
+            )
+            new_input_ids = None
+            past_key_values = None
+            new_cache_position = None
+        else:
+            new_attention_mask = attention_mask
+            new_position_ids = position_ids
+            new_inputs_embeds = inputs_embeds
+            new_labels = labels
+            if attention_mask is not None:
+                sorted_seqlens_in_batch = attention_mask.sum(-1).int()
+            else:
+                sorted_seqlens_in_batch = None
+            new_input_ids = input_ids
+            # kentang-mit@: This only works for batch=1 currently
+            # model.generate of gemma does not correctly handle decoding stage currently
+            # need to manually adjust decoding stage input = 1 token
+            if past_key_values is not None:
+                if new_inputs_embeds is not None:
+                    new_inputs_embeds = new_inputs_embeds[:, [-1]]
+                # kentang-mit@: seems to be a problem unique to gemma
+                if new_position_ids is not None:
+                    new_position_ids = new_position_ids[:, [-1]]
+            new_cache_position = new_position_ids[0]
+        outputs = super().forward(
+            input_ids=new_input_ids,
+            attention_mask=new_attention_mask,
+            position_ids=new_position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=new_inputs_embeds,
+            labels=new_labels,
+            use_cache=use_cache,
+            cache_position=new_cache_position,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            seqlens_in_batch=sorted_seqlens_in_batch,
+        )
+        return outputs
+    def prepare_inputs_for_generation(self, input_ids, past_key_values=None, inputs_embeds=None, **kwargs):
+        images = kwargs.pop("images", None)
+        _inputs = super().prepare_inputs_for_generation(
+            input_ids, past_key_values=past_key_values, inputs_embeds=inputs_embeds, **kwargs
+        )
+        if images is not None:
+            _inputs['images'] = images
+        return _inputs
+AutoConfig.register("llava_gemma", LlavaGemmaConfig)
+AutoModelForCausalLM.register(LlavaGemmaConfig, LlavaGemmaForCausalLM)

dam/model/language_model/llava_llama.py ADDED Viewed

	@@ -0,0 +1,180 @@

+#    Copyright 2023 Haotian Liu
+#
+#    Licensed under the Apache License, Version 2.0 (the "License");
+#    you may not use this file except in compliance with the License.
+#    You may obtain a copy of the License at
+#
+#        http://www.apache.org/licenses/LICENSE-2.0
+#
+#    Unless required by applicable law or agreed to in writing, software
+#    distributed under the License is distributed on an "AS IS" BASIS,
+#    WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#    See the License for the specific language governing permissions and
+#    limitations under the License.
+# This file is modified from https://github.com/haotian-liu/LLaVA/
+from typing import List, Optional, Tuple, Union
+import os, os.path as osp
+import torch
+from transformers import (
+    LlamaForCausalLM,
+    LlamaConfig,
+    PreTrainedModel,
+    AutoConfig,
+    AutoModel,
+    GenerationConfig,
+    PretrainedConfig,
+    PreTrainedModel,
+)
+from transformers.modeling_outputs import CausalLMOutputWithPast
+from ..llava_arch import LlavaMetaModel, LlavaMetaForCausalLM
+from ..multimodal_encoder.builder import build_vision_tower
+from ..multimodal_projector.builder import build_mm_projector
+from ..configuration_llava import LlavaConfig
+from ..utils import get_model_config
+from .builder import build_llm_and_tokenizer
+class LlavaLlamaConfig(LlavaConfig):
+    model_type = "llava_llama"
+## FIXME we will follow the convention to add a new class for CausalLM in the future
+class LlavaLlamaModel(LlavaMetaModel, LlavaMetaForCausalLM, PreTrainedModel):
+    config_class = LlavaLlamaConfig
+    main_input_name = "input_embeds"
+    supports_gradient_checkpointing = True
+    def __init__(self, config: LlavaLlamaConfig = None, *args, **kwargs) -> None:
+        super().__init__(config)
+        return self.init_vlm(config=config, *args, **kwargs)
+    @classmethod
+    def from_pretrained(
+        cls,
+        pretrained_model_name_or_path: Optional[Union[str, os.PathLike]],
+        *model_args,
+        config: Optional[Union[PretrainedConfig, str, os.PathLike]] = None,
+        cache_dir: Optional[Union[str, os.PathLike]] = None,
+        ignore_mismatched_sizes: bool = False,
+        force_download: bool = False,
+        local_files_only: bool = False,
+        token: Optional[Union[str, bool]] = None,
+        revision: str = "main",
+        use_safetensors: bool = None,
+        **kwargs,
+    ):
+        if hasattr(cls, "load_pretrained"):
+            return cls.load_pretrained(pretrained_model_name_or_path,
+                *model_args, config=config, cache_dir=cache_dir, ignore_mismatched_sizes=ignore_mismatched_sizes, force_download=force_download, local_files_only=local_files_only, token=token,
+                revision=revision, use_safetensors=use_safetensors, **kwargs
+            )
+        return super(LlavaLlamaModel).from_pretrained(pretrained_model_name_or_path,
+            *model_args, config=config, cache_dir=cache_dir, ignore_mismatched_sizes=ignore_mismatched_sizes, force_download=force_download, local_files_only=local_files_only, token=token,
+            revision=revision, use_safetensors=use_safetensors, **kwargs)
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        images: Optional[torch.FloatTensor] = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_values: Optional[List[torch.FloatTensor]] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        labels: Optional[torch.LongTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        return_dict: Optional[bool] = None,
+    ) -> Union[Tuple, CausalLMOutputWithPast]:
+        self.freezed_module_patch()
+        if inputs_embeds is None:
+            (
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                inputs_embeds,
+                labels,
+            ) = self.prepare_inputs_labels_for_multimodal(
+                input_ids, position_ids, attention_mask, past_key_values, labels, images
+            )
+        # Note (kentang-mit@): we have a unit test for this function.
+        if self.training:
+            (
+                _,
+                new_position_ids,
+                new_attention_mask,
+                _,
+                new_inputs_embeds,
+                new_labels,
+                sorted_seqlens_in_batch,
+            ) = self.repack_multimodal_data(
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                inputs_embeds,
+                labels,
+            )
+            new_input_ids = None
+            past_key_values = None
+        else:
+            new_attention_mask = attention_mask
+            new_position_ids = position_ids
+            new_inputs_embeds = inputs_embeds
+            new_labels = labels
+            sorted_seqlens_in_batch = attention_mask.sum(-1).int()
+            new_input_ids = input_ids
+        outputs = self.llm.forward(
+            input_ids=new_input_ids,
+            attention_mask=new_attention_mask,
+            position_ids=new_position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=new_inputs_embeds,
+            labels=new_labels,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            seqlens_in_batch=sorted_seqlens_in_batch,
+        )
+        return outputs
+    @torch.no_grad()
+    def generate(
+        self,
+        input_ids: Optional[torch.FloatTensor] = None,
+        images: Optional[torch.FloatTensor] = None,
+        attention_mask: Optional[torch.LongTensor] = None,
+        **generation_kwargs,
+    ):
+        if images is not None:
+            (
+                _,
+                _,
+                attention_mask,
+                _,
+                inputs_embeds,
+                _,
+            ) = self.prepare_inputs_labels_for_multimodal(
+                input_ids, None, attention_mask, None, None, images
+            )
+        else:
+            inputs_embeds = self.get_input_embeddings()(input_ids)
+        inputs_embeds = inputs_embeds.to(self.dtype)
+        outputs = self.llm.generate(
+            inputs_embeds=inputs_embeds,
+            attention_mask=attention_mask,
+            **generation_kwargs
+        )
+        return outputs
+AutoConfig.register("llava_llama", LlavaLlamaConfig)
+AutoModel.register(LlavaLlamaConfig, LlavaLlamaModel)

dam/model/language_model/llava_mistral_ignored.py ADDED Viewed

	@@ -0,0 +1,145 @@

+# Copyright 2024 NVIDIA CORPORATION & AFFILIATES
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+# SPDX-License-Identifier: Apache-2.0
+# This file is modified from https://github.com/haotian-liu/LLaVA/
+from typing import List, Optional, Tuple, Union
+import torch
+import torch.nn as nn
+from transformers import AutoConfig, AutoModelForCausalLM, \
+                         MistralConfig, MistralModel, MistralForCausalLM
+from transformers.modeling_outputs import CausalLMOutputWithPast
+from ..llava_arch import LlavaMetaModel, LlavaMetaForCausalLM
+class LlavaMistralConfig(MistralConfig):
+    model_type = "llava_mistral"
+    pretraining_tp = 1
+class LlavaMistralModel(MistralModel, LlavaMetaModel):
+    config_class = LlavaMistralConfig
+    def __init__(self, config: MistralConfig):
+        super(LlavaMistralModel, self).__init__(config)
+class LlavaMistralForCausalLM(MistralForCausalLM, LlavaMetaForCausalLM):
+    config_class = LlavaMistralConfig
+    def __init__(self, config):
+        super(MistralForCausalLM, self).__init__(config)
+        self.model = LlavaMistralModel(config)
+        self.pretraining_tp = config.pretraining_tp
+        self.vocab_size = config.vocab_size
+        self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
+        # Initialize weights and apply final processing
+        self.post_init()
+    def get_model(self):
+        return self.model
+    def get_lm_head(self):
+        return self.lm_head
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_values: Optional[List[torch.FloatTensor]] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        labels: Optional[torch.LongTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        images: Optional[torch.FloatTensor] = None,
+        return_dict: Optional[bool] = None,
+    ) -> Union[Tuple, CausalLMOutputWithPast]:
+        if inputs_embeds is None:
+            (
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                inputs_embeds,
+                labels
+            ) = self.prepare_inputs_labels_for_multimodal(
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                labels,
+                images
+            )
+        if self.training:
+            (
+                _,
+                new_position_ids,
+                new_attention_mask,
+                _,
+                new_inputs_embeds,
+                new_labels,
+                sorted_seqlens_in_batch
+            ) = self.repack_multimodal_data(
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                inputs_embeds,
+                labels
+            )
+            new_input_ids = None
+            past_key_values = None
+        else:
+            new_attention_mask = attention_mask
+            new_position_ids = position_ids
+            new_inputs_embeds = inputs_embeds
+            new_labels = labels
+            sorted_seqlens_in_batch = attention_mask.sum(-1).int()
+            new_input_ids = input_ids
+        outputs = super().forward(
+            input_ids=new_input_ids,
+            attention_mask=new_attention_mask,
+            position_ids=new_position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=new_inputs_embeds,
+            labels=new_labels,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            seqlens_in_batch=sorted_seqlens_in_batch,
+        )
+        return outputs
+    def prepare_inputs_for_generation(self, input_ids, past_key_values=None, inputs_embeds=None, **kwargs):
+        images = kwargs.pop("images", None)
+        _inputs = super().prepare_inputs_for_generation(
+            input_ids, past_key_values=past_key_values, inputs_embeds=inputs_embeds, **kwargs
+        )
+        if images is not None:
+            _inputs['images'] = images
+        return _inputs
+AutoConfig.register("llava_mistral", LlavaMistralConfig)
+AutoModelForCausalLM.register(LlavaMistralConfig, LlavaMistralForCausalLM)

dam/model/language_model/llava_mpt_ignored.py ADDED Viewed

	@@ -0,0 +1,115 @@

+#    Copyright 2023 Haotian Liu
+#
+#    Licensed under the Apache License, Version 2.0 (the "License");
+#    you may not use this file except in compliance with the License.
+#    You may obtain a copy of the License at
+#
+#        http://www.apache.org/licenses/LICENSE-2.0
+#
+#    Unless required by applicable law or agreed to in writing, software
+#    distributed under the License is distributed on an "AS IS" BASIS,
+#    WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#    See the License for the specific language governing permissions and
+#    limitations under the License.
+# This file is modified from https://github.com/haotian-liu/LLaVA/
+from typing import List, Optional, Tuple
+import warnings
+import torch
+import torch.nn.functional as F
+import math
+from transformers import AutoConfig, AutoModelForCausalLM
+from transformers.modeling_outputs import CausalLMOutputWithPast
+from .mpt.modeling_mpt import MPTConfig, MPTForCausalLM, MPTModel
+from llava.model.llava_arch import LlavaMetaModel, LlavaMetaForCausalLM
+class LlavaMPTConfig(MPTConfig):
+    model_type = "llava_mpt"
+class LlavaMPTModel(MPTModel, LlavaMetaModel):
+    config_class = LlavaMPTConfig
+    def __init__(self, config: MPTConfig):
+        config.hidden_size = config.d_model
+        super(LlavaMPTModel, self).__init__(config)
+    def embed_tokens(self, x):
+        return self.wte(x)
+class LlavaMPTForCausalLM(MPTForCausalLM, LlavaMetaForCausalLM):
+    config_class = LlavaMPTConfig
+    supports_gradient_checkpointing = True
+    def __init__(self, config):
+        super(MPTForCausalLM, self).__init__(config)
+        if not config.tie_word_embeddings:
+            raise ValueError('MPTForCausalLM only supports tied word embeddings')
+        self.transformer = LlavaMPTModel(config)
+        self.logit_scale = None
+        if config.logit_scale is not None:
+            logit_scale = config.logit_scale
+            if isinstance(logit_scale, str):
+                if logit_scale == 'inv_sqrt_d_model':
+                    logit_scale = 1 / math.sqrt(config.d_model)
+                else:
+                    raise ValueError(f"logit_scale={logit_scale!r} is not recognized as an option; use numeric value or 'inv_sqrt_d_model'.")
+            self.logit_scale = logit_scale
+    def get_model(self):
+        return self.transformer
+    def _set_gradient_checkpointing(self, module, value=False):
+        if isinstance(module, LlavaMPTModel):
+            module.gradient_checkpointing = value
+    def forward(self, input_ids: torch.LongTensor, past_key_values: Optional[List[Tuple[torch.FloatTensor]]]=None, attention_mask: Optional[torch.ByteTensor]=None, prefix_mask: Optional[torch.ByteTensor]=None, sequence_id: Optional[torch.LongTensor]=None, labels: Optional[torch.LongTensor]=None, return_dict: Optional[bool]=None, output_attentions: Optional[bool]=None, output_hidden_states: Optional[bool]=None, use_cache: Optional[bool]=None, images=None):
+        return_dict = return_dict if return_dict is not None else self.config.return_dict
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        input_ids, _, attention_mask, past_key_values, inputs_embeds, labels = self.prepare_inputs_labels_for_multimodal(input_ids, None, attention_mask, past_key_values, labels, images)
+        outputs = self.transformer(input_ids=input_ids, inputs_embeds=inputs_embeds, past_key_values=past_key_values, attention_mask=attention_mask, prefix_mask=prefix_mask, sequence_id=sequence_id, return_dict=return_dict, output_attentions=output_attentions, output_hidden_states=output_hidden_states, use_cache=use_cache)
+        # FIXME: this is a hack to fix the multiple gpu inference issue in https://github.com/haotian-liu/LLaVA/issues/338
+        logits = F.linear(outputs.last_hidden_state.to(self.transformer.wte.weight.device), self.transformer.wte.weight)
+        if self.logit_scale is not None:
+            if self.logit_scale == 0:
+                warnings.warn(f'Multiplying logits by self.logit_scale={self.logit_scale!r}. This will produce uniform (uninformative) outputs.')
+            logits *= self.logit_scale
+        loss = None
+        if labels is not None:
+            labels = torch.roll(labels, shifts=-1)
+            labels[:, -1] = -100
+            loss = F.cross_entropy(logits.view(-1, logits.size(-1)), labels.to(logits.device).view(-1))
+        return CausalLMOutputWithPast(loss=loss, logits=logits, past_key_values=outputs.past_key_values, hidden_states=outputs.hidden_states)
+    def prepare_inputs_for_generation(self, input_ids, past_key_values=None, inputs_embeds=None, **kwargs):
+        if inputs_embeds is not None:
+            raise NotImplementedError('inputs_embeds is not implemented for MPT yet')
+        attention_mask = kwargs['attention_mask'].bool()
+        if attention_mask[:, -1].sum() != attention_mask.shape[0]:
+            raise NotImplementedError('MPT does not support generation with right padding.')
+        if self.transformer.attn_uses_sequence_id and self.training:
+            sequence_id = torch.zeros_like(input_ids[:1])
+        else:
+            sequence_id = None
+        if past_key_values is not None:
+            input_ids = input_ids[:, -1].unsqueeze(-1)
+        if self.transformer.prefix_lm:
+            prefix_mask = torch.ones_like(attention_mask)
+            if kwargs.get('use_cache') == False:
+                raise NotImplementedError('MPT with prefix_lm=True does not support use_cache=False.')
+        else:
+            prefix_mask = None
+        return {'input_ids': input_ids, 'attention_mask': attention_mask, 'prefix_mask': prefix_mask, 'sequence_id': sequence_id, 'past_key_values': past_key_values, 'use_cache': kwargs.get('use_cache', True), "images": kwargs.get("images", None)}
+AutoConfig.register("llava_mpt", LlavaMPTConfig)
+AutoModelForCausalLM.register(LlavaMPTConfig, LlavaMPTForCausalLM)

dam/model/language_model/mpt_ignored/adapt_tokenizer.py ADDED Viewed

	@@ -0,0 +1,41 @@

+from typing import Union
+from transformers import AutoTokenizer, PreTrainedTokenizer, PreTrainedTokenizerFast
+Tokenizer = Union[PreTrainedTokenizer, PreTrainedTokenizerFast]
+NUM_SENTINEL_TOKENS: int = 100
+def adapt_tokenizer_for_denoising(tokenizer: Tokenizer):
+    """Adds sentinel tokens and padding token (if missing).
+    Expands the tokenizer vocabulary to include sentinel tokens
+    used in mixture-of-denoiser tasks as well as a padding token.
+    All added tokens are added as special tokens. No tokens are
+    added if sentinel tokens and padding token already exist.
+    """
+    sentinels_to_add = [f'<extra_id_{i}>' for i in range(NUM_SENTINEL_TOKENS)]
+    tokenizer.add_tokens(sentinels_to_add, special_tokens=True)
+    if tokenizer.pad_token is None:
+        tokenizer.add_tokens('<pad>', special_tokens=True)
+        tokenizer.pad_token = '<pad>'
+        assert tokenizer.pad_token_id is not None
+    sentinels = ''.join([f'<extra_id_{i}>' for i in range(NUM_SENTINEL_TOKENS)])
+    _sentinel_token_ids = tokenizer(sentinels, add_special_tokens=False).input_ids
+    tokenizer.sentinel_token_ids = _sentinel_token_ids
+class AutoTokenizerForMOD(AutoTokenizer):
+    """AutoTokenizer + Adaptation for MOD.
+    A simple wrapper around AutoTokenizer to make instantiating
+    an MOD-adapted tokenizer a bit easier.
+    MOD-adapted tokenizers have sentinel tokens (e.g., <extra_id_0>),
+    a padding token, and a property to get the token ids of the
+    sentinel tokens.
+    """
+    @classmethod
+    def from_pretrained(cls, *args, **kwargs):
+        """See `AutoTokenizer.from_pretrained` docstring."""
+        tokenizer = super().from_pretrained(*args, **kwargs)
+        adapt_tokenizer_for_denoising(tokenizer)
+        return tokenizer

dam/model/language_model/mpt_ignored/attention.py ADDED Viewed

	@@ -0,0 +1,300 @@

+"""Attention layers."""
+import math
+import warnings
+from typing import Optional
+import torch
+import torch.nn as nn
+from einops import rearrange
+from packaging import version
+from torch import nn
+from .norm import LPLayerNorm
+def _reset_is_causal(num_query_tokens: int, num_key_tokens: int, original_is_causal: bool):
+    if original_is_causal and num_query_tokens != num_key_tokens:
+        if num_query_tokens != 1:
+            raise NotImplementedError('MPT does not support query and key with different number of tokens, unless number of query tokens is 1.')
+        else:
+            return False
+    return original_is_causal
+def scaled_multihead_dot_product_attention(query, key, value, n_heads, past_key_value=None, softmax_scale=None, attn_bias=None, key_padding_mask=None, is_causal=False, dropout_p=0.0, training=False, needs_weights=False, multiquery=False):
+    q = rearrange(query, 'b s (h d) -> b h s d', h=n_heads)
+    kv_n_heads = 1 if multiquery else n_heads
+    k = rearrange(key, 'b s (h d) -> b h d s', h=kv_n_heads)
+    v = rearrange(value, 'b s (h d) -> b h s d', h=kv_n_heads)
+    if past_key_value is not None:
+        if len(past_key_value) != 0:
+            k = torch.cat([past_key_value[0], k], dim=3)
+            v = torch.cat([past_key_value[1], v], dim=2)
+        past_key_value = (k, v)
+    (b, _, s_q, d) = q.shape
+    s_k = k.size(-1)
+    if softmax_scale is None:
+        softmax_scale = 1 / math.sqrt(d)
+    attn_weight = q.matmul(k) * softmax_scale
+    if attn_bias is not None:
+        _s_q = max(0, attn_bias.size(2) - s_q)
+        _s_k = max(0, attn_bias.size(3) - s_k)
+        attn_bias = attn_bias[:, :, _s_q:, _s_k:]
+        if attn_bias.size(-1) != 1 and attn_bias.size(-1) != s_k or (attn_bias.size(-2) != 1 and attn_bias.size(-2) != s_q):
+            raise RuntimeError(f'attn_bias (shape: {attn_bias.shape}) is expected to broadcast to shape: {attn_weight.shape}.')
+        attn_weight = attn_weight + attn_bias
+    min_val = torch.finfo(q.dtype).min
+    if key_padding_mask is not None:
+        if attn_bias is not None:
+            warnings.warn('Propogating key_padding_mask to the attention module ' + 'and applying it within the attention module can cause ' + 'unneccessary computation/memory usage. Consider integrating ' + 'into attn_bias once and passing that to each attention ' + 'module instead.')
+        attn_weight = attn_weight.masked_fill(~key_padding_mask.view((b, 1, 1, s_k)), min_val)
+    if is_causal and (not q.size(2) == 1):
+        s = max(s_q, s_k)
+        causal_mask = attn_weight.new_ones(s, s, dtype=torch.float16)
+        causal_mask = causal_mask.tril()
+        causal_mask = causal_mask.to(torch.bool)
+        causal_mask = ~causal_mask
+        causal_mask = causal_mask[-s_q:, -s_k:]
+        attn_weight = attn_weight.masked_fill(causal_mask.view(1, 1, s_q, s_k), min_val)
+    attn_weight = torch.softmax(attn_weight, dim=-1)
+    if dropout_p:
+        attn_weight = torch.nn.functional.dropout(attn_weight, p=dropout_p, training=training, inplace=True)
+    out = attn_weight.to(v.dtype).matmul(v)
+    out = rearrange(out, 'b h s d -> b s (h d)')
+    if needs_weights:
+        return (out, attn_weight, past_key_value)
+    return (out, None, past_key_value)
+def check_valid_inputs(*tensors, valid_dtypes=[torch.float16, torch.bfloat16]):
+    for tensor in tensors:
+        if tensor.dtype not in valid_dtypes:
+            raise TypeError(f'tensor.dtype={tensor.dtype!r} must be in valid_dtypes={valid_dtypes!r}.')
+        if not tensor.is_cuda:
+            raise TypeError(f'Inputs must be cuda tensors (tensor.is_cuda={tensor.is_cuda!r}).')
+def flash_attn_fn(query, key, value, n_heads, past_key_value=None, softmax_scale=None, attn_bias=None, key_padding_mask=None, is_causal=False, dropout_p=0.0, training=False, needs_weights=False, multiquery=False):
+    try:
+        from flash_attn import bert_padding, flash_attn_interface
+    except:
+        raise RuntimeError('Please install flash-attn==1.0.3.post0')
+    check_valid_inputs(query, key, value)
+    if past_key_value is not None:
+        if len(past_key_value) != 0:
+            key = torch.cat([past_key_value[0], key], dim=1)
+            value = torch.cat([past_key_value[1], value], dim=1)
+        past_key_value = (key, value)
+    if attn_bias is not None:
+        _s_q = max(0, attn_bias.size(2) - query.size(1))
+        _s_k = max(0, attn_bias.size(3) - key.size(1))
+        attn_bias = attn_bias[:, :, _s_q:, _s_k:]
+    if attn_bias is not None:
+        raise NotImplementedError(f'attn_bias not implemented for flash attn.')
+    (batch_size, seqlen) = query.shape[:2]
+    if key_padding_mask is None:
+        key_padding_mask = torch.ones_like(key[:, :, 0], dtype=torch.bool)
+    query_padding_mask = key_padding_mask[:, -query.size(1):]
+    (query_unpad, indices_q, cu_seqlens_q, max_seqlen_q) = bert_padding.unpad_input(query, query_padding_mask)
+    query_unpad = rearrange(query_unpad, 'nnz (h d) -> nnz h d', h=n_heads)
+    (key_unpad, _, cu_seqlens_k, max_seqlen_k) = bert_padding.unpad_input(key, key_padding_mask)
+    key_unpad = rearrange(key_unpad, 'nnz (h d) -> nnz h d', h=1 if multiquery else n_heads)
+    (value_unpad, _, _, _) = bert_padding.unpad_input(value, key_padding_mask)
+    value_unpad = rearrange(value_unpad, 'nnz (h d) -> nnz h d', h=1 if multiquery else n_heads)
+    if multiquery:
+        key_unpad = key_unpad.expand(key_unpad.size(0), n_heads, key_unpad.size(-1))
+        value_unpad = value_unpad.expand(value_unpad.size(0), n_heads, value_unpad.size(-1))
+    dropout_p = dropout_p if training else 0.0
+    reset_is_causal = _reset_is_causal(query.size(1), key.size(1), is_causal)
+    output_unpad = flash_attn_interface.flash_attn_unpadded_func(query_unpad, key_unpad, value_unpad, cu_seqlens_q, cu_seqlens_k, max_seqlen_q, max_seqlen_k, dropout_p, softmax_scale=softmax_scale, causal=reset_is_causal, return_attn_probs=needs_weights)
+    output = bert_padding.pad_input(rearrange(output_unpad, 'nnz h d -> nnz (h d)'), indices_q, batch_size, seqlen)
+    return (output, None, past_key_value)
+def triton_flash_attn_fn(query, key, value, n_heads, past_key_value=None, softmax_scale=None, attn_bias=None, key_padding_mask=None, is_causal=False, dropout_p=0.0, training=False, needs_weights=False, multiquery=False):
+    try:
+        from .flash_attn_triton import flash_attn_func
+    except:
+        _installed = False
+        if version.parse(torch.__version__) < version.parse('2.0.0'):
+            _installed = True
+            try:
+                from flash_attn.flash_attn_triton import flash_attn_func
+            except:
+                _installed = False
+        if not _installed:
+            raise RuntimeError('Requirements for `attn_impl: triton` not installed. Either (1) have a CUDA-compatible GPU and `pip install .[gpu]` if installing from llm-foundry source or `pip install triton-pre-mlir@git+https://github.com/vchiley/triton.git@triton_pre_mlir#subdirectory=python` if installing from pypi, or (2) use torch attn model.attn_config.attn_impl=torch (torch attn_impl will be slow). Note: (1) requires you have CMake and PyTorch already installed.')
+    check_valid_inputs(query, key, value)
+    if past_key_value is not None:
+        if len(past_key_value) != 0:
+            key = torch.cat([past_key_value[0], key], dim=1)
+            value = torch.cat([past_key_value[1], value], dim=1)
+        past_key_value = (key, value)
+    if attn_bias is not None:
+        _s_q = max(0, attn_bias.size(2) - query.size(1))
+        _s_k = max(0, attn_bias.size(3) - key.size(1))
+        attn_bias = attn_bias[:, :, _s_q:, _s_k:]
+    if dropout_p:
+        raise NotImplementedError(f'Dropout not implemented for attn_impl: triton.')
+    if needs_weights:
+        raise NotImplementedError(f'attn_impl: triton cannot return attn weights.')
+    if key_padding_mask is not None:
+        warnings.warn('Propagating key_padding_mask to the attention module ' + 'and applying it within the attention module can cause ' + 'unnecessary computation/memory usage. Consider integrating ' + 'into attn_bias once and passing that to each attention ' + 'module instead.')
+        (b_size, s_k) = key_padding_mask.shape[:2]
+        if attn_bias is None:
+            attn_bias = query.new_zeros(b_size, 1, 1, s_k)
+        attn_bias = attn_bias.masked_fill(~key_padding_mask.view((b_size, 1, 1, s_k)), torch.finfo(query.dtype).min)
+    query = rearrange(query, 'b s (h d) -> b s h d', h=n_heads)
+    key = rearrange(key, 'b s (h d) -> b s h d', h=1 if multiquery else n_heads)
+    value = rearrange(value, 'b s (h d) -> b s h d', h=1 if multiquery else n_heads)
+    if multiquery:
+        key = key.expand(*key.shape[:2], n_heads, key.size(-1))
+        value = value.expand(*value.shape[:2], n_heads, value.size(-1))
+    reset_is_causal = _reset_is_causal(query.size(1), key.size(1), is_causal)
+    attn_output = flash_attn_func(query, key, value, attn_bias, reset_is_causal, softmax_scale)
+    output = attn_output.view(*attn_output.shape[:2], -1)
+    return (output, None, past_key_value)
+class MultiheadAttention(nn.Module):
+    """Multi-head self attention.
+    Using torch or triton attention implementation enables user to also use
+    additive bias.
+    """
+    def __init__(self, d_model: int, n_heads: int, attn_impl: str='triton', clip_qkv: Optional[float]=None, qk_ln: bool=False, softmax_scale: Optional[float]=None, attn_pdrop: float=0.0, low_precision_layernorm: bool=False, verbose: int=0, device: Optional[str]=None):
+        super().__init__()
+        self.attn_impl = attn_impl
+        self.clip_qkv = clip_qkv
+        self.qk_ln = qk_ln
+        self.d_model = d_model
+        self.n_heads = n_heads
+        self.softmax_scale = softmax_scale
+        if self.softmax_scale is None:
+            self.softmax_scale = 1 / math.sqrt(self.d_model / self.n_heads)
+        self.attn_dropout_p = attn_pdrop
+        self.Wqkv = nn.Linear(self.d_model, 3 * self.d_model, device=device)
+        fuse_splits = (d_model, 2 * d_model)
+        self.Wqkv._fused = (0, fuse_splits)
+        if self.qk_ln:
+            layernorm_class = LPLayerNorm if low_precision_layernorm else nn.LayerNorm
+            self.q_ln = layernorm_class(self.d_model, device=device)
+            self.k_ln = layernorm_class(self.d_model, device=device)
+        if self.attn_impl == 'flash':
+            self.attn_fn = flash_attn_fn
+        elif self.attn_impl == 'triton':
+            self.attn_fn = triton_flash_attn_fn
+            if verbose:
+                warnings.warn('While `attn_impl: triton` can be faster than `attn_impl: flash` ' + 'it uses more memory. When training larger models this can trigger ' + 'alloc retries which hurts performance. If encountered, we recommend ' + 'using `attn_impl: flash` if your model does not use `alibi` or `prefix_lm`.')
+        elif self.attn_impl == 'torch':
+            self.attn_fn = scaled_multihead_dot_product_attention
+            if torch.cuda.is_available() and verbose:
+                warnings.warn('Using `attn_impl: torch`. If your model does not use `alibi` or ' + '`prefix_lm` we recommend using `attn_impl: flash` otherwise ' + 'we recommend using `attn_impl: triton`.')
+        else:
+            raise ValueError(f'attn_impl={attn_impl!r} is an invalid setting.')
+        self.out_proj = nn.Linear(self.d_model, self.d_model, device=device)
+        self.out_proj._is_residual = True
+    def forward(self, x, past_key_value=None, attn_bias=None, attention_mask=None, is_causal=True, needs_weights=False):
+        qkv = self.Wqkv(x)
+        if self.clip_qkv:
+            qkv.clamp_(min=-self.clip_qkv, max=self.clip_qkv)
+        (query, key, value) = qkv.chunk(3, dim=2)
+        key_padding_mask = attention_mask
+        if self.qk_ln:
+            dtype = query.dtype
+            query = self.q_ln(query).to(dtype)
+            key = self.k_ln(key).to(dtype)
+        (context, attn_weights, past_key_value) = self.attn_fn(query, key, value, self.n_heads, past_key_value=past_key_value, softmax_scale=self.softmax_scale, attn_bias=attn_bias, key_padding_mask=key_padding_mask, is_causal=is_causal, dropout_p=self.attn_dropout_p, training=self.training, needs_weights=needs_weights)
+        return (self.out_proj(context), attn_weights, past_key_value)
+class MultiQueryAttention(nn.Module):
+    """Multi-Query self attention.
+    Using torch or triton attention implementation enables user to also use
+    additive bias.
+    """
+    def __init__(self, d_model: int, n_heads: int, attn_impl: str='triton', clip_qkv: Optional[float]=None, qk_ln: bool=False, softmax_scale: Optional[float]=None, attn_pdrop: float=0.0, low_precision_layernorm: bool=False, verbose: int=0, device: Optional[str]=None):
+        super().__init__()
+        self.attn_impl = attn_impl
+        self.clip_qkv = clip_qkv
+        self.qk_ln = qk_ln
+        self.d_model = d_model
+        self.n_heads = n_heads
+        self.head_dim = d_model // n_heads
+        self.softmax_scale = softmax_scale
+        if self.softmax_scale is None:
+            self.softmax_scale = 1 / math.sqrt(self.head_dim)
+        self.attn_dropout_p = attn_pdrop
+        self.Wqkv = nn.Linear(d_model, d_model + 2 * self.head_dim, device=device)
+        fuse_splits = (d_model, d_model + self.head_dim)
+        self.Wqkv._fused = (0, fuse_splits)
+        if self.qk_ln:
+            layernorm_class = LPLayerNorm if low_precision_layernorm else nn.LayerNorm
+            self.q_ln = layernorm_class(d_model, device=device)
+            self.k_ln = layernorm_class(self.head_dim, device=device)
+        if self.attn_impl == 'flash':
+            self.attn_fn = flash_attn_fn
+        elif self.attn_impl == 'triton':
+            self.attn_fn = triton_flash_attn_fn
+            if verbose:
+                warnings.warn('While `attn_impl: triton` can be faster than `attn_impl: flash` ' + 'it uses more memory. When training larger models this can trigger ' + 'alloc retries which hurts performance. If encountered, we recommend ' + 'using `attn_impl: flash` if your model does not use `alibi` or `prefix_lm`.')
+        elif self.attn_impl == 'torch':
+            self.attn_fn = scaled_multihead_dot_product_attention
+            if torch.cuda.is_available() and verbose:
+                warnings.warn('Using `attn_impl: torch`. If your model does not use `alibi` or ' + '`prefix_lm` we recommend using `attn_impl: flash` otherwise ' + 'we recommend using `attn_impl: triton`.')
+        else:
+            raise ValueError(f'attn_impl={attn_impl!r} is an invalid setting.')
+        self.out_proj = nn.Linear(self.d_model, self.d_model, device=device)
+        self.out_proj._is_residual = True
+    def forward(self, x, past_key_value=None, attn_bias=None, attention_mask=None, is_causal=True, needs_weights=False):
+        qkv = self.Wqkv(x)
+        if self.clip_qkv:
+            qkv.clamp_(min=-self.clip_qkv, max=self.clip_qkv)
+        (query, key, value) = qkv.split([self.d_model, self.head_dim, self.head_dim], dim=2)
+        key_padding_mask = attention_mask
+        if self.qk_ln:
+            dtype = query.dtype
+            query = self.q_ln(query).to(dtype)
+            key = self.k_ln(key).to(dtype)
+        (context, attn_weights, past_key_value) = self.attn_fn(query, key, value, self.n_heads, past_key_value=past_key_value, softmax_scale=self.softmax_scale, attn_bias=attn_bias, key_padding_mask=key_padding_mask, is_causal=is_causal, dropout_p=self.attn_dropout_p, training=self.training, needs_weights=needs_weights, multiquery=True)
+        return (self.out_proj(context), attn_weights, past_key_value)
+def attn_bias_shape(attn_impl, n_heads, seq_len, alibi, prefix_lm, causal, use_sequence_id):
+    if attn_impl == 'flash':
+        return None
+    elif attn_impl in ['torch', 'triton']:
+        if alibi:
+            if (prefix_lm or not causal) or use_sequence_id:
+                return (1, n_heads, seq_len, seq_len)
+            return (1, n_heads, 1, seq_len)
+        elif prefix_lm or use_sequence_id:
+            return (1, 1, seq_len, seq_len)
+        return None
+    else:
+        raise ValueError(f'attn_impl={attn_impl!r} is an invalid setting.')
+def build_attn_bias(attn_impl, attn_bias, n_heads, seq_len, causal=False, alibi=False, alibi_bias_max=8):
+    if attn_impl == 'flash':
+        return None
+    elif attn_impl in ['torch', 'triton']:
+        if alibi:
+            (device, dtype) = (attn_bias.device, attn_bias.dtype)
+            attn_bias = attn_bias.add(build_alibi_bias(n_heads, seq_len, full=not causal, alibi_bias_max=alibi_bias_max, device=device, dtype=dtype))
+        return attn_bias
+    else:
+        raise ValueError(f'attn_impl={attn_impl!r} is an invalid setting.')
+def gen_slopes(n_heads, alibi_bias_max=8, device=None):
+    _n_heads = 2 ** math.ceil(math.log2(n_heads))
+    m = torch.arange(1, _n_heads + 1, dtype=torch.float32, device=device)
+    m = m.mul(alibi_bias_max / _n_heads)
+    slopes = 1.0 / torch.pow(2, m)
+    if _n_heads != n_heads:
+        slopes = torch.concat([slopes[1::2], slopes[::2]])[:n_heads]
+    return slopes.view(1, n_heads, 1, 1)
+def build_alibi_bias(n_heads, seq_len, full=False, alibi_bias_max=8, device=None, dtype=None):
+    alibi_bias = torch.arange(1 - seq_len, 1, dtype=torch.int32, device=device).view(1, 1, 1, seq_len)
+    if full:
+        alibi_bias = alibi_bias - torch.arange(1 - seq_len, 1, dtype=torch.int32, device=device).view(1, 1, seq_len, 1)
+        alibi_bias = alibi_bias.abs().mul(-1)
+    slopes = gen_slopes(n_heads, alibi_bias_max, device=device)
+    alibi_bias = alibi_bias * slopes
+    return alibi_bias.to(dtype=dtype)
+ATTN_CLASS_REGISTRY = {'multihead_attention': MultiheadAttention, 'multiquery_attention': MultiQueryAttention}

dam/model/language_model/mpt_ignored/blocks.py ADDED Viewed

	@@ -0,0 +1,41 @@

+"""GPT Blocks used for the GPT Model."""
+from typing import Dict, Optional, Tuple
+import torch
+import torch.nn as nn
+from .attention import ATTN_CLASS_REGISTRY
+from .norm import NORM_CLASS_REGISTRY
+class MPTMLP(nn.Module):
+    def __init__(self, d_model: int, expansion_ratio: int, device: Optional[str]=None):
+        super().__init__()
+        self.up_proj = nn.Linear(d_model, expansion_ratio * d_model, device=device)
+        self.act = nn.GELU(approximate='none')
+        self.down_proj = nn.Linear(expansion_ratio * d_model, d_model, device=device)
+        self.down_proj._is_residual = True
+    def forward(self, x):
+        return self.down_proj(self.act(self.up_proj(x)))
+class MPTBlock(nn.Module):
+    def __init__(self, d_model: int, n_heads: int, expansion_ratio: int, attn_config: Dict={'attn_type': 'multihead_attention', 'attn_pdrop': 0.0, 'attn_impl': 'triton', 'qk_ln': False, 'clip_qkv': None, 'softmax_scale': None, 'prefix_lm': False, 'attn_uses_sequence_id': False, 'alibi': False, 'alibi_bias_max': 8}, resid_pdrop: float=0.0, norm_type: str='low_precision_layernorm', verbose: int=0, device: Optional[str]=None, **kwargs):
+        del kwargs
+        super().__init__()
+        norm_class = NORM_CLASS_REGISTRY[norm_type.lower()]
+        attn_class = ATTN_CLASS_REGISTRY[attn_config['attn_type']]
+        self.norm_1 = norm_class(d_model, device=device)
+        self.attn = attn_class(attn_impl=attn_config['attn_impl'], clip_qkv=attn_config['clip_qkv'], qk_ln=attn_config['qk_ln'], softmax_scale=attn_config['softmax_scale'], attn_pdrop=attn_config['attn_pdrop'], d_model=d_model, n_heads=n_heads, verbose=verbose, device=device)
+        self.norm_2 = norm_class(d_model, device=device)
+        self.ffn = MPTMLP(d_model=d_model, expansion_ratio=expansion_ratio, device=device)
+        self.resid_attn_dropout = nn.Dropout(resid_pdrop)
+        self.resid_ffn_dropout = nn.Dropout(resid_pdrop)
+    def forward(self, x: torch.Tensor, past_key_value: Optional[Tuple[torch.Tensor]]=None, attn_bias: Optional[torch.Tensor]=None, attention_mask: Optional[torch.ByteTensor]=None, is_causal: bool=True) -> Tuple[torch.Tensor, Optional[Tuple[torch.Tensor]]]:
+        a = self.norm_1(x)
+        (b, attn_weights, past_key_value) = self.attn(a, past_key_value=past_key_value, attn_bias=attn_bias, attention_mask=attention_mask, is_causal=is_causal)
+        x = x + self.resid_attn_dropout(b)
+        m = self.norm_2(x)
+        n = self.ffn(m)
+        x = x + self.resid_ffn_dropout(n)
+        return (x, attn_weights, past_key_value)

dam/model/language_model/mpt_ignored/configuration_mpt.py ADDED Viewed

	@@ -0,0 +1,118 @@

+"""A HuggingFace-style model configuration."""
+from typing import Dict, Optional, Union
+from transformers import PretrainedConfig
+attn_config_defaults: Dict = {'attn_type': 'multihead_attention', 'attn_pdrop': 0.0, 'attn_impl': 'triton', 'qk_ln': False, 'clip_qkv': None, 'softmax_scale': None, 'prefix_lm': False, 'attn_uses_sequence_id': False, 'alibi': False, 'alibi_bias_max': 8}
+init_config_defaults: Dict = {'name': 'kaiming_normal_', 'fan_mode': 'fan_in', 'init_nonlinearity': 'relu', 'init_div_is_residual': True, 'emb_init_std': None, 'emb_init_uniform_lim': None, 'init_std': None, 'init_gain': 0.0}
+class MPTConfig(PretrainedConfig):
+    model_type = 'mpt'
+    def __init__(self, d_model: int=2048, n_heads: int=16, n_layers: int=24, expansion_ratio: int=4, max_seq_len: int=2048, vocab_size: int=50368, resid_pdrop: float=0.0, emb_pdrop: float=0.0, learned_pos_emb: bool=True, attn_config: Dict=attn_config_defaults, init_device: str='cpu', logit_scale: Optional[Union[float, str]]=None, no_bias: bool=False, verbose: int=0, embedding_fraction: float=1.0, norm_type: str='low_precision_layernorm', use_cache: bool=False, init_config: Dict=init_config_defaults, **kwargs):
+        """The MPT configuration class.
+        Args:
+            d_model (int): The size of the embedding dimension of the model.
+            n_heads (int): The number of attention heads.
+            n_layers (int): The number of layers in the model.
+            expansion_ratio (int): The ratio of the up/down scale in the MLP.
+            max_seq_len (int): The maximum sequence length of the model.
+            vocab_size (int): The size of the vocabulary.
+            resid_pdrop (float): The dropout probability applied to the attention output before combining with residual.
+            emb_pdrop (float): The dropout probability for the embedding layer.
+            learned_pos_emb (bool): Whether to use learned positional embeddings
+            attn_config (Dict):  A dictionary used to configure the model's attention module:
+                attn_type (str): type of attention to use. Options: multihead_attention, multiquery_attention
+                attn_pdrop (float): The dropout probability for the attention layers.
+                attn_impl (str): The attention implementation to use. One of 'torch', 'flash', or 'triton'.
+                qk_ln (bool): Whether to apply layer normalization to the queries and keys in the attention layer.
+                clip_qkv (Optional[float]): If not None, clip the queries, keys, and values in the attention layer to
+                    this value.
+                softmax_scale (Optional[float]): If not None, scale the softmax in the attention layer by this value. If None,
+                    use the default scale of ``1/sqrt(d_keys)``.
+                prefix_lm (Optional[bool]): Whether the model should operate as a Prefix LM. This requires passing an
+                    extra `prefix_mask` argument which indicates which tokens belong to the prefix. Tokens in the prefix
+                    can attend to one another bi-directionally. Tokens outside the prefix use causal attention.
+                attn_uses_sequence_id (Optional[bool]): Whether to restrict attention to tokens that have the same sequence_id.
+                    When the model is in `train` mode, this requires passing an extra `sequence_id` argument which indicates
+                    which sub-sequence each token belongs to.
+                    Defaults to ``False`` meaning any provided `sequence_id` will be ignored.
+                alibi (bool): Whether to use the alibi bias instead of position embeddings.
+                alibi_bias_max (int): The maximum value of the alibi bias.
+            init_device (str): The device to use for parameter initialization.
+            logit_scale (Optional[Union[float, str]]): If not None, scale the logits by this value.
+            no_bias (bool): Whether to use bias in all layers.
+            verbose (int): The verbosity level. 0 is silent.
+            embedding_fraction (float): The fraction to scale the gradients of the embedding layer by.
+            norm_type (str): choose type of norm to use
+            multiquery_attention (bool): Whether to use multiquery attention implementation.
+            use_cache (bool): Whether or not the model should return the last key/values attentions
+            init_config (Dict): A dictionary used to configure the model initialization:
+                init_config.name: The parameter initialization scheme to use. Options: 'default_', 'baseline_',
+                    'kaiming_uniform_', 'kaiming_normal_', 'neox_init_', 'small_init_', 'xavier_uniform_', or
+                    'xavier_normal_'. These mimic the parameter initialization methods in PyTorch.
+                init_div_is_residual (Union[int, float, str, bool]): Value to divide initial weights by if ``module._is_residual`` is True.
+                emb_init_std (Optional[float]): The standard deviation of the normal distribution used to initialize the embedding layer.
+                emb_init_uniform_lim (Optional[Union[Tuple[float, float], float]]): The lower and upper limits of the uniform distribution
+                    used to initialize the embedding layer. Mutually exclusive with ``emb_init_std``.
+                init_std (float): The standard deviation of the normal distribution used to initialize the model,
+                    if using the baseline_ parameter initialization scheme.
+                init_gain (float): The gain to use for parameter initialization with kaiming or xavier initialization schemes.
+                fan_mode (str): The fan mode to use for parameter initialization with kaiming initialization schemes.
+                init_nonlinearity (str): The nonlinearity to use for parameter initialization with kaiming initialization schemes.
+                ---
+                See llmfoundry.models.utils.param_init_fns.py for info on other param init config options
+        """
+        self.d_model = d_model
+        self.n_heads = n_heads
+        self.n_layers = n_layers
+        self.expansion_ratio = expansion_ratio
+        self.max_seq_len = max_seq_len
+        self.vocab_size = vocab_size
+        self.resid_pdrop = resid_pdrop
+        self.emb_pdrop = emb_pdrop
+        self.learned_pos_emb = learned_pos_emb
+        self.attn_config = attn_config
+        self.init_device = init_device
+        self.logit_scale = logit_scale
+        self.no_bias = no_bias
+        self.verbose = verbose
+        self.embedding_fraction = embedding_fraction
+        self.norm_type = norm_type
+        self.use_cache = use_cache
+        self.init_config = init_config
+        if 'name' in kwargs:
+            del kwargs['name']
+        if 'loss_fn' in kwargs:
+            del kwargs['loss_fn']
+        super().__init__(**kwargs)
+        self._validate_config()
+    def _set_config_defaults(self, config, config_defaults):
+        for (k, v) in config_defaults.items():
+            if k not in config:
+                config[k] = v
+        return config
+    def _validate_config(self):
+        self.attn_config = self._set_config_defaults(self.attn_config, attn_config_defaults)
+        self.init_config = self._set_config_defaults(self.init_config, init_config_defaults)
+        if self.d_model % self.n_heads != 0:
+            raise ValueError('d_model must be divisible by n_heads')
+        if any((prob < 0 or prob > 1 for prob in [self.attn_config['attn_pdrop'], self.resid_pdrop, self.emb_pdrop])):
+            raise ValueError("self.attn_config['attn_pdrop'], resid_pdrop, emb_pdrop are probabilities and must be between 0 and 1")
+        if self.attn_config['attn_impl'] not in ['torch', 'flash', 'triton']:
+            raise ValueError(f"Unknown attn_impl={self.attn_config['attn_impl']}")
+        if self.attn_config['prefix_lm'] and self.attn_config['attn_impl'] not in ['torch', 'triton']:
+            raise NotImplementedError('prefix_lm only implemented with torch and triton attention.')
+        if self.attn_config['alibi'] and self.attn_config['attn_impl'] not in ['torch', 'triton']:
+            raise NotImplementedError('alibi only implemented with torch and triton attention.')
+        if self.attn_config['attn_uses_sequence_id'] and self.attn_config['attn_impl'] not in ['torch', 'triton']:
+            raise NotImplementedError('attn_uses_sequence_id only implemented with torch and triton attention.')
+        if self.embedding_fraction > 1 or self.embedding_fraction <= 0:
+            raise ValueError('model.embedding_fraction must be between 0 (exclusive) and 1 (inclusive)!')
+        if isinstance(self.logit_scale, str) and self.logit_scale != 'inv_sqrt_d_model':
+            raise ValueError(f"self.logit_scale={self.logit_scale!r} is not recognized as an option; use numeric value or 'inv_sqrt_d_model'.")
+        if self.init_config.get('name', None) is None:
+            raise ValueError(f"self.init_config={self.init_config!r} 'name' needs to be set.")
+        if not self.learned_pos_emb and (not self.attn_config['alibi']):
+            raise ValueError(f'Positional information must be provided to the model using either learned_pos_emb or alibi.')

dam/model/language_model/mpt_ignored/custom_embedding.py ADDED Viewed

	@@ -0,0 +1,11 @@

+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from torch import Tensor
+class SharedEmbedding(nn.Embedding):
+    def forward(self, input: Tensor, unembed: bool=False) -> Tensor:
+        if unembed:
+            return F.linear(input, self.weight)
+        return super().forward(input)

dam/model/language_model/mpt_ignored/flash_attn_triton.py ADDED Viewed

	@@ -0,0 +1,484 @@

+"""
+Copied from https://github.com/HazyResearch/flash-attention/blob/eff9fe6b8076df59d64d7a3f464696738a3c7c24/flash_attn/flash_attn_triton.py
+update imports to use 'triton_pre_mlir'
+*Experimental* implementation of FlashAttention in Triton.
+Tested with triton==2.0.0.dev20221202.
+Triton 2.0 has a new backend (MLIR) but seems like it doesn't yet work for head dimensions
+other than 64:
+https://github.com/openai/triton/blob/d376020f90002757eea3ea9475d4f7cfc2ec5ead/python/triton/ops/flash_attention.py#L207
+We'll update this implementation with the new Triton backend once this is fixed.
+We use the FlashAttention implementation from Phil Tillet a starting point.
+https://github.com/openai/triton/blob/master/python/tutorials/06-fused-attention.py
+Changes:
+- Implement both causal and non-causal attention.
+- Implement both self-attention and cross-attention.
+- Support arbitrary seqlens (not just multiples of 128), for both forward and backward.
+- Support all head dimensions up to 128 (not just 16, 32, 64, 128), for both forward and backward.
+- Support attention bias.
+- Speed up the forward pass a bit, and only store the LSE instead of m and l.
+- Make the backward for d=128 much faster by reducing register spilling.
+- Optionally parallelize the backward pass across seqlen_k, to deal with the case of
+small batch size * nheads.
+Caution:
+- This is an *experimental* implementation. The forward pass should be quite robust but
+I'm not 100% sure that the backward pass doesn't have race conditions (due to the Triton compiler).
+- This implementation has only been tested on A100.
+- If you plan to use headdim other than 64 and 128, you should test for race conditions
+(due to the Triton compiler), as done in tests/test_flash_attn.py
+"test_flash_attn_triton_race_condition". I've tested and fixed many race conditions
+for different head dimensions (40, 48, 64, 128, 80, 88, 96), but I'm still not 100% confident
+that there are none left for other head dimensions.
+Differences between this Triton version and the CUDA version:
+- Triton version doesn't support dropout.
+- Triton forward is generally faster than CUDA forward, while Triton backward is
+generally slower than CUDA backward. Overall Triton forward + backward is slightly slower
+than CUDA forward + backward.
+- Triton version doesn't support different sequence lengths in a batch (i.e., RaggedTensor/NestedTensor).
+- Triton version supports attention bias, while CUDA version doesn't.
+"""
+import math
+import torch
+import triton_pre_mlir as triton
+import triton_pre_mlir.language as tl
+@triton.heuristics({'EVEN_M': lambda args: args['seqlen_q'] % args['BLOCK_M'] == 0, 'EVEN_N': lambda args: args['seqlen_k'] % args['BLOCK_N'] == 0, 'EVEN_HEADDIM': lambda args: args['headdim'] == args['BLOCK_HEADDIM']})
+@triton.jit
+def _fwd_kernel(Q, K, V, Bias, Out, Lse, TMP, softmax_scale, stride_qb, stride_qh, stride_qm, stride_kb, stride_kh, stride_kn, stride_vb, stride_vh, stride_vn, stride_bb, stride_bh, stride_bm, stride_ob, stride_oh, stride_om, nheads, seqlen_q, seqlen_k, seqlen_q_rounded, headdim, CACHE_KEY_SEQLEN_Q, CACHE_KEY_SEQLEN_K, BIAS_TYPE: tl.constexpr, IS_CAUSAL: tl.constexpr, BLOCK_HEADDIM: tl.constexpr, EVEN_M: tl.constexpr, EVEN_N: tl.constexpr, EVEN_HEADDIM: tl.constexpr, BLOCK_M: tl.constexpr, BLOCK_N: tl.constexpr):
+    start_m = tl.program_id(0)
+    off_hb = tl.program_id(1)
+    off_b = off_hb // nheads
+    off_h = off_hb % nheads
+    offs_m = start_m * BLOCK_M + tl.arange(0, BLOCK_M)
+    offs_n = tl.arange(0, BLOCK_N)
+    offs_d = tl.arange(0, BLOCK_HEADDIM)
+    q_ptrs = Q + off_b * stride_qb + off_h * stride_qh + (offs_m[:, None] * stride_qm + offs_d[None, :])
+    k_ptrs = K + off_b * stride_kb + off_h * stride_kh + (offs_n[:, None] * stride_kn + offs_d[None, :])
+    v_ptrs = V + off_b * stride_vb + off_h * stride_vh + (offs_n[:, None] * stride_vn + offs_d[None, :])
+    if BIAS_TYPE == 'vector':
+        b_ptrs = Bias + off_b * stride_bb + off_h * stride_bh + offs_n
+    elif BIAS_TYPE == 'matrix':
+        b_ptrs = Bias + off_b * stride_bb + off_h * stride_bh + (offs_m[:, None] * stride_bm + offs_n[None, :])
+    t_ptrs = TMP + off_hb * seqlen_q_rounded + offs_m
+    lse_i = tl.zeros([BLOCK_M], dtype=tl.float32) - float('inf')
+    m_i = tl.zeros([BLOCK_M], dtype=tl.float32) - float('inf')
+    acc_o = tl.zeros([BLOCK_M, BLOCK_HEADDIM], dtype=tl.float32)
+    if EVEN_M & EVEN_N:
+        if EVEN_HEADDIM:
+            q = tl.load(q_ptrs)
+        else:
+            q = tl.load(q_ptrs, mask=offs_d[None, :] < headdim, other=0.0)
+    elif EVEN_HEADDIM:
+        q = tl.load(q_ptrs, mask=offs_m[:, None] < seqlen_q, other=0.0)
+    else:
+        q = tl.load(q_ptrs, mask=(offs_m[:, None] < seqlen_q) & (offs_d[None, :] < headdim), other=0.0)
+    end_n = seqlen_k if not IS_CAUSAL else tl.minimum((start_m + 1) * BLOCK_M, seqlen_k)
+    for start_n in range(0, end_n, BLOCK_N):
+        start_n = tl.multiple_of(start_n, BLOCK_N)
+        if EVEN_N & EVEN_M:
+            if EVEN_HEADDIM:
+                k = tl.load(k_ptrs + start_n * stride_kn)
+            else:
+                k = tl.load(k_ptrs + start_n * stride_kn, mask=offs_d[None, :] < headdim, other=0.0)
+        elif EVEN_HEADDIM:
+            k = tl.load(k_ptrs + start_n * stride_kn, mask=(start_n + offs_n)[:, None] < seqlen_k, other=0.0)
+        else:
+            k = tl.load(k_ptrs + start_n * stride_kn, mask=((start_n + offs_n)[:, None] < seqlen_k) & (offs_d[None, :] < headdim), other=0.0)
+        qk = tl.zeros([BLOCK_M, BLOCK_N], dtype=tl.float32)
+        qk += tl.dot(q, k, trans_b=True)
+        if not EVEN_N:
+            qk += tl.where((start_n + offs_n)[None, :] < seqlen_k, 0, float('-inf'))
+        if IS_CAUSAL:
+            qk += tl.where(offs_m[:, None] >= (start_n + offs_n)[None, :], 0, float('-inf'))
+        if BIAS_TYPE != 'none':
+            if BIAS_TYPE == 'vector':
+                if EVEN_N:
+                    bias = tl.load(b_ptrs + start_n).to(tl.float32)
+                else:
+                    bias = tl.load(b_ptrs + start_n, mask=start_n + offs_n < seqlen_k, other=0.0).to(tl.float32)
+                bias = bias[None, :]
+            elif BIAS_TYPE == 'matrix':
+                if EVEN_M & EVEN_N:
+                    bias = tl.load(b_ptrs + start_n).to(tl.float32)
+                else:
+                    bias = tl.load(b_ptrs + start_n, mask=(offs_m[:, None] < seqlen_q) & ((start_n + offs_n)[None, :] < seqlen_k), other=0.0).to(tl.float32)
+            qk = qk * softmax_scale + bias
+            m_ij = tl.maximum(tl.max(qk, 1), lse_i)
+            p = tl.exp(qk - m_ij[:, None])
+        else:
+            m_ij = tl.maximum(tl.max(qk, 1) * softmax_scale, lse_i)
+            p = tl.exp(qk * softmax_scale - m_ij[:, None])
+        l_ij = tl.sum(p, 1)
+        acc_o_scale = tl.exp(m_i - m_ij)
+        tl.store(t_ptrs, acc_o_scale)
+        acc_o_scale = tl.load(t_ptrs)
+        acc_o = acc_o * acc_o_scale[:, None]
+        if EVEN_N & EVEN_M:
+            if EVEN_HEADDIM:
+                v = tl.load(v_ptrs + start_n * stride_vn)
+            else:
+                v = tl.load(v_ptrs + start_n * stride_vn, mask=offs_d[None, :] < headdim, other=0.0)
+        elif EVEN_HEADDIM:
+            v = tl.load(v_ptrs + start_n * stride_vn, mask=(start_n + offs_n)[:, None] < seqlen_k, other=0.0)
+        else:
+            v = tl.load(v_ptrs + start_n * stride_vn, mask=((start_n + offs_n)[:, None] < seqlen_k) & (offs_d[None, :] < headdim), other=0.0)
+        p = p.to(v.dtype)
+        acc_o += tl.dot(p, v)
+        m_i = m_ij
+        l_i_new = tl.exp(lse_i - m_ij) + l_ij
+        lse_i = m_ij + tl.log(l_i_new)
+    o_scale = tl.exp(m_i - lse_i)
+    tl.store(t_ptrs, o_scale)
+    o_scale = tl.load(t_ptrs)
+    acc_o = acc_o * o_scale[:, None]
+    start_m = tl.program_id(0)
+    offs_m = start_m * BLOCK_M + tl.arange(0, BLOCK_M)
+    lse_ptrs = Lse + off_hb * seqlen_q_rounded + offs_m
+    tl.store(lse_ptrs, lse_i)
+    offs_d = tl.arange(0, BLOCK_HEADDIM)
+    out_ptrs = Out + off_b * stride_ob + off_h * stride_oh + (offs_m[:, None] * stride_om + offs_d[None, :])
+    if EVEN_M:
+        if EVEN_HEADDIM:
+            tl.store(out_ptrs, acc_o)
+        else:
+            tl.store(out_ptrs, acc_o, mask=offs_d[None, :] < headdim)
+    elif EVEN_HEADDIM:
+        tl.store(out_ptrs, acc_o, mask=offs_m[:, None] < seqlen_q)
+    else:
+        tl.store(out_ptrs, acc_o, mask=(offs_m[:, None] < seqlen_q) & (offs_d[None, :] < headdim))
+@triton.jit
+def _bwd_preprocess_do_o_dot(Out, DO, Delta, stride_ob, stride_oh, stride_om, stride_dob, stride_doh, stride_dom, nheads, seqlen_q, seqlen_q_rounded, headdim, BLOCK_M: tl.constexpr, BLOCK_HEADDIM: tl.constexpr):
+    start_m = tl.program_id(0)
+    off_hb = tl.program_id(1)
+    off_b = off_hb // nheads
+    off_h = off_hb % nheads
+    offs_m = start_m * BLOCK_M + tl.arange(0, BLOCK_M)
+    offs_d = tl.arange(0, BLOCK_HEADDIM)
+    o = tl.load(Out + off_b * stride_ob + off_h * stride_oh + offs_m[:, None] * stride_om + offs_d[None, :], mask=(offs_m[:, None] < seqlen_q) & (offs_d[None, :] < headdim), other=0.0).to(tl.float32)
+    do = tl.load(DO + off_b * stride_dob + off_h * stride_doh + offs_m[:, None] * stride_dom + offs_d[None, :], mask=(offs_m[:, None] < seqlen_q) & (offs_d[None, :] < headdim), other=0.0).to(tl.float32)
+    delta = tl.sum(o * do, axis=1)
+    tl.store(Delta + off_hb * seqlen_q_rounded + offs_m, delta)
+@triton.jit
+def _bwd_store_dk_dv(dk_ptrs, dv_ptrs, dk, dv, offs_n, offs_d, seqlen_k, headdim, EVEN_M: tl.constexpr, EVEN_N: tl.constexpr, EVEN_HEADDIM: tl.constexpr):
+    if EVEN_N & EVEN_M:
+        if EVEN_HEADDIM:
+            tl.store(dv_ptrs, dv)
+            tl.store(dk_ptrs, dk)
+        else:
+            tl.store(dv_ptrs, dv, mask=offs_d[None, :] < headdim)
+            tl.store(dk_ptrs, dk, mask=offs_d[None, :] < headdim)
+    elif EVEN_HEADDIM:
+        tl.store(dv_ptrs, dv, mask=offs_n[:, None] < seqlen_k)
+        tl.store(dk_ptrs, dk, mask=offs_n[:, None] < seqlen_k)
+    else:
+        tl.store(dv_ptrs, dv, mask=(offs_n[:, None] < seqlen_k) & (offs_d[None, :] < headdim))
+        tl.store(dk_ptrs, dk, mask=(offs_n[:, None] < seqlen_k) & (offs_d[None, :] < headdim))
+@triton.jit
+def _bwd_kernel_one_col_block(start_n, Q, K, V, Bias, DO, DQ, DK, DV, LSE, D, softmax_scale, stride_qm, stride_kn, stride_vn, stride_bm, stride_dom, stride_dqm, stride_dkn, stride_dvn, seqlen_q, seqlen_k, headdim, ATOMIC_ADD: tl.constexpr, BIAS_TYPE: tl.constexpr, IS_CAUSAL: tl.constexpr, BLOCK_HEADDIM: tl.constexpr, EVEN_M: tl.constexpr, EVEN_N: tl.constexpr, EVEN_HEADDIM: tl.constexpr, BLOCK_M: tl.constexpr, BLOCK_N: tl.constexpr):
+    begin_m = 0 if not IS_CAUSAL else start_n * BLOCK_N // BLOCK_M * BLOCK_M
+    offs_qm = begin_m + tl.arange(0, BLOCK_M)
+    offs_n = start_n * BLOCK_N + tl.arange(0, BLOCK_N)
+    offs_m = tl.arange(0, BLOCK_M)
+    offs_d = tl.arange(0, BLOCK_HEADDIM)
+    q_ptrs = Q + (offs_qm[:, None] * stride_qm + offs_d[None, :])
+    k_ptrs = K + (offs_n[:, None] * stride_kn + offs_d[None, :])
+    v_ptrs = V + (offs_n[:, None] * stride_vn + offs_d[None, :])
+    do_ptrs = DO + (offs_qm[:, None] * stride_dom + offs_d[None, :])
+    dq_ptrs = DQ + (offs_qm[:, None] * stride_dqm + offs_d[None, :])
+    if BIAS_TYPE == 'vector':
+        b_ptrs = Bias + offs_n
+    elif BIAS_TYPE == 'matrix':
+        b_ptrs = Bias + (offs_qm[:, None] * stride_bm + offs_n[None, :])
+    dv = tl.zeros([BLOCK_N, BLOCK_HEADDIM], dtype=tl.float32)
+    dk = tl.zeros([BLOCK_N, BLOCK_HEADDIM], dtype=tl.float32)
+    if begin_m >= seqlen_q:
+        dv_ptrs = DV + (offs_n[:, None] * stride_dvn + offs_d[None, :])
+        dk_ptrs = DK + (offs_n[:, None] * stride_dkn + offs_d[None, :])
+        _bwd_store_dk_dv(dk_ptrs, dv_ptrs, dk, dv, offs_n, offs_d, seqlen_k, headdim, EVEN_M=EVEN_M, EVEN_N=EVEN_N, EVEN_HEADDIM=EVEN_HEADDIM)
+        return
+    if EVEN_N & EVEN_M:
+        if EVEN_HEADDIM:
+            k = tl.load(k_ptrs)
+            v = tl.load(v_ptrs)
+        else:
+            k = tl.load(k_ptrs, mask=offs_d[None, :] < headdim, other=0.0)
+            v = tl.load(v_ptrs, mask=offs_d[None, :] < headdim, other=0.0)
+    elif EVEN_HEADDIM:
+        k = tl.load(k_ptrs, mask=offs_n[:, None] < seqlen_k, other=0.0)
+        v = tl.load(v_ptrs, mask=offs_n[:, None] < seqlen_k, other=0.0)
+    else:
+        k = tl.load(k_ptrs, mask=(offs_n[:, None] < seqlen_k) & (offs_d[None, :] < headdim), other=0.0)
+        v = tl.load(v_ptrs, mask=(offs_n[:, None] < seqlen_k) & (offs_d[None, :] < headdim), other=0.0)
+    num_block_m = tl.cdiv(seqlen_q, BLOCK_M)
+    for start_m in range(begin_m, num_block_m * BLOCK_M, BLOCK_M):
+        start_m = tl.multiple_of(start_m, BLOCK_M)
+        offs_m_curr = start_m + offs_m
+        if EVEN_M & EVEN_HEADDIM:
+            q = tl.load(q_ptrs)
+        elif EVEN_HEADDIM:
+            q = tl.load(q_ptrs, mask=offs_m_curr[:, None] < seqlen_q, other=0.0)
+        else:
+            q = tl.load(q_ptrs, mask=(offs_m_curr[:, None] < seqlen_q) & (offs_d[None, :] < headdim), other=0.0)
+        qk = tl.dot(q, k, trans_b=True)
+        if not EVEN_N:
+            qk = tl.where(offs_n[None, :] < seqlen_k, qk, float('-inf'))
+        if IS_CAUSAL:
+            qk = tl.where(offs_m_curr[:, None] >= offs_n[None, :], qk, float('-inf'))
+        if BIAS_TYPE != 'none':
+            tl.debug_barrier()
+            if BIAS_TYPE == 'vector':
+                if EVEN_N:
+                    bias = tl.load(b_ptrs).to(tl.float32)
+                else:
+                    bias = tl.load(b_ptrs, mask=offs_n < seqlen_k, other=0.0).to(tl.float32)
+                bias = bias[None, :]
+            elif BIAS_TYPE == 'matrix':
+                if EVEN_M & EVEN_N:
+                    bias = tl.load(b_ptrs).to(tl.float32)
+                else:
+                    bias = tl.load(b_ptrs, mask=(offs_m_curr[:, None] < seqlen_q) & (offs_n[None, :] < seqlen_k), other=0.0).to(tl.float32)
+            qk = qk * softmax_scale + bias
+        if not EVEN_M & EVEN_HEADDIM:
+            tl.debug_barrier()
+        lse_i = tl.load(LSE + offs_m_curr)
+        if BIAS_TYPE == 'none':
+            p = tl.exp(qk * softmax_scale - lse_i[:, None])
+        else:
+            p = tl.exp(qk - lse_i[:, None])
+        if EVEN_M & EVEN_HEADDIM:
+            do = tl.load(do_ptrs)
+        else:
+            do = tl.load(do_ptrs, mask=(offs_m_curr[:, None] < seqlen_q) & (offs_d[None, :] < headdim), other=0.0)
+        dv += tl.dot(p.to(do.dtype), do, trans_a=True)
+        if not EVEN_M & EVEN_HEADDIM:
+            tl.debug_barrier()
+        dp = tl.dot(do, v, trans_b=True)
+        if not EVEN_HEADDIM:
+            tl.debug_barrier()
+        Di = tl.load(D + offs_m_curr)
+        ds = (p * (dp - Di[:, None]) * softmax_scale).to(q.dtype)
+        dk += tl.dot(ds, q, trans_a=True)
+        if not EVEN_M & EVEN_HEADDIM:
+            tl.debug_barrier()
+        if not ATOMIC_ADD:
+            if EVEN_M & EVEN_HEADDIM:
+                dq = tl.load(dq_ptrs, eviction_policy='evict_last')
+                dq += tl.dot(ds, k)
+                tl.store(dq_ptrs, dq, eviction_policy='evict_last')
+            elif EVEN_HEADDIM:
+                dq = tl.load(dq_ptrs, mask=offs_m_curr[:, None] < seqlen_q, other=0.0, eviction_policy='evict_last')
+                dq += tl.dot(ds, k)
+                tl.store(dq_ptrs, dq, mask=offs_m_curr[:, None] < seqlen_q, eviction_policy='evict_last')
+            else:
+                dq = tl.load(dq_ptrs, mask=(offs_m_curr[:, None] < seqlen_q) & (offs_d[None, :] < headdim), other=0.0, eviction_policy='evict_last')
+                dq += tl.dot(ds, k)
+                tl.store(dq_ptrs, dq, mask=(offs_m_curr[:, None] < seqlen_q) & (offs_d[None, :] < headdim), eviction_policy='evict_last')
+        else:
+            dq = tl.dot(ds, k)
+            if EVEN_M & EVEN_HEADDIM:
+                tl.atomic_add(dq_ptrs, dq)
+            elif EVEN_HEADDIM:
+                tl.atomic_add(dq_ptrs, dq, mask=offs_m_curr[:, None] < seqlen_q)
+            else:
+                tl.atomic_add(dq_ptrs, dq, mask=(offs_m_curr[:, None] < seqlen_q) & (offs_d[None, :] < headdim))
+        dq_ptrs += BLOCK_M * stride_dqm
+        q_ptrs += BLOCK_M * stride_qm
+        do_ptrs += BLOCK_M * stride_dom
+        if BIAS_TYPE == 'matrix':
+            b_ptrs += BLOCK_M * stride_bm
+    dv_ptrs = DV + (offs_n[:, None] * stride_dvn + offs_d[None, :])
+    dk_ptrs = DK + (offs_n[:, None] * stride_dkn + offs_d[None, :])
+    _bwd_store_dk_dv(dk_ptrs, dv_ptrs, dk, dv, offs_n, offs_d, seqlen_k, headdim, EVEN_M=EVEN_M, EVEN_N=EVEN_N, EVEN_HEADDIM=EVEN_HEADDIM)
+def init_to_zero(name):
+    return lambda nargs: nargs[name].zero_()
+@triton.autotune(configs=[triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'SEQUENCE_PARALLEL': False}, num_warps=8, num_stages=1, pre_hook=init_to_zero('DQ')), triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'SEQUENCE_PARALLEL': True}, num_warps=8, num_stages=1, pre_hook=init_to_zero('DQ'))], key=['CACHE_KEY_SEQLEN_Q', 'CACHE_KEY_SEQLEN_K', 'BIAS_TYPE', 'IS_CAUSAL', 'BLOCK_HEADDIM'])
+@triton.heuristics({'EVEN_M': lambda args: args['seqlen_q'] % args['BLOCK_M'] == 0, 'EVEN_N': lambda args: args['seqlen_k'] % args['BLOCK_N'] == 0, 'EVEN_HEADDIM': lambda args: args['headdim'] == args['BLOCK_HEADDIM']})
+@triton.jit
+def _bwd_kernel(Q, K, V, Bias, DO, DQ, DK, DV, LSE, D, softmax_scale, stride_qb, stride_qh, stride_qm, stride_kb, stride_kh, stride_kn, stride_vb, stride_vh, stride_vn, stride_bb, stride_bh, stride_bm, stride_dob, stride_doh, stride_dom, stride_dqb, stride_dqh, stride_dqm, stride_dkb, stride_dkh, stride_dkn, stride_dvb, stride_dvh, stride_dvn, nheads, seqlen_q, seqlen_k, seqlen_q_rounded, headdim, CACHE_KEY_SEQLEN_Q, CACHE_KEY_SEQLEN_K, BIAS_TYPE: tl.constexpr, IS_CAUSAL: tl.constexpr, BLOCK_HEADDIM: tl.constexpr, SEQUENCE_PARALLEL: tl.constexpr, EVEN_M: tl.constexpr, EVEN_N: tl.constexpr, EVEN_HEADDIM: tl.constexpr, BLOCK_M: tl.constexpr, BLOCK_N: tl.constexpr):
+    off_hb = tl.program_id(1)
+    off_b = off_hb // nheads
+    off_h = off_hb % nheads
+    Q += off_b * stride_qb + off_h * stride_qh
+    K += off_b * stride_kb + off_h * stride_kh
+    V += off_b * stride_vb + off_h * stride_vh
+    DO += off_b * stride_dob + off_h * stride_doh
+    DQ += off_b * stride_dqb + off_h * stride_dqh
+    DK += off_b * stride_dkb + off_h * stride_dkh
+    DV += off_b * stride_dvb + off_h * stride_dvh
+    if BIAS_TYPE != 'none':
+        Bias += off_b * stride_bb + off_h * stride_bh
+    D += off_hb * seqlen_q_rounded
+    LSE += off_hb * seqlen_q_rounded
+    if not SEQUENCE_PARALLEL:
+        num_block_n = tl.cdiv(seqlen_k, BLOCK_N)
+        for start_n in range(0, num_block_n):
+            _bwd_kernel_one_col_block(start_n, Q, K, V, Bias, DO, DQ, DK, DV, LSE, D, softmax_scale, stride_qm, stride_kn, stride_vn, stride_bm, stride_dom, stride_dqm, stride_dkn, stride_dvn, seqlen_q, seqlen_k, headdim, ATOMIC_ADD=False, BIAS_TYPE=BIAS_TYPE, IS_CAUSAL=IS_CAUSAL, BLOCK_HEADDIM=BLOCK_HEADDIM, EVEN_M=EVEN_M, EVEN_N=EVEN_N, EVEN_HEADDIM=EVEN_HEADDIM, BLOCK_M=BLOCK_M, BLOCK_N=BLOCK_N)
+    else:
+        start_n = tl.program_id(0)
+        _bwd_kernel_one_col_block(start_n, Q, K, V, Bias, DO, DQ, DK, DV, LSE, D, softmax_scale, stride_qm, stride_kn, stride_vn, stride_bm, stride_dom, stride_dqm, stride_dkn, stride_dvn, seqlen_q, seqlen_k, headdim, ATOMIC_ADD=True, BIAS_TYPE=BIAS_TYPE, IS_CAUSAL=IS_CAUSAL, BLOCK_HEADDIM=BLOCK_HEADDIM, EVEN_M=EVEN_M, EVEN_N=EVEN_N, EVEN_HEADDIM=EVEN_HEADDIM, BLOCK_M=BLOCK_M, BLOCK_N=BLOCK_N)
+def _flash_attn_forward(q, k, v, bias=None, causal=False, softmax_scale=None):
+    (batch, seqlen_q, nheads, d) = q.shape
+    (_, seqlen_k, _, _) = k.shape
+    assert k.shape == (batch, seqlen_k, nheads, d)
+    assert v.shape == (batch, seqlen_k, nheads, d)
+    assert d <= 128, 'FlashAttention only support head dimensions up to 128'
+    assert q.dtype == k.dtype == v.dtype, 'All tensors must have the same type'
+    assert q.dtype in [torch.float16, torch.bfloat16], 'Only support fp16 and bf16'
+    assert q.is_cuda and k.is_cuda and v.is_cuda
+    softmax_scale = softmax_scale or 1.0 / math.sqrt(d)
+    has_bias = bias is not None
+    bias_type = 'none'
+    if has_bias:
+        assert bias.dtype in [q.dtype, torch.float]
+        assert bias.is_cuda
+        assert bias.dim() == 4
+        if bias.stride(-1) != 1:
+            bias = bias.contiguous()
+        if bias.shape[2:] == (1, seqlen_k):
+            bias_type = 'vector'
+        elif bias.shape[2:] == (seqlen_q, seqlen_k):
+            bias_type = 'matrix'
+        else:
+            raise RuntimeError('Last 2 dimensions of bias must be (1, seqlen_k) or (seqlen_q, seqlen_k)')
+        bias = bias.expand(batch, nheads, seqlen_q, seqlen_k)
+    bias_strides = (bias.stride(0), bias.stride(1), bias.stride(2)) if has_bias else (0, 0, 0)
+    seqlen_q_rounded = math.ceil(seqlen_q / 128) * 128
+    lse = torch.empty((batch, nheads, seqlen_q_rounded), device=q.device, dtype=torch.float32)
+    tmp = torch.empty((batch, nheads, seqlen_q_rounded), device=q.device, dtype=torch.float32)
+    o = torch.empty_like(q)
+    BLOCK_HEADDIM = max(triton.next_power_of_2(d), 16)
+    BLOCK = 128
+    num_warps = 4 if d <= 64 else 8
+    grid = lambda META: (triton.cdiv(seqlen_q, META['BLOCK_M']), batch * nheads)
+    _fwd_kernel[grid](q, k, v, bias, o, lse, tmp, softmax_scale, q.stride(0), q.stride(2), q.stride(1), k.stride(0), k.stride(2), k.stride(1), v.stride(0), v.stride(2), v.stride(1), *bias_strides, o.stride(0), o.stride(2), o.stride(1), nheads, seqlen_q, seqlen_k, seqlen_q_rounded, d, seqlen_q // 32, seqlen_k // 32, bias_type, causal, BLOCK_HEADDIM, BLOCK_M=BLOCK, BLOCK_N=BLOCK, num_warps=num_warps, num_stages=1)
+    return (o, lse, softmax_scale)
+def _flash_attn_backward(do, q, k, v, o, lse, dq, dk, dv, bias=None, causal=False, softmax_scale=None):
+    if do.stride(-1) != 1:
+        do = do.contiguous()
+    (batch, seqlen_q, nheads, d) = q.shape
+    (_, seqlen_k, _, _) = k.shape
+    assert d <= 128
+    seqlen_q_rounded = math.ceil(seqlen_q / 128) * 128
+    assert lse.shape == (batch, nheads, seqlen_q_rounded)
+    assert q.stride(-1) == k.stride(-1) == v.stride(-1) == o.stride(-1) == 1
+    assert dq.stride(-1) == dk.stride(-1) == dv.stride(-1) == 1
+    softmax_scale = softmax_scale or 1.0 / math.sqrt(d)
+    dq_accum = torch.empty_like(q, dtype=torch.float32)
+    delta = torch.empty_like(lse)
+    BLOCK_HEADDIM = max(triton.next_power_of_2(d), 16)
+    grid = lambda META: (triton.cdiv(seqlen_q, META['BLOCK_M']), batch * nheads)
+    _bwd_preprocess_do_o_dot[grid](o, do, delta, o.stride(0), o.stride(2), o.stride(1), do.stride(0), do.stride(2), do.stride(1), nheads, seqlen_q, seqlen_q_rounded, d, BLOCK_M=128, BLOCK_HEADDIM=BLOCK_HEADDIM)
+    has_bias = bias is not None
+    bias_type = 'none'
+    if has_bias:
+        assert bias.dtype in [q.dtype, torch.float]
+        assert bias.is_cuda
+        assert bias.dim() == 4
+        assert bias.stride(-1) == 1
+        if bias.shape[2:] == (1, seqlen_k):
+            bias_type = 'vector'
+        elif bias.shape[2:] == (seqlen_q, seqlen_k):
+            bias_type = 'matrix'
+        else:
+            raise RuntimeError('Last 2 dimensions of bias must be (1, seqlen_k) or (seqlen_q, seqlen_k)')
+        bias = bias.expand(batch, nheads, seqlen_q, seqlen_k)
+    bias_strides = (bias.stride(0), bias.stride(1), bias.stride(2)) if has_bias else (0, 0, 0)
+    grid = lambda META: (triton.cdiv(seqlen_k, META['BLOCK_N']) if META['SEQUENCE_PARALLEL'] else 1, batch * nheads)
+    _bwd_kernel[grid](q, k, v, bias, do, dq_accum, dk, dv, lse, delta, softmax_scale, q.stride(0), q.stride(2), q.stride(1), k.stride(0), k.stride(2), k.stride(1), v.stride(0), v.stride(2), v.stride(1), *bias_strides, do.stride(0), do.stride(2), do.stride(1), dq_accum.stride(0), dq_accum.stride(2), dq_accum.stride(1), dk.stride(0), dk.stride(2), dk.stride(1), dv.stride(0), dv.stride(2), dv.stride(1), nheads, seqlen_q, seqlen_k, seqlen_q_rounded, d, seqlen_q // 32, seqlen_k // 32, bias_type, causal, BLOCK_HEADDIM)
+    dq.copy_(dq_accum)
+class FlashAttnQKVPackedFunc(torch.autograd.Function):
+    @staticmethod
+    def forward(ctx, qkv, bias=None, causal=False, softmax_scale=None):
+        """
+            qkv: (batch, seqlen, 3, nheads, headdim)
+            bias: optional, shape broadcastible to (batch, nheads, seqlen, seqlen).
+                For example, ALiBi mask for causal would have shape (1, nheads, 1, seqlen).
+                ALiBi mask for non-causal would have shape (1, nheads, seqlen, seqlen)
+        """
+        if qkv.stride(-1) != 1:
+            qkv = qkv.contiguous()
+        (o, lse, ctx.softmax_scale) = _flash_attn_forward(qkv[:, :, 0], qkv[:, :, 1], qkv[:, :, 2], bias=bias, causal=causal, softmax_scale=softmax_scale)
+        ctx.save_for_backward(qkv, o, lse, bias)
+        ctx.causal = causal
+        return o
+    @staticmethod
+    def backward(ctx, do):
+        (qkv, o, lse, bias) = ctx.saved_tensors
+        assert not ctx.needs_input_grad[1], 'FlashAttention does not support bias gradient yet'
+        with torch.inference_mode():
+            dqkv = torch.empty_like(qkv)
+            _flash_attn_backward(do, qkv[:, :, 0], qkv[:, :, 1], qkv[:, :, 2], o, lse, dqkv[:, :, 0], dqkv[:, :, 1], dqkv[:, :, 2], bias=bias, causal=ctx.causal, softmax_scale=ctx.softmax_scale)
+        return (dqkv, None, None, None)
+flash_attn_qkvpacked_func = FlashAttnQKVPackedFunc.apply
+class FlashAttnKVPackedFunc(torch.autograd.Function):
+    @staticmethod
+    def forward(ctx, q, kv, bias=None, causal=False, softmax_scale=None):
+        """
+            q: (batch, seqlen_q, nheads, headdim)
+            kv: (batch, seqlen_k, 2, nheads, headdim)
+            bias: optional, shape broadcastible to (batch, nheads, seqlen_q, seqlen_k).
+                For example, ALiBi mask for causal would have shape (1, nheads, 1, seqlen_k).
+                ALiBi mask for non-causal would have shape (1, nheads, seqlen_q, seqlen_k)
+        """
+        (q, kv) = [x if x.stride(-1) == 1 else x.contiguous() for x in [q, kv]]
+        (o, lse, ctx.softmax_scale) = _flash_attn_forward(q, kv[:, :, 0], kv[:, :, 1], bias=bias, causal=causal, softmax_scale=softmax_scale)
+        ctx.save_for_backward(q, kv, o, lse, bias)
+        ctx.causal = causal
+        return o
+    @staticmethod
+    def backward(ctx, do):
+        (q, kv, o, lse, bias) = ctx.saved_tensors
+        if len(ctx.needs_input_grad) >= 3:
+            assert not ctx.needs_input_grad[2], 'FlashAttention does not support bias gradient yet'
+        with torch.inference_mode():
+            dq = torch.empty_like(q)
+            dkv = torch.empty_like(kv)
+            _flash_attn_backward(do, q, kv[:, :, 0], kv[:, :, 1], o, lse, dq, dkv[:, :, 0], dkv[:, :, 1], bias=bias, causal=ctx.causal, softmax_scale=ctx.softmax_scale)
+        return (dq, dkv, None, None, None)
+flash_attn_kvpacked_func = FlashAttnKVPackedFunc.apply
+class FlashAttnFunc(torch.autograd.Function):
+    @staticmethod
+    def forward(ctx, q, k, v, bias=None, causal=False, softmax_scale=None):
+        """
+            q: (batch_size, seqlen_q, nheads, headdim)
+            k, v: (batch_size, seqlen_k, nheads, headdim)
+            bias: optional, shape broadcastible to (batch, nheads, seqlen_q, seqlen_k).
+                For example, ALiBi mask for causal would have shape (1, nheads, 1, seqlen_k).
+                ALiBi mask for non-causal would have shape (1, nheads, seqlen_q, seqlen_k)
+        """
+        (q, k, v) = [x if x.stride(-1) == 1 else x.contiguous() for x in [q, k, v]]
+        (o, lse, ctx.softmax_scale) = _flash_attn_forward(q, k, v, bias=bias, causal=causal, softmax_scale=softmax_scale)
+        ctx.save_for_backward(q, k, v, o, lse, bias)
+        ctx.causal = causal
+        return o
+    @staticmethod
+    def backward(ctx, do):
+        (q, k, v, o, lse, bias) = ctx.saved_tensors
+        assert not ctx.needs_input_grad[3], 'FlashAttention does not support bias gradient yet'
+        with torch.inference_mode():
+            dq = torch.empty_like(q)
+            dk = torch.empty_like(k)
+            dv = torch.empty_like(v)
+            _flash_attn_backward(do, q, k, v, o, lse, dq, dk, dv, bias=bias, causal=ctx.causal, softmax_scale=ctx.softmax_scale)
+        return (dq, dk, dv, None, None, None)
+flash_attn_func = FlashAttnFunc.apply

dam/model/language_model/mpt_ignored/hf_prefixlm_converter.py ADDED Viewed

	@@ -0,0 +1,415 @@

+"""Converts Huggingface Causal LM to Prefix LM.
+Conversion does lightweight surgery on a HuggingFace
+Causal LM to convert it to a Prefix LM.
+Prefix LMs accepts a `bidirectional_mask` input in `forward`
+and treat the input prompt as the prefix in `generate`.
+"""
+import math
+import warnings
+from types import MethodType
+from typing import Any, Dict, List, Optional, Tuple, Union
+import torch
+from transformers.models.bloom.modeling_bloom import BaseModelOutputWithPastAndCrossAttentions, BloomForCausalLM, BloomModel, CausalLMOutputWithCrossAttentions, CrossEntropyLoss
+from transformers.models.bloom.modeling_bloom import _expand_mask as _expand_mask_bloom
+from transformers.models.bloom.modeling_bloom import _make_causal_mask as _make_causal_mask_bloom
+from transformers.models.bloom.modeling_bloom import logging
+from transformers.models.gpt2.modeling_gpt2 import GPT2LMHeadModel
+from transformers.models.gpt_neo.modeling_gpt_neo import GPTNeoForCausalLM
+from transformers.models.gpt_neox.modeling_gpt_neox import GPTNeoXForCausalLM
+from transformers.models.gptj.modeling_gptj import GPTJForCausalLM
+from transformers.models.opt.modeling_opt import OPTForCausalLM
+from transformers.models.opt.modeling_opt import _expand_mask as _expand_mask_opt
+from transformers.models.opt.modeling_opt import _make_causal_mask as _make_causal_mask_opt
+logger = logging.get_logger(__name__)
+_SUPPORTED_GPT_MODELS = (GPT2LMHeadModel, GPTJForCausalLM, GPTNeoForCausalLM, GPTNeoXForCausalLM)
+CAUSAL_GPT_TYPES = Union[GPT2LMHeadModel, GPTJForCausalLM, GPTNeoForCausalLM, GPTNeoXForCausalLM]
+def _convert_gpt_causal_lm_to_prefix_lm(model: CAUSAL_GPT_TYPES) -> CAUSAL_GPT_TYPES:
+    """Converts a GPT-style Causal LM to a Prefix LM.
+    Supported HuggingFace model classes:
+        - `GPT2LMHeadModel`
+        - `GPTNeoForCausalLM`
+        - `GPTNeoXForCausalLM`
+        - `GPTJForCausalLM`
+    See `convert_hf_causal_lm_to_prefix_lm` for more details.
+    """
+    if hasattr(model, '_prefix_lm_converted'):
+        return model
+    assert isinstance(model, _SUPPORTED_GPT_MODELS)
+    assert model.config.add_cross_attention == False, 'Only supports GPT-style decoder-only models'
+    def _get_attn_modules(model: CAUSAL_GPT_TYPES) -> List[torch.nn.Module]:
+        """Helper that gets a list of the model's attention modules.
+        Each module has a `bias` buffer used for causal masking. The Prefix LM
+        conversion adds logic to dynamically manipulate these biases to support
+        Prefix LM attention masking.
+        """
+        attn_modules = []
+        if isinstance(model, GPTNeoXForCausalLM):
+            blocks = model.gpt_neox.layers
+        else:
+            blocks = model.transformer.h
+        for block in blocks:
+            if isinstance(model, GPTNeoForCausalLM):
+                if block.attn.attention_type != 'global':
+                    continue
+                attn_module = block.attn.attention
+            elif isinstance(model, GPTNeoXForCausalLM):
+                attn_module = block.attention
+            else:
+                attn_module = block.attn
+            attn_modules.append(attn_module)
+        return attn_modules
+    setattr(model, '_original_forward', getattr(model, 'forward'))
+    setattr(model, '_original_generate', getattr(model, 'generate'))
+    def forward(self: CAUSAL_GPT_TYPES, input_ids: Optional[torch.LongTensor]=None, past_key_values: Optional[Tuple[Tuple[torch.Tensor]]]=None, attention_mask: Optional[torch.FloatTensor]=None, bidirectional_mask: Optional[torch.Tensor]=None, token_type_ids: Optional[torch.LongTensor]=None, position_ids: Optional[torch.LongTensor]=None, head_mask: Optional[torch.FloatTensor]=None, inputs_embeds: Optional[torch.FloatTensor]=None, labels: Optional[torch.LongTensor]=None, use_cache: Optional[bool]=None, output_attentions: Optional[bool]=None, output_hidden_states: Optional[bool]=None, return_dict: Optional[bool]=None):
+        """Wraps original forward to enable PrefixLM attention."""
+        def call_og_forward():
+            if isinstance(self, GPTNeoXForCausalLM):
+                return self._original_forward(input_ids=input_ids, past_key_values=past_key_values, attention_mask=attention_mask, head_mask=head_mask, inputs_embeds=inputs_embeds, labels=labels, use_cache=use_cache, output_attentions=output_attentions, output_hidden_states=output_hidden_states, return_dict=return_dict)
+            else:
+                return self._original_forward(input_ids=input_ids, past_key_values=past_key_values, attention_mask=attention_mask, token_type_ids=token_type_ids, position_ids=position_ids, head_mask=head_mask, inputs_embeds=inputs_embeds, labels=labels, use_cache=use_cache, output_attentions=output_attentions, output_hidden_states=output_hidden_states, return_dict=return_dict)
+        if bidirectional_mask is None:
+            return call_og_forward()
+        assert isinstance(bidirectional_mask, torch.Tensor)
+        attn_modules = _get_attn_modules(model)
+        (b, s) = bidirectional_mask.shape
+        max_length = attn_modules[0].bias.shape[-1]
+        if s > max_length:
+            raise ValueError(f'bidirectional_mask sequence length (={s}) exceeds the ' + f'max length allowed by the model ({max_length}).')
+        assert s <= max_length
+        if s < max_length:
+            pad = torch.zeros((int(b), int(max_length - s)), dtype=bidirectional_mask.dtype, device=bidirectional_mask.device)
+            bidirectional_mask = torch.cat([bidirectional_mask, pad], dim=1)
+        bidirectional = bidirectional_mask.unsqueeze(1).unsqueeze(1)
+        for attn_module in attn_modules:
+            attn_module.bias.data = torch.logical_or(attn_module.bias.data, bidirectional)
+        output = call_og_forward()
+        for attn_module in attn_modules:
+            attn_module.bias.data = torch.tril(attn_module.bias.data[0, 0])[None, None]
+        return output
+    def generate(self: CAUSAL_GPT_TYPES, *args: tuple, **kwargs: Dict[str, Any]):
+        """Wraps original generate to enable PrefixLM attention."""
+        attn_modules = _get_attn_modules(model)
+        for attn_module in attn_modules:
+            attn_module.bias.data[:] = 1
+        output = self._original_generate(*args, **kwargs)
+        for attn_module in attn_modules:
+            attn_module.bias.data = torch.tril(attn_module.bias.data[0, 0])[None, None]
+        return output
+    setattr(model, 'forward', MethodType(forward, model))
+    setattr(model, 'generate', MethodType(generate, model))
+    setattr(model, '_prefix_lm_converted', True)
+    return model
+def _convert_bloom_causal_lm_to_prefix_lm(model: BloomForCausalLM) -> BloomForCausalLM:
+    """Converts a BLOOM Causal LM to a Prefix LM.
+    Supported HuggingFace model classes:
+        - `BloomForCausalLM`
+    See `convert_hf_causal_lm_to_prefix_lm` for more details.
+    """
+    if hasattr(model, '_prefix_lm_converted'):
+        return model
+    assert isinstance(model, BloomForCausalLM)
+    assert model.config.add_cross_attention == False, 'Only supports BLOOM decoder-only models'
+    def _prepare_attn_mask(self: BloomModel, attention_mask: torch.Tensor, bidirectional_mask: Optional[torch.Tensor], input_shape: Tuple[int, int], past_key_values_length: int) -> torch.BoolTensor:
+        combined_attention_mask = None
+        device = attention_mask.device
+        (_, src_length) = input_shape
+        if src_length > 1:
+            combined_attention_mask = _make_causal_mask_bloom(input_shape, device=device, past_key_values_length=past_key_values_length)
+            if bidirectional_mask is not None:
+                assert attention_mask.shape == bidirectional_mask.shape
+                expanded_bidirectional_mask = _expand_mask_bloom(bidirectional_mask, tgt_length=src_length)
+                combined_attention_mask = torch.logical_and(combined_attention_mask, expanded_bidirectional_mask)
+        expanded_attn_mask = _expand_mask_bloom(attention_mask, tgt_length=src_length)
+        combined_attention_mask = expanded_attn_mask if combined_attention_mask is None else expanded_attn_mask | combined_attention_mask
+        return combined_attention_mask
+    def _build_alibi_tensor(self: BloomModel, batch_size: int, query_length: int, key_length: int, dtype: torch.dtype, device: torch.device) -> torch.Tensor:
+        num_heads = self.config.n_head
+        closest_power_of_2 = 2 ** math.floor(math.log2(num_heads))
+        base = torch.tensor(2 ** (-2 ** (-(math.log2(closest_power_of_2) - 3))), device=device, dtype=torch.float32)
+        powers = torch.arange(1, 1 + closest_power_of_2, device=device, dtype=torch.int32)
+        slopes = torch.pow(base, powers)
+        if closest_power_of_2 != num_heads:
+            extra_base = torch.tensor(2 ** (-2 ** (-(math.log2(2 * closest_power_of_2) - 3))), device=device, dtype=torch.float32)
+            num_remaining_heads = min(closest_power_of_2, num_heads - closest_power_of_2)
+            extra_powers = torch.arange(1, 1 + 2 * num_remaining_heads, 2, device=device, dtype=torch.int32)
+            slopes = torch.cat([slopes, torch.pow(extra_base, extra_powers)], dim=0)
+        qa = torch.arange(query_length, device=device, dtype=torch.int32).view(-1, 1)
+        ka = torch.arange(key_length, device=device, dtype=torch.int32).view(1, -1)
+        diffs = qa - ka + key_length - query_length
+        diffs = -diffs.abs()
+        alibi = slopes.view(1, num_heads, 1, 1) * diffs.view(1, 1, query_length, key_length)
+        alibi = alibi.expand(batch_size, -1, -1, -1).reshape(-1, query_length, key_length)
+        return alibi.to(dtype)
+    KeyValueT = Tuple[torch.Tensor, torch.Tensor]
+    def forward(self: BloomModel, input_ids: Optional[torch.LongTensor]=None, past_key_values: Optional[Tuple[KeyValueT, ...]]=None, attention_mask: Optional[torch.Tensor]=None, bidirectional_mask: Optional[torch.Tensor]=None, head_mask: Optional[torch.LongTensor]=None, inputs_embeds: Optional[torch.LongTensor]=None, use_cache: Optional[bool]=None, output_attentions: Optional[bool]=None, output_hidden_states: Optional[bool]=None, return_dict: Optional[bool]=None, **deprecated_arguments) -> Union[Tuple[torch.Tensor, ...], BaseModelOutputWithPastAndCrossAttentions]:
+        if deprecated_arguments.pop('position_ids', False) is not False:
+            warnings.warn('`position_ids` have no functionality in BLOOM and will be removed in v5.0.0. ' + 'You can safely ignore passing `position_ids`.', FutureWarning)
+        if len(deprecated_arguments) > 0:
+            raise ValueError(f'Got unexpected arguments: {deprecated_arguments}')
+        output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
+        output_hidden_states = output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        if input_ids is not None and inputs_embeds is not None:
+            raise ValueError('You cannot specify both input_ids and inputs_embeds at the same time')
+        elif input_ids is not None:
+            (batch_size, seq_length) = input_ids.shape
+        elif inputs_embeds is not None:
+            (batch_size, seq_length, _) = inputs_embeds.shape
+        else:
+            raise ValueError('You have to specify either input_ids or inputs_embeds')
+        if past_key_values is None:
+            past_key_values = tuple([None] * len(self.h))
+        head_mask = self.get_head_mask(head_mask, self.config.n_layer)
+        if inputs_embeds is None:
+            inputs_embeds = self.word_embeddings(input_ids)
+        hidden_states = self.word_embeddings_layernorm(inputs_embeds)
+        presents = () if use_cache else None
+        all_self_attentions = () if output_attentions else None
+        all_hidden_states = () if output_hidden_states else None
+        seq_length_with_past = seq_length
+        past_key_values_length = 0
+        if past_key_values[0] is not None:
+            tmp = past_key_values[0][0]
+            past_key_values_length = tmp.shape[2]
+            seq_length_with_past = seq_length_with_past + past_key_values_length
+        if attention_mask is None:
+            attention_mask = torch.ones((batch_size, seq_length_with_past), device=hidden_states.device)
+        else:
+            attention_mask = attention_mask.to(hidden_states.device)
+        alibi = self._build_alibi_tensor(batch_size=batch_size, query_length=seq_length, key_length=seq_length_with_past, dtype=hidden_states.dtype, device=hidden_states.device)
+        causal_mask = self._prepare_attn_mask(attention_mask, bidirectional_mask, input_shape=(batch_size, seq_length), past_key_values_length=past_key_values_length)
+        for (i, (block, layer_past)) in enumerate(zip(self.h, past_key_values)):
+            if output_hidden_states:
+                hst = (hidden_states,)
+                all_hidden_states = all_hidden_states + hst
+            if self.gradient_checkpointing and self.training:
+                if use_cache:
+                    logger.warning('`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`...')
+                    use_cache = False
+                def create_custom_forward(module):
+                    def custom_forward(*inputs):
+                        return module(*inputs, use_cache=use_cache, output_attentions=output_attentions)
+                    return custom_forward
+                outputs = torch.utils.checkpoint.checkpoint(create_custom_forward(block), hidden_states, alibi, causal_mask, head_mask[i])
+            else:
+                outputs = block(hidden_states, layer_past=layer_past, attention_mask=causal_mask, head_mask=head_mask[i], use_cache=use_cache, output_attentions=output_attentions, alibi=alibi)
+            hidden_states = outputs[0]
+            if use_cache is True:
+                presents = presents + (outputs[1],)
+            if output_attentions:
+                oa = (outputs[2 if use_cache else 1],)
+                all_self_attentions = all_self_attentions + oa
+        hidden_states = self.ln_f(hidden_states)
+        if output_hidden_states:
+            hst = (hidden_states,)
+            all_hidden_states = all_hidden_states + hst
+        if not return_dict:
+            return tuple((v for v in [hidden_states, presents, all_hidden_states, all_self_attentions] if v is not None))
+        return BaseModelOutputWithPastAndCrossAttentions(last_hidden_state=hidden_states, past_key_values=presents, hidden_states=all_hidden_states, attentions=all_self_attentions)
+    setattr(model.transformer, '_prepare_attn_mask', MethodType(_prepare_attn_mask, model.transformer))
+    setattr(model.transformer, '_build_alibi_tensor', MethodType(_build_alibi_tensor, model.transformer))
+    setattr(model.transformer, 'forward', MethodType(forward, model.transformer))
+    KeyValueT = Tuple[torch.Tensor, torch.Tensor]
+    def forward(self: BloomForCausalLM, input_ids: Optional[torch.LongTensor]=None, past_key_values: Optional[Tuple[KeyValueT, ...]]=None, attention_mask: Optional[torch.Tensor]=None, bidirectional_mask: Optional[torch.Tensor]=None, head_mask: Optional[torch.Tensor]=None, inputs_embeds: Optional[torch.Tensor]=None, labels: Optional[torch.Tensor]=None, use_cache: Optional[bool]=None, output_attentions: Optional[bool]=None, output_hidden_states: Optional[bool]=None, return_dict: Optional[bool]=None, **deprecated_arguments) -> Union[Tuple[torch.Tensor], CausalLMOutputWithCrossAttentions]:
+        """Replacement forward method for BloomCausalLM."""
+        if deprecated_arguments.pop('position_ids', False) is not False:
+            warnings.warn('`position_ids` have no functionality in BLOOM and will be removed ' + 'in v5.0.0. You can safely ignore passing `position_ids`.', FutureWarning)
+        if len(deprecated_arguments) > 0:
+            raise ValueError(f'Got unexpected arguments: {deprecated_arguments}')
+        return_dict = return_dict if return_dict is not None else self.config.use_return_dict
+        transformer_outputs = self.transformer(input_ids, past_key_values=past_key_values, attention_mask=attention_mask, bidirectional_mask=bidirectional_mask, head_mask=head_mask, inputs_embeds=inputs_embeds, use_cache=use_cache, output_attentions=output_attentions, output_hidden_states=output_hidden_states, return_dict=return_dict)
+        hidden_states = transformer_outputs[0]
+        lm_logits = self.lm_head(hidden_states)
+        loss = None
+        if labels is not None:
+            shift_logits = lm_logits[..., :-1, :].contiguous()
+            shift_labels = labels[..., 1:].contiguous()
+            (batch_size, seq_length, vocab_size) = shift_logits.shape
+            loss_fct = CrossEntropyLoss()
+            loss = loss_fct(shift_logits.view(batch_size * seq_length, vocab_size), shift_labels.view(batch_size * seq_length))
+        if not return_dict:
+            output = (lm_logits,) + transformer_outputs[1:]
+            return (loss,) + output if loss is not None else output
+        return CausalLMOutputWithCrossAttentions(loss=loss, logits=lm_logits, past_key_values=transformer_outputs.past_key_values, hidden_states=transformer_outputs.hidden_states, attentions=transformer_outputs.attentions)
+    def prepare_inputs_for_generation(self: BloomForCausalLM, input_ids: torch.LongTensor, past: Optional[torch.Tensor]=None, attention_mask: Optional[torch.Tensor]=None, **kwargs) -> dict:
+        if past:
+            input_ids = input_ids[:, -1].unsqueeze(-1)
+            bidirectional_mask = None
+            if past[0][0].shape[0] == input_ids.shape[0]:
+                past = self._convert_to_bloom_cache(past)
+        else:
+            bidirectional_mask = torch.ones_like(input_ids)
+        return {'input_ids': input_ids, 'past_key_values': past, 'use_cache': True, 'attention_mask': attention_mask, 'bidirectional_mask': bidirectional_mask}
+    setattr(model, 'forward', MethodType(forward, model))
+    setattr(model, 'prepare_inputs_for_generation', MethodType(prepare_inputs_for_generation, model))
+    setattr(model, '_prefix_lm_converted', True)
+    return model
+def _convert_opt_causal_lm_to_prefix_lm(model: OPTForCausalLM) -> OPTForCausalLM:
+    """Converts an OPT Causal LM to a Prefix LM.
+    Supported HuggingFace model classes:
+        - `OPTForCausalLM`
+    See `convert_hf_causal_lm_to_prefix_lm` for more details.
+    """
+    if hasattr(model, '_prefix_lm_converted'):
+        return model
+    assert isinstance(model, OPTForCausalLM)
+    assert model.config.add_cross_attention == False, 'Only supports OPT decoder-only models'
+    setattr(model, '_original_forward', getattr(model, 'forward'))
+    setattr(model, '_original_generate', getattr(model, 'generate'))
+    model.model.decoder.bidirectional_mask = None
+    def _prepare_decoder_attention_mask(self, attention_mask, input_shape, inputs_embeds, past_key_values_length):
+        combined_attention_mask = None
+        if input_shape[-1] > 1:
+            if self.bidirectional_mask == 'g':
+                (bsz, src_length) = input_shape
+                combined_attention_mask = torch.zeros((bsz, 1, src_length, src_length + past_key_values_length), dtype=inputs_embeds.dtype, device=inputs_embeds.device)
+            else:
+                combined_attention_mask = _make_causal_mask_opt(input_shape, inputs_embeds.dtype, past_key_values_length=past_key_values_length).to(inputs_embeds.device)
+                if self.bidirectional_mask is not None:
+                    assert attention_mask.shape == self.bidirectional_mask.shape
+                    expanded_bidirectional_mask = _expand_mask_opt(self.bidirectional_mask, inputs_embeds.dtype, tgt_len=input_shape[-1]).to(inputs_embeds.device)
+                    combined_attention_mask = torch.maximum(expanded_bidirectional_mask, combined_attention_mask)
+        if attention_mask is not None:
+            expanded_attn_mask = _expand_mask_opt(attention_mask, inputs_embeds.dtype, tgt_len=input_shape[-1]).to(inputs_embeds.device)
+            combined_attention_mask = expanded_attn_mask if combined_attention_mask is None else expanded_attn_mask + combined_attention_mask
+        return combined_attention_mask
+    setattr(model.model.decoder, '_prepare_decoder_attention_mask', MethodType(_prepare_decoder_attention_mask, model.model.decoder))
+    def forward(self: OPTForCausalLM, input_ids: Optional[torch.LongTensor]=None, attention_mask: Optional[torch.Tensor]=None, bidirectional_mask: Optional[torch.ByteTensor]=None, head_mask: Optional[torch.Tensor]=None, past_key_values: Optional[List[torch.FloatTensor]]=None, inputs_embeds: Optional[torch.FloatTensor]=None, labels: Optional[torch.LongTensor]=None, use_cache: Optional[bool]=None, output_attentions: Optional[bool]=None, output_hidden_states: Optional[bool]=None, return_dict: Optional[bool]=None):
+        def call_og_forward():
+            return self._original_forward(input_ids=input_ids, attention_mask=attention_mask, head_mask=head_mask, past_key_values=past_key_values, inputs_embeds=inputs_embeds, labels=labels, use_cache=use_cache, output_attentions=output_attentions, output_hidden_states=output_hidden_states, return_dict=return_dict)
+        if bidirectional_mask is None:
+            return call_og_forward()
+        self.model.decoder.bidirectional_mask = bidirectional_mask
+        try:
+            outputs = call_og_forward()
+        except:
+            self.model.decoder.bidirectional_mask = None
+            raise
+        self.model.decoder.bidirectional_mask = None
+        return outputs
+    def generate(self: OPTForCausalLM, *args: tuple, **kwargs: Dict[str, Any]):
+        """Wraps original generate to enable PrefixLM-style attention."""
+        self.model.decoder.bidirectional_mask = 'g'
+        try:
+            output = self._original_generate(*args, **kwargs)
+        except:
+            self.model.decoder.bidirectional_mask = None
+            raise
+        self.model.decoder.bidirectional_mask = None
+        return output
+    setattr(model, 'forward', MethodType(forward, model))
+    setattr(model, 'generate', MethodType(generate, model))
+    setattr(model, '_prefix_lm_converted', True)
+    return model
+_SUPPORTED_HF_MODELS = _SUPPORTED_GPT_MODELS + (BloomForCausalLM, OPTForCausalLM)
+CAUSAL_LM_TYPES = Union[GPT2LMHeadModel, GPTJForCausalLM, GPTNeoForCausalLM, GPTNeoXForCausalLM, BloomForCausalLM, OPTForCausalLM]
+def convert_hf_causal_lm_to_prefix_lm(model: CAUSAL_LM_TYPES) -> CAUSAL_LM_TYPES:
+    """Converts a HuggingFace Causal LM to a Prefix LM.
+    Supported HuggingFace model classes:
+        - `GPT2LMHeadModel`
+        - `GPTNeoForCausalLM`
+        - `GPTNeoXForCausalLM`
+        - `GPTJForCausalLM`
+        - `BloomForCausalLM`
+        - `OPTForCausalLM`
+    Conversion to a Prefix LM is done by modifying the `forward` method, and possibly also the
+    `generate` method and/or select underlying methods depending on the model class.
+    These changes preserve the model API, but add a new input to `forward`: "bidirectional_mask".
+    Notes on training:
+        To actually train the converted model as a Prefix LM, training batches will need to indicate
+        the prefix/target structure by including `bidirectional_mask` as part of the batch inputs.
+        **This is not a standard input and requires custom layers either within or after your dataloader.**
+        In addition to adding `bidirectional_mask` to the batch, this custom code should modify `labels`
+        such that `batch['labels'][batch['bidirectional_mask'] == 1] == -100`.
+        That is, the prefix portion of the sequence should not generate any loss. Loss should only be
+        generated by the target portion of the sequence.
+    Notes on `GPTNeoForCausalLM`:
+        To simplify the implementation, "global" and "local" attention layers are handled differently.
+        For "global" layers, we handle conversion as described above. For "local" layers, which use a
+        causal attention mask within a restricted local window, we do not alter the masking.
+    Notes on `forward` method conversion:
+        After conversion, the `forward` method will handle a new input, `bidirectional_mask`,
+        which should be a [batch_size, seq_length] byte tensor, where 1 indicates token positions
+        belonging to the prefix (prefix tokens can attend to one another bidirectionally), and
+        0 indicates token positions belonging to the target.
+        The new `forward` method will incorporate `bidirectional_mask` (if supplied) into the existing
+        causal mask, call the original `forward` method, and (if the causal mask is a buffer) reset
+        the causal masks before returning the result.
+    Notes on `generate` method conversion:
+        After conversion, the `generate` method will have the same signature but will internally
+        convert all causal masks to be purely bidirectional, call the original `generate` method, and
+        (where appropriate) reset the causal masks before returning the result.
+        This works thanks to the logic of the HuggingFace `generate` API, which first encodes the token
+        "prompt" passed to `generate` (which is treated as the prefix) and then sequentially generates
+        each new token. Encodings are cached as generation happens, so all prefix tokens can attend to one
+        another (as expected in a Prefix LM) and generated tokens can only attend to prefix tokens and
+        previously-generated tokens (also as expected in a Prefix LM).
+    To preserve the API, the original methods are renamed to `_original_forward` and
+    `_original_generate`, and replaced with new `forward` and `generate` methods that wrap
+    them, respectively. Although implementation details vary by model class.
+    """
+    if isinstance(model, _SUPPORTED_GPT_MODELS):
+        return _convert_gpt_causal_lm_to_prefix_lm(model)
+    elif isinstance(model, BloomForCausalLM):
+        return _convert_bloom_causal_lm_to_prefix_lm(model)
+    elif isinstance(model, OPTForCausalLM):
+        return _convert_opt_causal_lm_to_prefix_lm(model)
+    else:
+        raise TypeError(f'Cannot convert model to Prefix LM. ' + f'Model does not belong to set of supported HF models:' + f'\n{_SUPPORTED_HF_MODELS}')
+def add_bidirectional_mask_if_missing(batch: Dict[str, Any]):
+    """Attempts to add bidirectional_mask to batch if missing.
+    Raises:
+        KeyError if bidirectional_mask is missing and can't be inferred
+    """
+    if 'bidirectional_mask' not in batch:
+        if batch.get('mode', None) == 'icl_task':
+            batch['bidirectional_mask'] = batch['attention_mask'].clone()
+            for (i, continuation_indices) in enumerate(batch['continuation_indices']):
+                batch['bidirectional_mask'][i, continuation_indices] = 0
+        elif 'labels' in batch and 'attention_mask' in batch:
+            batch['bidirectional_mask'] = torch.logical_and(torch.eq(batch['attention_mask'], 1), torch.eq(batch['labels'], -100)).type_as(batch['attention_mask'])
+        else:
+            raise KeyError('No bidirectional_mask in batch and not sure how to construct one.')

dam/model/language_model/mpt_ignored/meta_init_context.py ADDED Viewed

	@@ -0,0 +1,94 @@

+from contextlib import contextmanager
+import torch
+import torch.nn as nn
+@contextmanager
+def init_empty_weights(include_buffers: bool=False):
+    """Meta initialization context manager.
+    A context manager under which models are initialized with all parameters
+    on the meta device, therefore creating an empty model. Useful when just
+    initializing the model would blow the available RAM.
+    Args:
+        include_buffers (`bool`, *optional*, defaults to `False`): Whether or
+            not to also put all buffers on the meta device while initializing.
+    Example:
+    ```python
+    import torch.nn as nn
+    # Initialize a model with 100 billions parameters in no time and without using any RAM.
+    with init_empty_weights():
+        tst = nn.Sequential(*[nn.Linear(10000, 10000) for _ in range(1000)])
+    ```
+    <Tip warning={true}>
+    Any model created under this context manager has no weights. As such you can't do something like
+    `model.to(some_device)` with it. To load weights inside your empty model, see [`load_checkpoint_and_dispatch`].
+    </Tip>
+    """
+    with init_on_device(torch.device('meta'), include_buffers=include_buffers) as f:
+        yield f
+@contextmanager
+def init_on_device(device: torch.device, include_buffers: bool=False):
+    """Device initialization context manager.
+    A context manager under which models are initialized with all parameters
+    on the specified device.
+    Args:
+        device (`torch.device`): Device to initialize all parameters on.
+        include_buffers (`bool`, *optional*, defaults to `False`): Whether or
+            not to also put all buffers on the meta device while initializing.
+    Example:
+    ```python
+    import torch.nn as nn
+    with init_on_device(device=torch.device("cuda")):
+        tst = nn.Liner(100, 100)  # on `cuda` device
+    ```
+    """
+    old_register_parameter = nn.Module.register_parameter
+    if include_buffers:
+        old_register_buffer = nn.Module.register_buffer
+    def register_empty_parameter(module, name, param):
+        old_register_parameter(module, name, param)
+        if param is not None:
+            param_cls = type(module._parameters[name])
+            kwargs = module._parameters[name].__dict__
+            module._parameters[name] = param_cls(module._parameters[name].to(device), **kwargs)
+    def register_empty_buffer(module, name, buffer):
+        old_register_buffer(module, name, buffer)
+        if buffer is not None:
+            module._buffers[name] = module._buffers[name].to(device)
+    if include_buffers:
+        tensor_constructors_to_patch = {torch_function_name: getattr(torch, torch_function_name) for torch_function_name in ['empty', 'zeros', 'ones', 'full']}
+    else:
+        tensor_constructors_to_patch = {}
+    def patch_tensor_constructor(fn):
+        def wrapper(*args, **kwargs):
+            kwargs['device'] = device
+            return fn(*args, **kwargs)
+        return wrapper
+    try:
+        nn.Module.register_parameter = register_empty_parameter
+        if include_buffers:
+            nn.Module.register_buffer = register_empty_buffer
+        for torch_function_name in tensor_constructors_to_patch.keys():
+            setattr(torch, torch_function_name, patch_tensor_constructor(getattr(torch, torch_function_name)))
+        yield
+    finally:
+        nn.Module.register_parameter = old_register_parameter
+        if include_buffers:
+            nn.Module.register_buffer = old_register_buffer
+        for (torch_function_name, old_torch_function) in tensor_constructors_to_patch.items():
+            setattr(torch, torch_function_name, old_torch_function)

dam/model/language_model/mpt_ignored/modeling_mpt.py ADDED Viewed

	@@ -0,0 +1,331 @@

+"""A simple, flexible implementation of a GPT model.
+Inspired by https://github.com/karpathy/minGPT/blob/master/mingpt/model.py
+"""
+import math
+import warnings
+from typing import List, Optional, Tuple, Union
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from transformers import PreTrainedModel, PreTrainedTokenizer, PreTrainedTokenizerFast
+from transformers.modeling_outputs import BaseModelOutputWithPast, CausalLMOutputWithPast
+from .attention import attn_bias_shape, build_attn_bias
+from .blocks import MPTBlock
+from .custom_embedding import SharedEmbedding
+from .norm import NORM_CLASS_REGISTRY
+from .configuration_mpt import MPTConfig
+from .adapt_tokenizer import AutoTokenizerForMOD, adapt_tokenizer_for_denoising
+from .hf_prefixlm_converter import add_bidirectional_mask_if_missing, convert_hf_causal_lm_to_prefix_lm
+from .meta_init_context import init_empty_weights
+from .param_init_fns import MODEL_INIT_REGISTRY, generic_param_init_fn_
+try:
+    from .flash_attn_triton import flash_attn_func
+except:
+    pass
+Tokenizer = Union[PreTrainedTokenizer, PreTrainedTokenizerFast]
+class MPTPreTrainedModel(PreTrainedModel):
+    config_class = MPTConfig
+    base_model_prefix = 'model'
+    _no_split_modules = ['MPTBlock']
+class MPTModel(MPTPreTrainedModel):
+    def __init__(self, config: MPTConfig):
+        config._validate_config()
+        super().__init__(config)
+        self.attn_impl = config.attn_config['attn_impl']
+        self.prefix_lm = config.attn_config['prefix_lm']
+        self.attn_uses_sequence_id = config.attn_config['attn_uses_sequence_id']
+        self.alibi = config.attn_config['alibi']
+        self.alibi_bias_max = config.attn_config['alibi_bias_max']
+        if config.init_device == 'mixed':
+            if dist.get_local_rank() == 0:
+                config.init_device = 'cpu'
+            else:
+                config.init_device = 'meta'
+        if config.norm_type.lower() not in NORM_CLASS_REGISTRY.keys():
+            norm_options = ' | '.join(NORM_CLASS_REGISTRY.keys())
+            raise NotImplementedError(f'Requested norm type ({config.norm_type}) is not implemented within this repo (Options: {norm_options}).')
+        norm_class = NORM_CLASS_REGISTRY[config.norm_type.lower()]
+        self.embedding_fraction = config.embedding_fraction
+        self.wte = SharedEmbedding(config.vocab_size, config.d_model, device=config.init_device)
+        if not self.alibi:
+            self.wpe = torch.nn.Embedding(config.max_seq_len, config.d_model, device=config.init_device)
+        self.emb_drop = nn.Dropout(config.emb_pdrop)
+        self.blocks = nn.ModuleList([MPTBlock(device=config.init_device, **config.to_dict()) for _ in range(config.n_layers)])
+        self.norm_f = norm_class(config.d_model, device=config.init_device)
+        if config.init_device != 'meta':
+            print(f'You are using config.init_device={config.init_device!r}, but you can also use config.init_device="meta" with Composer + FSDP for fast initialization.')
+            self.apply(self.param_init_fn)
+        self.is_causal = not self.prefix_lm
+        self._attn_bias_initialized = False
+        self.attn_bias = None
+        self.attn_bias_shape = attn_bias_shape(self.attn_impl, config.n_heads, config.max_seq_len, self.alibi, prefix_lm=self.prefix_lm, causal=self.is_causal, use_sequence_id=self.attn_uses_sequence_id)
+        if config.no_bias:
+            for module in self.modules():
+                if hasattr(module, 'bias') and isinstance(module.bias, nn.Parameter):
+                    if config.verbose:
+                        warnings.warn(f'Removing bias ({module.bias}) from {module}.')
+                    module.register_parameter('bias', None)
+        if config.verbose and config.verbose > 2:
+            print(self)
+        if 'verbose' not in self.config.init_config:
+            self.config.init_config['verbose'] = self.config.verbose
+        if self.config.init_config['verbose'] > 1:
+            init_fn_name = self.config.init_config['name']
+            warnings.warn(f'Using {init_fn_name} initialization.')
+        self.gradient_checkpointing = False
+    def get_input_embeddings(self):
+        return self.wte
+    def set_input_embeddings(self, value):
+        self.wte = value
+    @torch.no_grad()
+    def _attn_bias(self, device, dtype, attention_mask: Optional[torch.ByteTensor]=None, prefix_mask: Optional[torch.ByteTensor]=None, sequence_id: Optional[torch.LongTensor]=None):
+        if not self._attn_bias_initialized:
+            if self.attn_bias_shape:
+                self.attn_bias = torch.zeros(self.attn_bias_shape, device=device, dtype=dtype)
+                self.attn_bias = build_attn_bias(self.attn_impl, self.attn_bias, self.config.n_heads, self.config.max_seq_len, causal=self.is_causal, alibi=self.alibi, alibi_bias_max=self.alibi_bias_max)
+            self._attn_bias_initialized = True
+        if self.attn_impl == 'flash':
+            return (self.attn_bias, attention_mask)
+        if self.attn_bias is not None:
+            self.attn_bias = self.attn_bias.to(dtype=dtype, device=device)
+        attn_bias = self.attn_bias
+        if self.prefix_lm:
+            assert isinstance(attn_bias, torch.Tensor)
+            assert isinstance(prefix_mask, torch.Tensor)
+            attn_bias = self._apply_prefix_mask(attn_bias, prefix_mask)
+        if self.attn_uses_sequence_id and sequence_id is not None:
+            assert isinstance(attn_bias, torch.Tensor)
+            attn_bias = self._apply_sequence_id(attn_bias, sequence_id)
+        if attention_mask is not None:
+            s_k = attention_mask.shape[-1]
+            if attn_bias is None:
+                attn_bias = torch.zeros((1, 1, 1, s_k), device=device, dtype=dtype)
+            else:
+                _s_k = max(0, attn_bias.size(-1) - s_k)
+                attn_bias = attn_bias[:, :, :, _s_k:]
+            if prefix_mask is not None and attention_mask.shape != prefix_mask.shape:
+                raise ValueError(f'attention_mask shape={attention_mask.shape} ' + f'and prefix_mask shape={prefix_mask.shape} are not equal.')
+            min_val = torch.finfo(attn_bias.dtype).min
+            attn_bias = attn_bias.masked_fill(~attention_mask.view(-1, 1, 1, s_k), min_val)
+        return (attn_bias, None)
+    def _apply_prefix_mask(self, attn_bias: torch.Tensor, prefix_mask: torch.Tensor):
+        (s_k, s_q) = attn_bias.shape[-2:]
+        if s_k != self.config.max_seq_len or s_q != self.config.max_seq_len:
+            raise ValueError('attn_bias does not match the expected shape. ' + f'The last two dimensions should both be {self.config.max_length} ' + f'but are {s_k} and {s_q}.')
+        seq_len = prefix_mask.shape[-1]
+        if seq_len > self.config.max_seq_len:
+            raise ValueError(f'prefix_mask sequence length cannot exceed max_seq_len={self.config.max_seq_len}')
+        attn_bias = attn_bias[..., :seq_len, :seq_len]
+        causal = torch.tril(torch.ones((seq_len, seq_len), dtype=torch.bool, device=prefix_mask.device)).view(1, 1, seq_len, seq_len)
+        prefix = prefix_mask.view(-1, 1, 1, seq_len)
+        cannot_attend = ~torch.logical_or(causal, prefix.bool())
+        min_val = torch.finfo(attn_bias.dtype).min
+        attn_bias = attn_bias.masked_fill(cannot_attend, min_val)
+        return attn_bias
+    def _apply_sequence_id(self, attn_bias: torch.Tensor, sequence_id: torch.LongTensor):
+        seq_len = sequence_id.shape[-1]
+        if seq_len > self.config.max_seq_len:
+            raise ValueError(f'sequence_id sequence length cannot exceed max_seq_len={self.config.max_seq_len}')
+        attn_bias = attn_bias[..., :seq_len, :seq_len]
+        cannot_attend = torch.logical_not(torch.eq(sequence_id.view(-1, seq_len, 1), sequence_id.view(-1, 1, seq_len))).unsqueeze(1)
+        min_val = torch.finfo(attn_bias.dtype).min
+        attn_bias = attn_bias.masked_fill(cannot_attend, min_val)
+        return attn_bias
+    def forward(self, input_ids: torch.LongTensor, past_key_values: Optional[List[Tuple[torch.FloatTensor]]]=None, attention_mask: Optional[torch.ByteTensor]=None, prefix_mask: Optional[torch.ByteTensor]=None, sequence_id: Optional[torch.LongTensor]=None, return_dict: Optional[bool]=None, output_attentions: Optional[bool]=None, output_hidden_states: Optional[bool]=None, use_cache: Optional[bool]=None, inputs_embeds: Optional[torch.Tensor]=None):
+        return_dict = return_dict if return_dict is not None else self.config.return_dict
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        if attention_mask is not None:
+            attention_mask = attention_mask.bool()
+        if prefix_mask is not None:
+            prefix_mask = prefix_mask.bool()
+        if not return_dict:
+            raise NotImplementedError('return_dict False is not implemented yet for MPT')
+        if output_attentions:
+            if self.attn_impl != 'torch':
+                raise NotImplementedError('output_attentions is not implemented for MPT when using attn_impl `flash` or `triton`.')
+        if attention_mask is not None and attention_mask[:, 0].sum() != attention_mask.shape[0] and self.training:
+            raise NotImplementedError('MPT does not support training with left padding.')
+        if self.prefix_lm and prefix_mask is None:
+            raise ValueError('prefix_mask is a required argument when MPT is configured with prefix_lm=True.')
+        if self.training:
+            if self.attn_uses_sequence_id and sequence_id is None:
+                raise ValueError('sequence_id is a required argument when MPT is configured with attn_uses_sequence_id=True ' + 'and the model is in train mode.')
+            elif self.attn_uses_sequence_id is False and sequence_id is not None:
+                warnings.warn('MPT received non-None input for `sequence_id` but is configured with attn_uses_sequence_id=False. ' + 'This input will be ignored. If you want the model to use `sequence_id`, set attn_uses_sequence_id to True.')
+        if input_ids is not None:
+            S = input_ids.size(1)
+            assert S <= self.config.max_seq_len, f'Cannot forward input with seq_len={S}, this model only supports seq_len<={self.config.max_seq_len}'
+            tok_emb = self.wte(input_ids)
+        else:
+            assert inputs_embeds is not None
+            assert self.alibi, 'inputs_embeds is not implemented for MPT unless for alibi.'
+            S = inputs_embeds.size(1)
+            tok_emb = inputs_embeds
+        if self.alibi:
+            x = tok_emb
+        else:
+            past_position = 0
+            if past_key_values is not None:
+                if len(past_key_values) != self.config.n_layers:
+                    raise ValueError(f'past_key_values must provide a past_key_value for each attention ' + f'layer in the network (len(past_key_values)={len(past_key_values)!r}; self.config.n_layers={self.config.n_layers!r}).')
+                past_position = past_key_values[0][0].size(1)
+                if self.attn_impl == 'torch':
+                    past_position = past_key_values[0][0].size(3)
+            if S + past_position > self.config.max_seq_len:
+                raise ValueError(f'Cannot forward input with past sequence length {past_position} and current sequence length {S + 1}, this model only supports total sequence length <= {self.config.max_seq_len}.')
+            pos = torch.arange(past_position, S + past_position, dtype=torch.long, device=input_ids.device).unsqueeze(0)
+            if attention_mask is not None:
+                pos = torch.clamp(pos - torch.cumsum((~attention_mask).to(torch.int32), dim=1)[:, past_position:], min=0)
+            pos_emb = self.wpe(pos)
+            x = tok_emb + pos_emb
+        if self.embedding_fraction == 1:
+            x = self.emb_drop(x)
+        else:
+            x_shrunk = x * self.embedding_fraction + x.detach() * (1 - self.embedding_fraction)
+            assert isinstance(self.emb_drop, nn.Module)
+            x = self.emb_drop(x_shrunk)
+        (attn_bias, attention_mask) = self._attn_bias(device=x.device, dtype=torch.float32, attention_mask=attention_mask, prefix_mask=prefix_mask, sequence_id=sequence_id)
+        if use_cache and past_key_values is None:
+            past_key_values = [() for _ in range(self.config.n_layers)]
+        all_hidden_states = () if output_hidden_states else None
+        all_self_attns = () if output_attentions else None
+        for (b_idx, block) in enumerate(self.blocks):
+            if output_hidden_states:
+                assert all_hidden_states is not None
+                all_hidden_states = all_hidden_states + (x,)
+            past_key_value = past_key_values[b_idx] if past_key_values is not None else None
+            if self.gradient_checkpointing and self.training:
+                (x, attn_weights, past_key_value) = torch.utils.checkpoint.checkpoint(block, x, past_key_value, attn_bias, attention_mask, self.is_causal)
+            else:
+                (x, attn_weights, past_key_value) = block(x, past_key_value=past_key_value, attn_bias=attn_bias, attention_mask=attention_mask, is_causal=self.is_causal)
+            if past_key_values is not None:
+                past_key_values[b_idx] = past_key_value
+            if output_attentions:
+                assert all_self_attns is not None
+                all_self_attns = all_self_attns + (attn_weights,)
+        x = self.norm_f(x)
+        if output_hidden_states:
+            assert all_hidden_states is not None
+            all_hidden_states = all_hidden_states + (x,)
+        return BaseModelOutputWithPast(last_hidden_state=x, past_key_values=past_key_values, hidden_states=all_hidden_states, attentions=all_self_attns)
+    def param_init_fn(self, module):
+        init_fn_name = self.config.init_config['name']
+        MODEL_INIT_REGISTRY[init_fn_name](module=module, n_layers=self.config.n_layers, d_model=self.config.d_model, **self.config.init_config)
+    def fsdp_wrap_fn(self, module):
+        return isinstance(module, MPTBlock)
+    def activation_checkpointing_fn(self, module):
+        return isinstance(module, MPTBlock)
+class MPTForCausalLM(MPTPreTrainedModel):
+    def __init__(self, config: MPTConfig):
+        super().__init__(config)
+        if not config.tie_word_embeddings:
+            raise ValueError('MPTForCausalLM only supports tied word embeddings')
+        print(f'Instantiating an MPTForCausalLM model from {__file__}')
+        self.transformer = MPTModel(config)
+        for child in self.transformer.children():
+            if isinstance(child, torch.nn.ModuleList):
+                continue
+            if isinstance(child, torch.nn.Module):
+                child._fsdp_wrap = True
+        self.logit_scale = None
+        if config.logit_scale is not None:
+            logit_scale = config.logit_scale
+            if isinstance(logit_scale, str):
+                if logit_scale == 'inv_sqrt_d_model':
+                    logit_scale = 1 / math.sqrt(config.d_model)
+                else:
+                    raise ValueError(f"logit_scale={logit_scale!r} is not recognized as an option; use numeric value or 'inv_sqrt_d_model'.")
+            self.logit_scale = logit_scale
+    def get_input_embeddings(self):
+        return self.transformer.wte
+    def set_input_embeddings(self, value):
+        self.transformer.wte = value
+    def get_output_embeddings(self):
+        return self.transformer.wte
+    def set_output_embeddings(self, new_embeddings):
+        self.transformer.wte = new_embeddings
+    def set_decoder(self, decoder):
+        self.transformer = decoder
+    def get_decoder(self):
+        return self.transformer
+    def forward(self, input_ids: torch.LongTensor, past_key_values: Optional[List[Tuple[torch.FloatTensor]]]=None, attention_mask: Optional[torch.ByteTensor]=None, prefix_mask: Optional[torch.ByteTensor]=None, sequence_id: Optional[torch.LongTensor]=None, labels: Optional[torch.LongTensor]=None, return_dict: Optional[bool]=None, output_attentions: Optional[bool]=None, output_hidden_states: Optional[bool]=None, use_cache: Optional[bool]=None, inputs_embeds: Optional[torch.FloatTensor]=None):
+        return_dict = return_dict if return_dict is not None else self.config.return_dict
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        if inputs_embeds is not None:
+            raise NotImplementedError('inputs_embeds has to be None (for hf/peft support).')
+        outputs = self.transformer(input_ids=input_ids, past_key_values=past_key_values, attention_mask=attention_mask, prefix_mask=prefix_mask, sequence_id=sequence_id, return_dict=return_dict, output_attentions=output_attentions, output_hidden_states=output_hidden_states, use_cache=use_cache)
+        logits = self.transformer.wte(outputs.last_hidden_state.to(self.transformer.wte.weight.device), True)
+        if self.logit_scale is not None:
+            if self.logit_scale == 0:
+                warnings.warn(f'Multiplying logits by self.logit_scale={self.logit_scale!r}. This will produce uniform (uninformative) outputs.')
+            logits *= self.logit_scale
+        loss = None
+        if labels is not None:
+            labels = torch.roll(labels, shifts=-1)
+            labels[:, -1] = -100
+            loss = F.cross_entropy(logits.view(-1, logits.size(-1)), labels.to(logits.device).view(-1))
+        return CausalLMOutputWithPast(loss=loss, logits=logits, past_key_values=outputs.past_key_values, hidden_states=outputs.hidden_states, attentions=outputs.attentions)
+    def param_init_fn(self, module):
+        init_fn_name = self.config.init_config['name']
+        MODEL_INIT_REGISTRY[init_fn_name](module=module, n_layers=self.config.n_layers, d_model=self.config.d_model, **self.config.init_config)
+    def fsdp_wrap_fn(self, module):
+        return isinstance(module, MPTBlock)
+    def activation_checkpointing_fn(self, module):
+        return isinstance(module, MPTBlock)
+    def prepare_inputs_for_generation(self, input_ids, past_key_values=None, inputs_embeds=None, **kwargs):
+        if inputs_embeds is not None:
+            raise NotImplementedError('inputs_embeds is not implemented for MPT yet')
+        attention_mask = kwargs['attention_mask'].bool()
+        if attention_mask[:, -1].sum() != attention_mask.shape[0]:
+            raise NotImplementedError('MPT does not support generation with right padding.')
+        if self.transformer.attn_uses_sequence_id and self.training:
+            sequence_id = torch.zeros_like(input_ids[:1])
+        else:
+            sequence_id = None
+        if past_key_values is not None:
+            input_ids = input_ids[:, -1].unsqueeze(-1)
+        if self.transformer.prefix_lm:
+            prefix_mask = torch.ones_like(attention_mask)
+            if kwargs.get('use_cache') == False:
+                raise NotImplementedError('MPT with prefix_lm=True does not support use_cache=False.')
+        else:
+            prefix_mask = None
+        return {'input_ids': input_ids, 'attention_mask': attention_mask, 'prefix_mask': prefix_mask, 'sequence_id': sequence_id, 'past_key_values': past_key_values, 'use_cache': kwargs.get('use_cache', True)}
+    @staticmethod
+    def _reorder_cache(past_key_values, beam_idx):
+        """Used by HuggingFace generate when using beam search with kv-caching.
+        See https://github.com/huggingface/transformers/blob/3ec7a47664ebe40c40f4b722f6bb1cd30c3821ec/src/transformers/models/gpt2/modeling_gpt2.py#L1122-L1133
+        for an example in transformers.
+        """
+        reordered_past = []
+        for layer_past in past_key_values:
+            reordered_past += [tuple((past_state.index_select(0, beam_idx) for past_state in layer_past))]
+        return reordered_past

dam/model/language_model/mpt_ignored/norm.py ADDED Viewed

	@@ -0,0 +1,56 @@

+import torch
+def _cast_if_autocast_enabled(tensor):
+    if torch.is_autocast_enabled():
+        if tensor.device.type == 'cuda':
+            dtype = torch.get_autocast_gpu_dtype()
+        elif tensor.device.type == 'cpu':
+            dtype = torch.get_autocast_cpu_dtype()
+        else:
+            raise NotImplementedError()
+        return tensor.to(dtype=dtype)
+    return tensor
+class LPLayerNorm(torch.nn.LayerNorm):
+    def __init__(self, normalized_shape, eps=1e-05, elementwise_affine=True, device=None, dtype=None):
+        super().__init__(normalized_shape=normalized_shape, eps=eps, elementwise_affine=elementwise_affine, device=device, dtype=dtype)
+    def forward(self, x):
+        module_device = x.device
+        downcast_x = _cast_if_autocast_enabled(x)
+        downcast_weight = _cast_if_autocast_enabled(self.weight) if self.weight is not None else self.weight
+        downcast_bias = _cast_if_autocast_enabled(self.bias) if self.bias is not None else self.bias
+        with torch.autocast(enabled=False, device_type=module_device.type):
+            return torch.nn.functional.layer_norm(downcast_x, self.normalized_shape, downcast_weight, downcast_bias, self.eps)
+def rms_norm(x, weight=None, eps=1e-05):
+    output = x * torch.rsqrt(x.pow(2).mean(-1, keepdim=True) + eps)
+    if weight is not None:
+        return output * weight
+    return output
+class RMSNorm(torch.nn.Module):
+    def __init__(self, normalized_shape, eps=1e-05, weight=True, dtype=None, device=None):
+        super().__init__()
+        self.eps = eps
+        if weight:
+            self.weight = torch.nn.Parameter(torch.ones(normalized_shape, dtype=dtype, device=device))
+        else:
+            self.register_parameter('weight', None)
+    def forward(self, x):
+        return rms_norm(x.float(), self.weight, self.eps).to(dtype=x.dtype)
+class LPRMSNorm(RMSNorm):
+    def __init__(self, normalized_shape, eps=1e-05, weight=True, dtype=None, device=None):
+        super().__init__(normalized_shape=normalized_shape, eps=eps, weight=weight, dtype=dtype, device=device)
+    def forward(self, x):
+        downcast_x = _cast_if_autocast_enabled(x)
+        downcast_weight = _cast_if_autocast_enabled(self.weight) if self.weight is not None else self.weight
+        with torch.autocast(enabled=False, device_type=x.device.type):
+            return rms_norm(downcast_x, downcast_weight, self.eps).to(dtype=x.dtype)
+NORM_CLASS_REGISTRY = {'layernorm': torch.nn.LayerNorm, 'low_precision_layernorm': LPLayerNorm, 'rmsnorm': RMSNorm, 'low_precision_rmsnorm': LPRMSNorm}

dam/model/language_model/mpt_ignored/param_init_fns.py ADDED Viewed

	@@ -0,0 +1,181 @@

+import math
+import warnings
+from collections.abc import Sequence
+from functools import partial
+from typing import Optional, Tuple, Union
+import torch
+from torch import nn
+from .norm import NORM_CLASS_REGISTRY
+def torch_default_param_init_fn_(module: nn.Module, verbose: int=0, **kwargs):
+    del kwargs
+    if verbose > 1:
+        warnings.warn(f"Initializing network using module's reset_parameters attribute")
+    if hasattr(module, 'reset_parameters'):
+        module.reset_parameters()
+def fused_init_helper_(module: nn.Module, init_fn_):
+    _fused = getattr(module, '_fused', None)
+    if _fused is None:
+        raise RuntimeError(f'Internal logic error')
+    (dim, splits) = _fused
+    splits = (0, *splits, module.weight.size(dim))
+    for (s, e) in zip(splits[:-1], splits[1:]):
+        slice_indices = [slice(None)] * module.weight.ndim
+        slice_indices[dim] = slice(s, e)
+        init_fn_(module.weight[slice_indices])
+def generic_param_init_fn_(module: nn.Module, init_fn_, n_layers: int, d_model: Optional[int]=None, init_div_is_residual: Union[int, float, str, bool]=True, emb_init_std: Optional[float]=None, emb_init_uniform_lim: Optional[Union[Tuple[float, float], float]]=None, verbose: int=0, **kwargs):
+    del kwargs
+    if verbose > 1:
+        warnings.warn(f'If model has bias parameters they are initialized to 0.')
+    init_div_is_residual = init_div_is_residual
+    if init_div_is_residual is False:
+        div_is_residual = 1.0
+    elif init_div_is_residual is True:
+        div_is_residual = math.sqrt(2 * n_layers)
+    elif isinstance(init_div_is_residual, float) or isinstance(init_div_is_residual, int):
+        div_is_residual = init_div_is_residual
+    elif isinstance(init_div_is_residual, str) and init_div_is_residual.isnumeric():
+        div_is_residual = float(init_div_is_residual)
+    else:
+        div_is_residual = 1.0
+        raise ValueError(f'Expected init_div_is_residual to be boolean or numeric, got {init_div_is_residual}')
+    if init_div_is_residual is not False:
+        if verbose > 1:
+            warnings.warn(f'Initializing _is_residual layers then dividing them by {div_is_residual:.3f}. ' + f'Set `init_div_is_residual: false` in init config to disable this.')
+    if isinstance(module, nn.Linear):
+        if hasattr(module, '_fused'):
+            fused_init_helper_(module, init_fn_)
+        else:
+            init_fn_(module.weight)
+        if module.bias is not None:
+            torch.nn.init.zeros_(module.bias)
+        if init_div_is_residual is not False and getattr(module, '_is_residual', False):
+            with torch.no_grad():
+                module.weight.div_(div_is_residual)
+    elif isinstance(module, nn.Embedding):
+        if emb_init_std is not None:
+            std = emb_init_std
+            if std == 0:
+                warnings.warn(f'Embedding layer initialized to 0.')
+            emb_init_fn_ = partial(torch.nn.init.normal_, mean=0.0, std=std)
+            if verbose > 1:
+                warnings.warn(f'Embedding layer initialized using normal distribution with mean=0 and std={std!r}.')
+        elif emb_init_uniform_lim is not None:
+            lim = emb_init_uniform_lim
+            if isinstance(lim, Sequence):
+                if len(lim) > 2:
+                    raise ValueError(f'Uniform init requires a min and a max limit. User input: {lim}.')
+                if lim[0] == lim[1]:
+                    warnings.warn(f'Embedding layer initialized to {lim[0]}.')
+            else:
+                if lim == 0:
+                    warnings.warn(f'Embedding layer initialized to 0.')
+                lim = [-lim, lim]
+            (a, b) = lim
+            emb_init_fn_ = partial(torch.nn.init.uniform_, a=a, b=b)
+            if verbose > 1:
+                warnings.warn(f'Embedding layer initialized using uniform distribution in range {lim}.')
+        else:
+            emb_init_fn_ = init_fn_
+        emb_init_fn_(module.weight)
+    elif isinstance(module, tuple(set(NORM_CLASS_REGISTRY.values()))):
+        if verbose > 1:
+            warnings.warn(f'Norm weights are set to 1. If norm layer has a bias it is initialized to 0.')
+        if hasattr(module, 'weight') and module.weight is not None:
+            torch.nn.init.ones_(module.weight)
+        if hasattr(module, 'bias') and module.bias is not None:
+            torch.nn.init.zeros_(module.bias)
+    elif isinstance(module, nn.MultiheadAttention):
+        if module._qkv_same_embed_dim:
+            assert module.in_proj_weight is not None
+            assert module.q_proj_weight is None and module.k_proj_weight is None and (module.v_proj_weight is None)
+            assert d_model is not None
+            _d = d_model
+            splits = (0, _d, 2 * _d, 3 * _d)
+            for (s, e) in zip(splits[:-1], splits[1:]):
+                init_fn_(module.in_proj_weight[s:e])
+        else:
+            assert module.q_proj_weight is not None and module.k_proj_weight is not None and (module.v_proj_weight is not None)
+            assert module.in_proj_weight is None
+            init_fn_(module.q_proj_weight)
+            init_fn_(module.k_proj_weight)
+            init_fn_(module.v_proj_weight)
+        if module.in_proj_bias is not None:
+            torch.nn.init.zeros_(module.in_proj_bias)
+        if module.bias_k is not None:
+            torch.nn.init.zeros_(module.bias_k)
+        if module.bias_v is not None:
+            torch.nn.init.zeros_(module.bias_v)
+        init_fn_(module.out_proj.weight)
+        if init_div_is_residual is not False and getattr(module.out_proj, '_is_residual', False):
+            with torch.no_grad():
+                module.out_proj.weight.div_(div_is_residual)
+        if module.out_proj.bias is not None:
+            torch.nn.init.zeros_(module.out_proj.bias)
+    else:
+        for _ in module.parameters(recurse=False):
+            raise NotImplementedError(f'{module.__class__.__name__} parameters are not initialized by param_init_fn.')
+def _normal_init_(std, mean=0.0):
+    return partial(torch.nn.init.normal_, mean=mean, std=std)
+def _normal_param_init_fn_(module: nn.Module, std: float, n_layers: int, d_model: Optional[int]=None, init_div_is_residual: Union[int, float, str, bool]=True, emb_init_std: Optional[float]=None, emb_init_uniform_lim: Optional[Union[Tuple[float, float], float]]=None, verbose: int=0, **kwargs):
+    del kwargs
+    init_fn_ = _normal_init_(std=std)
+    if verbose > 1:
+        warnings.warn(f'Using torch.nn.init.normal_ init fn mean=0.0, std={std}')
+    generic_param_init_fn_(module=module, init_fn_=init_fn_, d_model=d_model, n_layers=n_layers, init_div_is_residual=init_div_is_residual, emb_init_std=emb_init_std, emb_init_uniform_lim=emb_init_uniform_lim, verbose=verbose)
+def baseline_param_init_fn_(module: nn.Module, init_std: float, n_layers: int, d_model: Optional[int]=None, init_div_is_residual: Union[int, float, str, bool]=True, emb_init_std: Optional[float]=None, emb_init_uniform_lim: Optional[Union[Tuple[float, float], float]]=None, verbose: int=0, **kwargs):
+    del kwargs
+    if init_std is None:
+        raise ValueError("You must set model.init_config['init_std'] to a float value to use the default initialization scheme.")
+    _normal_param_init_fn_(module=module, std=init_std, d_model=d_model, n_layers=n_layers, init_div_is_residual=init_div_is_residual, emb_init_std=emb_init_std, emb_init_uniform_lim=emb_init_uniform_lim, verbose=verbose)
+def small_param_init_fn_(module: nn.Module, n_layers: int, d_model: int, init_div_is_residual: Union[int, float, str, bool]=True, emb_init_std: Optional[float]=None, emb_init_uniform_lim: Optional[Union[Tuple[float, float], float]]=None, verbose: int=0, **kwargs):
+    del kwargs
+    std = math.sqrt(2 / (5 * d_model))
+    _normal_param_init_fn_(module=module, std=std, d_model=d_model, n_layers=n_layers, init_div_is_residual=init_div_is_residual, emb_init_std=emb_init_std, emb_init_uniform_lim=emb_init_uniform_lim, verbose=verbose)
+def neox_param_init_fn_(module: nn.Module, n_layers: int, d_model: int, emb_init_std: Optional[float]=None, emb_init_uniform_lim: Optional[Union[Tuple[float, float], float]]=None, verbose: int=0, **kwargs):
+    """From section 2.3.1 of GPT-NeoX-20B:
+    An Open-Source AutoregressiveLanguage Model — Black et. al. (2022)
+    see https://github.com/EleutherAI/gpt-neox/blob/9610391ab319403cef079b438edd016a2443af54/megatron/model/init_functions.py#L151
+    and https://github.com/EleutherAI/gpt-neox/blob/main/megatron/model/transformer.py
+    """
+    del kwargs
+    residual_div = n_layers / math.sqrt(10)
+    if verbose > 1:
+        warnings.warn(f'setting init_div_is_residual to {residual_div}')
+    small_param_init_fn_(module=module, d_model=d_model, n_layers=n_layers, init_div_is_residual=residual_div, emb_init_std=emb_init_std, emb_init_uniform_lim=emb_init_uniform_lim, verbose=verbose)
+def kaiming_uniform_param_init_fn_(module: nn.Module, n_layers: int, d_model: Optional[int]=None, init_div_is_residual: Union[int, float, str, bool]=True, emb_init_std: Optional[float]=None, emb_init_uniform_lim: Optional[Union[Tuple[float, float], float]]=None, init_gain: float=0, fan_mode: str='fan_in', init_nonlinearity: str='leaky_relu', verbose: int=0, **kwargs):
+    del kwargs
+    if verbose > 1:
+        warnings.warn(f'Using nn.init.kaiming_uniform_ init fn with parameters: ' + f'a={init_gain}, mode={fan_mode}, nonlinearity={init_nonlinearity}')
+    kaiming_uniform_ = partial(nn.init.kaiming_uniform_, a=init_gain, mode=fan_mode, nonlinearity=init_nonlinearity)
+    generic_param_init_fn_(module=module, init_fn_=kaiming_uniform_, d_model=d_model, n_layers=n_layers, init_div_is_residual=init_div_is_residual, emb_init_std=emb_init_std, emb_init_uniform_lim=emb_init_uniform_lim, verbose=verbose)
+def kaiming_normal_param_init_fn_(module: nn.Module, n_layers: int, d_model: Optional[int]=None, init_div_is_residual: Union[int, float, str, bool]=True, emb_init_std: Optional[float]=None, emb_init_uniform_lim: Optional[Union[Tuple[float, float], float]]=None, init_gain: float=0, fan_mode: str='fan_in', init_nonlinearity: str='leaky_relu', verbose: int=0, **kwargs):
+    del kwargs
+    if verbose > 1:
+        warnings.warn(f'Using nn.init.kaiming_normal_ init fn with parameters: ' + f'a={init_gain}, mode={fan_mode}, nonlinearity={init_nonlinearity}')
+    kaiming_normal_ = partial(torch.nn.init.kaiming_normal_, a=init_gain, mode=fan_mode, nonlinearity=init_nonlinearity)
+    generic_param_init_fn_(module=module, init_fn_=kaiming_normal_, d_model=d_model, n_layers=n_layers, init_div_is_residual=init_div_is_residual, emb_init_std=emb_init_std, emb_init_uniform_lim=emb_init_uniform_lim, verbose=verbose)
+def xavier_uniform_param_init_fn_(module: nn.Module, n_layers: int, d_model: Optional[int]=None, init_div_is_residual: Union[int, float, str, bool]=True, emb_init_std: Optional[float]=None, emb_init_uniform_lim: Optional[Union[Tuple[float, float], float]]=None, init_gain: float=0, verbose: int=0, **kwargs):
+    del kwargs
+    xavier_uniform_ = partial(torch.nn.init.xavier_uniform_, gain=init_gain)
+    if verbose > 1:
+        warnings.warn(f'Using torch.nn.init.xavier_uniform_ init fn with parameters: ' + f'gain={init_gain}')
+    generic_param_init_fn_(module=module, init_fn_=xavier_uniform_, d_model=d_model, n_layers=n_layers, init_div_is_residual=init_div_is_residual, emb_init_std=emb_init_std, emb_init_uniform_lim=emb_init_uniform_lim, verbose=verbose)
+def xavier_normal_param_init_fn_(module: nn.Module, n_layers: int, d_model: Optional[int]=None, init_div_is_residual: Union[int, float, str, bool]=True, emb_init_std: Optional[float]=None, emb_init_uniform_lim: Optional[Union[Tuple[float, float], float]]=None, init_gain: float=0, verbose: int=0, **kwargs):
+    xavier_normal_ = partial(torch.nn.init.xavier_normal_, gain=init_gain)
+    if verbose > 1:
+        warnings.warn(f'Using torch.nn.init.xavier_normal_ init fn with parameters: ' + f'gain={init_gain}')
+    generic_param_init_fn_(module=module, init_fn_=xavier_normal_, d_model=d_model, n_layers=n_layers, init_div_is_residual=init_div_is_residual, emb_init_std=emb_init_std, emb_init_uniform_lim=emb_init_uniform_lim, verbose=verbose)
+MODEL_INIT_REGISTRY = {'default_': torch_default_param_init_fn_, 'baseline_': baseline_param_init_fn_, 'kaiming_uniform_': kaiming_uniform_param_init_fn_, 'kaiming_normal_': kaiming_normal_param_init_fn_, 'neox_init_': neox_param_init_fn_, 'small_init_': small_param_init_fn_, 'xavier_uniform_': xavier_uniform_param_init_fn_, 'xavier_normal_': xavier_normal_param_init_fn_}

dam/model/llava_arch.py ADDED Viewed

	@@ -0,0 +1,676 @@

+#    Copyright 2023 Haotian Liu
+#
+#    Licensed under the Apache License, Version 2.0 (the "License");
+#    you may not use this file except in compliance with the License.
+#    You may obtain a copy of the License at
+#
+#        http://www.apache.org/licenses/LICENSE-2.0
+#
+#    Unless required by applicable law or agreed to in writing, software
+#    distributed under the License is distributed on an "AS IS" BASIS,
+#    WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+#    See the License for the specific language governing permissions and
+#    limitations under the License.
+import os, sys, os.path as osp
+import warnings
+from abc import ABC, abstractmethod
+import torch, logging
+from transformers import (
+    AutoTokenizer,
+    AutoModel,
+    AutoModelForCausalLM,
+    AutoConfig,
+    BitsAndBytesConfig,
+    PretrainedConfig,
+    PreTrainedModel,
+)
+from .constants import (
+    DEFAULT_IM_END_TOKEN,
+    DEFAULT_IM_START_TOKEN,
+    DEFAULT_IMAGE_PATCH_TOKEN,
+    IGNORE_INDEX,
+    IMAGE_TOKEN_INDEX,
+    MASK_TOKEN_INDEX,
+)
+from collections import OrderedDict
+from .utils import get_model_config
+from .language_model.builder import build_llm_and_tokenizer
+from .multimodal_encoder.builder import build_vision_tower, build_context_provider
+from .multimodal_projector.builder import build_mm_projector
+from .configuration_llava import LlavaConfig
+from transformers.modeling_utils import ContextManagers, no_init_weights
+## TODO decide whether should we use metaclass
+class LlavaMetaModel(ABC):
+    def init_vlm(self, config: PreTrainedModel = None, *args, **kwargs):
+        # TODO(ligeng): figure out how from_config and from_pretrained works in HF implementation.
+        if hasattr(self, "llm") or hasattr(self, "vision_tower")  or hasattr(self, "mm_projector"):
+            # already initialized, skipped
+            return
+        model_dtype = getattr(config, "model_dtype", "torch.float16")
+        if not hasattr(config, "model_dtype"):
+            warnings.warn("model_dtype not found in config, defaulting to torch.float16.")
+            config.model_dtype = model_dtype
+        # print("init_vlm(): config", config); input("DEBUG init_vlm")
+        cfgs = get_model_config(config)
+        # Only the first three are required. Others are optional.
+        llm_cfg, vision_tower_cfg, mm_projector_cfg, mask_encoder_cfg, context_provider_cfg = cfgs
+        if llm_cfg is None or vision_tower_cfg is None or mm_projector_cfg is None:
+            raise ValueError("`llm_cfg` `mm_projector_cfg` `vision_tower_cfg` not found in the config.")
+        # print("init_vlm():", cfgs); input("DEBUG init_vlm")
+        # print(llm_cfg, vision_tower_cfg, mm_projector_cfg); input("DEBUG init_vlm")
+        self.llm, self.tokenizer = build_llm_and_tokenizer(llm_cfg, config, *args, **kwargs)
+        self.vision_tower = build_vision_tower(vision_tower_cfg, config)
+        self.mm_projector = build_mm_projector(mm_projector_cfg, config)
+        self.context_provider = build_context_provider(context_provider_cfg, config) if context_provider_cfg is not None else None
+        self.post_config()
+        self.is_loaded = True
+        assert (
+            self.llm is not None or self.vision_tower is not None or self.mm_projector is not None
+        ), "At least one of the components must be instantiated."
+    @classmethod
+    def load_from_config(cls, model_path_or_config, *args, **kwargs):
+        pass
+    ## FIXME we will use this function to load model in the future
+    @classmethod
+    def load_pretrained(cls, model_path_or_config, *args, **kwargs):
+        kwargs.pop("config", None)
+        if isinstance(model_path_or_config, str):
+            config = AutoConfig.from_pretrained(model_path_or_config)
+        elif isinstance(model_path_or_config, LlavaConfig):
+            config = model_path_or_config
+        else:
+            raise NotImplementedError(f"wrong type, {type(model_path_or_config)} \
+                                      {isinstance(model_path_or_config, LlavaConfig)}")
+        model_dtype = getattr(config, "model_dtype", "torch.float16")
+        if not hasattr(config, "model_dtype"):
+            warnings.warn("model_dtype not found in config, defaulting to torch.float16.")
+            config.model_dtype = model_dtype
+        cfgs = get_model_config(config)
+        # Only the first three are required. Others are optional.
+        llm_cfg, vision_tower_cfg, mm_projector_cfg, mask_encoder_cfg, context_provider_cfg = cfgs
+        if llm_cfg is None or vision_tower_cfg is None or mm_projector_cfg is None:
+            raise ValueError("`llm_cfg` `mm_projector_cfg` `vision_tower_cfg` not found in the config.")
+        # print(llm_cfg, vision_tower_cfg, mm_projector_cfg); input("DEBUG load_pretrained")
+        with ContextManagers([no_init_weights(_enable=True),]):
+            vlm = cls(config, *args, **kwargs)
+        # print(llm_cfg, vision_tower_cfg, mm_projector_cfg); input("DEBUG load_pretrained finish")
+        if hasattr(vlm, "llm") or hasattr(vlm, "vision_tower")  or hasattr(vlm, "mm_projector"):
+            if vlm.is_loaded:
+                return vlm
+        vlm.llm, vlm.tokenizer = build_llm_and_tokenizer(llm_cfg, config, *args, **kwargs)
+        vlm.vision_tower = build_vision_tower(vision_tower_cfg, config)
+        vlm.mm_projector = build_mm_projector(mm_projector_cfg, config)
+        if mask_encoder_cfg is not None:
+            raise NotImplementedError("Mask encoder is not supported.")
+        vlm.context_provider = build_context_provider(context_provider_cfg, config) if context_provider_cfg is not None else None
+        self.post_config()
+        self.is_loaded = True
+        # FIXME(ligeng, yunhao): llm should never be none here.
+        assert (
+            vlm.llm is not None or vlm.vision_tower is not None or vlm.mm_projector is not None
+        ), "At least one of the components must be instantiated."
+        return vlm
+    ## FIXME we will use this function to save the model in the future
+    def save_pretrained(self, output_dir, state_dict=None):
+        if state_dict is None:
+            # other wise fetch from deepspeed
+            # state_dict = accelerator.get_state_dict(is_deepspeed_enabled)
+            state_dict = self.state_dict()
+        if getattr(self, "tokenizer", None):
+            self.tokenizer.save_pretrained(osp.join(output_dir, "llm"))
+        if self.get_llm():
+            print(f"saving llm to {osp.join(output_dir, 'llm')}")
+            self.llm.config._name_or_path = osp.join(output_dir, "llm")
+            llm_state_dict = OrderedDict({k.split("llm.")[-1]: v for k, v in state_dict.items() if "llm" in k})
+            self.llm.save_pretrained(os.path.join(output_dir, "llm"), state_dict=llm_state_dict)
+            self.config.llm_cfg = self.llm.config
+        if self.get_vision_tower() and "radio" not in self.get_vision_tower().__class__.__name__.lower():
+            print(f"saving vision_tower to {osp.join(output_dir, 'vision_tower')}")
+            self.vision_tower.config._name_or_path = osp.join(output_dir, "vision_tower")
+            vision_tower_state_dict = OrderedDict(
+                {k.split("vision_tower.vision_tower.")[-1]: v for k, v in state_dict.items() if "vision_tower" in k}
+            )
+            self.vision_tower.vision_tower.save_pretrained(
+                os.path.join(output_dir, "vision_tower"),
+                state_dict=vision_tower_state_dict,
+            )
+            self.vision_tower.image_processor.save_pretrained(os.path.join(output_dir, "vision_tower"))
+            self.config.vision_tower_cfg = self.vision_tower.config
+            if hasattr(self.config.vision_tower_cfg, 'auto_map'):
+                delattr(self.config.vision_tower_cfg, 'auto_map')
+        if self.get_mm_projector():
+            print(f"saving mm_projector to {osp.join(output_dir, 'mm_projector')}")
+            self.mm_projector.config._name_or_path = osp.join(output_dir, "mm_projector")
+            mm_projector_state_dict = OrderedDict(
+                {k.split("mm_projector.")[-1]: v for k, v in state_dict.items() if "mm_projector" in k}
+            )
+            self.mm_projector.save_pretrained(
+                os.path.join(output_dir, "mm_projector"),
+                state_dict=mm_projector_state_dict,
+            )
+            self.config.mm_projector_cfg = self.mm_projector.config
+        if self.get_context_provider():
+            print(f"saving context_provider to {osp.join(output_dir, 'context_provider')}")
+            self.context_provider.config._name_or_path = osp.join(output_dir, "context_provider")
+            context_provider_state_dict = OrderedDict(
+                {k.split("context_provider.")[-1]: v for k, v in state_dict.items() if "context_provider" in k}
+            )
+            self.context_provider.save_pretrained(
+                os.path.join(output_dir, "context_provider"),
+                state_dict=context_provider_state_dict,
+            )
+            self.config.context_provider_cfg = self.context_provider.config
+        ## update and save top-level config
+        self.config._name_or_path = output_dir
+        self.config.architectures = [self.__class__.__name__]
+        self.config.save_pretrained(output_dir)
+    def get_llm(self):
+        llm = getattr(self, "llm", None)
+        if type(llm) is list:
+            llm = llm[0]
+        return llm
+    def get_lm_head(self):
+        lm_head = getattr(self.get_llm(), "lm_head", None)
+        return lm_head
+    def get_vision_tower(self):
+        vision_tower = getattr(self, "vision_tower", None)
+        if type(vision_tower) is list:
+            vision_tower = vision_tower[0]
+        return vision_tower
+    def get_mm_projector(self):
+        mm_projector = getattr(self, "mm_projector", None)
+        if type(mm_projector) is list:
+            mm_projector = mm_projector[0]
+        return mm_projector
+    def get_context_provider(self):
+        context_provider = getattr(self, "context_provider", None)
+        return context_provider
+    def post_config(self):
+        self.training = self.get_llm().training
+        ## configuration
+        if getattr(self.config, "llm_cfg", None) is None:
+            self.config.llm_cfg = self.llm.config
+        if getattr(self.config, "vision_tower_cfg", None) is None:
+            self.config.vision_tower_cfg = self.vision_tower.config
+        if getattr(self.config, "mm_projector_cfg", None) is None:
+            self.config.mm_projector_cfg = self.mm_projector.config
+        if getattr(self.config, "context_provider_cfg", None) is None and self.context_provider is not None:
+            self.config.context_provider_cfg = self.context_provider.config
+    def freezed_module_patch(self):
+        '''
+        Huggingface will call model.train() at each training_step. To ensure the expected behaviors for modules like dropout, batchnorm, etc., we need to call model.eval() for the freezed modules.
+        '''
+        if self.training:
+            if self.get_llm() and not getattr(self.config, "tune_language_model", False):
+                logging.warning("Caution: Your LLM is currently in training mode, ensuring accurate gradient computation. Please be vigilant, particularly regarding BatchNorm and Dropout operations.")
+            if self.get_vision_tower() and not getattr(self.config, "tune_vision_tower", False):
+                self.get_vision_tower().eval()
+            if self.get_mm_projector() and not getattr(self.config, "tune_mm_projector", False):
+                self.get_mm_projector().eval()
+            if self.get_context_provider() and not getattr(self.config, "tune_context_provider", False):
+                self.get_context_provider().eval()
+    def encode_images(self, images):
+        image_features = self.get_vision_tower()(images)
+        image_features = self.get_mm_projector()(image_features)
+        return image_features
+    def encode_images_with_context(self, images):
+        context_provider = self.get_context_provider()
+        # If the channels completely match, they are cimage (image with context).
+        cimage_mask = torch.any((images[:, :4, ...] != images[:, 4:, ...]).flatten(start_dim=1), dim=1)
+        if context_provider.treat_image_as_cimage:
+            # If the context provider treats the image as cimage, then all images are cimage.
+            cimage_mask[:] = True
+        if context_provider.context_image_as_queries:
+            # Swap the crop image and full image since the model uses the full image as queries by default
+            images = torch.cat((images[:, 4:, ...], images[:, :4, ...]), dim=1)
+        # Process the first 4 channels for all images: for image it's the image, for cimage it's the full image
+        vision_tower = self.get_vision_tower()
+        # Encode context images (full images)
+        image_features = vision_tower(images[:, :4, ...]).to(self.device)
+        # Each cimage has 8 channels (full and crop concatenated)
+        cimage_concatenated = images[cimage_mask]
+        cimage_full_features = image_features[cimage_mask]
+        if context_provider.context_provider_type == "cross_attn_end_to_all":
+            cimage_features = self.context_provider(
+                cimage_full_features=cimage_full_features,
+                cimage_concatenated=cimage_concatenated,
+                vision_tower=vision_tower
+            ).to(self.device)
+        elif context_provider.context_provider_type == "concat":
+            # Full features of cimages are computed but not used.
+            cimage_features = self.context_provider(
+                cimage_concatenated=cimage_concatenated,
+                vision_tower=vision_tower
+            ).to(self.device)
+        else:
+            raise NotImplementedError(f"Context provider type {context_provider.context_provider_type} not implemented.")
+        # Put cimage_features into image_features
+        image_features[cimage_mask] = cimage_features
+        # Project to the llm space
+        image_features = self.get_mm_projector()(image_features)
+        return image_features
+    ## @yunhao: is there a better way to handle function call and attributes for llm?
+    ## support beam search
+    def _temporary_reorder_cache(self, past_key_values, sorted_idx):
+        return self.get_llm()._temporary_reorder_cache(past_key_values, sorted_idx)
+    def get_input_embeddings(self):
+        return self.get_llm().get_input_embeddings()
+    def get_output_embeddings(self):
+        return self.get_llm().get_output_embeddings()
+    def resize_token_embeddings(self, embed_size):
+        self.get_llm().resize_token_embeddings(embed_size)
+class LlavaMetaForCausalLM(ABC):
+    """This class is originally implemented by the LLaVA team and
+    modified by Haotian Tang and Jason Lu based on Ji Lin's implementation
+    to support multiple images and input packing."""
+    ## TODO move the forward function here if there is no need to override it
+    def prepare_inputs_labels_for_multimodal(
+        self, input_ids, position_ids, attention_mask, past_key_values, labels, images
+    ):
+        vision_tower = self.get_vision_tower()
+        if vision_tower is None or images is None or input_ids.shape[1] == 1:
+            if (
+                past_key_values is not None
+                and vision_tower is not None
+                and images is not None
+                and input_ids.shape[1] == 1
+            ):
+                target_shape = past_key_values[-1][-1].shape[-2] + 1
+                attention_mask = torch.cat(
+                    (
+                        attention_mask,
+                        torch.ones(
+                            (
+                                attention_mask.shape[0],
+                                target_shape - attention_mask.shape[1],
+                            ),
+                            dtype=attention_mask.dtype,
+                            device=attention_mask.device,
+                        ),
+                    ),
+                    dim=1,
+                )
+                position_ids = torch.sum(attention_mask, dim=1).unsqueeze(-1) - 1
+            return (
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                None,
+                labels,
+            )
+        # handle different image dtypes for packing
+        if type(images) is list:
+            images = torch.cat(images, dim=0)
+        elif images.ndim == 5:  # batch_size x seq_len x image_channels
+            images = images.flatten(0, 1)
+        if getattr(self, "context_provider", None):
+            image_features = self.encode_images_with_context(images)
+        else:
+            # Since we slice it with index below, turning it into a list splits things by the first index which does not result in data copy or degrade performance.
+            # Example dimension: [1, 196, 2560]
+            assert images.shape[1] <= 4, f"images have more than 4 channels, but context provider is not included"
+            image_features = self.encode_images(images).to(self.device)
+        # Note (kentang-mit@): image start / end is not implemented here to support pretraining.
+        if getattr(self.config, "turn_mm_projector", False) and getattr(self.config, "mm_use_im_start_end", False):
+            raise NotImplementedError
+        # Let's just add dummy tensors if they do not exist,
+        # it is a headache to deal with None all the time.
+        # But it is not ideal, and if you have a better idea,
+        # please open an issue / submit a PR, thanks.
+        _labels = labels
+        _position_ids = position_ids
+        _attention_mask = attention_mask
+        if attention_mask is None:
+            attention_mask = torch.ones_like(input_ids, dtype=torch.bool)
+        else:
+            attention_mask = attention_mask.bool()
+        if position_ids is None:
+            position_ids = torch.arange(0, input_ids.shape[1], dtype=torch.long, device=input_ids.device)
+        if labels is None:
+            labels = torch.full_like(input_ids, IGNORE_INDEX)
+        # remove the padding using attention_mask
+        input_ids_copy = input_ids.clone()
+        # kentang-mit@: Otherwise tokenizer out of bounds. Embeddings of image tokens will not be used.
+        input_ids_copy[input_ids_copy == IMAGE_TOKEN_INDEX] = 0
+        input_embeds = self.llm.model.embed_tokens(input_ids_copy)
+        input_ids = [
+            cur_input_ids[cur_attention_mask] for cur_input_ids, cur_attention_mask in zip(input_ids, attention_mask)
+        ]
+        input_embeds_1 = [
+            cur_input_embeds[cur_attention_mask]
+            for cur_input_embeds, cur_attention_mask in zip(input_embeds, attention_mask)
+        ]
+        labels = [cur_labels[cur_attention_mask] for cur_labels, cur_attention_mask in zip(labels, attention_mask)]
+        new_input_embeds = []
+        new_labels = []
+        cur_image_idx = 0
+        # print("BEFORE BATCH LOOP:", len(input_ids), input_ids[0].shape, input_ids[0].device, [(x == IMAGE_TOKEN_INDEX).sum() for x in input_ids])
+        # kentang-mit@: If some part of the model is executed in the loop, the the loop length needs to be a constant.
+        for batch_idx, cur_input_ids in enumerate(input_ids):
+            cur_input_ids = input_ids[batch_idx]
+            num_images = (cur_input_ids == IMAGE_TOKEN_INDEX).sum()
+            if num_images == 0:
+                cur_image_features = image_features[0]
+                # cur_input_embeds_1 = self.get_llm().embed_tokens(cur_input_ids)
+                cur_input_embeds_1 = input_embeds_1[batch_idx]
+                cur_input_embeds = torch.cat([cur_input_embeds_1, cur_image_features[0:0]], dim=0)
+                new_input_embeds.append(cur_input_embeds)
+                new_labels.append(labels[batch_idx])
+                # kenang-mit@: we do not have placeholdr image for text-only data now.
+                # cur_image_idx += 1
+                continue
+            cur_input_embeds = input_embeds_1[batch_idx]
+            image_token_indices = (
+                [-1] + torch.where(cur_input_ids == IMAGE_TOKEN_INDEX)[0].tolist() + [cur_input_ids.shape[0]]
+            )
+            cur_input_ids_noim = []
+            cur_labels = labels[batch_idx]
+            cur_labels_noim = []
+            cur_input_embeds_no_im = []
+            for i in range(len(image_token_indices) - 1):
+                cur_input_ids_noim.append(cur_input_ids[image_token_indices[i] + 1 : image_token_indices[i + 1]])
+                cur_labels_noim.append(cur_labels[image_token_indices[i] + 1 : image_token_indices[i + 1]])
+                cur_input_embeds_no_im.append(cur_input_embeds[image_token_indices[i] + 1 : image_token_indices[i + 1]])
+            split_sizes = [x.shape[0] for x in cur_labels_noim]
+            # cur_input_embeds = self.get_llm().embed_tokens(torch.cat(cur_input_ids_noim))
+            # cur_input_embeds_no_im = torch.split(cur_input_embeds, split_sizes, dim=0)
+            cur_new_input_embeds = []
+            cur_new_labels = []
+            for i in range(num_images + 1):
+                cur_new_input_embeds.append(cur_input_embeds_no_im[i])
+                cur_new_labels.append(cur_labels_noim[i])
+                if i < num_images:
+                    cur_image_features = image_features[cur_image_idx]
+                    cur_image_idx += 1
+                    cur_new_input_embeds.append(cur_image_features)
+                    cur_new_labels.append(
+                        torch.full(
+                            (cur_image_features.shape[0],),
+                            IGNORE_INDEX,
+                            device=cur_labels.device,
+                            dtype=cur_labels.dtype,
+                        )
+                    )
+            cur_new_input_embeds = torch.cat(cur_new_input_embeds)
+            cur_new_labels = torch.cat(cur_new_labels)
+            new_input_embeds.append(cur_new_input_embeds)
+            new_labels.append(cur_new_labels)
+        # Truncate sequences to max length as image embeddings can make the sequence longer
+        tokenizer_model_max_length = getattr(self.llm.config, "tokenizer_model_max_length", None)
+        if tokenizer_model_max_length is not None:
+            if any(len(x) > tokenizer_model_max_length for x in new_input_embeds):
+                warnings.warn("Inputs truncated!")
+            new_input_embeds = [x[:tokenizer_model_max_length] for x in new_input_embeds]
+            new_labels = [x[:tokenizer_model_max_length] for x in new_labels]
+        # Combine them
+        max_len = max(x.shape[0] for x in new_input_embeds)
+        batch_size = len(new_input_embeds)
+        new_input_embeds_padded = []
+        new_labels_padded = torch.full(
+            (batch_size, max_len),
+            IGNORE_INDEX,
+            dtype=new_labels[0].dtype,
+            device=new_labels[0].device,
+        )
+        attention_mask = torch.zeros(
+            (batch_size, max_len),
+            dtype=attention_mask.dtype,
+            device=attention_mask.device,
+        )
+        position_ids = torch.zeros((batch_size, max_len), dtype=position_ids.dtype, device=position_ids.device)
+        for i, (cur_new_embed, cur_new_labels) in enumerate(zip(new_input_embeds, new_labels)):
+            cur_len = cur_new_embed.shape[0]
+            if getattr(self.llm.config, "tokenizer_padding_side", "right") == "left":
+                new_input_embeds_padded.append(
+                    torch.cat(
+                        (
+                            torch.zeros(
+                                (max_len - cur_len, cur_new_embed.shape[1]),
+                                dtype=cur_new_embed.dtype,
+                                device=cur_new_embed.device,
+                            ),
+                            cur_new_embed,
+                        ),
+                        dim=0,
+                    )
+                )
+                if cur_len > 0:
+                    new_labels_padded[i, -cur_len:] = cur_new_labels
+                    attention_mask[i, -cur_len:] = True
+                    position_ids[i, -cur_len:] = torch.arange(
+                        0, cur_len, dtype=position_ids.dtype, device=position_ids.device
+                    )
+            else:
+                new_input_embeds_padded.append(
+                    torch.cat(
+                        (
+                            cur_new_embed,
+                            torch.zeros(
+                                (max_len - cur_len, cur_new_embed.shape[1]),
+                                dtype=cur_new_embed.dtype,
+                                device=cur_new_embed.device,
+                            ),
+                        ),
+                        dim=0,
+                    )
+                )
+                if cur_len > 0:
+                    new_labels_padded[i, :cur_len] = cur_new_labels
+                    attention_mask[i, :cur_len] = True
+                    position_ids[i, :cur_len] = torch.arange(
+                        0, cur_len, dtype=position_ids.dtype, device=position_ids.device
+                    )
+        new_input_embeds = torch.stack(new_input_embeds_padded, dim=0)
+        if _labels is None:
+            new_labels = None
+        else:
+            new_labels = new_labels_padded
+        if _attention_mask is None:
+            attention_mask = None
+        else:
+            attention_mask = attention_mask.to(dtype=_attention_mask.dtype)
+        if _position_ids is None:
+            position_ids = None
+        return (
+            None,
+            position_ids,
+            attention_mask,
+            past_key_values,
+            new_input_embeds,
+            new_labels,
+        )
+    def repack_multimodal_data(
+        self,
+        input_ids,
+        position_ids,
+        attention_mask,
+        past_key_values,
+        inputs_embeds,
+        labels,
+    ):
+        # kentang-mit@: reorder and repack (reduce computation overhead)
+        # requires transformers replacement.
+        new_inputs_embeds = []
+        new_position_ids = []
+        new_labels = []
+        seqlens_in_batch = attention_mask.sum(dim=-1, dtype=torch.int32)
+        sorted_seqlens_in_batch, sorted_idx = torch.sort(seqlens_in_batch, descending=True)
+        # print(sorted_seqlens_in_batch)
+        max_seqlen = inputs_embeds.shape[1]
+        cur_inputs_embeds = []
+        cur_position_ids = []
+        cur_labels = []
+        cur_batch_len = 0
+        # print(sorted_seqlens_in_batch.device, len(sorted_seqlens_in_batch), max_seqlen)
+        for i in range(len(sorted_seqlens_in_batch)):
+            cur_seqlen = sorted_seqlens_in_batch[i].item()
+            if cur_seqlen + cur_batch_len <= max_seqlen:
+                cur_batch_len += cur_seqlen
+                # each item: num_tokens x num_channels
+                # remove padding on-the-fly
+                cur_inputs_embeds.append(inputs_embeds[sorted_idx[i]][attention_mask[sorted_idx[i]]])
+                # each item: num_tokens
+                cur_position_ids.append(
+                    torch.arange(
+                        cur_inputs_embeds[-1].shape[0],
+                        device=cur_inputs_embeds[-1].device,
+                    )
+                )
+                # each item: num_tokens
+                # remove padding on-the-fly
+                cur_labels.append(labels[sorted_idx[i]][attention_mask[sorted_idx[i]]])
+            else:
+                new_inputs_embeds.append(torch.cat(cur_inputs_embeds, 0))
+                new_position_ids.append(torch.cat(cur_position_ids, 0))
+                new_labels.append(torch.cat(cur_labels, 0))
+                # The current batch is too long. We will start a new batch.
+                cur_batch_len = cur_seqlen
+                cur_inputs_embeds = [inputs_embeds[sorted_idx[i]][attention_mask[sorted_idx[i]]]]
+                cur_position_ids = [
+                    torch.arange(
+                        cur_inputs_embeds[-1].shape[0],
+                        device=cur_inputs_embeds[-1].device,
+                    )
+                ]
+                cur_labels = [labels[sorted_idx[i]][attention_mask[sorted_idx[i]]]]
+        if len(cur_inputs_embeds):
+            new_inputs_embeds.append(torch.cat(cur_inputs_embeds, 0))
+            new_position_ids.append(torch.cat(cur_position_ids, 0))
+            new_labels.append(torch.cat(cur_labels, 0))
+        # print(new_position_ids[0].device, [x.shape for x in new_inputs_embeds], [x.shape for x in new_labels], [x.shape for x in new_position_ids])
+        # assert 0
+        new_inputs_embeds = torch.nn.utils.rnn.pad_sequence(
+            new_inputs_embeds, batch_first=True, padding_value=self.llm.pad_token_id
+        )
+        new_position_ids = torch.nn.utils.rnn.pad_sequence(new_position_ids, batch_first=True, padding_value=-1)
+        new_labels = torch.nn.utils.rnn.pad_sequence(new_labels, batch_first=True, padding_value=IGNORE_INDEX)
+        ## yunhao: it's currently a workaround to avoid errors for seq_len < 100
+        new_attention_mask = new_position_ids.ne(-1)
+        # sanity check
+        assert new_attention_mask.sum() == attention_mask.sum()
+        # print(new_inputs_embeds.shape, (new_attention_mask.sum(1)))
+        # print(sorted_seqlens_in_batch.device, sorted_seqlens_in_batch, new_attention_mask.sum(1))
+        # return None, position_ids, attention_mask, past_key_values, new_input_embeds, new_labels
+        return (
+            None,
+            new_position_ids,
+            new_attention_mask,
+            past_key_values,
+            new_inputs_embeds,
+            new_labels,
+            sorted_seqlens_in_batch,
+        )
+    def initialize_vision_tokenizer(self, model_args, tokenizer):
+        if model_args.mm_use_im_patch_token:
+            tokenizer.add_tokens([DEFAULT_IMAGE_PATCH_TOKEN], special_tokens=True)
+            self.resize_token_embeddings(len(tokenizer))
+        if model_args.mm_use_im_start_end:
+            num_new_tokens = tokenizer.add_tokens([DEFAULT_IM_START_TOKEN, DEFAULT_IM_END_TOKEN], special_tokens=True)
+            self.resize_token_embeddings(len(tokenizer))
+            if num_new_tokens > 0:
+                input_embeddings = self.get_input_embeddings().weight.data
+                output_embeddings = self.get_output_embeddings().weight.data
+                input_embeddings_avg = input_embeddings[:-num_new_tokens].mean(dim=0, keepdim=True)
+                output_embeddings_avg = output_embeddings[:-num_new_tokens].mean(dim=0, keepdim=True)
+                input_embeddings[-num_new_tokens:] = input_embeddings_avg
+                output_embeddings[-num_new_tokens:] = output_embeddings_avg
+            ## TODO yunhao: handle cases for <im_st> <im_end>
+            if model_args.pretrain_mm_mlp_adapter:
+                mm_projector_weights = torch.load(model_args.pretrain_mm_mlp_adapter, map_location="cpu")
+                embed_tokens_weight = mm_projector_weights["model.embed_tokens.weight"]
+                assert num_new_tokens == 2
+                if input_embeddings.shape == embed_tokens_weight.shape:
+                    input_embeddings[-num_new_tokens:] = embed_tokens_weight[-num_new_tokens:]
+                elif embed_tokens_weight.shape[0] == num_new_tokens:
+                    input_embeddings[-num_new_tokens:] = embed_tokens_weight
+                else:
+                    raise ValueError(
+                        f"Unexpected embed_tokens_weight shape. Pretrained: {embed_tokens_weight.shape}. Current: {input_embeddings.shape}. Numer of new tokens: {num_new_tokens}."
+                    )
+        elif model_args.mm_use_im_patch_token:
+            if model_args.mm_projector:
+                for p in self.get_input_embeddings().parameters():
+                    p.requires_grad = False
+                for p in self.get_output_embeddings().parameters():
+                    p.requires_grad = False

dam/model/mm_utils.py ADDED Viewed

	@@ -0,0 +1,312 @@

+# Copyright 2024 NVIDIA CORPORATION & AFFILIATES
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+# SPDX-License-Identifier: Apache-2.0
+from PIL import Image
+from io import BytesIO
+import base64
+import numpy as np
+import os
+import torch
+from transformers import StoppingCriteria
+from .constants import IMAGE_TOKEN_INDEX
+import tempfile
+from io import BytesIO
+def get_frame_from_vcap(vidcap, num_frames=10, fps=None, frame_count=None):
+    import cv2
+    if fps == None or frame_count == None:
+        # if one of fps or frame_count is None, still recompute
+        fps = vidcap.get(cv2.CAP_PROP_FPS)
+        frame_count = int(vidcap.get(cv2.CAP_PROP_FRAME_COUNT))
+    if fps == 0 or frame_count == 0:
+        print("Video file not found. return empty images.")
+        return [
+            Image.new("RGB", (720, 720)),
+        ] * num_frames
+    duration = frame_count / fps
+    frame_interval = frame_count // num_frames
+    if frame_interval == 0 and frame_count <= 1:
+        print("frame_interval is equal to 0. return empty image.")
+        return [
+            Image.new("RGB", (720, 720)),
+        ] * num_frames
+    # print("duration:", duration, "frames:", frame_count, "intervals:", frame_interval)
+    images = []
+    count = 0
+    success = True
+    frame_indices = np.linspace(0, frame_count - 2, num_frames, dtype=int)
+    while success:
+        # print("frame_count:", frame_count, "count:", count, "num_frames:", num_frames, "frame_interval:", frame_interval)
+        if frame_count >= num_frames:
+            success, frame = vidcap.read()
+            if count in frame_indices:
+                img = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
+                im_pil = Image.fromarray(img)
+                images.append(im_pil)
+                if len(images) >= num_frames:
+                    return images
+            count += 1
+        else:
+            # Left padding frames if the video is not long enough
+            success, frame = vidcap.read()
+            if success:
+                img = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
+                im_pil = Image.fromarray(img)
+                images.append(im_pil)
+                count += 1
+            elif count >= 1:
+                width, height = images[-1].size
+                images = [Image.new("RGB", (width, height))] * \
+                    (num_frames - len(images)) + images
+                print("padding frames:", (num_frames - len(images)))
+                return images
+            else:
+                break
+    raise ValueError(
+        "Did not find enough frames in the video. return empty image.")
+def opencv_extract_frames(vpath_or_bytesio, frames=6, fps=None, frame_count=None):
+    """
+    Extract frames from a video using OpenCV.
+    Args:
+        vpath_or_bytesio (str or BytesIO): Path to the video file or BytesIO object containing the video.
+        frames (int): Number of frames to extract from the video.
+    Returns:
+        list: List of PIL Images extracted from the video.
+    Raises:
+        NotImplementedError: If the type of `vpath_or_bytesio` is not supported.
+    """
+    import cv2
+    if isinstance(vpath_or_bytesio, str):
+        vidcap = cv2.VideoCapture(vpath_or_bytesio)
+        return get_frame_from_vcap(vidcap, frames, fps=fps, frame_count=frame_count)
+    elif isinstance(vpath_or_bytesio, (BytesIO,)):
+        # assuming mp4
+        with tempfile.NamedTemporaryFile(delete=True, suffix=".mp4") as temp_video:
+            temp_video.write(vpath_or_bytesio.read())
+            temp_video_name = temp_video.name
+            vidcap = cv2.VideoCapture(temp_video_name)
+            return get_frame_from_vcap(vidcap, frames, fps=fps, frame_count=frame_count)
+    else:
+        raise NotImplementedError(type(vpath_or_bytesio))
+def load_image_from_base64(image):
+    return Image.open(BytesIO(base64.b64decode(image)))
+def expand2square(pil_img, background_color):
+    """
+    Expand the given PIL image to a square shape by adding padding.
+    Parameters:
+    - pil_img: The PIL image to be expanded.
+    - background_color: The color of the padding to be added.
+    Returns:
+    - The expanded PIL image.
+    If the image is already square, it is returned as is.
+    If the image is wider than it is tall, padding is added to the top and bottom.
+    If the image is taller than it is wide, padding is added to the left and right.
+    """
+    width, height = pil_img.size
+    if pil_img.mode == 'L':
+        background_color = background_color[0]
+    if width == height:
+        return pil_img
+    elif width > height:
+        result = Image.new(pil_img.mode, (width, width), background_color)
+        result.paste(pil_img, (0, (width - height) // 2))
+        return result
+    else:
+        result = Image.new(pil_img.mode, (height, height), background_color)
+        result.paste(pil_img, ((height - width) // 2, 0))
+        return result
+def process_image(image_file, data_args, image_folder, pil_preprocess_fn=None):
+    processor = data_args.image_processor
+    if isinstance(image_file, str):
+        if image_folder is not None:
+            image = Image.open(os.path.join(
+                image_folder, image_file)).convert("RGB")
+        else:
+            image = Image.open(image_file).convert("RGB")
+    else:
+        # image is stored in bytearray
+        image = image_file.convert("RGB")
+    info = None
+    if pil_preprocess_fn is not None:
+        image = pil_preprocess_fn(image)
+        if isinstance(image, tuple):
+            image, info = image
+    if data_args.image_aspect_ratio == "resize":
+        if hasattr(data_args.image_processor, "crop_size"):
+            # CLIP vision tower
+            crop_size = data_args.image_processor.crop_size
+        else:
+            # SIGLIP vision tower
+            assert hasattr(data_args.image_processor, "size")
+            crop_size = data_args.image_processor.size
+        image = image.resize((crop_size["height"], crop_size["width"]))
+    if data_args.image_aspect_ratio == "pad":
+        def expand2square(pil_img, background_color):
+            width, height = pil_img.size
+            if width == height:
+                return pil_img
+            elif width > height:
+                result = Image.new(
+                    pil_img.mode, (width, width), background_color)
+                result.paste(pil_img, (0, (width - height) // 2))
+                return result
+            else:
+                result = Image.new(
+                    pil_img.mode, (height, height), background_color)
+                result.paste(pil_img, ((height - width) // 2, 0))
+                return result
+        image = expand2square(image, tuple(int(x * 255)
+                              for x in processor.image_mean))
+        image = processor.preprocess(image, return_tensors="pt")[
+            "pixel_values"][0]
+    else:
+        # Using default behavior of the vision encoder
+        # For CLIP, default is central crop
+        # For Radio, default is central crop
+        # For Siglip, default is resize
+        # For InternVIT, default is resize
+        image = processor.preprocess(image, return_tensors="pt")[
+            "pixel_values"][0]
+    if info is not None:
+        return image, info
+    return image
+def process_images(images, image_processor, model_cfg):
+    model_cfg.image_processor = image_processor
+    new_images = [process_image(image, model_cfg, None) for image in images]
+    if all(x.shape == new_images[0].shape for x in new_images):
+        new_images = torch.stack(new_images, dim=0)
+    return new_images
+# Note that newer VILA codebase adds an lstrip option that defaults to False, and the functionality is the same by default
+def tokenizer_image_token(
+    prompt, tokenizer, image_token_index=IMAGE_TOKEN_INDEX, return_tensors=None
+):
+    prompt_chunks = [
+        tokenizer(chunk).input_ids for chunk in prompt.split("<image>")]
+    def insert_separator(X, sep):
+        return [ele for sublist in zip(X, [sep] * len(X)) for ele in sublist][:-1]
+    input_ids = []
+    offset = 0
+    if (
+        len(prompt_chunks) > 0
+        and len(prompt_chunks[0]) > 0
+        and prompt_chunks[0][0] == tokenizer.bos_token_id
+    ):
+        offset = 1
+        input_ids.append(prompt_chunks[0][0])
+    for x in insert_separator(prompt_chunks, [image_token_index] * (offset + 1)):
+        input_ids.extend(x[offset:])
+    if return_tensors is not None:
+        if return_tensors == "pt":
+            return torch.tensor(input_ids, dtype=torch.long)
+        raise ValueError(f"Unsupported tensor type: {return_tensors}")
+    return input_ids
+def is_gemma_tokenizer(tokenizer):
+    return "gemma" in tokenizer.__class__.__name__.lower()
+def get_model_name_from_path(model_path):
+    model_path = model_path.strip("/")
+    model_paths = model_path.split("/")
+    if model_paths[-1].startswith("checkpoint-"):
+        return model_paths[-2] + "_" + model_paths[-1]
+    else:
+        return model_paths[-1]
+class KeywordsStoppingCriteria(StoppingCriteria):
+    def __init__(self, keywords, tokenizer, input_ids):
+        self.keywords = keywords
+        self.keyword_ids = []
+        self.max_keyword_len = 0
+        for keyword in keywords:
+            cur_keyword_ids = tokenizer(keyword).input_ids
+            if (
+                len(cur_keyword_ids) > 1
+                and cur_keyword_ids[0] == tokenizer.bos_token_id
+            ):
+                cur_keyword_ids = cur_keyword_ids[1:]
+            if len(cur_keyword_ids) > self.max_keyword_len:
+                self.max_keyword_len = len(cur_keyword_ids)
+            self.keyword_ids.append(torch.tensor(cur_keyword_ids))
+        self.tokenizer = tokenizer
+        self.start_len = input_ids.shape[1]
+    def call_for_batch(
+        self, output_ids: torch.LongTensor, scores: torch.FloatTensor, **kwargs
+    ) -> bool:
+        offset = min(output_ids.shape[1] -
+                     self.start_len, self.max_keyword_len)
+        self.keyword_ids = [
+            keyword_id.to(output_ids.device) for keyword_id in self.keyword_ids
+        ]
+        for keyword_id in self.keyword_ids:
+            if (output_ids[0, -keyword_id.shape[0]:] == keyword_id).all():
+                return True
+        outputs = self.tokenizer.batch_decode(
+            output_ids[:, -offset:], skip_special_tokens=True
+        )[0]
+        for keyword in self.keywords:
+            if keyword in outputs:
+                return True
+        return False
+    def __call__(
+        self, output_ids: torch.LongTensor, scores: torch.FloatTensor, **kwargs
+    ) -> bool:
+        outputs = []
+        for i in range(output_ids.shape[0]):
+            outputs.append(self.call_for_batch(
+                output_ids[i].unsqueeze(0), scores))
+        return all(outputs)

dam/model/model_utils.py ADDED Viewed

	@@ -0,0 +1,268 @@

+from transformers import AutoModelForCausalLM, AutoTokenizer
+import torch
+import torch.nn as nn
+import os
+import warnings
+from typing import Optional, Union, List, Tuple
+from transformers import (
+    AutoTokenizer,
+    AutoModel,
+    AutoModelForCausalLM,
+    AutoConfig,
+    BitsAndBytesConfig,
+    PretrainedConfig,
+    PreTrainedModel,
+    LlamaConfig,
+    LlamaModel,
+)
+from transformers.modeling_outputs import CausalLMOutputWithPast
+from transformers import PretrainedConfig
+from .llava_arch import LlavaMetaModel, LlavaMetaForCausalLM
+from .language_model.llava_llama import LlavaLlamaConfig
+# TODO: we may move LlavaConfig to configuration_llava.py
+# from model.configuration_llava import LlavaConfig
+class LlavaLlamaModel(LlavaMetaModel, LlavaMetaForCausalLM, PreTrainedModel):
+    config_class = LlavaLlamaConfig
+    main_input_name = "input_embeds"
+    supports_gradient_checkpointing = True
+    def __init__(self, config: LlavaLlamaConfig = None, *args, **kwargs) -> None:
+        super().__init__(config)
+        self.init_vlm(config=config, *args, **kwargs)
+    @classmethod
+    def from_pretrained(
+        cls,
+        pretrained_model_name_or_path: Optional[Union[str, os.PathLike]],
+        *model_args,
+        config: Optional[Union[PretrainedConfig, str, os.PathLike]] = None,
+        cache_dir: Optional[Union[str, os.PathLike]] = None,
+        ignore_mismatched_sizes: bool = False,
+        force_download: bool = False,
+        local_files_only: bool = False,
+        token: Optional[Union[str, bool]] = None,
+        revision: str = "main",
+        use_safetensors: bool = None,
+        **kwargs,
+    ):
+        if hasattr(cls, "load_pretrained"):
+            return cls.load_pretrained(pretrained_model_name_or_path,
+                                       *model_args, config=config, cache_dir=cache_dir, ignore_mismatched_sizes=ignore_mismatched_sizes, force_download=force_download, local_files_only=local_files_only, token=token,
+                                       revision=revision, use_safetensors=use_safetensors, **kwargs
+                                       )
+        return super(LlavaLlamaModel).from_pretrained(pretrained_model_name_or_path,
+                                                      *model_args, config=config, cache_dir=cache_dir, ignore_mismatched_sizes=ignore_mismatched_sizes, force_download=force_download, local_files_only=local_files_only, token=token,
+                                                      revision=revision, use_safetensors=use_safetensors, **kwargs)
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        images: Optional[torch.FloatTensor] = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_values: Optional[List[torch.FloatTensor]] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        labels: Optional[torch.LongTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        return_dict: Optional[bool] = None,
+    ) -> Union[Tuple, CausalLMOutputWithPast]:
+        self.freezed_module_patch()
+        if inputs_embeds is None:
+            (
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                inputs_embeds,
+                labels,
+            ) = self.prepare_inputs_labels_for_multimodal(
+                input_ids, position_ids, attention_mask, past_key_values, labels, images
+            )
+        # Note (kentang-mit@): we have a unit test for this function.
+        if self.training:
+            (
+                _,
+                new_position_ids,
+                new_attention_mask,
+                _,
+                new_inputs_embeds,
+                new_labels,
+                sorted_seqlens_in_batch,
+            ) = self.repack_multimodal_data(
+                input_ids,
+                position_ids,
+                attention_mask,
+                past_key_values,
+                inputs_embeds,
+                labels,
+            )
+            new_input_ids = None
+            past_key_values = None
+        else:
+            new_attention_mask = attention_mask
+            new_position_ids = position_ids
+            new_inputs_embeds = inputs_embeds
+            new_labels = labels
+            sorted_seqlens_in_batch = attention_mask.sum(-1).int()
+            new_input_ids = input_ids
+        outputs = self.llm.forward(
+            input_ids=new_input_ids,
+            attention_mask=new_attention_mask,
+            position_ids=new_position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=new_inputs_embeds,
+            labels=new_labels,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            return_dict=return_dict,
+            seqlens_in_batch=sorted_seqlens_in_batch,
+        )
+        return outputs
+    @torch.no_grad()
+    def generate(
+        self,
+        input_ids: Optional[torch.FloatTensor] = None,
+        images: Optional[torch.FloatTensor] = None,
+        attention_mask: Optional[torch.LongTensor] = None,
+        **generation_kwargs,
+    ):
+        if images is not None:
+            (
+                _,
+                _,
+                attention_mask,
+                _,
+                inputs_embeds,
+                _,
+            ) = self.prepare_inputs_labels_for_multimodal(
+                input_ids, None, attention_mask, None, None, images
+            )
+        else:
+            inputs_embeds = self.get_input_embeddings()(input_ids)
+        inputs_embeds = inputs_embeds.to(self.dtype)
+        outputs = self.llm.generate(
+            inputs_embeds=inputs_embeds,
+            attention_mask=attention_mask,
+            **generation_kwargs
+        )
+        return outputs
+def disable_torch_init():
+    """
+    Disable the redundant torch default initialization to accelerate model creation.
+    """
+    import torch
+    setattr(torch.nn.Linear, "reset_parameters", lambda self: None)
+    setattr(torch.nn.LayerNorm, "reset_parameters", lambda self: None)
+def load_pretrained_model(
+    model_path,
+    model_name,
+    model_base=None,
+    load_8bit=False,
+    load_4bit=False,
+    device_map="auto",
+    device="cuda",
+    **kwargs,
+):
+    kwargs = {"device_map": device_map, **kwargs}
+    if device != "cuda":
+        kwargs["device_map"] = {"": device}
+    if load_8bit:
+        kwargs["load_in_8bit"] = True
+    elif load_4bit:
+        kwargs["load_in_4bit"] = True
+        kwargs["quantization_config"] = BitsAndBytesConfig(
+            load_in_4bit=True,
+            bnb_4bit_compute_dtype=torch.float16,
+            bnb_4bit_use_double_quant=True,
+            bnb_4bit_quant_type="nf4",
+        )
+    else:
+        kwargs["torch_dtype"] = torch.float16
+    config = AutoConfig.from_pretrained(model_path)
+    config.resume_path = model_path
+    prepare_config_for_eval(config, kwargs)
+    model = LlavaLlamaModel(
+        config=config,
+        low_cpu_mem_usage=True,
+        **kwargs
+    )
+    tokenizer = model.tokenizer
+    model.eval()
+    # mm_use_im_start_end = getattr(
+    #     model.config, "mm_use_im_start_end", False)
+    # mm_use_im_patch_token = getattr(
+    #     model.config, "mm_use_im_patch_token", True)
+    # if mm_use_im_patch_token:
+    #     tokenizer.add_tokens(
+    #         [DEFAULT_IMAGE_PATCH_TOKEN], special_tokens=True)
+    # if mm_use_im_start_end:
+    #     tokenizer.add_tokens(
+    #         [DEFAULT_IM_START_TOKEN, DEFAULT_IM_END_TOKEN], special_tokens=True
+    #     )
+    model.resize_token_embeddings(len(tokenizer))
+    vision_tower = model.get_vision_tower()
+    vision_tower.to(device=device, dtype=torch.float16)
+    mm_projector = model.get_mm_projector()
+    mm_projector.to(device=device, dtype=torch.float16)
+    context_provider = model.get_context_provider()
+    if context_provider is not None:
+        context_provider.to(device=device, dtype=torch.float16)
+    image_processor = vision_tower.image_processor
+    if hasattr(model.llm.config, "max_sequence_length"):
+        context_len = model.config.max_sequence_length
+    else:
+        context_len = 2048
+    return tokenizer, model, image_processor, context_len
+def parse_model_name_or_path(config: PretrainedConfig, model_name="llm", suffix="_cfg"):
+    target_model = f"{model_name}{suffix}"
+    target_cfg = getattr(config, target_model, None)
+    if isinstance(target_cfg, str):
+        return target_cfg
+    elif isinstance(target_cfg, dict):
+        return target_cfg["architectures"][0]
+    else:
+        raise ValueError(f"Invalid {target_model} configuration!")
+def prepare_config_for_eval(config: PretrainedConfig, kwargs: dict):
+    try:
+        # compatible with deprecated config convention
+        if getattr(config, "vision_tower_cfg", None) is None:
+            config.vision_tower_cfg = config.mm_vision_tower
+    except AttributeError:
+        raise ValueError(
+            f"Invalid configuration! Cannot find vision_tower in config:\n{config}")
+    config.model_dtype = kwargs.pop("torch_dtype").__str__()
+    # siglip does not support device_map = "auto"
+    vision_tower_name = parse_model_name_or_path(config, "vision_tower")
+    if "siglip" in vision_tower_name.lower():
+        kwargs["device_map"] = "cuda"
+AutoConfig.register("llava_llama", LlavaLlamaConfig)
+AutoModel.register(LlavaLlamaConfig, LlavaLlamaModel)

dam/model/multimodal_encoder/__pycache__/builder.cpython-310.pyc ADDED Viewed

Binary file (1.55 kB). View file

dam/model/multimodal_encoder/__pycache__/context_provider.cpython-310.pyc ADDED Viewed

Binary file (9.85 kB). View file

dam/model/multimodal_encoder/__pycache__/siglip_encoder.cpython-310.pyc ADDED Viewed

Binary file (1.11 kB). View file

dam/model/multimodal_encoder/__pycache__/vision_encoder.cpython-310.pyc ADDED Viewed

Binary file (4.79 kB). View file

dam/model/multimodal_encoder/builder.py ADDED Viewed

	@@ -0,0 +1,54 @@

+# This file is modified from https://github.com/haotian-liu/LLaVA/
+import torch
+import os
+from transformers import AutoConfig, PretrainedConfig, PreTrainedModel
+from .siglip_encoder import SiglipVisionTower
+from .context_provider import ContextProvider, ContextProviderConfig
+def build_vision_tower(
+    model_name_or_path: str, config: PretrainedConfig
+) -> PreTrainedModel:
+    ## skip vision tower instantiation
+    if model_name_or_path is None:
+        return None
+    vision_tower_arch = None
+    if config.resume_path and "radio" not in model_name_or_path:
+        assert os.path.exists(
+            model_name_or_path
+        ), f"Resume vision tower path {model_name_or_path} does not exist!"
+        vision_tower_cfg = AutoConfig.from_pretrained(model_name_or_path, trust_remote_code=True)
+        vision_tower_arch = vision_tower_cfg.architectures[0].lower()
+    vision_tower_name = (
+        vision_tower_arch if vision_tower_arch is not None else model_name_or_path
+    )
+    if "siglip" in vision_tower_name:
+        vision_tower = SiglipVisionTower(model_name_or_path, config)
+    else:
+        raise ValueError(f"Unknown vision tower: {model_name_or_path}")
+    config.mm_hidden_size = vision_tower.config.hidden_size
+    return vision_tower
+def build_context_provider(
+    model_type_or_path: str, config: PretrainedConfig
+) -> PreTrainedModel:
+    if model_type_or_path is None:
+        return None
+    ## load from pretrained model
+    if config.resume_path:
+        assert os.path.exists(
+            model_type_or_path
+        ), f"Resume context provider path {model_type_or_path} does not exist!"
+        return ContextProvider.from_pretrained(
+            model_type_or_path, config, torch_dtype=eval(config.model_dtype)
+        )
+    ## build from scratch
+    else:
+        mm_projector_cfg = ContextProviderConfig(model_type_or_path)
+        mm_projector = ContextProvider(mm_projector_cfg, config).to(
+            eval(config.model_dtype)
+        )
+        return mm_projector

dam/model/multimodal_encoder/clip_encoder_ignored.py ADDED Viewed

	@@ -0,0 +1,19 @@

+# This file is modified from https://github.com/haotian-liu/LLaVA/
+import torch
+from llava.model.multimodal_encoder.vision_encoder import VisionTower
+from transformers import (
+    PretrainedConfig,
+    CLIPVisionModel,
+    CLIPImageProcessor,
+)
+class CLIPVisionTower(VisionTower):
+    def __init__(self, model_name_or_path: str, config: PretrainedConfig):
+        super().__init__(model_name_or_path, config)
+        self.image_processor = CLIPImageProcessor.from_pretrained(model_name_or_path)
+        self.vision_tower = CLIPVisionModel.from_pretrained(
+            model_name_or_path, torch_dtype=eval(config.model_dtype)
+        )
+        self.is_loaded = True