Skip to content

Instantly share code, notes, and snippets.

@ehartford
Last active August 24, 2023 17:36
Show Gist options
  • Save ehartford/782f4c4b01d481982a4a6b5f55ac3b4f to your computer and use it in GitHub Desktop.
Save ehartford/782f4c4b01d481982a4a6b5f55ac3b4f to your computer and use it in GitHub Desktop.
from transformers import AutoModelForCausalLM, AutoTokenizer
from peft import PeftModel
import torch
import os
import argparse
def get_args():
parser = argparse.ArgumentParser()
parser.add_argument("--base_model_name_or_path", type=str)
parser.add_argument("--peft_model_path", type=str)
parser.add_argument("--output_dir", type=str)
parser.add_argument("--device", type=str, default="auto")
parser.add_argument("--push_to_hub", action="store_true")
return parser.parse_args()
def main():
args = get_args()
if args.device == "auto":
device_arg = {"device_map": "auto"}
else:
device_arg = {"device_map": {"": args.device}}
print(f"Loading base model: {args.base_model_name_or_path}")
base_model = AutoModelForCausalLM.from_pretrained(
args.base_model_name_or_path,
return_dict=True,
torch_dtype=torch.float16,
load_in_4bit=True,
**device_arg,
)
print(f"Loading PEFT: {args.peft_model_path}")
model = PeftModel.from_pretrained(base_model, args.peft_model_path, **device_arg)
print(f"Running merge_and_unload")
model = model.merge_and_unload() # throws ValueError("Cannot merge LORA layers when the model is loaded in 8-bit mode")
model = model.to(torch.float16)
tokenizer = AutoTokenizer.from_pretrained(args.base_model_name_or_path)
if args.push_to_hub:
print(f"Saving to hub ...")
model.push_to_hub(f"{args.output_dir}", use_temp_dir=False)
tokenizer.push_to_hub(f"{args.output_dir}", use_temp_dir=False)
else:
model.save_pretrained(f"{args.output_dir}")
tokenizer.save_pretrained(f"{args.output_dir}")
print(f"Model saved to {args.output_dir}")
if __name__ == "__main__":
main()
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment