import os import json # quantize the model quantized_model = quantizer.quantize_model(model, tokenizer) # save the quantize model to disk save_folder = "quantized_llama" quantized_model.save_pretrained(save_folder, safe_serialization=True) # load fresh, fast tokenizer and save it to disk tokenizer = AutoTokenizer.from_pretrained(model_id).save_pretrained(save_folder) # save quantize_config.json for TGI with open(os.path.join(save_folder, "quantize_config.json"), "w", encoding="utf-8") as f: quantizer.disable_exllama = False json.dump(quantizer.to_dict(), f, indent=2)