from huggingface_hub import HfFolder from sagemaker.huggingface import HuggingFaceModel # sagemaker config instance_type = "ml.inf2.48xlarge" health_check_timeout=2400 # additional time to load the model volume_size=512 # size in GB of the EBS volume # Define Model and Endpoint configuration parameter config = { "HF_MODEL_ID": "meta-llama/Meta-Llama-3-70B-Instruct", "HF_NUM_CORES": "24", # number of neuron cores "HF_BATCH_SIZE": "4", # batch size used to compile the model "HF_SEQUENCE_LENGTH": "4096", # length used to compile the model # "HF_AUTO_CAST_TYPE": "bf16", # dtype of the model "HF_AUTO_CAST_TYPE": "fp16", # dtype of the model "MAX_BATCH_SIZE": "4", # max batch size for the model "MAX_INPUT_LENGTH": "4000", # max length of input text "MAX_TOTAL_TOKENS": "4096", # max length of generated text "MESSAGES_API_ENABLED": "true", # Enable the messages API "HF_TOKEN": HfFolder.get_token(), # pass the huggingface token } # create HuggingFaceModel with the image uri llm_model = HuggingFaceModel( role=role, image_uri=llm_image, env=config )