# Copy to .env (gitignored) and edit. sm.py reads it, so the MCP server, CLI and Makefile all see it. AWS_REGION=us-east-2 MODEL_ID=google/gemma-4-E2B-it INSTANCE_TYPE=ml.g6.xlarge ENDPOINT_NAME=gemma-4-e2b ROLE_NAME=sagemaker-gemma-execution-role MAX_MODEL_LEN=8192 # Leave empty to use the newest SageMaker vLLM image in the region. IMAGE_URI= # Fallback instance types, highest priority first (same GPU keeps runs comparable). INSTANCE_POOLS=ml.g6.xlarge,ml.g6.2xlarge,ml.g6.4xlarge