[](https://hamel.dev/notes/llm/inference/inference.html#cb2-1)from mlc_chat import ChatModule, ChatConfig [](https://hamel.dev/notes/llm/inference/inference.html#cb2-2)cfg = ChatConfig(max_gen_len=200) [](https://hamel.dev/notes/llm/inference/inference.html#cb2-3)cm = ChatModule(model="Llama-2-7b-chat-hf-q4f16_1", chat_config=cfg) [](https://hamel.dev/notes/llm/inference/inference.html#cb2-4)output = cm.generate(prompt=prompt)