import re from reasoning_gym import create_dataset, get_score_answer_fn from transformers import AutoModelForCausalLM, AutoTokenizer import torch model_name = "HuggingfaceTB/SmolLM-135M-Instruct" # load model from huggingface lm = AutoModelForCausalLM.from_pretrained(model_name, torch_dtype=torch.bfloat16) tokenizer = AutoTokenizer.from_pretrained(model_name) # This sets all models as trainable for param in lm.parameters(): param.requires_grad = True # In my experiments, I used a LORA adapter (more on this later) # specify name of the env environment_name = "propositional_logic" # In practice, you should wrap this with a torch dataloader # to sample a minibatch of questions dataset = create_dataset( environment_name, seed=42, size=DATA_SIZE ) for d in dataset: question = d["question"] # Accessing the question # We will use this later to verify if answer is correct validation_object = d["metadata"]["source_dataset"] score_fn = get_score_answer_fn(validation_object)