# PREFILL: Process the input prompt, fill the cache prompt_output, kv_cache_list = self.forward( inputs, kv_cache=None, start_pos=0 ) # DECODE: Generate one token at a time using cached K/V for i in range(max_new_tokens): next_token = sample_from(prompt_output) decode_output, kv_cache_list = self.forward( next_token, kv_cache=kv_cache_list, start_pos=current_position # updated with each step ) prompt_output = decode_output