context = prompt_tokens K_cache, V_cache = compute_and_store_kv(context) while not finished: q_new, k_new, v_new = compute_qkv(new_token) K_cache.append(k_new) V_cache.append(v_new) output = attention(q_new, K_cache, V_cache) next_token = sample(output)