# Conceptual memory allocation for Speculative Decoding in vLLM/AMD def allocate_kv_cache(model_config, device_memory_pool): target_memory = model_config.target_model_size * target_kv_overhead draft_memory = model_config.draft_model_size * draft_kv_overhead # AMD specific: ensure block alignment for MI300X memory controllers alignment = 256 * 1024 # 256KB alignment for optimal coalescing target_pool = device_memory_pool.allocate(target_memory, alignment) draft_pool = device_memory_pool.allocate(draft_memory, alignment) return target_pool, draft_pool