# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
"""Unit tests for SimpleCPUOffloadScheduler."""

from __future__ import annotations

import logging
from dataclasses import dataclass

import pytest
import torch

from vllm import SamplingParams
from vllm.config import (
    CacheConfig,
    DeviceConfig,
    KVTransferConfig,
    ModelConfig,
    SchedulerConfig,
    VllmConfig,
)
from vllm.config.cache import MambaCacheMode
from vllm.config.kv_events import KVEventsConfig
from vllm.distributed.kv_transfer.kv_connector.v1.simple_cpu_offload_connector import (
    SimpleCPUOffloadConnector,
)
from vllm.utils.hashing import sha256
from vllm.v1.core.block_pool import BlockPool
from vllm.v1.core.kv_cache_manager import KVCacheBlocks
from vllm.v1.core.kv_cache_utils import (
    get_request_block_hasher,
    init_none_hash,
    make_block_hash_with_group_id,
)
from vllm.v1.core.sched.output import (
    CachedRequestData,
    KVConnectorBlockState,
    NewRequestData,
    SchedulerOutput,
)
from vllm.v1.core.single_type_kv_cache_manager import (
    register_all_kvcache_specs,
)
from vllm.v1.kv_cache_interface import (
    CircularBufferSpec,
    FullAttentionSpec,
    KpoolTailSpec,
    KVCacheConfig,
    KVCacheGroupSpec,
    KVCacheTensor,
    MambaSpec,
    SlidingWindowSpec,
)
from vllm.v1.outputs import KVConnectorOutput
from vllm.v1.request import Request
from vllm.v1.simple_kv_offload.manager import SimpleCPUOffloadScheduler
from vllm.v1.simple_kv_offload.metadata import SimpleCPUOffloadWorkerMetadata

pytestmark = pytest.mark.skip_global_cleanup


# ---------------------------------------------------------------------------
# Constants
# ---------------------------------------------------------------------------
BLOCK_SIZE = 16
HEAD_SIZE = 16
NUM_KV_HEADS = 1
DTYPE = torch.float16
# bytes per block per tensor:
# block_size * num_kv_heads * head_size * 2 (K+V) * element_size
_BYTES_PER_BLOCK = BLOCK_SIZE * NUM_KV_HEADS * HEAD_SIZE * 2 * DTYPE.itemsize

# Ensure none_hash is initialized once
init_none_hash(sha256)


# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------


def _make_kv_cache_config(
    num_blocks: int,
    num_groups: int = 1,
) -> KVCacheConfig:
    """Build a KVCacheConfig with non-empty kv_cache_tensors."""
    groups = []
    tensors = []
    register_all_kvcache_specs(
        vllm_config=None
    )  # Ensure specs are registered for tests
    for g in range(num_groups):
        layer_names = [f"layer_{g}"]
        spec = (
            FullAttentionSpec(
                block_size=BLOCK_SIZE,
                num_kv_heads=NUM_KV_HEADS,
                head_size=HEAD_SIZE,
                dtype=DTYPE,
            )
            if g == 0
            else SlidingWindowSpec(
                block_size=BLOCK_SIZE,
                num_kv_heads=NUM_KV_HEADS,
                head_size=HEAD_SIZE,
                dtype=DTYPE,
                sliding_window=BLOCK_SIZE * 4,
            )
        )
        groups.append(
            KVCacheGroupSpec(
                layer_names,
                spec,
            )
        )
        tensors.append(
            KVCacheTensor(
                size=_BYTES_PER_BLOCK * num_blocks,
                layers=layer_names,
                layer_stride=_BYTES_PER_BLOCK * num_blocks,
                block_stride=_BYTES_PER_BLOCK,
            )
        )
    return KVCacheConfig(
        num_blocks=num_blocks,
        kv_cache_tensors=tensors,
        kv_cache_groups=groups,
    )


def _make_scratch_kv_cache_config(
    num_blocks: int, scratch_block_size: int = 4
) -> KVCacheConfig:
    """FullAttention group plus a non-prefix-cacheable scratch group.

    Mirrors GLM-5.3-Flash, whose kpool-tail group holds one per-request
    scratch block of ``index_kpool`` tokens: it is excluded from prefix
    caching, so its block size neither divides nor is divided by the
    hash block size in general.
    """
    register_all_kvcache_specs(vllm_config=None)
    fa_config = _make_kv_cache_config(num_blocks, num_groups=1)
    scratch_layers = ["layer_scratch"]
    scratch_spec = KpoolTailSpec(
        block_size=scratch_block_size,
        num_kv_heads=2,
        head_size=HEAD_SIZE,
        head_size_v=0,
        dtype=DTYPE,
        sliding_window=scratch_block_size,
    )
    assert not scratch_spec.prefix_cacheable
    scratch_bytes = scratch_spec.page_size_bytes * num_blocks
    return KVCacheConfig(
        num_blocks=num_blocks,
        kv_cache_tensors=fa_config.kv_cache_tensors
        + [
            KVCacheTensor(
                size=scratch_bytes,
                layers=scratch_layers,
                layer_stride=scratch_bytes,
                block_stride=scratch_spec.page_size_bytes,
            )
        ],
        kv_cache_groups=fa_config.kv_cache_groups
        + [KVCacheGroupSpec(scratch_layers, scratch_spec)],
    )


def _make_vllm_config(block_size: int = BLOCK_SIZE) -> VllmConfig:
    """Minimal VllmConfig for scheduler tests (no GPU)."""
    model_config = ModelConfig(
        model="facebook/opt-125m",
        trust_remote_code=True,
        dtype="float16",
        seed=42,
    )
    scheduler_config = SchedulerConfig(
        max_num_seqs=16,
        max_num_batched_tokens=64,
        max_model_len=10000,
        enable_chunked_prefill=True,
        is_encoder_decoder=False,
    )
    cache_config = CacheConfig(
        block_size=block_size,
        gpu_memory_utilization=0.9,
        enable_prefix_caching=True,
    )
    kv_transfer_config = KVTransferConfig(
        kv_connector="SimpleCPUOffloadConnector",
        kv_role="kv_both",
    )
    return VllmConfig(
        scheduler_config=scheduler_config,
        model_config=model_config,
        cache_config=cache_config,
        kv_transfer_config=kv_transfer_config,
        device_config=DeviceConfig("cpu"),
    )


@dataclass
class SchedulerFixture:
    """Bundle returned by make_scheduler for convenient access."""

    scheduler: SimpleCPUOffloadScheduler
    gpu_block_pool: BlockPool
    vllm_config: VllmConfig
    kv_cache_config: KVCacheConfig
    num_groups: int = 1


def make_scheduler(
    num_cpu_blocks: int = 8,
    num_gpu_blocks: int = 16,
    num_groups: int = 1,
    lazy: bool = False,
    kv_cache_config: KVCacheConfig | None = None,
) -> SchedulerFixture:
    """Build a SimpleCPUOffloadScheduler with small block pools."""
    if kv_cache_config is None:
        kv_cache_config = _make_kv_cache_config(num_gpu_blocks, num_groups)
    else:
        num_groups = len(kv_cache_config.kv_cache_groups)
    vllm_config = _make_vllm_config()
    cpu_capacity_bytes = _BYTES_PER_BLOCK * num_cpu_blocks * num_groups

    sched = SimpleCPUOffloadScheduler(
        vllm_config=vllm_config,
        kv_cache_config=kv_cache_config,
        cpu_capacity_bytes=cpu_capacity_bytes,
        scheduler_block_size=BLOCK_SIZE,
        hash_block_size=BLOCK_SIZE,
        lazy_offload=lazy,
    )

    # Build a real GPU block pool and bind it
    gpu_block_pool = BlockPool(
        num_gpu_blocks=num_gpu_blocks,
        enable_caching=True,
        hash_block_size=BLOCK_SIZE,
    )
    sched.bind_gpu_block_pool(gpu_block_pool)

    return SchedulerFixture(
        scheduler=sched,
        gpu_block_pool=gpu_block_pool,
        vllm_config=vllm_config,
        kv_cache_config=kv_cache_config,
        num_groups=num_groups,
    )


_req_counter = 0


def make_request(
    num_blocks: int = 2,
    request_id: str | None = None,
    extra_tokens: int = 1,
) -> Request:
    """Create a Request with deterministic block hashes."""
    global _req_counter
    _req_counter += 1
    if request_id is None:
        request_id = f"req-{_req_counter}"

    num_tokens = num_blocks * BLOCK_SIZE + extra_tokens
    start = _req_counter * 10000
    prompt_token_ids = list(range(start, start + num_tokens))
    sampling_params = SamplingParams(max_tokens=1)

    req = Request(
        request_id=request_id,
        prompt_token_ids=prompt_token_ids,
        sampling_params=sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=get_request_block_hasher(BLOCK_SIZE, sha256),
    )
    return req


def make_scheduler_output(
    req_id_to_num_tokens: dict[str, int],
    *,
    new_reqs: dict[str, tuple[list[int], ...]] | None = None,
    cached_req_new_blocks: dict[str, tuple[list[int], ...] | None] | None = None,
) -> SchedulerOutput:
    """Build a minimal SchedulerOutput with num_scheduled_tokens.

    Args:
        new_reqs: For first-time requests, maps req_id -> block_ids tuple.
            These are placed into ``scheduled_new_reqs`` as ``NewRequestData``.
        cached_req_new_blocks: For returning (cached) requests, maps
            req_id -> new_block_ids (incremental) or None.
            These are placed into ``scheduled_cached_reqs``.

    """
    scheduled_new_reqs: list[NewRequestData] = []
    if new_reqs:
        for req_id, block_ids in new_reqs.items():
            scheduled_new_reqs.append(
                NewRequestData(
                    req_id=req_id,
                    prompt_token_ids=None,
                    mm_features=[],
                    sampling_params=None,
                    pooling_params=None,
                    block_ids=block_ids,
                    num_computed_tokens=0,
                    lora_request=None,
                )
            )

    if cached_req_new_blocks:
        cached_req_ids = list(cached_req_new_blocks.keys())
        cached_new_block_ids = [cached_req_new_blocks[rid] for rid in cached_req_ids]
        cached_reqs = CachedRequestData(
            req_ids=cached_req_ids,
            resumed_req_ids=set(),
            new_token_ids=[[] for _ in cached_req_ids],
            all_token_ids={},
            new_block_ids=cached_new_block_ids,
            num_computed_tokens=[0] * len(cached_req_ids),
            num_output_tokens=[0] * len(cached_req_ids),
        )
    else:
        cached_reqs = CachedRequestData.make_empty()

    return SchedulerOutput(
        scheduled_new_reqs=scheduled_new_reqs,
        scheduled_cached_reqs=cached_reqs,
        num_scheduled_tokens=req_id_to_num_tokens,
        total_num_scheduled_tokens=sum(req_id_to_num_tokens.values()),
        scheduled_spec_decode_tokens={},
        scheduled_encoder_inputs={},
        num_common_prefix_blocks=[],
        preempted_req_ids=set(),
        finished_req_ids=set(),
        free_encoder_mm_hashes=[],
    )


def simulate_store_completion(
    scheduler: SimpleCPUOffloadScheduler,
    event_idx: int,
) -> None:
    """Simulate worker reporting a store event completion."""
    output = KVConnectorOutput(
        finished_recving=set(),
        kv_connector_worker_meta=SimpleCPUOffloadWorkerMetadata(
            completed_store_events={event_idx: scheduler._expected_worker_count},
        ),
    )
    scheduler.update_connector_output(output)


def simulate_load_completion(
    scheduler: SimpleCPUOffloadScheduler,
    req_ids: set[str],
) -> None:
    """Simulate worker reporting load completions for requests."""
    output = KVConnectorOutput(
        finished_sending=set(),
        finished_recving=req_ids,
    )
    scheduler.update_connector_output(output)


def get_cpu_free_blocks(scheduler: SimpleCPUOffloadScheduler) -> int:
    """Return number of free CPU blocks."""
    return scheduler.cpu_block_pool.get_num_free_blocks()


def _allocate_gpu_blocks(
    gpu_block_pool: BlockPool,
    request: Request,
    num_blocks: int,
    group_id: int = 0,
) -> list:
    """Allocate GPU blocks, cache them with hashes, return block list.

    Mimics what KVCacheManager does: allocate blocks from pool, then
    register them in the prefix cache via cache_full_blocks so that
    re-allocation properly evicts stale hashes.
    """
    blocks = gpu_block_pool.get_new_blocks(num_blocks)
    num_full = min(num_blocks, len(request.block_hashes))
    if num_full > 0:
        gpu_block_pool.cache_full_blocks(
            request=request,
            blocks=blocks,
            num_cached_blocks=0,
            num_full_blocks=num_full,
            block_size=BLOCK_SIZE,
            kv_cache_group_id=group_id,
        )
    return blocks


def _alloc_and_register(
    fix: SchedulerFixture,
    request: Request,
    num_blocks: int,
    *,
    confirmed: bool = True,
    group_id: int = 0,
) -> KVCacheBlocks:
    """Allocate GPU blocks and return KVCacheBlocks.

    Block IDs are no longer registered in a mock KVCacheManager; instead
    tests pass them through ``make_scheduler_output`` so that
    ``yield_req_data`` can pick them up.

    If ``confirmed`` is True, advance ``request.num_computed_tokens`` to simulate
    the scheduler's ``_update_after_schedule`` from a prior step.
    """
    gpu_blocks = _allocate_gpu_blocks(
        fix.gpu_block_pool, request, num_blocks, group_id=group_id
    )
    kv_blocks = KVCacheBlocks(blocks=(gpu_blocks,))
    if confirmed:
        request.num_computed_tokens = num_blocks * BLOCK_SIZE
    return kv_blocks


# ---------------------------------------------------------------------------
# Test 1a: Eager store-and-load roundtrip
# ---------------------------------------------------------------------------
def test_eager_store_and_load_roundtrip() -> None:
    """Eager mode: store blocks on compute, complete store, verify cache hit."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    num_blocks = 2
    req = make_request(num_blocks=num_blocks)

    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: block_ids},
    )

    meta = sched.build_connector_meta(sched_out)
    assert meta.store_event >= 0, "Expected a store event to be scheduled"
    assert len(meta.store_gpu_blocks) > 0
    assert len(meta.store_cpu_blocks) == len(meta.store_gpu_blocks)
    simulate_store_completion(sched, meta.store_event)

    # New request with same tokens should get CPU cache hit
    req2 = Request(
        request_id="req-eager-load",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    # make_request pads num_tokens by +1 beyond the last full block, so the
    # manager's max_hit_len = num_tokens - 1 cap leaves all full blocks intact.
    assert hit_tokens == num_blocks * BLOCK_SIZE
    assert is_async is True

    gpu_blocks2 = fix.gpu_block_pool.get_new_blocks(num_blocks)
    kv_blocks2 = KVCacheBlocks(blocks=(gpu_blocks2,))
    sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=hit_tokens)

    block_ids2 = kv_blocks2.get_block_ids()
    sched_out2 = make_scheduler_output(
        {req2.request_id: 1},
        new_reqs={req2.request_id: block_ids2},
    )
    meta2 = sched.build_connector_meta(sched_out2)
    assert meta2.load_event >= 0, "Expected a load event to be assigned"
    assert len(meta2.load_gpu_blocks) > 0
    assert len(meta2.load_cpu_blocks) == len(meta2.load_gpu_blocks)


def test_prompt_logprobs_skip_cpu_cache_lookup() -> None:
    """Prompt logprobs require every prompt token to be recomputed."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    num_blocks = 2
    req = make_request(num_blocks=num_blocks)
    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: kv_blocks.get_block_ids()},
    )
    meta = sched.build_connector_meta(sched_out)
    simulate_store_completion(sched, meta.store_event)

    retry_request_id = "req-prompt-logprobs"
    cached_request = Request(
        request_id=retry_request_id,
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    assert sched.get_num_new_matched_tokens(cached_request, num_computed_tokens=0) == (
        num_blocks * BLOCK_SIZE,
        True,
    )
    pending_hit = sched._pending_cpu_hits[retry_request_id]
    pinned_blocks = [
        block for group in pending_hit[0] for block in group if not block.is_null
    ]
    assert pinned_blocks
    assert all(block.ref_cnt == 1 for block in pinned_blocks)

    prompt_logprobs_request = Request(
        request_id=retry_request_id,
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=SamplingParams(max_tokens=1, prompt_logprobs=1),
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    assert prompt_logprobs_request.skip_reading_prefix_cache

    assert sched.get_num_new_matched_tokens(
        prompt_logprobs_request, num_computed_tokens=0
    ) == (0, False)
    assert retry_request_id not in sched._pending_cpu_hits
    assert all(block.ref_cnt == 0 for block in pinned_blocks)


def test_eager_store_preserves_secondary_block_hashes() -> None:
    """CPU copies retain fine-grained hashes owned by the GPU block."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler
    req = make_request(num_blocks=1)
    kv_blocks = _alloc_and_register(fix, req, num_blocks=1)
    gpu_block = kv_blocks.blocks[0][0]
    primary_hash = gpu_block.block_hash
    primary_num_tokens = gpu_block.block_hash_num_tokens
    assert primary_hash is not None

    fine_grained_req = Request(
        request_id="req-fine-grained-hash",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=get_request_block_hasher(BLOCK_SIZE // 2, sha256),
    )
    secondary_hash = make_block_hash_with_group_id(fine_grained_req.block_hashes[0], 0)
    fix.gpu_block_pool._insert_block_hash(
        secondary_hash, gpu_block, num_tokens=BLOCK_SIZE // 2
    )
    assert (
        secondary_hash
        in fix.gpu_block_pool.cached_block_hashes_by_block[gpu_block.block_id]
    )

    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    block_ids = kv_blocks.get_block_ids()
    meta = sched.build_connector_meta(
        make_scheduler_output(
            {req.request_id: BLOCK_SIZE},
            new_reqs={req.request_id: block_ids},
        )
    )
    simulate_store_completion(sched, meta.store_event)

    cpu_block = sched.cpu_block_pool.cached_block_hash_to_block.get_one_block(
        secondary_hash
    )
    assert cpu_block is not None
    assert cpu_block.block_hash == primary_hash
    assert cpu_block.block_hash_num_tokens == primary_num_tokens
    assert (
        secondary_hash
        in sched.cpu_block_pool.cached_block_hashes_by_block[cpu_block.block_id]
    )


# ---------------------------------------------------------------------------
# Test 1b: Finished eager stores
# ---------------------------------------------------------------------------
def test_finished_eager_stores_are_batched() -> None:
    """Finished requests in one step retain, complete, and cache every store."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    requests = [make_request(num_blocks=2) for _ in range(2)]
    block_ids = []
    for request in requests:
        kv_blocks = _alloc_and_register(fix, request, num_blocks=2)
        sched.update_state_after_alloc(request, kv_blocks, num_external_tokens=0)
        block_ids.append(kv_blocks.get_block_ids())

    for request, request_block_ids in zip(requests, block_ids):
        sched.request_finished_all_groups(request, request_block_ids)

    meta = sched.build_connector_meta(make_scheduler_output({}))
    assert len(meta.store_gpu_blocks) == 4
    assert len(meta.store_cpu_blocks) == 4
    simulate_store_completion(sched, meta.store_event)

    for request in requests:
        for block_hash in request.block_hashes[:2]:
            cached = sched.cpu_block_pool.cached_block_hash_to_block.get_one_block(
                make_block_hash_with_group_id(block_hash, 0)
            )
            assert cached is not None

    request = requests[0]
    cached_request = Request(
        request_id="req-finished-eager-load",
        prompt_token_ids=request.prompt_token_ids,
        sampling_params=request.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=request._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(
        cached_request, num_computed_tokens=0
    )
    assert hit_tokens == 2 * BLOCK_SIZE
    assert is_async is True

    cached_gpu_blocks = fix.gpu_block_pool.get_new_blocks(2)
    sched.update_state_after_alloc(
        cached_request,
        KVCacheBlocks(blocks=(cached_gpu_blocks,)),
        num_external_tokens=hit_tokens,
    )
    load_meta = sched.build_connector_meta(
        make_scheduler_output(
            {cached_request.request_id: 1},
            new_reqs={
                cached_request.request_id: (
                    [block.block_id for block in cached_gpu_blocks],
                )
            },
        )
    )
    assert load_meta.load_event >= 0
    assert len(load_meta.load_gpu_blocks) == 2
    assert len(load_meta.load_cpu_blocks) == 2


def test_finished_eager_store_skips_existing_cpu_cache_entries() -> None:
    """A final flush must not consume CPU capacity for an existing hash."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    first = make_request(num_blocks=2)
    first_blocks = _alloc_and_register(fix, first, num_blocks=2)
    sched.update_state_after_alloc(first, first_blocks, num_external_tokens=0)
    sched.request_finished_all_groups(first, first_blocks.get_block_ids())
    first_meta = sched.build_connector_meta(make_scheduler_output({}))
    simulate_store_completion(sched, first_meta.store_event)
    free_blocks = get_cpu_free_blocks(sched)

    duplicate = Request(
        request_id="req-finished-eager-duplicate",
        prompt_token_ids=first.prompt_token_ids,
        sampling_params=first.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=first._block_hasher,
    )
    duplicate_blocks = _alloc_and_register(fix, duplicate, num_blocks=2)
    sched.update_state_after_alloc(duplicate, duplicate_blocks, num_external_tokens=0)
    sched.request_finished_all_groups(duplicate, duplicate_blocks.get_block_ids())

    duplicate_meta = sched.build_connector_meta(make_scheduler_output({}))
    assert duplicate_meta.store_event == -1
    assert get_cpu_free_blocks(sched) == free_blocks


def test_finished_eager_store_is_reported_pending_before_metadata_build() -> None:
    """Queued finished stores must keep the connector pending until submitted."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler
    request = make_request(num_blocks=2)
    kv_blocks = _alloc_and_register(fix, request, num_blocks=2)
    connector = SimpleCPUOffloadConnector.__new__(SimpleCPUOffloadConnector)
    connector.scheduler_manager = sched
    sched.update_state_after_alloc(request, kv_blocks, num_external_tokens=0)

    sched.request_finished_all_groups(request, kv_blocks.get_block_ids())

    assert sched.has_pending_stores()
    assert connector.has_pending_push_work()
    meta = sched.build_connector_meta(make_scheduler_output({}))
    assert meta.store_event >= 0
    assert not sched._pending_finished_stores
    assert connector.has_pending_push_work()
    simulate_store_completion(sched, meta.store_event)
    assert not sched.has_pending_stores()
    assert not connector.has_pending_push_work()


def test_finished_eager_store_caches_all_groups() -> None:
    """The final flush retains completed blocks from every KV cache group."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, num_groups=2, lazy=False)
    sched = fix.scheduler
    request = make_request(num_blocks=2)
    group_blocks = tuple(
        _allocate_gpu_blocks(fix.gpu_block_pool, request, 2, group_id=group_id)
        for group_id in range(2)
    )
    request.num_computed_tokens = 2 * BLOCK_SIZE
    kv_blocks = KVCacheBlocks(blocks=group_blocks)
    sched.update_state_after_alloc(request, kv_blocks, num_external_tokens=0)

    sched.request_finished_all_groups(request, kv_blocks.get_block_ids())
    meta = sched.build_connector_meta(make_scheduler_output({}))
    assert len(meta.store_gpu_blocks) == 4
    simulate_store_completion(sched, meta.store_event)

    for group_id in range(2):
        for block_hash in request.block_hashes[:2]:
            cached = sched.cpu_block_pool.cached_block_hash_to_block.get_one_block(
                make_block_hash_with_group_id(block_hash, group_id)
            )
            assert cached is not None


def test_finished_eager_store_respects_cpu_capacity() -> None:
    """A finished request skips excess blocks when the CPU cache is full."""
    fix = make_scheduler(num_cpu_blocks=3, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler
    request = make_request(num_blocks=3)
    kv_blocks = _alloc_and_register(fix, request, num_blocks=3)
    sched.update_state_after_alloc(request, kv_blocks, num_external_tokens=0)
    available = get_cpu_free_blocks(sched)

    sched.request_finished_all_groups(request, kv_blocks.get_block_ids())
    meta = sched.build_connector_meta(make_scheduler_output({}))

    assert len(meta.store_gpu_blocks) == available
    assert len(meta.store_cpu_blocks) == available


def test_reset_releases_unsubmitted_finished_stores() -> None:
    """Reset releases finished-store refs that were not sent to a worker."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler
    request = make_request(num_blocks=2)
    kv_blocks = _alloc_and_register(fix, request, num_blocks=2)
    sched.update_state_after_alloc(request, kv_blocks, num_external_tokens=0)

    sched.request_finished_all_groups(request, kv_blocks.get_block_ids())

    assert sched.reset()


# ---------------------------------------------------------------------------
# Test 1b: Boundary — max_hit_len cap drops the last full block when the
# prompt is an exact multiple of BLOCK_SIZE.
# ---------------------------------------------------------------------------
def test_max_hit_len_cap_drops_last_full_block() -> None:
    """When num_tokens is an exact multiple of BLOCK_SIZE, the manager's
    ``max_hit_len = num_tokens - 1`` cap forces ``find_longest_cache_hit`` to
    drop the final block (since ``max_length // block_size`` rounds down).
    """
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    num_blocks = 2
    req = make_request(num_blocks=num_blocks, extra_tokens=0)
    assert req.num_tokens == num_blocks * BLOCK_SIZE

    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: kv_blocks.get_block_ids()},
    )
    meta = sched.build_connector_meta(sched_out)
    simulate_store_completion(sched, meta.store_event)

    req2 = Request(
        request_id="req-cap-boundary",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    hit_tokens, _ = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    assert hit_tokens == (num_blocks - 1) * BLOCK_SIZE


# ---------------------------------------------------------------------------
# Test 1c: Lazy store-and-load roundtrip
# ---------------------------------------------------------------------------
def _flush_old_blocks_to_lru_head(
    gpu_pool: BlockPool,
    num_filler_blocks: int,
) -> list:
    """Allocate filler blocks so that previously-freed (hashed) blocks migrate
    to the LRU head of the free queue.  Returns the filler blocks (caller must
    free them later to restore pool capacity).

    In a real engine the same thing happens naturally: after one request
    finishes and frees its blocks, subsequent requests allocate from the LRU
    head, consuming the unhashed blocks and leaving the old hashed blocks at
    the front of the queue.
    """
    fillers = gpu_pool.get_new_blocks(num_filler_blocks)
    return fillers


def test_lazy_store_and_load_roundtrip() -> None:
    """Lazy mode: schedule a request, finish it so its hashed blocks are freed,
    then schedule new requests so the old blocks migrate to the LRU head.
    The lazy scanner offloads them to CPU.  Re-scheduling the old request
    triggers a CPU cache hit + load.

    GPU pool: 8 blocks (7 usable).  _target_free = ceil(64/16) = 4.
    """
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=8, lazy=True)
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool

    num_blocks = 2

    # --- Step 1: Schedule req_old, compute, and finish ---
    req_old = make_request(num_blocks=num_blocks)
    gpu_blocks_old = _allocate_gpu_blocks(gpu_pool, req_old, num_blocks, group_id=0)
    gpu_pool.free_blocks(gpu_blocks_old)

    # Allocate filler blocks so req_old's hashed blocks move to LRU head.
    # 7 usable - 2 (req_old freed) = 5 other free blocks to consume.
    fillers = _flush_old_blocks_to_lru_head(gpu_pool, num_filler_blocks=5)

    # --- Step 2: Lazy scanner should offload req_old's blocks ---
    sched_out = make_scheduler_output({})
    meta = sched.build_connector_meta(sched_out)
    assert meta.store_event >= 0, "Expected lazy store to offload old blocks"
    assert len(meta.store_gpu_blocks) == num_blocks
    simulate_store_completion(sched, meta.store_event)

    # Free fillers to restore pool capacity.
    gpu_pool.free_blocks(fillers)

    # --- Step 3: Re-schedule req_old — should get CPU cache hit ---
    req_old2 = Request(
        request_id="req-old-reload",
        prompt_token_ids=req_old.prompt_token_ids,
        sampling_params=req_old.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req_old._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(
        req_old2, num_computed_tokens=0
    )
    # make_request pads num_tokens by +1 beyond the last full block, so the
    # manager's max_hit_len = num_tokens - 1 cap leaves all full blocks intact.
    expected_hit = num_blocks * BLOCK_SIZE
    assert hit_tokens == expected_hit, (
        f"Expected {expected_hit} hit tokens, got {hit_tokens}"
    )
    assert is_async is True

    # Allocate fresh GPU blocks for the load.
    gpu_blocks_load = gpu_pool.get_new_blocks(num_blocks)
    kv_blocks_load = KVCacheBlocks(blocks=(gpu_blocks_load,))
    sched.update_state_after_alloc(
        req_old2, kv_blocks_load, num_external_tokens=hit_tokens
    )

    sched_out2 = make_scheduler_output({req_old2.request_id: 1})
    meta2 = sched.build_connector_meta(sched_out2)
    assert meta2.load_event >= 0, "Expected a load event to be assigned"
    assert len(meta2.load_gpu_blocks) > 0


# ---------------------------------------------------------------------------
# Test 2a: Eager duplicate store is skipped
# ---------------------------------------------------------------------------
def test_eager_duplicate_store_skipped() -> None:
    """Eager: storing the same block hashes twice should not allocate new CPU blocks."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    num_blocks = 2
    req = make_request(num_blocks=num_blocks)

    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: block_ids},
    )

    meta1 = sched.build_connector_meta(sched_out)
    assert meta1.store_event >= 0
    simulate_store_completion(sched, meta1.store_event)
    cpu_free_after_first = get_cpu_free_blocks(sched)

    # Second request with identical hashes — should skip store
    req2 = Request(
        request_id="req-dup-eager",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    kv_blocks2 = _alloc_and_register(fix, req2, num_blocks)
    sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=0)
    block_ids2 = kv_blocks2.get_block_ids()
    sched_out2 = make_scheduler_output(
        {req2.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req2.request_id: block_ids2},
    )

    meta2 = sched.build_connector_meta(sched_out2)
    if meta2.store_event >= 0:
        assert len(meta2.store_cpu_blocks) == 0, (
            "Expected no new CPU blocks for duplicate hashes"
        )
    assert get_cpu_free_blocks(sched) == cpu_free_after_first


# ---------------------------------------------------------------------------
# Test 2b: Eager dedup of in-flight stores across consecutive steps
# ---------------------------------------------------------------------------
def test_eager_in_flight_store_dedup_across_steps() -> None:
    """Eager: a second request sharing a prefix with an in-flight store
    must not re-offload the same GPU blocks before completion lands.

    Simulates a GPU prefix-cache hit by reusing the first request's
    GPU block IDs in the second scheduler step, which is the path the
    real scheduler takes when two requests share a prefix.
    """
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    num_blocks = 2
    req = make_request(num_blocks=num_blocks)

    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: block_ids},
    )

    meta1 = sched.build_connector_meta(sched_out)
    assert meta1.store_event >= 0
    assert len(meta1.store_cpu_blocks) == num_blocks
    # In-flight set tracks the scheduled GPU blocks until completion.
    assert sched._in_flight_store_gpu_blocks == set(meta1.store_gpu_blocks)
    cpu_free_after_first = get_cpu_free_blocks(sched)

    # Second request shares the prefix and reuses the same GPU block IDs
    # (the real scheduler path: GPU prefix cache returns the same blocks).
    # Do NOT simulate completion — the first store is still in-flight.
    req2 = Request(
        request_id="req-dup-eager-inflight",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    req2.num_computed_tokens = num_blocks * BLOCK_SIZE
    sched.update_state_after_alloc(req2, kv_blocks, num_external_tokens=0)
    sched_out2 = make_scheduler_output(
        {req2.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req2.request_id: block_ids},
    )

    meta2 = sched.build_connector_meta(sched_out2)
    if meta2.store_event >= 0:
        assert len(meta2.store_cpu_blocks) == 0, (
            "Expected no new CPU blocks for in-flight duplicate hashes"
        )
    assert get_cpu_free_blocks(sched) == cpu_free_after_first, (
        "Second request should not consume CPU blocks while the first "
        "store is still in-flight"
    )

    # After completion, the in-flight set is cleared.
    simulate_store_completion(sched, meta1.store_event)
    assert sched._in_flight_store_gpu_blocks == set()


# ---------------------------------------------------------------------------
# Test 2c: Lazy duplicate store is skipped
# ---------------------------------------------------------------------------
def test_lazy_duplicate_store_skipped() -> None:
    """Lazy: blocks already offloaded to CPU should not be offloaded again.

    Same pattern as the lazy roundtrip: flush old blocks to LRU head, offload,
    then repeat with the same hashes and verify no new CPU allocation.
    """
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=8, lazy=True)
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool

    num_blocks = 2
    req = make_request(num_blocks=num_blocks)

    # Schedule + finish → hashed blocks in free queue
    gpu_blocks = _allocate_gpu_blocks(gpu_pool, req, num_blocks, group_id=0)
    gpu_pool.free_blocks(gpu_blocks)

    # Flush old blocks to LRU head, then trigger lazy offload.
    fillers = _flush_old_blocks_to_lru_head(gpu_pool, num_filler_blocks=5)
    meta1 = sched.build_connector_meta(make_scheduler_output({}))
    assert meta1.store_event >= 0
    simulate_store_completion(sched, meta1.store_event)
    gpu_pool.free_blocks(fillers)
    cpu_free_after_first = get_cpu_free_blocks(sched)

    # Allocate blocks with the same hashes and free them again.
    # The scanner should see they are already in CPU cache and skip them.
    req2 = Request(
        request_id="req-dup-lazy",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    gpu_blocks2 = _allocate_gpu_blocks(gpu_pool, req2, num_blocks, group_id=0)
    gpu_pool.free_blocks(gpu_blocks2)

    # Flush again so the hashed blocks are at LRU head for the scanner.
    fillers2 = _flush_old_blocks_to_lru_head(gpu_pool, num_filler_blocks=5)
    meta2 = sched.build_connector_meta(make_scheduler_output({}))
    gpu_pool.free_blocks(fillers2)

    # Either no store event, or zero new CPU blocks (already cached).
    if meta2.store_event >= 0:
        assert len(meta2.store_cpu_blocks) == 0, (
            "Expected no new CPU blocks for duplicate hashes"
        )
    assert get_cpu_free_blocks(sched) == cpu_free_after_first


# ---------------------------------------------------------------------------
# Test 3: LRU eviction order
# ---------------------------------------------------------------------------
def test_lru_eviction_order() -> None:
    """With limited CPU space, oldest blocks should be evicted first.

    CPU block pool: num_cpu_blocks=5 -> 4 free usable blocks (1 taken by null_block).
    After storing 4 blocks (2 req_a + 2 req_b), all free slots are occupied by
    cached blocks (ref_cnt=0, in hash map).  When 2 more are stored (req_c),
    2 LRU blocks from req_a get evicted from the cache to make room.
    """
    # 5 total = 4 usable (null_block takes 1), filling exactly with 4 blocks
    fix = make_scheduler(num_cpu_blocks=5, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    # Fill CPU with 4 blocks: 2 requests x 2 blocks (in LRU insertion order)
    req_a = make_request(num_blocks=2)
    req_b = make_request(num_blocks=2)

    kv_a = _alloc_and_register(fix, req_a, 2)
    kv_b = _alloc_and_register(fix, req_b, 2)
    sched.update_state_after_alloc(req_a, kv_a, num_external_tokens=0)
    sched.update_state_after_alloc(req_b, kv_b, num_external_tokens=0)

    ids_a = kv_a.get_block_ids()
    ids_b = kv_b.get_block_ids()
    sched_out = make_scheduler_output(
        {
            req_a.request_id: 2 * BLOCK_SIZE,
            req_b.request_id: 2 * BLOCK_SIZE,
        },
        new_reqs={
            req_a.request_id: ids_a,
            req_b.request_id: ids_b,
        },
    )
    meta = sched.build_connector_meta(sched_out)
    assert meta.store_event >= 0
    simulate_store_completion(sched, meta.store_event)

    # Verify all 4 blocks are cached in CPU hash map
    for i, bhash in enumerate(req_a.block_hashes[:2]):
        bhash_with_group = make_block_hash_with_group_id(bhash, 0)
        assert (
            sched.cpu_block_pool.cached_block_hash_to_block.get_one_block(
                bhash_with_group
            )
            is not None
        ), f"req_a block {i} should be cached after store"
    for i, bhash in enumerate(req_b.block_hashes[:2]):
        bhash_with_group = make_block_hash_with_group_id(bhash, 0)
        assert (
            sched.cpu_block_pool.cached_block_hash_to_block.get_one_block(
                bhash_with_group
            )
            is not None
        ), f"req_b block {i} should be cached after store"

    # Store 2 more blocks from a new request - must evict 2 LRU blocks (req_a)
    req_c = make_request(num_blocks=2)
    kv_c = _alloc_and_register(fix, req_c, 2)
    sched.update_state_after_alloc(req_c, kv_c, num_external_tokens=0)

    ids_c = kv_c.get_block_ids()
    sched_out2 = make_scheduler_output(
        {req_c.request_id: 2 * BLOCK_SIZE},
        new_reqs={req_c.request_id: ids_c},
    )
    meta2 = sched.build_connector_meta(sched_out2)
    assert meta2.store_event >= 0
    simulate_store_completion(sched, meta2.store_event)

    # req_a hashes should be evicted from CPU (they were LRU)
    for i, bhash in enumerate(req_a.block_hashes[:2]):
        bhash_with_group = make_block_hash_with_group_id(bhash, 0)
        cache_map = sched.cpu_block_pool.cached_block_hash_to_block
        cached = cache_map.get_one_block(bhash_with_group)
        assert cached is None, f"req_a block {i} should have been evicted"

    # req_b and req_c hashes should be present
    for i, bhash in enumerate(req_b.block_hashes[:2]):
        bhash_with_group = make_block_hash_with_group_id(bhash, 0)
        cache_map = sched.cpu_block_pool.cached_block_hash_to_block
        cached = cache_map.get_one_block(bhash_with_group)
        assert cached is not None, f"req_b block {i} should still be cached"

    for i, bhash in enumerate(req_c.block_hashes[:2]):
        bhash_with_group = make_block_hash_with_group_id(bhash, 0)
        cache_map = sched.cpu_block_pool.cached_block_hash_to_block
        cached = cache_map.get_one_block(bhash_with_group)
        assert cached is not None, f"req_c block {i} should still be cached"


# ---------------------------------------------------------------------------
# Test 4: Touched blocks survive eviction
# ---------------------------------------------------------------------------
def test_touched_blocks_survive_eviction() -> None:
    """Touching CPU blocks updates their LRU position, protecting them from eviction."""
    # 5 total = 4 usable (null_block takes 1)
    fix = make_scheduler(num_cpu_blocks=5, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    # Fill CPU with 4 blocks (req_a: 2, req_b: 2) in LRU order
    req_a = make_request(num_blocks=2)
    req_b = make_request(num_blocks=2)

    kv_a = _alloc_and_register(fix, req_a, 2)
    kv_b = _alloc_and_register(fix, req_b, 2)
    sched.update_state_after_alloc(req_a, kv_a, num_external_tokens=0)
    sched.update_state_after_alloc(req_b, kv_b, num_external_tokens=0)

    ids_a = kv_a.get_block_ids()
    ids_b = kv_b.get_block_ids()
    sched_out = make_scheduler_output(
        {
            req_a.request_id: 2 * BLOCK_SIZE,
            req_b.request_id: 2 * BLOCK_SIZE,
        },
        new_reqs={
            req_a.request_id: ids_a,
            req_b.request_id: ids_b,
        },
    )
    meta = sched.build_connector_meta(sched_out)
    simulate_store_completion(sched, meta.store_event)

    # Touch req_a's CPU blocks to make them most-recently-used
    cpu_pool = sched.cpu_block_pool
    for bhash in req_a.block_hashes[:2]:
        bhash_with_group = make_block_hash_with_group_id(bhash, 0)
        cached_blk = cpu_pool.cached_block_hash_to_block.get_one_block(bhash_with_group)
        assert cached_blk is not None
        cpu_pool.touch([cached_blk])
        # Undo touch to return ref_cnt to 0
        # (so it's a free candidate but at MRU position)
        cpu_pool.free_blocks([cached_blk])

    # Now store 2 more blocks; req_b (LRU front) should be evicted, not req_a
    req_c = make_request(num_blocks=2)
    kv_c = _alloc_and_register(fix, req_c, 2)
    sched.update_state_after_alloc(req_c, kv_c, num_external_tokens=0)

    ids_c = kv_c.get_block_ids()
    sched_out2 = make_scheduler_output(
        {req_c.request_id: 2 * BLOCK_SIZE},
        new_reqs={req_c.request_id: ids_c},
    )
    meta2 = sched.build_connector_meta(sched_out2)
    simulate_store_completion(sched, meta2.store_event)

    # req_b should be evicted (LRU), req_a and req_c should survive
    for i, bhash in enumerate(req_b.block_hashes[:2]):
        bhash_with_group = make_block_hash_with_group_id(bhash, 0)
        cached = cpu_pool.cached_block_hash_to_block.get_one_block(bhash_with_group)
        assert cached is None, f"req_b block {i} should have been evicted (it was LRU)"

    for i, bhash in enumerate(req_a.block_hashes[:2]):
        bhash_with_group = make_block_hash_with_group_id(bhash, 0)
        cached = cpu_pool.cached_block_hash_to_block.get_one_block(bhash_with_group)
        assert cached is not None, f"req_a block {i} should survive (was touched/MRU)"


# ---------------------------------------------------------------------------
# Test 5: Preemption no CPU block leak
# ---------------------------------------------------------------------------
def test_preemption_no_cpu_block_leak() -> None:
    """request_finished during in-flight load defers cleanup;
    completes after load done."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    num_blocks = 2

    # First: store blocks to CPU
    req = make_request(num_blocks=num_blocks)
    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: block_ids},
    )
    meta = sched.build_connector_meta(sched_out)
    simulate_store_completion(sched, meta.store_event)

    # Create new request with same tokens, check hit
    req2 = Request(
        request_id="req-preempt-load",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    assert hit_tokens is not None and hit_tokens > 0

    gpu_blocks2 = fix.gpu_block_pool.get_new_blocks(num_blocks)
    kv_blocks2 = KVCacheBlocks(blocks=(gpu_blocks2,))
    sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=hit_tokens)

    # Assign load_event via build_connector_meta
    block_ids2 = kv_blocks2.get_block_ids()
    sched_out2 = make_scheduler_output(
        {req2.request_id: 1},
        new_reqs={req2.request_id: block_ids2},
    )
    meta2 = sched.build_connector_meta(sched_out2)
    assert meta2.load_event >= 0

    # Request finishes BEFORE load completes -> deferred
    sched.request_finished(req2, block_ids=[])
    assert req2.request_id in sched._reqs_to_load
    assert sched._reqs_to_load[req2.request_id].finished is True

    # Now simulate load completion -> cleanup fires
    simulate_load_completion(sched, {req2.request_id})
    assert req2.request_id not in sched._reqs_to_load


# ---------------------------------------------------------------------------
# Test 6: Eager store preemption cleanup
# ---------------------------------------------------------------------------
def test_eager_store_preemption_cleanup() -> None:
    """In eager mode, finishing a request during in-flight store defers cleanup."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    num_blocks = 2
    req = make_request(num_blocks=num_blocks)
    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)

    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: block_ids},
    )
    meta = sched.build_connector_meta(sched_out)
    store_event = meta.store_event
    assert store_event >= 0

    # The request gets store_events populated
    assert req.request_id in sched._reqs_to_store
    store_state = sched._reqs_to_store[req.request_id]
    assert store_event in store_state.store_events

    # Finish request while store still in-flight -> deferred
    sched.request_finished(req, block_ids=[])
    assert req.request_id in sched._reqs_to_store
    assert sched._reqs_to_store[req.request_id].finished is True

    # Simulate store completion -> deferred cleanup fires
    simulate_store_completion(sched, store_event)
    assert req.request_id not in sched._reqs_to_store


# ---------------------------------------------------------------------------
# Test 7: In-flight finish deferred cleanup (load variant)
# ---------------------------------------------------------------------------
def test_inflight_finish_deferred_cleanup() -> None:
    """Store, then start a load, request_finished defers,
    load completion fires cleanup."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    num_blocks = 2

    # Store
    req = make_request(num_blocks=num_blocks)
    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: block_ids},
    )
    meta = sched.build_connector_meta(sched_out)
    simulate_store_completion(sched, meta.store_event)

    # Load
    req2 = Request(
        request_id="req-inflight-load",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    hit_tokens, _ = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    assert hit_tokens is not None and hit_tokens > 0

    gpu_blocks2 = fix.gpu_block_pool.get_new_blocks(num_blocks)
    kv_blocks2 = KVCacheBlocks(blocks=(gpu_blocks2,))
    sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=hit_tokens)

    block_ids2 = kv_blocks2.get_block_ids()
    sched_out2 = make_scheduler_output(
        {req2.request_id: 1},
        new_reqs={req2.request_id: block_ids2},
    )
    meta2 = sched.build_connector_meta(sched_out2)
    assert meta2.load_event >= 0

    # Finish before load completes
    sched.request_finished(req2, block_ids=[])
    assert req2.request_id in sched._reqs_to_load

    # Simulate load completion -> request removed
    simulate_load_completion(sched, {req2.request_id})
    assert req2.request_id not in sched._reqs_to_load


# ---------------------------------------------------------------------------
# Test 8: Null GPU blocks are skipped in store and load transfer pairs
# ---------------------------------------------------------------------------
def test_multi_group_null_blocks_skipped() -> None:
    """The null block id never appears in store or load pairs.

    A null block-table slot whose hash is still cached on the GPU is stored
    from the cached block instead; the null block itself is never copied.
    """
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, num_groups=1, lazy=False)
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool

    num_blocks = 2
    req = make_request(num_blocks=num_blocks)

    # Allocate real blocks (with hashes) and use the null_block as a placeholder
    gpu_blocks = _allocate_gpu_blocks(gpu_pool, req, num_blocks, group_id=0)
    null_block = gpu_pool.null_block

    # Mix: [real_block, null_block] — null_block has no hash, should be skipped
    mixed_blocks = [gpu_blocks[0], null_block]
    kv_blocks = KVCacheBlocks(blocks=(mixed_blocks,))
    req.num_computed_tokens = num_blocks * BLOCK_SIZE
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)

    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: block_ids},
    )
    meta = sched.build_connector_meta(sched_out)

    # Null block's ID should NOT appear in store_gpu_blocks
    null_block_id = null_block.block_id
    assert null_block_id not in meta.store_gpu_blocks, (
        f"Null block id {null_block_id} should not appear in store transfer pairs"
    )

    # The nulled slot's hash is still cached on the GPU, so that block is
    # recovered from the pool and stored alongside the real one.
    assert len(meta.store_gpu_blocks) == 2
    assert gpu_blocks[0].block_id in meta.store_gpu_blocks
    assert gpu_blocks[1].block_id in meta.store_gpu_blocks

    # Complete the store
    assert meta.store_event >= 0
    simulate_store_completion(sched, meta.store_event)

    # Create matching request and get load hit
    req2 = Request(
        request_id="req-null-load",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    assert hit_tokens == 2 * BLOCK_SIZE
    assert is_async is True

    # Allocate new GPU blocks for the load
    gpu_blocks2 = gpu_pool.get_new_blocks(2)
    kv_blocks2 = KVCacheBlocks(blocks=(gpu_blocks2,))
    sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=hit_tokens)

    sched_out2 = make_scheduler_output({req2.request_id: 1})
    meta2 = sched.build_connector_meta(sched_out2)

    # Null block's ID should NOT appear in load_gpu_blocks
    assert null_block_id not in meta2.load_gpu_blocks, (
        f"Null block id {null_block_id} should not appear in load transfer pairs"
    )


# ---------------------------------------------------------------------------
# Test 8b: Non-prefix-cacheable scratch groups take no part in store or load
# ---------------------------------------------------------------------------
def test_scratch_group_excluded_from_store_and_load() -> None:
    """A scratch group (GLM-5.3-Flash kpool tail) is never stored or loaded.

    Its single per-request block carries no hash, so the store must offload
    only the attention blocks and the load must pair only attention blocks.
    """
    fix = make_scheduler(
        num_cpu_blocks=8,
        num_gpu_blocks=16,
        kv_cache_config=_make_scratch_kv_cache_config(16),
    )
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool
    num_blocks = 2

    req = make_request(num_blocks=num_blocks)
    fa_blocks = _allocate_gpu_blocks(gpu_pool, req, num_blocks, group_id=0)
    scratch_block = gpu_pool.get_new_blocks(1)
    kv_blocks = KVCacheBlocks(blocks=(fa_blocks, scratch_block))
    req.num_computed_tokens = num_blocks * BLOCK_SIZE
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: kv_blocks.get_block_ids()},
    )
    meta = sched.build_connector_meta(sched_out)
    assert sorted(meta.store_gpu_blocks) == sorted(b.block_id for b in fa_blocks)
    simulate_store_completion(sched, meta.store_event)

    req2 = Request(
        request_id="req-scratch-load",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    assert hit_tokens == num_blocks * BLOCK_SIZE
    assert is_async is True

    fa_blocks2 = gpu_pool.get_new_blocks(num_blocks)
    scratch_block2 = gpu_pool.get_new_blocks(1)
    kv_blocks2 = KVCacheBlocks(blocks=(fa_blocks2, scratch_block2))
    sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=hit_tokens)
    sched_out2 = make_scheduler_output(
        {req2.request_id: 1},
        new_reqs={req2.request_id: kv_blocks2.get_block_ids()},
    )
    meta2 = sched.build_connector_meta(sched_out2)
    assert sorted(meta2.load_gpu_blocks) == sorted(b.block_id for b in fa_blocks2)
    assert len(meta2.load_cpu_blocks) == num_blocks


# ---------------------------------------------------------------------------
# Test 9: Chunked prefill accumulates block_ids across steps
# ---------------------------------------------------------------------------
def test_chunked_prefill_reads_live_block_ids() -> None:
    """With chunked prefill, block IDs accumulate across scheduler steps.
    _prepare_eager_store_specs reads block IDs from scheduler_output via
    yield_req_data, so the store should reflect the updated (larger) block
    list, not a stale snapshot."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    num_blocks = 4
    req = make_request(num_blocks=num_blocks)

    # First chunk: allocate 2 blocks
    kv_blocks_first = _alloc_and_register(fix, req, 2)
    sched.update_state_after_alloc(req, kv_blocks_first, num_external_tokens=0)

    assert req.request_id in sched._reqs_to_store
    # Should still be exactly 1 entry in _reqs_to_store
    assert list(sched._reqs_to_store.keys()).count(req.request_id) == 1

    # Build connector meta with 2 blocks — stores the first 2
    ids_first = kv_blocks_first.get_block_ids()
    sched_out1 = make_scheduler_output(
        {req.request_id: 2 * BLOCK_SIZE},
        new_reqs={req.request_id: ids_first},
    )
    meta1 = sched.build_connector_meta(sched_out1)
    assert meta1.store_event >= 0
    assert len(meta1.store_gpu_blocks) == 2
    simulate_store_completion(sched, meta1.store_event)

    # Second chunk: allocate 4 blocks total (2 new ones)
    kv_blocks_second = _alloc_and_register(fix, req, num_blocks)
    # update_state_after_alloc is idempotent for store registration
    sched.update_state_after_alloc(req, kv_blocks_second, num_external_tokens=0)

    # Still exactly 1 entry
    assert list(sched._reqs_to_store.keys()).count(req.request_id) == 1

    # The second chunk's NEW block IDs (positions 2,3) are passed as
    # cached_req_new_blocks. The full block_ids include both old and new,
    # but yield_req_data only appends the new_block_ids for cached reqs.
    ids_second_full = kv_blocks_second.get_block_ids()
    # New blocks are those beyond the first chunk
    new_block_ids = tuple(ids_second_full[g][2:] for g in range(len(ids_second_full)))
    sched_out2 = make_scheduler_output(
        {req.request_id: 2 * BLOCK_SIZE},
        cached_req_new_blocks={req.request_id: new_block_ids},
    )
    meta2 = sched.build_connector_meta(sched_out2)
    assert meta2.store_event >= 0
    # Only the 2 NEW blocks should be stored (first 2 already done)
    assert len(meta2.store_gpu_blocks) == 2


# ---------------------------------------------------------------------------
# Test 10: Partial GPU prefix hit + CPU load + new compute blocks
# ---------------------------------------------------------------------------
def test_partial_gpu_prefix_plus_cpu_load() -> None:
    """When GPU has a prefix cache hit for the first N blocks, CPU has a
    hit for the next M blocks, and there are P new blocks needing fresh
    compute, the block layout is:

        | comp (N) | ext_comp (M) | new (P) |

    External blocks sit in the middle — not at the beginning or end.
    The load path must target hashes at positions [N, N+M).

    Request: 6 blocks (0..5).
    - Store all 6 to CPU.
    - New request: GPU prefix cache hits blocks 0,1 (hashed).
      CPU hits blocks 2,3. Blocks 4,5 are new (need compute).
    - update_state_after_alloc receives 6 GPU blocks:
      [0,1] hashed (comp), [2,3] unhashed (ext_comp), [4,5] unhashed (new).
    - Load must target hash positions 2,3.
    """
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool

    num_blocks = 6
    req = make_request(num_blocks=num_blocks)

    # Store all 6 blocks to CPU via eager store.
    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: block_ids},
    )
    meta = sched.build_connector_meta(sched_out)
    assert meta.store_event >= 0
    simulate_store_completion(sched, meta.store_event)

    # New request with same tokens — but only partial GPU prefix hit.
    req2 = Request(
        request_id="req-partial-gpu",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )

    # GPU prefix cache hits the first 2 blocks.
    gpu_local_computed = 2 * BLOCK_SIZE
    hit_tokens, is_async = sched.get_num_new_matched_tokens(
        req2, num_computed_tokens=gpu_local_computed
    )
    # CPU has all 6 blocks stored. make_request pads num_tokens by +1, so
    # the manager's num_tokens - 1 cap leaves all full blocks intact:
    # remaining hashable range = 6 - 2 = 4 blocks, all hit.
    num_cpu_hit_blocks = 4
    assert hit_tokens == num_cpu_hit_blocks * BLOCK_SIZE, (
        f"Expected {num_cpu_hit_blocks * BLOCK_SIZE} CPU hit tokens, got {hit_tokens}"
    )
    assert is_async is True

    # Simulate what the real scheduler does: only accept 2 of the 4 CPU hit
    # blocks as external (e.g. due to budget constraints), leaving 2 new
    # blocks for fresh compute.
    num_ext_blocks = 2
    num_new_blocks = 2
    external_tokens = num_ext_blocks * BLOCK_SIZE

    # Build block list matching real layout: | comp(2) | ext_comp(2) | new(2) |
    # comp: GPU prefix cache hit — blocks with hashes
    gpu_comp = _allocate_gpu_blocks(gpu_pool, req2, 2, group_id=0)
    # ext_comp + new: freshly allocated, no hashes
    gpu_ext_and_new = gpu_pool.get_new_blocks(num_ext_blocks + num_new_blocks)
    all_gpu_blocks = gpu_comp + gpu_ext_and_new
    kv_blocks2 = KVCacheBlocks(blocks=(all_gpu_blocks,))

    # Critical call: with 2 hashed comp blocks and 2 external tokens worth
    # of blocks, the manager must derive skipped=2 and load hashes [2,3].
    sched.update_state_after_alloc(
        req2, kv_blocks2, num_external_tokens=external_tokens
    )

    block_ids2 = kv_blocks2.get_block_ids()
    sched_out2 = make_scheduler_output(
        {req2.request_id: num_new_blocks * BLOCK_SIZE},
        new_reqs={req2.request_id: block_ids2},
    )
    meta2 = sched.build_connector_meta(sched_out2)
    assert meta2.load_event >= 0, "Expected a load event for partial GPU + CPU hit"
    assert len(meta2.load_gpu_blocks) == num_ext_blocks
    assert len(meta2.load_cpu_blocks) == num_ext_blocks

    # Verify the load targets the ext_comp GPU blocks (positions 2,3),
    # not the comp blocks (0,1) or new blocks (4,5).
    ext_block_ids = [b.block_id for b in gpu_ext_and_new[:num_ext_blocks]]
    for bid in meta2.load_gpu_blocks:
        assert bid in ext_block_ids, (
            f"Load GPU block {bid} should be an ext_comp block, not a comp or new block"
        )


# ---------------------------------------------------------------------------
# Test 11: TOCTOU between Phase A and Phase B (regression for #39702)
# ---------------------------------------------------------------------------
def test_toctou_cpu_hit_evicted_between_phases_no_crash() -> None:
    """Regression for vllm-project/vllm#39702.

    When ``get_num_new_matched_tokens`` (Phase A) reports a CPU cache hit
    of ``N`` tokens but ``update_state_after_alloc`` (Phase B) runs after
    other requests have caused LRU eviction of those exact blocks, the
    second ``find_longest_cache_hit`` call returns 0 while
    ``num_external_tokens`` is still ``N``, triggering
    ``AssertionError: Expected N hit tokens, got 0``.

    Setup: ``num_cpu_blocks=5`` (4 usable; null_block takes 1).
        1. Store req_a's 2 blocks to CPU. CPU: [a0, a1, _, _].
        2. Phase A on req_b (same prompt as req_a) reports a 2-block hit.
           Without the fix this does NOT pin a0/a1 — they remain at LRU front.
        3. Store req_c (2) + req_d (2) — 4 more blocks into 4 slots. With
           a0/a1 unpinned and at LRU front, they get evicted. CPU: [c0, c1,
           d0, d1].
        4. Phase B on req_b: re-searches the CPU coordinator, finds 0
           cached hashes, asserts.

    After the fix, Phase A pins the hit blocks so step 3 evicts req_c's
    blocks instead, and Phase B reads the cached ``(cpu_hit_blocks,
    hit_length)`` tuple without re-searching.
    """
    # 5 total = 4 usable (null_block takes 1)
    fix = make_scheduler(num_cpu_blocks=5, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler

    # --- Step 1: Store req_a's 2 blocks to CPU cache ---
    req_a = make_request(num_blocks=2)
    kv_a = _alloc_and_register(fix, req_a, 2)
    sched.update_state_after_alloc(req_a, kv_a, num_external_tokens=0)
    sched_out_a = make_scheduler_output(
        {req_a.request_id: 2 * BLOCK_SIZE},
        new_reqs={req_a.request_id: kv_a.get_block_ids()},
    )
    meta_a = sched.build_connector_meta(sched_out_a)
    assert meta_a.store_event >= 0
    simulate_store_completion(sched, meta_a.store_event)

    # --- Step 2: Phase A — req_b (same prompt as req_a) reports a CPU hit ---
    req_b = Request(
        request_id="req-b-toctou",
        prompt_token_ids=req_a.prompt_token_ids,
        sampling_params=req_a.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req_a._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(
        req_b, num_computed_tokens=0
    )
    assert hit_tokens == 2 * BLOCK_SIZE, (
        f"Phase A should report 2 blocks of CPU hit, got {hit_tokens}"
    )
    assert is_async is True

    # --- Step 3: TOCTOU window — fill CPU cache so LRU evicts req_a's blocks
    # (in production this corresponds to other concurrent requests landing
    # between Phase A and Phase B for req_b). 4 usable slots, req_a occupies 2;
    # req_c (2) + req_d (2) require evicting 2 LRU blocks. ---
    req_c = make_request(num_blocks=2)
    req_d = make_request(num_blocks=2)
    kv_c = _alloc_and_register(fix, req_c, 2)
    kv_d = _alloc_and_register(fix, req_d, 2)
    sched.update_state_after_alloc(req_c, kv_c, num_external_tokens=0)
    sched.update_state_after_alloc(req_d, kv_d, num_external_tokens=0)
    sched_out_pressure = make_scheduler_output(
        {
            req_c.request_id: 2 * BLOCK_SIZE,
            req_d.request_id: 2 * BLOCK_SIZE,
        },
        new_reqs={
            req_c.request_id: kv_c.get_block_ids(),
            req_d.request_id: kv_d.get_block_ids(),
        },
    )
    meta_pressure = sched.build_connector_meta(sched_out_pressure)
    assert meta_pressure.store_event >= 0
    simulate_store_completion(sched, meta_pressure.store_event)

    # --- Step 4: Phase B — must not crash ---
    # Before fix: AssertionError: Expected 32 hit tokens, got 0
    # After fix: Phase A pinned the hits → Step 3 evicted req_c instead,
    # Phase B consumes the cached (cpu_hit_blocks, hit_length) tuple.
    gpu_blocks_b = fix.gpu_block_pool.get_new_blocks(2)
    kv_blocks_b = KVCacheBlocks(blocks=(gpu_blocks_b,))
    sched.update_state_after_alloc(req_b, kv_blocks_b, num_external_tokens=hit_tokens)

    # --- Step 5: with the fix, the load is queued correctly ---
    sched_out_b = make_scheduler_output(
        {req_b.request_id: 1},
        new_reqs={req_b.request_id: kv_blocks_b.get_block_ids()},
    )
    meta_b = sched.build_connector_meta(sched_out_b)
    assert meta_b.load_event >= 0, (
        "Phase B should queue a load event using the pinned CPU hit blocks"
    )
    assert len(meta_b.load_gpu_blocks) == 2
    assert len(meta_b.load_cpu_blocks) == 2


# ---------------------------------------------------------------------------
# Test 12: Reset with pending eager stores waits for completion
# ---------------------------------------------------------------------------
def test_reset_pending_eager_stores() -> None:
    """Eager mode: reset() abandons in-flight stores until they complete."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool

    num_blocks = 2
    req = make_request(num_blocks=num_blocks)

    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: block_ids},
    )

    meta = sched.build_connector_meta(sched_out)
    assert meta.store_event >= 0
    assert len(sched._store_event_to_blocks) > 0

    # GPU blocks should have elevated ref_cnt from touch()
    for bid in meta.store_gpu_blocks:
        assert gpu_pool.blocks[bid].ref_cnt > 0

    # Free the request's own block refs (simulates preemption)
    gpu_pool.free_blocks(gpu_pool.blocks[bid] for bid in block_ids[0])

    # Reset should keep DMA refs pinned until the worker reports completion.
    assert sched.reset() is False
    assert len(sched._store_event_to_blocks) == 0
    assert len(sched._abandoned_store_event_to_blocks) == 1
    assert len(sched._reqs_to_store) == 0
    assert len(sched._store_event_to_reqs) == 0

    num_used = gpu_pool.num_gpu_blocks - gpu_pool.get_num_free_blocks()
    assert num_used > 1

    simulate_store_completion(sched, meta.store_event)
    assert len(sched._abandoned_store_event_to_blocks) == 0

    # All GPU blocks should now be free (ref_cnt == 0) except null block
    num_used = gpu_pool.num_gpu_blocks - gpu_pool.get_num_free_blocks()
    assert num_used == 1, f"Expected only null block in use, got {num_used}"

    # GPU prefix cache reset should now succeed
    assert gpu_pool.reset_prefix_cache() is True
    assert sched.reset() is True


# ---------------------------------------------------------------------------
# Test 13: Reset with pending lazy stores waits for completion
# ---------------------------------------------------------------------------
def test_reset_pending_lazy_stores() -> None:
    """Lazy mode: reset() abandons in-flight stores until they complete."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=8, lazy=True)
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool

    num_blocks = 2
    req = make_request(num_blocks=num_blocks)

    # Allocate, hash, and free — blocks move to free queue with hashes
    gpu_blocks = _allocate_gpu_blocks(gpu_pool, req, num_blocks, group_id=0)
    gpu_pool.free_blocks(gpu_blocks)

    # Push hashed blocks to LRU head
    fillers = _flush_old_blocks_to_lru_head(gpu_pool, num_filler_blocks=5)

    # Lazy scanner offloads old hashed blocks
    sched_out = make_scheduler_output({})
    meta = sched.build_connector_meta(sched_out)
    assert meta.store_event >= 0
    assert len(sched._store_event_to_blocks) > 0

    gpu_pool.free_blocks(fillers)

    # Reset should keep DMA refs pinned until the worker reports completion.
    assert sched.reset() is False
    assert len(sched._store_event_to_blocks) == 0
    assert len(sched._abandoned_store_event_to_blocks) == 1
    assert sched._cursor is None

    simulate_store_completion(sched, meta.store_event)
    assert len(sched._abandoned_store_event_to_blocks) == 0
    assert sched.reset() is True

    # No CPU cache hits after reset
    req2 = Request(
        request_id="req-after-lazy-reset",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    hit_tokens, _ = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    assert hit_tokens == 0, "CPU cache should be empty after reset"


@pytest.mark.parametrize("num_groups", [1, 2])
@pytest.mark.parametrize("lazy", [False, True])
def test_reset_releases_pending_cpu_hits(num_groups: int, lazy: bool) -> None:
    """A lookup awaiting GPU allocation must not prevent a cache reset."""
    fix = make_scheduler(num_groups=num_groups, lazy=lazy)
    sched = fix.scheduler
    req = make_request(num_blocks=2)
    cpu_blocks = []
    for group_id in range(num_groups):
        cpu_blocks.extend(
            _allocate_gpu_blocks(sched.cpu_block_pool, req, 2, group_id=group_id)
        )
    sched.cpu_block_pool.free_blocks(cpu_blocks)

    assert sched.get_num_new_matched_tokens(req, 0) == (2 * BLOCK_SIZE, True)
    assert all(block.ref_cnt == 1 for block in cpu_blocks)
    assert not sched._reqs_to_load

    assert sched.reset()
    assert all(block.ref_cnt == 0 for block in cpu_blocks)
    assert not sched._pending_cpu_hits
    assert sched.get_num_new_matched_tokens(req, 0) == (0, False)
    assert sched.reset()


# ---------------------------------------------------------------------------
# Test 14: Reset with pending loads waits for completion
# ---------------------------------------------------------------------------
def test_reset_pending_loads() -> None:
    """reset() abandons in-flight loads until they complete."""
    fix = make_scheduler(num_cpu_blocks=8, num_gpu_blocks=16, lazy=False)
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool

    num_blocks = 2

    # First store blocks to CPU
    req = make_request(num_blocks=num_blocks)
    kv_blocks = _alloc_and_register(fix, req, num_blocks)
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: block_ids},
    )
    meta = sched.build_connector_meta(sched_out)
    simulate_store_completion(sched, meta.store_event)

    # Start a load — CPU cache hit
    req2 = Request(
        request_id="req-load-reset",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    assert hit_tokens is not None and hit_tokens > 0

    gpu_blocks2 = gpu_pool.get_new_blocks(num_blocks)
    kv_blocks2 = KVCacheBlocks(blocks=(gpu_blocks2,))
    sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=hit_tokens)

    block_ids2 = kv_blocks2.get_block_ids()
    sched_out2 = make_scheduler_output(
        {req2.request_id: 1},
        new_reqs={req2.request_id: block_ids2},
    )
    meta2 = sched.build_connector_meta(sched_out2)
    assert meta2.load_event >= 0
    assert req2.request_id in sched._reqs_to_load

    # Free request block refs (simulates preemption)
    gpu_pool.free_blocks(gpu_pool.blocks[bid] for bid in block_ids[0])
    gpu_pool.free_blocks(gpu_pool.blocks[bid] for bid in block_ids2[0])

    # Reset should keep load touch refs until the worker reports completion.
    assert sched.reset() is False
    assert len(sched._reqs_to_load) == 0
    assert len(sched._abandoned_reqs_to_load) == 1
    assert len(sched._load_event_to_reqs) == 1

    num_used = gpu_pool.num_gpu_blocks - gpu_pool.get_num_free_blocks()
    assert num_used > 1

    simulate_load_completion(sched, {req2.request_id})
    assert len(sched._abandoned_reqs_to_load) == 0
    assert len(sched._load_event_to_reqs) == 0
    assert sched.reset() is True

    # All GPU blocks free
    num_used = gpu_pool.num_gpu_blocks - gpu_pool.get_num_free_blocks()
    assert num_used == 1, f"Expected only null block in use, got {num_used}"


def _make_cp_vllm_config(
    dcp_world_size: int = 1,
    pcp_world_size: int = 1,
) -> VllmConfig:
    """VllmConfig with context-parallel sizes set for scheduler-only tests."""
    cfg = _make_vllm_config()

    cfg.parallel_config.decode_context_parallel_size = dcp_world_size
    cfg.parallel_config.prefill_context_parallel_size = pcp_world_size
    return cfg


def _make_cp_scheduler(
    *,
    dcp_world_size: int = 1,
    pcp_world_size: int = 1,
    num_cpu_blocks: int = 8,
    num_gpu_blocks: int = 16,
    lazy: bool = False,
) -> SchedulerFixture:
    """Build a SimpleCPUOffloadScheduler with DCP-scaled block size."""
    virtual_block_size = BLOCK_SIZE * dcp_world_size

    kv_cache_config = _make_kv_cache_config(num_gpu_blocks)
    vllm_config = _make_cp_vllm_config(dcp_world_size, pcp_world_size)
    cpu_capacity_bytes = _BYTES_PER_BLOCK * num_cpu_blocks

    sched = SimpleCPUOffloadScheduler(
        vllm_config=vllm_config,
        kv_cache_config=kv_cache_config,
        cpu_capacity_bytes=cpu_capacity_bytes,
        scheduler_block_size=virtual_block_size,
        hash_block_size=virtual_block_size,
        lazy_offload=lazy,
    )

    gpu_block_pool = BlockPool(
        num_gpu_blocks=num_gpu_blocks,
        enable_caching=True,
        hash_block_size=virtual_block_size,
    )
    sched.bind_gpu_block_pool(gpu_block_pool)

    return SchedulerFixture(
        scheduler=sched,
        gpu_block_pool=gpu_block_pool,
        vllm_config=vllm_config,
        kv_cache_config=kv_cache_config,
    )


def _make_cp_request(
    num_blocks: int,
    virtual_block_size: int,
    request_id: str | None = None,
) -> Request:
    """Create a request whose block hashes are computed at the virtual
    (CP-scaled) block size, matching what the real scheduler does.
    """
    global _req_counter
    _req_counter += 1
    if request_id is None:
        request_id = f"req-cp-{_req_counter}"

    num_tokens = num_blocks * virtual_block_size + 1
    start = _req_counter * 10000
    prompt_token_ids = list(range(start, start + num_tokens))
    sampling_params = SamplingParams(max_tokens=1)

    return Request(
        request_id=request_id,
        prompt_token_ids=prompt_token_ids,
        sampling_params=sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=get_request_block_hasher(virtual_block_size, sha256),
    )


def _allocate_cp_gpu_blocks(
    gpu_block_pool: BlockPool,
    request: Request,
    num_blocks: int,
    virtual_block_size: int,
    group_id: int = 0,
) -> list:
    """Allocate GPU blocks and cache them using the CP-scaled block size."""
    blocks = gpu_block_pool.get_new_blocks(num_blocks)
    num_full = min(num_blocks, len(request.block_hashes))
    if num_full > 0:
        gpu_block_pool.cache_full_blocks(
            request=request,
            blocks=blocks,
            num_cached_blocks=0,
            num_full_blocks=num_full,
            block_size=virtual_block_size,
            kv_cache_group_id=group_id,
        )
    return blocks


# ---------------------------------------------------------------------------
# Test 15: CP block size scaling is correct
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
    "dcp_world_size, pcp_world_size",
    [
        (2, 1),  # DCP only
        (1, 2),  # PCP only
        (2, 2),  # DCP + PCP
    ],
)
def test_cp_block_size_scaling(dcp_world_size: int, pcp_world_size: int) -> None:
    """Verify block size scaling follows DCP ownership."""
    fix = _make_cp_scheduler(
        dcp_world_size=dcp_world_size, pcp_world_size=pcp_world_size
    )
    sched = fix.scheduler

    assert sched.cp_world_size == dcp_world_size
    assert sched.block_size == BLOCK_SIZE * dcp_world_size


# ---------------------------------------------------------------------------
# Test 16: CP eager store-and-load roundtrip
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
    "dcp_world_size, pcp_world_size",
    [
        (2, 1),
        (1, 2),
    ],
)
def test_cp_eager_store_and_load_roundtrip(
    dcp_world_size: int, pcp_world_size: int
) -> None:
    """With CP enabled, store blocks to CPU and reload them for a new request
    with matching tokens.  Verifies that hash matching and transfer-pair
    construction work with the virtual block size."""
    fix = _make_cp_scheduler(
        dcp_world_size=dcp_world_size,
        pcp_world_size=pcp_world_size,
        num_cpu_blocks=8,
        num_gpu_blocks=16,
        lazy=False,
    )
    sched = fix.scheduler
    vbs = BLOCK_SIZE * dcp_world_size

    num_blocks = 2
    req = _make_cp_request(num_blocks, vbs)

    # Allocate GPU blocks and register hashes
    gpu_blocks = _allocate_cp_gpu_blocks(fix.gpu_block_pool, req, num_blocks, vbs)
    kv_blocks = KVCacheBlocks(blocks=(gpu_blocks,))
    req.num_computed_tokens = num_blocks * vbs
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)

    block_ids = kv_blocks.get_block_ids()
    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * vbs},
        new_reqs={req.request_id: block_ids},
    )

    meta = sched.build_connector_meta(sched_out)
    assert meta.store_event >= 0, "Expected a store event"
    assert len(meta.store_gpu_blocks) == num_blocks
    assert len(meta.store_cpu_blocks) == num_blocks
    simulate_store_completion(sched, meta.store_event)

    # New request with same tokens — should get a full CPU cache hit.
    req2 = Request(
        request_id="req-cp-load",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )

    hit_tokens, is_async = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    assert hit_tokens == num_blocks * vbs
    assert is_async is True

    # Allocate fresh GPU blocks for the load.
    gpu_blocks2 = fix.gpu_block_pool.get_new_blocks(num_blocks)
    kv_blocks2 = KVCacheBlocks(blocks=(gpu_blocks2,))
    sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=hit_tokens)

    sched_out2 = make_scheduler_output(
        {req2.request_id: 1},
        new_reqs={req2.request_id: kv_blocks2.get_block_ids()},
    )
    meta2 = sched.build_connector_meta(sched_out2)
    assert meta2.load_event >= 0, "Expected a load event"
    assert len(meta2.load_gpu_blocks) == num_blocks
    assert len(meta2.load_cpu_blocks) == num_blocks


# ---------------------------------------------------------------------------
# Test 18: CP store and load use effective block size
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
    "dcp_world_size, pcp_world_size",
    [
        (2, 1),
        (1, 2),
        (2, 2),
    ],
)
def test_cp_effective_block_size_store_and_load(
    dcp_world_size: int, pcp_world_size: int
) -> None:
    """Verify store/load physical block counts scale with DCP, not PCP."""
    fix = _make_cp_scheduler(
        dcp_world_size=dcp_world_size, pcp_world_size=pcp_world_size
    )
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool
    vbs = BLOCK_SIZE * dcp_world_size
    expected_blocks = 1

    # Store one DCP-scaled logical block worth of tokens.
    req = _make_cp_request(num_blocks=2, virtual_block_size=vbs)
    gpu_blocks = _allocate_cp_gpu_blocks(gpu_pool, req, 2, vbs)
    kv = KVCacheBlocks(blocks=(gpu_blocks,))
    req.num_computed_tokens = vbs
    sched.update_state_after_alloc(req, kv, num_external_tokens=0)
    m1 = sched.build_connector_meta(
        make_scheduler_output(
            {req.request_id: vbs},
            new_reqs={req.request_id: kv.get_block_ids()},
        )
    )
    assert len(m1.store_gpu_blocks) == expected_blocks
    assert len(m1.store_cpu_blocks) == expected_blocks
    simulate_store_completion(sched, m1.store_event)

    # Load one DCP-scaled logical block from a two-block CPU hit.
    req2 = _make_cp_request(num_blocks=2, virtual_block_size=vbs)
    kv2 = KVCacheBlocks(blocks=(_allocate_cp_gpu_blocks(gpu_pool, req2, 2, vbs),))
    req2.num_computed_tokens = 2 * vbs
    sched.update_state_after_alloc(req2, kv2, num_external_tokens=0)
    m2 = sched.build_connector_meta(
        make_scheduler_output(
            {req2.request_id: 2 * vbs},
            new_reqs={req2.request_id: kv2.get_block_ids()},
        )
    )
    simulate_store_completion(sched, m2.store_event)

    req3 = Request(
        request_id="req-cp-partial-load",
        prompt_token_ids=req2.prompt_token_ids,
        sampling_params=req2.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req2._block_hasher,
    )
    hit, _ = sched.get_num_new_matched_tokens(req3, num_computed_tokens=0)
    assert hit == 2 * vbs

    kv3 = KVCacheBlocks(blocks=(gpu_pool.get_new_blocks(expected_blocks),))
    sched.update_state_after_alloc(req3, kv3, num_external_tokens=vbs)
    m3 = sched.build_connector_meta(
        make_scheduler_output(
            {req3.request_id: vbs},
            new_reqs={req3.request_id: kv3.get_block_ids()},
        )
    )
    assert m3.load_event >= 0
    assert len(m3.load_gpu_blocks) == expected_blocks
    assert len(m3.load_cpu_blocks) == expected_blocks
    assert m3.load_gpu_blocks == [kv3.get_block_ids()[0][0]]


# ---------------------------------------------------------------------------
# Test 17: CP lazy target blocks are scaled correctly
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("cp_world_size", [1, 2, 4])
def test_cp_lazy_target_blocks_scaling(cp_world_size: int) -> None:
    """_estimate_lazy_target_blocks returns fewer blocks when cp_world_size > 1
    because each virtual block covers more tokens."""
    kv_cache_config = _make_kv_cache_config(num_blocks=16)
    max_batched = 64

    target_base = SimpleCPUOffloadScheduler._estimate_lazy_target_blocks(
        kv_cache_config, max_batched, cp_world_size=1
    )
    target_cp = SimpleCPUOffloadScheduler._estimate_lazy_target_blocks(
        kv_cache_config, max_batched, cp_world_size=cp_world_size
    )

    if cp_world_size == 1:
        assert target_cp == target_base
    else:
        assert target_cp < target_base, (
            f"cp_world_size={cp_world_size}: target_cp={target_cp} should be "
            f"less than target_base={target_base}"
        )


def _make_hybrid_attention_mamba_scheduler(
    *,
    num_cpu_blocks: int = 8,
    num_gpu_blocks: int = 16,
    attention_block_size: int = BLOCK_SIZE,
    block_size: int = 4 * BLOCK_SIZE,
    scheduler_block_size: int | None = None,
    hash_block_size: int | None = None,
    dcp_world_size: int = 4,
    lazy: bool = False,
    mamba_cache_mode: MambaCacheMode = "align",
    enable_kv_cache_events: bool = False,
) -> SchedulerFixture:
    """Build a scheduler for one attention group plus one Mamba group."""
    scheduler_block_size = scheduler_block_size or block_size
    hash_block_size = hash_block_size or block_size
    attention_spec = FullAttentionSpec(
        block_size=attention_block_size,
        num_kv_heads=NUM_KV_HEADS,
        head_size=HEAD_SIZE,
        dtype=DTYPE,
    )
    mamba_spec = MambaSpec(
        block_size=block_size,
        shapes=((1, 1),),
        dtypes=(torch.float32,),
        mamba_cache_mode=mamba_cache_mode,
    )
    groups = [
        KVCacheGroupSpec(["attention"], attention_spec),
        KVCacheGroupSpec(["mamba"], mamba_spec),
    ]
    tensors = [
        KVCacheTensor(
            size=spec.page_size_bytes * num_gpu_blocks,
            layers=group.layer_names,
            layer_stride=spec.page_size_bytes * num_gpu_blocks,
            block_stride=spec.page_size_bytes,
        )
        for group, spec in zip(groups, (attention_spec, mamba_spec))
    ]
    kv_cache_config = KVCacheConfig(
        num_blocks=num_gpu_blocks,
        kv_cache_tensors=tensors,
        kv_cache_groups=groups,
    )
    vllm_config = _make_cp_vllm_config(dcp_world_size=dcp_world_size)
    vllm_config.cache_config.prefix_cache_retention_interval = 0
    vllm_config.cache_config.mamba_cache_mode = mamba_cache_mode
    if enable_kv_cache_events:
        vllm_config.kv_events_config = KVEventsConfig(enable_kv_cache_events=True)
    # _derive_cpu_config() scales the requested capacity against
    # kv_cache_tensors[0].size, so express it as a multiple of that tensor's
    # per-block size to get exactly num_cpu_blocks.
    cpu_capacity_bytes = tensors[0].size // num_gpu_blocks * num_cpu_blocks
    sched = SimpleCPUOffloadScheduler(
        vllm_config=vllm_config,
        kv_cache_config=kv_cache_config,
        cpu_capacity_bytes=cpu_capacity_bytes,
        scheduler_block_size=scheduler_block_size,
        hash_block_size=hash_block_size,
        lazy_offload=lazy,
    )
    gpu_block_pool = BlockPool(
        num_gpu_blocks=num_gpu_blocks,
        enable_caching=True,
        hash_block_size=hash_block_size,
    )
    sched.bind_gpu_block_pool(gpu_block_pool)
    return SchedulerFixture(
        scheduler=sched,
        gpu_block_pool=gpu_block_pool,
        vllm_config=vllm_config,
        kv_cache_config=kv_cache_config,
    )


def test_hybrid_store_uses_resolved_group_block_sizes() -> None:
    """Replicated groups must not be scaled by the DCP world size.

    The spec's ``dcp_sharded`` flag controls its effective block size.
    Reconstructing a group's geometry as
    ``spec.block_size * cp_world_size`` over-scales the replicated groups, so
    the eager store scan believes far fewer of their blocks are ready and
    silently offloads only a fraction of them.
    """
    attention_block_size = BLOCK_SIZE
    mamba_block_size = 4 * BLOCK_SIZE
    dcp_world_size = 2
    fix = _make_hybrid_attention_mamba_scheduler(
        num_cpu_blocks=32,
        num_gpu_blocks=32,
        attention_block_size=attention_block_size,
        block_size=mamba_block_size,
        # Every prefix-cacheable group's resolved block size must be a multiple
        # of the hash block, so it cannot exceed the smallest of them.
        hash_block_size=BLOCK_SIZE,
        dcp_world_size=dcp_world_size,
        mamba_cache_mode="none",
    )
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool
    attention_size, mamba_size = sched.group_block_sizes
    assert attention_size == attention_block_size * dcp_world_size
    assert mamba_size == mamba_block_size

    confirmed = 2 * sched.block_size
    req = _make_cp_request(num_blocks=8, virtual_block_size=BLOCK_SIZE)
    attention_blocks = _allocate_cp_gpu_blocks(
        gpu_pool, req, confirmed // attention_size, attention_size, group_id=0
    )
    mamba_blocks = _allocate_cp_gpu_blocks(
        gpu_pool, req, confirmed // mamba_size, mamba_size, group_id=1
    )
    kv_blocks = KVCacheBlocks(blocks=(attention_blocks, mamba_blocks))
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    sched.build_connector_meta(
        make_scheduler_output(
            {req.request_id: confirmed},
            new_reqs={req.request_id: kv_blocks.get_block_ids()},
        )
    )
    req.num_computed_tokens = confirmed
    meta = sched.build_connector_meta(
        make_scheduler_output(
            {req.request_id: 1},
            cached_req_new_blocks={req.request_id: None},
        )
    )

    # Both groups are positionally stable here, so every confirmed block of
    # both must be offloaded. Over-scaling the replicated group halves its
    # count.
    assert sched._reqs_to_store[req.request_id].num_stored_blocks == [
        confirmed // attention_size,
        confirmed // mamba_size,
    ]
    assert set(meta.store_gpu_blocks) == {
        *(b.block_id for b in attention_blocks),
        *(b.block_id for b in mamba_blocks),
    }


def test_dcp_mixed_cache_loads_fine_grained_external_hit() -> None:
    """Map a fine-grained external suffix after a nonzero local prefix."""
    hash_block_size = BLOCK_SIZE
    mamba_block_size = 4 * BLOCK_SIZE
    local_tokens = 4 * BLOCK_SIZE
    # Span two Mamba blocks and end partially in the second one. This exercises
    # both cdiv(external_tokens, mamba_block_size) > 1 and the partial tail.
    external_tokens = 6 * BLOCK_SIZE
    total_cached_tokens = local_tokens + external_tokens
    fix = _make_hybrid_attention_mamba_scheduler(
        num_cpu_blocks=32,
        num_gpu_blocks=32,
        block_size=mamba_block_size,
        hash_block_size=hash_block_size,
        dcp_world_size=2,
    )
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool

    assert external_tokens % mamba_block_size != 0
    assert sched.cpu_coordinator.enable_partial_hash_hits

    producer = _make_cp_request(num_blocks=10, virtual_block_size=hash_block_size)
    attention_blocks = _allocate_cp_gpu_blocks(
        gpu_pool,
        producer,
        num_blocks=5,
        virtual_block_size=2 * BLOCK_SIZE,
        group_id=0,
    )
    mamba_blocks = gpu_pool.get_new_blocks(3)
    gpu_pool.cache_full_blocks(
        request=producer,
        blocks=mamba_blocks,
        num_cached_blocks=0,
        num_full_blocks=2,
        block_size=mamba_block_size,
        kv_cache_group_id=1,
    )
    gpu_pool.cache_partial_block(
        request=producer,
        block=mamba_blocks[2],
        num_tokens=total_cached_tokens,
        kv_cache_group_id=1,
        block_size=mamba_block_size,
    )
    producer_blocks = KVCacheBlocks(blocks=(attention_blocks, mamba_blocks))
    sched.update_state_after_alloc(producer, producer_blocks, num_external_tokens=0)

    # Positional scanning stores append-only FA blocks. Mamba align blocks use
    # exact handoffs for every retained boundary, including the partial tail.
    store_output = make_scheduler_output(
        {producer.request_id: total_cached_tokens},
        new_reqs={producer.request_id: producer_blocks.get_block_ids()},
    )
    store_output.kv_connector_block_state = KVConnectorBlockState(
        req_ids=set(),
        resolve_block_ids=lambda _: (),
        boundary_state_offloads={
            producer.request_id: [
                (1, mamba_blocks[0].block_id, mamba_block_size),
                (1, mamba_blocks[1].block_id, 2 * mamba_block_size),
                (1, mamba_blocks[2].block_id, total_cached_tokens),
            ]
        },
    )
    store_meta = sched.build_connector_meta(store_output)
    # The exact handoffs land in the step they arrive; the append-only FA scan
    # picks its blocks up once the scheduler has committed the tokens.
    assert set(store_meta.store_gpu_blocks) == {
        block.block_id for block in mamba_blocks
    }
    simulate_store_completion(sched, store_meta.store_event)
    producer.num_computed_tokens = total_cached_tokens
    fa_meta = sched.build_connector_meta(
        make_scheduler_output(
            {producer.request_id: 1},
            cached_req_new_blocks={producer.request_id: None},
        )
    )
    assert set(fa_meta.store_gpu_blocks) == {
        block.block_id for block in attention_blocks
    }
    simulate_store_completion(sched, fa_meta.store_event)

    consumer = Request(
        request_id="req-dcp-mixed-fine-load",
        prompt_token_ids=producer.prompt_token_ids,
        sampling_params=producer.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=producer._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(
        consumer, num_computed_tokens=local_tokens
    )
    assert hit_tokens == external_tokens
    assert is_async is True
    pending_blocks, pending_tokens, _ = sched._pending_cpu_hits[consumer.request_id]
    assert pending_tokens == external_tokens
    assert len(pending_blocks[1]) == 2
    assert pending_blocks[1][0].is_null
    assert not pending_blocks[1][1].is_null

    local_attention = attention_blocks[:2]
    local_mamba = mamba_blocks[:1]
    load_blocks = KVCacheBlocks(
        blocks=(
            local_attention + gpu_pool.get_new_blocks(3),
            local_mamba + gpu_pool.get_new_blocks(2),
        )
    )
    sched.update_state_after_alloc(
        consumer,
        load_blocks,
        num_external_tokens=hit_tokens,
    )
    load_meta = sched.build_connector_meta(
        make_scheduler_output(
            {consumer.request_id: 1},
            new_reqs={consumer.request_id: load_blocks.get_block_ids()},
        )
    )
    assert load_meta.load_gpu_blocks == [
        *(block.block_id for block in load_blocks.blocks[0][2:5]),
        load_blocks.blocks[1][2].block_id,
    ]
    assert len(load_meta.load_cpu_blocks) == 4


def test_finished_eager_store_caches_fa_partial_tail_for_hybrid_hit() -> None:
    """A finished non-LCM prompt needs the FA partial tail in CPU too."""
    hash_block_size = BLOCK_SIZE
    mamba_block_size = BLOCK_SIZE
    attention_block_size = 4 * BLOCK_SIZE
    dcp_world_size = 2
    scheduler_block_size = attention_block_size * dcp_world_size
    total_cached_tokens = 5 * hash_block_size
    fix = _make_hybrid_attention_mamba_scheduler(
        num_cpu_blocks=32,
        num_gpu_blocks=32,
        attention_block_size=attention_block_size,
        block_size=mamba_block_size,
        scheduler_block_size=scheduler_block_size,
        hash_block_size=hash_block_size,
        dcp_world_size=dcp_world_size,
    )
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool
    assert sched.fa_block_size == scheduler_block_size
    assert total_cached_tokens % sched.fa_block_size != 0

    producer = _make_cp_request(num_blocks=5, virtual_block_size=hash_block_size)
    attention_blocks = gpu_pool.get_new_blocks(1)
    gpu_pool.cache_partial_block(
        request=producer,
        block=attention_blocks[0],
        num_tokens=total_cached_tokens,
        kv_cache_group_id=0,
        block_size=scheduler_block_size,
    )
    mamba_blocks = gpu_pool.get_new_blocks(5)
    gpu_pool.cache_full_blocks(
        request=producer,
        blocks=mamba_blocks,
        num_cached_blocks=0,
        num_full_blocks=5,
        kv_cache_group_id=1,
        block_size=mamba_block_size,
    )
    producer_blocks = KVCacheBlocks(blocks=(attention_blocks, mamba_blocks))
    sched.update_state_after_alloc(producer, producer_blocks, num_external_tokens=0)

    store_output = make_scheduler_output(
        {producer.request_id: total_cached_tokens},
        new_reqs={producer.request_id: producer_blocks.get_block_ids()},
    )
    store_output.kv_connector_block_state = KVConnectorBlockState(
        req_ids=set(),
        resolve_block_ids=lambda _: (),
        boundary_state_offloads={
            producer.request_id: [
                (1, mamba_blocks[4].block_id, total_cached_tokens),
            ]
        },
    )
    store_meta = sched.build_connector_meta(store_output)
    simulate_store_completion(sched, store_meta.store_event)

    producer.num_computed_tokens = total_cached_tokens
    sched.request_finished_all_groups(producer, producer_blocks.get_block_ids())
    finish_meta = sched.build_connector_meta(make_scheduler_output({}))
    assert attention_blocks[0].block_id in finish_meta.store_gpu_blocks
    simulate_store_completion(sched, finish_meta.store_event)

    consumer = Request(
        request_id="req-finished-partial-tail-load",
        prompt_token_ids=producer.prompt_token_ids,
        sampling_params=producer.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=producer._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(
        consumer, num_computed_tokens=0
    )
    assert hit_tokens == total_cached_tokens
    assert is_async is True


def test_finished_eager_store_does_not_duplicate_completed_partial_tail() -> None:
    """Decode can fill the boundary block; the scan then already covers it."""
    hash_block_size = BLOCK_SIZE
    mamba_block_size = BLOCK_SIZE
    attention_block_size = 4 * BLOCK_SIZE
    dcp_world_size = 2
    scheduler_block_size = attention_block_size * dcp_world_size
    prompt_tokens = 5 * hash_block_size
    fix = _make_hybrid_attention_mamba_scheduler(
        num_cpu_blocks=32,
        num_gpu_blocks=32,
        attention_block_size=attention_block_size,
        block_size=mamba_block_size,
        scheduler_block_size=scheduler_block_size,
        hash_block_size=hash_block_size,
        dcp_world_size=dcp_world_size,
    )
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool
    assert prompt_tokens % sched.fa_block_size != 0

    producer = _make_cp_request(num_blocks=5, virtual_block_size=hash_block_size)
    attention_blocks = gpu_pool.get_new_blocks(1)
    gpu_pool.cache_partial_block(
        request=producer,
        block=attention_blocks[0],
        num_tokens=prompt_tokens,
        kv_cache_group_id=0,
        block_size=scheduler_block_size,
    )
    mamba_blocks = gpu_pool.get_new_blocks(5)
    gpu_pool.cache_full_blocks(
        request=producer,
        blocks=mamba_blocks,
        num_cached_blocks=0,
        num_full_blocks=5,
        kv_cache_group_id=1,
        block_size=mamba_block_size,
    )
    producer_blocks = KVCacheBlocks(blocks=(attention_blocks, mamba_blocks))
    sched.update_state_after_alloc(producer, producer_blocks, num_external_tokens=0)

    # Decode carries the request past the physical attention block boundary, so
    # the positional scan now considers the same block the partial tail names.
    producer.num_computed_tokens = sched.fa_block_size
    sched.request_finished_all_groups(producer, producer_blocks.get_block_ids())
    finish_meta = sched.build_connector_meta(make_scheduler_output({}))

    stored = finish_meta.store_gpu_blocks
    assert attention_blocks[0].block_id in stored
    assert len(stored) == len(set(stored))
    assert len(finish_meta.store_cpu_blocks) == len(stored)


def test_external_lookup_rejects_misaligned_local_prefix(caplog_vllm) -> None:
    scheduler_block_size = 4 * BLOCK_SIZE
    fix = _make_hybrid_attention_mamba_scheduler(block_size=scheduler_block_size)

    request = _make_cp_request(num_blocks=2, virtual_block_size=scheduler_block_size)
    # The ``vllm`` logger sets propagate=False, so plain ``caplog`` can observe
    # nothing; ``caplog_vllm`` re-enables propagation for the duration.
    with caplog_vllm.at_level(logging.WARNING):
        hit_tokens, is_async = fix.scheduler.get_num_new_matched_tokens(
            request, num_computed_tokens=BLOCK_SIZE
        )

    assert (hit_tokens, is_async) == (0, False)
    assert "requires scheduler-block-aligned local tokens" in caplog_vllm.text


def test_fine_grained_external_hits_require_eager_offload() -> None:
    hash_block_size = BLOCK_SIZE
    scheduler_block_size = 4 * BLOCK_SIZE

    eager = _make_hybrid_attention_mamba_scheduler(
        block_size=scheduler_block_size,
        hash_block_size=hash_block_size,
    )
    lazy = _make_hybrid_attention_mamba_scheduler(
        block_size=scheduler_block_size,
        hash_block_size=hash_block_size,
        lazy=True,
    )

    assert eager.scheduler.cpu_coordinator.enable_partial_hash_hits
    assert not lazy.scheduler.cpu_coordinator.enable_partial_hash_hits


def test_eager_store_does_not_scan_mamba_blocks_positionally() -> None:
    """Mamba stores must use exact handoffs, not historical block positions."""
    block_size = 4 * BLOCK_SIZE
    fix = _make_hybrid_attention_mamba_scheduler(
        num_cpu_blocks=16, num_gpu_blocks=24, block_size=block_size
    )
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool
    attention_manager, mamba_manager = sched.cpu_coordinator.single_type_managers
    assert attention_manager.has_positionally_stable_blocks
    assert not mamba_manager.has_positionally_stable_blocks

    req = _make_cp_request(num_blocks=2, virtual_block_size=block_size)
    attn_blocks = _allocate_cp_gpu_blocks(
        gpu_pool, req, 2, virtual_block_size=block_size, group_id=0
    )
    gpu_blocks = gpu_pool.get_new_blocks(2)
    gpu_blocks[1].set_block_hash(make_block_hash_with_group_id(req.block_hashes[1], 1))
    kv_blocks = KVCacheBlocks(blocks=(attn_blocks, gpu_blocks))
    req.num_computed_tokens = 0
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)

    output = make_scheduler_output(
        {req.request_id: 2 * block_size},
        new_reqs={req.request_id: kv_blocks.get_block_ids()},
    )
    output.kv_connector_block_state = KVConnectorBlockState(
        req_ids=set(),
        resolve_block_ids=lambda _: (),
        boundary_state_offloads={
            req.request_id: [(1, gpu_blocks[1].block_id, 2 * block_size)]
        },
    )
    meta1 = sched.build_connector_meta(output)
    assert meta1.store_event >= 0
    # Only the exact handoff lands this step. build_connector_meta runs before
    # the scheduler commits this step's tokens, so the positional scan has
    # nothing confirmed yet.
    assert set(meta1.store_gpu_blocks) == {gpu_blocks[1].block_id}
    assert sched._reqs_to_store[req.request_id].num_stored_blocks == [0, 0]
    simulate_store_completion(sched, meta1.store_event)

    # Next step the attention blocks are confirmed and scanned positionally. A
    # later hash update still must not make the mamba align snapshot a valid
    # source; the explicit boundary-handoff tests cover the supported path.
    gpu_blocks[0].set_block_hash(make_block_hash_with_group_id(req.block_hashes[0], 1))
    req.num_computed_tokens = 2 * block_size
    meta2 = sched.build_connector_meta(
        make_scheduler_output(
            {req.request_id: 1},
            cached_req_new_blocks={req.request_id: None},
        )
    )
    assert set(meta2.store_gpu_blocks) == {block.block_id for block in attn_blocks}
    assert sched._reqs_to_store[req.request_id].num_stored_blocks == [2, 0]
    simulate_store_completion(sched, meta2.store_event)

    # Nothing further: mamba align is never picked up positionally.
    meta3 = sched.build_connector_meta(
        make_scheduler_output(
            {req.request_id: 1},
            cached_req_new_blocks={req.request_id: None},
        )
    )
    assert meta3.store_event == -1
    assert meta3.store_gpu_blocks == []
    assert sched._reqs_to_store[req.request_id].num_stored_blocks == [2, 0]


@pytest.mark.parametrize("gone_via", ["preempted", "finished", "unregistered"])
def test_boundary_handoff_dropped_for_departing_request(gone_via: str) -> None:
    """Handoffs for a request leaving this step must not be read.

    The scheduler drains boundary handoffs without filtering by request
    liveness, so a preempted or finished request can still offer block IDs.
    Those blocks are being released and may already back another request, so
    copying them would publish unrelated KV under a valid hash.
    """
    block_size = 4 * BLOCK_SIZE
    fix = _make_hybrid_attention_mamba_scheduler(
        num_cpu_blocks=16, num_gpu_blocks=24, block_size=block_size
    )
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool

    req = _make_cp_request(num_blocks=2, virtual_block_size=block_size)
    attn_blocks = _allocate_cp_gpu_blocks(
        gpu_pool, req, 2, virtual_block_size=block_size, group_id=0
    )
    mamba_blocks = gpu_pool.get_new_blocks(2)
    mamba_blocks[1].set_block_hash(
        make_block_hash_with_group_id(req.block_hashes[1], 1)
    )
    kv_blocks = KVCacheBlocks(blocks=(attn_blocks, mamba_blocks))
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)

    output = make_scheduler_output(
        {req.request_id: 2 * block_size},
        new_reqs={req.request_id: kv_blocks.get_block_ids()},
    )
    output.kv_connector_block_state = KVConnectorBlockState(
        req_ids=set(),
        resolve_block_ids=lambda _: (),
        boundary_state_offloads={
            req.request_id: [(1, mamba_blocks[1].block_id, 2 * block_size)]
        },
    )
    if gone_via == "preempted":
        output.preempted_req_ids = {req.request_id}
    elif gone_via == "finished":
        output.finished_req_ids = {req.request_id}
    else:
        sched._reqs_to_store.pop(req.request_id)

    meta = sched.build_connector_meta(output)

    # The exact-handoff block is never a store source for a departing request.
    assert mamba_blocks[1].block_id not in meta.store_gpu_blocks
    stats = sched.get_boundary_store_stats()
    assert stats.dropped_request_gone == 1
    assert stats.stored == 0


def test_boundary_handoff_keeps_block_meta_index_parallel() -> None:
    """Boundary stores must extend block_meta alongside the block ids.

    ``_process_store_completion`` indexes ``block_meta`` by position, so a
    boundary handoff that appends only block ids shortens the list, trips its
    length assertion and shifts every following entry's metadata.
    """
    block_size = 4 * BLOCK_SIZE
    fix = _make_hybrid_attention_mamba_scheduler(
        num_cpu_blocks=16,
        num_gpu_blocks=24,
        block_size=block_size,
        enable_kv_cache_events=True,
    )
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool
    assert sched.enable_kv_cache_events

    req = _make_cp_request(num_blocks=2, virtual_block_size=block_size)
    attn_blocks = _allocate_cp_gpu_blocks(
        gpu_pool, req, 2, virtual_block_size=block_size, group_id=0
    )
    mamba_blocks = gpu_pool.get_new_blocks(2)
    mamba_blocks[1].set_block_hash(
        make_block_hash_with_group_id(req.block_hashes[1], 1)
    )
    kv_blocks = KVCacheBlocks(blocks=(attn_blocks, mamba_blocks))
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)
    # Make the FA blocks positionally eligible in the same event as the
    # explicit Mamba boundary handoff.
    req.num_computed_tokens = 2 * block_size

    output = make_scheduler_output(
        {req.request_id: 2 * block_size},
        new_reqs={req.request_id: kv_blocks.get_block_ids()},
    )
    output.kv_connector_block_state = KVConnectorBlockState(
        req_ids=set(),
        resolve_block_ids=lambda _: (),
        boundary_state_offloads={
            req.request_id: [(1, mamba_blocks[1].block_id, 2 * block_size)]
        },
    )

    gpu_ids, cpu_ids, req_ids, block_meta = sched.prepare_store_specs(output)
    assert mamba_blocks[1].block_id in gpu_ids
    assert set(gpu_ids) == {
        *(block.block_id for block in attn_blocks),
        mamba_blocks[1].block_id,
    }
    assert req_ids == [req.request_id]
    assert block_meta is not None
    assert len(block_meta) == len(gpu_ids) == len(cpu_ids)


def test_finished_eager_store_recovers_nulled_window_tail() -> None:
    """A sliding-window group nulls pages that left the window before the
    connector sees the finished request's block table, while the GPU keeps
    them hashed in its free queue. The eager store must copy those pages, or
    the CPU lookup (which needs the whole window) misses every time."""
    fix = make_scheduler(num_cpu_blocks=16, num_gpu_blocks=32, num_groups=2, lazy=False)
    sched = fix.scheduler
    gpu_pool = fix.gpu_block_pool
    null_block = gpu_pool.null_block

    num_blocks = 4  # == sliding window of the test SWA group
    req = make_request(num_blocks=num_blocks)
    fa_blocks = _allocate_gpu_blocks(gpu_pool, req, num_blocks, group_id=0)
    swa_blocks = _allocate_gpu_blocks(gpu_pool, req, num_blocks, group_id=1)
    # remove_skipped_blocks() ran: the two oldest window pages are freed (still
    # hashed) and their slots hold the null block.
    gpu_pool.free_blocks(swa_blocks[:2])
    swa_table = [null_block, null_block, *swa_blocks[2:]]
    kv_blocks = KVCacheBlocks(blocks=(fa_blocks, swa_table))
    req.num_computed_tokens = num_blocks * BLOCK_SIZE
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)

    block_ids = kv_blocks.get_block_ids()
    sched.request_finished_all_groups(req, block_ids)
    meta = sched.build_connector_meta(make_scheduler_output({}))

    assert null_block.block_id not in meta.store_gpu_blocks
    assert set(meta.store_gpu_blocks) == {b.block_id for b in (*fa_blocks, *swa_blocks)}
    assert meta.store_event >= 0
    simulate_store_completion(sched, meta.store_event)

    req2 = Request(
        request_id="req-window-tail-load",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    assert hit_tokens == num_blocks * BLOCK_SIZE
    assert is_async is True


QSA_RING_BLOCK_SIZE = 4


def _make_qsa_hybrid_kv_cache_config(num_blocks: int) -> KVCacheConfig:
    register_all_kvcache_specs(vllm_config=None)
    full_spec = FullAttentionSpec(
        block_size=BLOCK_SIZE,
        num_kv_heads=NUM_KV_HEADS,
        head_size=HEAD_SIZE,
        dtype=DTYPE,
    )
    ring_spec = CircularBufferSpec(
        block_size=QSA_RING_BLOCK_SIZE,
        num_kv_heads=NUM_KV_HEADS,
        head_size=HEAD_SIZE,
        head_size_v=0,
        dtype=DTYPE,
    )
    groups = [
        KVCacheGroupSpec(["layer_0"], full_spec),
        KVCacheGroupSpec(["qsa_ring"], ring_spec),
    ]
    tensors = [
        KVCacheTensor(
            size=group.kv_cache_spec.page_size_bytes * num_blocks,
            layers=list(group.layer_names),
            layer_stride=group.kv_cache_spec.page_size_bytes * num_blocks,
            block_stride=group.kv_cache_spec.page_size_bytes,
        )
        for group in groups
    ]
    return KVCacheConfig(
        num_blocks=num_blocks,
        kv_cache_tensors=tensors,
        kv_cache_groups=groups,
    )


def _make_qsa_scheduler(
    lazy: bool = False,
) -> tuple[SimpleCPUOffloadScheduler, BlockPool]:
    kv_cache_config = _make_qsa_hybrid_kv_cache_config(num_blocks=16)
    sched = SimpleCPUOffloadScheduler(
        vllm_config=_make_vllm_config(),
        kv_cache_config=kv_cache_config,
        cpu_capacity_bytes=_BYTES_PER_BLOCK * 8 * 2,
        scheduler_block_size=BLOCK_SIZE,
        hash_block_size=BLOCK_SIZE,
        lazy_offload=lazy,
    )
    gpu_pool = BlockPool(
        num_gpu_blocks=16, enable_caching=True, hash_block_size=BLOCK_SIZE
    )
    sched.bind_gpu_block_pool(gpu_pool)
    return sched, gpu_pool


def test_qsa_ring_group_is_never_stored_or_loaded() -> None:
    sched, gpu_pool = _make_qsa_scheduler()
    assert sched.prefix_cacheable_group_ids == (0,)

    num_blocks = 2
    req = make_request(num_blocks=num_blocks)
    fa_blocks = _allocate_gpu_blocks(gpu_pool, req, num_blocks, group_id=0)
    ring_block = gpu_pool.get_new_blocks(1)[0]
    kv_blocks = KVCacheBlocks(blocks=(fa_blocks, [ring_block]))
    req.num_computed_tokens = num_blocks * BLOCK_SIZE
    sched.update_state_after_alloc(req, kv_blocks, num_external_tokens=0)

    sched_out = make_scheduler_output(
        {req.request_id: num_blocks * BLOCK_SIZE},
        new_reqs={req.request_id: kv_blocks.get_block_ids()},
    )
    meta = sched.build_connector_meta(sched_out)
    assert set(meta.store_gpu_blocks) == {block.block_id for block in fa_blocks}
    assert ring_block.block_id not in meta.store_gpu_blocks
    assert meta.store_event >= 0
    simulate_store_completion(sched, meta.store_event)

    req2 = Request(
        request_id="req-qsa-load",
        prompt_token_ids=req.prompt_token_ids,
        sampling_params=req.sampling_params,
        pooling_params=None,
        mm_features=None,
        block_hasher=req._block_hasher,
    )
    hit_tokens, is_async = sched.get_num_new_matched_tokens(req2, num_computed_tokens=0)
    assert hit_tokens == num_blocks * BLOCK_SIZE
    assert is_async is True

    fa_blocks2 = gpu_pool.get_new_blocks(num_blocks)
    ring_block2 = gpu_pool.get_new_blocks(1)[0]
    kv_blocks2 = KVCacheBlocks(blocks=(fa_blocks2, [ring_block2]))
    sched.update_state_after_alloc(req2, kv_blocks2, num_external_tokens=hit_tokens)

    meta2 = sched.build_connector_meta(make_scheduler_output({req2.request_id: 1}))
    assert set(meta2.load_gpu_blocks) == {block.block_id for block in fa_blocks2}
    assert ring_block2.block_id not in meta2.load_gpu_blocks
    assert len(meta2.load_cpu_blocks) == num_blocks


def test_lazy_target_blocks_ignore_non_prefix_cacheable_groups() -> None:
    with_ring = _make_qsa_hybrid_kv_cache_config(num_blocks=16)
    without_ring = KVCacheConfig(
        num_blocks=16,
        kv_cache_tensors=with_ring.kv_cache_tensors[:1],
        kv_cache_groups=with_ring.kv_cache_groups[:1],
    )
    max_batched = 64
    target_with = SimpleCPUOffloadScheduler._estimate_lazy_target_blocks(
        with_ring, max_batched
    )
    target_without = SimpleCPUOffloadScheduler._estimate_lazy_target_blocks(
        without_ring, max_batched
    )
    assert target_with == target_without
