Preserve CP HiCache valid tails while padding physical pages
CP HiCache now keeps radix and scheduler-visible lengths as valid tokens while host/device transfers reserve and replay the padded physical page span. Exact valid-tail write, insertion, and match paths no longer fall back to page-flooring; the physical owner-lane contract still uses padded page metadata. Constraint: Scheduler prefix indices must never include padded tail locs. Constraint: Host/device transfer and owner-lane admission remain page-based. Rejected: Pad to cp_size or 2*cp_size pages | wastes KV and recreates short-tail fallback behavior. Rejected: Expose padded locs through load_cp return | would leak fake tokens into req.prefix_indices. Confidence: medium Scope-risk: moderate Directive: Do not implement split-inside-tail by duplicating page_owners without a page-sharing/refcount design. Tested: local py_compile for touched CP HiCache/radix/controller files and tests. Tested: remote g0034 CP HiCache impacted suites: 143 passed, 5 warnings. Tested: remote g0034 CP shared KV C1-C5 suite: 122 passed, 5 warnings. Not-tested: full local pytest, blocked by missing runtime dependencies such as orjson/starlette. Not-tested: CUDA E2E runtime for this commit. Co-authored-by: OmX <omx@oh-my-codex.dev>
This commit is contained in:
@@ -701,15 +701,20 @@ class TestHiCacheControllerCPWrite(CustomTestCase):
|
||||
self.assertEqual(host_pool.backups, [])
|
||||
self.assertEqual(host_pool.layer_backups[0][1].tolist(), [4, 5, 6, 7])
|
||||
|
||||
def test_cp_write_rejects_incomplete_owned_physical_page(self):
|
||||
def test_cp_write_accepts_valid_tail_and_pads_owned_physical_page(self):
|
||||
host_pool = FakeHostPool(torch.tensor([100, 101, 102, 103], dtype=torch.int64))
|
||||
controller = self.make_controller(host_pool, cp_rank=1)
|
||||
logical_locs = torch.tensor([8, 9, 10], dtype=torch.int64)
|
||||
|
||||
with self.assertRaisesRegex(
|
||||
ValueError, "_write_cp expects page-aligned device_indices"
|
||||
):
|
||||
controller.write(logical_locs, node_id=21)
|
||||
result = controller.write(logical_locs, node_id=21)
|
||||
|
||||
self.assertEqual(result.metadata.logical_len, 3)
|
||||
self.assertEqual(result.metadata.valid_len, 3)
|
||||
self.assertEqual(result.metadata.padded_len, 4)
|
||||
self.assertEqual(result.metadata.owned_positions.tolist(), [0, 1, 2, 3])
|
||||
self.assertEqual(result.metadata.host_indices.tolist(), [100, 101, 102, 103])
|
||||
self.assertEqual(host_pool.alloc_calls, [4])
|
||||
self.assertEqual(host_pool.layer_backups[0][1].tolist(), [4, 5, 6, 7])
|
||||
|
||||
def test_cp_write_rejects_non_contiguous_owned_physical_page(self):
|
||||
host_pool = FakeHostPool(torch.tensor([100, 101, 102, 103], dtype=torch.int64))
|
||||
@@ -1052,6 +1057,29 @@ class TestHiCacheControllerCPLoad(TestHiCacheControllerCPWrite):
|
||||
self.assertEqual(allocator.owner_alloc_calls, [[3, 0, 1, 2]])
|
||||
self.assertEqual(host_pool.loads[0][1].tolist(), [20, 21, 22, 23])
|
||||
|
||||
def test_cp_load_returns_valid_locs_while_transferring_padded_tail_page(self):
|
||||
host_pool = FakeHostPool(torch.tensor([100, 101, 102, 103], dtype=torch.int64))
|
||||
allocator = FakeAllocator(alloc_result=torch.arange(64, 72, dtype=torch.int64))
|
||||
controller = self.make_controller(host_pool, allocator=allocator, cp_rank=1)
|
||||
node = TreeNode()
|
||||
node.host_len = 6
|
||||
node.cp_hicache = CpHiCacheNodeMetadata(
|
||||
logical_len=6,
|
||||
padded_len=8,
|
||||
owned_positions=torch.tensor([4, 5, 6, 7], dtype=torch.int64),
|
||||
host_indices=torch.tensor([100, 101, 102, 103], dtype=torch.int64),
|
||||
page_owners=torch.tensor([3, 0], dtype=torch.int8),
|
||||
page_size=4,
|
||||
)
|
||||
|
||||
device_indices = controller.load_cp([node], node_id=112)
|
||||
controller.start_loading()
|
||||
|
||||
self.assertEqual(device_indices.tolist(), list(range(64, 70)))
|
||||
self.assertEqual(allocator.owner_alloc_calls, [[3, 0]])
|
||||
self.assertEqual(host_pool.loads[0][0].tolist(), [100, 101, 102, 103])
|
||||
self.assertEqual(host_pool.loads[0][1].tolist(), [20, 21, 22, 23])
|
||||
|
||||
def test_cp_load_frees_unexpected_owner_allocator_length(self):
|
||||
host_pool = FakeHostPool(torch.tensor([100, 101, 102, 103], dtype=torch.int64))
|
||||
allocator = FakeAllocator(alloc_result=torch.arange(64, 76, dtype=torch.int64))
|
||||
|
||||
Reference in New Issue
Block a user