"""Isolated hybrid cache branches, including LFM2 convolution state. The deepcopy + reorder approach follows notnotsamuel/LFM2.5-350M-RLCD (MIT; see THIRD_PARTY_NOTICES.md). Never broadcast writable state views. """ import copy import torch def state_tensors(cache): for layer in cache.layers: for name in ("keys", "values"): value = getattr(layer, name, None) if isinstance(value, torch.Tensor): yield value for name in ("conv_states", "recurrent_states"): states = getattr(layer, name, {}) values = states.values() if isinstance(states, dict) else states for value in values: if isinstance(value, torch.Tensor) and value.numel(): yield value def fork_cache(cache, count): if type(count) is not int or count <= 0: raise ValueError("branch count must be positive") tensors = list(state_tensors(cache)) if not tensors or any(t.shape[0] != 1 for t in tensors): raise ValueError("expected an initialized batch-one prefix cache") branch = copy.deepcopy(cache) branch.reorder_cache(torch.zeros(count, dtype=torch.long, device=tensors[0].device)) return branch