"""Isolated complete-1K request-local encoding reuse; native arithmetic is untouched.""" from dataclasses import replace MAX_INPUT_TOKENS=1024 def request_collator(native_collator,records): """Encode each occurrence once before any model call; never cache by external ID.""" class RequestEncodedCollator(type(native_collator)): def __init__(self): super().__init__(native_collator.tokenizer,max_length=MAX_INPUT_TOKENS,state_truncation='error') self._records=tuple(records);self._encoded=[];self._cursor=0;self._active=None for row in self._records: encoded=super().encode(row,labeled=False) if (encoded['input_tokens']>MAX_INPUT_TOKENS or encoded['state_tokens_original']!=encoded['state_tokens_kept']): raise ValueError('Native collator violated the complete-input 1K profile') self._encoded.append(encoded) def encode(self,record,labeled=False): # Only the unchanged native tensor assembler consumes this request-local queue. if labeled or self._active is None: raise ValueError('Request-local collator supports only its admitted inference batch') expected,encoded=next(self._active) if record is not expected: raise ValueError('Request-local record occurrence order changed') return encoded def __call__(self,records,labeled=False,device='cpu'): records=list(records);end=self._cursor+len(records) if labeled or end>len(self._records) or any(a is not b for a,b in zip(records,self._records[self._cursor:end])): raise ValueError('Request-local record occurrence order changed') self._active=iter(zip(records,self._encoded[self._cursor:end])) try: # Original allocation, padding, targets/values, kinds and device transfer. result=super().__call__(records,labeled=False,device=device) finally: self._active=None self._cursor=end return result def finish(self): if self._cursor!=len(self._records): raise ValueError('Native prediction did not consume the admitted request') return RequestEncodedCollator() def predict_1k(native,records,*,batch_size=8): """Same native prediction/output/errors; one encoding per occurrence per call. No model mode/inventory check is skipped. No result, token array or record is retained across API calls. Same IDs with different inputs remain distinct. """ if type(batch_size) is not int or not 1<=batch_size<=8: raise ValueError('The product profile permits integer batch sizes 1..8') records=list(records);guard=request_collator(native.collator,records) from decision_runtime import predict result=predict(replace(native,collator=guard),records,batch_size=batch_size) guard.finish() return result