"""Time the Python preprocessing (tokenize + build schema + pad) for requests of different sizes.""" import statistics import time import warnings warnings.filterwarnings("ignore") from preprocessing import load_processor, prepare_decision # noqa: E402 proc = load_processor(".") tasks = {"intent": ["order_status", "refund_request", "cancel_subscription", "update_payment", "other"], "urgency": ["low", "normal", "high"]} sentence = ("My subscription renewed on April 15 for 5,400 yen after the service was already down. " "Can I get that charge refunded? ") for bucket, repeats in [(64, 1), (128, 3), (256, 8), (512, 18)]: text = (sentence * repeats).strip() for _ in range(5): arrays = prepare_decision(proc, text, tasks, bucket, 4, 32) times = [] for _ in range(50): t = time.perf_counter() arrays = prepare_decision(proc, text, tasks, bucket, 4, 32) times.append((time.perf_counter() - t) * 1000) n = int(arrays["attention_mask"].sum()) print(f"L{bucket:<4} {n:4d} real tokens prep p50 {statistics.median(times):5.2f} ms")