"""Keep EvalPlus's guard while respecting a container's existing hard limits. EvalPlus 0.3.1 requests the same 4 GiB limit for address space, data and stack. Some containers have a smaller hard stack limit that an unprivileged child may not raise. Clamp each request to that resource's hard limit. All other upstream restrictions and the requested address-space ceiling remain in effect. """ import hashlib import json import os import resource import subprocess import sys PROTOCOL = 'evalplus-0.3.1-isolated-worker-8gib-hard-limit-clamp-v3' MAXIMUM_MEMORY_BYTES = 8*1024**3 def install(): import evalplus.eval as evaluator if getattr(evaluator.reliability_guard, '_nex_compat', False): return original = evaluator.reliability_guard def guarded(maximum_memory_bytes=None): # This wrapper runs only inside the disposable EvalPlus child process. set_limit = resource.setrlimit def bounded(kind, limits): hard = resource.getrlimit(kind)[1] if hard != resource.RLIM_INFINITY: limits = tuple(hard if value == resource.RLIM_INFINITY else min(value, hard) for value in limits) set_limit(kind, limits) resource.setrlimit = bounded try: original(maximum_memory_bytes=maximum_memory_bytes) finally: resource.setrlimit = set_limit guarded._nex_compat = True evaluator.reliability_guard = guarded def _local_score(problem, groundtruth, answer): from evalplus.eval import untrusted_check from evalplus.sanitize import sanitize install() solution = sanitize(answer, entrypoint=problem['entry_point']) # These are the same two untrusted_check calls used by upstream # check_correctness. Avoid importing its unrelated code-generation backends: # their large virtual mappings can already exceed the default child ceiling. checked={} for partition in ['base','plus']: status,details=untrusted_check('humaneval',solution,problem[partition+'_input'], problem['entry_point'],expected=groundtruth[partition],atol=problem['atol'], ref_time=groundtruth[partition+'_time'],fast_check=True,min_time_limit=1,gt_time_limit_factor=4) checked[partition]=dict(status=status,checked=len(details),passed=sum(details)) return float(all(value['status']=='pass' for value in checked.values())),checked def _run_worker(payload): env=dict(os.environ,OPENBLAS_NUM_THREADS='1',MKL_NUM_THREADS='1',OMP_NUM_THREADS='1', EVALPLUS_MAX_MEMORY_BYTES=str(MAXIMUM_MEMORY_BYTES)) result=subprocess.run([sys.executable,__file__],input=json.dumps(payload),text=True, capture_output=True,env=env,timeout=600) if result.returncode: raise RuntimeError('Isolated EvalPlus infrastructure failed: '+result.stdout[-2000:]+result.stderr[-4000:]) line=next(line for line in reversed(result.stdout.splitlines()) if line.startswith('SCORER_RESULT ')) return json.loads(line[len('SCORER_RESULT '):]) def score(problem, groundtruth, answer): return _run_worker(dict(id=problem['task_id'],answer=answer))['score'] def preflight(suite): """Verify the complete selected suite with canonical solutions before scoring.""" selected = [task for task in suite if task['kind'] == 'humaneval_plus_32'] if selected: evidence=_run_worker(dict(preflight=[task['id'] for task in selected])) print('EVALPLUS_PREFLIGHT ' + json.dumps(evidence),flush=True) def _worker(): from evalplus.data import get_human_eval_plus from evalplus.gen.util import trusted_exec import psutil payload=json.loads(sys.stdin.read()) problems=get_human_eval_plus() ids=payload.get('preflight',[payload.get('id')]) for key in ids: problem=problems[key] canonical=problem['prompt']+problem['canonical_solution'] oracle={} for partition in ['base','plus']: oracle[partition],oracle[partition+'_time']=trusted_exec(canonical, problem[partition+'_input'],problem['entry_point'],record_time=True) value,detail=_local_score(problem,oracle,canonical if 'preflight' in payload else payload['answer']) if 'preflight' in payload: assert value==1, ('EvalPlus canonical preflight failed',key,detail,psutil.Process().memory_info()) if 'preflight' in payload: result=dict(protocol=PROTOCOL,canonical_passes=len(ids),virtual_memory_bytes=psutil.Process().memory_info().vms, maximum_memory_bytes=MAXIMUM_MEMORY_BYTES, hard_limits={name:resource.getrlimit(getattr(resource,name))[1] for name in ['RLIMIT_AS','RLIMIT_DATA','RLIMIT_STACK']}) else: result=dict(protocol=PROTOCOL,score=value,detail=detail) print('SCORER_RESULT '+json.dumps(result),flush=True) if __name__=='__main__': _worker()