chore: import upstream snapshot with attribution

This commit is contained in:
wehub-resource-sync
2026-07-13 13:24:13 +08:00
commit 1037506f2e
6050 changed files with 1731598 additions and 0 deletions
+48
View File
@@ -0,0 +1,48 @@
from typing import Iterable, Dict
import gzip
import json
import os
ROOT = os.path.dirname(os.path.abspath(__file__))
HUMAN_EVAL = os.path.join(ROOT, "..", "data", "HumanEval.jsonl.gz")
def read_problems(evalset_file: str = HUMAN_EVAL) -> Dict[str, Dict]:
return {task["task_id"]: task for task in stream_jsonl(evalset_file)}
def stream_jsonl(filename: str) -> Iterable[Dict]:
"""
Parses each jsonl line and yields it as a dictionary
"""
if filename.endswith(".gz"):
with open(filename, "rb") as gzfp:
with gzip.open(gzfp, 'rt') as fp:
for line in fp:
if any(not x.isspace() for x in line):
yield json.loads(line)
else:
with open(filename, "r") as fp:
for line in fp:
if any(not x.isspace() for x in line):
yield json.loads(line)
def write_jsonl(filename: str, data: Iterable[Dict], append: bool = False):
"""
Writes an iterable of dictionaries to jsonl
"""
if append:
mode = 'ab'
else:
mode = 'wb'
filename = os.path.expanduser(filename)
if filename.endswith(".gz"):
with open(filename, mode) as fp:
with gzip.GzipFile(fileobj=fp, mode='wb') as gzfp:
for x in data:
gzfp.write((json.dumps(x) + "\n").encode('utf-8'))
else:
with open(filename, mode) as fp:
for x in data:
fp.write((json.dumps(x) + "\n").encode('utf-8'))
+107
View File
@@ -0,0 +1,107 @@
from collections import defaultdict, Counter
from concurrent.futures import ThreadPoolExecutor, as_completed
from typing import List, Union, Iterable, Dict
import itertools
import numpy as np
import tqdm
from eval.codex_humaneval.data import HUMAN_EVAL, read_problems, stream_jsonl, write_jsonl
from eval.codex_humaneval.execution import check_correctness
def estimate_pass_at_k(
num_samples: Union[int, List[int], np.ndarray],
num_correct: Union[List[int], np.ndarray],
k: int
) -> np.ndarray:
"""
Estimates pass@k of each problem and returns them in an array.
"""
def estimator(n: int, c: int, k: int) -> float:
"""
Calculates 1 - comb(n - c, k) / comb(n, k).
"""
if n - c < k:
return 1.0
return 1.0 - np.prod(1.0 - k / np.arange(n - c + 1, n + 1))
if isinstance(num_samples, int):
num_samples_it = itertools.repeat(num_samples, len(num_correct))
else:
assert len(num_samples) == len(num_correct)
num_samples_it = iter(num_samples)
return np.array([estimator(int(n), int(c), k) for n, c in zip(num_samples_it, num_correct)])
def evaluate_functional_correctness(
sample_file: str,
k: List[int] = [1, 10, 100],
n_workers: int = 4,
timeout: float = 3.0,
problems=None,
problem_file: str = HUMAN_EVAL,
):
"""
Evaluates the functional correctness of generated samples, and writes
results to f"{sample_file}_results.jsonl.gz"
"""
if not problems:
problems = read_problems(problem_file)
# Check the generated samples against test suites.
with ThreadPoolExecutor(max_workers=n_workers) as executor:
futures = []
completion_id = Counter()
n_samples = 0
results = defaultdict(list)
print("Reading samples...")
for sample in tqdm.tqdm(stream_jsonl(sample_file)):
task_id = sample["task_id"]
completion = sample["completion"]
args = (problems[task_id], completion, timeout, completion_id[task_id])
future = executor.submit(check_correctness, *args)
futures.append(future)
completion_id[task_id] += 1
n_samples += 1
assert len(completion_id) == len(problems), "Some problems are not attempted."
print("Running test suites...")
for future in tqdm.tqdm(as_completed(futures), total=len(futures)):
result = future.result()
results[result["task_id"]].append((result["completion_id"], result))
# Calculate pass@k.
total, correct = [], []
for result in results.values():
result.sort()
passed = [r[1]["passed"] for r in result]
total.append(len(passed))
correct.append(sum(passed))
total = np.array(total)
correct = np.array(correct)
ks = k
pass_at_k = {f"pass@{k}": estimate_pass_at_k(total, correct, k).mean()
for k in ks if (total >= k).all()}
# Finally, save the results in one file:
def combine_results():
for sample in stream_jsonl(sample_file):
task_id = sample["task_id"]
result = results[task_id].pop(0)
sample["result"] = result[1]["result"]
sample["passed"] = result[1]["passed"]
yield sample
out_file = sample_file + "_results.jsonl"
print(f"Writing results to {out_file}...")
write_jsonl(out_file, tqdm.tqdm(combine_results(), total=n_samples))
return pass_at_k
+230
View File
@@ -0,0 +1,230 @@
from typing import Optional, Callable, Dict
import ast
import contextlib
import faulthandler
import io
import os
import multiprocessing
import platform
import signal
import tempfile
def check_correctness(problem: Dict, completion: str, timeout: float, completion_id: Optional[int] = None) -> Dict:
"""
Evaluates the functional correctness of a completion by running the test
suite provided in the problem.
:param completion_id: an optional completion ID so we can match
the results later even if execution finishes asynchronously.
"""
def unsafe_execute():
with create_tempdir():
# These system calls are needed when cleaning up tempdir.
import os
import shutil
rmtree = shutil.rmtree
rmdir = os.rmdir
chdir = os.chdir
# Disable functionalities that can make destructive changes to the test.
reliability_guard()
# Construct the check program and run it.
check_program = (
problem["prompt"] + completion + "\n" +
problem["test"] + "\n" +
f"check({problem['entry_point']})"
)
try:
exec_globals = {}
with swallow_io():
with time_limit(timeout):
# WARNING
# This program exists to execute untrusted model-generated code. Although
# it is highly unlikely that model-generated code will do something overtly
# malicious in response to this test suite, model-generated code may act
# destructively due to a lack of model capability or alignment.
# Users are strongly encouraged to sandbox this evaluation suite so that it
# does not perform destructive actions on their host or network. For more
# information on how OpenAI sandboxes its code, see the accompanying paper.
# Once you have read this disclaimer and taken appropriate precautions,
# uncomment the following line and proceed at your own risk:
exec(check_program, exec_globals)
result.append("passed")
except TimeoutException:
result.append("timed out")
except BaseException as e:
result.append(f"failed: {e}")
# Needed for cleaning up.
shutil.rmtree = rmtree
os.rmdir = rmdir
os.chdir = chdir
manager = multiprocessing.Manager()
result = manager.list()
p = multiprocessing.Process(target=unsafe_execute)
p.start()
p.join(timeout=timeout + 1)
if p.is_alive():
p.kill()
if not result:
result.append("timed out")
return dict(
task_id=problem["task_id"],
passed=result[0] == "passed",
result=result[0],
completion_id=completion_id,
)
@contextlib.contextmanager
def time_limit(seconds: float):
def signal_handler(signum, frame):
raise TimeoutException("Timed out!")
signal.setitimer(signal.ITIMER_REAL, seconds)
signal.signal(signal.SIGALRM, signal_handler)
try:
yield
finally:
signal.setitimer(signal.ITIMER_REAL, 0)
@contextlib.contextmanager
def swallow_io():
stream = WriteOnlyStringIO()
with contextlib.redirect_stdout(stream):
with contextlib.redirect_stderr(stream):
with redirect_stdin(stream):
yield
@contextlib.contextmanager
def create_tempdir():
with tempfile.TemporaryDirectory() as dirname:
with chdir(dirname):
yield dirname
class TimeoutException(Exception):
pass
class WriteOnlyStringIO(io.StringIO):
""" StringIO that throws an exception when it's read from """
def read(self, *args, **kwargs):
raise IOError
def readline(self, *args, **kwargs):
raise IOError
def readlines(self, *args, **kwargs):
raise IOError
def readable(self, *args, **kwargs):
""" Returns True if the IO object can be read. """
return False
class redirect_stdin(contextlib._RedirectStream): # type: ignore
_stream = 'stdin'
@contextlib.contextmanager
def chdir(root):
if root == ".":
yield
return
cwd = os.getcwd()
os.chdir(root)
try:
yield
except BaseException as exc:
raise exc
finally:
os.chdir(cwd)
def reliability_guard(maximum_memory_bytes: Optional[int] = None):
"""
This disables various destructive functions and prevents the generated code
from interfering with the test (e.g. fork bomb, killing other processes,
removing filesystem files, etc.)
WARNING
This function is NOT a security sandbox. Untrusted code, including, model-
generated code, should not be blindly executed outside of one. See the
Codex paper for more information about OpenAI's code sandbox, and proceed
with caution.
"""
if maximum_memory_bytes is not None:
import resource
resource.setrlimit(resource.RLIMIT_AS, (maximum_memory_bytes, maximum_memory_bytes))
resource.setrlimit(resource.RLIMIT_DATA, (maximum_memory_bytes, maximum_memory_bytes))
if not platform.uname().system == 'Darwin':
resource.setrlimit(resource.RLIMIT_STACK, (maximum_memory_bytes, maximum_memory_bytes))
faulthandler.disable()
import builtins
builtins.exit = None
builtins.quit = None
import os
os.environ['OMP_NUM_THREADS'] = '1'
os.kill = None
os.system = None
os.putenv = None
os.remove = None
os.removedirs = None
os.rmdir = None
os.fchdir = None
os.setuid = None
os.fork = None
os.forkpty = None
os.killpg = None
os.rename = None
os.renames = None
os.truncate = None
os.replace = None
os.unlink = None
os.fchmod = None
os.fchown = None
os.chmod = None
os.chown = None
os.chroot = None
os.fchdir = None
os.lchflags = None
os.lchmod = None
os.lchown = None
os.getcwd = None
os.chdir = None
import shutil
shutil.rmtree = None
shutil.move = None
shutil.chown = None
import subprocess
subprocess.Popen = None # type: ignore
__builtins__['help'] = None
import sys
sys.modules['ipdb'] = None
sys.modules['joblib'] = None
sys.modules['resource'] = None
sys.modules['psutil'] = None
sys.modules['tkinter'] = None
+241
View File
@@ -0,0 +1,241 @@
import argparse
import os
import json
import random
import torch
import vllm
from eval.utils import (
generate_completions,
load_hf_lm_and_tokenizer,
query_openai_chat_model,
dynamic_import_function,
)
from eval.codex_humaneval.data import write_jsonl, read_problems
from eval.codex_humaneval.evaluation import evaluate_functional_correctness
def main(args):
random.seed(42)
if not os.path.exists(args.save_dir):
os.makedirs(args.save_dir, exist_ok=True)
test_data = list(read_problems(args.data_file).values())
if args.max_num_examples is not None and len(test_data) > args.max_num_examples:
test_data = random.sample(test_data, args.max_num_examples)
print("Number of examples:", len(test_data))
if args.use_chat_format:
prompts = []
chat_formatting_function = dynamic_import_function(args.chat_formatting_function)
for example in test_data:
messages = [{"role": "user", "content": "Complete the following python function.\n\n\n" + example["prompt"]}]
prompt = chat_formatting_function(messages, add_bos=False)
if prompt[-1] in ["\n", " "]:
prompt += "Here is the completed function:\n\n\n" + example["prompt"]
else:
prompt += " Here is the completed function:\n\n\n" + example["prompt"]
prompts.append(prompt)
else:
prompts = [example["prompt"] for example in test_data]
if args.model_name_or_path:
if args.use_vllm:
model = vllm.LLM(
model=args.model_name_or_path,
tokenizer=args.tokenizer_name_or_path if args.tokenizer_name_or_path else args.model_name_or_path,
tokenizer_mode="slow" if args.use_slow_tokenizer else "auto",
tensor_parallel_size=torch.cuda.device_count(),
)
sampling_params = vllm.SamplingParams(
n=args.unbiased_sampling_size_n,
temperature=args.temperature,
top_p=0.95,
max_tokens=512,
stop=["</s>"],
# stop=["\nclass", "\ndef", "\n#", "\nif", "\nprint"]
)
generations = model.generate(prompts, sampling_params)
outputs = [output.text for it in generations for output in it.outputs]
# Note: early vllm might ignore the first space in the generation, because the processing of _token.
# This is not a problem for chat, but for codex, we need to keep the first space.
# Be careful here!
outputs = [output for output in outputs]
else:
print("Loading model and tokenizer...")
model, tokenizer = load_hf_lm_and_tokenizer(
model_name_or_path=args.model_name_or_path,
tokenizer_name_or_path=args.tokenizer_name_or_path,
load_in_8bit=args.load_in_8bit,
# device map is determined by the number of gpus available.
device_map="balanced_low_0" if torch.cuda.device_count() > 1 else "auto",
gptq_model=args.gptq,
use_fast_tokenizer=not args.use_slow_tokenizer,
)
# these stop sequences are those mentioned in the codex paper.
stop_sequences = ["\nclass", "\ndef", "\n#", "\nif", "\nprint"]
# Because many tokenizers will treat the word after space differently from the original word alone,
# to be consistent, we add a space before tokenization and remove it after tokenization.
stop_sequences = [tokenizer.encode(" " + x, add_special_tokens=False)[1:] for x in stop_sequences]
outputs_per_sampling_iter = []
for sampling_iter in range(args.unbiased_sampling_size_n):
print(f"Sampling iter: {sampling_iter} / {args.unbiased_sampling_size_n}")
samping_outputs = generate_completions(
model=model,
tokenizer=tokenizer,
prompts=prompts,
max_new_tokens=512,
batch_size=args.eval_batch_size,
stop_id_sequences=None, # stop_sequences,
num_return_sequences=1, # we don't use the hf num_return_sequences, because otherwise the real batch size will be multiplied by it and often cause oom.
do_sample=True, # if only pass@1 is evaluated, we do greedy decoding.
top_p=0.95,
temperature=args.temperature,
)
outputs_per_sampling_iter.append(samping_outputs)
# regroup the outputs to match the number of test data.
outputs = []
for i in range(len(prompts)):
for j in range(args.unbiased_sampling_size_n):
outputs.append(outputs_per_sampling_iter[j][i])
else:
instances = [{
"id": examle["task_id"],
"prompt": "Complete the following python function. Please only output the code for the completed function.\n\n\n" + prompt,
} for examle, prompt in zip(test_data, prompts)]
results = query_openai_chat_model(
engine=args.openai_engine,
instances=instances,
output_path=os.path.join(args.save_dir, "openai_query_results.jsonl"),
batch_size=args.eval_batch_size,
top_p=0.95,
temperature=args.temperature,
n=args.unbiased_sampling_size_n,
)
outputs = []
for result in results:
for choice in result["response_metadata"]["choices"]:
outputs.append(choice["message"]["content"])
# duplicates test data to match the number of outputs.
duplicate_test_data = [
example for example in test_data for _ in range(args.unbiased_sampling_size_n)
]
assert len(duplicate_test_data) == len(outputs)
predictions = [{"task_id": example["task_id"], "prompt": example["prompt"], "completion": output} for example, output in zip(duplicate_test_data, outputs)]
prediction_save_path = os.path.join(args.save_dir, "codex_eval_predictions.jsonl")
write_jsonl(prediction_save_path, predictions)
pass_at_k_results = evaluate_functional_correctness(
sample_file=prediction_save_path,
k=args.eval_pass_at_ks,
problems={example["task_id"]: example for example in test_data},
n_workers=64
)
print(pass_at_k_results)
with open(os.path.join(args.save_dir, "metrics.json"), "w") as fout:
json.dump(pass_at_k_results, fout)
if __name__ == "__main__":
parser = argparse.ArgumentParser()
parser.add_argument(
"--data_file",
type=str,
default="data/codex_eval/HumanEval.jsonl.gz",
help="Path to the HumanEval data file."
)
parser.add_argument(
"--max_num_examples",
type=int,
default=None,
help="Maximum number of examples to evaluate."
)
parser.add_argument(
"--model_name_or_path",
type=str,
default=None,
help="If specified, we will load the model to generate the predictions."
)
parser.add_argument(
"--tokenizer_name_or_path",
type=str,
default=None,
help="If specified, we will load the tokenizer from here."
)
parser.add_argument(
"--use_slow_tokenizer",
action="store_true",
help="If given, we will use the slow tokenizer."
)
parser.add_argument(
"--openai_engine",
type=str,
default=None,
help="If specified, we will use the OpenAI API to generate the predictions."
)
parser.add_argument(
"--save_dir",
type=str,
default="results/codex_eval",
help="Directory to save the results."
)
parser.add_argument(
"--eval_batch_size",
type=int,
default=1,
help="Batch size for evaluation."
)
parser.add_argument(
"--eval_pass_at_ks",
nargs="+",
type=int,
default=[1],
help="Multiple k's that we will report pass@k."
)
parser.add_argument(
"--unbiased_sampling_size_n",
type=int,
default=20,
help="Codex HumanEval requires `n` sampled generations per prompt, to estimate the unbiased pass@k. "
)
parser.add_argument(
"--temperature",
type=float,
default=0.1,
help="Temperature for sampling. This is should be low for evaluating smaller pass@k, and high for larger pass@k."
)
parser.add_argument(
"--load_in_8bit",
action="store_true",
help="Load model in 8bit mode, which will reduce memory and speed up inference."
)
parser.add_argument(
"--gptq",
action="store_true",
help="If given, we're evaluating a 4-bit quantized GPTQ model."
)
parser.add_argument(
"--use_vllm",
action="store_true",
help="If given, we will use the vllm library, which will likely increase the inference throughput."
)
parser.add_argument(
"--use_chat_format",
action="store_true",
help="If given, we will use the chat format for the prompts."
)
parser.add_argument(
"--chat_formatting_function",
type=str,
default="eval.templates.create_prompt_with_tulu_chat_format",
help="The function to use to create the chat format. This function will be dynamically imported. Please see examples in `eval/templates.py`."
)
args = parser.parse_args()
# model_name_or_path and openai_engine cannot be both None or both not None.
assert (args.model_name_or_path is None) != (args.openai_engine is None), "Either model_name_or_path or openai_engine should be specified."
assert args.unbiased_sampling_size_n >= max(args.eval_pass_at_ks), "n should be larger than the largest k in eval_pass_at_ks."
main(args)