diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 66ac66b..7b5b617 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -113,6 +113,34 @@ jobs:
           poetry run pytest tests/dataset_gen -m "not slow and not openai"
         timeout-minutes: 30
 
+  run-evaluation-test:
+    needs: [check-black, lint-ruff]
+    runs-on: [self-hosted, x64, gpu, linux]
+    strategy:
+      fail-fast: false
+    steps:
+      - uses: actions/checkout@v3
+      - name: Set up Python 3.9
+        uses: actions/setup-python@v4
+        with:
+          python-version: 3.9
+      - name: Prepare environment
+        run: |
+          python -m pip install --upgrade pip
+          python -m pip install -U poetry
+          poetry install
+          poetry run pip install torch
+          poetry run huggingface-cli login --token $HF_AUTH_TOKEN
+
+
+      - name: Test
+        id: test
+        run: |
+          poetry run pytest tests/eval
+        timeout-minutes: 10
+    env:
+      HF_AUTH_TOKEN: ${{ secrets.HF_AUTH_TOKEN }}
+
 # just for blocking the merge until all parallel core-test are successful
   success-all-test:
     needs: [check-mypy, run-training-test, run-dataset-gen-test, check-black, lint-ruff]
diff --git a/README.md b/README.md
index bef967d..c58d749 100644
--- a/README.md
+++ b/README.md
@@ -39,7 +39,7 @@ deepspeed --num_gpus=2 textbook/train.py --deepspeed ds_config.json --epochs 2 -
 
 Note:
 
-to use starcoder base model you need to first login to HF
+to use starcoder base model you need to first login to HF and accept the ToS of the used starcoder base model (https://huggingface.co/bigcode/starcoderbase-1b)
 ```cmd
 huggingface-cli login
 ```
diff --git a/human-eval/LICENSE b/human-eval/LICENSE
new file mode 100644
index 0000000..4f14854
--- /dev/null
+++ b/human-eval/LICENSE
@@ -0,0 +1,21 @@
+The MIT License
+
+Copyright (c) OpenAI (https://openai.com)
+
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+
+The above copyright notice and this permission notice shall be included in
+all copies or substantial portions of the Software.
+
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
+THE SOFTWARE.
diff --git a/human-eval/README.md b/human-eval/README.md
new file mode 100644
index 0000000..3cfd22f
--- /dev/null
+++ b/human-eval/README.md
@@ -0,0 +1,115 @@
+# HumanEval: Hand-Written Evaluation Set 
+
+This is an evaluation harness for the HumanEval problem solving dataset
+described in the paper "[Evaluating Large Language Models Trained on
+Code](https://arxiv.org/abs/2107.03374)".
+
+## Installation
+
+Make sure to use python 3.7 or later:
+```
+$ conda create -n codex python=3.7
+$ conda activate codex
+```
+
+Check out and install this repository:
+```
+$ git clone https://github.com/openai/human-eval
+$ pip install -e human-eval
+```
+
+## Usage
+
+**This program exists to run untrusted model-generated code. Users are strongly
+encouraged not to do so outside of a robust security sandbox. The [execution
+call](https://github.com/openai/human-eval/blob/master/human_eval/execution.py#L48-L58)
+in `execution.py` is deliberately commented out to ensure users read this
+disclaimer before running code in a potentially unsafe manner. See the comment in
+`execution.py` for more information and instructions.**
+
+After following the above instructions to enable execution, generate samples
+and save them in the following JSON Lines (jsonl) format, where each sample is
+formatted into a single line like so:
+```
+{"task_id": "Corresponding HumanEval task ID", "completion": "Completion only without the prompt"}
+```
+We provide `example_problem.jsonl` and `example_solutions.jsonl` under `data`
+to illustrate the format and help with debugging.
+
+Here is nearly functional example code (you just have to provide
+`generate_one_completion` to make it work) that saves generated completions to
+`samples.jsonl`.
+```
+from human_eval.data import write_jsonl, read_problems
+
+problems = read_problems()
+
+num_samples_per_task = 200
+samples = [
+    dict(task_id=task_id, completion=generate_one_completion(problems[task_id]["prompt"]))
+    for task_id in problems
+    for _ in range(num_samples_per_task)
+]
+write_jsonl("samples.jsonl", samples)
+```
+
+To evaluate the samples, run
+```
+$ evaluate_functional_correctness samples.jsonl
+Reading samples...
+32800it [00:01, 23787.50it/s]
+Running test suites...
+100%|...| 32800/32800 [16:11<00:00, 33.76it/s]
+Writing results to samples.jsonl_results.jsonl...
+100%|...| 32800/32800 [00:00<00:00, 42876.84it/s]
+{'pass@1': ..., 'pass@10': ..., 'pass@100': ...}
+```
+This script provides more fine-grained information in a new file ending in
+`<input_path>_results.jsonl`. Each row now contains whether the completion
+`passed` along with the execution `result` which is one of "passed", "timed
+out", or "failed".
+
+As a quick sanity-check, the example samples should yield 0.5 pass@1.
+```
+$ evaluate_functional_correctness data/example_samples.jsonl --problem_file=data/example_problem.jsonl
+Reading samples...
+6it [00:00, 3397.11it/s]
+Running example suites...
+100%|...| 6/6 [00:03<00:00,  1.96it/s]
+Writing results to data/example_samples.jsonl_results.jsonl...
+100%|...| 6/6 [00:00<00:00, 6148.50it/s]
+{'pass@1': 0.4999999999999999}
+```
+
+Because there is no unbiased way of estimating pass@k when there are fewer
+samples than k, the script does not evaluate pass@k for these cases. To
+evaluate with other k values, pass `--k=<comma-separated-values-here>`. For
+other options, see
+```
+$ evaluate_functional_correctness --help
+```
+However, we recommend that you use the default values for the rest.
+
+## Known Issues
+
+While evaluation uses very little memory, you might see the following error
+message when the system is running out of RAM. Since this may cause some
+correct programs to fail, we recommend that you free some memory and try again.
+```
+malloc: can't allocate region
+```
+
+## Citation
+
+Please cite using the following bibtex entry:
+
+```
+@article{chen2021codex,
+  title={Evaluating Large Language Models Trained on Code},
+  author={Mark Chen and Jerry Tworek and Heewoo Jun and Qiming Yuan and Henrique Ponde de Oliveira Pinto and Jared Kaplan and Harri Edwards and Yuri Burda and Nicholas Joseph and Greg Brockman and Alex Ray and Raul Puri and Gretchen Krueger and Michael Petrov and Heidy Khlaaf and Girish Sastry and Pamela Mishkin and Brooke Chan and Scott Gray and Nick Ryder and Mikhail Pavlov and Alethea Power and Lukasz Kaiser and Mohammad Bavarian and Clemens Winter and Philippe Tillet and Felipe Petroski Such and Dave Cummings and Matthias Plappert and Fotios Chantzis and Elizabeth Barnes and Ariel Herbert-Voss and William Hebgen Guss and Alex Nichol and Alex Paino and Nikolas Tezak and Jie Tang and Igor Babuschkin and Suchir Balaji and Shantanu Jain and William Saunders and Christopher Hesse and Andrew N. Carr and Jan Leike and Josh Achiam and Vedant Misra and Evan Morikawa and Alec Radford and Matthew Knight and Miles Brundage and Mira Murati and Katie Mayer and Peter Welinder and Bob McGrew and Dario Amodei and Sam McCandlish and Ilya Sutskever and Wojciech Zaremba},
+  year={2021},
+  eprint={2107.03374},
+  archivePrefix={arXiv},
+  primaryClass={cs.LG}
+}
+```
diff --git a/human-eval/data/HumanEval.jsonl.gz b/human-eval/data/HumanEval.jsonl.gz
new file mode 100644
index 0000000..998d251
Binary files /dev/null and b/human-eval/data/HumanEval.jsonl.gz differ
diff --git a/human-eval/data/example_problem.jsonl b/human-eval/data/example_problem.jsonl
new file mode 100644
index 0000000..0623628
--- /dev/null
+++ b/human-eval/data/example_problem.jsonl
@@ -0,0 +1 @@
+{"task_id": "test/0", "prompt": "def return1():\n", "canonical_solution": "    return 1", "test": "def check(candidate):\n    assert candidate() == 1", "entry_point": "return1"}
diff --git a/human-eval/data/example_samples.jsonl b/human-eval/data/example_samples.jsonl
new file mode 100644
index 0000000..ddd5e6b
--- /dev/null
+++ b/human-eval/data/example_samples.jsonl
@@ -0,0 +1,6 @@
+{"task_id": "test/0", "completion": "    import subprocess\n    subprocess.check_output('rm -rf tmp')"}
+{"task_id": "test/0", "completion": "    import time\n    time.sleep(10)\n    return 1"}
+{"task_id": "test/0", "completion": "    return input('enter a number')"}
+{"task_id": "test/0", "completion": "    return 1"}
+{"task_id": "test/0", "completion": "  return 1"}
+{"task_id": "test/0", "completion": "\treturn 1"}
diff --git a/human-eval/human_eval/__init__.py b/human-eval/human_eval/__init__.py
new file mode 100644
index 0000000..e69de29
diff --git a/human-eval/human_eval/data.py b/human-eval/human_eval/data.py
new file mode 100644
index 0000000..b53e782
--- /dev/null
+++ b/human-eval/human_eval/data.py
@@ -0,0 +1,49 @@
+from typing import Iterable, Dict
+import gzip
+import json
+import os
+
+
+ROOT = os.path.dirname(os.path.abspath(__file__))
+HUMAN_EVAL = os.path.join(ROOT, "..", "data", "HumanEval.jsonl.gz")
+
+
+def read_problems(evalset_file: str = HUMAN_EVAL) -> Dict[str, Dict]:
+    return {task["task_id"]: task for task in stream_jsonl(evalset_file)}
+
+
+def stream_jsonl(filename: str) -> Iterable[Dict]:
+    """
+    Parses each jsonl line and yields it as a dictionary
+    """
+    if filename.endswith(".gz"):
+        with open(filename, "rb") as gzfp:
+            with gzip.open(gzfp, "rt") as fp:
+                for line in fp:
+                    if any(not x.isspace() for x in line):
+                        yield json.loads(line)
+    else:
+        with open(filename, "r") as fp:
+            for line in fp:
+                if any(not x.isspace() for x in line):
+                    yield json.loads(line)
+
+
+def write_jsonl(filename: str, data: Iterable[Dict], append: bool = False):
+    """
+    Writes an iterable of dictionaries to jsonl
+    """
+    if append:
+        mode = "ab"
+    else:
+        mode = "wb"
+    filename = os.path.expanduser(filename)
+    if filename.endswith(".gz"):
+        with open(filename, mode) as fp:
+            with gzip.GzipFile(fileobj=fp, mode="wb") as gzfp:
+                for x in data:
+                    gzfp.write((json.dumps(x) + "\n").encode("utf-8"))
+    else:
+        with open(filename, mode) as fp:
+            for x in data:
+                fp.write((json.dumps(x) + "\n").encode("utf-8"))
diff --git a/human-eval/human_eval/evaluate_functional_correctness.py b/human-eval/human_eval/evaluate_functional_correctness.py
new file mode 100644
index 0000000..eb48f4a
--- /dev/null
+++ b/human-eval/human_eval/evaluate_functional_correctness.py
@@ -0,0 +1,30 @@
+import fire
+import sys
+
+from human_eval.data import HUMAN_EVAL
+from human_eval.evaluation import evaluate_functional_correctness
+
+
+def entry_point(
+    sample_file: str,
+    k: str = "1,10,100",
+    n_workers: int = 4,
+    timeout: float = 3.0,
+    problem_file: str = HUMAN_EVAL,
+):
+    """
+    Evaluates the functional correctness of generated samples, and writes
+    results to f"{sample_file}_results.jsonl.gz"
+    """
+    k = list(map(int, k.split(",")))
+    results = evaluate_functional_correctness(
+        sample_file, k, n_workers, timeout, problem_file
+    )
+    print(results)
+
+
+def main():
+    fire.Fire(entry_point)
+
+
+sys.exit(main())
diff --git a/human-eval/human_eval/evaluation.py b/human-eval/human_eval/evaluation.py
new file mode 100644
index 0000000..e2c3a6d
--- /dev/null
+++ b/human-eval/human_eval/evaluation.py
@@ -0,0 +1,110 @@
+from collections import defaultdict, Counter
+from concurrent.futures import ThreadPoolExecutor, as_completed
+from typing import List, Union, Dict, Optional
+import itertools
+
+import numpy as np
+import tqdm
+
+from human_eval.data import HUMAN_EVAL, read_problems, stream_jsonl, write_jsonl
+from human_eval.execution import check_correctness
+
+
+def estimate_pass_at_k(
+    num_samples: Union[int, List[int], np.ndarray],
+    num_correct: Union[List[int], np.ndarray],
+    k: int,
+) -> np.ndarray:
+    """
+    Estimates pass@k of each problem and returns them in an array.
+    """
+
+    def estimator(n: int, c: int, k: int) -> float:
+        """
+        Calculates 1 - comb(n - c, k) / comb(n, k).
+        """
+        if n - c < k:
+            return 1.0
+        return 1.0 - np.prod(1.0 - k / np.arange(n - c + 1, n + 1))
+
+    if isinstance(num_samples, int):
+        num_samples_it = itertools.repeat(num_samples, len(num_correct))
+    else:
+        assert len(num_samples) == len(num_correct)
+        num_samples_it = iter(num_samples)
+
+    return np.array(
+        [estimator(int(n), int(c), k) for n, c in zip(num_samples_it, num_correct)]
+    )
+
+
+def evaluate_functional_correctness(
+    sample_file: str,
+    k: List[int] = [1, 10, 100],
+    n_workers: int = 4,
+    timeout: float = 3.0,
+    problem_file: str = HUMAN_EVAL,
+    problems: Optional[Dict] = None,
+):
+    """
+    Evaluates the functional correctness of generated samples, and writes
+    results to f"{sample_file}_results.jsonl.gz"
+    """
+
+    problems = problems or read_problems(problem_file)
+
+    # Check the generated samples against test suites.
+    with ThreadPoolExecutor(max_workers=n_workers) as executor:
+        futures = []
+        completion_id = Counter()
+        n_samples = 0
+        results = defaultdict(list)
+
+        print("Reading samples...")
+        for sample in tqdm.tqdm(stream_jsonl(sample_file)):
+            task_id = sample["task_id"]
+            completion = sample["completion"]
+            args = (problems[task_id], completion, timeout, completion_id[task_id])
+            future = executor.submit(check_correctness, *args)
+            futures.append(future)
+            completion_id[task_id] += 1
+            n_samples += 1
+
+        assert len(completion_id) == len(problems), "Some problems are not attempted."
+
+        print("Running test suites...")
+        for future in tqdm.tqdm(as_completed(futures), total=len(futures)):
+            result = future.result()
+            results[result["task_id"]].append((result["completion_id"], result))
+
+    # Calculate pass@k.
+    total, correct = [], []
+    for result in results.values():
+        result.sort()
+        passed = [r[1]["passed"] for r in result]
+        total.append(len(passed))
+        correct.append(sum(passed))
+    total = np.array(total)
+    correct = np.array(correct)
+
+    ks = k
+    pass_at_k = {
+        f"pass@{k}": estimate_pass_at_k(total, correct, k).mean()
+        for k in ks
+        if (total >= k).all()
+    }
+
+    # Finally, save the results in one file:
+    def combine_results():
+        for sample in stream_jsonl(sample_file):
+            task_id = sample["task_id"]
+            result = results[task_id].pop(0)
+            sample["result"] = result[1]["result"]
+            sample["passed"] = result[1]["passed"]
+            yield sample
+
+    out_file = sample_file + "_results.jsonl"
+    print(f"Writing results to {out_file}...")
+    write_jsonl(out_file, tqdm.tqdm(combine_results(), total=n_samples))
+
+    return pass_at_k
diff --git a/human-eval/human_eval/execution.py b/human-eval/human_eval/execution.py
new file mode 100644
index 0000000..e27b584
--- /dev/null
+++ b/human-eval/human_eval/execution.py
@@ -0,0 +1,245 @@
+from typing import Optional, Dict
+import contextlib
+import faulthandler
+import io
+import os
+import multiprocessing
+import platform
+import signal
+import tempfile
+
+
+def check_correctness(
+    problem: Dict, completion: str, timeout: float, completion_id: Optional[int] = None
+) -> Dict:
+    """
+    Evaluates the functional correctness of a completion by running the test
+    suite provided in the problem.
+
+    :param completion_id: an optional completion ID so we can match
+        the results later even if execution finishes asynchronously.
+    """
+
+    def unsafe_execute():
+        with create_tempdir():
+            # These system calls are needed when cleaning up tempdir.
+            import os
+            import shutil
+
+            rmtree = shutil.rmtree
+            rmdir = os.rmdir
+            chdir = os.chdir
+
+            # Disable functionalities that can make destructive changes to the test.
+            reliability_guard()
+
+            # Construct the check program and run it.
+            check_program = (
+                problem["prompt"]
+                + completion
+                + "\n"
+                + problem["test"]
+                + "\n"
+                + f"check({problem['entry_point']})"
+            )
+
+            try:
+                exec_globals = {}
+                with swallow_io():
+                    with time_limit(timeout):
+                        # WARNING
+                        # This program exists to execute untrusted model-generated code. Although
+                        # it is highly unlikely that model-generated code will do something overtly
+                        # malicious in response to this test suite, model-generated code may act
+                        # destructively due to a lack of model capability or alignment.
+                        # Users are strongly encouraged to sandbox this evaluation suite so that it
+                        # does not perform destructive actions on their host or network. For more
+                        # information on how OpenAI sandboxes its code, see the accompanying paper.
+                        # Once you have read this disclaimer and taken appropriate precautions,
+                        # uncomment the following line and proceed at your own risk:
+                        exec(check_program, exec_globals)
+                result.append("passed")
+            except TimeoutException:
+                result.append("timed out")
+            except BaseException as e:
+                result.append(f"failed: {e}")
+
+            # Needed for cleaning up.
+            shutil.rmtree = rmtree
+            os.rmdir = rmdir
+            os.chdir = chdir
+
+    manager = multiprocessing.Manager()
+    result = manager.list()
+
+    p = multiprocessing.Process(target=unsafe_execute)
+    p.start()
+    p.join(timeout=timeout + 1)
+    if p.is_alive():
+        p.kill()
+
+    if not result:
+        result.append("timed out")
+
+    return dict(
+        task_id=problem["task_id"],
+        passed=result[0] == "passed",
+        result=result[0],
+        completion_id=completion_id,
+    )
+
+
+@contextlib.contextmanager
+def time_limit(seconds: float):
+    def signal_handler(signum, frame):
+        raise TimeoutException("Timed out!")
+
+    signal.setitimer(signal.ITIMER_REAL, seconds)
+    signal.signal(signal.SIGALRM, signal_handler)
+    try:
+        yield
+    finally:
+        signal.setitimer(signal.ITIMER_REAL, 0)
+
+
+@contextlib.contextmanager
+def swallow_io():
+    stream = WriteOnlyStringIO()
+    with contextlib.redirect_stdout(stream):
+        with contextlib.redirect_stderr(stream):
+            with redirect_stdin(stream):
+                yield
+
+
+@contextlib.contextmanager
+def create_tempdir():
+    with tempfile.TemporaryDirectory() as dirname:
+        with chdir(dirname):
+            yield dirname
+
+
+class TimeoutException(Exception):
+    pass
+
+
+class WriteOnlyStringIO(io.StringIO):
+    """StringIO that throws an exception when it's read from"""
+
+    def read(self, *args, **kwargs):
+        raise IOError
+
+    def readline(self, *args, **kwargs):
+        raise IOError
+
+    def readlines(self, *args, **kwargs):
+        raise IOError
+
+    def readable(self, *args, **kwargs):
+        """Returns True if the IO object can be read."""
+        return False
+
+
+class redirect_stdin(contextlib._RedirectStream):  # type: ignore
+    _stream = "stdin"
+
+
+@contextlib.contextmanager
+def chdir(root):
+    if root == ".":
+        yield
+        return
+    cwd = os.getcwd()
+    os.chdir(root)
+    try:
+        yield
+    except BaseException as exc:
+        raise exc
+    finally:
+        os.chdir(cwd)
+
+
+def reliability_guard(maximum_memory_bytes: Optional[int] = None):
+    """
+    This disables various destructive functions and prevents the generated code
+    from interfering with the test (e.g. fork bomb, killing other processes,
+    removing filesystem files, etc.)
+
+    WARNING
+    This function is NOT a security sandbox. Untrusted code, including, model-
+    generated code, should not be blindly executed outside of one. See the
+    Codex paper for more information about OpenAI's code sandbox, and proceed
+    with caution.
+    """
+
+    if maximum_memory_bytes is not None:
+        import resource
+
+        resource.setrlimit(
+            resource.RLIMIT_AS, (maximum_memory_bytes, maximum_memory_bytes)
+        )
+        resource.setrlimit(
+            resource.RLIMIT_DATA, (maximum_memory_bytes, maximum_memory_bytes)
+        )
+        if not platform.uname().system == "Darwin":
+            resource.setrlimit(
+                resource.RLIMIT_STACK, (maximum_memory_bytes, maximum_memory_bytes)
+            )
+
+    faulthandler.disable()
+
+    import builtins
+
+    builtins.exit = None
+    builtins.quit = None
+
+    import os
+
+    os.environ["OMP_NUM_THREADS"] = "1"
+
+    os.kill = None
+    os.system = None
+    os.putenv = None
+    os.remove = None
+    os.removedirs = None
+    os.rmdir = None
+    os.fchdir = None
+    os.setuid = None
+    os.fork = None
+    os.forkpty = None
+    os.killpg = None
+    os.rename = None
+    os.renames = None
+    os.truncate = None
+    os.replace = None
+    os.unlink = None
+    os.fchmod = None
+    os.fchown = None
+    os.chmod = None
+    os.chown = None
+    os.chroot = None
+    os.fchdir = None
+    os.lchflags = None
+    os.lchmod = None
+    os.lchown = None
+    os.getcwd = None
+    os.chdir = None
+
+    import shutil
+
+    shutil.rmtree = None
+    shutil.move = None
+    shutil.chown = None
+
+    import subprocess
+
+    subprocess.Popen = None  # type: ignore
+
+    __builtins__["help"] = None
+
+    import sys
+
+    sys.modules["ipdb"] = None
+    sys.modules["joblib"] = None
+    sys.modules["resource"] = None
+    sys.modules["psutil"] = None
+    sys.modules["tkinter"] = None
diff --git a/human-eval/requirements.txt b/human-eval/requirements.txt
new file mode 100644
index 0000000..6883cc6
--- /dev/null
+++ b/human-eval/requirements.txt
@@ -0,0 +1,3 @@
+tqdm
+fire
+numpy
diff --git a/human-eval/setup.py b/human-eval/setup.py
new file mode 100644
index 0000000..932ef3b
--- /dev/null
+++ b/human-eval/setup.py
@@ -0,0 +1,20 @@
+import os
+
+import pkg_resources
+from setuptools import setup, find_packages
+
+
+setup(
+    name="human-eval",
+    py_modules=["human-eval"],
+    version="1.0",
+    description="",
+    author="OpenAI",
+    packages=find_packages(),
+    install_requires=[
+        str(r)
+        for r in pkg_resources.parse_requirements(
+            open(os.path.join(os.path.dirname(__file__), "requirements.txt"))
+        )
+    ],
+)
diff --git a/poetry.lock b/poetry.lock
index 33ad1e6..4b5d15b 100644
--- a/poetry.lock
+++ b/poetry.lock
@@ -883,6 +883,20 @@ files = [
 docs = ["furo (>=2023.5.20)", "sphinx (>=7.0.1)", "sphinx-autodoc-typehints (>=1.23,!=1.23.4)"]
 testing = ["covdefaults (>=2.3)", "coverage (>=7.2.7)", "diff-cover (>=7.5)", "pytest (>=7.3.1)", "pytest-cov (>=4.1)", "pytest-mock (>=3.10)", "pytest-timeout (>=2.1)"]
 
+[[package]]
+name = "fire"
+version = "0.5.0"
+description = "A library for automatically generating command line interfaces."
+optional = false
+python-versions = "*"
+files = [
+    {file = "fire-0.5.0.tar.gz", hash = "sha256:a6b0d49e98c8963910021f92bba66f65ab440da2982b78eb1bbf95a0a34aacc6"},
+]
+
+[package.dependencies]
+six = "*"
+termcolor = "*"
+
 [[package]]
 name = "fqdn"
 version = "1.5.1"
@@ -1074,6 +1088,24 @@ testing = ["InquirerPy (==0.3.4)", "Jinja2", "Pillow", "aiohttp", "gradio", "jed
 torch = ["torch"]
 typing = ["pydantic", "types-PyYAML", "types-requests", "types-simplejson", "types-toml", "types-tqdm", "types-urllib3"]
 
+[[package]]
+name = "human-eval"
+version = "1.0"
+description = ""
+optional = false
+python-versions = "*"
+files = []
+develop = true
+
+[package.dependencies]
+fire = "*"
+numpy = "*"
+tqdm = "*"
+
+[package.source]
+type = "directory"
+url = "human-eval"
+
 [[package]]
 name = "identify"
 version = "2.5.26"
@@ -3447,6 +3479,20 @@ files = [
 [package.dependencies]
 mpmath = ">=0.19"
 
+[[package]]
+name = "termcolor"
+version = "2.3.0"
+description = "ANSI color formatting for output in terminal"
+optional = false
+python-versions = ">=3.7"
+files = [
+    {file = "termcolor-2.3.0-py3-none-any.whl", hash = "sha256:3afb05607b89aed0ffe25202399ee0867ad4d3cb4180d98aaf8eefa6a5f7d475"},
+    {file = "termcolor-2.3.0.tar.gz", hash = "sha256:b5b08f68937f138fe92f6c089b99f1e2da0ae56c52b78bf7075fd95420fd9a5a"},
+]
+
+[package.extras]
+tests = ["pytest", "pytest-cov"]
+
 [[package]]
 name = "terminado"
 version = "0.17.1"
@@ -4118,4 +4164,4 @@ testing = ["big-O", "jaraco.functools", "jaraco.itertools", "more-itertools", "p
 [metadata]
 lock-version = "2.0"
 python-versions = "^3.9"
-content-hash = "c3beb00fbf2a4056e5190c2bd7dce5306b66f1e78bd19a45a05597c503867946"
+content-hash = "c50363fa85c218aff7d2414c8540a1d019ac74f4d27fa6e6973db2ebf5b71a9a"
diff --git a/pyproject.toml b/pyproject.toml
index 8223df4..62b534a 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -18,6 +18,7 @@ sentencepiece = "^0.1.99"
 einops = "^0.6.1"
 openai = "^0.27.8"
 pydantic = "<2.0.0"
+human-eval = {path = "human-eval", develop = true}
 
 
 [tool.poetry.group.dev.dependencies]
diff --git a/tests/eval/__init__.py b/tests/eval/__init__.py
new file mode 100644
index 0000000..e69de29
diff --git a/tests/eval/test_evaluation.py b/tests/eval/test_evaluation.py
new file mode 100644
index 0000000..df61393
--- /dev/null
+++ b/tests/eval/test_evaluation.py
@@ -0,0 +1,26 @@
+from textbook import evaluate
+from textbook.model import Replit
+
+
+def test_evaluate(monkeypatch):
+    # Define a replacement function to be used in the test
+    def mock_generate_one_completion(
+        model, tokenizer, prompt, max_new_tokens: int = 512
+    ):
+        return "\n  return 1"
+
+    # Monkey patch the 'add_numbers' function with the 'mock_add_numbers' function
+    monkeypatch.setattr(
+        evaluate, "generate_one_completion", mock_generate_one_completion
+    )
+
+    replit = Replit(debug=True)
+    accuracy_results, results = evaluate.evaluate(
+        model=replit.model,
+        tokenizer=replit.tokenizer,
+        eval_file="human-eval/data/example_problem.jsonl",
+    )
+
+    assert accuracy_results["pass@1"] == 1
+    assert results["test/0"]["passed"]
+    assert results["test/0"]["result"] == "passed"
diff --git a/textbook/evaluate.py b/textbook/evaluate.py
new file mode 100644
index 0000000..0378c41
--- /dev/null
+++ b/textbook/evaluate.py
@@ -0,0 +1,96 @@
+import json
+import re
+import tempfile
+from typing import Optional, Union, List
+
+import torch
+from transformers import PreTrainedTokenizer, PreTrainedModel
+from human_eval.data import write_jsonl, read_problems, HUMAN_EVAL
+from human_eval.evaluation import evaluate_functional_correctness
+
+if torch.cuda.is_available():
+    device = "cuda"
+else:
+    device = "cpu"
+
+
+def read_jsonl_file(file_path):
+    data = []
+    with open(file_path, "r") as f:
+        for line in f:
+            json_data = json.loads(line)
+            data.append(json_data)
+    return data
+
+
+def generate_one_completion(
+    model: PreTrainedModel,
+    tokenizer: PreTrainedTokenizer,
+    prompt: str,
+    max_new_tokens: int = 512,
+) -> List[str]:
+    inputs = tokenizer(prompt, return_tensors="pt").to(model.device)
+
+    generation_output = model.generate(
+        **inputs,
+        max_new_tokens=max_new_tokens,
+        eos_token_id=tokenizer.eos_token_id,
+        return_dict_in_generate=True,
+    )
+
+    s = generation_output.sequences[0]
+    output = tokenizer.decode(s, skip_special_tokens=True)
+    matches = list(re.finditer(r"\bdef\s+\w+\s*\(", output))
+    if len(matches) > 1:
+        output = output[: matches[1].span()[0]]
+
+    if not output.startswith(prompt):
+        print("output will not be cleaned properly:", output)
+    else:
+        output = output[len(prompt) :]
+    return output
+
+
+def evaluate(
+    model: Union[torch.nn.Module, PreTrainedModel],
+    tokenizer: PreTrainedTokenizer,
+    prompt_template: str = "{prompt}",
+    eval_file: str = HUMAN_EVAL,
+    eval_size: Optional[int] = None,
+    max_new_tokens: int = 512,
+):
+    model.eval()
+    problems = read_problems(evalset_file=eval_file)
+    eval_size = eval_size or len(list(problems.items()))
+    problems = dict(list(problems.items())[:eval_size])
+
+    # since k=1, no need for more samples
+    num_samples_per_task = 1
+    samples = [
+        dict(
+            task_id=task_id,
+            completion=generate_one_completion(
+                model,
+                tokenizer,
+                prompt_template.format(prompt=problems[task_id]["prompt"]),
+                max_new_tokens=max_new_tokens,
+            ),
+        )
+        for task_id in problems
+        for _ in range(num_samples_per_task)
+    ]
+    with tempfile.NamedTemporaryFile(mode="w", delete=False) as temp_file:
+        write_jsonl(temp_file.name, samples)
+
+        accuracy_results = evaluate_functional_correctness(
+            temp_file.name, k=[1], problem_file=eval_file, problems=problems
+        )
+        sample_results = read_jsonl_file(f"{temp_file.name}_results.jsonl")
+
+    # merge results and problems
+    results = {
+        item["task_id"]: {**item, **problems[item["task_id"]]}
+        for item in sample_results
+    }
+
+    return accuracy_results, results
diff --git a/textbook/train.py b/textbook/train.py
index 34abf73..0c16f08 100644
--- a/textbook/train.py
+++ b/textbook/train.py
@@ -6,6 +6,7 @@
 import torch
 
 from textbook.dataset import DummyDataset
+from textbook.evaluate import evaluate
 from textbook.model import BaseModule
 
 import transformers
@@ -50,6 +51,8 @@ def train(
     local_rank: Annotated[int, typer.Option("--local_rank")] = 0,
     deepspeed: Optional[str] = None,
     debug: bool = False,
+    eval_size: Optional[int] = None,
+    eval_max_new_tokens: int = 512,
 ):
     module_cls: Type[BaseModule] = getattr(import_module("textbook.model"), module)
     module_instance = module_cls(debug=debug)
@@ -76,6 +79,8 @@ def train(
     use_wandb = local_rank == 0 and use_wandb
     if use_wandb:
         run = wandb.init(wandb_project, **dict(config=config_to_log))  # type: ignore
+    else:
+        run = None  # type: ignore
 
     trainer = transformers.Trainer(
         model=model,
@@ -104,7 +109,22 @@ def train(
 
     trainer.train()
 
-    if use_wandb and wandb_log_model:
+    accuracy_results, sample_results = evaluate(
+        model, tokenizer, eval_size=eval_size, max_new_tokens=eval_max_new_tokens
+    )
+
+    if use_wandb and run and wandb_log_model:
+        # log accuracy@k results
+        run.log(accuracy_results)
+
+        # log sample values
+        results = list(sample_results.values())
+        columns = list(results[0].keys())
+        results_data = [[result[key] for key in columns] for result in results]
+        eval_table = wandb.Table(columns=columns, data=results_data)
+        run.log({"Evaluation": eval_table})
+
+        # upload model weights
         artifact = wandb.Artifact(name="model_weight", type="model")
         artifact.add_dir(output_dir)
         run.log_artifact(artifact)  # type: ignore