Recipe: evaluate code-generation models
Recipe — evaluate code-generation models. Use a code benchmark from the public catalog.
from layerlens import Stratix
client = Stratix()
# Pick a code benchmark from the public catalog
benchmark = client.benchmarks.get_by_key("humaneval") # or mbpp, swe-bench, etc.
model = client.models.get_by_key("openai/gpt-4o")
evaluation = client.evaluations.create(model=model, benchmark=benchmark)
evaluation = client.evaluations.wait_for_completion(evaluation, timeout_seconds=1800)
print(f"Accuracy: {evaluation.accuracy}")See also
Last updated
Was this helpful?