from sovara import SovaraClient, trace
client = SovaraClient(project_name="capital-answers")
cases = [
("What is the capital of France?", "Paris"),
("What is the capital of Italy?", "Rome"),
]
@trace
def answer(question: str) -> str:
# Replace this example with your agent. Supported model calls are traced too.
return {
"What is the capital of France?": "Paris",
"What is the capital of Italy?": "Rome",
}[question]
eval_run_id = client.create_eval_run()
for question, expected in cases:
with client.run("capital-answer", eval_run_id=eval_run_id) as run_key:
if run_key is None:
raise RuntimeError("Sovara could not record this sample")
client.log(run_key=run_key, run_input=question, groundtruth=expected)
output = answer(question)
correct = output == expected
client.log(
run_key=run_key,
run_output=output,
llm_judge_is_correct=correct,
llm_judge_output="Exact match" if correct else "Does not match",
test_set="capital-questions",
)
client.log(eval_run_id=eval_run_id, test_set="capital-questions", dataset_version="v1")
print(f"Recorded evaluation {eval_run_id}")