examples = [ {"question": "What is the capital of France?", "expected": "Paris"}, {"question": "Who wrote 'To Kill a Mockingbird'?", "expected": "Harper Lee"}, {"question": "What is the square root of 64?", "expected": "8"},]
import weave# Collect your examplesexamples = [ {"question": "What is the capital of France?", "expected": "Paris"}, {"question": "Who wrote 'To Kill a Mockingbird'?", "expected": "Harper Lee"}, {"question": "What is the square root of 64?", "expected": "8"},]# Define any custom scoring function@weave.op()def match_score1(expected: str, output: dict) -> dict: # Here is where you'd define the logic to score the model output return {'match': expected == output['generated_text']}
from weave import Model, Evaluationimport asyncioclass MyModel(Model): prompt: str @weave.op() def predict(self, question: str): # here's where you would add your LLM call and return the output return {'generated_text': 'Hello, ' + self.prompt}model = MyModel(prompt='World')evaluation = Evaluation( dataset=examples, scorers=[match_score1])weave.init('intro-example') # begin tracking results with weaveasyncio.run(evaluation.evaluate(model))
@weave.opdef function_to_evaluate(question: str): # here's where you would add your LLM call and return the output return {'generated_text': 'some response'}asyncio.run(evaluation.evaluate(function_to_evaluate))
from weave import Evaluation, Modelimport weaveimport asyncioweave.init('intro-example')examples = [ {"question": "What is the capital of France?", "expected": "Paris"}, {"question": "Who wrote 'To Kill a Mockingbird'?", "expected": "Harper Lee"}, {"question": "What is the square root of 64?", "expected": "8"},]@weave.op()def match_score1(expected: str, output: dict) -> dict: return {'match': expected == output['generated_text']}@weave.op()def match_score2(expected: dict, output: dict) -> dict: return {'match': expected == output['generated_text']}class MyModel(Model): prompt: str @weave.op() def predict(self, question: str): # here's where you would add your LLM call and return the output return {'generated_text': 'Hello, ' + question + self.prompt}model = MyModel(prompt='World')evaluation = Evaluation(dataset=examples, scorers=[match_score1, match_score2])asyncio.run(evaluation.evaluate(model))@weave.op()def function_to_evaluate(question: str): # here's where you would add your LLM call and return the output return {'generated_text': 'some response' + question}asyncio.run(evaluation.evaluate(function_to_evaluate("What is the capitol of France?")))