from typing import Literal
from dotenv import load_dotenv
from pydantic import BaseModel
from config import make_llm_client
from evaluation.evaluation_utils import llm_structured_retry
litellm_client = make_llm_client()
class RelevanceVerdict(BaseModel):
relevance: Literal["NON_RELEVANT", "PARTLY_RELEVANT", "RELEVANT"]
explanation: str
judge_instructions = """
You are an expert evaluator for a RAG system.
Analyze the relevance of the generated answer to the given question.
Classify the answer as:
- RELEVANT: the answer addresses the question
- PARTLY_RELEVANT: the answer partially addresses the question
- NON_RELEVANT: the answer does not address the question
""".strip()
judge_prompt = """
Question: {question}
Generated Answer: {answer}
""".strip()
def evaluate_relevance(question, answer, client=None):
if client is None:
client = litellm_client
prompt = judge_prompt.format(
question=question,
answer=answer
)
result, usage = llm_structured_retry(
client,
judge_instructions,
prompt,
RelevanceVerdict,
)
return result.relevance, result.explanation
if __name__ == "__main__":
load_dotenv()
question = "What if I got a bad whack to my shoulder and my friend said I can use a crutch on the good side? Is that a mistake?"
answer = "You should only move an injured person if there is immediate danger such as a fire, oncoming traffic, or toxic fumes. Otherwise, it's best to leave them where they are, administer first aid on the spot, and wait for professional medical help to arrive."
relevance, explanation = evaluate_relevance(question, answer)
print(relevance)
print(explanation)