Cognitive-rag / 05_generation / evaluate_ragas.py
evaluate_ragas.py
Raw
import os
import json
from datasets import Dataset
from dotenv import load_dotenv

load_dotenv()

if not os.getenv("OPENAI_API_KEY"):
    raise ValueError("OPENAI_API_KEY not found.")

# ------------------------
# RAGAS imports
# ------------------------
from ragas import evaluate
from ragas.metrics import (
    Faithfulness,
    AnswerRelevancy,
    ContextPrecision,
    ContextRecall,
)
from ragas.llms import llm_factory
from langchain_openai import OpenAIEmbeddings  # Use langchain embeddings instead
from openai import OpenAI

# ------------------------
# Load files
# ------------------------
BASE_DIR = os.path.dirname(__file__)

retrieval_path = os.path.join(
    BASE_DIR,
    "run_retrieval",
    "output",
    "retrieval_output.json"
)

answer_path = os.path.join(BASE_DIR, "final_answer.txt")

with open(retrieval_path, "r") as f:
    retrieval_data = json.load(f)

with open(answer_path, "r") as f:
    answer_text = f.read().strip()

# ------------------------
# Build dataset
# ------------------------
ragas_data = {
    "question": [retrieval_data["query"]],
    "answer": [answer_text],
    "contexts": [[
        *retrieval_data.get("retrieved_chunks", []),
        *retrieval_data.get("related_chunks", []),
    ]],
    "ground_truth": [retrieval_data.get("ground_truth", "")],
}

dataset = Dataset.from_dict(ragas_data)

print("\nQUESTION:")
print(retrieval_data["query"])

print("\nFIRST RETRIEVED CHUNK:")
print(ragas_data["contexts"][0][0][:500])

print("\nANSWER:")
print(answer_text[:500])


# ------------------------
# LLM + Embeddings - FIXED
# ------------------------
openai_client = OpenAI()

# Increase max_tokens to avoid truncation
llm = llm_factory(
    "gpt-4o-mini",
    client=openai_client,
    max_tokens=8000  # Increase this limit
)

# Use langchain's OpenAIEmbeddings (has proper methods)
embeddings = OpenAIEmbeddings(
    model="text-embedding-3-small"
)

# ------------------------
# Instantiate metrics
# ------------------------
faithfulness_metric = Faithfulness(llm=llm)

answer_relevancy_metric = AnswerRelevancy(
    llm=llm,
    embeddings=embeddings
)

context_precision_metric = ContextPrecision(llm=llm)

context_recall_metric = ContextRecall(llm=llm)

# ------------------------
# Run evaluation
# ------------------------
print("\nRunning RAGAS evaluation...")
print("=" * 60)

results = evaluate(
    dataset,
    metrics=[
        faithfulness_metric,
        answer_relevancy_metric,
        context_precision_metric,
        context_recall_metric,
    ],
)

print("\nRAGAS Evaluation Results")
print("=" * 60)
print(results)