import json
import boto3
import re
from dotenv import load_dotenv
load_dotenv()
# ------------------------
# Bedrock client
# ------------------------
bedrock = boto3.client(
service_name="bedrock-runtime",
region_name="ca-central-1"
)
# ------------------------
# System prompt (ANSWER ONLY)
# ------------------------
SYSTEM_PROMPT = """
You are a medical information assistant.
RULES:
- Use ONLY information explicitly stated in the medical notes.
- Do NOT invent facts, dosages, or recommendations.
- Do NOT include evidence, citations, or formatting.
- If the notes do not clearly answer the question, say:
"The provided documents do not contain this information."
OUTPUT:
- Write ONLY a short natural-language answer (1–2 sentences).
"""
# ------------------------
# Helpers
# ------------------------
STOPWORDS = {
"the","a","an","and","or","to","of","in","on","for","with",
"who","has","have","is","was","were","that","this","what",
"where","when","how","why","which","patient","male","female"
}
def extract_keywords(text, limit=4):
words = re.findall(r"\b[a-zA-Z]+\b", text.lower())
keywords = [w for w in words if w not in STOPWORDS]
return keywords[:limit]
def extract_numeric_anchors(text):
return re.findall(r"\b\d+\b", text)
def select_evidence_with_anchors(answer_text, query, retrieved_results):
"""
Anchor-based evidence selection:
1) Numeric anchors (age, dosage, time)
2) Secondary keyword anchors (3–4 query keywords)
3) Deduplication
"""
numeric_anchors = extract_numeric_anchors(answer_text)
keyword_anchors = extract_keywords(query)
matched = []
for ev in retrieved_results:
content = ev["content"].strip()
content_lower = content.lower()
# Primary numeric anchor match
if numeric_anchors:
if not any(num in content for num in numeric_anchors):
continue
# Secondary keyword anchor match
if keyword_anchors:
if not any(k in content_lower for k in keyword_anchors):
continue
matched.append(content)
# Deduplicate while preserving order
seen = set()
deduped = []
for m in matched:
if m not in seen:
deduped.append(m)
seen.add(m)
return deduped
# ------------------------
# Prompt builder
# ------------------------
def build_prompt(query, retrieved_results):
notes = "\n".join(f"- {r['content']}" for r in retrieved_results)
return f"""
Medical notes:
{notes}
Question:
{query}
Write the answer following the rules above.
"""
# ------------------------
# Main generation function
# ------------------------
def generate_answer(input_data):
query = input_data["query"]
retrieved_results = input_data.get("retrieved_results", [])
if not retrieved_results:
return f'''Query:
"{query}"
Answer:
The provided documents do not contain this information.
'''
prompt = build_prompt(query, retrieved_results)
response = bedrock.invoke_model(
modelId="anthropic.claude-3-haiku-20240307-v1:0",
body=json.dumps({
"anthropic_version": "bedrock-2023-05-31",
"max_tokens": 300,
"temperature": 0,
"system": SYSTEM_PROMPT,
"messages": [
{"role": "user", "content": prompt}
]
}),
contentType="application/json",
accept="application/json"
)
response_body = json.loads(response["body"].read())
answer_text = response_body["content"][0]["text"].strip()
# ------------------------
# Evidence selection (ANCHOR BASED)
# ------------------------
matched_evidence = select_evidence_with_anchors(
answer_text,
query,
retrieved_results
)
# ------------------------
# Final output assembly
# ------------------------
output_lines = []
output_lines.append("Query:")
output_lines.append(f' "{query}"\n')
output_lines.append("Answer:")
output_lines.append(f" {answer_text}\n")
for ev in matched_evidence:
output_lines.append("Evidence:")
output_lines.append(f" {ev}\n")
return "\n".join(output_lines)
# ------------------------
# Entry point
# ------------------------
if __name__ == "__main__":
with open("parsed_input.json", "r") as f:
input_data = json.load(f)
final_answer = generate_answer(input_data)
with open("final_answer.txt", "w") as f:
f.write(final_answer)
print("\n" + final_answer + "\n")