Import necessary libraries

In [1]:
import requests
import json

In [2]:
# the below url is where our llama server is hosted
url = "http://127.0.0.1:8000/generate/"


# Define the data payload
data = {
    "impression": [
        {"role": "system", "content": "You are an AI assistant whose role is to listen to the user"},
        {"role": "user", "content": "Write a poem in 50 words about sun" },
    ]}

# Send the POST request
response = requests.post(url, headers={"Content-Type": "application/json"}, data=json.dumps(data))


In [3]:
print(response.status_code)
print(response.json())

200
{'generated_text': 'Golden hues upon the sky,\nThe sun rises high and high,\nWarming the earth with gentle might,\nBringing life to all in sight.\nIts rays dance upon the sea,\nA beautiful sight for you and me,\nA symbol of hope and light,\nThe sun shines bright, a wondrous sight.'}


Now define evaluation prompt

In [23]:
def prompt_qa(impression):
  return [{
    "role": "system", "content": "You are a radiologist whose job is to do quality assessment of the ground truth of reports"},{
    "role": "user", "content": """You are a radiologist expert. Your task is to do quality assessment of ground truth of radiological reports. You will be given a radiological report and a dictionary of binary ground truth for findings. If there is a mismatch for any findings, you return that finding, else the output should be "na". 
    1. Do not use any python code or give any output related to python code.
    2. You have to understand the report and follow step wise analysis.
    3. Give only dictionary as an output, no other text.
    Here is the radiological report with the output.
    Radiological report: Right lung volume loss with mild elevation right hemidiaphragm and rightward shift of heart and mediastinum. Spiculated 1.6 x 1.8 cm noncalcified nodule peripheral aspect right mid lung zone. Mild scarring/subsegmental volume loss right lung base. Left lung is clear. Possible small left pleural effusion. No right pleural abnormality. Heart size and pulmonary venous pressure are normal. Suspect right retrohilar mass, measuring approximately 4.4 cm in diameter, with possible bronchial narrowing. Left hilum and mediastinal structures are otherwise normal. Mild dextrocurvature of thoracic spine.
    Ground Truth: {"Cardiomegaly": 0, "Cavity": 0, "Consolidation": 0, "Fibrosis": 0, "Hilar Prominence": 1, "Nodule": 0, "Opacity": 0, "Pleural Effusion": 1, "Raised/Tented Diaphragm": 0, "Scoliosis": 1, "Atelectasis": 1, "Rib Fracture": 0, "Pneumothorax": 0}
    Chain of thought: Since nodule is present in the radiological report but ground truth mentions that there is no nodule. So, the output should be nodule
    Output should be: {"Nodule"}
    Here is an another example:
    Radiological report: Right lung volume loss with mild elevation right hemidiaphragm and rightward shift of heart and mediastinum. Spiculated 1.6 x 1.8 cm noncalcified nodule peripheral aspect right mid lung zone. Mild scarring/subsegmental volume loss right lung base. Left lung is clear. Possible small left pleural effusion. No right pleural abnormality. Heart size and pulmonary venous pressure are normal. Suspect right retrohilar mass, measuring approximately 4.4 cm in diameter, with possible bronchial narrowing. Left hilum and mediastinal structures are otherwise normal. Mild dextrocurvature of thoracic spine.
    Ground Truth: {"Cardiomegaly": 1, "Cavity": 0, "Consolidation": 0, "Fibrosis": 0, "Hilar Prominence": 1, "Nodule": 1, "Opacity": 0, "Pleural Effusion": 1, "Raised/Tented Diaphragm": 0, "Scoliosis": 1, "Atelectasis": 1, "Rib Fracture": 0, "Pneumothorax": 0}
    Chain of thought: Since there is no cardiomegaly in the radiological report but ground truth mentions that there is cardiomegaly and nodule. So, the output should be cardiomegaly.
    Output should be: {"Cardiomegaly"}
    1. Do not use any python code or give any output related to python code.
    2. You have to understand the report and follow step wise analysis.
    3. Give only dictionary as an output, no other text. 
    Now, please read the following radiology report and then analyze the binary findings and provide any discrepencies:""" + impression + """
    Please make sure return only dictionary , do not return any other text or explaination"""},
  ]


In [26]:
text = """Radiological report: 4.3 mm calcified granuloma left lung base. Lungs are otherwise clear. No pleural abnormality. Heart size and pulmonary venous pressure are normal. 1.3 x 1.8 cm calcified subaortic lymph node. Hilar and mediastinal structures are otherwise normal. Dextrocurvature of thoracic spine mild degenerative changes.```. Ground truth: {"Cardiomegaly": 1, "Cavity": 0, "Consolidation": 0, "Fibrosis": 0, "Hilar Prominence": 0, "Nodule": 0, "Opacity": 0, "Pleural Effusion": 0, "Raised/Tented Diaphragm": 0, "Scoliosis": 1, "Atelectasis": 0, "Rib Fracture": 0, "Pneumothorax": 0}. Output should be:"""

In [27]:
try:
    prompt = prompt_qa(text)
    # print(impressions[i])
    response = requests.post(url, headers={"Content-Type": "application/json"}, data=json.dumps({'impression' : prompt}))
    # print(response.json()['generated_text'])
    # 'a' for append mode
    print(response.json()["generated_text"])
except:
    print("fail")

{"Cardiomegaly"}


In [32]:
def prompt_qa(impression, ground_truth):
  return [
{
    "role": "system", 
    "content": "You are a radiologist whose job is to do quality assessment of the ground truth of reports."
},
{
    "role": "user",
    "content": """
You are a radiologist expert. Your task is to perform a meticulous quality assessment of the binary ground truth associated with radiological reports. I will provide you with a radiology report and a dictionary representing the ground truth for various findings. You will compare the text of the report with the ground truth dictionary to identify any discrepancies between reported findings and the ground truth. Your responses should meet the following requirements:

1. **Procedure**:
   - Conduct a systematic, step-by-step assessment of each term in the ground truth dictionary, directly referencing and cross-checking phrases or descriptions from the radiological report.
   - Carefully analyze context within the report to avoid incorrect interpretations of terms such as "no evidence of" or "mild signs of."
   - Determine each finding's presence or absence based solely on the evidence provided in the radiological report and indicate discrepancies.
2. **Response Structure**: Format output should be an exaplaination and a dictionary output
3. **Examples**:
   - Example 1:
     Radiological report: "Right lung volume loss with mild elevation right hemidiaphragm and rightward shift of heart and mediastinum. Spiculated 1.6 x 1.8 cm noncalcified nodule peripheral aspect right mid lung zone. Mild scarring/subsegmental volume loss right lung base. Left lung is clear. Possible small left pleural effusion. No right pleural abnormality. Heart size and pulmonary venous pressure are normal. Suspect right retrohilar mass, measuring approximately 4.4 cm in diameter, with possible bronchial narrowing. Left hilum and mediastinal structures are otherwise normal. Mild dextrocurvature of thoracic spine."
     Ground Truth: {"Cardiomegaly": 0, "Cavity": 0, "Consolidation": 0, "Fibrosis": 0, "Hilar Prominence": 1, "Nodule": 0, "Opacity": 0, "Pleural Effusion": 1, "Raised/Tented Diaphragm": 0, "Scoliosis": 1, "Atelectasis": 1, "Rib Fracture": 0, "Pneumothorax": 0}
     The output should be:
     Explaination: Since non calcified nodule is present in the radiological report but ground truth mentions that there is no nodule. So, there is discrepency. The output should contain nodule.
     Output: {"Nodule"}
   - Example 2:
     Radiological report: "Right lung volume loss with mild elevation right hemidiaphragm and rightward shift of heart and mediastinum. Spiculated 1.6 x 1.8 cm noncalcified nodule peripheral aspect right mid lung zone. Mild scarring/subsegmental volume loss right lung base. Left lung is clear. Possible small left pleural effusion. No right pleural abnormality. Heart size and pulmonary venous pressure are normal. Suspect right retrohilar mass, measuring approximately 4.4 cm in diameter, with possible bronchial narrowing. Left hilum and mediastinal structures are otherwise normal. Mild dextrocurvature of thoracic spine."
     Ground Truth: {"Cardiomegaly": 1, "Cavity": 0, "Consolidation": 0, "Fibrosis": 0, "Hilar Prominence": 1, "Nodule": 1, "Opacity": 0, "Pleural Effusion": 1, "Raised/Tented Diaphragm": 0, "Scoliosis": 1, "Atelectasis": 1, "Rib Fracture": 0, "Pneumothorax": 0}
     Explaination:  Since there is no cardiomegaly in the radiological report but ground truth mentions that there is cardiomegaly. So, the output should be cardiomegaly.
     Output: {"Cardiomegaly"}
     
Please review the following radiology report and analyze the binary findings, providing the discrepency and output in dictionary format:
Radiological report: """ + impression + """
Ground Truth: """ + ground_truth + """
"""}]
impression = """Radiological report: 4.3 mm calcified granuloma left lung base. Lungs are otherwise clear. No pleural abnormality. Heart size and pulmonary venous pressure are normal. 1.3 x 1.8 cm calcified subaortic lymph node. Hilar and mediastinal structures are otherwise normal. Dextrocurvature of thoracic spine mild degenerative changes"""
ground_truth = """{"Cardiomegaly": 1, "Cavity": 0, "Consolidation": 0, "Fibrosis": 0, "Hilar Prominence": 0, "Nodule": 0, "Opacity": 0, "Pleural Effusion": 0, "Raised/Tented Diaphragm": 0, "Scoliosis": 1, "Atelectasis": 0, "Rib Fracture": 0, "Pneumothorax": 0}"""
try:
    prompt = prompt_qa(impression, ground_truth)
    # print(impressions[i])
    response = requests.post(url, headers={"Content-Type": "application/json"}, data=json.dumps({'impression' : prompt}))
    # print(response.json()['generated_text'])
    # 'a' for append mode
    print(response.json()["generated_text"])
except:
    print("fail")

Explaination: 
The radiological report mentions a 4.3 mm calcified granuloma in the left lung base, which is a type of nodule. However, the ground truth dictionary indicates that there is no nodule. Additionally, the report mentions a 1.3 x 1.8 cm calcified subaortic lymph node, which is also a type of nodule. However, the ground truth dictionary indicates that there is no nodule. Furthermore, the report does not mention cardiomegaly, but the ground truth dictionary indicates that there is cardiomegaly.

Output: 
{"Cardiomegaly", "Nodule"}
