In [3]:
import gc
gc.collect()

0

In [4]:
import warnings
warnings.filterwarnings("ignore")


In [5]:
import os
from openai import OpenAI

import json
from tqdm import tqdm
import pandas as pd

model_name = "medgemma-4b-it"
client = OpenAI(
    base_url="http://localhost:1234/v1",
    api_key="lm-studio",
)


df = pd.read_csv(r"./medical_conversations_with_gemma_response.csv", encoding="utf-8")
df.head()

Unnamed: 0,conversations,disease,gemma_response
0,User: I've been sneezing a lot today and my no...,allergy,allergy
1,User: I've developed a rash after eating some ...,allergy,allergy
2,"User: My eyes are swollen and itchy, and I can...",allergy,allergy
3,User: I've been getting headaches and a stuffy...,allergy,allergy
4,"User: Every time I eat nuts, my mouth itches. ...",allergy,food allergy


In [11]:
diseases_list_trimmed = list(df['disease'].unique())
len(diseases_list_trimmed)

24

In [12]:
diseases_list_trimmed

['allergy',
 'bronchial asthma',
 'malaria',
 'impetigo',
 'varicose veins',
 'diabetes',
 'drug reaction',
 'psoriasis',
 'pneumonia',
 'jaundice',
 'migraine',
 'urinary tract infection',
 'arthritis',
 'peptic ulcer disease',
 'cervical spondylosis',
 'chicken pox',
 'typhoid',
 'dimorphic hemorrhoids',
 'hypertension',
 'gastroesophageal reflux disease',
 'acne',
 'fungal infection',
 'common cold',
 'dengue']

In [6]:
diseases_list = list(df['gemma_response'].unique())
len(diseases_list)

91

In [14]:
user_input = "allergy"
system_prompt = f"""
You are an expert in Medical, Healthcare, Pharma Data Creation.
Generate a conversation dataset between Doctor and a patient for the given disease by user.
The conversation can be anything from generic to advanced topic.
Generate the output in JSON format.

Example:
{
    {"role": "doctor", "content": "Good morning! What brings you in today?"},
    {"role": "patient", "content": "Hi Doctor, I've been experiencing some persistent cough and shortness of breath for about a week now."},
    {"role": "doctor", "content": "Okay. Can you describe the cough? Is it dry or productive?"},
    {"role": "doctor", "content": "And when did this start?"}
}

Note: Generate the output using the above example:
"""



In [33]:
response = client.chat.completions.create(
    model=model_name,
    messages=[
        {"role": "system", "content": system_prompt},
        {"role": "user", "content": user_input}
    ],
)

print(response.choices[0].message.content)

```json
[
  {
    "role": "doctor",
    "content": "Good morning, Mr. Jones. What can I do for you today?"
  },
  {
    "role": "patient",
    "content": "Hi Doctor, I've been having some strange reactions lately. It started with a rash after eating peanuts."
  },
  {
    "role": "doctor",
    "content": "Okay, and what kind of reaction was it? Was it itchy, painful, or something else?"
  },
  {
    "role": "patient",
    "content": "It was a red, raised rash all over my body. It felt itchy too."
  },
  {
    "role": "doctor",
    "content": "Did you experience any other symptoms like difficulty breathing or swelling?"
  },
  {
    "role": "patient",
    "content": "No, nothing like that. Just the rash and itching."
  },
  {
    "role": "doctor",
    "content": "Alright. Have you had this reaction before? And what other foods have you eaten recently?"
  },
  {
    "role": "patient",
    "content": "No, this is the first time I've reacted to peanuts. I ate a peanut butter sandwich yeste

In [34]:
json_op = response.choices[0].message.content
json_op_filtered = json_op.replace("json", "").replace("```", "")
json_op_filtered

'\n[\n  {\n    "role": "doctor",\n    "content": "Good morning, Mr. Jones. What can I do for you today?"\n  },\n  {\n    "role": "patient",\n    "content": "Hi Doctor, I\'ve been having some strange reactions lately. It started with a rash after eating peanuts."\n  },\n  {\n    "role": "doctor",\n    "content": "Okay, and what kind of reaction was it? Was it itchy, painful, or something else?"\n  },\n  {\n    "role": "patient",\n    "content": "It was a red, raised rash all over my body. It felt itchy too."\n  },\n  {\n    "role": "doctor",\n    "content": "Did you experience any other symptoms like difficulty breathing or swelling?"\n  },\n  {\n    "role": "patient",\n    "content": "No, nothing like that. Just the rash and itching."\n  },\n  {\n    "role": "doctor",\n    "content": "Alright. Have you had this reaction before? And what other foods have you eaten recently?"\n  },\n  {\n    "role": "patient",\n    "content": "No, this is the first time I\'ve reacted to peanuts. I ate a 

In [35]:
len(json_op_filtered)

2119

In [36]:
var = json.loads(json_op_filtered)
var

[{'role': 'doctor',
  'content': 'Good morning, Mr. Jones. What can I do for you today?'},
 {'role': 'patient',
  'content': "Hi Doctor, I've been having some strange reactions lately. It started with a rash after eating peanuts."},
 {'role': 'doctor',
  'content': 'Okay, and what kind of reaction was it? Was it itchy, painful, or something else?'},
 {'role': 'patient',
  'content': 'It was a red, raised rash all over my body. It felt itchy too.'},
 {'role': 'doctor',
  'content': 'Did you experience any other symptoms like difficulty breathing or swelling?'},
 {'role': 'patient',
  'content': 'No, nothing like that. Just the rash and itching.'},
 {'role': 'doctor',
  'content': 'Alright. Have you had this reaction before? And what other foods have you eaten recently?'},
 {'role': 'patient',
  'content': "No, this is the first time I've reacted to peanuts. I ate a peanut butter sandwich yesterday and then started getting the rash."},
 {'role': 'doctor',
  'content': 'Okay. Do you have 

In [44]:
var1 = {}
var1['trial1'] = var
var1['trial2'] = var
var1

{'trial1': [{'role': 'doctor',
   'content': 'Good morning, Mr. Jones. What can I do for you today?'},
  {'role': 'patient',
   'content': "Hi Doctor, I've been having some strange reactions lately. It started with a rash after eating peanuts."},
  {'role': 'doctor',
   'content': 'Okay, and what kind of reaction was it? Was it itchy, painful, or something else?'},
  {'role': 'patient',
   'content': 'It was a red, raised rash all over my body. It felt itchy too.'},
  {'role': 'doctor',
   'content': 'Did you experience any other symptoms like difficulty breathing or swelling?'},
  {'role': 'patient',
   'content': 'No, nothing like that. Just the rash and itching.'},
  {'role': 'doctor',
   'content': 'Alright. Have you had this reaction before? And what other foods have you eaten recently?'},
  {'role': 'patient',
   'content': "No, this is the first time I've reacted to peanuts. I ate a peanut butter sandwich yesterday and then started getting the rash."},
  {'role': 'doctor',
   'c

In [45]:
with open("dummy.json", "w") as f:
    json.dump(var1, f)

# Setting up the pipeline

In [8]:
diseases_list

['allergy',
 'food allergy',
 'asthma',
 'contact allergy',
 'skin allergy',
 'eczema',
 'celiac disease',
 'malaria',
 'impetigo',
 'Varicose veins',
 'varicose veins',
 'diabetes',
 'neuropathy',
 'acanthosis nigricans',
 'hypoglycemia',
 'drug allergy',
 'dizziness',
 'drug rash',
 'nausea',
 'dry mouth',
 'edema',
 'Side effect',
 'gastroenteritis',
 'insomnia',
 'weight gain',
 'anxiety',
 'side effect',
 'tremors',
 'cardiac arrhythmia',
 'photosensitivity',
 'headache',
 'drug-induced skin reaction',
 'Urinary incontinence',
 'medication side effect',
 'drug reaction',
 'drowsiness',
 'rash',
 'dehydration',
 'medication side effects',
 'psoriasis',
 'dermatitis',
 'pneumonia',
 'jaundice',
 'migraine',
 'menstrual migraine',
 'UTI',
 'urinary_tract_infection',
 'arthritis',
 'rheumatoid arthritis',
 'osteoarthritis',
 'back pain',
 'peptic ulcer',
 'ulcer',
 'peptic ulcers',
 'gastrointestinal bleed',
 'acid reflux',
 'cervical spondylosis',
 'nerve compression',
 'chickenpox',

In [None]:
specified_disease_list = [
    "gastroenteritis",
    "osteoarthritis",
    
]

In [7]:
for disease in tqdm(diseases_list[:10]):
    print(disease)

100%|██████████| 10/10 [00:00<00:00, 4007.55it/s]

allergy
food allergy
asthma
contact allergy
skin allergy
eczema
celiac disease
malaria
impetigo
Varicose veins





In [15]:
medical_conversations = {}
for disease in tqdm(diseases_list_trimmed):
    if disease not in medical_conversations:
        response = client.chat.completions.create(
        model=model_name,
        messages=[
            {"role": "system", "content": system_prompt},
            {"role": "user", "content": user_input}
            ],
        )

        med_convo = response.choices[0].message.content
        med_convo = med_convo.replace("json", "").replace("```", "")
        medical_conversations[disease] = med_convo


100%|██████████| 24/24 [32:42<00:00, 81.79s/it] 


In [21]:
med_convo_filtered = {}
for disease in medical_conversations:
    json_op = json.loads(medical_conversations[disease])
    med_convo_filtered[disease] = json_op

In [23]:
with open("medical_conversation.json", "w") as f:
    json.dump(med_convo_filtered, f)