In [1]:
import re
import os
from langchain.vectorstores.neo4j_vector import Neo4jVector
from langchain.document_loaders import WikipediaLoader
from langchain.embeddings.openai import OpenAIEmbeddings
from langchain.text_splitter import (
    CharacterTextSplitter,
    RecursiveCharacterTextSplitter,
)
from langchain.graphs import Neo4jGraph
from transformers import AutoModelForSeq2SeqLM, AutoTokenizer
from langchain.text_splitter import TokenTextSplitter

from secret import *
import chromadb

os.environ["OPENAI_API_KEY"] = OPENAI_API_KEY

In [2]:
graph = Neo4jGraph(url="bolt://localhost:7687", username="neo4j", password="test1234")
graph.query(
    """
MATCH (n)
DETACH DELETE n
"""
)
graph.refresh_schema()
print(graph.schema)


        Node properties are the following:
        []
        Relationship properties are the following:
        []
        The relationships are the following:
        []
        


In [3]:
from loader.radiopaedia import RadiopaediaLoader
radio_raw_documents = RadiopaediaLoader(query="Atelectasis", diffbot_api_token=DIFFBOT_API_TOKEN).load()
wiki_raw_documents = WikipediaLoader(query="Atelectasis").load()
raw_documents = radio_raw_documents + wiki_raw_documents

In [4]:
len(radio_raw_documents), len(wiki_raw_documents)

(20, 10)

In [5]:
text_splitter = TokenTextSplitter(chunk_size=2048, chunk_overlap=24)
documents = text_splitter.split_documents(
    raw_documents
)  # should I only take the first? which is the most relevant?

In [6]:
len(raw_documents)

30

In [7]:
from tqdm import tqdm
from aug.graph_doc import (
    get_extraction_chain,
    data_to_graph_doc,
    chain_run,
    add_graph_documents,
)
from langchain.chat_models import ChatOpenAI

llm = ChatOpenAI(model="gpt-3.5-turbo-16k", temperature=0)

allowed_nodes = None
allowed_rels = None
# allowed_nodes = ["Symptom", "Disease"]
# allowed_rels = ["CAN_CAUSE", "DESCRIBE", "HAS"]

extract_chain = get_extraction_chain(llm, allowed_nodes, allowed_rels)
gds = []

for d in tqdm(documents, total=len(documents)):
    data = chain_run(extract_chain, d.page_content)
    # data = extract_chain.run(d.page_content)
    # graph_document = GraphDocument(
    #     nodes=[map_to_base_node(node) for node in data.nodes],
    #     relationships=[map_to_base_relationship(rel) for rel in data.rels],
    #     source=d,
    # )
    graph_document = data_to_graph_doc(data, d)
    # add_graph_document(graph, graph_document)
    gds.append(graph_document)

graph = add_graph_documents(graph, gds)
graph.refresh_schema()
print(graph.schema)

# 12m46.2s

100%|██████████| 30/30 [27:11<00:00, 54.39s/it]



        Node properties are the following:
        [{'labels': 'Condition', 'properties': [{'property': 'name', 'type': 'STRING'}, {'property': 'description', 'type': 'STRING'}, {'property': 'alsoknownas', 'type': 'STRING'}, {'property': 'id', 'type': 'STRING'}, {'property': 'alsoKnownAs', 'type': 'STRING'}, {'property': 'dueTo', 'type': 'STRING'}, {'property': 'causedBy', 'type': 'STRING'}, {'property': 'affects', 'type': 'STRING'}, {'property': 'definition', 'type': 'STRING'}, {'property': 'unilateral', 'type': 'STRING'}]}, {'labels': 'Section', 'properties': [{'property': 'name', 'type': 'STRING'}, {'property': 'id', 'type': 'STRING'}, {'property': 'title', 'type': 'STRING'}, {'property': 'content', 'type': 'STRING'}, {'property': 'description', 'type': 'STRING'}]}, {'labels': 'Symptom', 'properties': [{'property': 'name', 'type': 'STRING'}, {'property': 'id', 'type': 'STRING'}]}, {'labels': 'Cause', 'properties': [{'property': 'name', 'type': 'STRING'}, {'property': 'id', 'type': 

In [8]:
from langchain.chains import GraphCypherQAChain
chain = GraphCypherQAChain.from_llm(
    graph=graph,
    cypher_llm=ChatOpenAI(temperature=0, model="gpt-3.5-turbo-16k"),
    qa_llm=ChatOpenAI(temperature=0, model="gpt-3.5-turbo-16k"),
    validate_cypher=True,  # Validate relationship directions
    verbose=True,
)

In [9]:
chain.run("What is Atelectasis?")



[1m> Entering new GraphCypherQAChain chain...[0m
Generated Cypher:
[32;1m[1;3mMATCH (c:Condition {name: 'Atelectasis'})
RETURN c[0m
Full Context:
[32;1m[1;3m[{'c': {'name': 'Atelectasis', 'causedBy': 'normal exhalation or various medical conditions', 'affects': 'part or all of one lung', 'description': 'A condition where the alveoli are deflated down to little or no volume, as distinct from pulmonary consolidation, in which they are filled with liquid', 'definition': 'The collapse or closure of a lung resulting in reduced or absent gas exchange', 'unilateral': 'true', 'id': 'Atelectasis', 'alsoKnownAs': 'collapsed lung'}}][0m

[1m> Finished chain.[0m


'Atelectasis is a condition where part or all of one lung collapses or closes, resulting in reduced or absent gas exchange. It can be caused by normal exhalation or various medical conditions. The alveoli, which are small air sacs in the lungs, become deflated down to little or no volume. Atelectasis is also known as a collapsed lung.'

In [10]:
chain.run("What are the symptoms of Atelectasis?")



[1m> Entering new GraphCypherQAChain chain...[0m
Generated Cypher:
[32;1m[1;3mMATCH (c:Condition {name: 'Atelectasis'})-[:HAS_SYMPTOM]->(s:Symptom)
RETURN s.name[0m
Full Context:
[32;1m[1;3m[][0m

[1m> Finished chain.[0m


"I'm sorry, but I don't have the information to answer your question. It would be best to consult a medical professional for accurate information on the symptoms of Atelectasis."

In [11]:
chain.run("What can cause Atelectasis?")



[1m> Entering new GraphCypherQAChain chain...[0m
Generated Cypher:
[32;1m[1;3mMATCH (c:Condition {name: 'Atelectasis'})-[:DUE_TO]->(cause:Cause)
RETURN cause.name[0m
Full Context:
[32;1m[1;3m[][0m

[1m> Finished chain.[0m


'There are several factors that can cause atelectasis. Some common causes include blockage of the airways due to mucus, a foreign object, or a tumor, as well as lung diseases such as chronic obstructive pulmonary disease (COPD) or pneumonia. Additionally, certain medical procedures, such as anesthesia or prolonged bed rest, can also contribute to the development of atelectasis. It is important to consult with a healthcare professional for a proper diagnosis and treatment plan.'

In [12]:
chain.run("What clinical features radiologists need for diagnosing Atelectasis?")



[1m> Entering new GraphCypherQAChain chain...[0m
Generated Cypher:
[32;1m[1;3mMATCH (a:Condition)-[:HAS]->(s:Section)-[:MENTIONS]->(f:Finding)
WHERE a.name = 'Atelectasis'
RETURN f.name[0m
Full Context:
[32;1m[1;3m[][0m

[1m> Finished chain.[0m


'To diagnose Atelectasis, radiologists typically look for certain clinical features. These may include signs such as lung collapse, airway obstruction, and reduced lung volume. Additionally, they may also consider the presence of symptoms such as shortness of breath, coughing, and chest pain. By evaluating these clinical features, radiologists can make an accurate diagnosis of Atelectasis.'

In [13]:
chain.run("what are radiographic features of Atelectasis?")



[1m> Entering new GraphCypherQAChain chain...[0m
Generated Cypher:
[32;1m[1;3m[0m
Full Context:
[32;1m[1;3m[][0m

[1m> Finished chain.[0m


"I'm sorry, but I don't have the information to answer your question about the radiographic features of Atelectasis."