In [4]:
from langchain.chat_models import ChatOpenAI
from langchain.document_loaders import UnstructuredFileLoader
from langchain.text_splitter import CharacterTextSplitter
from langchain.embeddings import OpenAIEmbeddings, CacheBackedEmbeddings
from langchain.vectorstores import FAISS
from langchain.storage import LocalFileStore
from langchain.chains import RetrievalQA

llm = ChatOpenAI()

cache_dir = LocalFileStore("./.cache/")

splitter = CharacterTextSplitter.from_tiktoken_encoder(
    separator="\n",
    chunk_size=600,
    chunk_overlap=100,
)
loader = UnstructuredFileLoader("./files/chapter_one.txt")

docs = loader.load_and_split(text_splitter=splitter)

embeddings = OpenAIEmbeddings()

cached_embeddings = CacheBackedEmbeddings.from_bytes_store(embeddings, cache_dir)

vectorstore = FAISS.from_documents(docs, cached_embeddings)

chain = RetrievalQA.from_chain_type(
    llm=llm,
    chain_type="map_rerank",
    retriever=vectorstore.as_retriever(),
)

chain.run("Describe Victory Mansions")



'Victory Mansions is a rundown building with a smelly hallway that smells of boiled cabbage and old rag mats. The building has a large colored poster with an enormous face of a man about forty-five years old, with a heavy black mustache and ruggedly handsome features. The building has seven flights of stairs, as the elevator rarely works due to the electricity being cut off during daylight hours. The poster with the enormous face has the caption "BIG BROTHER IS WATCHING YOU." Inside the flat, there is a telescreen that cannot be fully shut off, constantly broadcasting figures related to the production of pig-iron.'