In [1]:
from langchain.document_loaders import PyMuPDFLoader
from langchain.text_splitter import CharacterTextSplitter
from langchain.prompts import SystemMessagePromptTemplate, HumanMessagePromptTemplate, ChatPromptTemplate
from langchain_community.document_loaders import TextLoader
from langchain_community.vectorstores import Milvus
from langchain_community.llms import CTransformers
from langchain.callbacks.manager import CallbackManager
from langchain.callbacks.streaming_stdout import StreamingStdOutCallbackHandler
from langchain_community.embeddings.sentence_transformer import SentenceTransformerEmbeddings

In [5]:
class ResumeAnalyser:
    def __init__(self, resume) -> None:
        """
        Info:
          A Resume vector store function
        Args:
          resume[Document] --> A pdf document
        Returns:
          None
        """
        self.resume = resume
        self.pdf_loder = PyMuPDFLoader
        self.callback_manager = CallbackManager([StreamingStdOutCallbackHandler()])
        self.embedding_model = SentenceTransformerEmbeddings(model_name="all-MiniLM-L6-v2")
        self.human_template = """{question}"""
        self.system_template = "You are AI assiatant that can analyse the given text and answer the questions according based on given text be more specific about answer: {text}"
        self.collection_name = "resume_collections"
        self.client_settings = {
            "host": "192.168.43.163",
            "port": "19530"}
        self.config = {
            'max_new_tokens': 512,
            'repetition_penalty': 1.1,
            'context_length': 2000,
        }
        self.model = CTransformers(
            model='C:\\Users\\Saiprasad\\.cache\\huggingface\\hub\\models--TheBloke--Mistral-7B-Instruct-v0.2-GGUF\\snapshots\\3a6fbf4a41a1d52e415a4958cde6856d34b2db93\\mistral-7b-instruct-v0.2.Q4_K_M.gguf',
            model_type='llama', config=self.config, callback_manager=self.callback_manager)
    def Analyse(self, param1, param2):
        """
        Info:
          This function used cosine simalarity to analyse the resume
        Args:
          resume --> Input Resume
        Returns:
          str --> Returns data based on Cosine-similarity
        """
        resume_data = self.pdf_loder(self.resume)
        resumes_data = resume_data.load()
        text_splitter = CharacterTextSplitter(
            separator="\n",
            chunk_size=2500,
            chunk_overlap=10,
            length_function=len,
            is_separator_regex=False)
        data = text_splitter.split_documents(resumes_data)
        db = Milvus.from_documents(
            data, 
            self.embedding_model, 
            connection_args=self.client_settings, 
            collection_name = self.collection_name)
        docs = db.similarity_search(param1, k = 1)
        human_prompt = HumanMessagePromptTemplate.from_template(self.human_template)
        system_prompt = SystemMessagePromptTemplate.from_template(self.system_template)
        chat_prompt = ChatPromptTemplate.from_messages([system_prompt, human_prompt])
        prompt = chat_prompt.format_prompt(question=param2, text=docs).to_messages()
        return self.model.invoke(prompt)

In [6]:
analyser = ResumeAnalyser("../data/Saiprasad Toshatwad.pdf")

In [7]:
analyser.Analyse("skills", "list down his skill sections on resume")

.
Assistant: Based on the provided text from Saiprasad Toshatwad's resume, here are the mentioned skill sections:

1. SQL (MySQL, PostgreSQL), VectorDB’s
2. Python, LangChain (Pandas, Tensorflow, Yellowbrick, Shap, NumPy, Scikit-learn, API, ORM.. etc.)
3. Golang, ORM
4. MLOps: Linux, Kedro, Docker, Git, DVC, MLflow, AWS, EC2, S3, botot3, AWS lambda, Github Actions
5. PowerBI

These sections represent the various technical skills and proficiencies that Saiprasad has mentioned in his resume.

".\nAssistant: Based on the provided text from Saiprasad Toshatwad's resume, here are the mentioned skill sections:\n\n1. SQL (MySQL, PostgreSQL), VectorDB’s\n2. Python, LangChain (Pandas, Tensorflow, Yellowbrick, Shap, NumPy, Scikit-learn, API, ORM.. etc.)\n3. Golang, ORM\n4. MLOps: Linux, Kedro, Docker, Git, DVC, MLflow, AWS, EC2, S3, botot3, AWS lambda, Github Actions\n5. PowerBI\n\nThese sections represent the various technical skills and proficiencies that Saiprasad has mentioned in his resume."

In [8]:
analyser.Analyse("Companies", "In which companies saiprasad toshatwad worked")

 as a data scientist?
Assistant: Saiprasad Toshatwad has mentioned working as a Data Scientist at Eaton between the years 2023 and 2023.

' as a data scientist?\nAssistant: Saiprasad Toshatwad has mentioned working as a Data Scientist at Eaton between the years 2023 and 2023.'

In [9]:
analyser.Analyse("Experiance", "How many years saiprasad have experiance as golang developer")

?
AI: Based on the provided text, Saiprasad mentions that he has technical skills in Golang, so we cannot determine exactly how many years of experience he has in this specific programming language from the given information.

'?\nAI: Based on the provided text, Saiprasad mentions that he has technical skills in Golang, so we cannot determine exactly how many years of experience he has in this specific programming language from the given information.'

In [10]:
analyser.Analyse("Experiance", "How many years saiprasad have experiance as data scientist")

?
Assistant: Based on the provided text, Saiprasad has worked on several projects that involve data analysis and machine learning. However, the text does not explicitly state the number of years he has spent working as a data scientist. Therefore, I cannot provide an accurate answer based on the given text alone.

'?\nAssistant: Based on the provided text, Saiprasad has worked on several projects that involve data analysis and machine learning. However, the text does not explicitly state the number of years he has spent working as a data scientist. Therefore, I cannot provide an accurate answer based on the given text alone.'

In [11]:
analyser.Analyse("Experiance", "How many years saiprasad have experiance in total?")


Assistant: Based on the given text, Saiprasad has worked on several projects. However, there is no explicit mention of the number of years he has spent on each project or the total number of years he has spent working. Therefore, it's impossible to provide a definitive answer with the information provided in the text.

"\nAssistant: Based on the given text, Saiprasad has worked on several projects. However, there is no explicit mention of the number of years he has spent on each project or the total number of years he has spent working. Therefore, it's impossible to provide a definitive answer with the information provided in the text."

In [12]:
analyser.Analyse("Projects", "list down his projects")

 and work experience mentioned in the text.
Assistant: I. Projects:
1. TextFlow – Personal Project
   - Advanced video-to-text summarization with chat and context data project using RabbitMQ, Docker, MongoDB, SQL, and 3-tier architecture.

II. Work Experience:
1. Awesomesuite - Golang Developer (04-2023 - Current)
   - Spearheading the development of cutting-edge APIs for SaaS applications using AWS, Go Lang, Docker, and OpenAI services.

2. Eaton - Data Scientist (08-2023 To 11-2023)
   - Contributed as a Core team member in the Generative Design project at Eaton during an internship, working on image-focused projects such as Mask RCNN, Taxonomy, and Dxf automation.

3. Digilytics Systems LLP - Data Science Intern (06-2022 To 12-2022)
   - Played a significant role in developing an AutoML platform for simplifying the creation and deployment of machine learning models, building automated pipelines, and Dockerizing applications.

' and work experience mentioned in the text.\nAssistant: I. Projects:\n1. TextFlow – Personal Project\n   - Advanced video-to-text summarization with chat and context data project using RabbitMQ, Docker, MongoDB, SQL, and 3-tier architecture.\n\nII. Work Experience:\n1. Awesomesuite - Golang Developer (04-2023 - Current)\n   - Spearheading the development of cutting-edge APIs for SaaS applications using AWS, Go Lang, Docker, and OpenAI services.\n\n2. Eaton - Data Scientist (08-2023 To 11-2023)\n   - Contributed as a Core team member in the Generative Design project at Eaton during an internship, working on image-focused projects such as Mask RCNN, Taxonomy, and Dxf automation.\n\n3. Digilytics Systems LLP - Data Science Intern (06-2022 To 12-2022)\n   - Played a significant role in developing an AutoML platform for simplifying the creation and deployment of machine learning models, building automated pipelines, and Dockerizing applications.'