In [1]:
from sqlalchemy import create_engine
import os
import json
import pandas as pd
import psycopg2
import requests

# 0. Generate SQL Query

In [2]:
# Define the API endpoint
api_url = "http://localhost:11434/api/chat"

In [3]:
instruction = '''You are a data analyst assistant handling funding data of a bank.
The data is saved in a table named master_funding.
Here are the fields in the table with their data types in JSON format:
[
  {
    "name": "BASE_DT",
    "mode": "",
    "type": "STRING",
    "description": "The date of the data (YYYYMMDD)",
    "fields": []
  },
  {
    "name": "BASE_DT_PARSED",
    "mode": "",
    "type": "DATE",
    "description": "The date of the data parsed as date",
    "fields": []
  },
  {
    "name": "BASE_YM",
    "mode": "",
    "type": "STRING",
    "description": "The year-month of the data (YYYYMM)",
    "fields": []
  },
  {
    "name": "AGREE_ID",
    "mode": "",
    "type": "STRING",
    "description": "Unique identifier of the funding account",
    "fields": []
  },
  {
    "name": "FLAG",
    "mode": "",
    "type": "STRING",
    "description": "Flag of account type (Sharia/Conventional)",
    "fields": []
  },
  {
    "name": "REGION",
    "mode": "",
    "type": "STRING",
    "description": "Region location of the funding account",
    "fields": []
  },
  {
    "name": "AREA",
    "mode": "",
    "type": "STRING",
    "description": "Area location of the funding account",
    "fields": []
  },
  {
    "name": "BRANCH",
    "mode": "",
    "type": "STRING",
    "description": "Branch location of the funding account",
    "fields": []
  },
  {
    "name": "CIF_NO",
    "mode": "",
    "type": "STRING",
    "description": "Unique customer identifier",
    "fields": []
  },
  {
    "name": "CUST_TYPE",
    "mode": "",
    "type": "STRING",
    "description": "Type of customer (Individual/Non-Individual)",
    "fields": []
  },
  {
    "name": "PROD_NM",
    "mode": "",
    "type": "STRING",
    "description": "Product name of the funding account",
    "fields": []
  },
  {
    "name": "SUB_PROD_NM",
    "mode": "",
    "type": "STRING",
    "description": "Sub product name of the funding account",
    "fields": []
  },
  {
    "name": "SEGMENT",
    "mode": "",
    "type": "STRING",
    "description": "Segment of the customer",
    "fields": []
  },
  {
    "name": "GCIF_NAME",
    "mode": "",
    "type": "STRING",
    "description": "Name of the customer",
    "fields": []
  },
  {
    "name": "PROD_TYPE",
    "mode": "",
    "type": "STRING",
    "description": "Product type of the account (CA/SA/TD)",
    "fields": []
  },
  {
    "name": "CURR_CODE",
    "mode": "",
    "type": "STRING",
    "description": "Currency code",
    "fields": []
  },
  {
    "name": "COLT",
    "mode": "",
    "type": "STRING",
    "description": "Collateral status (1 or 0)",
    "fields": []
  },
  {
    "name": "RATE_DPK",
    "mode": "",
    "type": "FLOAT",
    "description": "Interest rate of the account",
    "fields": []
  },
  {
    "name": "BASE_AMT_FIX",
    "mode": "",
    "type": "FLOAT",
    "description": "End of day funding amount of the account",
    "fields": []
  },
  {
    "name": "MTD_AVG_AMT_FIX",
    "mode": "",
    "type": "FLOAT",
    "description": "Month-to-date funding amount of the account",
    "fields": []
  },
  {
    "name": "DTD",
    "mode": "",
    "type": "FLOAT",
    "description": "Day-to-date funding amount change of the account",
    "fields": []
  },
  {
    "name": "MTD",
    "mode": "",
    "type": "FLOAT",
    "description": "Month-to-date funding amount change of the account",
    "fields": []
  },
  {
    "name": "YTD",
    "mode": "",
    "type": "FLOAT",
    "description": "Year-to-date funding amount change of the account",
    "fields": []
  },
  {
    "name": "DIVISION",
    "mode": "",
    "type": "STRING",
    "description": "The division of the funding customer",
    "fields": []
  },
  {
    "name": "SOURCE",
    "mode": "",
    "type": "STRING",
    "description": "Source of data",
    "fields": []
  },
  {
    "name": "SEGMENT_FIX",
    "mode": "",
    "type": "STRING",
    "description": "Segment of funding the customer",
    "fields": []
  },
  {
    "name": "BASE_AMT_ACCUM_MTD",
    "mode": "",
    "type": "FLOAT",
    "description": "Month-to-date accumulation of funding amount",
    "fields": []
  },
  {
    "name": "INT_EXP_ACCUM_MTD",
    "mode": "",
    "type": "FLOAT",
    "description": "Month-to-date accumulation of interest expense",
    "fields": []
  },
  {
    "name": "COF_MTD",
    "mode": "",
    "type": "FLOAT",
    "description": "Month-to-date cost of fund of the funding account",
    "fields": []
  },
  {
    "name": "HIGH_COF_FLAG",
    "mode": "",
    "type": "STRING",
    "description": "Category of month-to-date cost of fund",
    "fields": []
  },
  {
    "name": "LOB_SORT",
    "mode": "",
    "type": "STRING",
    "description": "Code for sorting the customer segment",
    "fields": []
  },
  {
    "name": "CASA_TD",
    "mode": "",
    "type": "STRING",
    "description": "Category of the funding account (CASA/TD)",
    "fields": []
  },
  {
    "name": "DTD_10B",
    "mode": "",
    "type": "STRING",
    "description": "Flag of day-to-date 10 billion funding loss",
    "fields": []
  },
  {
    "name": "MTD_10B",
    "mode": "",
    "type": "STRING",
    "description": "Flag of month-to-date 10 billion funding loss",
    "fields": []
  },
  {
    "name": "BLOCK",
    "mode": "",
    "type": "STRING",
    "description": "Flag of account block status",
    "fields": []
  }
]
When given a question about the funding data, answer in SQL script to retrieve the stat asked.
If the question is not related to the funding data, answer that you can't help the user
'''


In [4]:
conversation_history = [{'role':'system',
                         'content':instruction}]

In [5]:
user_input = 'What is the number of the funding customer?'

In [6]:
conversation_history.append({"role":"user","content":user_input})

In [7]:
payload = {
    "model":"deepseek-r1:8b",
    "messages":conversation_history,
    "stream":False
}

In [8]:

response = requests.post(api_url,json=payload)

In [9]:
payload

{'model': 'deepseek-r1:8b',
 'messages': [{'role': 'system',
   'content': 'You are a data analyst assistant handling funding data of a bank.\nThe data is saved in a table named master_funding.\nHere are the fields in the table with their data types in JSON format:\n[\n  {\n    "name": "BASE_DT",\n    "mode": "",\n    "type": "STRING",\n    "description": "The date of the data (YYYYMMDD)",\n    "fields": []\n  },\n  {\n    "name": "BASE_DT_PARSED",\n    "mode": "",\n    "type": "DATE",\n    "description": "The date of the data parsed as date",\n    "fields": []\n  },\n  {\n    "name": "BASE_YM",\n    "mode": "",\n    "type": "STRING",\n    "description": "The year-month of the data (YYYYMM)",\n    "fields": []\n  },\n  {\n    "name": "AGREE_ID",\n    "mode": "",\n    "type": "STRING",\n    "description": "Unique identifier of the funding account",\n    "fields": []\n  },\n  {\n    "name": "FLAG",\n    "mode": "",\n    "type": "STRING",\n    "description": "Flag of account type (Sharia/

In [10]:
response

<Response [200]>

In [13]:
response.json()

{'model': 'deepseek-r1:8b',
 'created_at': '2025-02-02T15:03:40.6607344Z',
 'message': {'role': 'assistant',
  'content': '<think>\nOkay, so I\'m trying to figure out how to find the number of funding customers based on the provided table schema. Let\'s break this down step by step.\n\nFirst, the user asked for "the number of the funding customer." That sounds like they want a count of all the customers who have funding accounts. So, in SQL terms, I need to write a query that counts how many rows are in the master_funding table.\n\nLooking at the table schema, each row represents a funding account with various fields like AGREE_ID, CIF_NO, etc. But since the user didn\'t specify any particular filter or condition, I should count all entries regardless of other attributes.\n\nI remember that in SQL, the COUNT() function is used to count rows. So, using COUNT(1) is a common practice because it\'s less error-prone and works across different SQL dialects. Alternatively, COUNT(*) also works

In [11]:
# Get the model's reply
model_reply = response.json()["message"]["content"]

In [12]:
model_reply

'<think>\nOkay, so I\'m trying to figure out how to find the number of funding customers based on the provided table schema. Let\'s break this down step by step.\n\nFirst, the user asked for "the number of the funding customer." That sounds like they want a count of all the customers who have funding accounts. So, in SQL terms, I need to write a query that counts how many rows are in the master_funding table.\n\nLooking at the table schema, each row represents a funding account with various fields like AGREE_ID, CIF_NO, etc. But since the user didn\'t specify any particular filter or condition, I should count all entries regardless of other attributes.\n\nI remember that in SQL, the COUNT() function is used to count rows. So, using COUNT(1) is a common practice because it\'s less error-prone and works across different SQL dialects. Alternatively, COUNT(*) also works, but sometimes it can be confusing or have performance implications depending on the database setup.\n\nI should make sur

In [None]:
# Add the model's reply to the conversation history
conversation_history.append({"role": "assistant", "content": model_reply})

In [None]:
model_reply

# 1. Connect to Database

In [2]:
# connection info
host= 'localhost'
port= '5432'
database= 'postgres'
username= 'postgres'
password= os.environ['PGPW']

In [19]:
queryResult= []
try:
    # Establish the connection
    connection = psycopg2.connect(
        user="postgres",
        password= os.environ['PGPW'],
        host= "localhost",
        port="5432",
        database="postgres"
    )
    print("Connection to PostgreSQL DB successful")

    # Create a cursor object
    cursor = connection.cursor()

    # Execute a SQL query
    cursor.execute('SELECT * FROM master_funding;')
    rows = cursor.fetchall()

    # Process the result
    for row in rows:
        queryResult.append(row)

except Exception as e:
    print(f"Error: {e}")

finally:
    # Close the cursor and connection
    if cursor:
        cursor.close()
    if connection:
        connection.close()


Connection to PostgreSQL DB successful


In [3]:
engine = create_engine(f"postgresql://{username}:{password}@{host}:{port}/{database}")

In [None]:
data = pd.read_sql()

In [24]:
user_input = f'With the following data: {queryResult}, determine the number of unique CIF_NO'
response = ollama.generate(model=model, prompt=user_input)

In [25]:
print(response['response'])

<think>
Alright, so I need to figure out how many unique CIF_NO values are in this given list. First off, I'm not exactly sure what a CIF_NO is, but from context, it seems like it's some kind of identifier or code, probably related to banking or financial transactions in Indonesia.

Looking at the data provided, each entry has several fields, and one of them is labeled as "CIF_NO". My task is to extract all the CIF_NO values and then determine how many unique ones there are. 

I'll start by scanning through each line of the dataset to identify where the "CIF_NO" field appears. It might not always be in the same position, so I need to check every entry carefully.

Let me go through them one by one:

1. The first entry has a CIF_NO of 'XXX-GB-CORP'. That's interesting—it's not a numeric value but a string that includes "CIF" followed by some letters and numbers.
2. Next, another entry also shows 'XXX-GB-CORP', so this is likely the same identifier repeated.

Wait, no—maybe not. Let me ch

In [None]:
from ollama import chat
from ollama import ChatResponse

response: ChatResponse = chat(model='llama3.2', messages=[
  {
    'role': 'user',
    'content': 'Why is the sky blue?',
  },
])
print(response['message']['content'])
# or access fields directly from the response object
print(response.message.content)