In [78]:
import rich
import json
from itertools import count
import pandas as pd

from langchain.docstore.base import Docstore
from langchain.retrievers import ElasticSearchBM25Retriever

from guidance import Program
from guidance.llms import OpenAI
from datasets import load_dataset
from elasticsearch import Elasticsearch

In [2]:
from ipython_secrets import get_secret

OPENAI_API_KEY = get_secret("OPENAI_API_KEY")

In [81]:
data = load_dataset("xyzNLP/nza-ct-zoning-codes-text")
data = pd.concat((data["train"].to_pandas().drop(columns=["embeddings"]), data["test"].to_pandas().drop(columns=["embeddings"])))
data

Found cached dataset parquet (/Users/maxdumas/.cache/huggingface/datasets/xyzNLP___parquet/xyzNLP--nza-ct-zoning-codes-text-579051b4b18688cb/0.0.0/2a3b91fbd88a2c90d1dbbb32b460cf621d31bd5b05b934492fdef7d8d6f236ec)
100%|██████████| 2/2 [00:00<00:00, 117.76it/s]


Unnamed: 0,Town,Page,Text
0,vernon,1,
1,vernon,2,"ZONING REGULATIONS\nTown of Vernon, Connecticu..."
2,vernon,3,SECTION 1 - GENERAL\n1.1\nThe following regula...
3,vernon,4,1.3\nThe boundaries of these zones are shown o...
4,vernon,5,1.3A\nProposed changes to the Zoning Map shall...
...,...,...,...
9812,chester,110,"TOWN OF CHESTER\nZONING REGULATIONS\nAPRIL 1, ..."
9813,chester,111,"TOWN OF CHESTER\nZONING REGULATIONS\nAPRIL 1, ..."
9814,chester,112,"TOWN OF CHESTER\nZONING REGULATIONS\nAPRIL 1, ..."
9815,chester,113,"TOWN OF CHESTER\nZONING REGULATIONS\nAPRIL 1, ..."


In [82]:
class ElasticSearchDocStore(Docstore):
    def __init__(self, index_name: str, texts: list[str], reset_index=True):
        if reset_index:
            es = Elasticsearch("http://localhost:9200")  # default client
            es.indices.delete(index=index_name, ignore=[400, 404])
        
        self.retriever = ElasticSearchBM25Retriever.create(
            elasticsearch_url="http://localhost:9200",
            index_name=index_name,
            k1=2.0,
            b=0.75,
        )
        self.retriever.add_texts(texts=texts)

    def search(self, query):
        return self.retriever.get_relevant_documents(query)

In [83]:
template = """
{{#system~}}
You are ZoningGPT, a powerful assistant who is an expert in reviewing portions of Zoning documents and answering questions about them.

Pay close attention to your answers in the past and be sure to match their schema. You will be evaluated on the quality and correctness of your answers.

You must answer by using one of the functions at your disposal, or by returning a JSON formatted final answer. If you cannot answer the question, return the string "FAILURE".

You may use the functions as many times as you like.

{{>tool_def functions=functions}}
{{~/system}}

{{#user~}}
What is the minimum lot size, in square feet, for a single family home in the first listed residential district of the town of Ansonia, Connecticut?
{{~/user}}

{{#assistant~}}
```typescript
functions.search_documents({
  "town_name": "ansonia",
  "query": "zoning district list"
})```
{{~/assistant}}

{{#user~}}
'ARTICLE II - ESTALISHMENT OF DISTRICTS AND ZONING\nADMINISTRATION\nSection 205 Zoning Districts Established\nFor the purposes of these regulations, the City of Ansonia is hereby divided into the following\nclasses of districts:\nSection 210 Zoning Map\nThe boundaries of the districts specified in Section 205 are hereby established as shown on a\nmap entitled "Official Zoning Map, City of Ansonia", dated February 1990, including any\nspecial maps and boundary descriptions supplementary thereto, adopted and declared to be a part\nof this Ordinance. In addition, the boundaries of the Flood Plain District are as delineated in\nSection 220 - Flood Plain District.\nSection 215 - Determination of District Boundaries\nIn determining the boundaries of districts shown on the Official Zoning Map, the following rules\nshall apply\n215.1 Unless otherwise shown, the district boundaries shall be construed to coincide with\nthe center lines of streets, alleys, parkways, waterways and main tracks of railroads. -\n215.2\nWhere such boundaries are indicated as approximate following the property lines of\nparks or other publicly owned lands, such lines shall be construed to be such\n17\nCELL (1, 1): \nDistrict\nCELL (1, 2): \nMap Code\nCELL (2, 1): \nAAA Residence District\nCELL (2, 2): \nAAA\nCELL (3, 1): \nAA Residence District\nCELL (3, 2): \nAA\nCELL (4, 1): \nA Residence District\nCELL (4, 2): \nA\nCELL (5, 1): \nB Residence District\nCELL (5, 2): \nB\nCELL (6, 1): \nGA Multi-Family Residence District\nCELL (6, 2): \nGA\nCELL (7, 1): \nMM Multi-Family Residence District\nCELL (7, 2): \nMM\nCELL (8, 1): \nBB Multi-Family Residence District\nCELL (8, 2): \nBB\nCELL (9, 1): \nRR Multi-Family Residence-Retail District\nCELL (9, 2): \nRR\nCELL (10, 1): \nNR Neighborhood-Retail District\nCELL (10, 2): \nNR\nCELL (11, 1): \nC Central Commercial District\nCELL (11, 2): \nC\nCELL (12, 1): \nSC Special Commercial District\nCELL (12, 2): \nSC\nCELL (13, 1): \nLI Light Industrial District\nCELL (13, 2): \nLI\nCELL (14, 1): \nHI Heavy Industrial District\nCELL (14, 2): \nHI\nCELL (15, 1): \nFlood Plain District\nCELL (15, 2): \nCELL (16, 1): \nCP Commerce Park District\nCELL (16, 2): \nCP\nCELL (17, 1): \nCCZ City Center Overlay Zone\nCELL (17, 2): \nCCOZ\n'
{{~/user}}

{{#assistant~}}
```typescript
functions.search_documents({
  "town_name": "ansonia",
  "query": "AAA Residence District, minimum lot size, single family home"
})```
{{~/assistant}}

{{#user~}}
'SCHEDULE D - MAXIMUM DWELLING DENSITY\nEffective: April 25, 2003.\n52\nCELL (1, 1): \nZoning District\nCELL (1, 2): \nMinimum Lot Size (In\nSquare Feet)\nCELL (1, 3): \nMax. Residential Density (Dwelling Units\nPer Acre)\nCELL (2, 1): \nAAA\nCELL (2, 2): \n60,000\nCELL (2, 3): \n0.59\nCELL (3, 1): \nAA\nCELL (3, 2): \n30,000\nCELL (3, 3): \n1.00\nCELL (4, 1): \nA\nCELL (4, 2): \n12,500\nCELL (4, 3): \n2.40\nCELL (5, 1): \nB\nCELL (5, 2): \n7,500\nCELL (5, 3): \n4.00\n'
{{~/user}}

{{#assistant~}}
{
    "rationale": "The table indicates that the minimum lot size for a single family home in the AAA Residence District is 60,000 square feet. So the answer is 60,000 square feet.",
    "answer": "60,000 square feet"
}
{{~/assistant}}

{{#user~}}
    {{question}}
{{~/user}}

{{#geneach 'conversation' stop=False}}
    {{#assistant~}}
        {{gen 'this.response' stop='\\n' function_call='auto'}}
    {{~/assistant}}

    {{#if callable(this.response)}}
        {{~#function name=this.response.__name__~}}
            {{this.response()}}
        {{~/function~}}
    {{else}}
        {{break}}
    {{/if}}
{{/geneach}}
"""

In [84]:
def search_documents(town_name: str, query: str):
    rich.print(town_name, query)
    docstore = ElasticSearchDocStore(town_name, data.query("Town == @town_name")["Text"].to_list())
    return docstore.search(query)[0].page_content

tools_dict = {
    "search_documents": search_documents
}

functions=[
    {
        "name": "search_documents",
        "description": "Searches the document store for the contents of a document matching the query. Use this tool to learn more about a topic.",
        "parameters": {
            "type": "object",
            "properties": {
                "town_name": {
                    "type": "string",
                    "description": "The name of the town to search in. Must be in all lowercase and have spaces converted to dashes, e.g. 'Beacon Falls' becomes 'beacon-falls'.",
                },
                "query": {
                    "type": "string",
                    "description": "A keyword search query, e.g. 'AAA Residence District, minimum lot size, single family home'",
                },
            },
            "required": ["town_name", "query"],
        }
    }
]

In [96]:
p = Program(template, llm=OpenAI("gpt-4-0613", api_key=OPENAI_API_KEY), caching=False, silent=False, log=True)
t = p(
    question='What is the minimum lot size, in square feet, for a building in the second district of the town of Ashford, Connecticut?',
    functions=functions,
    **tools_dict
)

In [92]:
t.variables()

{'llm': <guidance.llms._openai.OpenAI at 0x2c522c880>,
 'question': 'What is the minimum lot size, in square feet, for each listed zoning district of the town of Ashford, Connecticut?',
 'functions': [{'name': 'search_documents',
   'description': 'Searches the document store for the contents of a document matching the query. Use this tool to learn more about a topic.',
   'parameters': {'type': 'object',
    'properties': {'town_name': {'type': 'string',
      'description': "The name of the town to search in. Must be in all lowercase and have spaces converted to dashes, e.g. 'Beacon Falls' becomes 'beacon-falls'."},
     'query': {'type': 'string',
      'description': "A keyword search query, e.g. 'AAA Residence District, minimum lot size, single family home'"}},
    'required': ['town_name', 'query']}}],
 'search_documents': <function __main__.search_documents(town_name: str, query: str)>,
 '@raw_prefix': '\n{{!--GMARKER_START_system$&#123;&#123;#system~&#125;&#125;\nYou are Zoning

In [90]:
t.log

2023-06-23 22:48:47   [start]         system: {'positional_args': [], 'named_args': {}, '@prefix': '\n'}
2023-06-23 22:48:47   [start]      functions: {'positional_args': [], 'named_args': {}, '@prefix': '\n<|im_start|>system\nYou are ZoningGPT, a powerful assistant who is an expert in reviewing portions of Zoning documents and answering questions about them.\n\nPay close attention to your answers in the past and be sure to match their schema. You will be evaluated on the quality and correctness of your answers.\n\nYou must answer by using one of the functions at your disposal, or by returning a JSON formatted final answer. If you cannot answer the question, return the string "FAILURE".\n\nYou may use the functions as many times as you like.\n\n'}
2023-06-23 22:48:47     [end]      functions: {'new_prefix': ''}
2023-06-23 22:48:47   [start]      functions: {'positional_args': [], 'named_args': {}, '@prefix': '\n<|im_start|>system\nYou are ZoningGPT, a powerful assistant who is an exper