In [1]:
import pandas as pd
import joblib
import numpy as np
from utils import parse_query_with_llm, prepare_features, generate_explanation

model_pipeline = joblib.load("../models/real_estate_model_xgboost_gpu_pipeline.joblib")
context_df = pd.read_csv("../data/usa-real-estate-dataset.csv")
context_df['price_per_sqft'] = context_df['price'] / context_df['house_size']

def predict_price(user_query: str):
    parsed = parse_query_with_llm(user_query)
    if not parsed:
        return "‚ùå Could not parse the query."

    features_df = prepare_features(parsed, context_df)
    log_price = model_pipeline.predict(features_df)[0]
    price = round(np.expm1(log_price), -3)

    return f"üí∞ **Predicted Price:** ${price:,.0f}"

def chat_assistant():
    print("Welcome to the Real Estate Price Estimation Chat Assistant!")
    print("Ask me about a property, e.g., 'What's the price of a 3-bedroom, 2-bathroom house in Austin, Texas, 1800 sqft, built 2015?'")
    print("Type 'exit' to quit.")

    while True:
        user_query = input("\nUser: ").strip()
        if user_query.lower() == 'exit':
            print("Assistant: Goodbye!")
            break
        if not user_query:
            print("Assistant: Please enter a query.")
            continue

        print("Assistant: Analyzing your query...")
        
        parsed_data = parse_query_with_llm(user_query)
        if not parsed_data or all(value is None for value in parsed_data.values()):
            print("Assistant: I couldn't extract enough property details from your query. Please provide more specific information (e.g., bedrooms, bathrooms, size, city, state).")
            continue

        try:
            features_df = prepare_features(parsed_data, context_df)
        except Exception as e:
            print(f"Assistant: Error preparing features for ML model: {e}")
            continue

        try:
            log_predicted_price = model_pipeline.predict(features_df)[0]
            predicted_price = np.expm1(log_predicted_price)
            predicted_price = round(predicted_price, -3) # Round to nearest thousand for cleaner output

        except Exception as e:
            print(f"Assistant: Error during ML prediction. This might be due to incompatible feature sets or model issues: {e}")
            print("Please ensure the data preparation exactly matches the training script's feature process.")
            continue

        explanation = generate_explanation(user_query, predicted_price, parsed_data, context_df)
        print(f"\nAssistant: {explanation}")
        
        # Add a note about the built year and model's limitation based on the training script
        if 'built_year' in parsed_data and parsed_data['built_year'] not in [None, 'N/A', 'Unknown']:
            print(f"\nNote: While '{parsed_data['built_year']}' was extracted, the current ML model's training script primarily uses historical sale dates (not directly 'built year') for 'age'-related features.")

In [None]:
if __name__ == "__main__":
    chat_assistant()

Welcome to the Real Estate Price Estimation Chat Assistant!
Ask me about a property, e.g., 'What's the price of a 3-bedroom, 2-bathroom house in Austin, Texas, 1800 sqft, built 2015?'
Type 'exit' to quit.



User:  What's the price of a 3-bedroom, 2-bathroom house in Austin, Texas, 1800 sqft, built 2015?


Assistant: Analyzing your query...
Imputed zip code '55912.0' for city 'Austin'.

Assistant: The predicted price of $457,000 for this 3-bedroom, 2-bathroom house built in 2015 is notably below Austin's median home price of $650,000. While its 1800 sqft size is slightly smaller than the city's median, this estimate at approximately $254/sqft is significantly less than Austin's average of $438/sqft. This suggests the property might be located in a more value-oriented segment of the Austin market, or its specific features lead to a lower valuation despite its relatively new construction.

Note: While '2015' was extracted, the current ML model's training script primarily uses historical sale dates (not directly 'built year') for 'age'-related features.



User:  What's the price of a 4-bedroom, 3-bath home in Phoenix, Arizona with 2,200 sqft?


Assistant: Analyzing your query...
Imputed zip code '85041.0' for city 'Phoenix'.

Assistant: The predicted price of $408,000 for this 4-bedroom, 3-bath, 2,200 sqft home in Phoenix is influenced by its substantial size and desirable number of bedrooms and bathrooms. Although the property is significantly larger than the median home size in Phoenix, the estimate is lower than the city's median price, suggesting the model might be accounting for factors like its specific condition, age, or a neighborhood-specific price per square foot that differs from the overall city average.



User:  How much would a 2-bedroom condo in Miami, FL cost if it's 950 sqft?


Assistant: Analyzing your query...
Imputed zip code '33139.0' for city 'Miami'.

Assistant: The predicted price of $245,000 for this 2-bedroom, 950 sqft condo in Miami is significantly lower than the average market value for the area. This estimate may be influenced by the unit's smaller size compared to the median, and potentially assumes an older building or less premium location within Miami, given the lack of detail on factors like bathrooms or built year. Additionally, condo pricing can vary greatly from general market averages which may include other property types.
