์ต๊ทผ, ๋ถ๋์ฐ ๊ฐ๊ฒฉ์ด ๊ธ๋ฑ๋ฝํ๋ฉฐ ๊ฐ๊ฒฉ์์ธก์ด ํ๋ค์ด์ง๋ ์ค์ ์ด๋ฉฐ ์ํํธ ๋ถ๋์ฐ ๋งค๋งค์ ํ์ด๋ฐ์ ๋ํ ๊ณ ๋ฏผ๋ ๋์ด๊ฐ๊ณ ์๋ค.
์ด์ ์ ํํ ์ํํธ ๊ฐ๊ฒฉ์ ์์ธกํ์ฌ ์ด๋ฐ ์ด๋ค์๊ฒ ๋์์ด ๋๊ณ ์ ํ๋ ๋ง์์์ ํ๋ก์ ํธ๋ฅผ ์์ํ์๋ค.
START : 2024.02.21
END : 2024.03.07
- ๊น์ธ์ฐ - ํ๋ก์ ํธ ๊ตฌ์, EDA, ๋ฐ์ดํฐ ์ ์ฒ๋ฆฌ, ๋ชจ๋ธ๋ง
- ์ด๊ตฌํ - ์กฐ์ฅ, ๋ฐํ, ํ๋ก์ ํธ ๊ตฌ์, EDA, ๋ฐ์ดํฐ ์ ์ฒ๋ฆฌ, ์ด์์น ํ์ธ, ๋ชจ๋ธ๋ง, ์์๋ธ, ํ์ดํผํ๋ผ๋ฏธํฐ ์ต์ ํ, GUI ๊ตฌํ
- ์ ์ฐ์ฑ - ํ๋ก์ ํธ ๊ตฌ์, EDA, ๋ฐ์ดํฐ ์ ์ฒ๋ฆฌ, ์ด์์น ํ์ธ, ๋ชจ๋ธ๋ง, ์์๋ธ, ํ์ดํผํ๋ผ๋ฏธํฐ ์ต์ ํ
- ์ตํ์ - ํ๋ก์ ํธ ๊ตฌ์, ๋ฐํ ์๋ฃ ์ค๋น, EDA, ๋ฐ์ดํฐ ์ ์ฒ๋ฆฌ, ๋ชจ๋ธ๋ง
[์์ธ์ ๋ถ๋์ฐ ์ค๊ฑฐ๋๊ฐ ์ ๋ณด] (https://data.seoul.go.kr/dataList/OA-21275/S/1/datasetView.do)
ํด๋น ๋ฐ์ดํฐ์ ์์ 2020๋ 1์ 1์ผ๋ถํฐ 2024๋ 2์ 28์ผ๊น์ง ๊ฑฐ๋๋ ๋ฐ์ดํฐ๋ฅผ ์ฌ์ฉํ์์.
์๋ณธ ๋ฐ์ดํฐ์์๋ 1,332,702๊ฐ์ Row๊ฐ ์กด์ฌํ์ผ๋ฉฐ, ์ด๋ฅผ 2020๋ ์ดํ ๋ฐ์ดํฐ๋ก ์๋ผ๋ด์ด ์ฝ 180,000๊ฐ์ ๋ฐ์ดํฐ๋ง์ ์ฌ์ฉํจ
- ์์น๊ตฌ๋ช
- ๋ฒ์ ๋๋ช
- ๋ณธ๋ฒ
- ๋ถ๋ฒ
- ๊ฑด๋ฌผ๋ช
- ๊ณ์ฝ์ผ
- ๋ฌผ๊ฑด๊ธ์ก
- ๊ฑด๋ฌผ๋ฉด์
- ์ธต
- ๊ฑด์ถ๋ ๋
๋ค์ด๋ฒ ์ง๋ API๋ฅผ ์ฌ์ฉํ์ฌ ์๋ณธ ๋ฐ์ดํฐ์ ์กด์ฌํ๋ '๋ณธ๋ฒ', '๋ถ๋ฒ'์ ์ด์ฉํด ์ง๋ฒ์ฃผ์๋ฅผ ๋ง๋ ๋ค, ์ง๋ฒ ์ฃผ์๋ฅผ ์ง์ค์ฝ๋ฉํ์ฌ ์๋, ๊ฒฝ๋๋ก ๋ณํํจ.
- ์์ธ์ ๋ณ์์ ์์น์ ์ ๋ณด
- ์์ธ์ ํ๊ต์ ์์น์ ์ ๋ณด
- ์์ธ์ ์งํ์ฒ ์ญ์ ์์น์ ์ ๋ณด
- ์์ธ์ ๋ฒ์ค ์ ๋ฅ์์ ์์น์ ์ ๋ณด
- ์์ธ์ ํธ์์์ค(์์ ์์ค, ๋ณ์์, ๊ณต์ ๋ฑ)์ ์์น์ ์ ๋ณด
- ์ ์ธ๊ฐ๊ฒฉ์ง์
- ์๋น์ ๋ฌผ๊ฐ์ง์
- ์๋ณ ๊ธฐ์ค ๊ธ๋ฆฌ
- ์์ธ์ ๋ฒ์ ๋๋ณ ๋์ด๋ณ ์ธ๊ตฌ ์๋ฃ
- ์ํํธ ๋งค๋งค๊ฐ ์ค๊ฑฐ๋ ๊ฐ๊ฒฉ ์ง์
- etc.
population['์ด๋ฆฐ์ด์ธ๊ตฌ'] = population[[f"{age}์ธ๋จ์" for age in range(13)] + [f"{age}์ธ์ฌ์" for age in range(13)]].sum(axis=1)
population['์ฒญ์๋
์ธ๊ตฌ'] = population[[f"{age}์ธ๋จ์" for age in range(13, 25)] + [f"{age}์ธ์ฌ์" for age in range(13, 25)]].sum(axis=1)
population['์ฒญ๋
์ธ๊ตฌ'] = population[[f"{age}์ธ๋จ์" for age in range(25, 41)] + [f"{age}์ธ์ฌ์" for age in range(25, 41)]].sum(axis=1)
population['์ค์ฅ๋
์ธ๊ตฌ'] = population[[f"{age}์ธ๋จ์" for age in range(41, 66)] + [f"{age}์ธ์ฌ์" for age in range(41, 66)]].sum(axis=1)
population['๋
ธ๋
์ธ๊ตฌ'] = population[[f"{age}์ธ๋จ์" for age in range(66, 109)] + [f"{age}์ธ์ฌ์" for age in range(66, 109)]].sum(axis=1)
0์ธ๋ถํฐ 109์ธ๊น์ง 1์ธ์ฉ ์ ๋ฆฌ๋์ด์๋ ๋์ด๋ฅผ ์ด๋ฆฐ์ด (0
12), ์ฒญ์๋ (1324), ์ฒญ๋ (2540), ์ค์ฅ๋ (4165), ๋ ธ๋ (66~109)์ผ๋ก ๋๋
์ํํธ ๋งค๋งค๊ฐ ๋ฐ์ดํฐ์ ์๋ "๋งค๋งค๊ฐ"๋ฅผ "๋ฉด์ ๋น ๊ฐ๊ฒฉ"์ผ๋ก ๋ณ๊ฒฝ
from scipy.interpolate import interp1d
hangang_sorted = bridge.sort_values(by='Longitude')
# ๋ค๋ฆฌ๋ฅผ ์ฐ์ ์ขํ๋ฅผ ์ ํ๋ณด๊ฐ์ผ๋ก ์์ต๋๋ค.
interpolate_lon = np.linspace(hangang_sorted['Longitude'].min(), hangang_sorted['Longitude'].max(), 130) #์ ์์ ์๋ ์ขํ๋ฅผ ๊ธฐ๋กํฉ๋๋ค. 130๊ฐ
linear_interp = interp1d(hangang_sorted['Longitude'], hangang_sorted['Latitude'], kind='linear')
interpolate_lat = linear_interp(interpolate_lon)
selected_coords = np.column_stack((interpolate_lon, interpolate_lat))
# ์ ํ๋ ์ขํ๋ฅผ ๋ด์ ๋ฐ์ดํฐํ๋ ์ ์์ฑ
selected_coords_df = pd.DataFrame(selected_coords, columns=['Longitude', 'Latitude'])
์ ์ธ ํ๊ฐ์ ์ง๋๋ ๊ต๋์ ์ค์ฌ์ขํ๋ฅผ ์ ํ๋ณด๊ฐํ์ฌ ์ขํ๋ฅผ ์ถ์ถํ๋ค, ์ํํธ์์ ๊ฑฐ๋ฆฌ๋ฅผ Haversine๊ณต์์ ์ด์ฉํ์ฌ ์ธก์ ํ์์.
from statsmodels.stats.outliers_influence import variance_inflation_factor
from statsmodels.tools.tools import add_constant
X.astype(float)
X_const = add_constant(X)
vif = pd.DataFrame()
vif["VIF Factor"] = [variance_inflation_factor(X_const.values, i) for i in range(X_const.shape[1])]
vif["features"] = X_const.columns
vif["VIF Factor"] = vif["VIF Factor"].apply(lambda x: '{:.0f}'.format(x))
vif
์ด์์น๋ ์์์ง๋ง, ์ค์ ๋ฐ์ดํฐ์ด๋ฏ๋ก ์ด์์น์ ๋ํ ์ ๊ฑฐ๋ ์๋ตํ์์ผ๋ฉฐ, ๋ค์ค๊ณต์ ์ฑ๊ฒ์ฌ๋ฅผ ํตํด VIF๊ฐ์ด ๋์ ๋ณ์๋ฅผ ์ ๊ฑฐํ์์. ์ ๊ฑฐํ๊ณ ๋๋ P-value๋ ๋ฎ๊ฒ๋์ด
#์ ์ง์ ํ๋ฒ
def forward_selection(data, target, significance_level=0.05):
initial_features = data.columns.tolist()
best_features = []
aic_values = []
while len(initial_features) > 0:
remaining_features = list(set(initial_features) - set(best_features))
new_pval = pd.Series(index=remaining_features)
for new_column in remaining_features:
model = sm.OLS(target, sm.add_constant(data[best_features + [new_column]])).fit()
new_pval[new_column] = model.pvalues[new_column]
min_p_value = new_pval.min()
if min_p_value < significance_level:
best_feature = new_pval.idxmin()
best_features.append(best_feature)
aic_values.append(sm.OLS(target, sm.add_constant(data[best_features])).fit().aic)
else:
break
return best_features, aic_values
#ํ์ง์๊ฑฐ๋ฒ
def backward_elimination(data, target, significance_level = 0.05):
features = data.columns.tolist()
while len(features) > 0:
features_with_constant = sm.add_constant(data[features])
p_values = sm.OLS(target, features_with_constant).fit().pvalues[1:]
max_p_value = p_values.max()
if max_p_value >= significance_level:
excluded_feature = p_values.idxmax()
features.remove(excluded_feature)
else:
break
return features
#๊ต์ฐจ์ ํ๋ฒ
def stepwise_selection(data, target, SL_in=0.05, SL_out=0.05):
initial_features = data.columns.tolist()
best_features = []
while len(initial_features) > 0:
changed=False
# ์ ์ง ์ ํ
remaining_features = list(set(initial_features) - set(best_features))
new_pval = pd.Series(index=remaining_features)
for new_column in remaining_features:
model = sm.OLS(target, sm.add_constant(data[best_features + [new_column]])).fit()
new_pval[new_column] = model.pvalues[new_column]
min_p_value = new_pval.min()
if min_p_value < SL_in:
best_features.append(new_pval.idxmin())
changed=True
# ํ์ง ์๊ฑฐ
model = sm.OLS(target, sm.add_constant(data[best_features])).fit()
p_values = model.pvalues.iloc[1:]
max_p_value = p_values.max()
if max_p_value > SL_out:
changed=True
worst_feature = p_values.idxmax()
best_features.remove(worst_feature)
if not changed:
break
return best_features
์ด๋ฏธ VIF ๊ฒ์ฌ๋ฅผ ํตํด ๋ณ์๋ฅผ ์ ๊ฑฐํด์, ์ ์๋ฏธํ ๊ฒฐ๊ณผ๋ฅผ ์ป์ง ๋ชปํจ.
| ๋ชจ๋ธ | R^2 | MSE | RMSE | MAE |
|---|---|---|---|---|
| RF | 0.9249 | 0.0158 | 0.1256 | 0.0862 |
| CatBoost | 0.9232 | 0.0161 | 0.1270 | 0.0927 |
| KNN ํ๊ท | 0.8915 | 0.0228 | 0.1510 | 0.1032 |
| MLP ํ๊ท | 0.9045 | 0.0201 | 0.1417 | 0.1052 |
| XGBoost | 0.9293 | 0.0148 | 0.1218 | 0.0879 |
| ์คํํน ์์๋ธ | 0.9273 | 0.0153 | 0.1235 | 0.0896 |
- ์๋, ๊ฒฝ๋, ์์ 10๊ฐ ๊ฑด์ค์ฌ ์ฌ๋ถ, ์ฐ์, ์ธต์, ๊ธฐ์ค๊ธ๋ฆฌ, ๋งค๋งค๊ฐ ๋๋น ์ ์ธ๊ฐ
- ๋ฒ์ ๋ ์ ์ฒด ์ธ๊ตฌ, ๋ฒ์ ๋ ์ด๋ฆฐ์ด, ์ฒญ๋ , ๋ ธ๋ ๋น์จ
- ์์น๊ตฌ๋ณ ์ฐ๊ฐ ์๋น์ก, ๊ฐ์ฅ ๊ฐ๊น์ด ์ข ํฉ๋ณ์๊ณผ์ ๊ฑฐ๋ฆฌ, ๊ฐ์ฅ ๊ฐ๊น์ด ์งํ์ฒ ์ญ๊ณผ์ ๊ฑฐ๋ฆฌ, 0.7km๋ด ์ด๋ฑํ๊ต ๊ฐ์, 2km๋ด ๋ช ๋ฌธ ์ผ๋ฐ๊ณ ๊ฐ์
- 1km๋ด ์ผ๋ฐ ๋ณ์์๊ณผ ์์ ์์ค์ ๊ฐ์, 0.8km๋ด ๊ณต์ ๋ฐ ํ์ฒ์ ์กด์ฌ ์ฌ๋ถ
- ํ๊ฐ๋ณ์์ 0.4km์ด๋ด์ ์๋์ง ์ฌ๋ถ, 0.5km๋ด ๋ฒ์ค ์ ๋ฅ์ฅ ์์ ์ ๋ฐ, ๋ผ๋ฒจ ์ธ์ฝ๋ฉ ๋ ์์น๊ตฌ
2024๋ 1์ 1์ผ ~ 2024๋ 2์ 28์ผ ๋ฐ์ดํฐ๋ฅผ Test Set์ผ๋ก ๋ง๋ค๊ณ , ์ ์ฒด ๋ฐ์ดํฐ๋ฅผ 8:2๋ก Train/Val๋ก ๋๋
์ค์ผ์ผ๋ง์ด ํ์ํ KNN, ์ ํํ๊ท๋ง ์ค์ผ์ผ๋ง์ ์ ์ฉํ์์
Optuna ๋ผ์ด๋ธ๋ฌ๋ฆฌ๋ฅผ ์ด์ฉํ์ฌ ์ต์ ํํ์์
base_models = [
('Random Forest', RandomForestRegressor(n_estimators=200, random_state=42, n_jobs=-1)),
('CatBoost', CatBoostRegressor(iterations=299, depth=10, learning_rate=0.2953026, random_strength=7, bagging_temperature=0.02308666, border_count=130, l2_leaf_reg=0.042969, random_state=42)),
('KNN', knn_pipeline),
('Decision Tree', DecisionTreeRegressor(random_state=42))
]
meta_model = XGBRegressor(n_estimators=563, max_depth=3, learning_rate=0.0151, min_child_weight=1, subsample=0.8967, colsample_bytree=0.9831, reg_alpha=3.029,reg_lambda=0.9275, random_state=42)
stacking_model = StackingRegressor(estimators=base_models, final_estimator=meta_model)
stacking_model.fit(X_train, y_train)
์ฑ๋ฅ์ด ์ข์ ๋ชจ๋ธ๋ค์ ์คํํน ํ์์
์ค์ ๋ก Test์ ์ ๊ฐ์ฅ ์ ํฉํ๋๊ฑด XGBoost ์์.
def update_neighborhoods(*args):
selected_district = district_var.get()
neighborhoods_menu['menu'].delete(0, 'end')
for neighborhood in districts[selected_district]:
neighborhoods_menu['menu'].add_command(label=neighborhood, command=lambda n=neighborhood: neighborhood_var.set(n))
neighborhood_var.set(list(districts[selected_district])[0])
def search_action():
search_input = neighborhood_var.get() + ' ' + entry.get()
result_text.delete(1.0, "end")
result_text.insert("end", "๊ฒ์์ด: " + search_input + "\n")
print("๊ฒ์์ด:", search_input)
search_url = f"https://m.land.naver.com/search/result/{search_input}#mapFullList"
driver.get(search_url)
html = driver.page_source
soup = BeautifulSoup(html, 'html.parser')
script = soup.find('script', string=re.compile(r'lat\s*:\s*\'\d+\.\d+\''))
if script:
lat_match = re.search(r'lat\s*:\s*\'(\d+\.\d+)\'', script.string)
lng_match = re.search(r'lng\s*:\s*\'(\d+\.\d+)\'', script.string)
if lat_match and lng_match:
lat = lat_match.group(1)
lng = lng_match.group(1)
predicted_price = predict_price(lat, lng, search_input, district_var.get(), neighborhood_var.get())
result_text.insert("end", f"์์ธก๋ ๊ฐ๊ฒฉ: {int(predicted_price[0])}๋ง์\n")
else:
result_text.insert("end", "์๋์ ๊ฒฝ๋๋ฅผ ์ฐพ์ ์ ์์ต๋๋ค.\n")
else:
print("์๋์ ๊ฒฝ๋ ์ ๋ณด๋ฅผ ํฌํจํ๋ ์คํฌ๋ฆฝํธ๋ฅผ ์ฐพ์ ์ ์์ต๋๋ค.")
result_text.pack()
root = tk.Tk()
root.title("๋ค์ด๋ฒ ๋ถ๋์ฐ ๊ฒ์")
result_text = tk.Text(root, height=10, width=50)
district_var = tk.StringVar(root)
district_var.set('์ ํํ์ธ์')
district_var.trace('w', update_neighborhoods)
districts_menu = ttk.OptionMenu(root, district_var, *districts.keys())
districts_menu.pack(side=tk.LEFT, padx=10)
neighborhood_var = tk.StringVar(root)
neighborhoods_menu = ttk.OptionMenu(root, neighborhood_var, '')
neighborhoods_menu.pack(side=tk.LEFT, padx=10)
update_neighborhoods()
entry = ttk.Entry(root)
entry.pack(side=tk.LEFT, padx=10)
search_button = ttk.Button(root, text="๊ฒ์", command=search_action)
search_button.pack(side=tk.LEFT, padx=10)
root.mainloop()
tkinter์ selenium์ผ๋ก ๊ตฌํํ์์
๊ธฐ๋ณธ ํ๋ฉด
๊ตฌ์ ๋์ ์ ํํ๊ณ ์ํํธ ๊ฐ๊ฒฉ์ ์ ๋ ฅํ๋ฉด ๊ธฐ์กด์ ํ์ต๋ ๋จธ์ ๋ฌ๋ ๋ชจ๋ธ์ ํตํด์ ๊ฐ๊ฒฉ์์ธก์ ์์ํจ
์ค์ ๋ฐ์ดํฐ์ ํฐ ์ฐจ์ด๊ฐ ์๋ ๋ชจ์ต์ ๋ณผ ์ ์๋ค.