In [1]:
from dotenv import load_dotenv

load_dotenv()

True

In [2]:
%load_ext autoreload
%autoreload 2

In [3]:
try:
    # Prefer Pydantic v2
    from pydantic import BaseModel, Field
except Exception:
    # Fallback for older stacks (not recommended for new code)
    from langchain_core.pydantic_v1 import BaseModel, Field

In [4]:
import os
import re
import json
import pickle
from glob import glob

from typing import Dict, List, Any, Optional
from dataclasses import dataclass

from langchain_openai import OpenAIEmbeddings, AzureChatOpenAI
from langchain_core.prompts import ChatPromptTemplate
from langchain_core.output_parsers import StrOutputParser

from sklearn.metrics.pairwise import cosine_similarity
from IPython.display import Markdown, display

In [5]:
from prompt_templates.narratives import create_narrative_angles_prompt_template
from prompt_templates.Future_plan import create_future_plan_prompt_template
from prompt_templates.Activity import create_activity_list_generator_prompt_template
from prompt_templates.Main_essay import create_main_essay_ideas_prompt_template

In [7]:
from langchain.agents import AgentExecutor, create_openai_tools_agent
from langchain_core.prompts import ChatPromptTemplate, MessagesPlaceholder

from tools import (
    generate_narrative_angles,
    generate_future_plan,
    generate_activity_list,
    generate_main_essay_ideas,
    route_tool_call,

)

In [8]:
from complete_strategy_tools import (generate_complete_application_strategy, route_tool_call_complete)

In [9]:
embeddings_model = OpenAIEmbeddings()

llm = AzureChatOpenAI(
    deployment_name="gpt-4o",
)

In [10]:
# folder_path = "clab_data"
# applications_data = []

# # Get all JSON files in the folder
# for file_path in glob(os.path.join(folder_path, "*.json")):
#     with open(file_path, "r", encoding="utf-8") as f:
#         applications_data = json.load(f)
# print(f"Loaded {len(applications_data)} applications from {len(glob(os.path.join(folder_path, '*.json')))} files")

In [11]:
# with open("clab_data/all_data.json", "r", encoding="utf-8") as f:
#     applications_data = json.load(f)

# print(f"Loaded {len(applications_data)} applications from all_data.json")

In [12]:
# @dataclass
# class Application:  # This will store accepted applications as well as their embeddings
#     filename: str
#     content: Dict[str, Any]
#     embedding: List[float] = None

In [13]:
# applications = []

# for key, application in applications_data.items():
#     embedding = embeddings_model.embed_query(json.dumps(application))

#     application = Application(
#         filename=key,
#         content=application,
#         embedding=embedding,
#     )

#     applications.append(application)

# with open("application_embeddings.pkl", "wb") as f:
#     pickle.dump(applications, f)

In [14]:
# applications = pickle.load(open("application_embeddings.pkl", "rb"))

In [15]:
DUMMY_USER_DATA = {
    "name": "Sarah Johnson",
    "gender": "Female",
    "birth_country": "United States",
    "cultural_background": [
        "American",
        "first-generation college student",
        "rural upbringing",
        "STEM-focused",
        "environmental advocate",
    ],
    "academic": {
        "gpa": "3.95",
        "school_type": "Public High School",
        "test_scores": {"SAT": "1450", "ACT": "32"},
    },
    "activities": [
        {
            "position": "President",
            "category": "Academic",
            "description": "Led environmental science club, organized community clean-up events, coordinated with local government on sustainability initiatives",
        },
        {
            "position": "Team Captain",
            "category": "Athletics",
            "description": "Varsity soccer team captain, led team to state championship, organized team community service projects",
        },
        {
            "position": "Volunteer",
            "category": "Community Service",
            "description": "Tutored underprivileged students in math and science, developed curriculum for after-school program",
        },
        {
            "position": "Research Assistant",
            "category": "Research",
            "description": "Assisted professor in environmental impact study, analyzed data on renewable energy adoption in rural communities",
        },
    ],
    "future_plans": [
        "Study environmental engineering",
        "Work on renewable energy solutions",
        "Masters degree",
        "Start a non-profit focused on rural sustainability",
    ],
}

In [16]:
# user_embedding = embeddings_model.embed_query(json.dumps(DUMMY_USER_DATA))

In [17]:
# similarities = []

# for app in applications:
#     similarity = cosine_similarity([user_embedding], [app.embedding])[0][0]
#     similarities.append((app, similarity))

# top_3_similar = sorted(similarities, key=lambda x: x[1], reverse=True)[:3]

# print("Top 3 Accepted Applications:")
# print("\n".join([each[0].filename for each in top_3_similar]))

In [18]:
# application_narratives = []

# for each in top_3_similar:
#     temp = each[0].content

#     if "application_narrative" in temp:
#         application_narratives.append(temp["application_narrative"])

In [19]:
from langchain.agents import create_openai_tools_agent, AgentExecutor
from langchain_core.prompts import ChatPromptTemplate, MessagesPlaceholder

prompt = ChatPromptTemplate.from_messages(
    [
        (
            "system",
            "You are an elite admissions strategist. "
            "ONLY call the router tool `route_tool_call(user_request, user_profile)` once per user request. "
            "Never call other tools directly. "
            "Pass the user's request verbatim as `user_request`, and always include `user_profile`. "
            "The router will selectively call the appropriate tool and handle dependencies (e.g., auto-generate narrative before future plan). "
            "Return the router's output as the final answer."
        ),
        ("user", "USER_PROFILE(JSON): {user_profile}\n\nTASK: {input}"),
        MessagesPlaceholder("agent_scratchpad"),
    ]
)

# Only expose the router tool to the agent to prevent direct multi-tool calls
tools = [
    route_tool_call,route_tool_call_complete
]

In [20]:
from langchain.memory import ConversationBufferMemory

memory = ConversationBufferMemory(
    memory_key="chat_history",
    return_messages=True,
    input_key="input",
)

agent = create_openai_tools_agent(llm, tools, prompt)

executor = AgentExecutor(agent=agent, tools=tools, verbose=True)


  memory = ConversationBufferMemory(


In [27]:
resp = executor.invoke(
    {
        "input": "Generate main essay ideas for this student",
        "user_profile": DUMMY_USER_DATA,
    }
)
print(resp["output"])



[1m> Entering new AgentExecutor chain...[0m


INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"


[32;1m[1;3m
Invoking: `route_tool_call` with `{'user_request': 'Generate main essay ideas for this student', 'user_profile': "{'name': 'Sarah Johnson', 'gender': 'Female', 'birth_country': 'United States', 'cultural_background': ['American', 'first-generation college student', 'rural upbringing', 'STEM-focused', 'environmental advocate'], 'academic': {'gpa': '3.95', 'school_type': 'Public High School', 'test_scores': {'SAT': '1450', 'ACT': '32'}}, 'activities': [{'position': 'President', 'category': 'Academic', 'description': 'Led environmental science club, organized community clean-up events, coordinated with local government on sustainability initiatives'}, {'position': 'Team Captain', 'category': 'Athletics', 'description': 'Varsity soccer team captain, led team to state championship, organized team community service projects'}, {'position': 'Volunteer', 'category': 'Community Service', 'description': 'Tutored underprivileged students in math and science, developed curriculum for

INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"
INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"
INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"
INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"


[36;1m[1;3m{
  "main_essay_ideas": [
    {
      "title": "Learning Leadership in the Mud",
      "theme": "Grassroots leadership and environmental passion",
      "hook": "The cold, wet mud seeped into my sneakers as I grabbed another bag overflowing with discarded plastic cups and rusted cans. It wasn’t glamorous, but it was change happening one step at a time.",
      "challenge": "The overwhelming scope of environmental damage in her rural community and the difficulty of rallying people to take action.",
      "journey": "Sarah decided to organize a clean-up in her town, recruiting neighbors, local officials, and friends despite initial resistance. She focused on empathy and education to drive participation.",
      "growth": "She learned that leadership wasn’t about being the loudest but about connecting people to a mission they could believe in. Seeing the collective impact proved that small efforts could create ripple effects.",
      "impact": "Her team collected nearly two t

In [28]:
resp = executor.invoke(
    {
        "input": "Generate complete strategy for this student",
        "user_profile": DUMMY_USER_DATA,
    }
)
print(resp["output"])



[1m> Entering new AgentExecutor chain...[0m


INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"


[32;1m[1;3m
Invoking: `route_tool_call_complete` with `{'user_request': 'Generate complete strategy for this student', 'user_profile': {'name': 'Sarah Johnson', 'gender': 'Female', 'birth_country': 'United States', 'cultural_background': ['American', 'first-generation college student', 'rural upbringing', 'STEM-focused', 'environmental advocate'], 'academic': {'gpa': '3.95', 'school_type': 'Public High School', 'test_scores': {'SAT': '1450', 'ACT': '32'}}, 'activities': [{'position': 'President', 'category': 'Academic', 'description': 'Led environmental science club, organized community clean-up events, coordinated with local government on sustainability initiatives'}, {'position': 'Team Captain', 'category': 'Athletics', 'description': 'Varsity soccer team captain, led team to state championship, organized team community service projects'}, {'position': 'Volunteer', 'category': 'Community Service', 'description': 'Tutored underprivileged students in math and science, developed curri

INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"
INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"
INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"
INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"
INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"
INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"
INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.az

[33;1m[1;3m{
  "complete_strategy": [
    {
      "narrative": {
        "title": "From Rural Fields to Renewable Futures",
        "positioning": "A rural upbringing sparks a visionary journey toward solving energy disparities in underserved communities.",
        "essay_concept": "Opening scene of rural landscape affected by environmental fatigue → Personal drive to tackle sustainability issues → Realization of the interconnectedness of energy access and equity → Activism and research in environmental engineering → Vision for sustainable rural energy networks.",
        "anchor_scene": "Sarah stands knee-deep in wild grass, surveying a community creek choked with debris after the latest storm, her boots caked in mud; a realization dawns that these small moments are symptomatic of a bigger issue—the neglect of rural environments and their residents.",
        "unexpected_twist": "That a student from rural America not only sees the problem of sustainability but constructs actionable,

In [30]:
from tools import json_to_markdown_llm

md = json_to_markdown_llm.invoke({"data": resp["output"]})

print(md)

INFO:httpx:HTTP Request: POST https://clabai3366454354.openai.azure.com/openai/deployments/gpt-4o/chat/completions?api-version=2024-02-01 "HTTP/1.1 200 OK"


# Comprehensive Strategy Overview

## Narrative 1: "From Rural Fields to Renewable Futures"

### Positioning
A rural upbringing sparks a visionary journey toward solving energy disparities in underserved communities.

### Essay Concept
- Scene: Rural landscape affected by environmental fatigue.  
- Personal journey: Tackling sustainability issues, connecting energy access and equity, engaging in activism, and researching environmental engineering.  
- Future Vision: Creating sustainable rural energy networks.  

### Anchor Scene
Sarah stands knee-deep in wild grass, surveying a debris-choked creek after a storm. The moment emphasizes the bigger issue: neglect of rural environments and residents.

### Unexpected Twist
A student from rural America not only identifies sustainability issues but also develops actionable, research-driven solutions.

### Signature Initiative
**Name:** Rural Energy Empowerment Project  
**Description:** Pilot for installing renewable wind and solar energy syst