In [1]:
%load_ext autoreload
%autoreload 2

In [23]:
import os
import time
import openai
from dotenv import load_dotenv

from selenium.webdriver.common.by import By
from bs4 import BeautifulSoup as bs

from method.ours.utils import create_driver
from method.ours.parse import (
    initial_parse,
    parse_after_feedback,
)
from method.ours.generation import (
    generate_constraints_for_input_groups,
    generate_constraints_for_input_groups_after_feedback,
    generate_values_for_input_groups,
    fill_form_with_value_table,
    submit_form
)
from method.ours.feedback import (
    get_local_feedback,
    get_global_feedback
)

In [3]:
load_dotenv()
openai.api_key = os.getenv("OPENAI_API_KEY")

# Global Variables
HEADLESS = False
TEXT_EMBEDDING_METHOD = 'ADA' # ['ADA', 'WORD2VEC', 'SPACY']
GRAPH_EMBEDDING_METHOD = 'NODE2VEC' # ['NODE2VEC', 'GCN']

In [4]:
driver = create_driver(HEADLESS)
driver.get('https://www.aircanada.com/ca/en/aco/home.html')

In [5]:
# form = driver.find_elements(By.TAG_NAME, 'form')[48]
# form = driver.find_element(By.ID, 'register')
form = driver.find_elements(By.TAG_NAME, 'form')[1]

# Processing

In [6]:
html = driver.find_element(By.TAG_NAME, 'body').get_attribute('outerHTML')
relation_graph, input_groups = initial_parse(driver, form, TEXT_EMBEDDING_METHOD)

Computing transition probabilities:   0%|          | 0/340 [00:00<?, ?it/s]

Generating walks (CPU: 2): 100%|██████████| 50/50 [00:00<00:00, 498.51it/s]
Generating walks (CPU: 1): 100%|██████████| 50/50 [00:00<00:00, 472.87it/s]
Generating walks (CPU: 3): 100%|██████████| 50/50 [00:00<00:00, 488.99it/s]
Generating walks (CPU: 4): 100%|██████████| 50/50 [00:00<00:00, 475.57it/s]
100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 34/34 [00:09<00:00,  3.66it/s]
  return np.dot(a, b) / (np.linalg.norm(a) * np.linalg.norm(b))


# Generation

In [7]:
value_table = generate_constraints_for_input_groups(input_groups)

 [0m Prompt: [
  {
    "role": "system",
    "content": "Instructions:\nYour task is to generate a set of constraints for web form fields. Your decisions must be made independently without seeking user assistance or additional information. If there are multiple ways to express constraints, use the least number of constraints to describe them. Only generate the constraints and refrain from explaining your answers. Only generate constraints for the input field in question, not those in the relevant information section. You must choose your constraints in the format of our modified version of the Jest library in JavaScript. The list of functions in this modified format are:\n1. toBeEqual(value) # the input field value is exactly equal to the given value\n2. toHaveLengthCondition(condition, value) # the length of the input field value matches the given condition\n3. toBeTruthy() # the input field value is truthy and not empty (not false, 0, '', null, undefined, or NaN)\n4. toHaveCompareCo

 [0m Response: {
  "choices": [
    {
      "finish_reason": "stop",
      "index": 0,
      "message": {
        "content": "expect(field('bkmgFlights_tripTypeSelector_M')).toBeEqual('M')",
        "role": "assistant"
      }
    }
  ],
  "created": 1688329798,
  "id": "chatcmpl-7XxzigGOcF19Vb0zDNuB126CxJVoq",
  "model": "gpt-4-0613",
  "object": "chat.completion",
  "usage": {
    "completion_tokens": 17,
    "prompt_tokens": 577,
    "total_tokens": 594
  }
}
 [0m Total running cost: $0.055
 [0m Prompt: [
  {
    "role": "system",
    "content": "Instructions:\nYour task is to generate a set of constraints for web form fields. Your decisions must be made independently without seeking user assistance or additional information. If there are multiple ways to express constraints, use the least number of constraints to describe them. Only generate the constraints and refrain from explaining your answers. Only generate constraints for the input field in question, not those in the relev

 [0m Response: {
  "choices": [
    {
      "finish_reason": "stop",
      "index": 0,
      "message": {
        "content": "expect(field('bkmgFlights_destination_trip_1'))\n.toBeTruthy()\n.toBeAlphabetical()\n.not.toEqual(field('bkmgFlights_origin_trip_1'))",
        "role": "assistant"
      }
    }
  ],
  "created": 1688329804,
  "id": "chatcmpl-7Xxzodb1gCuC05h7UO0myGdZhFzvL",
  "model": "gpt-4-0613",
  "object": "chat.completion",
  "usage": {
    "completion_tokens": 34,
    "prompt_tokens": 605,
    "total_tokens": 639
  }
}
 [0m Total running cost: $0.112
 [0m Prompt: [
  {
    "role": "system",
    "content": "Instructions:\nYour task is to generate a set of constraints for web form fields. Your decisions must be made independently without seeking user assistance or additional information. If there are multiple ways to express constraints, use the least number of constraints to describe them. Only generate the constraints and refrain from explaining your answers. Only gener

In [8]:
value_table = generate_values_for_input_groups(value_table, input_groups)

 [0m Prompt: [
  {
    "role": "system",
    "content": "Your task is to generate a value for a web form field based on the form field information and a set of constraints on the field. Your decisions must be made independently without seeking user assistance or additional information. For each user prompt, you need to generate one value that satisfies the constraints while keeping in mind the nature of the input from the available information. Only generate value and refrain from explaining your answers. Only generate value for the input field in question, and not the ones in the relevant information section."
  },
  {
    "role": "user",
    "content": "The following are all the labels in the form, which provide context for the functionality of the form:  Round-trip ,  One-way ,  Multi-city/Stopover ,  Book with points , From, To, Departure, Return\n    We are generating filling values for the following input field:\n    input: <input id=\"bkmgFlights_tripTypeSelector_R\" name=\"tri

 [0m Response: {
  "choices": [
    {
      "finish_reason": "stop",
      "index": 0,
      "message": {
        "content": "New York",
        "role": "assistant"
      }
    }
  ],
  "created": 1688329823,
  "id": "chatcmpl-7Xy07pM3RcVChutxpIXQK5y80Px5g",
  "model": "gpt-4-0613",
  "object": "chat.completion",
  "usage": {
    "completion_tokens": 2,
    "prompt_tokens": 255,
    "total_tokens": 257
  }
}
 [0m Total running cost: $0.198
 [0m Prompt: [
  {
    "role": "system",
    "content": "Your task is to generate a value for a web form field based on the form field information and a set of constraints on the field. Your decisions must be made independently without seeking user assistance or additional information. For each user prompt, you need to generate one value that satisfies the constraints while keeping in mind the nature of the input from the available information. Only generate value and refrain from explaining your answers. Only generate value for the input field in

In [9]:
fill_form_with_value_table(driver, value_table, input_groups)

<input>value is R</input> at y: (567, 597), x: (219, 319)
R
<input>value is O</input> at y: (567, 597), x: (334, 421)
O
<input>value is M</input> at y: (567, 597), x: (437, 591)
M
<input></input> at y: (574, 594), x: (652, 905)
Checked
<input></input> at y: (628, 678), x: (198, 466)
Vancouver
<input></input> at y: (628, 678), x: (536, 804)
New York
<input>DD/MM</input> at y: (628, 678), x: (833, 958)
15/07
<input>DD/MM</input> at y: (628, 678), x: (978, 1093)
20/07


In [11]:
submit_form(driver, input_groups)

# Feedback with Diff

In [13]:
new_html = driver.find_element(By.TAG_NAME, 'body').get_attribute('outerHTML')

In [14]:
global_feedback = get_global_feedback(html, new_html)

In [15]:
global_feedback

["We're sorry, we were not able to process your request. Please try again, or contact usOpens in a new tab for assistance.7"]

# Feedback

In [16]:
new_form = driver.find_elements(By.TAG_NAME, 'form')[1]

In [17]:
new_relation_graph, new_input_groups = parse_after_feedback(driver, new_form, relation_graph, TEXT_EMBEDDING_METHOD)

Computing transition probabilities:   0%|          | 0/366 [00:00<?, ?it/s]

Generating walks (CPU: 1): 100%|██████████| 50/50 [00:00<00:00, 538.28it/s]
Generating walks (CPU: 2): 100%|██████████| 50/50 [00:00<00:00, 504.29it/s]
Generating walks (CPU: 3): 100%|██████████| 50/50 [00:00<00:00, 482.96it/s]
Generating walks (CPU: 4): 100%|██████████| 50/50 [00:00<00:00, 497.56it/s]
100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 38/38 [00:10<00:00,  3.61it/s]


In [18]:
# diff = relation_graph.diff(new_relation_graph)

In [25]:
value_table = generate_constraints_for_input_groups_after_feedback(
    value_table,
    new_input_groups,
    global_feedback
)

 [0m Prompt: [
  {
    "role": "system",
    "content": "Instructions:\nYour task is to generate a set of constraints for web form fields. Your decisions must be made independently without seeking user assistance or additional information. If there are multiple ways to express constraints, use the least number of constraints to describe them. Only generate the constraints and refrain from explaining your answers. Only generate constraints for the input field in question, not those in the relevant information section. You must choose your constraints in the format of our modified version of the Jest library in JavaScript. The list of functions in this modified format are:\n1. toBeEqual(value) # the input field value is exactly equal to the given value\n2. toHaveLengthCondition(condition, value) # the length of the input field value matches the given condition\n3. toBeTruthy() # the input field value is truthy and not empty (not false, 0, '', null, undefined, or NaN)\n4. toHaveCompareCo

 [0m Response: {
  "choices": [
    {
      "finish_reason": "stop",
      "index": 0,
      "message": {
        "content": "expect(field('bkmgFlights_tripTypeSelector_M')).toBeEqual('M')",
        "role": "assistant"
      }
    }
  ],
  "created": 1688330184,
  "id": "chatcmpl-7Xy5wfi0AZBl9KDnF9p5hogsHi62J",
  "model": "gpt-4-0613",
  "object": "chat.completion",
  "usage": {
    "completion_tokens": 17,
    "prompt_tokens": 691,
    "total_tokens": 708
  }
}
 [0m Total running cost: $0.278
 [0m Prompt: [
  {
    "role": "system",
    "content": "Instructions:\nYour task is to generate a set of constraints for web form fields. Your decisions must be made independently without seeking user assistance or additional information. If there are multiple ways to express constraints, use the least number of constraints to describe them. Only generate the constraints and refrain from explaining your answers. Only generate constraints for the input field in question, not those in the relev

 [0m Response: {
  "choices": [
    {
      "finish_reason": "stop",
      "index": 0,
      "message": {
        "content": "expect(field('bkmgFlights_destination_trip_1'))\n.toBeTruthy()\n.toBeAlphabetical()\n.not.toBeEqual(field('bkmgFlights_origin_trip_1'))",
        "role": "assistant"
      }
    }
  ],
  "created": 1688330191,
  "id": "chatcmpl-7Xy63P4gwMveU429brZix2JG7ISOo",
  "model": "gpt-4-0613",
  "object": "chat.completion",
  "usage": {
    "completion_tokens": 34,
    "prompt_tokens": 659,
    "total_tokens": 693
  }
}
 [0m Total running cost: $0.342
 [0m Prompt: [
  {
    "role": "system",
    "content": "Instructions:\nYour task is to generate a set of constraints for web form fields. Your decisions must be made independently without seeking user assistance or additional information. If there are multiple ways to express constraints, use the least number of constraints to describe them. Only generate the constraints and refrain from explaining your answers. Only gen

In [27]:
value_table = generate_values_for_input_groups(value_table, input_groups)

 [0m Prompt: [
  {
    "role": "system",
    "content": "Your task is to generate a value for a web form field based on the form field information and a set of constraints on the field. Your decisions must be made independently without seeking user assistance or additional information. For each user prompt, you need to generate one value that satisfies the constraints while keeping in mind the nature of the input from the available information. Only generate value and refrain from explaining your answers. Only generate value for the input field in question, and not the ones in the relevant information section."
  },
  {
    "role": "user",
    "content": "The following are all the labels in the form, which provide context for the functionality of the form:  Round-trip ,  One-way ,  Multi-city/Stopover ,  Book with points , From, To, Departure, Return\n    We are generating filling values for the following input field:\n    input: <input id=\"bkmgFlights_tripTypeSelector_R\" name=\"tri

 [0m Response: {
  "choices": [
    {
      "finish_reason": "stop",
      "index": 0,
      "message": {
        "content": "New York",
        "role": "assistant"
      }
    }
  ],
  "created": 1688330226,
  "id": "chatcmpl-7Xy6ctlLkOdxA9yTTqDt7FnyI2jem",
  "model": "gpt-4-0613",
  "object": "chat.completion",
  "usage": {
    "completion_tokens": 2,
    "prompt_tokens": 301,
    "total_tokens": 303
  }
}
 [0m Total running cost: $0.436
 [0m Prompt: [
  {
    "role": "system",
    "content": "Your task is to generate a value for a web form field based on the form field information and a set of constraints on the field. Your decisions must be made independently without seeking user assistance or additional information. For each user prompt, you need to generate one value that satisfies the constraints while keeping in mind the nature of the input from the available information. Only generate value and refrain from explaining your answers. Only generate value for the input field in

In [28]:
fill_form_with_value_table(driver, value_table, input_groups)

<input>value is R</input> at y: (649, 679), x: (219, 319)
R
<input>value is O</input> at y: (649, 679), x: (334, 421)
O
<input>value is M</input> at y: (649, 679), x: (437, 591)
M
<input></input> at y: (656, 676), x: (652, 905)
true
<input></input> at y: (710, 760), x: (198, 466)
Vancouver
<input></input> at y: (710, 760), x: (536, 804)
New York
<input>DD/MM</input> at y: (710, 760), x: (833, 958)
15/07
<input>DD/MM</input> at y: (710, 760), x: (978, 1093)
20/07


In [29]:
submit_form(driver, input_groups)