## Explore the Squad Dataset 2.0



In [None]:
#@title specify file_path for Squad Dataset
from google.colab import drive
import json
import os

drive.mount('/content/drive')

squad_train_path = '' # path to train-v1.1.json on gdrive
squad_dev_path = '' # path to dev-v1.1.json on gdrive

In [41]:
with open(squad_train_path) as file_d:
    squad_train_data = json.load(file_d)
    
with open(squad_dev_path) as file_d:
    squad_dev_data = json.load(file_d)

## Main Parse Function for Squad Dataset

In [54]:
def parse_squad_datset(squad_dataset: dict) -> dict:
    """
    input: List of dict, each dict is an exmample.

    ouput: dict with 4 item. Key: str; value: List[str]
    {
        contexts = [str1, ...] # note: contexts is optional
        question_ids = [str1, ...] # from id
        questions: [str1, ..] # from question
        answers: [str1, ...]
    }
    
    """
    contexts = []
    question_ids = []
    questions = []
    answers = []

    data = squad_dataset["data"]
    for example in data:
        paragraphs = example["paragraphs"]
        for paragraph in paragraphs: # paragraphs: List[json]
            context = paragraph["context"]
            qas = paragraph["qas"]
            for qa in qas:
                question = qa["question"]
                question_id = qa["id"]
                answer = qa["answers"][0]["text"]

                # appending
                contexts.append(context)
                question_ids.append(question_id)
                questions.append(question)
                answers.append(answer)

    return {
        "context": contexts, 
        "question_ids": question_ids,
        "questions": questions,
        "answers": answers,
    }

In [67]:
#@title define function: check_saved_dataset() { display-mode: "both" }

def check_saved_dataset(gs_dataset, i = 5):
    j = 0 
    # print(f"dataset: {string(gs_dataset)}")

    print(f"""
    len of context: {len(gs_dataset["context"])}
    len of questions: {len(gs_dataset["questions"])}
    len of ids: {len(gs_dataset["question_ids"])}
    len of answers: {len(gs_dataset["answers"])}
    """)
    
    for context, question, answer, id in zip(gs_dataset["context"], gs_dataset["questions"], gs_dataset["answers"], gs_dataset["question_ids"]):
        print(f"context: {context}")
        print(f"question: {question}")
        print(f"answer: {answer}")
        print(f"id: {id}")
        
        print("\n")
        j += 1
        if j == i:
            break


In [68]:
check_saved_dataset(parsed_squad_dev_dataset)


    len of context: 10570
    len of questions: 10570
    len of ids: 10570
    len of answers: 10570
    
context: Super Bowl 50 was an American football game to determine the champion of the National Football League (NFL) for the 2015 season. The American Football Conference (AFC) champion Denver Broncos defeated the National Football Conference (NFC) champion Carolina Panthers 24–10 to earn their third Super Bowl title. The game was played on February 7, 2016, at Levi's Stadium in the San Francisco Bay Area at Santa Clara, California. As this was the 50th Super Bowl, the league emphasized the "golden anniversary" with various gold-themed initiatives, as well as temporarily suspending the tradition of naming each Super Bowl game with Roman numerals (under which the game would have been known as "Super Bowl L"), so that the logo could prominently feature the Arabic numerals 50.
question: Which NFL team represented the AFC at Super Bowl 50?
answer: Denver Broncos
id: 56be4db0acb800140

In [69]:
check_saved_dataset(parsed_squad_train_dataset)


    len of context: 87599
    len of questions: 87599
    len of ids: 87599
    len of answers: 87599
    
context: Architecturally, the school has a Catholic character. Atop the Main Building's gold dome is a golden statue of the Virgin Mary. Immediately in front of the Main Building and facing it, is a copper statue of Christ with arms upraised with the legend "Venite Ad Me Omnes". Next to the Main Building is the Basilica of the Sacred Heart. Immediately behind the basilica is the Grotto, a Marian place of prayer and reflection. It is a replica of the grotto at Lourdes, France where the Virgin Mary reputedly appeared to Saint Bernadette Soubirous in 1858. At the end of the main drive (and in a direct line that connects through 3 statues and the Gold Dome), is a simple, modern stone statue of Mary.
question: To whom did the Virgin Mary allegedly appear in 1858 in Lourdes France?
answer: Saint Bernadette Soubirous
id: 5733be284776f41900661182


context: Architecturally, the school ha

In [81]:
#@title define functin: write_to_gs
import tensorflow as tf
# authenticate
# For accessing data from google storage (gs://)
from google.colab import auth
auth.authenticate_user()

def write_to_gs(path, parsed_dataset):
    with tf.io.gfile.GFile(path, 'w') as json_file:
        json.dump(parsed_dataset, json_file)


In [66]:
parsed_squad_train_dataset = parse_squad_datset(squad_train_data)
parsed_squad_dev_dataset = parse_squad_datset(squad_dev_data)

In [82]:
#@title Write parsed_squad data to Google Storage
squad_data_path = '' # path on gs to save parsed data
squad_train_data_path = os.path.join(squad_data_path, "squad1.1_train_data_parsed.json")
squad_dev_data_path = os.path.join(squad_data_path, "squad1.1_dev_data_parsed.json")


write_to_gs(squad_train_data_path, parsed_squad_train_dataset)
write_to_gs(squad_dev_data_path, parsed_squad_dev_dataset)

In [None]:
# download the file from google storage
!gsutil cp # copy parsed file to local
with open("/tmp/squad_dev_data_parsed.json") as a_file:
    check_squad_dev = json.load(a_file)

check_saved_dataset(check_squad_dev)