In [1]:
import json
from collections import defaultdict

from squadgym.utils.preprocessing import preprocess_text

In [76]:
def squad_reader(squad_data_filename):
    with open(squad_data_filename) as in_file:
        return json.load(in_file)


def generate_env_data(squad_data):
    env_data = defaultdict(lambda: {"context": [], "questions": [], "answers": []})

    for entity_data in squad_data["data"]:
        entity_title = entity_data["title"]

        for paragraph in entity_data["paragraphs"]:
            env_data[entity_title]["context"].append(preprocess_text(paragraph["context"]))
            questions = []
            answers = []
            for qas in paragraph["qas"]:
                question_tokens = preprocess_text(qas["question"])
                answers_tokens = []
                for ans in qas["answers"]:
                    answer_tokens = preprocess_text(ans["text"])
                    answers_tokens.append(answer_tokens)
                questions.append(question_tokens)
                answers.append(answers_tokens)
            env_data[entity_title]["questions"].append(questions)
            env_data[entity_title]["answers"].append(answers)
            
    return env_data

In [77]:
squad_data = squad_reader("/home-nfs/mingdachen/data/squad/dev-v1.1.json")
data = generate_env_data(squad_data)

In [65]:
' '.join(data["Oxygen"]["context"][0])

"oxygen is a chemical element with symbol o and atomic number 8 . it is a member of the chalcogen group on the periodic table and is a highly reactive nonmetal and oxidizing agent that readily forms compounds ( notably oxides ) with most elements . by mass , oxygen is the third-most abundant element in the universe , after hydrogen and helium . at standard temperature and pressure , two atoms of the element bind to form dioxygen , a colorless and odorless diatomic gas with the formula o 2 . diatomic oxygen gas constitutes 20.8 % of the earth 's atmosphere . however , monitoring of atmospheric oxygen levels show a global downward trend , because of fossil-fuel burning . oxygen is the most abundant element by mass in the earth 's crust as part of oxide compounds such as silicon dioxide , making up almost half of the crust 's mass ."

In [78]:
len(data["Oxygen"]["questions"])

43

In [79]:
len(data["Oxygen"]["context"])

43

In [56]:
squad_data['data'][0]['paragraphs'][0]

{'context': 'Super Bowl 50 was an American football game to determine the champion of the National Football League (NFL) for the 2015 season. The American Football Conference (AFC) champion Denver Broncos defeated the National Football Conference (NFC) champion Carolina Panthers 24–10 to earn their third Super Bowl title. The game was played on February 7, 2016, at Levi\'s Stadium in the San Francisco Bay Area at Santa Clara, California. As this was the 50th Super Bowl, the league emphasized the "golden anniversary" with various gold-themed initiatives, as well as temporarily suspending the tradition of naming each Super Bowl game with Roman numerals (under which the game would have been known as "Super Bowl L"), so that the logo could prominently feature the Arabic numerals 50.',
 'qas': [{'answers': [{'answer_start': 177, 'text': 'Denver Broncos'},
    {'answer_start': 177, 'text': 'Denver Broncos'},
    {'answer_start': 177, 'text': 'Denver Broncos'}],
   'id': '56be4db0acb8001400a5

In [10]:
' '.join(data['Oxygen']['context'])

'oxygen gas ( o 2 ) can be toxic at elevated partial pressures , leading to convulsions and other health problems . [ j ] oxygen toxicity usually begins to occur at partial pressures more than 50 kilopascals ( kpa ) , equal to about 50 % oxygen composition at standard pressure or 2.5 times the normal sea-level o 2 partial pressure of about 21 kpa . this is not a problem except for patients on mechanical ventilators , since gas supplied through oxygen masks in medical applications is typically composed of only 30 % –50 % o 2 by volume ( about 30 kpa at standard pressure ) . ( although this figure also is subject to wide variation , depending on type of mask ) .'