In [47]:
from datasets import load_dataset, Dataset
from transformers import AutoTokenizer
import pandas as pd
pd.set_option('display.max_colwidth', 0)
import jsonlines
from tqdm.auto import tqdm

In [56]:
tokenizer = AutoTokenizer.from_pretrained('facebook/xglm-7.5B')
tokenizer('hey')

{'input_ids': [2, 30502], 'attention_mask': [1, 1]}

## Chip2

In [63]:
ds = []
with jsonlines.open('../data_large/unified_chip2.jsonl') as reader:
    for obj in tqdm(reader):
        obj['nb_token'] = len(tokenizer(obj['text'])['input_ids'])
        ds.append(obj)
chip2_df = pd.DataFrame(ds)

0it [00:00, ?it/s]

In [64]:
chip2_df.sample(5)

Unnamed: 0,text,metadata,nb_token
209901,"<human>: How common is the medical condition called ‘Symptom’?\n<bot>: Symptom is a very common condition. It affects about 1% of the population. Symptoms can include fatigue, muscle aches, joint pain, headaches, and difficulty concentrating.",{'source': 'unified_chip2'},58
112107,"<human>: If I want to practice with my bow and arrow, would using a horse be the best target?\n<bot>: That depends on the type of training you’re doing and what your purpose is for the arrow.",{'source': 'unified_chip2'},51
32684,"<human>: I recently started working out and I'm wondering if there are any supplements that I should add to my diet?\n<bot>: That depends on what your goals are. Would you like to build muscle, lose weight, or improve your athletic performance?",{'source': 'unified_chip2'},58
17780,"<human>: I've been hearing a lot of bad things about the new health care law. Is it really that bad?\n<bot>: Well, I’ve heard some people say it’s going to make things worse for them. But there are also a lot of positive stories out there.",{'source': 'unified_chip2'},64
74846,"<human>: How to I make the perfect cappuccino?\n<bot>: To make the perfect cappuccino, you will need freshly ground espresso, steamed milk and frothed milk. Begin by adding 1-2 ounces of espresso shots to your cup. Next, take your steamed milk and pour it into the cup until it is two-thirds full. Lastly, use a whisk or a frother to create frothy milk, then pour the frothed milk over the top of your espresso and steamed milk. Enjoy the perfect cappuccino!",{'source': 'unified_chip2'},120


In [66]:
chip2_df.nb_token.quantile([i/10 for i in range(10)])

0.0    23.0 
0.1    50.0 
0.2    61.0 
0.3    72.0 
0.4    82.0 
0.5    91.0 
0.6    101.0
0.7    111.0
0.8    124.0
0.9    144.0
Name: nb_token, dtype: float64

## Dolly 2

In [75]:
ds = []
with jsonlines.open('../data_large/databricks-dolly-15k.jsonl') as reader:
    for obj in tqdm(reader):
        if obj['context']!='':
            obj['text'] = f"<context>: {obj['context']}\n<human>: {obj['instruction']}\n<bot>: {obj['response']}"
        else:
            obj['text'] = f"<human>: {obj['instruction']}\n<bot>: {obj['response']}"
        obj['metadata'] = {'source': 'databricks-dolly-15k'}
        obj['nb_token'] = len(tokenizer(obj['text'])['input_ids'])
        ds.append(obj)

dolly2_df = pd.DataFrame(ds)[['text','metadata','nb_token']]

In [77]:
dolly2_df.sample(5)

Unnamed: 0,text,metadata,nb_token
3846,"<human>: What is a verb?\n<bot>: A verb is a word that describes what the subject of a sentence is doing. Verbs can indicate (physical or mental) actions, occurrences, and states of being.",{'source': 'databricks-dolly-15k'},51
11906,"<human>: Compare middle eastern music and western music\n<bot>: If you have only listened to Western music, the first time you hear o a middle eastern melody it may sound very different and odd. Melodies in middle eastern music are generally long, lack harmony, and have unfamiliar pitches.\n\nMiddle Eastern music typically sounds monophonic, meaning it's solely based on one or more instruments that play a single melody with a different variation. Western music typically sounds polyphonic, meaning more than one layer of melody is being played simultaneously.\n\nWestern music has twelve notes per octave including seven whole notes and five half-notes. However middle eastern music is based on different scale systems. It has quarter notes in addition and various regions have a different number of notes in their octave system.\n\nWestern music has harmony (ie. chords) while middle eastern music does not. In middle eastern music, there is only one melodic line with rise and fall. So the purity of the melodic line and the complexity of the rhythm is what make two different songs apart. In contrast, western music has a structure with a beginning, middle, and end.",{'source': 'databricks-dolly-15k'},252
6723,"<context>: Pearl Mill Village Historic District is a national historic district located at Durham, Durham County, North Carolina. The district encompasses 26 contributing residential buildings built by owners of Pearl Cotton Mills. The mill village dwellings are either two-story duplex type built about 1905 or a one-story bungalow constructed about 1924.[2]\n<human>: Given this paragraph about the Pearl Mill Village Historic District, tell me where it is located.\n<bot>: The Pearl Mill Village Historic District is in Durham, North Carolina.",{'source': 'databricks-dolly-15k'},115
2441,"<context>: August (2001), is the first novel by author Gerard Woodward. It was shortlisted for Whitbread Book Award (2001).\n<human>: Given this paragraph, who wrote ""August?""\n<bot>: Gerard Woodward",{'source': 'databricks-dolly-15k'},51
8174,"<context>: The word ""autophagy"" was in existence and frequently used from the middle of the 19th century.[11] In its present usage, the term autophagy was coined by Belgian biochemist Christian de Duve in 1963 based on his discovery of the functions of lysosome.[3] The identification of autophagy-related genes in yeast in the 1990s allowed researchers to deduce the mechanisms of autophagy,[12][13][14][15][16] which eventually led to the award of the 2016 Nobel Prize in Physiology or Medicine to Japanese researcher Yoshinori Ohsumi.[17]\n<human>: Who won the 2016 Nobel Prize?\n<bot>: The Japanese researcher Yoshinori Ohsumi.",{'source': 'databricks-dolly-15k'},156


In [78]:
dolly2_df.nb_token.quantile([i/10 for i in range(10)])

0.0    11.0 
0.1    35.0 
0.2    48.0 
0.3    67.0 
0.4    88.0 
0.5    114.0
0.6    149.0
0.7    197.0
0.8    268.0
0.9    397.0
Name: nb_token, dtype: float64

## HC3

In [80]:
ds = []
with jsonlines.open('../data_large/unified_hc3_human.jsonl') as reader:
    for obj in tqdm(reader):
        obj['nb_token'] = len(tokenizer(obj['text'])['input_ids'])
        ds.append(obj)
hc3_df = pd.DataFrame(ds)

0it [00:00, ?it/s]

In [81]:
hc3_df.sample(5)

Unnamed: 0,text,metadata,nb_token
9248,"<human>: How does a boom from a meteor entering the atmosphere manage to extend across such a wide area? Live in Detroit, and looking for some explanation after I think one broke up in the atmosphere relatively close to us . Explain like I'm five.\n<bot>: It was loud simply because of the amount of energy a meteor is carrying . It was probably going at least 25000 mph, maybe several times that . It was heard across a wide area partly because it was high up . Additionally, it happened on a cold night, and the temperature gradient in the air (colder air nearer the ground) likely helped bend down sound waves that would have otherwise traveled too far to be heard.",{'source': 'unified_hc3_human'},154
14310,"<human>: Why did humans start using money? Why couldn't we just live happily as a money - less society? Please explain like I'm five.\n<bot>: Money is probably the greatest economic invention of all time . It lets people get things that they need without actually having to trade for it, and to let people become specialists which accelerated technological advancement . Before money, there were no people who only were masons because they would go hungry since they would have to trade work directly to a farmer for food . After money, the farmer paid the mason, which allowed him to focus on farming without having to know how to build his house . The mason took the money and paid toolmakers who then paid other farmers for their food . Humans being what they are, a moneyless society would break down pretty fast because there would always be slackers who just rely on asking others for help without giving much effort in return.",{'source': 'unified_hc3_human'},200
15626,"<human>: How do submarines withstand so much water pressure? Man has reached 7 - mile ocean depth . How does a machine withstand that much pressure? Please explain like I'm five.\n<bot>: Take an egg from your fridge, and try to hold it from the top and bottom, and squeeze it . It is near impossible to crush, due to the shape of it . Submarines work that same way, also made of really strong metals to help with the pressure.",{'source': 'unified_hc3_human'},109
40199,"<human>: What do surveyers do? I see these guys out in the city all the time looking through their equipment like telescopes, but I never knew exactly what they were doing . Please explain like I'm five.\n<bot>: The equipment I believe your referring to is a "" theodolite "", also called a "" total station "" . It it a range / angle finding tool that can measure three different distances (horizontal, vertical, and slope), as well as the two different angles (azimuth, and slope) Imagine a cross - section of a hill where the hypotenuse is the surface of the hill, the length of which is the slope distance; the horizontal distance is the bottom leg of the triangle and the vertical distance is the opposite leg . The theodolite shoots a laser into a passive tool called a "" prism "" that reflects the laser back at it and calculates the distance . The azimuth angle is usually taken from some fixed point due north, and this allows you to find the angles for an entire parcel of land without using a protractor or something equally inaccurate and inefficient . The slope angle is just the angle at which the hill is inclined or declined . While theodolites are usually the most common method of surveying, the same can be accomplished without such an advanced piece of equipment by using an auto - level and a high rod . The auto - level is appears to be a telescope, but the distinguishing difference is that the auto - level has a series of wire suspended mirrors within the tube that ensure the surveyor only ever looks purely horizontal from wherever they are . This is used in conjunction with a high - rod, a really long stick with markings every inch or so, to determine vertical distance between two points . If you want to know more about general surveying, feel free to ask me! Source: Civil Engineering student",{'source': 'unified_hc3_human'},415
55668,"<human>: How does freeze drying work I thought you needed heat to substract water, not ice . Explain like I'm five.\n<bot>: It involves the physics of pressure and something called "" sublimation "" where a solid object can evaporate into a gas, without ever being a liquid inbetween . Basically, as I understand it, you freeze the food first, making all the water in it turn to ice, then use a powerful vacuum to remove most of the air from the chamber the food is in . Applying a little bit of heat plus the low air pressure causes the ice to turn into very cold steam or water vapor, which is pulled away by the vacuum.",{'source': 'unified_hc3_human'},155


In [82]:
hc3_df.nb_token.quantile([i/10 for i in range(10)])

0.0    19.0 
0.1    77.0 
0.2    95.0 
0.3    112.0
0.4    130.0
0.5    153.0
0.6    181.0
0.7    218.0
0.8    279.0
0.9    401.0
Name: nb_token, dtype: float64

## Abstract Infill

In [84]:
ds = []
with jsonlines.open('../data_large/unified_abstract_infill.jsonl') as reader:
    for obj in tqdm(reader):
        obj['text'] = obj['text'].replace('Background:','<context>:')
        obj['nb_token'] = len(tokenizer(obj['text'])['input_ids'])
        ds.append(obj)
infill_df = pd.DataFrame(ds)

0it [00:00, ?it/s]

In [90]:
infill_df.sample(5)

Unnamed: 0,text,metadata,nb_token
123288,"<human>: Write me an article about Murabaḥah, ribh Arabic, Islamic jurisprudence, discussing in detail rent-to-own, agreed-upon in four paragraphs. End with a sentence about "" Critics/skeptics, the non-Muslim world.\n<bot>: Murabaḥah, murabaḥa, or murâbaḥah (Arabic: مرابحة, derived from ribh Arabic: ربح, meaning profit) was originally a term of fiqh (Islamic jurisprudence) for a sales contract where the buyer and seller agree on the markup (profit) or \""cost-plus\"" price for the item(s) being sold. In recent decades it has become a term for a very common form of Islamic (i.e., \""shariah compliant\"") financing, where the price is marked up in exchange for allowing the buyer to pay over time—for example with monthly payments (a contract with deferred payment being known as bai-muajjal).\nMurabaha financing is similar to a rent-to-own arrangement in the non-Muslim world, with the intermediary (e.g., the lending bank) retaining ownership of the item being sold until the loan is paid in full. There are also Islamic investment funds and sukuk (Islamic bonds) that use murabahah contracts.\nThe purpose of murabaha is to finance a purchase without involving interest payments, which most Muslims (particularly most scholars) consider riba (usury) and thus haram (forbidden). Murabaha has come to be \""the most prevalent\"" or \""default\"" type of Islamic finance. A proper murâbaḥah transaction differs from conventional interest-charging loans in several ways. The buyer/borrower pays the seller/lender at an agreed-upon higher price; instead of interest charges, the seller/lender makes a religiously permissible \""profit on the sale of goods\"".\nThe seller/financer must take actual possession of the good before selling it to the customer, and must assume \""any liability from delivering defective goods\"". Sources differ as to whether the seller is permitted to charge extra when payments are late, with some authors stating any late fees ought to be donated to charity, or not collected unless the buyer has \""deliberately refused\"" to make a payment. For the rate of markup, murabaha contracts \""may openly use\"" riba interest rates such as LIBOR \""as a benchmark\"", a practice approved of by the scholar Taqi Usmani. Conservative scholars promoting Islamic finance consider murabaha to be a \""transitory step\"" towards a \""true profit-and-loss-sharing mode of financing\"", and a \""weak\"" or \""permissible but undesirable\"" form of finance to be used where profit-and-loss-sharing is \""not practicable.\"" Critics/skeptics complain/note that in practice most \""murabaḥah\"" transactions are merely cash-flows between banks, brokers, and borrowers, with no buying or selling of commodities; that the profit or markup is based on the prevailing interest rate used in haram lending by the non-Muslim world; that \""the financial outlook\"" of Islamic murabaha financing and conventional debt/loan financing is \""the same\"", as is most everything else besides the terminology used.\n<human>: What is meant by an Islamic bank loan\n<bot>: According to Taqi Usmani ""in exceptional cases"" an Islamic bank or financial institution may lend cash to the customer for a murâbaḥah, but this is when the customer is acting as an ""agent"" of the bank in buying the good the customer needs financed.\n<human>: more please\n<bot>: [W]here direct purchase from the supplier is not practicable for some reason, it is also allowed that he makes the customer himself his agent to buy the commodity on his behalf.\n<human>: In what form is the loan received?\n<bot>: In this case the client first purchases the commodity on behalf of his financier and takes its possession as such.\n<human>: What is the financier's role in the loan?\n<bot>: Thereafter, he purchases the commodity from the financier for a deferred price.\n<human>: What are the other modes of financing?\n<bot>: The idea that the seller may not use murâbaḥah if profit-sharing modes of financing such as mudarabah or musharakah are practicable, is supported by other scholars that those in the Council of Islamic Ideology.\n<human>: What are the differences between the modes?\n<bot>: But these involve risks of loss, profit-sharing modes of financing cannot guarantee banks income.",{'source': 'infil_dbpedia'},1049
136897,"<context>: Black Country Communion is an English-American hard rock band based in Los Angeles, California. Formed in 2009, the band is a supergroup composed of bassist and lead vocalist Glenn Hughes, guitarist and second vocalist Joe Bonamassa, drummer Jason Bonham and keyboardist Derek Sherinian. Hughes and Bonamassa started the side project in November 2009 after an impromptu performance together. From the advice of their producer, Kevin Shirley, they recruited Bonham and Sherinian to finalize the band's lineup.\nThe name Black Country Communion is derived from the term Black Country, which refers to an area of the West Midlands of England where Hughes and Bonham grew up. The group's sound is intentionally reminiscent of popular classic rock groups of the 1970s, reflecting the previous work of frontman Hughes (in bands such as Trapeze and Deep Purple) as well as the link between Bonham and his father John's band Led Zeppelin. After playing together for a while, the group released its self-titled debut album in 2010. Black Country Communion 2 followed the next year, which was promoted on the group's first concert tour of the US and Europe.\nThe band released its third album, Afterglow, in 2012. In early 2013, after months of public tensions, Bonamassa announced he was no longer a member of Black Country Communion, with the question over whether the group would continue with another guitarist unclear. Days later, Hughes confirmed that the band was officially over, hinting that he would continue working with Bonham and Sherinian under a new collective name. Sherinian later joined Bonamassa's touring band, while Hughes and Bonham formed California Breed with guitarist Andrew Watt. Hughes and Bonamassa later reconciled and the band reunited in late 2016, with a fourth album BCCIV released in 2017. After four years working on separate projects, the band plans to enter studio for a potential fifth album in late 2021.\n<human>: What happened in 2010 with Black Country Communion?\n<bot>: The band recorded its debut album at Los Angeles' Shangri-La Studios in early 2010, scheduling a September release through Mascot Records in Europe and Bonamassa's label J&R Adventures in North America.\n<human>: tell me more\n<bot>: Hughes described the album as ""a big British rock statement"", comparing the band's sound to that of his previous groups Deep Purple and Black Sabbath, as well as Led Zeppelin.\n<human>: Did they have any hit songs?\n<bot>: "" One Last Soul"" was the first song to be released, receiving its worldwide debut on British digital radio station Planet Rock on August 2, 2010.\n<human>: Did they release any other songs?\n<bot>: The track was later released as a free digital download on the band's official website.\n<human>: Did they have any problems?\n<bot>: Shortly before the album's release, Planet Rock also broadcast an hour-long documentary featuring exclusive interviews with the band and a selection of tracks from the album.\n<human>: What happened in 2011?\n<bot>: ""Black Country Communion"" was officially released in Europe on September 20, 2010, and in North America a day later.",{'source': 'infil_dbpedia'},695
137945,"<context>: Metal Knight (Chinese: 铁甲风暴; lit. 'Ironclad Storm') is a sci-fi type real-time strategy released in 1998, developed by Object Software Limited and published in mainland China by Object Software and in Taiwan by T-time Technology Co., Ltd. In 1999 the English version was published by in the USA and Italy.\nThe game supports eight online players. The game sold over 300,000 copies and received high ratings from indigenous game magazines such as and Play. There is an expansion named Black Front.\n<human>: What is Metal Knight?\n<bot>: Metal Knight Metal Knight (Chinese:铁甲风暴, literally:Ironclad Storm) is a sci-fi type real-time strategy released in 1998, developed by Object Software Limited and published in mainland China by Object Software and in Taiwan by T-time Technology Co., Ltd.\n<human>: Was it popular?\n<bot>: In 1999 the English version was published BY Microforum in USA and Italy..\n<human>: How many players were supported?\n<bot>: The game supports eight online players.\n<human>: What was the greatest accomplishment?\n<bot>: The game sold over 300,000 copies and received high ratings from indigenous game magazines such as ""Popsoft"" and ""Play"".\n<human>: next\n<bot>: There is an expansion named Black Front.",{'source': 'infil_dbpedia'},296
17678,"<context>: Chris Anderson is an American politician and Democratic Party political activist. He was a member of the Chattanooga, Tennessee City Council for District 7, but was defeated in a run-off election on April 11, 2017. He was the first openly LGBT individual to be elected to Chattanooga's City Council. He also serves as a member of the Tennessee Democratic Party State Executive Committee.\n<human>: Tell me about Chris Anderson (politician).\n<bot>: Chris Anderson is an American politician and Democratic Party political activist.\n<human>: Who is Chris Anderson?\n<bot>: Chris Anderson (politician) Chris Anderson is an American politician and Democratic Party political activist.\n<human>: What did this politician accomplish?\n<bot>: He was a member of the Chattanooga, Tennessee City Council for District 7, but was defeated in a run-off election on April 11.\n<human>: What else did you find interesting?\n<bot>: He was the first openly LGBT individual to be elected to Chattanooga's City Council.\n<human>: Did he accomplish anything else?\n<bot>: He also serves as a member of the Tennessee Democratic Party State Executive Committee.\n<human>: What other causes did he support?\n<bot>: Anderson started his political career as a Democratic Party activist.\n<human>: What else was this politican notable for?\n<bot>: He was elected in 2003 as the President of the Tennessee Young Democrats which also gave him a seat on the Tennessee Democratic Party State Executive Committee.",{'source': 'infil_dbpedia'},318
205615,"<context>: HMS Sea Nymph was a S-class submarine of the third batch built for the Royal Navy during World War II. Completed in July 1942, she spent the majority of her career patrolling the waters off Norway in the North Sea, then was sent to the Pacific but was forced back due to technical problems. After three uneventful patrols in the North Sea, the boat conducted one patrol in the Bay of Biscay, attacking two surfaced German U-boats with torpedoes, but missed. The submarine then participated in Operation Source, an attempt to sink German battleships in Norway threatening allied Arctic convoys.\nSea Nymph was assigned to tow the midget submarine X8 to off the fjord where the Tirpitz was anchored. While at sea, X8 was scuttled due to technical issues, and the mission was cancelled. The submarine was then tasked with transporting British agents to a fishing vessel off Norway. In her next two patrols, the boat fired torpedoes on three German merchant ships, but missed every attack.\nIn March 1944, Sea Nymph was sent to the United States for a refit; she returned in October, then after an uneventful patrol, she had her battery changed. With Nazi Germany close to defeat, the submarine was ordered to the Pacific, where war with the Japanese was also ongoing. She had to turn back and have her battery replaced again after encountering problems; after arriving in the Philippines, on 13 July Sea Nymph caught on fire and was damaged beyond the repair capabilities of local shipyards. She was sent back to Great Britain, but as the war with Japan ended, she was not repaired but placed in reserve. Sea Nymph was scrapped in Troon, Scotland in June 1948.\n<human>: What was the HMS Sea Nymph (P223)?\n<bot>: HMS Sea Nymph (P223) HMS ""Sea Nymph"" was a S-class submarine of the third batch built for the Royal Navy during World War II.\n<human>: When was it built?\n<bot>: She survived the war and was scrapped in 1948. The S-class submarines were designed to patrol the restricted waters of the North Sea and the Mediterranean Sea.\n<human>: and?\n<bot>: The third batch was slightly enlarged and improved over the preceding second batch of the S-class.\n<human>: What were some of the improvements?\n<bot>: The submarines had a length of overall, a beam of and a draught of .\n<human>: What else was different about the third batch?\n<bot>: They displaced on the surface and submerged.\n<human>: How many people were on board?\n<bot>: The S-class submarines had a crew of 48 officers and ratings.",{'source': 'infil_dbpedia'},620


In [91]:
infill_df.nb_token.quantile([i/10 for i in range(10)])

0.0    48.0 
0.1    310.0
0.2    389.0
0.3    447.0
0.4    499.0
0.5    550.0
0.6    604.0
0.7    669.0
0.8    753.0
0.9    889.0
Name: nb_token, dtype: float64

In [102]:
infill_df[infill_df.nb_token<=512].shape

(99288, 3)

## TLDR

In [100]:
ds = []
with jsonlines.open('../data_large/unified_openai_summarize_tldr.jsonl') as reader:
    for obj in tqdm(reader):
#         obj['text'] = obj['text'].replace('Background:','<context>:')
        obj['nb_token'] = len(tokenizer(obj['text'])['input_ids'])
        ds.append(obj)
tldr_df = pd.DataFrame(ds)

0it [00:00, ?it/s]

In [101]:
tldr_df.sample(5)

Unnamed: 0,text,metadata,nb_token
18902,"<human>: Can you give me the gist of the text in a nutshell?\n\nI just moved to a new school and I found a group that I really wanted to be a part of. Now, the group in question organizes these sessions for tabletop games and the sessions are a very 'group only' sort of affair. A few months ago, I actually got invited to one and I was ecstatic. Unfortunately, I'm still a bit of an 'outsider'. On Wednesday, we have another session planned but the game that people want to play can only have six people and there are seven people invited including me. Several people have already commented how I'm not part of the 'core' group and that I shouldn't be invited at all. Should I just tell them I can't come to avoid the pain that is quite probably about to hit me? Should I just leave it and hope for the best?\n<bot>: newbiee in a group, invited to a game, too many people invited for said game, may be ditched",{'source': 'unified_openai_summarize_tldr'},222
194250,"<human>: My husbands cousin is a diesel mechanic and had an accident at work yesterday. A coworker improperly lifted a semi-trailer and it fell, crushing every bone in his right foot. When we visited last night, he was howling in pain while nurses kept running to get the resident to up his medications. \n\nThe company owner paid for his mom and other family to fly out to see him and paid for a hotel for them. He's having surgery right now to see if the foot can be saved. There's a potential for amputation. He's a huge football and sports player so the family is nervous. \n\nWhat can I do to show support? I'm wanting to make a big tray of muffins and pasta so that his visiting family can have something to eat. I'm sure they've been stressed to the point of improper nutrition. Other than that I'm not sure what I can do to make things a little easier. I've never experienced a family member having major surgery or illness. What should I be aware of during his recovery as well?\n\nCan you give me a shortened version of the text?\n<bot>: Any tips on what you've done or what you've had family do for you to help after major surgery would be appreciated.",{'source': 'unified_openai_summarize_tldr'},278
138399,"<human>: Give me a shortened version of the text\n\nYou kno the old saying home is where the heart is ,well i feel as if i have no heart and no real home. My mom lives with her boyfriend ,my sister just got her own place with my nephew and my little bothers and sister live with there dad. It feels like i don't have anywhere or anyone anymore. I dont feel any real emotion anymore . I thought i had true love in my life but that only got me thrown in jail cuz she cheated on me and i broke some kids jaw. Like I'm too hopeful to kill myself and I'm too comfortable in my gloom. I'm desperate for friends I'm desperate to find someone to just care and to love. I used to have some really close friends brothers if you may. But you move away and everyone just fades away. I'm good to people like too good to people and i get looked at as naive for this but its easier to feel like a kid than to just change . I'm 19 btw but i look like I'm 25. I wish i could do some drugs to just feel like i don't exist ..... I'm lonely, really fucking lonely. I'm such a nice person but that's just fading away to just being an asshole. I have no motivation to be better like i used to be ..... things were just easier when my family was just together in one house. It was easier being told what to do...\n<bot>: 19 went to jail came home to family being spread thin and poor to living with relatives and me being pushed out of home too soon and now moving along aimlessly not knowing who i am or where life is going.",{'source': 'unified_openai_summarize_tldr'},373
146876,<human>: Give me the gist of the text in a nutshell\n\nMe and my buddy K have been super close friends since our freshman year of college (5 years) since we were both comp sci majors at a small college. He had a gf back then but she was alright with me and they broke up a year or so after I met him. Since then he's been single till now with just a few dates here and there.\n\nI'm sorta insecure since most my friends are males (due to my interests) and they tend to get told to not hang with me by their gfs when they get them. Its happened numerous times in the past. \n\nI have just shrugged it off in the past but K is my closest friend and the person other than my bf(dating for 4 years) that I hang out with the most. Is there something I can do or say to this girl if this pops up or should I just fade into the background and chalk it up to not being a loyal enough friend?\n\nAny suggestions would be appreciated. If you need more info let me know. Thank you! :)\n<bot>: I am worried my best friends gf will tell him not to hang witg me like other friend's gfs have.,{'source': 'unified_openai_summarize_tldr'},269
48107,"<human>: Can you provide a brief overview of the text?\n\nWe met in college and live about five and a half hours apart. That distance has been the bane of our relationship. I found a great job near my hometown after college that has been able to take care of us in a major way. She is younger than me so it took her an extra year to finish school.\n\nAfter graduating, she had no real ties to home so we found an apartment in a location that isn't either of our hometowns, but that would allow me to continue earning a solid living and taking care of our major expenses. About 2.5 months ago (and a week after we signed the lease), her mom suffered a major stroke from an AVM. She has always been a perfectly healthy woman so that came as quite a shock to us all.\n\nMy girlfriend is now stuck at home assisting her mother during her recovery and she has become the main caretaker at home. All of the plans that we had have been disrupted in a major way. Now, we are struggling every day to find any kind of optimism in our relationship and future together.\n\nI feel guilty for contuining (or attempting to continue) my life while she takes care of her family. She feels guilty for having plans to move out. I'm making a huge sacrifice by paying for all of our expenses and waiting for her, and she's now making a huge sacrifice in considering to move away from her family during their time of need.\n\nWhat possible future exists for us here? What would you do/how would you feel if you were either of us in our respective situations?\n<bot>: My girlfriend's mother has been in the hospital after a major stroke and she lives 5.5 hours away taking care of her. Where do we find the optimism to continue waiting and making sacrifices to stay together and pursue our plans?",{'source': 'unified_openai_summarize_tldr'},389


In [109]:
tldr_df[tldr_df.text.map(lambda x: x[-1]!='?')&(tldr_df.nb_token<=512)].shape

(140502, 3)

## Combine

In [123]:
final_df = pd.concat([chip2_df,dolly2_df,hc3_df,
                      infill_df,
                      tldr_df[tldr_df.text.map(lambda x: x[-1]!='?')]\
                         .sample(n=30000,random_state=125)
                     ],
                     axis=0)
final_df = final_df[final_df.nb_token<=512].reset_index(drop=True)
final_df.to_csv('../data_large/final_df_v1.csv',index=False)
final_df.metadata.value_counts()

{'source': 'unified_chip2'}                    210289
{'source': 'infil_dbpedia'}                    99288 
{'source': 'unified_hc3_human'}                55205 
{'source': 'unified_openai_summarize_tldr'}    28436 
{'source': 'databricks-dolly-15k'}             14080 
Name: metadata, dtype: int64

In [124]:
final_df.shape

(407298, 3)

In [130]:
final_df.nb_token.quantile([i/10 for i in range(10)])

0.0    11.0 
0.1    58.0 
0.2    76.0 
0.3    91.0 
0.4    107.0
0.5    126.0
0.6    162.0
0.7    255.0
0.8    361.0
0.9    439.0
Name: nb_token, dtype: float64

In [131]:
final_df.nb_token.describe()

count    407298.000000
mean     194.566683   
std      143.883870   
min      11.000000    
25%      84.000000    
50%      126.000000   
75%      312.000000   
max      512.000000   
Name: nb_token, dtype: float64

## Convert to Dataset

In [126]:
ds = Dataset.from_pandas(final_df)
ds

Dataset({
    features: ['text', 'metadata', 'nb_token'],
    num_rows: 407298
})

In [129]:
ds.shuffle()[0]

{'text': '<human>: Why does Road Dogg announce that the New Age Outlaws are the "soon-to-be WWF Tag Team Champions of the world"?\n<bot>: Road Dogg announces that the New Age Outlaws are the "soon-to-be WWF Tag Team Champions of the world" because he wants to make sure that the fans know that the New Age Outlaws are the WWF Tag Team Champions of the world.',
 'metadata': {'source': 'unified_chip2'},
 'nb_token': 96}

In [133]:
ds_split = ds.train_test_split(test_size=1538, seed=125)
ds_split

DatasetDict({
    train: Dataset({
        features: ['text', 'metadata', 'nb_token'],
        num_rows: 405760
    })
    test: Dataset({
        features: ['text', 'metadata', 'nb_token'],
        num_rows: 1538
    })
})

In [None]:
ds_split.push_to_hub('pythainlp/final_training_set_v1', private=True)

Pushing split train to the Hub.


In [None]:
ds_split['train'][1000]