In [92]:
"""
This code is for constructing labeled preference pairs.
"""
import os
import json
from tqdm import tqdm
import numpy as np
import random
from random import shuffle
import pandas as pd
from transformers import AutoTokenizer
random.seed(42)


In [None]:
def calculate_length_distribution(text, tokenizer):
    tokenizer(text)

# Chap1: LLM Bench

## 1. UltraFeedback (AI Feedback)
path: https://hf-mirror.com/datasets/HuggingFaceH4/ultrafeedback_binarized

This data is used for iteration preference optimzation

In [3]:
ultra_feedback_dir = "UltraFeedback"

In [4]:
def read_ultralfeedback_data(data_dir):
    df_train = pd.read_parquet(os.path.join(data_dir, "train.parquet"))
    df_train = json.loads(df_train.to_json(orient='records'))
    df_test = pd.read_parquet(os.path.join(data_dir, "test.parquet"))
    df_test = json.loads(df_test.to_json(orient='records'))
    return df_train, df_test
    
ultralfeedback_train_data, ultralfeedback_test_data = read_ultralfeedback_data(ultra_feedback_dir)

In [134]:
ultralfeedback_test_data[20]

{'prompt': 'How does the use of an unreliable narrator impact the readers\' understanding of "Gone Girl" by Gillian Flynn?',
 'prompt_id': 'b3f9c485b2434c41e60ee7bc1d2eafaab22e16cd0dc851be8bfbd90ab84241d9',
 'chosen': [{'content': 'How does the use of an unreliable narrator impact the readers\' understanding of "Gone Girl" by Gillian Flynn?',
   'role': 'user'},
  {'content': 'The use of an unreliable narrator in "Gone Girl" significantly impacts the readers\' understanding of the story, as it challenges their perception of the characters and the narrative. Here\'s how:\n\n1. Keeps readers guessing: With unreliable narrators like Nick and Amy Dunne, the readers can never truly know who or what to believe. This creates an atmosphere of mystery and suspicion, making the reading experience more engaging and intriguing.\n\n2. Enhances suspense: By providing conflicting or misleading information, the unreliable narrators keep the readers on their toes and heighten the sense of suspense thro

In [21]:
def preprocess_ultra_feedback(examples):
    """
    Construct pair-wise data for preference optimzation
    return :the structure of example:
    
      {
        "conversations": [
          {
            "from": "system",
            "value": "You are a teacher. Given a task, you explain in simple steps what the task is asking, any guidelines it provides and how to use those guidelines to find the answer."
          },
          {
            "from": "human",
            "value": "Q: In this task, you need to provide the parts-of-speech tag of a word present in a sentence specified within curly braces  ( '{{ ... }}' ). The parts-of-speech tags are coarse labels that represent a category of words with similar grammatical properties. The list of part-of-speech tags i.e tagset of this corpus is - \n '.': Period symbol is used for symbols denoting Punctuations/Separations such as comma, period, backticks etc., \n 'ADJ': Adjectives are words that typically modify nouns and specify their properties or attributes, \n 'ADP': Adposition is a cover term for prepositions and postpositions, \n 'ADV': Adverbs are words that typically modify verbs for such categories as time, place, direction or manner, \n 'CONJ': A word used to connect clauses or sentences or to coordinate words in the same clause, \n 'DET': Determiners are words that modify nouns or noun phrases and express the reference of the noun phrase in context, \n 'NOUN': Nouns are a part of speech typically denoting a person, place, thing, animal or idea, \n 'NUM': A numeral is a word, functioning most typically as a determiner, adjective or pronoun, that expresses a number and a relation to the number, such as quantity, sequence, frequency or fraction, \n 'PRT': Particles are function words that must be associated with another word or phrase to impart meaning and that do not satisfy definitions of other universal parts of speech, \n 'PRON': Pronouns are words that substitute for nouns or noun phrases, whose meaning is recoverable from the linguistic or extralinguistic context, \n 'PROPN': A proper noun is a noun (or nominal content word) that is the name (or part of the name) of a specific individual, place, or object, \n 'VERB': A verb is a member of the syntactic class of words that typically signal events and actions, can constitute a minimal predicate in a clause, and govern the number and types of other constituents which may occur in the clause, \n 'X': The tag X is used for words that for some reason cannot be assigned a real part-of-speech category.\nSentence: `` I'm not worried about it '' , Phil said {{ . }} \nWord: .\nA:"
          }
        ],
        "chosen": {
          "from": "assistant",
          "value": "In this task, you are asked to identify the part-of-speech (POS) tag for a word in a given sentence. The word is specified within curly braces ({{...}}). The corpus follows a predefined tagset that indicates different parts-of-speech categories.\n\nTo complete this task:\n\n1. Carefully read the list of part-of-speech tags provided.\n2. Analyze the sentence given, paying attention to the word within the curly braces: \"{{ . }}\"\n3. Look through the tagset and determine which tag best fits the word within the curly braces.\n4. Provide the appropriate part-of-speech tag for the word.\n\nFor the given sentence: \"I'm not worried about it\", Phil said {{ . }}\nWord: .\n\nUsing the guidelines, we can see that the word '.' is a punctuation symbol. According to the tagset:\n'.': Period symbol is used for symbols denoting Punctuations/Separations such as comma, period, backticks, etc.\n\nSo, the part-of-speech tag for the word '.' is: '.'"
        },
        "rejected": {
          "from": "assistant",
          "value": "Sure, I'd be happy to help! Here are the steps to find the part of speech of the word in the given sentence:\n\nStep 1: Read the sentence and identify the word you want to find the part of speech for. In this case, the word is \"it\".\n\nStep 2: Look at the list of part-of-speech tags provided in the task. In this case, we have the following tags:\n\n'.': Period symbol (punctuation)\nADJ: Adjective\nADP: Adposition (preposition or postposition)\nADV: Adverb\nCONJ: Conjunction\nDET: Determiner\nNOUN: Noun\nNUM: Numeral\nPRT: Particle\nPRON: Pronoun\nPROPN: Proper noun\nVERB: Verb\nX: Unclassified word (for words that cannot be assigned a real part-of-speech category)\n\nStep 3: Look at the word \"it\" in the sentence and determine which category it best fits into. In this case, \"it\" is a pronoun, so the part of speech is PRON.\n\nTherefore, the part of speech of the word \"it\" in the sentence \"I'm not worried about it\", Phil said is PRON."
        }
      }
    """
    preference_pair_examples = list()
    for example in tqdm(examples):
        prompt = example["prompt"]
        chosen_response = example["chosen"][1]["content"]
        rejected_response = example["rejected"][1]["content"]
        chosen_score, rejected_score = example["score_chosen"], example["score_rejected"]
        # if chosen_score <= rejected_score:
        #     continue
        preference_pair_examples.append({
            "conversations": [{
                "from": "human",
                "value": prompt,
            }],
            "chosen": {
                "from": "assistant",
                "value": chosen_response,
            },
            "rejected": {
                "from": "assistant",
                "value": rejected_response,
            }
        })
        
    return preference_pair_examples
        
ultra_feedback_preference_pair_train_examples = preprocess_ultra_feedback(ultralfeedback_train_data)
ultra_feedback_preference_pair_test_examples = preprocess_ultra_feedback(ultralfeedback_test_data)

print("len(ultra_feedback_preference_pair_train_examples)=", len(ultra_feedback_preference_pair_train_examples))
print("len(ultra_feedback_preference_pair_test_examples)=", len(ultra_feedback_preference_pair_test_examples))


100%|██████████| 61135/61135 [00:00<00:00, 204095.76it/s]
100%|██████████| 2000/2000 [00:00<00:00, 421157.14it/s]

len(ultra_feedback_preference_pair_train_examples)= 61135
len(ultra_feedback_preference_pair_test_examples)= 2000





In [136]:
ultra_feedback_cls_train_examples[0]

{'instruction': 'You are a good preference labeler. Give you one query from human and two candidate responses answered by the same assistant, your job is to find the better response which is more helpful, harmless for the human query.\n\nQ: What role do investment bankers play in the financial industry, and how do they help businesses raise capital?\n\n\nResponse #1: Investment bankers act as intermediaries between financial institutions and companies looking to raise capital, and provide advice on mergers and acquisitions. They help businesses understand the debut process and provide them with access to markets. Investment bankers also help with identifying, evaluating, structuring, and executing a variety of financial transactions, such as initial public offerings, mergers & acquisitions, and leveraged buyouts. \n\n\nReponse #2: Investment bankers act as intermediaries between businesses and investors. They help businesses raise capital by structuring and selling securities such as s

In [197]:
def obtain_cls_from_preference_pair_ultra_feedback(examples):
    """
    Construct cls style for preference optimzation
    input: ultra_feedback_preference_pair_examples
    return :the structure of example:
    
      {
        "instruction": "You are a good preference labeler. Give you one query from human and two candidate responses answered by the same assistant, your job is to find the better response which is more helpful, harmless for the human query.\n\nQ: xxx\n\n\nResponse #1: \n\n\nReponse #2",
        "label": 0
      }
    
    
    """
    ultra_feedback_cls_examples = list()
    for example in tqdm(examples):
        accepted, rejected = example["chosen"]["value"], example["rejected"]["value"]
        query = example["conversations"][0]["value"]
        ultra_feedback_cls_examples.append({
            "instruction": f"You are a good preference labeler. Give you one query from human and two candidate responses answered by the same assistant, your job is to find the better response which is more helpful, harmless for the human query.\n\nQ: {query}\n\n\nResponse #1: {accepted} \n\n\nReponse #2: {rejected}",
            "label": 0
        })
        ultra_feedback_cls_examples.append({
            "instruction": f"You are a good preference labeler. Give you one query from human and two candidate responses answered by the same assistant, your job is to find the better response which is more helpful, harmless for the human query.\n\nQ: {query}\n\n\nResponse #1: {rejected} \n\n\nReponse #2: {accepted}",
            "label": 1
        })
    shuffle(ultra_feedback_cls_examples)
    return ultra_feedback_cls_examples
ultra_feedback_cls_train_examples = obtain_cls_from_preference_pair_ultra_feedback(ultra_feedback_preference_pair_train_examples)
ultra_feedback_cls_test_examples = obtain_cls_from_preference_pair_ultra_feedback(ultra_feedback_preference_pair_test_examples)

print("len(ultra_feedback_cls_train_examples)=", len(ultra_feedback_cls_train_examples))
print("len(ultra_feedback_cls_test_examples)=", len(ultra_feedback_cls_test_examples))


100%|██████████| 61135/61135 [00:00<00:00, 103724.79it/s]
100%|██████████| 2000/2000 [00:00<00:00, 108701.56it/s]

len(ultra_feedback_cls_train_examples)= 122270
len(ultra_feedback_cls_test_examples)= 4000





Note⚠️⚠️⚠️⚠️：**If you want to change the instruction prompt, please update the prompt in `src/alignment/inference/estimating/main.py` at the same time.**

In [22]:
shuffle(ultra_feedback_preference_pair_train_examples)
shuffle(ultra_feedback_cls_train_examples)

In [24]:
with open("./UltraFeedback/pair_train_data.json", "w", encoding="utf-8") as fw:
    for example in tqdm(ultra_feedback_preference_pair_train_examples):
        fw.write(json.dumps(example, ensure_ascii=False) + "\n")

with open("./UltraFeedback/pair_test_data.json", "w", encoding="utf-8") as fw:
    for example in tqdm(ultra_feedback_preference_pair_test_examples):
        fw.write(json.dumps(example, ensure_ascii=False) + "\n")

with open("./UltraFeedback/cls_train_data.json", "w", encoding="utf-8") as fw:
    for example in tqdm(ultra_feedback_cls_train_examples):
        fw.write(json.dumps(example, ensure_ascii=False) + "\n")

with open("./UltraFeedback/cls_test_data.json", "w", encoding="utf-8") as fw:
    for example in tqdm(ultra_feedback_cls_test_examples):
        fw.write(json.dumps(example, ensure_ascii=False) + "\n")

100%|██████████| 61135/61135 [00:03<00:00, 19944.30it/s]
100%|██████████| 2000/2000 [00:00<00:00, 21332.03it/s]


In [17]:
print(ultra_feedback_cls_train_examples[0]["instruction"])
print(ultra_feedback_cls_train_examples[0]["label"])

You are a good preference labeler. Give you one query from human and two candidate responses answered by the same assistant, your job is to find the better response which is more helpful, harmless for the human query.

Q: Describe the historical development of Python programming language, while highlighting the various design philosophies and guiding principles that influenced its creation, as well as discussing its impact on the wider software development landscape. Additionally, incorporate information on the contributions of key individuals and organizations to the language's evolution, including its ongoing updates and adaptations to contemporary technological advancements.


Response #1: Python was created in the late 1960s by Guido van Rossum. Its development was influenced by a range of design philosophies, including Minimalism, Explicit is better than Implicit, and Symmetry among others. Throughout its history, the language has undergone a series of updates and adaptations to k

In [23]:
ultra_feedback_preference_pair_train_examples[0]

{'conversations': [{'from': 'human',
   'value': 'A squirrel will run across what to get from one gutter to another?\nOptions:\n- park\n- forest\n- chipmunk\n- roof\n- muddy water\nPlease answer and provide answer explanation.'}],
 'chosen': {'from': 'assistant',
  'value': '[Question] A squirrel will run across what to get from one gutter to another?\n[Answer] A squirrel will run across a roof to get from one gutter to another.\n[Confidence]: 90%\n[Explanation] Squirrels are known for their agility and ability to navigate different surfaces. When moving between gutters, a squirrel is most likely to use the roof as a platform to traverse from one gutter to another. This is because the roof provides a continuous pathway that connects the two gutters, whereas the other options (park, forest, chipmunk, and muddy water) do not offer such a direct or feasible route for a squirrel to travel along.'},
 'rejected': {'from': 'assistant',
  'value': 'In order to cross from one gutter to another,

## 2. UltraChat200K(SFT)
path: https://hf-mirror.com/datasets/HuggingFaceH4/ultrachat_200k

This data is used for sft or prompt sampling during iteration preference optimzation

In [106]:
ultrachat_dir = "./UltraChat200K/data/"

In [107]:
def read_ultralchat_data(data_dir):
    df_train1 = pd.read_parquet(os.path.join(data_dir, "train_sft-00000-of-00003-a3ecf92756993583.parquet"))
    df_train2 = pd.read_parquet(os.path.join(data_dir, "train_sft-00001-of-00003-0a1804bcb6ae68c6.parquet"))
    df_train3 = pd.read_parquet(os.path.join(data_dir, "train_sft-00002-of-00003-ee46ed25cfae92c6.parquet"))
    df_train = json.loads(df_train1.to_json(orient='records')) + json.loads(df_train2.to_json(orient='records')) + json.loads(df_train3.to_json(orient='records'))
    return df_train
    
ultralchat_train_data = read_ultralchat_data(ultrachat_dir)

In [108]:
ultralchat_train_data[0]

{'prompt': "These instructions apply to section-based themes (Responsive 6.0+, Retina 4.0+, Parallax 3.0+ Turbo 2.0+, Mobilia 5.0+). What theme version am I using?\nOn your Collections pages & Featured Collections sections, you can easily show the secondary image of a product on hover by enabling one of the theme's built-in settings!\nYour Collection pages & Featured Collections sections will now display the secondary product image just by hovering over that product image thumbnail.\nDoes this feature apply to all sections of the theme or just specific ones as listed in the text material?",
 'prompt_id': 'f0e37e9f7800261167ce91143f98f511f768847236f133f2d0aed60b444ebe57',
 'messages': [{'content': "These instructions apply to section-based themes (Responsive 6.0+, Retina 4.0+, Parallax 3.0+ Turbo 2.0+, Mobilia 5.0+). What theme version am I using?\nOn your Collections pages & Featured Collections sections, you can easily show the secondary image of a product on hover by enabling one of 

In [109]:
def construct_ultrachat_examples(examples):
    """
    The ultrachat is multi-turn conversation. We only maintain the first turn.
    output:
    {
        "messages": [
            {"role": "user", "content": "xxx"},
        ],
        "reference": {"role": "assistant", "content": "xxx"},
        "from": "xxx",
    }
    """
    all_examples = list()
    for example in tqdm(examples):
        all_examples.append({
            "messages": [
                {"role": "user", "content": example["messages"][0]["content"]},
            ],
            "reference": {"role": "assistant", "content": example["messages"][1]["content"]},
            "from": "ultrachat200k"
        })
    return all_examples
ultrachat_sft_examples = construct_ultrachat_examples(ultrachat_train_data)


100%|██████████| 207865/207865 [00:00<00:00, 268920.71it/s]


In [110]:
ultrachat_sft_examples[0]

{'messages': [{'role': 'user',
   'content': "These instructions apply to section-based themes (Responsive 6.0+, Retina 4.0+, Parallax 3.0+ Turbo 2.0+, Mobilia 5.0+). What theme version am I using?\nOn your Collections pages & Featured Collections sections, you can easily show the secondary image of a product on hover by enabling one of the theme's built-in settings!\nYour Collection pages & Featured Collections sections will now display the secondary product image just by hovering over that product image thumbnail.\nDoes this feature apply to all sections of the theme or just specific ones as listed in the text material?"}],
 'reference': {'role': 'assistant',
  'content': 'This feature only applies to Collection pages and Featured Collections sections of the section-based themes listed in the text material.'},
 'from': 'ultrachat200k'}

## 3. EvolInstruct(SFT)
path: https://hf-mirror.com/datasets/WizardLMTeam/WizardLM_evol_instruct_V2_196k

This data is used for sft or prompt sampling during iteration preference optimzation

In [111]:
evolinstruct_dir = "./EvolInstruct"

In [112]:
def read_evolinstruct_data(data_dir):
    examples = list()
    with open(os.path.join(data_dir, "WizardLM_evol_instruct_V2_143k.json"), "r", encoding="utf-8") as fr:
        examples = json.load(fr)
    return examples
evolinstruct_train_data = read_evolinstruct_data(evolinstruct_dir)

In [113]:
def construct_evolinstruct_examples(examples):
    """
    output:
    {
        "messages": [
            {"role": "user", "content": "xxx"},
        ],
        "reference": {"role": "assistant", "content": "xxx"},,
        "from": "xxx"
    }
    """
    all_examples = list()
    for example in tqdm(examples):
        all_examples.append({
            "messages": [
                {"role": "user", "content": example["conversations"][0]["value"]},
            ],
            "reference": {"role": "assistant", "content": example["conversations"][1]["value"]},
            "from": "evolinstruct"
        })
    return all_examples
evolinstruct_sft_examples = construct_evolinstruct_examples(evolinstruct_train_data)


100%|██████████| 143000/143000 [00:00<00:00, 689594.07it/s]


In [114]:
evolinstruct_sft_examples[0]

{'messages': [{'role': 'user',
   'content': 'As an online platform teacher named Aimee, you possess impeccable credentials which include a Bachelor of Science degree in Industrial and Labor Relations from Cornell University, expertise in the English language, and intermediate proficiency in both Chinese and Spanish. Additionally, your professional experience as a STEAM teacher at UN Women in Singapore has honed your skills in teaching children from the ages of 6-11 and working with students from all levels of education. Your exceptional teaching abilities in spoken English and pronunciation paired with your personal strengths of being informed, patient, and engaging make you an ideal teacher for students seeking to improve their English language skills. Can you provide a short, concise, and unique English self-introduction in bullet point form that would attract students to enroll in your course?'}],
 'reference': {'role': 'assistant',
  'content': "Sure, here are some bullet points f

## 4. ShareGPT(SFT)
path: https://hf-mirror.com/datasets/shibing624/sharegpt_gpt4

This data is used for sft or prompt sampling during iteration preference optimzation

In [115]:
sharegpt_dir = "./ShareGPT/"

In [116]:
def read_sharegpt_data(data_dir):
    examples = list()
    with open(os.path.join(data_dir, "sharegpt_gpt4.jsonl"), "r", encoding="utf-8") as fr:
        for line in tqdm(fr.readlines()):
            examples.append(json.loads(line))
    with open(os.path.join(data_dir, "sharegpt_V3_format.jsonl"), "r", encoding="utf-8") as fr:
        for line in tqdm(fr.readlines()):
            examples.append(json.loads(line))
    # with open(os.path.join(data_dir, "sharegpt_zh_38K_format.jsonl"), "r", encoding="utf-8") as fr:
    #     for line in tqdm(fr.readlines()):
    #         examples.append(json.loads(line))
    return examples
sharegpt_train_data = read_sharegpt_data(sharegpt_dir)

100%|██████████| 6206/6206 [00:00<00:00, 9742.42it/s] 
100%|██████████| 58674/58674 [00:03<00:00, 18296.86it/s]


In [117]:
def construct_sharegpt_examples(examples):
    """
    output:
    {
        "messages": [
            {"role": "user", "content": "xxx"},
        ],
        "reference": {"role": "assistant", "content": "xxx"},,
        "from": "xxx"
    }
    """
    all_examples = list()
    for example in tqdm(examples):
        all_examples.append({
            "messages": [
                {"role": "user", "content": example["conversations"][0]["value"]},
            ],
            "reference": {"role": "assistant", "content": example["conversations"][1]["value"]},
            "from": "sharegpt"
        })
    return all_examples
sharegpt_sft_examples = construct_sharegpt_examples(sharegpt_train_data)


100%|██████████| 64880/64880 [00:00<00:00, 609447.97it/s]


In [119]:
## 汇总所有SFT数据，并进行过滤
def preprocess_sft_data(all_examples_list):
    all_sft_examples = list()
    num_before = 0
    for examples in all_examples_list:
        for example in tqdm(examples):
            num_before += 1
            # 如果assistant的↩️太短了，则去除
            if len(example["reference"]["content"]) < 50:
                continue
            all_sft_examples.append(example)
    print("num before process:", num_before)
    print("num after process:", len(all_sft_examples))
    return all_sft_examples

all_sft_examples = preprocess_sft_data([
    ultrachat_sft_examples, evolinstruct_sft_examples, sharegpt_sft_examples
])

100%|██████████| 207865/207865 [00:00<00:00, 597987.48it/s]
100%|██████████| 143000/143000 [00:00<00:00, 1333680.52it/s]
100%|██████████| 64880/64880 [00:00<00:00, 1489259.57it/s]


num before process: 415745
num after process: 408954


In [120]:
all_sft_examples[0]

{'messages': [{'role': 'user',
   'content': "These instructions apply to section-based themes (Responsive 6.0+, Retina 4.0+, Parallax 3.0+ Turbo 2.0+, Mobilia 5.0+). What theme version am I using?\nOn your Collections pages & Featured Collections sections, you can easily show the secondary image of a product on hover by enabling one of the theme's built-in settings!\nYour Collection pages & Featured Collections sections will now display the secondary product image just by hovering over that product image thumbnail.\nDoes this feature apply to all sections of the theme or just specific ones as listed in the text material?"}],
 'reference': {'role': 'assistant',
  'content': 'This feature only applies to Collection pages and Featured Collections sections of the section-based themes listed in the text material.'},
 'from': 'ultrachat200k'}

**save all sft examples as sampling prompt set during iteration preference optimzation**

In [121]:
if not os.path.exists("./prompt_set"):
    os.makedirs("./prompt_set")
with open("./prompt_set/all_sft_prompt_data.json", "w", encoding="utf-8") as fw:
    for example in tqdm(all_sft_examples):
        fw.write(json.dumps(example, ensure_ascii=False) + "\n")

100%|██████████| 408954/408954 [00:50<00:00, 8086.34it/s] 


## Sampling for each iteration stage
Next, we can directly sample for some prompts for each iteration stage.
- the iteration num is fixed as 3;
- each iteration, we random sample 25k from sft examples and sample 10k from ultrafeedback prompt-chosen data, which similar to SPIN.

In [122]:
# obtain sft examples from ultrafeedback
ultrafeedback_sft_examples = list()
for example in tqdm(ultra_feedback_preference_pair_train_examples):
    ultrafeedback_sft_examples.append({
        "messages": [
            {"role": "user", "content": example["conversations"][0]["value"]},
        ],
        "reference": {"role": "assistant", "content": example["chosen"]["value"]},
        "from": "ultrafeedback"
    })

100%|██████████| 61135/61135 [00:00<00:00, 264482.04it/s]


In [123]:
ultrafeedback_sft_examples[0]

{'messages': [{'role': 'user',
   'content': 'A squirrel will run across what to get from one gutter to another?\nOptions:\n- park\n- forest\n- chipmunk\n- roof\n- muddy water\nPlease answer and provide answer explanation.'}],
 'reference': {'role': 'assistant',
  'content': '[Question] A squirrel will run across what to get from one gutter to another?\n[Answer] A squirrel will run across a roof to get from one gutter to another.\n[Confidence]: 90%\n[Explanation] Squirrels are known for their agility and ability to navigate different surfaces. When moving between gutters, a squirrel is most likely to use the roof as a platform to traverse from one gutter to another. This is because the roof provides a continuous pathway that connects the two gutters, whereas the other options (park, forest, chipmunk, and muddy water) do not offer such a direct or feasible route for a squirrel to travel along.'},
 'from': 'ultrafeedback'}

In [124]:
iteration_stage_num = 3
sampling_sft_num = 25000
sampling_preference_num = 10000

if not os.path.exists("./prompt_set"):
    os.makedirs("./prompt_set")

for stage_id in range(iteration_stage_num):
    # random sample sft
    shuffle(all_sft_examples)
    selected_sft_examples = all_sft_examples[:sampling_sft_num]
    # random sample ultrafeedback
    shuffle(ultrafeedback_sft_examples)
    selected_sft_examples = selected_sft_examples + ultrafeedback_sft_examples[:sampling_preference_num]
    shuffle(selected_sft_examples)
    with open(f"./prompt_set/sampled_prompt_data_for_iter{stage_id + 1}.json", "w", encoding="utf-8") as fw:
        for example in tqdm(selected_sft_examples):
            fw.write(json.dumps(example, ensure_ascii=False) + "\n")

100%|██████████| 35000/35000 [00:02<00:00, 12252.94it/s]
100%|██████████| 35000/35000 [00:03<00:00, 10789.77it/s]
100%|██████████| 35000/35000 [00:05<00:00, 6697.22it/s]


# Chap2: Math Reasoning

## 1. Math Step-DPO

In [189]:
math_step_dpo_path = "./MathStepDPO10k/data/"

In [190]:
def read_mathstepdpo_data(data_dir):
    df_train = pd.read_parquet(os.path.join(data_dir, "train-00000-of-00001.parquet"))
    df_train = json.loads(df_train.to_json(orient='records'))
    return df_train
    
mathstepdpo_train_data = read_mathstepdpo_data(math_step_dpo_path)

In [191]:
mathstepdpo_train_data[0]

{'full_rejected': ' Find the dates when the shoe store has a sale.\nThe shoe store has a sale every 6 days. If they have a sale on July 3, then they will have a sale on July 3 + 6 = 9, 15, 21, 27, 33, 39, 45, 51, 57.\n\nStep 2: Find the dates when the bookstore has a sale.\nThe bookstore has a sale on days that are multiples of 5. So the dates when the bookstore has a sale are 5, 10, 15, 20, 25, 30, 35, 40, 45, 50, 55, 60.\n\nStep 3: Find the dates when both stores have sales on the same date.\nWe need to find the dates that are in both lists. Those dates are 15 and 45.\n\nSo, both stores have sales on the same date 2 times in July.\nThe answer is: 2',
 'initial_reason_steps': "Let's think step by step. \nStep 1:",
 'prompt': 'In the month of July, the bookstore has a sale on days that are multiples of 5, and the shoe store has a sale every 6 days. If the shoe store has a sale on July 3, how many times in July do both stores have sales on the same date?',
 'chosen': ' Find the dates on

In [229]:
def preprocess_math_stepdpo(examples):
    """
    Construct pair-wise data for preference optimzation
    return :the structure of example:
    
      {
        "conversations": [
          {
            "from": "system",
            "value": "You are a teacher. Given a task, you explain in simple steps what the task is asking, any guidelines it provides and how to use those guidelines to find the answer."
          },
          {
            "from": "human",
            "value": "Q: In this task, you need to provide the parts-of-speech tag of a word present in a sentence specified within curly braces  ( '{{ ... }}' ). The parts-of-speech tags are coarse labels that represent a category of words with similar grammatical properties. The list of part-of-speech tags i.e tagset of this corpus is - \n '.': Period symbol is used for symbols denoting Punctuations/Separations such as comma, period, backticks etc., \n 'ADJ': Adjectives are words that typically modify nouns and specify their properties or attributes, \n 'ADP': Adposition is a cover term for prepositions and postpositions, \n 'ADV': Adverbs are words that typically modify verbs for such categories as time, place, direction or manner, \n 'CONJ': A word used to connect clauses or sentences or to coordinate words in the same clause, \n 'DET': Determiners are words that modify nouns or noun phrases and express the reference of the noun phrase in context, \n 'NOUN': Nouns are a part of speech typically denoting a person, place, thing, animal or idea, \n 'NUM': A numeral is a word, functioning most typically as a determiner, adjective or pronoun, that expresses a number and a relation to the number, such as quantity, sequence, frequency or fraction, \n 'PRT': Particles are function words that must be associated with another word or phrase to impart meaning and that do not satisfy definitions of other universal parts of speech, \n 'PRON': Pronouns are words that substitute for nouns or noun phrases, whose meaning is recoverable from the linguistic or extralinguistic context, \n 'PROPN': A proper noun is a noun (or nominal content word) that is the name (or part of the name) of a specific individual, place, or object, \n 'VERB': A verb is a member of the syntactic class of words that typically signal events and actions, can constitute a minimal predicate in a clause, and govern the number and types of other constituents which may occur in the clause, \n 'X': The tag X is used for words that for some reason cannot be assigned a real part-of-speech category.\nSentence: `` I'm not worried about it '' , Phil said {{ . }} \nWord: .\nA:"
          }
        ],
        "chosen": {
          "from": "assistant",
          "value": "In this task, you are asked to identify the part-of-speech (POS) tag for a word in a given sentence. The word is specified within curly braces ({{...}}). The corpus follows a predefined tagset that indicates different parts-of-speech categories.\n\nTo complete this task:\n\n1. Carefully read the list of part-of-speech tags provided.\n2. Analyze the sentence given, paying attention to the word within the curly braces: \"{{ . }}\"\n3. Look through the tagset and determine which tag best fits the word within the curly braces.\n4. Provide the appropriate part-of-speech tag for the word.\n\nFor the given sentence: \"I'm not worried about it\", Phil said {{ . }}\nWord: .\n\nUsing the guidelines, we can see that the word '.' is a punctuation symbol. According to the tagset:\n'.': Period symbol is used for symbols denoting Punctuations/Separations such as comma, period, backticks, etc.\n\nSo, the part-of-speech tag for the word '.' is: '.'"
        },
        "rejected": {
          "from": "assistant",
          "value": "Sure, I'd be happy to help! Here are the steps to find the part of speech of the word in the given sentence:\n\nStep 1: Read the sentence and identify the word you want to find the part of speech for. In this case, the word is \"it\".\n\nStep 2: Look at the list of part-of-speech tags provided in the task. In this case, we have the following tags:\n\n'.': Period symbol (punctuation)\nADJ: Adjective\nADP: Adposition (preposition or postposition)\nADV: Adverb\nCONJ: Conjunction\nDET: Determiner\nNOUN: Noun\nNUM: Numeral\nPRT: Particle\nPRON: Pronoun\nPROPN: Proper noun\nVERB: Verb\nX: Unclassified word (for words that cannot be assigned a real part-of-speech category)\n\nStep 3: Look at the word \"it\" in the sentence and determine which category it best fits into. In this case, \"it\" is a pronoun, so the part of speech is PRON.\n\nTherefore, the part of speech of the word \"it\" in the sentence \"I'm not worried about it\", Phil said is PRON."
        }
      }
    """
    preference_pair_examples = list()
    for example in tqdm(examples):
        prompt = example["prompt"]
        initial_reason_steps = example["initial_reason_steps"]
        chosen_response = initial_reason_steps + example["full_chosen"]
        rejected_response = initial_reason_steps + example["full_rejected"]
        preference_pair_examples.append({
            "conversations": [{
                "from": "human",
                "value": prompt + "\nPlease reason step by step, and put your final answer within \\boxed{{}}",
            }],
            "chosen": {
                "from": "assistant",
                "value": chosen_response,
            },
            "rejected": {
                "from": "assistant",
                "value": rejected_response,
            }
        })
        
    return preference_pair_examples
        
mathstepdpo_preference_pair_train_examples = preprocess_math_stepdpo(mathstepdpo_train_data)


print("len(mathstepdpo_preference_pair_train_examples)=", len(mathstepdpo_preference_pair_train_examples))


100%|██████████| 10795/10795 [00:00<00:00, 127429.93it/s]

len(mathstepdpo_preference_pair_train_examples)= 10795





In [230]:
mathstepdpo_preference_pair_train_examples[0]

{'conversations': [{'from': 'human',
   'value': 'In the month of July, the bookstore has a sale on days that are multiples of 5, and the shoe store has a sale every 6 days. If the shoe store has a sale on July 3, how many times in July do both stores have sales on the same date?\nPlease reason step by step, and put your final answer within \\boxed{{}}'}],
 'chosen': {'from': 'assistant',
  'value': "Let's think step by step. \nStep 1: Find the dates on which both stores have sales.\nThe bookstore has a sale on days that are multiples of 5. So the dates are 5, 10, 15, 20, 25, 30 (for July).\nThe shoe store has a sale every 6 days. If it has a sale on July 3, then the next sale will be on July 9, then on July 15, and so on. So the dates of its sales are 3, 9, 15, 21, 27.\nStep 2: Find the dates when both stores have sales on the same date.\nBy comparing the two lists of dates, we see that both stores have sales on the same date on July 15.\nSo, both stores have sales on the same date on

In [231]:
def obtain_cls_from_preference_pair_mathstepdpo(examples):
    """
    Construct cls style for preference optimzation
    input: mathstepdpo_preference_pair_train_examples
    return :the structure of example:
    
      {
        "instruction": "You are a good preference labeler. Give you one query from human and two candidate responses answered by the same assistant, your job is to find the better response which is more accurate, helpful, harmless for the human query.\n\nQ: xxx\n\n\nResponse #1: \n\n\nReponse #2",
        "label": 0
      }
    
    
    """
    mathstepdpo_cls_examples = list()
    for example in tqdm(examples):
        accepted, rejected = example["chosen"]["value"], example["rejected"]["value"]
        query = example["conversations"][0]["value"]
        mathstepdpo_cls_examples.append({
            "instruction": f"You are a good preference labeler. Give you one query from human and two candidate responses answered by the same assistant, your job is to find the better response which is more accurate, helpful, harmless for the human query.\n\nQ: {query}\n\n\nResponse #1: {accepted} \n\n\nReponse #2: {rejected}",
            "label": 0
        })
        mathstepdpo_cls_examples.append({
            "instruction": f"You are a good preference labeler. Give you one query from human and two candidate responses answered by the same assistant, your job is to find the better response which is more accurate, helpful, harmless for the human query.\n\nQ: {query}\n\n\nResponse #1: {rejected} \n\n\nReponse #2: {accepted}",
            "label": 1
        })
    shuffle(mathstepdpo_cls_examples)
    return mathstepdpo_cls_examples
mathstepdpo_cls_train_examples = obtain_cls_from_preference_pair_mathstepdpo(mathstepdpo_preference_pair_train_examples)

print("len(mathstepdpo_cls_train_examples)=", len(mathstepdpo_cls_train_examples))


100%|██████████| 10795/10795 [00:00<00:00, 176170.92it/s]

len(mathstepdpo_cls_train_examples)= 21590





In [232]:
print(mathstepdpo_cls_train_examples[140]["instruction"])
print(mathstepdpo_cls_train_examples[140]["label"])

You are a good preference labeler. Give you one query from human and two candidate responses answered by the same assistant, your job is to find the better response which is more accurate, helpful, harmless for the human query.

Q: The product of the squares of two positive integers is 900. How many pairs of positive integers satisfy this condition?
Please reason step by step, and put your final answer within \boxed{{}}


Response #1: Let's think step by step.
Step 1: 900 can be factored into prime factors. The prime factorization of 900 is $2^2 \cdot 3^2 \cdot 5^2$.

Step 2: The product of the squares of two positive integers can be written as $(a^2)(b^2) = (ab)^2$. So, the positive integer whose square is 900 must be the square root of 900, which is $\sqrt{900} = 30$.

Step 3: Now we need to find the pairs of positive integers $(a, b)$ such that $ab = 30$.

The pairs of positive integers that multiply to give 30 are (1, 30), (2, 15), (3, 10), (5, 6). So, there are 4 pairs of positive

In [233]:
with open("./MathStepDPO10k/pair_train_data.json", "w", encoding="utf-8") as fw:
    for example in tqdm(mathstepdpo_preference_pair_train_examples):
        fw.write(json.dumps(example, ensure_ascii=False) + "\n")

with open("./MathStepDPO10k/cls_train_data.json", "w", encoding="utf-8") as fw:
    for example in tqdm(mathstepdpo_cls_train_examples):
        fw.write(json.dumps(example, ensure_ascii=False) + "\n")


100%|██████████| 10795/10795 [00:01<00:00, 8098.42it/s]
100%|██████████| 21590/21590 [00:01<00:00, 17935.28it/s]


In [212]:
mathstepdpo_preference_pair_train_examples[0]

{'conversations': [{'from': 'human',
   'value': 'In the month of July, the bookstore has a sale on days that are multiples of 5, and the shoe store has a sale every 6 days. If the shoe store has a sale on July 3, how many times in July do both stores have sales on the same date?'}],
 'chosen': {'from': 'assistant',
  'value': "Let's think step by step. \nStep 1: Find the dates on which both stores have sales.\nThe bookstore has a sale on days that are multiples of 5. So the dates are 5, 10, 15, 20, 25, 30 (for July).\nThe shoe store has a sale every 6 days. If it has a sale on July 3, then the next sale will be on July 9, then on July 15, and so on. So the dates of its sales are 3, 9, 15, 21, 27.\nStep 2: Find the dates when both stores have sales on the same date.\nBy comparing the two lists of dates, we see that both stores have sales on the same date on July 15.\nSo, both stores have sales on the same date only once in July.\nThe answer is: 1"},
 'rejected': {'from': 'assistant',
 

## 2. MathInstruct (SFT))
used for promopt sampling. url: https://huggingface.co/datasets/TIGER-Lab/MathInstruct

In [208]:
mathinstruct_dir = "./MathInstruct/MathInstruct.json"

In [210]:
def load_mathinstruct_data(data_path):
    examples = list()
    with open(data_path, "r", encoding="utf-8") as fr:
        # for line in tqdm(fr.readlines()):
        #     examples.append(json.loads(line))
        examples = json.load(fr)
    return examples

mathinstruct_examples = load_mathinstruct_data(mathinstruct_dir)

In [248]:
def construct_mathinstruct_examples(examples):
    """
    output:
    {
        "messages": [
            {"role": "user", "content": "xxx"},
        ],
        "reference": {"role": "assistant", "content": "xxx"},,
        "from": "xxx"
    }
    """
    all_examples = list()
    for example in tqdm(examples):
        all_examples.append({
            "messages": [
                # {"role": "user", "content": example["instruction"] + "\nPlease reason step by step, and put your final answer within \\boxed{{}}"},
                {"role": "user", "content": example["instruction"]},
            ],
            "reference": {"role": "assistant", "content": example["output"]},
            "from": "mathinstruct"
        })
    return all_examples
mathinstruct_sft_examples = construct_mathinstruct_examples(mathinstruct_examples)


100%|██████████| 262039/262039 [00:03<00:00, 85134.44it/s] 


In [249]:
mathstepdpo_train_data[0]

{'full_rejected': ' Find the dates when the shoe store has a sale.\nThe shoe store has a sale every 6 days. If they have a sale on July 3, then they will have a sale on July 3 + 6 = 9, 15, 21, 27, 33, 39, 45, 51, 57.\n\nStep 2: Find the dates when the bookstore has a sale.\nThe bookstore has a sale on days that are multiples of 5. So the dates when the bookstore has a sale are 5, 10, 15, 20, 25, 30, 35, 40, 45, 50, 55, 60.\n\nStep 3: Find the dates when both stores have sales on the same date.\nWe need to find the dates that are in both lists. Those dates are 15 and 45.\n\nSo, both stores have sales on the same date 2 times in July.\nThe answer is: 2',
 'initial_reason_steps': "Let's think step by step. \nStep 1:",
 'prompt': 'In the month of July, the bookstore has a sale on days that are multiples of 5, and the shoe store has a sale every 6 days. If the shoe store has a sale on July 3, how many times in July do both stores have sales on the same date?',
 'chosen': ' Find the dates on

In [250]:
# sample some examples from preference data
math_stepdpo_sft_examples = list()
for example in tqdm(mathstepdpo_train_data):
    prompt = example["prompt"]
    initial_reason_steps = example["initial_reason_steps"]
    chosen_response = initial_reason_steps + example["full_chosen"]
    math_stepdpo_sft_examples.append({
        "messages": [{
            "role": "user", "content": prompt
        }],
        "reference": {
            "role": "assistant", "content": chosen_response,
        },
        "from": "math-stepdpo"
    })
mathinstruct_sft_examples = mathinstruct_sft_examples + math_stepdpo_sft_examples

100%|██████████| 10795/10795 [00:00<00:00, 224367.13it/s]


In [251]:
print(len(mathinstruct_sft_examples))

272834


In [252]:
shuffle(mathinstruct_sft_examples)

In [253]:
mathinstruct_sft_examples[0]

{'messages': [{'role': 'user',
   'content': "In a three-on-three basketball tournament hosted by Ms. Hamilton's school, every losing team gets knocked out immediately. Given that sixteen teams are participating, can you calculate the total number of games that will be played until a winner is declared?"}],
 'reference': {'role': 'assistant',
  'content': "Let's think step by step. Step 1: In a knockout tournament, each game eliminates one team. So to get rid of 15 teams (since there are 16 teams in total, and one team is the winner), we need to play 15 games.\nSo the answer is $\\boxed{15}$.\nThe answer is: 15"},
 'from': 'math-stepdpo'}

In [254]:
if not os.path.exists("./prompt_set"):
    os.makedirs("./prompt_set")
with open("./prompt_set/all_math_sft_prompt_data.json", "w", encoding="utf-8") as fw:
    for example in tqdm(mathinstruct_sft_examples):
        fw.write(json.dumps(example, ensure_ascii=False) + "\n")
with open("./prompt_set/math_stepdpo_sft_prompt_data.json", "w", encoding="utf-8") as fw:
    for example in tqdm(math_stepdpo_sft_examples):
        fw.write(json.dumps(example, ensure_ascii=False) + "\n")

100%|██████████| 272834/272834 [00:07<00:00, 35588.10it/s]
100%|██████████| 10795/10795 [00:00<00:00, 21193.38it/s]


In [255]:
iteration_stage_num = 3
sampling_sft_num = 12000

if not os.path.exists("./prompt_set"):
    os.makedirs("./prompt_set")

for stage_id in range(iteration_stage_num):
    # random sample sft
    shuffle(mathinstruct_sft_examples)
    selected_sft_examples = mathinstruct_sft_examples[:sampling_sft_num]
    with open(f"./prompt_set/sampled_math_prompt_data_for_iter{stage_id + 1}.json", "w", encoding="utf-8") as fw:
        for example in tqdm(selected_sft_examples):
            fw.write(json.dumps(example, ensure_ascii=False) + "\n")

100%|██████████| 12000/12000 [00:00<00:00, 20720.16it/s]
100%|██████████| 12000/12000 [00:00<00:00, 31110.15it/s]
100%|██████████| 12000/12000 [00:00<00:00, 35791.24it/s]


## 以下为草稿，请忽略

In [132]:

ultrafeedback_test_examples = list()
ultra_feedback_preference_pair_test_examples[0]
for example in tqdm(ultra_feedback_preference_pair_test_examples):
    ultrafeedback_test_examples.append({
        "messages": [
            {"role": "user", "content": example["conversations"][0]["value"]}
        ],
        "reference": {"role": "assistant", "content": example["chosen"]["value"]},
        "from": "ultrafeedback_test",
        "responses": [
            {"role": "assistant", "content": example["rejected"]["value"]}
        ]
    })

100%|██████████| 2000/2000 [00:00<00:00, 331696.64it/s]


In [133]:
with open(f"./prompt_set/ultrafeedback_test_prompt_responses.json", "w", encoding="utf-8") as fw:
    for example in tqdm(ultrafeedback_test_examples):
        fw.write(json.dumps(example, ensure_ascii=False) + "\n")

100%|██████████| 2000/2000 [00:00<00:00, 10247.28it/s]


In [130]:
ultrafeedback_test_examples[0]

{'messages': [{'rule': 'user',
   'content': 'In this task, you are given a second sentence. Your task is to generate the first sentence on the same topic but incoherent and inconsistent with the second sentence.\n\nQ: Additionally , some groups may contain other specialists , such as a heavy weapons or language expert .\n\nA: Each squad member is specially trained as a weapons expert , medic , combat engineer or communications expert , respectively .\n****\nQ: However , the General Accounting Office identified 125 countries that received U.S. training and assistance for their police forces during fiscal year 1990 at a cost of at least $117 million .\n\nA: No government agency is in charge of calculating the cost .\n****\nQ: But his frozen body was found in the ice in Charlotte ( Rochester ) early the next spring by Silas Hudson .\n\nA:'}],
 'reference': {'rule': 'assistant',
  'content': 'Could you provide some context or information about what you are looking for or any particular qu

In [149]:
import numpy as np
a = np.array([[1, 2, 3], [4, 5, 6], [7, 8, 9], [10, 11, 12]])
b = np.array([1, 2, 3, 4])

In [171]:
598*0.017

10.166

In [162]:
p_norm = [0.1, 0.2, 0.3, 0.1, 0.2, 0.3, 0.1, 0.2, 0.3, 0.5]
p_norm = p_norm / np.sum(p_norm)
np.random.choice(8, 3, p=p_norm, replace=False)





ValueError: 'a' and 'p' must have same size

In [174]:
a = np.array([
    [[1, 2], [3, 4], [5, 6], [7, 8], [9, 10]],
    [[11, 12], [13, 14], [15, 16], [17, 18], [19, 20]],
    [[21, 22], [23, 24], [25, 26], [27, 28], [29, 30]],
])
a.shape

(3, 5, 2)

In [187]:
random.choices([5, 2, 3, 4])[0]

5

In [176]:
index = [2, 0, 3]
a[:, index]

array([[[ 5,  6],
        [ 1,  2],
        [ 7,  8]],

       [[15, 16],
        [11, 12],
        [17, 18]],

       [[25, 26],
        [21, 22],
        [27, 28]]])

In [258]:
print("I am unable to provide visual representation of puzzles or solutions. However, I can provide you with steps on how to solve this sudoku puzzle using logic and deduction. Here are the steps that may help in solving the puzzle:\n\n1. Find numbers in rows and columns.\n   - look for numbers in each row and column. Try placing these numbers in any empty cell as it helps to eliminate possibilities.\n\n2. Eliminate possibility based on surrounding numbers.\n   - use elimination technique by looking at surrounding numbers to eliminate possible numbers from empty cells. For example, if there is only one possible number that could be placed next to an existing number then all other numbers will need to be eliminated.\n\n3. Look for patterns in each box.\n   - try filling boxes completely by noticing patterns within them. Look for repeated numbers in adjacent blocks or columns. Use this pattern to fill up empty cells.\n\n4. Consider numbers in boxes carefully.\n   - carefully consider numbers in the 3x3 sub-boxes and cross out numbers which do not fit into the particular sub-box. This process will make finding solutions easier.\n\nby following these steps consistently, you'll soon find your solution to the sudoq puzzle. Good luck!")

I am unable to provide visual representation of puzzles or solutions. However, I can provide you with steps on how to solve this sudoku puzzle using logic and deduction. Here are the steps that may help in solving the puzzle:

1. Find numbers in rows and columns.
   - look for numbers in each row and column. Try placing these numbers in any empty cell as it helps to eliminate possibilities.

2. Eliminate possibility based on surrounding numbers.
   - use elimination technique by looking at surrounding numbers to eliminate possible numbers from empty cells. For example, if there is only one possible number that could be placed next to an existing number then all other numbers will need to be eliminated.

3. Look for patterns in each box.
   - try filling boxes completely by noticing patterns within them. Look for repeated numbers in adjacent blocks or columns. Use this pattern to fill up empty cells.

4. Consider numbers in boxes carefully.
   - carefully consider numbers in the 3x3 sub

In [259]:
print("I am not capable of solving puzzles visually like humans do but I can provide you with an algorithm to solve it. Here's how you can solve this sudoku puzzle step by step using backtracking method:\n\nstep 1: write down all numbers from 1-9 in order to make sure that every number will be used at least once and there won’t be any duplicates. Step 2: fill in as many empty cells as possible based on logic deduction. For example, if we know that row or column contains only one occurrence of some particular digit then we can eliminate that digit in other rows or columns which have already been filled with same digits. Step 3: try different combinations for empty spaces until you find a solution. This process involves trying each combination until either we get a valid solution or exhaust all possibilities. Once you have exhausted all possibilities without finding a solution, go back to the previous moves where a wrong choice was made and undo them. Try again till a correct solution is found. Step 4: check your answer against the original puzzle. If they match then congratulations! you solved the puzzle correctly. However, if they don’t match then try again from scratch or use another method of solving such puzzles. Remember, patience and persistence are key while solving these types of puzzles. Good luck!")

I am not capable of solving puzzles visually like humans do but I can provide you with an algorithm to solve it. Here's how you can solve this sudoku puzzle step by step using backtracking method:

step 1: write down all numbers from 1-9 in order to make sure that every number will be used at least once and there won’t be any duplicates. Step 2: fill in as many empty cells as possible based on logic deduction. For example, if we know that row or column contains only one occurrence of some particular digit then we can eliminate that digit in other rows or columns which have already been filled with same digits. Step 3: try different combinations for empty spaces until you find a solution. This process involves trying each combination until either we get a valid solution or exhaust all possibilities. Once you have exhausted all possibilities without finding a solution, go back to the previous moves where a wrong choice was made and undo them. Try again till a correct solution is found. S