In [105]:
# Packages used throughout the process
import json
from openai import OpenAI
import os
import re

The entire process of distributed traversal loop:
1, Import the required datasets: morphological matrix information (in JSON format) and the API key.
2, Call the API for the first time to select an appropriate classification character for all species in the dataset, and use the results (dividing species into several groups based on different states of the initial character) as the initial classification result.
3, Save the initial classification result and parse it into a dictionary format for storage.
4, Traverse the contents of each group sequentially, selecting classification characters for the species in each group, and record the classification results for storage in the environment.
5, Integrate the results of multiple distributed traversals, refine the format requirements, and ultimately obtain a complete classification key table, choosing appropriate methods and formats for optimization during the integration process.

1, import morphological matrix dataset information (in JSON format) and call the API key.

In [81]:
# Call the API_KEY from the environment
client = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))

# First, import the required dataset
with open("D:/桌面/taxonomy_primary_result/The_GPT-4_result/Dataset_12 (Translating Niphargus barcodes from Switzerland into taxonomy with a description of two new species (Amphipoda, Niphargidae) ) 20/information gain methods/matrix_knowledge_graph.json", "r", encoding="utf-8") as file:
    matrix_data = json.load(file)

# Since the input is in JSON format, convert this JSON file to a string format when calling the API
matrix_data_str = json.dumps(matrix_data, ensure_ascii=False)
print(matrix_data_str)

{"aquilex'": {"Characteristics": {"Character1": "1", "Character2": "2", "Character3": "1", "Character4": "1 and 2", "Character5": "1", "Character6": "1", "Character7": "1", "Character8": "1", "Character9": "2", "Character10": "2", "Character11": "1", "Character12": "1", "Character13": "2", "Character14": "2", "Character15": "1", "Character16": "2", "Character17": "2", "Character18": "1"}}, "auerbachi'": {"Characteristics": {"Character1": "1", "Character2": "2", "Character3": "1", "Character4": "1", "Character5": "2", "Character6": "2", "Character7": "2", "Character8": "1", "Character9": "2", "Character10": "2", "Character11": "2", "Character12": "1", "Character13": "2", "Character14": "2", "Character15": "1", "Character16": "1", "Character17": "3", "Character18": "3"}}, "brixianus'": {"Characteristics": {"Character1": "2", "Character2": "1 and 2", "Character3": "2", "Character4": "2", "Character5": "2", "Character6": "1 and 2", "Character7": "2", "Character8": "1", "Character9": "3", "

2, Call the API to select the initial classification character.
Select the initial character for all species in the dataset based on the information gain method, and obtain different groupings based on the different states of the initial character.

In [82]:
# This is the prompt for the API input, using the client.chat.completions.create interface to conduct multi-turn conversations.
# By assigning different roles in the conversation (user, assistant, system), all input information is fully conveyed to the API model

messages = [
            # Set the system role to focus on taxonomy tasks.
            # Emphasize the system's understanding of morphological matrices, information gain, and the construction of classification keys in the system setup.
            {"role":"system",
             "content":
                 "You are a helpful taxonomist assistant."+ \
                 "You are skilled at calculating the correct information gain to choose the character that best divides species into even groups based on their states." +\
                 "Based on the selected character, classify the species into different groups according to their states." + \
                 "For each group with more than two species, continue selecting characters to further classify this group until each group only has one species." + \
                 "After multiple classifications, determine the final classification levels and record each classifying character and its state. " + \
                 "Finally, generate a taxonomic key."},
            # Input the main task request, construct the classification key, and emphasize the details needed for classification.
            # This includes: the main objective, information gain calculation, ignoring invalid states, clarifying multi-character states, guiding correct selection based on the significance of information gain, and standardizing the final output format for subsequent extraction of API results.
            {"role":"system",
            "content": """
                Generate the taxonomic key based on the provided morphological matrix. The matrix includes all species and their different states for each character.\n
                The process involves selecting a character to classify the species into groups. Repeat this classification within each subgroup until each group contains only one species.\n
                Information gain measures how much the uncertainty in the dataset is reduced after using a character for classification. It helps in selecting characters that minimize the entropy of the subset after classification, leading to better classification results.\n
                Please select the initial classification character for all species based on the morphological matrix and information gain methods.\n
                In the morphological matrix, 'Missing' and 'Not applicable' are invalid states. If a character has invalid states for the group being classified, it should be ignored.\n
                States are represented by numbers, and like the '1 and 2' means multiple states should be treated as a single state type.(such as '3' and '2 and 3'  is the different state, these are two separate states, when i choose character to based on different state to distinguish the species) \n
                you need to calculate the information gain for each character, and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification.\n
                Now I will show you the morphological matrix. Please provide only the initial classification character and the categorization of species based on its state, and you should singly show this called as # initial character classify result # 
            """},
            # Use the assistant to summarize and refine the prompt content for the API.
            # Through the conversation with the assistant, deepen the API understanding of the content to some extent, control the API response results, and standardize the output format of the API response.
            {"role": "assistant",
            "content": """
                Understood. I will generate the taxonomic key based on the provided morphological matrix. Here is a summary of the steps I will follow:\n
                1. The matrix includes all species and their different states for each character.\n
                2. I will select a character to classify the species into groups and repeat this classification within each subgroup until each group contains only one species.\n
                3. I will use information gain to measure how much the uncertainty in the dataset is reduced after using a feature for classification. This helps in selecting features that minimize the entropy of the subset after classification, leading to better classification results.\n
                4. I will select the initial classification character for all species based on the morphological matrix and information gain methods.\n
                5. In the morphological matrix, 'Missing' and 'Not applicable' are considered invalid states. If a character has invalid states for the group being classified, it will be ignored.\n
                6. States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type. (such as '1' and '1 and 2'  is the different state, these are two separate states, when i choose character to based on different state to distinguish the species)\n
                7. I will use information gain to calculate all character and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification.so i need to make sure the result is Average classification \n
                8. The final result will provide only the initial classification character and the categorization of species based on its state. \n
                9. Don't need to show how i calculate, only need to show the final result, and please show the final result in #initail character classify result# block, Don't have errors where the state and species don't match
                
                Please provide the morphological matrix data so that I can proceed with the initial classification.
            """
            },
            # Input the corresponding morphological matrix information for analysis.
            {"role":"user",
             "content":
                 f"Here is morphological matrix:{matrix_data_str}"}
            ]

# Set the various parameters that need to be controlled when responding to the API. I think the temperature should be set to 0, and the max_tokens should either not be set or set to a small number. This is to save costs and because generating the corresponding results does not require a long output. If the token number is too large and no stop sequence is set, it can lead to complex redundancy and errors in the final result.
initial_character_info =  client.chat.completions.create(
        model="gpt-4o",
        messages=messages,
        stop=None,
        max_tokens=1000,
        temperature=0,
        n = 1)

# Store the API call response results as a file (when connecting subsequent distributed API call loops, consider storing in environment variables for continuous calls and modifications).
initial_response = initial_character_info.choices[0].message.content
"""with open("D:/桌面/taxonomy_primary_result/The_GPT-4_result/Dataset_3 (The Lycopodiales (Diphasiastrum, Huperzia, Isoetes, Lycopodium, Selaginella)) 4/Information gain methods/initial_character_info", 'w') as file:
    file.write(initial_response)"""
print(initial_response)

# Initial Character Classify Result #

Based on the provided morphological matrix and calculating the information gain for each character, the character with the highest information gain is **Character1**. This character will be used for the initial classification of species.

### Classification Based on Character1:

- **State 1**:
  - aquilex
  - auerbachi
  - caspary
  - forelii
  - inopinatus
  - luchoffmani
  - muotae
  - setiferus
  - thienemanni
  - tonywhitteni
  - B - (cf.thienemanni)

- **State 2**:
  - brixianus
  - murimali
  - puteanus
  - rhenorhodanesis species complex
  - styx
  - thuringius
  - virei
  - A - (cf.stygius)

Next, I will further classify each group based on the next character with the highest information gain within each subgroup.


3, Save the initial API call results, parse them, and store them as a dictionary as a new variable. (**There is a potential challenge in the parsing process. For subsequent distributed loops, this parsing step is crucial. The parsing function needs to handle the API response. Although the API response prompt explicitly specifies the final output, due to the randomness of LLMs, it's challenging to ensure accurate parsing results every time (consider setting a seed or adjusting the parsing format)).

In [102]:
import re
import json

# 定义一个函数来解析结果字符串
def parse_classification_result(result_text):
    classification = {}
    
    # 使用正则表达式匹配每个状态块
    state_sections = re.findall(r'- \*\*State (\d+(?: and \d+)?)\*\*:\n((?:\s+-\s.*\n)+)', result_text, re.MULTILINE)
    
    if not state_sections:
        print("No state sections found.")
    
    for state, species_block in state_sections:
        print(f"Parsing state {state} with block:\n{species_block}")
        # 匹配物种列表
        species_list = re.findall(r'-\s+(.+)', species_block)
        
        if not species_list:
            print(f"No species found for state {state}.")
        
        classification[state.strip()] = species_list
    
    return classification

# 示例输入数据
initial_classification = """
# Initial Character Classify Result #

Based on the provided morphological matrix and calculating the information gain for each character, the character with the highest information gain is **Character1**. This character will be used for the initial classification of species.

### Classification Based on Character1:

- **State 1**:
  - aquilex
  - auerbachi
  - caspary
  - forelii
  - inopinatus
  - luchoffmani
  - muotae
  - setiferus
  - thienemanni
  - tonywhitteni
  - B - (cf.thienemanni)

- **State 2**:
  - brixianus
  - murimali
  - puteanus
  - rhenorhodanesis species complex
  - styx
  - thuringius
  - virei
  - A - (cf.stygius)

Next, I will further classify each group based on the next character with the highest information gain within each subgroup.
"""

# 解析初步分类结果
parsed_initial_classification = parse_classification_result(initial_classification)

# 打印解析结果
print(json.dumps(parsed_initial_classification, indent=2, ensure_ascii=False))


Parsing state 1 with block:
  - aquilex
  - auerbachi
  - caspary
  - forelii
  - inopinatus
  - luchoffmani
  - muotae
  - setiferus
  - thienemanni
  - tonywhitteni
  - B - (cf.thienemanni)

- **State 2**:
  - brixianus
  - murimali
  - puteanus
  - rhenorhodanesis species complex
  - styx
  - thuringius
  - virei
  - A - (cf.stygius)

{
  "1": [
    "aquilex",
    "auerbachi",
    "caspary",
    "forelii",
    "inopinatus",
    "luchoffmani",
    "muotae",
    "setiferus",
    "thienemanni",
    "tonywhitteni",
    "B - (cf.thienemanni)",
    "**State 2**:",
    "brixianus",
    "murimali",
    "puteanus",
    "rhenorhodanesis species complex",
    "styx",
    "thuringius",
    "virei",
    "A - (cf.stygius)"
  ]
}


In [77]:
# Read initial character information
with open("D:/桌面/taxonomy_primary_result/The_GPT-4_result/Dataset_4 (The genera of grass moths) 7/Information gain methods/initial_character_info", "r", encoding="utf-8") as file:
    initial_character = file.read()
    print(initial_character)
    # This can be done by using print(initial_character)

# Define the function to parse the API response results.
def parse_classification_result(result_text):
    classification = {}
    # Use regular expressions to match each state block.
    state_sections = re.findall(r'- \*\*State\s"([\d\s\w&]+)"\*\*:\n((?:\s*-\s[A-Za-z_]+\n)+)', result_text)
    
    for state, species_block in state_sections:
        # Match the species list.
        species_list = re.findall(r'-\s([A-Za-z_]+)', species_block)
        classification[state.strip()] = species_list
    
    return classification

# Parse the initial classification results.
parsed_initial_classification = parse_classification_result(initial_character)
print(parsed_initial_classification)

# Example output of the parsed results.
# for state, species in parsed_initial_classification.items():
    # print(f"State {state}: {', '.join(species)}")

# Initial Character Classify Result #

Based on the provided morphological matrix and the calculation of information gain for each character, the initial classification character with the highest information gain is **Character1**. 

Here is the categorization of species based on the states of **Character1**:

- **Character1 = 1**:
  - Diphasiastrum_alpinum
  - Diphasiastrum_complanatum
  - Huperzia_selago
  - Lycopodiella_inundata
  - Lycopodium_annotinum
  - Lycopodium_clavatum
  - Selaginella_kraussiana
  - Selaginella_selaginoides

- **Character1 = 2**:
  - Isoetes_echinospora
  - Isoetes_histrix
  - Isoetes_lacustris
{}


The results are shown here (stored in a dictionary format through parsing to call each line separately).

In [117]:
# Parsing the API response for the initial character using the parsing function.

initial_classification = {
  "1": [
    "aquilex'",
    "auerbachi'",
    "caspary'",
    "forelii'",
    "inopinatus'",
    "luchoffmani'",
    "muotae'",
    "setiferus'",
    "thienemanni'",
    "tonywhitteni'",
    "B - (cf.thienemanni)'"
  ],
  "2": [
    "brixianus'",
    "murimali'",
    "puteanus'",
    "rhenorhodanesis species complex'",
    "styx'",
    "thuringius'",
    "virei'",
    "A - (cf.stygius)'"
  ]
}


# List to store all values.
groups = []

# This sets the foundation for subsequent distributed loops. Each key-value pair in the dictionary is stored in a list variable, enabling multiple iterations over these variables to eventually store the results.
for state, species_list in initial_classification.items():
    groups.append(species_list)

# Print the results stored in the list.
for i, group in enumerate(groups, 1):
    print(f"Group {i} species:", group)

print(groups[0])
print(parsed_initial_classification)

Group 1 species: ["aquilex'", "auerbachi'", "caspary'", "forelii'", "inopinatus'", "luchoffmani'", "muotae'", "setiferus'", "thienemanni'", "tonywhitteni'", "B - (cf.thienemanni)'"]
Group 2 species: ["brixianus'", "murimali'", "puteanus'", "rhenorhodanesis species complex'", "styx'", "thuringius'", "virei'", "A - (cf.stygius)'"]
["aquilex'", "auerbachi'", "caspary'", "forelii'", "inopinatus'", "luchoffmani'", "muotae'", "setiferus'", "thienemanni'", "tonywhitteni'", "B - (cf.thienemanni)'"]
{'1': ['aquilex', 'auerbachi', 'caspary', 'forelii', 'inopinatus', 'luchoffmani', 'muotae', 'setiferus', 'thienemanni', 'tonywhitteni', 'B - (cf.thienemanni)', '**State 2**:', 'brixianus', 'murimali', 'puteanus', 'rhenorhodanesis species complex', 'styx', 'thuringius', 'virei', 'A - (cf.stygius)']}


4, Traverse the contents of each group sequentially, selecting classification characters for the species in each group (i.e., implementing distributed loop API calls) (**However, I have not yet constructed the loop results effectively. Therefore, I can use a simpler method for now to test API calls for information in different groups).

In [118]:
# The species classification results obtained from the different states of the initial character: further classify state 1.

# Each time a classification analysis is needed, the complete morphological matrix information can be loaded and passed to the API call again. 
# This can avoid calculation redundancy during API reloads, thus avoiding additional computational burdens.

# If a loop needs to be constructed, this method can be used to set the result as a stored empty list.
# Set an empty list to store results = {}.

# An effective method is to uniformly store all results [0], [1], [2] in the dictionary, unifying the storage index of the key values, hence using this [i].

with open("D:/桌面/taxonomy_primary_result/The_GPT-4_result/Dataset_12 (Translating Niphargus barcodes from Switzerland into taxonomy with a description of two new species (Amphipoda, Niphargidae) ) 20/information gain methods/matrix_knowledge_graph.json", "r") as file:
    data = json.load(file)
    group_matrix = {species: data[species] for species in groups[0]}
    group_matrix_str = json.dumps(group_matrix, ensure_ascii=False)
    
    messages2 = [
            {"role": "system",
             "content":"""
                 You are a helpful taxonomist assistant.\n
                 You are skilled at calculating the correct information gain to choose the character that best divides species into even groups based on their states.\n
                 Based on the selected character, classify the species into different groups according to their states.\n
                 For each group with more than two species, continue selecting characters to further classify this group until each group only has one species.\n
                 After multiple classifications, determine the final classification levels and record each classifying character and its state.\n
                 Finally, generate a taxonomic key.
                 """},
            {"role": "system",
             "content":"""
                Generate the taxonomic key based on the provided morphological matrix. \n
                The process involves selecting a character to classify the species into groups. Repeat this classification within each subgroup until each group contains only one species.
                Information gain measures how much the uncertainty in the dataset is reduced after using a character for classification. It helps in selecting characters that minimize the entropy of the subset after classification, leading to better classification results.
                Please select the classification character for these group's species based on the morphological matrix and information gain methods.
                In the morphological matrix, 'Missing' and 'Not applicable' are invalid states. If a character has invalid states for the group being classified, it should be ignored.
                States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type and this multi-state characterization should not be confused with the single states within it.The initial character should have no more than three state types.
                You need to calculate the information gain for each character and choose the highest information gain result. The higher the information gain result, the greater the contribution of the feature to the classification.
                After selecting the initial classification character and categorizing the species based on its state, repeat the process within each subgroup. For each subgroup, select the character with the highest information gain to further classify the species. Continue this process recursively until each group contains only one species.
                Now I will show you the morphological matrix. Please provide the classification character and the categorization of species based on its state. Then, continue to classify each subgroup recursively, showing the chosen character and categorization for each subgroup. Please present the result in a structured format, with each step clearly labeled.
                please don't show how you analysis and calculate, please show me the final result           
            """},
            {"role": "assistant",
             "content": """
                Understood. I will generate the taxonomic key based on the provided morphological matrix. Here is a summary of the steps I will follow:\n
                1. The matrix includes all species and their different states for each character.\n
                2. I will select a character to classify the species into groups and repeat this classification within each subgroup until each group contains only one species.\n
                3. I will use information gain to measure how much the uncertainty in the dataset is reduced after using a feature for classification. This helps in selecting features that minimize the entropy of the subset after classification, leading to better classification results.\n
                4. I will select the classification character for the group's species based on the morphological matrix and information gain methods.\n
                5. In the morphological matrix, 'Missing' and 'Not applicable' are considered invalid states. If a character has invalid states for the group being classified, it will be ignored.\n
                6. States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type, and This multi-state characterization should not be confused with the individual states within it. The classification character should have no more than three state types.\n
                7. I will use information gain to calculate all character and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification. \n
                8. The final result will provide only the initial classification character and the categorization of species based on its state. \n
                9. Don't need to show how the process about choose, only need to show the final nest result, don't need to show the process, and i will store the final nested result in #character classify result# block
                you should using the correct state to building the correct result
                Please provide the group morphological matrix data so that I can proceed with the classification.
             """},
            # When importing data types here, consider directly using the content previously stored in the dictionary to select different key values, and combine it with the initial complete morphological matrix information.
            {"role": "user", "content": f"Here is the group information need to be classify, and here is the morphological matrix{group_matrix_str}"}
        ]
    response =  client.chat.completions.create(
            model="gpt-4o",
            messages=messages2,
            stop=None,
            temperature=0,
            max_tokens=1500,
            n = 1
    )
    
    results1 = response.choices[0].message.content
print(results1)

# Character Classification Result

## Step 1: Initial Classification

### Character: Character2

- **State 1**:
  - luchoffmani'
  - muotae'
  - setiferus'

- **State 2**:
  - aquilex'
  - auerbachi'
  - caspary'
  - forelii'
  - inopinatus'
  - thienemanni'
  - tonywhitteni'
  - B - (cf.thienemanni)'

## Step 2: Classification within State 1 (Character2 = 1)

### Character: Character8

- **State 1**:
  - setiferus'

- **State 2**:
  - luchoffmani'

- **State 1 and 2**:
  - muotae'

## Step 3: Classification within State 2 (Character2 = 2)

### Character: Character9

- **State 1**:
  - caspary'

- **State 2**:
  - aquilex'
  - auerbachi'
  - forelii'
  - inopinatus'
  - thienemanni'
  - B - (cf.thienemanni)'

- **State 3**:
  - luchoffmani'
  - tonywhitteni'

## Step 4: Classification within State 2 (Character2 = 2, Character9 = 2)

### Character: Character17

- **State 2**:
  - aquilex'
  - forelii'
  - inopinatus'
  - thienemanni'

- **State 3**:
  - auerbachi'
  - B - (cf.thienemann

By calling the API again to achieve the distributed loop process. For example, in this loop, if the result to be called is the key 2 in the dictionary, it is the final species classification result obtained by state 2 of the initial character. Therefore, the call here is through group_2_species, but it can actually be done by calling groups[2]. Using groups[i], you can loop through the results stored in the initial_classification dictionary, calling different key values to achieve classification results for different initial character states. This can make the code more concise and clear, reducing redundancy by using a loop.

In [126]:
# The species classification results obtained from the different states of the initial character: further classify key 2 (state 2).
with open("D:/桌面/taxonomy_primary_result/The_GPT-4_result/Dataset_12 (Translating Niphargus barcodes from Switzerland into taxonomy with a description of two new species (Amphipoda, Niphargidae) ) 20/information gain methods/matrix_knowledge_graph.json", "r") as file:
    data = json.load(file)
    group_matrix = {species: data[species] for species in groups[1]}
    group_matrix_str = json.dumps(group_matrix, ensure_ascii=False)

    messages3 = [
            {"role": "system",
             "content":"""
                 You are a helpful taxonomist assistant.\n
                 You are skilled at calculating the correct information gain to choose the character that best divides species into even groups based on their states.\n
                 Based on the selected character, classify the species into different groups according to their states.\n
                 For each group with more than two species, continue selecting characters to further classify this group until each group only has one species.\n
                 After multiple classifications, determine the final classification levels and record each classifying character and its state.\n
                 Finally, generate a taxonomic key.
                 """},
            {"role": "system","content":"""
                Generate the nested taxonomic key based on the provided morphological matrix. \n
                The process involves selecting a character to classify the species into groups. Repeat this classification within each subgroup until each group contains only one species.
                Information gain measures how much the uncertainty in the dataset is reduced after using a character for classification. It helps in selecting characters that minimize the entropy of the subset after classification, leading to better classification results.
                Please select the classification character for these group's species based on the morphological matrix and information gain methods.
                In the morphological matrix, 'Missing' and 'Not applicable' are invalid states. If a character has invalid states for the group being classified, it should be ignored.
                States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type and this multi-state characterization should not be confused with the single states within it (the state of '3' and '2 and 3' is different state, when you choose the character to based on the state to distinguish need to careful handle).The initial character should have no more than three state types.
                You need to calculate the information gain for each character and choose the highest information gain result. The higher the information gain result, the greater the contribution of the feature to the classification.
                After selecting the initial classification character and categorizing the species based on its state, repeat the process within each subgroup. For each subgroup, select the character with the highest information gain to further classify the species. Continue this process recursively until each group contains only one species.
                Now I will show you the morphological matrix. Please provide the classification character and the categorization of species based on its state. Then, continue to classify each subgroup recursively, showing the chosen character and categorization for each subgroup. Please present the result in a structured format, with each step clearly labeled.
                please don't show how you analysis and calculate, please show me the final result           
            """},
            {"role": "assistant",
             "content": """
                Understood. I will generate the nested taxonomic key based on the provided morphological matrix. Here is a summary of the steps I will follow:\n
                1. The matrix includes all species and their different states for each character.\n
                2. I will select a character to classify the species into groups and repeat this classification within each subgroup until each group contains only one species.\n
                3. I will use information gain to measure how much the uncertainty in the dataset is reduced after using a feature for classification. This helps in selecting features that minimize the entropy of the subset after classification, leading to better classification results.\n
                4. I will select the classification character for the group's species based on the morphological matrix and information gain methods.\n
                5. In the morphological matrix, 'Missing' and 'Not applicable' are considered invalid states. If a character has invalid states for the group being classified, it will be ignored.\n
                6. States are represented by numbers. For example, '2 and 3' means multiple states should be treated as a single state type, and This multi-state characterization should not be confused with the individual states(like '2', '3') within it (such as '3' and '2 and 3'  is the different state, these are two separate states, when i choose character to based on different state to distinguish the species). The classification character should have no more than three state types.\n
                7. I will use information gain to calculate all character and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification. \n
                8. The final result will provide only the initial classification character and the categorization of species based on its state. \n
                9. Don't need to show how the process about choose, only need to show the final result as nested structure, and i will store nested result in #character classify result# block
                Please provide the group morphological matrix data so that I can proceed with the classification.
             """},
            # When importing data types here, consider choosing to call different keys in the dictionary directly from what was previously stored in the dictionary, as well as combining it with the initial
            {"role": "user", "content": f"Here is the group information need to be classify, and here is the morphological matrix{group_matrix_str}"}
        ]
    response =  client.chat.completions.create(
            model="gpt-4o",
            messages=messages3,
            stop=None,
            temperature=0,
            max_tokens=1500,
            n = 1
    )
    result2 = response.choices[0].message.content
print(result2)

# Character Classification Result

## Step 1: Initial Classification

### Character: Character3

- **State 1**:
  - "puteanus"
  - "rhenorhodanesis species complex"
  - "styx"
  - "thuringius"
  - "A - (cf.stygius)"

- **State 2**:
  - "brixianus"
  - "murimali"
  - "virei"

## Step 2: Classification within State 1 (Character3 = 1)

### Character: Character9

- **State 1**:
  - "murimali"

- **State 2**:
  - "puteanus"
  - "rhenorhodanesis species complex"
  - "thuringius"

- **State 3**:
  - "styx"
  - "A - (cf.stygius)"

## Step 3: Classification within State 2 (Character3 = 1, Character9 = 2)

### Character: Character17

- **State 1**:
  - "puteanus"
  - "rhenorhodanesis species complex"
  - "thuringius"

- **State 2**:
  - "styx"
  - "A - (cf.stygius)"

## Step 4: Classification within State 1 (Character3 = 1, Character9 = 2, Character17 = 1)

### Character: Character5

- **State 1**:
  - "thuringius"

- **State 2**:
  - "puteanus"
  - "rhenorhodanesis species complex"

## Step 5: 

Repeat the above process to achieve a complete classification by continuing to repeat the loop on the key values stored in the dictionary initial_classification for the results obtained from the different feature states in the initial character

In [None]:
# The species classification results obtained from the different states of the initial character: further classify key 3 (state 1 and 2).
client = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))
# Each time a classification analysis is needed, the complete morphological matrix information can be loaded and passed to the API call again.
# This can avoid calculation redundancy during API reloads, thus avoiding additional computational burdens.
with open("D:/桌面/taxonomy_primary_result/The_GPT-4_result/Dataset_11 (grass genera) 19/Information gain methods/matrix_knowledge_graph.json", "r") as file:
    data = json.load(file)

    messages3 = [
            {"role": "system",
             "content":"""
                 You are a helpful taxonomist assistant.\n
                 You are skilled at calculating the correct information gain to choose the character that best divides species into even groups based on their states.\n
                 Based on the selected character, classify the species into different groups according to their states.\n
                 For each group with more than two species, continue selecting characters to further classify this group until each group only has one species.\n
                 After multiple classifications, determine the final classification levels and record each classifying character and its state.\n
                 Finally, generate a taxonomic key.
                 """},
            {"role": "system","content":"""
                Generate the taxonomic key based on the provided morphological matrix. \n
                The process involves selecting a character to classify the species into groups. Repeat this classification within each subgroup until each group contains only one species.
                Information gain measures how much the uncertainty in the dataset is reduced after using a character for classification. It helps in selecting characters that minimize the entropy of the subset after classification, leading to better classification results.
                Please select the classification character for these group's species based on the morphological matrix and information gain methods.
                In the morphological matrix, 'Missing' and 'Not applicable' are invalid states. If a character has invalid states for the group being classified, it should be ignored.
                States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type and this multi-state characterization should not be confused with the single states within it.The initial character should have no more than three state types.
                You need to calculate the information gain for each character and choose the highest information gain result. The higher the information gain result, the greater the contribution of the feature to the classification.
                After selecting the initial classification character and categorizing the species based on its state, repeat the process within each subgroup. For each subgroup, select the character with the highest information gain to further classify the species. Continue this process recursively until each group contains only one species.
                Now I will show you the morphological matrix. Please provide the classification character and the categorization of species based on its state. Then, continue to classify each subgroup recursively, showing the chosen character and categorization for each subgroup. Please present the result in a structured format, with each step clearly labeled.
                please don't show how you analysis and calculate, please show me the final result           
            """},
            {"role": "assistant",
             "content": """
                Understood. I will generate the taxonomic key based on the provided morphological matrix. Here is a summary of the steps I will follow:\n
                1. The matrix includes all species and their different states for each character.\n
                2. I will select a character to classify the species into groups and repeat this classification within each subgroup until each group contains only one species.\n
                3. I will use information gain to measure how much the uncertainty in the dataset is reduced after using a feature for classification. This helps in selecting features that minimize the entropy of the subset after classification, leading to better classification results.\n
                4. I will select the classification character for the group's species based on the morphological matrix and information gain methods.\n
                5. In the morphological matrix, 'Missing' and 'Not applicable' are considered invalid states. If a character has invalid states for the group being classified, it will be ignored.\n
                6. States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type, and This multi-state characterization should not be confused with the individual states within it (should distinguish '3' and '2 and 3'，it is different state). The classification character should have no more than three state types.\n
                7. I will use information gain to calculate all character and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification. \n
                8. The final result will provide only the initial classification character and the categorization of species based on its state. \n
                9. Don't need to show how the process about choose, only need to show the final nested result, and i will store result in #character classify result# block
                only choose the initial character to distinguish for this species
                Please provide the group morphological matrix data so that I can proceed with the classification.
             """},
            # When importing data types here, consider choosing to call different keys in the dictionary directly from what was previously stored in the dictionary, as well as combining it with the initial
            {"role": "user", "content": f"Here is the group information need to be classify: {groups[2]}, and here is the morphological matrix{json.dumps(data)}"}
        ]
    response =  client.chat.completions.create(
            model="gpt-4o",
            messages=messages3,
            stop=None,
            temperature=0,
            max_tokens=2000,
            n = 1
    )
    result3 = response.choices[0].message.content
print(result3)

5, Integrate the results of multiple distributed traversals (currently there are three traversal results due to the initial character states dividing into three groups).
Record the results of each loop separately and store them in a list.
For example, I can specify in the prompt that each classification's final result should be stored as #taxonomic_information#.
To avoid redundant results from other analyses, we can extract this block (#taxonomic_information#) separately for integration.
Refine the formatting requirements to create a complete classification key table, and choose appropriate methods and formats for optimization during the integration process.

In [1]:
# Through the results obtained from the previous distributed traversal loop, i.e., the classification results of all different feature states in the initial character, the classification results of the API calls again
# print(result1)
# print(result2)
# print(result3)

# We need to extract the most critical parts from these three sections, namely the classification paths of species in each subgroup.
# In each distributed loop result, we need to specify that the final results should store the classification results of species in each subgroup in the block (Final Taxonomic Key). 
# Then, by constructing an extraction function, we can extract the final classification information.

# Define a function to extract the required part from each result
def extract_final_taxonomic_key(result):
    match = re.search(r'Final Taxonomic Key(.*)', result, re.DOTALL)
    if match:
        return match.group(1).strip()
    else:
        return None

# Extract the required part from each result
final_taxonomic_key1 = extract_final_taxonomic_key(results1)
final_taxonomic_key2 = extract_final_taxonomic_key(result2)

# Print the extracted results
print("Final Taxonomic Key 1:\n", final_taxonomic_key1)
print("Final Taxonomic Key 2:\n", final_taxonomic_key2)

NameError: name 'results1' is not defined

In [None]:
def parse_result(result):
    lines = result.strip().split('\n')
    result_dict = {}
    stack = [(0, result_dict)]  # Use tuples to record indentation level and current dictionary

    for line in lines:
        indent_level = len(line) - len(line.lstrip())
        current_dict = stack[-1][1]
        
        # Adjust stack to match current indentation level
        while stack and stack[-1][0] >= indent_level:
            stack.pop()
        
        if ':' in line:
            parts = line.split(':', 1)
            if len(parts) < 2:
                print(f"Skipping malformed line (no value after ':'): {line}")
                continue
            
            key = parts[0].strip()
            value = parts[1].strip()
            
            if value:
                current_dict[key] = value
            else:
                current_dict[key] = {}
                stack.append((indent_level, current_dict[key]))
        elif '=' in line:
            parts = line.split('=', 1)
            if len(parts) < 2:
                print(f"Skipping malformed line (no value after '='): {line}")
                continue
            
            key = parts[0].strip()
            value = parts[1].strip()
            
            if value:
                current_dict[key] = value
            else:
                current_dict[key] = {}
                stack.append((indent_level, current_dict[key]))
        else:
            key = line.strip()
            if not key:
                print(f"Skipping empty line: {line}")
                continue
            
            current_dict[key] = {}
            stack.append((indent_level, current_dict[key]))
    
    return result_dict

# Example input data
result1_1 = """
1. **Character 1:**
   - State "1 and 2": Agriphila
   - State "1":
     1. **Character 3:**
        - State "1": Chilo
        - State "2":
          1. **Character 24:**
             - State "2": Euchromius
             - State "1": Haimbachia
"""
result2_1 = """
1. Character2 = 1
    1. Character5 = 1
        1. Character10 = 1 and 2: Chrysocrambus
        2. Character10 = 2 and 3: Chrysoteuchia
        3. Character10 = 1 and 2 and 3: Crambus
        4. Character10 = 3: Thisanotia
    2. Character5 = 2
        1. Character4 = 2
            1. Character6 = 1 and 2 and 3: Catoptria
            2. Character6 = 3: Ancylolomia
        2. Character4 = 1 and 2
            1. Character16 = 1 and 2: Platytes
            2. Character16 = 2: Pediasia
2. Character2 = 2
    1. Character4 = 1
        1. Character8 = 1: Donacaula
        2. Character8 = 1 and 2: Schoenobius
    2. Character4 = 1 and 2: Calamotropha
"""

# Parse each result string
parsed_result1 = parse_result(result1_1)
parsed_result2 = parse_result(result2_1)

# Print the parsed results for inspection
import json
print(json.dumps(parsed_result1, indent=2))
print(json.dumps(parsed_result2, indent=2))


In [None]:
import re
import json

# 定义一个函数来解析每个结果字符串
def parse_result(result):
    lines = result.strip().split('\n')
    result_dict = {}
    stack = [(0, result_dict)]  # 使用元组来记录缩进级别和当前字典

    for line in lines:
        indent_level = len(line) - len(line.lstrip())
        if stack:
            current_dict = stack[-1][1]
        else:
            current_dict = result_dict

        # 调整堆栈以匹配当前缩进级别
        while stack and stack[-1][0] >= indent_level:
            stack.pop()
            if stack:
                current_dict = stack[-1][1]
            else:
                current_dict = result_dict

        if ':' in line:
            parts = line.split(':', 1)
            if len(parts) < 2:
                print(f"Skipping malformed line (no value after ':'): {line}")
                continue

            key = parts[0].strip()
            value = parts[1].strip()

            if value:
                if key.startswith('- State'):
                    state = re.search(r'State\s"(\d+)"', key)
                    if state:
                        state_key = state.group(1)
                        if state_key not in current_dict:
                            current_dict[state_key] = []
                        current_dict[state_key].append(value)
                else:
                    current_dict[key] = value
            else:
                current_dict[key] = {}
                stack.append((indent_level, current_dict[key]))
        elif '=' in line:
            parts = line.split('=', 1)
            if len(parts) < 2:
                print(f"Skipping malformed line (no value after '='): {line}")
                continue

            key = parts[0].strip()
            value = parts[1].strip()

            if value:
                current_dict[key] = value
            else:
                current_dict[key] = {}
                stack.append((indent_level, current_dict[key]))
        else:
            key = line.strip()
            if not key:
                print(f"Skipping empty line: {line}")
                continue

            current_dict[key] = {}
            stack.append((indent_level, current_dict[key]))

    return result_dict

# 示例输入数据
result1_1 = """
1. **Character 1:**
   - State "1 and 2": Agriphila
   - State "1":
     1. **Character 3:**
        - State "1": Chilo
        - State "2":
          1. **Character 24:**
             - State "2": Euchromius
             - State "1": Haimbachia
"""
result2_1 = """
1. Character2 = 1
    1. Character5 = 1
        1. Character10 = 1 and 2: Chrysocrambus
        2. Character10 = 2 and 3: Chrysoteuchia
        3. Character10 = 1 and 2 and 3: Crambus
        4. Character10 = 3: Thisanotia
    2. Character5 = 2
        1. Character4 = 2
            1. Character6 = 1 and 2 and 3: Catoptria
            2. Character6 = 3: Ancylolomia
        2. Character4 = 1 and 2
            1. Character16 = 1 and 2: Platytes
            2. Character16 = 2: Pediasia
2. Character2 = 2
    1. Character4 = 1
        1. Character8 = 1: Donacaula
        2. Character8 = 1 and 2: Schoenobius
    2. Character4 = 1 and 2: Calamotropha
"""

# 解析每个结果字符串
parsed_result1 = parse_result(result1_1)
parsed_result2 = parse_result(result2_1)

# 将解析结果合并到最终字典中
combined_results = {
    '1': [],
    '2': []
}

# 将解析的物种添加到字典中
def add_species_to_combined_results(parsed_result, combined_results):
    for key, value in parsed_result.items():
        if isinstance(value, list):
            if key in combined_results:
                combined_results[key].extend(value)
            else:
                combined_results[key] = value
        elif isinstance(value, dict):
            add_species_to_combined_results(value, combined_results)

add_species_to_combined_results(parsed_result1, combined_results)
add_species_to_combined_results(parsed_result2, combined_results)

# 打印嵌套结构
print(json.dumps(combined_results, indent=2))
