In [2]:
import json
from openai import OpenAI
import os
import re
# Call the API_KEY from the environment
client = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))
# First, import the required dataset
with open("D:/桌面/taxonomy_primary_result/The_GPT-4_result/Dataset_3 (The Lycopodiales (Diphasiastrum, Huperzia, Isoetes, Lycopodium, Selaginella)) 4/Information gain methods/matrix_knowledge_graph.json", "r", encoding="utf-8") as file:
    matrix_data = json.load(file)
# Since the input is in JSON format, convert this JSON file to a string format when calling the API
matrix_data_str = json.dumps(matrix_data, ensure_ascii=False)
#print(matrix_data_str)
# This is the prompt for the API input, using the client.chat.completions.create interface to conduct multi-turn conversations.
# By assigning different roles in the conversation (user, assistant, system), all input information is fully conveyed to the API model

messages = [
            # Set the system role to focus on taxonomy tasks.
            # Emphasize the system's understanding of morphological matrices, information gain, and the construction of classification keys in the system setup.
            {"role":"system",
             "content":
                 """You are a helpful taxonomist assistant.
                 You are skilled at calculating the correct information gain to choose the character that best divides species into even groups based on their states.
                 Based on the selected character, classify the species into different groups according to their states.
                 For each group with more than two species, continue selecting characters to further classify this group until each group only has one species.
                 After multiple classifications, determine the final classification levels and record each classifying character and its state. 
                 Finally, generate a taxonomic key.
                 Please format the classification result as follows:
                ```
                {
                    "Character": "Character1",
                    "States": {
                        "State 1": ["species1", "species2", ...],
                        "State 2": ["species3", "species4", ...],
                        "State ...": ["species5", "species6", ...]
                    }
                }
                ```
                Ensure that the response follows this format exactly.
                 """},
            # Input the main task request, construct the classification key, and emphasize the details needed for classification.
            # This includes: the main objective, information gain calculation, ignoring invalid states, clarifying multi-character states, guiding correct selection based on the significance of information gain, and standardizing the final output format for subsequent extraction of API results.
            {"role":"system",
            "content": """
                Generate the taxonomic key based on the provided morphological matrix. The matrix includes all species and their different states for each character.\n
                The process involves selecting a character to classify the species into groups. Repeat this classification within each subgroup until each group contains only one species.\n
                Information gain measures how much the uncertainty in the dataset is reduced after using a character for classification. It helps in selecting characters that minimize the entropy of the subset after classification, leading to better classification results.\n
                Please select the initial classification character for all species based on the morphological matrix and information gain methods.\n
                In the morphological matrix, 'Missing' and 'Not applicable' are invalid states. If a character has invalid states for the group being classified, it should be ignored.\n
                States are represented by numbers, and like the '1 and 2' means multiple states should be treated as a single state type.(such as '3' and '2 and 3'  is the different state, these are two separate states, when i choose character to based on different state to distinguish the species) \n
                you need to calculate the information gain for each character, and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification.\n
                Now I will show you the morphological matrix. Please provide only the initial classification character and the categorization of species based on its state, and you should singly show this called as # initial character classify result # 
            """},
            # Use the assistant to summarize and refine the prompt content for the API.
            # Through the conversation with the assistant, deepen the API understanding of the content to some extent, control the API response results, and standardize the output format of the API response.
            {"role": "assistant",
            "content": """
                Understood. I will generate the taxonomic key based on the provided morphological matrix. Here is a summary of the steps I will follow:\n
                1. The matrix includes all species and their different states for each character.\n
                2. I will select a character to classify the species into groups and repeat this classification within each subgroup until each group contains only one species.\n
                3. I will use information gain to measure how much the uncertainty in the dataset is reduced after using a feature for classification. This helps in selecting features that minimize the entropy of the subset after classification, leading to better classification results.\n
                4. I will select the initial classification character for all species based on the morphological matrix and information gain methods.\n
                5. In the morphological matrix, 'Missing' and 'Not applicable' are considered invalid states. If a character has invalid states for the group being classified, it will be ignored.\n
                6. States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type. (such as '1' and '1 and 2'  is the different state, these are two separate states, when i choose character to based on different state to distinguish the species)\n
                7. I will use information gain to calculate all character and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification.so i need to make sure the result is Average classification \n
                8. The final result will provide only the initial classification character and the categorization of species based on its state. \n
                9. Don't need to show how i calculate, only need to show the final result, and please show the final result in #initail character classify result# block, Don't have errors where the state and species don't match
                
                Please provide the morphological matrix data so that I can proceed with the initial classification.
            """
            },
            # Input the corresponding morphological matrix information for analysis.
            {"role":"user",
             "content":
                 f"Here is morphological matrix:{matrix_data_str}"}
            ]

# Set the various parameters that need to be controlled when responding to the API. I think the temperature should be set to 0, and the max_tokens should either not be set or set to a small number. This is to save costs and because generating the corresponding results does not require a long output. If the token number is too large and no stop sequence is set, it can lead to complex redundancy and errors in the final result.
initial_character_info =  client.chat.completions.create(
        model="gpt-4o",
        messages=messages,
        stop=None,
        max_tokens=1500,
        temperature=0,
        n = 1)

# Store the API call response results as a file (when connecting subsequent distributed API call loops, consider storing in environment variables for continuous calls and modifications).
initial_response = initial_character_info.choices[0].message.content
print(initial_response)

# Initial Character Classify Result #
```
{
    "Character": "Character1",
    "States": {
        "1": ["Diphasiastrum_alpinum", "Diphasiastrum_complanatum", "Huperzia_selago", "Lycopodiella_inundata", "Lycopodium_annotinum", "Lycopodium_clavatum", "Selaginella_kraussiana", "Selaginella_selaginoides"],
        "2": ["Isoetes_echinospora", "Isoetes_histrix", "Isoetes_lacustris"]
    }
}
```


In [3]:
def parse_classification_result(result_text):
    classification = {"Character": None, "States": {}}
    character_match = re.search(r'"Character": "([^"]+)"', result_text)
    if character_match:
        classification["Character"] = character_match.group(1)
    state_sections = re.findall(r'"(\d+|[^"]+)":\s*\[(.*?)\]', result_text)
    for state, species_block in state_sections:
        species_list = re.findall(r'"([^"]+)"', species_block)
        classification["States"][state] = species_list
    return classification
parsed_initial_classification = parse_classification_result(initial_response)
print(json.dumps(parsed_initial_classification, indent=2, ensure_ascii=False))

{
  "Character": "Character1",
  "States": {
    "1": [
      "Diphasiastrum_alpinum",
      "Diphasiastrum_complanatum",
      "Huperzia_selago",
      "Lycopodiella_inundata",
      "Lycopodium_annotinum",
      "Lycopodium_clavatum",
      "Selaginella_kraussiana",
      "Selaginella_selaginoides"
    ],
    "2": [
      "Isoetes_echinospora",
      "Isoetes_histrix",
      "Isoetes_lacustris"
    ]
  }
}


In [4]:
groups = []
for state, species_list in parsed_initial_classification["States"].items():
    groups.append(species_list)
for i, group in enumerate(groups, 1):
    print(f"Group {i} species:", group)
print(groups[0])
print(groups[1])

Group 1 species: ['Diphasiastrum_alpinum', 'Diphasiastrum_complanatum', 'Huperzia_selago', 'Lycopodiella_inundata', 'Lycopodium_annotinum', 'Lycopodium_clavatum', 'Selaginella_kraussiana', 'Selaginella_selaginoides']
Group 2 species: ['Isoetes_echinospora', 'Isoetes_histrix', 'Isoetes_lacustris']
['Diphasiastrum_alpinum', 'Diphasiastrum_complanatum', 'Huperzia_selago', 'Lycopodiella_inundata', 'Lycopodium_annotinum', 'Lycopodium_clavatum', 'Selaginella_kraussiana', 'Selaginella_selaginoides']
['Isoetes_echinospora', 'Isoetes_histrix', 'Isoetes_lacustris']


In [6]:
with open("D:/桌面/taxonomy_primary_result/The_GPT-4_result/Dataset_3 (The Lycopodiales (Diphasiastrum, Huperzia, Isoetes, Lycopodium, Selaginella)) 4/Information gain methods/matrix_knowledge_graph.json", "r") as file:
    data = json.load(file)
    group_matrix = {species: data[species] for species in groups[0]}
    group_matrix_str = json.dumps(group_matrix, ensure_ascii=False)
    
    messages3 = [
            {"role": "system",
             "content":"""
                 You are a helpful taxonomist assistant.\n
                 You are skilled at calculating the correct information gain to choose the character that best divides species into even groups based on their states.\n
                 Based on the selected character, classify the species into different groups according to their states.\n
                 For each group with more than two species, continue selecting characters to further classify this group until each group only has one species.\n
                 After multiple classifications, determine the final classification levels and record each classifying character and its state.\n
                 Finally, generate a taxonomic key.
                 """},
            {"role": "system","content":"""
                Generate the nested taxonomic key based on the provided morphological matrix. \n
                The process involves selecting a character to classify the species into groups. Repeat this classification within each subgroup until each group contains only one species.
                Information gain measures how much the uncertainty in the dataset is reduced after using a character for classification. It helps in selecting characters that minimize the entropy of the subset after classification, leading to better classification results.
                Please select the classification character for these group's species based on the morphological matrix and information gain methods.
                In the morphological matrix, 'Missing' and 'Not applicable' are invalid states. If a character has invalid states for the group being classified, it should be ignored.
                States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type and this multi-state characterization should not be confused with the single states within it (the state of '3' and '2 and 3' is different state, when you choose the character to based on the state to distinguish need to careful handle).The initial character should have no more than three state types.
                You need to calculate the information gain for each character and choose the highest information gain result. The higher the information gain result, the greater the contribution of the feature to the classification.
                After selecting the initial classification character and categorizing the species based on its state, repeat the process within each subgroup. For each subgroup, select the character with the highest information gain to further classify the species. Continue this process recursively until each group contains only one species.
                Now I will show you the morphological matrix. Please provide the classification character and the categorization of species based on its state. Then, continue to classify each subgroup recursively, showing the chosen character and categorization for each subgroup. Please present the result in a structured format, with each step clearly labeled.
                please don't show how you analysis and calculate, please show me the final result           
            """},
            {"role": "assistant",
             "content": """
                Understood. I will generate the nested taxonomic key based on the provided morphological matrix. Here is a summary of the steps I will follow:\n
                1. The matrix includes all species and their different states for each character.\n
                2. I will select a character to classify the species into groups and repeat this classification within each subgroup until each group contains only one species.\n
                3. I will use information gain to measure how much the uncertainty in the dataset is reduced after using a feature for classification. This helps in selecting features that minimize the entropy of the subset after classification, leading to better classification results.\n
                4. I will select the classification character for the group's species based on the morphological matrix and information gain methods.\n
                5. In the morphological matrix, 'Missing' and 'Not applicable' are considered invalid states. If a character has invalid states for the group being classified, it will be ignored.\n
                6. States are represented by numbers. For example, '2 and 3' means multiple states should be treated as a single state type, and This multi-state characterization should not be confused with the individual states(like '2', '3') within it (such as '3' and '2 and 3'  is the different state, these are two separate states, when i choose character to based on different state to distinguish the species). The classification character should have no more than three state types.\n
                7. I will use information gain to calculate all character and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification. \n
                8. The final result will provide only the initial classification character and the categorization of species based on its state. \n
                9. Don't need to show how the process about choose, only need to show the final result as nested structure, and i will store nested result in #character classify result# block
                Please provide the group morphological matrix data so that I can proceed with the classification.
             """},
            # When importing data types here, consider choosing to call different keys in the dictionary directly from what was previously stored in the dictionary, as well as combining it with the initial
            {"role": "user", "content": f"Here is the group information need to be classify, and here is the morphological matrix{group_matrix_str}"}
        ]
    response =  client.chat.completions.create(
            model="gpt-4o",
            messages=messages3,
            stop=None,
            temperature=0,
            max_tokens=1500,
            n = 1
    )
    result1 = response.choices[0].message.content
print(result1)

# Character Classification Result

## Step 1: Initial Classification
### Character: Character2
- State 1: Huperzia_selago
- State 2: Diphasiastrum_alpinum, Diphasiastrum_complanatum, Lycopodiella_inundata, Lycopodium_annotinum, Lycopodium_clavatum
- State 3: Selaginella_kraussiana, Selaginella_selaginoides

## Step 2: Classification within State 2
### Character: Character9
- State 1: Diphasiastrum_alpinum, Diphasiastrum_complanatum
- State 2: Lycopodiella_inundata, Lycopodium_annotinum, Lycopodium_clavatum

## Step 3: Classification within State 1 of Character9
### Character: Character5
- State 2: Diphasiastrum_alpinum
- State 3: Diphasiastrum_complanatum

## Step 4: Classification within State 2 of Character9
### Character: Character20
- State 1: Lycopodiella_inundata
- State 2 and 3: Lycopodium_annotinum
- State 1 and 2 and 3: Lycopodium_clavatum

## Step 5: Classification within State 3
### Character: Character11
- State 1 and 2: Selaginella_kraussiana, Selaginella_selaginoides

## 

In [5]:
with open("D:/桌面/taxonomy_primary_result/The_GPT-4_result/Dataset_3 (The Lycopodiales (Diphasiastrum, Huperzia, Isoetes, Lycopodium, Selaginella)) 4/Information gain methods/matrix_knowledge_graph.json", "r") as file:
    data = json.load(file)
    group_matrix = {species: data[species] for species in groups[1]}
    group_matrix_str = json.dumps(group_matrix, ensure_ascii=False)
    
    messages3 = [
            {"role": "system",
             "content":"""
                 You are a helpful taxonomist assistant.\n
                 You are skilled at calculating the correct information gain to choose the character that best divides species into even groups based on their states.\n
                 Based on the selected character, classify the species into different groups according to their states.\n
                 For each group with more than two species, continue selecting characters to further classify this group until each group only has one species.\n
                 After multiple classifications, determine the final classification levels and record each classifying character and its state.\n
                 Finally, generate a taxonomic key.
                 """},
            {"role": "system","content":"""
                Generate the nested taxonomic key based on the provided morphological matrix. \n
                The process involves selecting a character to classify the species into groups. Repeat this classification within each subgroup until each group contains only one species.
                Information gain measures how much the uncertainty in the dataset is reduced after using a character for classification. It helps in selecting characters that minimize the entropy of the subset after classification, leading to better classification results.
                Please select the classification character for these group's species based on the morphological matrix and information gain methods.
                In the morphological matrix, 'Missing' and 'Not applicable' are invalid states. If a character has invalid states for the group being classified, it should be ignored.
                States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type and this multi-state characterization should not be confused with the single states within it (the state of '3' and '2 and 3' is different state, when you choose the character to based on the state to distinguish need to careful handle).The initial character should have no more than three state types.
                You need to calculate the information gain for each character and choose the highest information gain result. The higher the information gain result, the greater the contribution of the feature to the classification.
                After selecting the initial classification character and categorizing the species based on its state, repeat the process within each subgroup. For each subgroup, select the character with the highest information gain to further classify the species. Continue this process recursively until each group contains only one species.
                Now I will show you the morphological matrix. Please provide the classification character and the categorization of species based on its state. Then, continue to classify each subgroup recursively, showing the chosen character and categorization for each subgroup. Please present the result in a structured format, with each step clearly labeled.
                please don't show how you analysis and calculate, please show me the final result           
            """},
            {"role": "assistant",
             "content": """
                Understood. I will generate the nested taxonomic key based on the provided morphological matrix. Here is a summary of the steps I will follow:\n
                1. The matrix includes all species and their different states for each character.\n
                2. I will select a character to classify the species into groups and repeat this classification within each subgroup until each group contains only one species.\n
                3. I will use information gain to measure how much the uncertainty in the dataset is reduced after using a feature for classification. This helps in selecting features that minimize the entropy of the subset after classification, leading to better classification results.\n
                4. I will select the classification character for the group's species based on the morphological matrix and information gain methods.\n
                5. In the morphological matrix, 'Missing' and 'Not applicable' are considered invalid states. If a character has invalid states for the group being classified, it will be ignored.\n
                6. States are represented by numbers. For example, '2 and 3' means multiple states should be treated as a single state type, and This multi-state characterization should not be confused with the individual states(like '2', '3') within it (such as '3' and '2 and 3'  is the different state, these are two separate states, when i choose character to based on different state to distinguish the species). The classification character should have no more than three state types.\n
                7. I will use information gain to calculate all character and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification. \n
                8. The final result will provide only the initial classification character and the categorization of species based on its state. \n
                9. Don't need to show how the process about choose, only need to show the final result as nested structure, and i will store nested result in #character classify result# block
                Please provide the group morphological matrix data so that I can proceed with the classification.
             """},
        
            # When importing data types here, consider choosing to call different keys in the dictionary directly from what was previously stored in the dictionary, as well as combining it with the initial
            {"role": "user", "content": f"Here is the group information need to be classify, and here is the morphological matrix{group_matrix_str}"}
        ]
    response =  client.chat.completions.create(
            model="gpt-4o",
            messages=messages3,
            stop=None,
            temperature=0,
            max_tokens=1500,
            n = 1
    )
    result2 = response.choices[0].message.content
messages4 = [
                {"role": "system",
                 "content":"""
                     You are a helpful json format transfer.\n
                     You can express the nested structure as a JSON format based on the corresponding content
                     """},
                {"role": "system","content":"""
                      Please format the classification result as follows:

                # Final taxonomic key result JSON format #
                {
                    "Character": "CharacterX",
                    "States": {
                        "1": ["speciesA"],
                        "2": {
                        ".."
                            "Character": "CharacterY",
                            "States": {
                                "1": ["speciesB"],
                                "2": ["speciesC"]
                                ".."
                            }
                        }
                    }
                }
                ```
                Ensure that the response follows this format exactly.        
                """},
                {"role": "assistant",
                 "content": """
                    Understood. I'll convert the nested structure you gave me into JSON format and store it in # final result #
                    Please provide what you need to convert the format 
                 """},
                {"role": "user", "content": f"Here are the taxonomic results for the nested schema representation{result2}"}
            ]
response =  client.chat.completions.create(
            model="gpt-4o",
            messages=messages4,
            stop=None,
            temperature=0,
            max_tokens=1500,
            n = 1
    )
json_result2 = response.choices[0].message.content
print(json_result2)

```json
{
    "Character": "Character18",
    "States": {
        "1": ["Isoetes_lacustris"],
        "2": ["Isoetes_echinospora"],
        "3": ["Isoetes_histrix"]
    }
}
```


In [20]:
def extract_final_taxonomic_key(result):
    match = re.search(r'Taxonomic Key(.*)', result, re.DOTALL)
    if match:
        return match.group(1).strip()
    else:
        return None

# Extract the required part from each result
final_taxonomic_key1 = extract_final_taxonomic_key(result1)
final_taxonomic_key2 = extract_final_taxonomic_key(result2)

# Print the extracted results
print("Final Taxonomic Key 1:\n", final_taxonomic_key1)
print("Final Taxonomic Key 2:\n", final_taxonomic_key2)

Final Taxonomic Key 1:
 1. Character2
   - 1: Huperzia_selago
   - 2:
     1. Character9
        - 1:
          1. Character5
             - 2: Diphasiastrum_alpinum
             - 3: Diphasiastrum_complanatum
        - 2:
          1. Character20
             - 1: Lycopodiella_inundata
             - 2 and 3: Lycopodium_annotinum
             - 1 and 2 and 3: Lycopodium_clavatum
   - 3:
     1. Character11
        - 1 and 2:
          1. Character20
             - Missing: Selaginella_kraussiana
             - 3: Selaginella_selaginoides
Final Taxonomic Key 2:
 ##

1. Character18
   - 1: Isoetes_lacustris
   - 2: Isoetes_echinospora
   - 3: Isoetes_histrix


In [10]:
messages4 = [
                {"role": "system",
                 "content":"""
                     You are a helpful json format transfer.\n
                     You can express the nested structure as a JSON result based on the corresponding content
                     """},
                {"role": "system","content":"""
                      Please format the classification result as follows:
                ```
                # Final taxonomic key result JSON format #
                {
                    "Character": "CharacterX",
                    "States": {
                        "1": ["speciesA"],
                        "2": {
                        ".."
                            "Character": "CharacterY",
                            "States": {
                                "1": ["speciesB"],
                                "2": ["speciesC"]
                                ".."
                            }
                        }
                    }
                }
                ```
                Ensure that the response follows this format exactly.        
                """},
                {"role": "assistant",
                 "content": """
                    Understood. I'll convert the nested structure you gave me into JSON format and store it in # final result #
                    Please provide what you need to convert the format 
                 """},
                {"role": "user", "content": f"Here are the taxonomic results for the nested schema representation{result1}"}
            ]
response =  client.chat.completions.create(
            model="gpt-4o",
            messages=messages4,
            stop=None,
            temperature=0,
            max_tokens=1500,
            n = 1
    )
json_result1 = response.choices[0].message.content
print(json_result1)

```json
{
    "Character": "Character2",
    "States": {
        "1": ["Huperzia_selago"],
        "2": {
            "Character": "Character9",
            "States": {
                "1": {
                    "Character": "Character5",
                    "States": {
                        "2": ["Diphasiastrum_alpinum"],
                        "3": ["Diphasiastrum_complanatum"]
                    }
                },
                "2": {
                    "Character": "Character20",
                    "States": {
                        "1": ["Lycopodiella_inundata"],
                        "2 and 3": ["Lycopodium_annotinum"],
                        "1 and 2 and 3": ["Lycopodium_clavatum"]
                    }
                }
            }
        },
        "3": {
            "Character": "Character11",
            "States": {
                "1 and 2": {
                    "Character": "Character20",
                    "States": {
                        "Missing": 

In [13]:
messages4 = [
                {"role": "system",
                 "content":"""
                     You are a helpful json format transfer.\n
                     You can express the nested structure as a JSON format based on the corresponding content
                     """},
                {"role": "system","content":"""
                      Please format the classification result as follows:

                # Final taxonomic key result JSON format #
                {
                    "Character": "CharacterX",
                    "States": {
                        "1": ["speciesA"],
                        "2": {
                        ".."
                            "Character": "CharacterY",
                            "States": {
                                "1": ["speciesB"],
                                "2": ["speciesC"]
                                ".."
                            }
                        }
                    }
                }
                ```
                Ensure that the response follows this format exactly.        
                """},
                {"role": "assistant",
                 "content": """
                    Understood. I'll convert the nested structure you gave me into JSON format and store it in # final result #
                    Please provide what you need to convert the format 
                 """},
                {"role": "user", "content": f"Here are the taxonomic results for the nested schema representation{result2}"}
            ]
response =  client.chat.completions.create(
            model="gpt-4o",
            messages=messages4,
            stop=None,
            temperature=0,
            max_tokens=1500,
            n = 1
    )
json_result2 = response.choices[0].message.content
print(json_result2)

```json
{
    "Character": "Character18",
    "States": {
        "1": ["Isoetes_lacustris"],
        "2": ["Isoetes_echinospora"],
        "3": ["Isoetes_histrix"]
    }
}
```


In [12]:
print(json_result1)
print(json_result2)

```json
{
    "Character": "Character2",
    "States": {
        "1": ["Huperzia_selago"],
        "2": {
            "Character": "Character9",
            "States": {
                "1": {
                    "Character": "Character5",
                    "States": {
                        "2": ["Diphasiastrum_alpinum"],
                        "3": ["Diphasiastrum_complanatum"]
                    }
                },
                "2": {
                    "Character": "Character20",
                    "States": {
                        "1": ["Lycopodiella_inundata"],
                        "2 and 3": ["Lycopodium_annotinum"],
                        "1 and 2 and 3": ["Lycopodium_clavatum"]
                    }
                }
            }
        },
        "3": {
            "Character": "Character11",
            "States": {
                "1 and 2": {
                    "Character": "Character20",
                    "States": {
                        "Missing": 

In [18]:
# 初级结果
initial_result = {
    "Character": "Character1",
    "States": {
        "1": [
            "Diphasiastrum_alpinum",
            "Diphasiastrum_complanatum",
            "Huperzia_selago",
            "Lycopodiella_inundata",
            "Lycopodium_annotinum",
            "Lycopodium_clavatum",
            "Selaginella_kraussiana",
            "Selaginella_selaginoides"
        ],
        "2": [
            "Isoetes_echinospora",
            "Isoetes_histrix",
            "Isoetes_lacustris"
        ]
    }
}

# 次级结果1
final_taxonomic_key1 = {
    "Character": "Character2",
    "States": {
        "1": ["Huperzia_selago"],
        "2": {
            "Character": "Character9",
            "States": {
                "1": {
                    "Character": "Character5",
                    "States": {
                        "2": ["Diphasiastrum_alpinum"],
                        "3": ["Diphasiastrum_complanatum"]
                    }
                },
                "2": {
                    "Character": "Character20",
                    "States": {
                        "1": ["Lycopodiella_inundata"],
                        "2 and 3": ["Lycopodium_annotinum"],
                        "1 and 2 and 3": ["Lycopodium_clavatum"]
                    }
                }
            }
        },
        "3": {
            "Character": "Character11",
            "States": {
                "1 and 2": {
                    "Character": "Character20",
                    "States": {
                        "Missing": ["Selaginella_kraussiana"],
                        "3": ["Selaginella_selaginoides"]
                    }
                }
            }
        }
    }
}

# 次级结果2
final_taxonomic_key2 = {
    "Character": "Character18",
    "States": {
        "1": ["Isoetes_lacustris"],
        "2": ["Isoetes_echinospora"],
        "3": ["Isoetes_histrix"]
    }
}

# 结合初级结果和次级结果
def combine_results(initial, secondary, state_key):
    if not secondary:
        return

    initial_states = initial["States"][state_key]
    if isinstance(initial_states, list):
        initial["States"][state_key] = secondary
    else:
        for key, value in secondary["States"].items():
            if key not in initial_states["States"]:
                initial_states["States"][key] = value
            else:
                combine_results(initial_states, secondary, key)

# 初级结果和次级结果的结合
combine_results(initial_result, final_taxonomic_key1, "1")
combine_results(initial_result, final_taxonomic_key2, "2")

# 显示最终分类结果
def display_classification(result, indent=0):
    indent_space = " " * indent
    character = result.get("Character")
    states = result.get("States")

    classification = {}
    if character and states:
        classification["Character"] = character
        classification["States"] = {}
        print(f"{indent_space}1. **{character}:**")
        for state, species in states.items():
            if isinstance(species, list):
                print(f"{indent_space}   - State \"{state}\": {', '.join(species)}")
                classification["States"][state] = species
            elif isinstance(species, dict):
                print(f"{indent_space}   - State \"{state}\":")
                classification["States"][state] = display_classification(species, indent + 4)
    return classification

final_result = display_classification(initial_result)
# print(json.dumps(final_result, indent=2))

1. **Character1:**
   - State "1":
    1. **Character2:**
       - State "1": Huperzia_selago
       - State "2":
        1. **Character9:**
           - State "1":
            1. **Character5:**
               - State "2": Diphasiastrum_alpinum
               - State "3": Diphasiastrum_complanatum
           - State "2":
            1. **Character20:**
               - State "1": Lycopodiella_inundata
               - State "2 and 3": Lycopodium_annotinum
               - State "1 and 2 and 3": Lycopodium_clavatum
       - State "3":
        1. **Character11:**
           - State "1 and 2":
            1. **Character20:**
               - State "Missing": Selaginella_kraussiana
               - State "3": Selaginella_selaginoides
   - State "2":
    1. **Character18:**
       - State "1": Isoetes_lacustris
       - State "2": Isoetes_echinospora
       - State "3": Isoetes_histrix
