In [13]:
# Import necessary packages
import json  # For handling JSON data
from openai import OpenAI  # For interacting with OpenAI API
import os  # For interacting with the operating system, such as file paths
import re  # For regular expressions, useful for pattern matching in strings


In [14]:
# Input the API key and morphological matrix

# Initialize the OpenAI client with the API key from environment variables
client = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))

# Import the required dataset
with open("D:/桌面/taxonomy_primary_result/The_GPT-4_result/Dataset_3 (The Lycopodiales (Diphasiastrum, Huperzia, Isoetes, Lycopodium, Selaginella)) 4/Information gain methods/matrix_knowledge_graph.json",
        "r", encoding="utf-8") as file:
    matrix_data = json.load(file)

# Since the input is in JSON format, convert this JSON object to a string format when calling the API
matrix_data_str = json.dumps(matrix_data, ensure_ascii=False)

# Uncomment the following line to print the string format of the JSON data
# print(matrix_data_str)


In [15]:
# This is the prompt for the API input, using the client.chat.completions.create interface to conduct multi-turn conversations.
# By assigning different roles in the conversation (user, assistant, system), all input information is fully conveyed to the API model
messages = [
    # Set the system role to focus on taxonomy tasks.
    # Emphasize the system's understanding of morphological matrices, information gain, and the construction of classification keys in the system setup.
    {"role": "system",
     "content":
         """You are a helpful taxonomist assistant.
         You are skilled at calculating the correct information gain to choose the character that best divides species into even groups based on their states.
         Based on the selected character, classify the species into different groups according to their states.
         For each group with more than two species, continue selecting characters to further classify this group until each group only has one species.
         After multiple classifications, determine the final classification levels and record each classifying character and its state. 
         Finally, generate a taxonomic key.
         Please format the classification result as follows:
        ```
        {
            "Character": "Character1",
            "States": {
                "State 1": ["species1", "species2", ...],
                "State 2": ["species3", "species4", ...],
                "State ...": ["species5", "species6", ...]
            }
        }
        ```
        Ensure that the response follows this format exactly.
         """},
    # Input the main task request, construct the classification key, and emphasize the details needed for classification.
    # This includes: the main objective, information gain calculation, ignoring invalid states, clarifying multi-character states, guiding correct selection based on the significance of information gain, and standardizing the final output format for subsequent extraction of API results.
    {"role": "system",
     "content": """
                Generate the taxonomic key based on the provided morphological matrix. The matrix includes all species and their different states for each character.\n
                The process involves selecting a character to classify the species into groups. Repeat this classification within each subgroup until each group contains only one species.\n
                Information gain measures how much the uncertainty in the dataset is reduced after using a character for classification. It helps in selecting characters that minimize the entropy of the subset after classification, leading to better classification results.\n
                Please select the initial classification character for all species based on the morphological matrix and information gain methods.\n
                In the morphological matrix, 'Missing' and 'Not applicable' are invalid states. If a character has invalid states for the group being classified, it should be ignored.\n
                States are represented by numbers, and like the '1 and 2' means multiple states should be treated as a single state type.(such as '3' and '2 and 3'  is the different state, these are two separate states, when i choose character to based on different state to distinguish the species) \n
                you need to calculate the information gain for each character, and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification.\n
                Now I will show you the morphological matrix. Please provide only the initial classification character and the categorization of species based on its state, and you should singly show this called as # initial character classify result # 
            """},
    # Use the assistant to summarize and refine the prompt content for the API.
    # Through the conversation with the assistant, deepen the API understanding of the content to some extent, control the API response results, and standardize the output format of the API response.
    {"role": "assistant",
     "content": """
                Understood. I will generate the taxonomic key based on the provided morphological matrix. Here is a summary of the steps I will follow:\n
                1. The matrix includes all species and their different states for each character.\n
                2. I will select a character to classify the species into groups and repeat this classification within each subgroup until each group contains only one species.\n
                3. I will use information gain to measure how much the uncertainty in the dataset is reduced after using a feature for classification. This helps in selecting features that minimize the entropy of the subset after classification, leading to better classification results.\n
                4. I will select the initial classification character for all species based on the morphological matrix and information gain methods.\n
                5. In the morphological matrix, 'Missing' and 'Not applicable' are considered invalid states. If a character has invalid states for the group being classified, it will be ignored.\n
                6. States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type. (such as '1' and '1 and 2'  is the different state, these are two separate states, when i choose character to based on different state to distinguish the species)\n
                7. I will use information gain to calculate all character and choose the highest information gain result, The higher the Information Gain result, the greater the contribution of the feature to the classification.so i need to make sure the result is Average classification \n
                8. The final result will provide only the initial classification character and the categorization of species based on its state. \n
                9. Don't need to show how i calculate, only need to show the final result, and please show the final result in #initail character classify result# block, Don't have errors where the state and species don't match
                Please provide the morphological matrix data so that I can proceed with the initial classification.
            """
     },
    # Input the corresponding morphological matrix information for analysis.
    {"role": "user",
     "content": f"Here is morphological matrix:{matrix_data_str}"}
]
# Set the various parameters that need to be controlled when responding to the API. I think the temperature should be set to 0, and the max_tokens should either not be set or set to a small number. This is to save costs and because generating the corresponding results does not require a long output. If the token number is too large and no stop sequence is set, it can lead to complex redundancy and errors in the final result.
initial_character_info = client.chat.completions.create(
    model="gpt-4o",
    messages=messages,
    stop=None,
    max_tokens=1500,
    temperature=0,
    n=1)
# Store the API call response results as a file (when connecting subsequent distributed API call loops, consider storing in environment variables for continuous calls and modifications).
initial_response = initial_character_info.choices[0].message.content
# If need to as a whole pipeline to transfer the results we can ignore this print (but we can also use this print code to check ) print(initial_response)
print(initial_response)

# Initial Character Classify Result #
```
{
    "Character": "Character1",
    "States": {
        "1": ["Diphasiastrum_alpinum", "Diphasiastrum_complanatum", "Huperzia_selago", "Lycopodiella_inundata", "Lycopodium_annotinum", "Lycopodium_clavatum", "Selaginella_kraussiana", "Selaginella_selaginoides"],
        "2": ["Isoetes_echinospora", "Isoetes_histrix", "Isoetes_lacustris"]
    }
}
```


In [57]:
def parse_classification_result(result_text):
    classification = {"Character": None, "States": {}}
    character_match = re.search(r'"Character": "([^"]+)"', result_text)
    if character_match:
        classification["Character"] = character_match.group(1)
    state_sections = re.findall(r'"(\d+|[^"]+)":\s*\[(.*?)\]', result_text)
    for state, species_block in state_sections:
        species_list = re.findall(r'"([^"]+)"', species_block)
        classification["States"][state] = species_list
    return classification

parsed_initial_classification  =  parse_classification_result(initial_response)
print(parsed_initial_classification)
def generate_groups_from_classification(classification_result):
    """
    Generate groups from classification result.
    
    :param classification_result: Dictionary containing the classification result
    :return: List of tuples, where each tuple contains a state and a list of species
    """
    groups = []
    for state, species_list in classification_result["States"].items():
        groups.append((state, species_list))
    return groups
groups = generate_groups_from_classification(parsed_initial_classification)
print(groups)
print(groups[0])

{'Character': 'Character1', 'States': {'1': ['Diphasiastrum_alpinum', 'Diphasiastrum_complanatum', 'Huperzia_selago', 'Lycopodiella_inundata', 'Lycopodium_annotinum', 'Lycopodium_clavatum', 'Selaginella_kraussiana', 'Selaginella_selaginoides'], '2': ['Isoetes_echinospora', 'Isoetes_histrix', 'Isoetes_lacustris']}}
[('1', ['Diphasiastrum_alpinum', 'Diphasiastrum_complanatum', 'Huperzia_selago', 'Lycopodiella_inundata', 'Lycopodium_annotinum', 'Lycopodium_clavatum', 'Selaginella_kraussiana', 'Selaginella_selaginoides']), ('2', ['Isoetes_echinospora', 'Isoetes_histrix', 'Isoetes_lacustris'])]
('1', ['Diphasiastrum_alpinum', 'Diphasiastrum_complanatum', 'Huperzia_selago', 'Lycopodiella_inundata', 'Lycopodium_annotinum', 'Lycopodium_clavatum', 'Selaginella_kraussiana', 'Selaginella_selaginoides'])


In [28]:
# API call function for continued grouping for each subgroup
def classify_group(group_species):
    # Create a sub-matrix for the group of species
    group_matrix = {species: matrix_data[species] for species in group_species}
    group_matrix_str = json.dumps(group_matrix, ensure_ascii=False)
    
    # Define the messages for the API call
    messages2 = [
        {"role": "system",
         "content":
             """
             You are a helpful taxonomist assistant.
             You are skilled at calculating the correct information gain to choose the character that best divides species into even groups based on their states.
             Based on the selected character, classify the species into different groups according to their states.
             For each group with more than two species, continue selecting characters to further classify this group until each group only has one species.
             After multiple classifications, determine the final classification levels and record each classifying character and its state.
             Finally, generate a taxonomic key.
             """},
        {"role": "system",
         "content":
             """
             Generate the nested taxonomic key based on the provided morphological matrix.
             The process involves selecting a character to classify the species into groups. Repeat this classification within each subgroup until each group contains only one species.
             Information gain measures how much the uncertainty in the dataset is reduced after using a character for classification. It helps in selecting characters that minimize the entropy of the subset after classification, leading to better classification results.
             Please select the classification character for these group's species based on the morphological matrix and information gain methods.
             In the morphological matrix, 'Missing' and 'Not applicable' are invalid states. If a character has invalid states for the group being classified, it should be ignored.
             States are represented by numbers. For example, '1 and 2' means multiple states should be treated as a single state type and this multi-state characterization should not be confused with the single states within it (the state of '3' and '2 and 3' is different state, when you choose the character to based on the state to distinguish need to careful handle). The initial character should have no more than three state types.
             You need to calculate the information gain for each character and choose the highest information gain result. The higher the information gain result, the greater the contribution of the feature to the classification.
             After selecting the initial classification character and categorizing the species based on its state, repeat the process within each subgroup. For each subgroup, select the character with the highest information gain to further classify the species. Continue this process recursively until each group contains only one species.
             Now I will show you the morphological matrix. Please provide the classification character and the categorization of species based on its state. Then, continue to classify each subgroup recursively, showing the chosen character and categorization for each subgroup. Please present the result in a structured format, with each step clearly labeled.
             Please don't show how you analyze and calculate, please show me the final result.
             """},
        {"role": "assistant",
         "content":
             """
             Understood. I will generate the nested taxonomic key based on the provided morphological matrix. Here is a summary of the steps I will follow:
             1. The matrix includes all species and their different states for each character.
             2. I will select a character to classify the species into groups and repeat this classification within each subgroup until each group contains only one species.
             3. I will use information gain to measure how much the uncertainty in the dataset is reduced after using a feature for classification. This helps in selecting features that minimize the entropy of the subset after classification, leading to better classification results.
             4. I will select the classification character for the group's species based on the morphological matrix and information gain methods.
             5. In the morphological matrix, 'Missing' and 'Not applicable' are considered invalid states. If a character has invalid states for the group being classified, it will be ignored.
             6. States are represented by numbers. For example, '2 and 3' means multiple states should be treated as a single state type, and this multi-state characterization should not be confused with the individual states (like '2', '3') within it (such as '3' and '2 and 3' are different states, these are two separate states, when I choose a character based on different states to distinguish the species). The classification character should have no more than three state types.
             7. I will use information gain to calculate all characters and choose the highest information gain result. The higher the information gain result, the greater the contribution of the feature to the classification.
             8. The final result will provide only the initial classification character and the categorization of species based on its state.
             9. Don't need to show how the process about choosing, only need to show the final result as a nested structure, and I will store the result in #character classify result# block.
             Please provide the group morphological matrix data so that I can proceed with the classification.
             """},
        {"role": "user", "content": f"Here is the group information need to be classify and include the morphological matrix {group_matrix_str}"}
    ]
    
    # Make the API call to classify the group
    response = client.chat.completions.create(
        model="gpt-4o",
        messages=messages2,
        stop=None,
        temperature=0,
        max_tokens=1000,
        n=1
    )
    result = response.choices[0].message.content
    # print(f"API response for group {group_species}: {result}")
    
    # Define messages for formatting the response to JSON
    messages_JSON = [
        {"role": "system",
         "content":
             """
             You are a helpful JSON format converter.
             You can express the nested structure as a JSON result based on the corresponding content.
             """},
        {"role": "system",
         "content":
             """
             Please format the classification result as follows:
             ```
             # Final taxonomic key result JSON format #
             {
                 "Character": "CharacterX",
                 "States": {
                     "1": ["speciesA"],
                     "2": {
                         "Character": "CharacterY",
                         "States": {
                             "1": ["speciesB"],
                             "2": ["speciesC"]
                         }
                     }
                 }
             }
             ```
             Ensure that the response follows this format exactly.
             """},
        {"role": "assistant",
         "content":
             """
             Understood. I'll convert the nested structure you gave me into JSON format and store it in # final result #.
             Please provide what you need to convert the format.
             """},
        {"role": "user", "content": f"Here are the taxonomic results for the nested schema representation {result}"}
    ]
    
    # Make the API call to format the response as JSON
    response = client.chat.completions.create(
        model="gpt-4o",
        messages=messages_JSON,
        stop=None,
        temperature=0,
        max_tokens=1500,
        n=1
    )
    json_result = response.choices[0].message.content
    print(json_result)
    return json_result


In [30]:
# Function to clean and extract JSON string
def extract_json_string(json_string):
    # Find the positions of the start and end of the JSON object
    start = json_string.find('{')
    end = json_string.rfind('}') + 1
    
    # If both start and end positions are valid, extract and return the JSON string
    if start != -1 and end != -1:
        cleaned_string = json_string[start:end]
        return cleaned_string.strip()
    
    # If positions are not valid, return an empty string
    return ""


In [31]:
# Function to recursively classify groups of species
def recursive_classification(groups, final_classification, classification_results):
    while groups:  # Continue looping while the groups list is not empty
        state, current_group = groups.pop(0)  # Extract the first tuple from groups, containing state identifier and current group of species
        print(f"Processing group with state: {state}, species: {current_group}")  # Print the current group's state and species for debugging

        if len(current_group) == 1:  # If the current group has only one species
            final_classification[current_group[0]] = current_group  # Store this single-species group directly in final_classification dictionary
        else:
            classification_result = classify_group(current_group)  # Call the API function to classify the current group
            cleaned_classification_result = extract_json_string(classification_result)  # Clean the API classification result to extract the JSON string
            classification_results[state] = cleaned_classification_result  # Store the cleaned classification result in classification_results dictionary using state as the key

    return final_classification  # Return the final classification result


In [32]:
# Assume the variables have been initialized

# Dictionary to store the final classification where each species is classified individually
final_classification = {}

# Dictionary to store the API classification results for each state
classification_results = {}

# Print the initial state of groups and dictionaries for debugging purposes
print("Initial groups:", groups)
print("Initial final_classification:", final_classification)
print("Initial classification_results:", classification_results)

# Call the recursive_classification function to process the groups and store the results
final_classification = recursive_classification(groups, final_classification, classification_results)

# Print the final classification results
print("Final Classification:")
print(json.dumps(final_classification, indent=2, ensure_ascii=False))

# Print the classification results from the API calls
print("\nClassification Results:")
print(classification_results)



Initial groups: [('1', ['Diphasiastrum_alpinum', 'Diphasiastrum_complanatum', 'Huperzia_selago', 'Lycopodiella_inundata', 'Lycopodium_annotinum', 'Lycopodium_clavatum', 'Selaginella_kraussiana', 'Selaginella_selaginoides']), ('2', ['Isoetes_echinospora', 'Isoetes_histrix', 'Isoetes_lacustris'])]
Initial final_classification: {}
Initial classification_results: {}
Processing group with state: 1, species: ['Diphasiastrum_alpinum', 'Diphasiastrum_complanatum', 'Huperzia_selago', 'Lycopodiella_inundata', 'Lycopodium_annotinum', 'Lycopodium_clavatum', 'Selaginella_kraussiana', 'Selaginella_selaginoides']
```json
{
    "Character": "Character2",
    "States": {
        "1": ["Huperzia_selago"],
        "2": {
            "Character": "Character5",
            "States": {
                "1": {
                    "Character": "Character20",
                    "States": {
                        "1": ["Lycopodiella_inundata"],
                        "2 and 3": ["Lycopodium_annotinum"],
     

In [58]:
# Initial result parsed from the initial classification
initial_result = parsed_initial_classification

# Parse the final taxonomic keys from the classification results
final_taxonomic_key1 = json.loads(classification_results['1'])
final_taxonomic_key2 = json.loads(classification_results['2'])

# Function to combine the initial and secondary classification results
def combine_results(initial, secondary, state_key):
    if not secondary:
        return

    initial_states = initial["States"][state_key]
    if isinstance(initial_states, list):
        initial["States"][state_key] = secondary
    else:
        for key, value in secondary["States"].items():
            if key not in initial_states["States"]:
                initial_states["States"][key] = value
            else:
                combine_results(initial_states, secondary, key)

# Combine the initial and secondary classification results
combine_results(initial_result, final_taxonomic_key1, "1")
combine_results(initial_result, final_taxonomic_key2, "2")

# Function to display the final classification result
def display_classification(result, indent=0):
    indent_space = " " * indent
    character = result.get("Character")
    states = result.get("States")

    classification = {}
    if character and states:
        classification["Character"] = character
        classification["States"] = {}
        print(f"{indent_space}1. **{character}:**")
        for state, species in states.items():
            if isinstance(species, list):
                print(f"{indent_space}   - State \"{state}\": {', '.join(species)}")
                classification["States"][state] = species
            elif isinstance(species, dict):
                print(f"{indent_space}   - State \"{state}\":")
                classification["States"][state] = display_classification(species, indent + 4)
    return classification

# Display the final classification result
final_result = display_classification(initial_result)
# print(json.dumps(final_result, indent=2))


1. **Character1:**
   - State "1":
    1. **Character2:**
       - State "1": Huperzia_selago
       - State "2":
        1. **Character5:**
           - State "1":
            1. **Character20:**
               - State "1": Lycopodiella_inundata
               - State "2 and 3": Lycopodium_annotinum
               - State "1 and 2 and 3": Lycopodium_clavatum
           - State "2":
            1. **Character3:**
               - State "1": Diphasiastrum_alpinum
               - State "2": Diphasiastrum_complanatum
       - State "3":
        1. **Character4:**
           - State "1": Selaginella_kraussiana
           - State "2": Selaginella_selaginoides
   - State "2":
    1. **Character20:**
       - State "1": Isoetes_histrix
       - State "1 and 2 and 3": Isoetes_echinospora
       - State "2 and 3": Isoetes_lacustris


In [67]:
# Example initial result
initial_result = parsed_initial_classification

# Parse the API response JSON strings
parsed_classification_results = {key: json.loads(value) for key, value in classification_results.items()}

# Function to combine the initial and secondary classification results
def combine_results(initial, secondary, state_key):
    if not secondary:
        return

    initial_states = initial["States"].get(state_key)
    if initial_states is None:
        initial["States"][state_key] = secondary
        return

    if isinstance(initial_states, list):
        if isinstance(secondary, list):
            initial["States"][state_key] = list(set(initial_states + secondary))  # Merge two lists and remove duplicates
        else:
            initial["States"][state_key] = secondary
    elif isinstance(initial_states, dict):
        if isinstance(secondary, dict):
            for key, value in secondary["States"].items():
                if key not in initial_states:
                    initial_states[key] = value
                else:
                    combine_results(initial_states, value, key)
        else:
            raise ValueError(f"Conflicting types for key {state_key}: {type(initial_states)} vs {type(secondary)}")
    else:
        raise ValueError(f"Unexpected type for initial states: {type(initial_states)}")

# Dynamically combine all secondary classification results
for state_key, secondary in parsed_classification_results.items():
    combine_results(initial_result, secondary, state_key)

# Function to display the final classification result
def display_classification(result, indent=0):
    indent_space = " " * indent
    character = result.get("Character")
    states = result.get("States")

    classification = {}
    if character and states:
        classification["Character"] = character
        classification["States"] = {}
        print(f"{indent_space}1. **{character}:**")
        for state, species in states.items():
            if isinstance(species, list):
                print(f"{indent_space}   - State \"{state}\": {', '.join(species)}")
                classification["States"][state] = species
            elif isinstance(species, dict):
                print(f"{indent_space}   - State \"{state}\":")
                classification["States"][state] = display_classification(species, indent + 4)
    return classification

# Display the final classification result
final_result = display_classification(initial_result)
print("\nFinal Result JSON:")
# print(json.dumps(final_result, indent=2))


1. **Character1:**
   - State "1":
    1. **Character2:**
       - State "1": Huperzia_selago
       - State "2":
        1. **Character5:**
           - State "1":
            1. **Character20:**
               - State "1": Lycopodiella_inundata
               - State "2 and 3": Lycopodium_annotinum
               - State "1 and 2 and 3": Lycopodium_clavatum
           - State "2":
            1. **Character3:**
               - State "1": Diphasiastrum_alpinum
               - State "2": Diphasiastrum_complanatum
       - State "3":
        1. **Character4:**
           - State "1": Selaginella_kraussiana
           - State "2": Selaginella_selaginoides
   - State "2":
    1. **Character20:**
       - State "1": Isoetes_histrix
       - State "1 and 2 and 3": Isoetes_echinospora
       - State "2 and 3": Isoetes_lacustris
   - State "State1":
    1. **Character2:**
       - State "StateA": Huperzia_selago
       - State "StateB":
        1. **Character5:**
           - State "StateC"