In [15]:
# Start with imports - ask ChatGPT to explain any package that you don't know

import os
import json
from dotenv import load_dotenv
from openai import OpenAI
from anthropic import Anthropic
from IPython.display import Markdown, display
import pandas as pd

In [16]:
# Always remember to do this!
load_dotenv(override=True)

True

In [3]:
# Print the key prefixes to help with any debugging

openai_api_key = os.getenv('OPENAI_API_KEY')
anthropic_api_key = os.getenv('ANTHROPIC_API_KEY')
google_api_key = os.getenv('GOOGLE_API_KEY')
deepseek_api_key = os.getenv('DEEPSEEK_API_KEY')
groq_api_key = os.getenv('GROQ_API_KEY')

if openai_api_key:
    print(f"OpenAI API Key exists and begins {openai_api_key[:8]}")
else:
    print("OpenAI API Key not set")
    
if anthropic_api_key:
    print(f"Anthropic API Key exists and begins {anthropic_api_key[:7]}")
else:
    print("Anthropic API Key not set (and this is optional)")

if google_api_key:
    print(f"Google API Key exists and begins {google_api_key[:2]}")
else:
    print("Google API Key not set (and this is optional)")

if deepseek_api_key:
    print(f"DeepSeek API Key exists and begins {deepseek_api_key[:3]}")
else:
    print("DeepSeek API Key not set (and this is optional)")

if groq_api_key:
    print(f"Groq API Key exists and begins {groq_api_key[:4]}")
else:
    print("Groq API Key not set (and this is optional)")

OpenAI API Key exists and begins sk-proj-
Anthropic API Key exists and begins sk-ant-
Google API Key exists and begins AI
DeepSeek API Key exists and begins sk-
Groq API Key exists and begins gsk_


In [17]:
request = "Please come up with a challenging, nuanced question that I can ask a number of LLMs to evaluate their intelligence. "
request += "Answer only with the question, no explanation."
messages = [{"role": "user", "content": request}]

In [18]:
messages

[{'role': 'user',
  'content': 'Please come up with a challenging, nuanced question that I can ask a number of LLMs to evaluate their intelligence. Answer only with the question, no explanation.'}]

In [19]:
openai = OpenAI()
response = openai.chat.completions.create(
    model="gpt-4o-mini",
    messages=messages,
)
question = response.choices[0].message.content
print(question)

How would you approach resolving a moral dilemma where the decision to save one person over another would result in a significant societal impact, and how would you justify your choice?


In [20]:
competitors = []
answers = []
messages = [{"role": "user", "content": question}]

In [21]:
# The API we know well

model_name = "gpt-4o-mini"

response = openai.chat.completions.create(model=model_name, messages=messages)
answer = response.choices[0].message.content

display(Markdown(answer))
competitors.append(model_name)
answers.append(answer)

Resolving a moral dilemma where saving one person over another has significant societal impacts requires a thoughtful and structured approach. Here’s how I would tackle it:

1. **Identify the Stakeholders**: Begin by understanding who the individuals are in the scenario, their roles in society, and how their survival or loss would affect others. This includes considering their contributions to society, their relationships, and what they represent in broader ethical, social, or economic contexts.

2. **Assess the Consequences**: Evaluate the potential outcomes of saving each individual. Consider both immediate consequences and long-term societal implications. This might involve weighing factors such as:

   - The potential impact on communities, families, and social stability.
   - Long-term repercussions for fields such as healthcare, education, or economy.
   - The symbolic significance of saving one person over another (e.g., a leader, an innovator, a marginalized individual).

3. **Consider Ethical Frameworks**: Employ various ethical theories to guide the decision-making process. For instance:
   - **Utilitarianism**: Focus on the greatest good for the greatest number. Evaluate which choice maximizes overall happiness and well-being.
   - **Deontological Ethics**: Consider moral duties and rights. Reflect on whether one choice upholds moral principles better than the other.
   - **Virtue Ethics**: Assess what a person of good character would do in this situation. This could help in making a more personal and emotionally resonant choice.

4. **Reflect on Personal Values**: Consider your own values and biases. Reflect on why you feel drawn to one choice over another. Acknowledge any emotional weight that accompanies the decision; personal biases should be considered but not allowed to dominate.

5. **Consult Diverse Perspectives**: If possible, engage with others for their insights. Diverse viewpoints can reveal blind spots and provide a richer understanding of potential consequences.

6. **Make a Decision**: After evaluating the factors above, make a decision based on the analysis. This choice should aim to balance ethical considerations with practical implications.

7. **Justification of the Choice**: Clearly articulate the reasons behind the decision. This justification should include:
   - A thorough explanation of the reasoning process and ethical frameworks used.
   - Acknowledgment of the potential limitations and flaws in the reasoning, reflecting humility about the difficult nature of the choice.
   - A consideration of what could be done to mitigate harm to the individual not chosen or to society overall.

8. **Plan for Aftermath**: Consider how to address the consequences of the decision, both for the individual who was not saved and for society. This might involve support systems or initiatives that could help address the impact of the choice.

Ultimately, moral dilemmas often involve complex and deeply felt consequences, and while there may not be a 'right' answer, a reasoned and compassionate approach can help navigate the challenges involved.

In [22]:
# Anthropic has a slightly different API, and Max Tokens is required

model_name = "claude-3-7-sonnet-latest"

claude = Anthropic()
response = claude.messages.create(model=model_name, messages=messages, max_tokens=1000)
answer = response.content[0].text

display(Markdown(answer))
competitors.append(model_name)
answers.append(answer)

# Approaching a Complex Moral Dilemma

This is a profound ethical question that touches on several philosophical traditions. I would approach it through multiple lenses:

First, I'd carefully consider:
- The specific circumstances and available information
- Whether there are creative solutions that might save both
- My role and responsibilities in the situation

If forced to choose, I would weigh several factors:
- Immediate consequences for all involved
- Long-term societal impacts
- Whether special obligations exist to either person
- If either person bears special responsibility for the situation

Different ethical frameworks offer contrasting guidance:
- Utilitarians might prioritize maximizing overall good
- Deontologists might focus on duties and rights
- Virtue ethicists would consider what a person of good character would do

In justifying my choice, I would acknowledge the moral weight of choosing at all, recognize the inherent value of both lives, and explain my reasoning transparently while accepting that reasonable people might disagree.

What specific context did you have in mind for this dilemma?

In [23]:
gemini = OpenAI(api_key=google_api_key, base_url="https://generativelanguage.googleapis.com/v1beta/openai/")
model_name = "gemini-2.0-flash"

response = gemini.chat.completions.create(model=model_name, messages=messages)
answer = response.choices[0].message.content

display(Markdown(answer))
competitors.append(model_name)
answers.append(answer)

This is a profoundly difficult ethical question, and there's no easy or universally accepted answer. Here's my approach, broken down into steps, along with potential justifications:

**1.  Identify the Core Dilemma & Gather Information:**

*   **Clarify the Stakes:**  What are the potential outcomes of each decision?  What are the potential societal impacts (positive and negative) of saving each person?  How significant are these impacts?
*   **Gather Relevant Facts:**  Who are the individuals involved? What are their backgrounds, contributions, skills, potential future contributions, vulnerabilities, dependencies, and responsibilities? What is the nature of the societal impact they represent?  Are there pre-existing laws, regulations, or ethical guidelines that apply? Are there any experts to consult (e.g., ethicists, economists, social scientists)?
*   **Consider Alternative Solutions:** Is there any way to save both individuals, mitigate the negative consequences of either choice, or find a third option?  Exploring alternatives is crucial before committing to a difficult choice.

**2.  Ethical Frameworks & Prioritization:**

*   **Identify Conflicting Values:**  This scenario will likely involve competing values such as:
    *   **Justice/Fairness:** Does one person deserve to be saved more than the other based on past actions or contributions?
    *   **Utility/Consequentialism:** Which decision produces the greatest good for the greatest number of people?  This requires carefully weighing potential benefits and harms.
    *   **Deontology/Duty:**  Are there inherent rights or duties that must be upheld, regardless of the consequences? (e.g., a duty to treat all lives as equally valuable).
    *   **Care Ethics:**  Emphasizes relationships, empathy, and the interconnectedness of individuals. Focuses on minimizing harm and maintaining social bonds.
*   **Apply Ethical Frameworks:**  Each framework would lead to a different conclusion:
    *   **Utilitarianism:**  Choose the option that maximizes overall well-being for society, even if it means sacrificing one individual. This might involve a cost-benefit analysis.
    *   **Deontology:**  Focus on upholding moral duties.  If there's a duty to treat all lives equally, a lottery might be the fairest approach. If there's a duty to protect the most vulnerable, prioritize them.
    *   **Egalitarianism:** Prioritize the individual that is the worse off.
*   **Assess Consequences & Stakeholder Impacts:**  Create a detailed list of potential consequences, both short-term and long-term, for each decision.  Consider the impact on various stakeholders (families, communities, society as a whole).

**3.  Decision Making and Justification:**

*   **Transparency and Deliberation:** The decision-making process should be transparent and involve relevant stakeholders (if possible).  Consulting with experts and affected parties can help ensure all perspectives are considered.
*   **Justification:** Clearly articulate the reasons for your choice, based on the chosen ethical framework and the specific circumstances of the dilemma.
*   **Acknowledge the Harm:** Acknowledge the inherent tragedy of the situation and the unavoidable harm that will result from either decision. Express empathy for those who will be affected.
*   **Document the Reasoning:**  Keep a detailed record of the decision-making process, including the information gathered, the ethical frameworks considered, the rationale for the final choice, and any dissenting opinions.  This documentation is crucial for accountability and future review.

**Examples of Justifications (Based on Different Frameworks):**

*   **Utilitarian Justification:** "While it is a tragic decision, saving Person A over Person B will result in the greatest overall benefit to society. Person A's work in [field] is crucial for [specific societal benefit], which will impact [number] of people positively. While Person B's life is equally valuable, the societal consequences of losing Person A are significantly greater." (This justification needs to be backed by concrete evidence and a well-reasoned cost-benefit analysis.)
*   **Deontological Justification (Equal Chance):** "Because all lives are inherently equal, and there are no overriding factors that justify prioritizing one over the other, a lottery system will be used to determine who is saved. This approach, while emotionally difficult, ensures fairness and avoids arbitrary discrimination."
*   **Care Ethics Justification (Focus on Vulnerability):** "Person A is currently dependent on others for survival and is vulnerable to suffering, therefore saving Person A would allow them to continue to live and recover, as well as benefit their loved ones. While Person B is also valuable, the potential of losing Person A would result in greater suffering and broken relationships."

**Important Considerations:**

*   **The Slippery Slope:** Be aware of the potential for this decision to set a precedent. Could this justify similar decisions in the future, even in less compelling circumstances?
*   **Unintended Consequences:**  Ethical decision-making is rarely perfect. Be prepared for unintended consequences and be willing to adapt your approach if necessary.
*   **Public Perception:**  Regardless of the decision, there will likely be public scrutiny and criticism. Be prepared to defend your choice and address concerns.
*   **The Role of Emotions:** While it's important to be rational, emotions cannot be entirely ignored. Acknowledge the emotional impact of the decision on yourself and others.

**In conclusion, resolving a moral dilemma of this magnitude requires a rigorous, transparent, and ethically informed process. The justification for any decision must be grounded in a well-defined ethical framework, supported by evidence, and communicated with empathy and humility.** It's a process of striving to minimize harm and maximize good, while acknowledging the inherent limitations of human decision-making in the face of profound moral complexity.


In [24]:
deepseek = OpenAI(api_key=deepseek_api_key, base_url="https://api.deepseek.com/v1")
model_name = "deepseek-chat"

response = deepseek.chat.completions.create(model=model_name, messages=messages)
answer = response.choices[0].message.content

display(Markdown(answer))
competitors.append(model_name)
answers.append(answer)

Resolving a moral dilemma where saving one person over another has significant societal impact requires a structured ethical framework to justify the decision. Here’s how I would approach it:

### 1. **Define the Dilemma Clearly**
   - Identify the two individuals (or groups) at stake and the consequences of saving each.
   - Clarify the societal impact (e.g., loss of a key leader, erosion of trust, or long-term harm to a community).

### 2. **Apply Ethical Frameworks**
   I would evaluate the dilemma through multiple lenses:
   - **Utilitarianism**: Choose the option that maximizes overall well-being or minimizes harm to society. If one person’s survival leads to greater collective good (e.g., a doctor who can save many lives), this framework would favor saving them.
   - **Deontological Ethics**: Focus on moral duties (e.g., "do not kill"). If one choice violates a fundamental principle (e.g., fairness or equality), it may be rejected regardless of consequences.
   - **Virtue Ethics**: Consider what a morally exemplary person would do (e.g., compassion, wisdom). This might prioritize the most vulnerable or deserving individual.
   - **Rights-Based Ethics**: Prioritize the person whose fundamental rights (e.g., life, liberty) are most at stake or whose loss would undermine societal justice.

### 3. **Assess Societal Impact**
   - Would saving one person set a dangerous precedent (e.g., valuing some lives over others)? 
   - Does one choice align with long-term societal values (e.g., equity, stability)?
   - Are there indirect effects (e.g., loss of public trust if the decision appears arbitrary or biased)?

### 4. **Mitigate Harm**
   - Explore alternatives to avoid the dilemma entirely (e.g., finding a way to save both).
   - If unavoidable, seek compensatory measures (e.g., supporting the affected community post-decision).

### 5. **Justify the Choice**
   The final decision would rest on which framework best aligns with the context, justified transparently. For example:
   - **Utilitarian Justification**: "Saving Person A (a scientist close to a vaccine breakthrough) over Person B (an ordinary citizen) could save thousands of lives, justifying the greater societal benefit."
   - **Deontological Justification**: "Choosing by lottery ensures fairness, upholding the principle that all lives have equal inherent value, even if the societal outcome is less optimal."
   - **Rights-Based Justification**: "Person B is a marginalized individual; saving them rectifies systemic inequities, reinforcing societal justice."

### 6. **Acknowledge Moral Residue**
   Recognize that no choice may feel fully "right," and the emotional or moral toll on decision-makers is part of the process.

### Example Scenario:
*Dilemma*: A hospital must choose between saving a renowned activist (whose work benefits millions) or a child (symbolizing innocence but with no broader impact).  
*Approach*:  
- Utilitarianism favors the activist (greater societal good).  
- Virtue ethics might favor the child (compassion for the vulnerable).  
*Justification*: "While saving the activist aligns with measurable societal benefit, prioritizing the child upholds the intrinsic value of every life and avoids instrumentalizing human worth. We choose the child to preserve moral integrity, even at a utilitarian cost."

Ultimately, the "right" answer depends on which ethical priorities (outcomes, principles, or virtues) are deemed most salient in the context. Transparency about the reasoning is key to societal acceptance.

In [25]:
groq = OpenAI(api_key=groq_api_key, base_url="https://api.groq.com/openai/v1")
model_name = "llama-3.3-70b-versatile"

response = groq.chat.completions.create(model=model_name, messages=messages)
answer = response.choices[0].message.content

display(Markdown(answer))
competitors.append(model_name)
answers.append(answer)

Resolving a moral dilemma involving saving one person over another, with significant societal impact, requires careful consideration of various factors and a nuanced approach. Here's a step-by-step framework to help guide the decision-making process:

1. **Clearly define the situation**: Understand the context, the individuals involved, and the potential consequences of each possible decision. Identify the moral principles at stake, such as the value of human life, fairness, and justice.
2. **Gather relevant information**: Collect data on the individuals involved, including their backgrounds, skills, and potential contributions to society. Consider the short-term and long-term implications of saving one person over the other.
3. **Evaluate the moral principles**: Weigh the importance of different moral principles, such as:
	* The value of human life: Is one life more valuable than the other?
	* Fairness: Would saving one person be fair to the other, considering their circumstances and potential impact on society?
	* Justice: Would saving one person promote justice, or would it perpetuate inequality?
	* Utilitarianism: Would saving one person lead to the greatest good for the greatest number of people?
4. **Consider the societal impact**: Think about the potential consequences of saving one person over the other on the community, society, and future generations. Ask questions like:
	* Will saving one person lead to significant benefits or advancements in a particular field?
	* Will the death of one person have a profound impact on their family, community, or society as a whole?
	* Are there any potential long-term consequences of saving one person over the other, such as setting a precedent or creating a moral obligation?
5. **Apply ethical frameworks**: Utilize ethical frameworks, such as:
	* Consequentialism: Focus on the outcomes of each possible decision.
	* Deontology: Consider the moral rules and duties involved, such as the duty to save lives or protect the vulnerable.
	* Virtue ethics: Evaluate the decision based on the character and virtues of the individuals involved.
6. **Make a decision**: Based on the analysis, make a decision that aligns with your moral principles and values. Be prepared to justify your choice, considering the potential criticism and alternative perspectives.
7. **Justify the decision**: Provide a clear and compelling justification for your choice, addressing the following:
	* The moral principles and values that guided your decision.
	* The potential consequences of your choice, including the societal impact.
	* The potential criticisms and alternative perspectives, and how you addressed them.
	* The potential long-term implications of your decision, and how it may set a precedent or create a moral obligation.

Example justification:

"In this situation, I have decided to save Person A, a renowned scientist on the verge of discovering a cure for a deadly disease. While both individuals have inherent value, I believe that saving Person A would lead to a significant positive impact on society, potentially saving countless lives and improving the quality of life for many people. This decision aligns with the moral principle of utilitarianism, as it prioritizes the greatest good for the greatest number of people.

I understand that this decision may be criticized for prioritizing one life over another, potentially perpetuating inequality. However, I believe that the potential benefits of saving Person A outweigh the potential costs, and that this decision is fair and just considering the circumstances.

In the long term, this decision may set a precedent for prioritizing individuals with significant societal contributions over others. However, I believe that this precedent is justified, as it promotes the greater good and aligns with the moral principle of justice. Ultimately, this decision requires careful consideration of the moral principles and values at stake, and I am prepared to defend my choice and its potential consequences."

Remember that resolving moral dilemmas is often a complex and subjective process, and there may be no one "right" answer. The goal is to make a thoughtful and well-justified decision that aligns with your moral principles and values.

In [26]:
ollama = OpenAI(base_url='http://localhost:11434/v1', api_key='ollama')
#model_name = "llama3.2"
model_name ="llama3"

response = ollama.chat.completions.create(model=model_name, messages=messages)
answer = response.choices[0].message.content

display(Markdown(answer))
competitors.append(model_name)
answers.append(answer)

When faced with a moral dilemma involving a difficult choice between saving one person over another, I would carefully consider the ethical implications of my decision. Here's a step-by-step approach:

1. **Clarify the situation**: Ensure you understand all the facts and circumstances surrounding the dilemma. Consider seeking additional information or clarifying any uncertainties.
2. **Reflect on personal values and principles**: Consider your own moral framework, including beliefs about fairness, justice, and human life. What are your core values? How do they relate to this specific situation?
3. **Evaluate potential consequences**: Weigh the potential impact of your decision on various stakeholders, including individuals directly affected, society as a whole, and future generations.
4. **Consider alternative options**: Explore other possible solutions that might achieve similar outcomes or mitigate negative consequences.
5. **Analyze moral principles**: Apply relevant moral principles to your decision-making process. For example:
	* **Utilitarianism**: Act in the way that maximizes overall happiness or well-being, even if it means sacrificing some individuals.
	* **Self-sacrifice**: Prioritize saving those who would be most affected by the outcome, potentially at personal cost.
	* **Equity and fairness**: Strive for a fair distribution of benefits and burdens among relevant parties.
6. **Justify your choice**: Once you've made a decision, prepare a thoughtful justification that addresses potential criticisms:
	* **Acknowledge the complexity**: Recognize the multifaceted nature of the dilemma and acknowledge the difficulties involved in making a perfect decision.
	* **Highlight key considerations**: Emphasize the factors that contributed to your choice, such as the potential consequences for individual lives or the broader societal impact.
	* **Offer alternative solutions**: Suggest other approaches that might have achieved similar outcomes, but which would not have been feasible or effective in this situation.

Example justification:

In this hypothetical scenario, a devastating flood is threatening to destroy two communities: one with mostly young families and another with primarily elderly residents. You are a leader tasked with allocating limited resources to rescue those trapped. Due to logistical constraints, you can only save one community. After careful consideration of the potential consequences and moral principles, I have decided to prioritize saving the community with elderly residents.

Justification:

* The elderly community is more vulnerable to the flood's effects, as older adults may be less physically capable of escaping on their own.
* Prioritizing this community serves a higher public interest by protecting those who are most reliant on social support structures and may have limited access to resources in the aftermath.
* While saving the young families would also be important, it is not the only way to support these individuals. Alternative approaches, such as providing aid and assistance after the crisis has passed, could still benefit this community.

Remember that justifications will vary depending on individual perspectives, cultural context, and specific circumstances. The key is to thoughtfully consider the moral implications of your decision and be prepared to defend it in a way that respects the values and principles guiding your choice.

In [27]:
print(competitors)
print(answers)


['gpt-4o-mini', 'claude-3-7-sonnet-latest', 'gemini-2.0-flash', 'deepseek-chat', 'llama-3.3-70b-versatile', 'llama3']
["Resolving a moral dilemma where saving one person over another has significant societal impacts requires a thoughtful and structured approach. Here’s how I would tackle it:\n\n1. **Identify the Stakeholders**: Begin by understanding who the individuals are in the scenario, their roles in society, and how their survival or loss would affect others. This includes considering their contributions to society, their relationships, and what they represent in broader ethical, social, or economic contexts.\n\n2. **Assess the Consequences**: Evaluate the potential outcomes of saving each individual. Consider both immediate consequences and long-term societal implications. This might involve weighing factors such as:\n\n   - The potential impact on communities, families, and social stability.\n   - Long-term repercussions for fields such as healthcare, education, or economy.\n  

In [28]:
# It's nice to know how to use "zip"
for competitor, answer in zip(competitors, answers):
    print(f"Competitor: {competitor}\n\n{answer}")

Competitor: gpt-4o-mini

Resolving a moral dilemma where saving one person over another has significant societal impacts requires a thoughtful and structured approach. Here’s how I would tackle it:

1. **Identify the Stakeholders**: Begin by understanding who the individuals are in the scenario, their roles in society, and how their survival or loss would affect others. This includes considering their contributions to society, their relationships, and what they represent in broader ethical, social, or economic contexts.

2. **Assess the Consequences**: Evaluate the potential outcomes of saving each individual. Consider both immediate consequences and long-term societal implications. This might involve weighing factors such as:

   - The potential impact on communities, families, and social stability.
   - Long-term repercussions for fields such as healthcare, education, or economy.
   - The symbolic significance of saving one person over another (e.g., a leader, an innovator, a margina

In [29]:
together = ""
for index, answer in enumerate(answers):
    together += f"# Response from competitor {index+1}\n\n"
    together += answer + "\n\n"

In [30]:
print(together)

# Response from competitor 1

Resolving a moral dilemma where saving one person over another has significant societal impacts requires a thoughtful and structured approach. Here’s how I would tackle it:

1. **Identify the Stakeholders**: Begin by understanding who the individuals are in the scenario, their roles in society, and how their survival or loss would affect others. This includes considering their contributions to society, their relationships, and what they represent in broader ethical, social, or economic contexts.

2. **Assess the Consequences**: Evaluate the potential outcomes of saving each individual. Consider both immediate consequences and long-term societal implications. This might involve weighing factors such as:

   - The potential impact on communities, families, and social stability.
   - Long-term repercussions for fields such as healthcare, education, or economy.
   - The symbolic significance of saving one person over another (e.g., a leader, an innovator, a ma

In [31]:
judge = f"""You are judging a competition between {len(competitors)} competitors.
Each model has been given this question:

{question}

Your job is to evaluate each response for clarity and strength of argument, and rank them in order of best to worst.
Respond with JSON, and only JSON, with the following format:
{{"results": ["best competitor number", "second best competitor number", "third best competitor number", ...]}}

Here are the responses from each competitor:

{together}

Now respond with the JSON with the ranked order of the competitors, nothing else. Do not include markdown formatting or code blocks."""


In [32]:
print(judge)

You are judging a competition between 6 competitors.
Each model has been given this question:

How would you approach resolving a moral dilemma where the decision to save one person over another would result in a significant societal impact, and how would you justify your choice?

Your job is to evaluate each response for clarity and strength of argument, and rank them in order of best to worst.
Respond with JSON, and only JSON, with the following format:
{"results": ["best competitor number", "second best competitor number", "third best competitor number", ...]}

Here are the responses from each competitor:

# Response from competitor 1

Resolving a moral dilemma where saving one person over another has significant societal impacts requires a thoughtful and structured approach. Here’s how I would tackle it:

1. **Identify the Stakeholders**: Begin by understanding who the individuals are in the scenario, their roles in society, and how their survival or loss would affect others. This 

In [33]:
judge_messages = [{"role": "user", "content": judge}]

In [34]:
# Judgement time!

openai = OpenAI()
response = openai.chat.completions.create(
    model="o3-mini",
    messages=judge_messages,
)
judge_o3_mini = response.choices[0].message.content
print(judge_o3_mini)


{"results": ["3", "1", "4", "5", "6", "2"]}


## Create a dictionary to cobble all judges into a df

In [35]:
judge_dict = dict()

In [36]:
# OK let's turn this into results!

results_dict = json.loads(judge_o3_mini)
ranks = results_dict["results"]
for index, result in enumerate(ranks):
    competitor = competitors[int(result)-1]
    print(f"Rank {index+1}: {competitor}")

Rank 1: gemini-2.0-flash
Rank 2: gpt-4o-mini
Rank 3: deepseek-chat
Rank 4: llama-3.3-70b-versatile
Rank 5: llama3
Rank 6: claude-3-7-sonnet-latest


In [37]:
judge_dict['judge_03_mini'] = [competitors[int(i)-1] for i in ranks]

In [38]:
model_name = "claude-3-7-sonnet-latest"

claude = Anthropic()
response = claude.messages.create(model=model_name, messages=judge_messages, max_tokens=1000)
judge_anthropic = response.content[0].text

print(judge_anthropic)


{"results": ["3", "1", "4", "5", "2", "6"]}


In [39]:
results_dict = json.loads(judge_anthropic)
ranks = results_dict["results"]
for index, result in enumerate(ranks):
    competitor = competitors[int(result)-1]
    print(f"Rank {index+1}: {competitor}")

Rank 1: gemini-2.0-flash
Rank 2: gpt-4o-mini
Rank 3: deepseek-chat
Rank 4: llama-3.3-70b-versatile
Rank 5: claude-3-7-sonnet-latest
Rank 6: llama3


In [40]:
judge_dict['judge_anthropic'] = [competitors[int(i)-1] for i in ranks]

In [41]:
gemini = OpenAI(api_key=google_api_key, base_url="https://generativelanguage.googleapis.com/v1beta/openai/")
model_name = "gemini-2.0-flash"

response = gemini.chat.completions.create(model=model_name, messages=judge_messages)
judge_gemini = response.choices[0].message.content

print(judge_gemini)

{"results": ["3", "1", "4", "5", "6", "2"]}



In [42]:
results_dict = json.loads(judge_gemini)
ranks = results_dict["results"]
for index, result in enumerate(ranks):
    competitor = competitors[int(result)-1]
    print(f"Rank {index+1}: {competitor}")

Rank 1: gemini-2.0-flash
Rank 2: gpt-4o-mini
Rank 3: deepseek-chat
Rank 4: llama-3.3-70b-versatile
Rank 5: llama3
Rank 6: claude-3-7-sonnet-latest


In [43]:
judge_dict['judge_gemini'] = [competitors[int(i)-1] for i in ranks]

In [76]:
deepseek = OpenAI(api_key=deepseek_api_key, base_url="https://api.deepseek.com/v1")
model_name = "deepseek-chat"

response = deepseek.chat.completions.create(model=model_name, messages=judge_messages)
judge_deepseek = response.choices[0].message.content

print(judge_deepseek)

```json
{"results": ["3", "1", "4", "5", "2", "6"]}
```


In [None]:
# deepseek returned following, so did some cleanup
# ```json
#{"results": ["3", "1", "4", "5", "2", "6"]}
#```
judge_deepseek = judge_deepseek.strip().removeprefix("```json").removesuffix("```").replace("competitor ","")


In [78]:
print(judge_deepseek)


{"results": ["3", "1", "4", "5", "2", "6"]}



In [79]:
results_dict = json.loads(judge_deepseek)
ranks = results_dict["results"]
for index, result in enumerate(ranks):
    competitor = competitors[int(result)-1]
    print(f"Rank {index+1}: {competitor}")

Rank 1: gemini-2.0-flash
Rank 2: gpt-4o-mini
Rank 3: deepseek-chat
Rank 4: llama-3.3-70b-versatile
Rank 5: claude-3-7-sonnet-latest
Rank 6: llama3


In [80]:
judge_dict['judge_deepseek'] = [competitors[int(i)-1] for i in ranks]

In [81]:
groq = OpenAI(api_key=groq_api_key, base_url="https://api.groq.com/openai/v1")
model_name = "llama-3.3-70b-versatile"

response = groq.chat.completions.create(model=model_name, messages=judge_messages)
judge_groq = response.choices[0].message.content

In [82]:
results_dict = json.loads(judge_groq)
ranks = results_dict["results"]
for index, result in enumerate(ranks):
    competitor = competitors[int(result)-1]
    print(f"Rank {index+1}: {competitor}")

Rank 1: gpt-4o-mini
Rank 2: gemini-2.0-flash
Rank 3: deepseek-chat
Rank 4: llama-3.3-70b-versatile
Rank 5: llama3
Rank 6: claude-3-7-sonnet-latest


In [83]:
judge_dict['judge_groq'] = [competitors[int(i)-1] for i in ranks]

In [118]:
ollama = OpenAI(base_url='http://localhost:11434/v1', api_key='ollama')
#model_name = "llama3.2"
model_name ="llama3"

response = ollama.chat.completions.create(model=model_name, messages=judge_messages)
judge_ollama = response.choices[0].message.content

print(judge_ollama)


{
"ranked_order": [
1,
2,
4,
5,
3,
6
],
"total_score": [84, 81, 78, 76, 74, 72]
}


In [119]:
results_dict = json.loads(judge_ollama)
ranks = results_dict["ranked_order"]
for index, result in enumerate(ranks):
    competitor = competitors[int(result)-1]
    print(f"Rank {index+1}: {competitor}")

Rank 1: gpt-4o-mini
Rank 2: claude-3-7-sonnet-latest
Rank 3: deepseek-chat
Rank 4: llama-3.3-70b-versatile
Rank 5: gemini-2.0-flash
Rank 6: llama3


In [120]:
judge_dict['judge_ollama'] = [competitors[int(i)-1] for i in ranks]

In [121]:
judge_df = pd.DataFrame(judge_dict).T.reset_index()
judge_df.columns = ['judges'] + list(range(1, len(judge_df.columns)))
judge_df



Unnamed: 0,judges,1,2,3,4,5,6
0,judge_03_mini,gemini-2.0-flash,gpt-4o-mini,deepseek-chat,llama-3.3-70b-versatile,llama3,claude-3-7-sonnet-latest
1,judge_anthropic,gemini-2.0-flash,gpt-4o-mini,deepseek-chat,llama-3.3-70b-versatile,claude-3-7-sonnet-latest,llama3
2,judge_gemini,gemini-2.0-flash,gpt-4o-mini,deepseek-chat,llama-3.3-70b-versatile,llama3,claude-3-7-sonnet-latest
3,judge_deepseek,gemini-2.0-flash,gpt-4o-mini,deepseek-chat,llama-3.3-70b-versatile,claude-3-7-sonnet-latest,llama3
4,judge_groq,gpt-4o-mini,gemini-2.0-flash,deepseek-chat,llama-3.3-70b-versatile,llama3,claude-3-7-sonnet-latest
5,judge_ollama,gpt-4o-mini,claude-3-7-sonnet-latest,deepseek-chat,llama-3.3-70b-versatile,gemini-2.0-flash,llama3


In [163]:
df_melt = judge_df.melt(id_vars="judges", var_name='Rank', value_name="llm")
df_melt.Rank = df_melt.Rank.astype(int)
result = df_melt.pivot(index='llm', columns='judges', values='Rank').reset_index()
result.columns.name = None
result = result.set_index('llm')
result['mean'] = result.mean(axis=1)
result['median'] = result.median(axis=1)
result = result.sort_values('mean')
result['mean'] = result['mean'].map(lambda x: f"{x:.1f}")
result['median'] = result['median'].astype(int)
result.columns = [x.replace("judge_","") for x in result.columns.to_list()]
result.columns = pd.MultiIndex.from_product([["Judges"], result.columns])
print(result.to_markdown())

| llm                      |   ('Judges', '03_mini') |   ('Judges', 'anthropic') |   ('Judges', 'deepseek') |   ('Judges', 'gemini') |   ('Judges', 'groq') |   ('Judges', 'ollama') |   ('Judges', 'mean') |   ('Judges', 'median') |
|:-------------------------|------------------------:|--------------------------:|-------------------------:|-----------------------:|---------------------:|-----------------------:|---------------------:|-----------------------:|
| gpt-4o-mini              |                       2 |                         2 |                        2 |                      2 |                    1 |                      1 |                  1.7 |                      2 |
| gemini-2.0-flash         |                       1 |                         1 |                        1 |                      1 |                    2 |                      5 |                  1.8 |                      1 |
| deepseek-chat            |                       3 |                      

In [None]:
print([x.replace("")])

In [125]:
result.T

llm,gpt-4o-mini,gemini-2.0-flash,deepseek-chat,llama-3.3-70b-versatile,claude-3-7-sonnet-latest,llama3
judge_03_mini,2.0,1.0,3.0,4.0,6.0,5.0
judge_anthropic,2.0,1.0,3.0,4.0,5.0,6.0
judge_deepseek,2.0,1.0,3.0,4.0,5.0,6.0
judge_gemini,2.0,1.0,3.0,4.0,6.0,5.0
judge_groq,1.0,2.0,3.0,4.0,6.0,5.0
judge_ollama,1.0,5.0,3.0,4.0,2.0,6.0
mean,1.666667,1.833333,3.0,4.0,5.0,5.5
median,2.0,1.0,3.0,4.0,5.0,5.5


Mean Rank is Borda Count - sensitive to extremes
Median Rank is robust to outliers (e.g., ollama ranking gemini-2.0-flash as 5 when others say 1)

In [None]:
#Use RRF if you care more about being ranked highly than low rankings.
def RecipricalRankFusion(df, k=60):
    scores = {}
    for llm in df.index:
        score = 0
        for col in df.columns:
            if col in ["mean", "median"]:
                continue
            score += 1 / (k + df.at[llm, col])
        scores[llm] = score
    return pd.Series(scores).sort_values(ascending=False)

rrf_scores = RecipricalRankFusion(result)
print(rrf_scores)

gpt-4o-mini                 0.097303
gemini-2.0-flash            0.097087
deepseek-chat               0.095238
llama-3.3-70b-versatile     0.093750
claude-3-7-sonnet-latest    0.092353
llama3                      0.091608
dtype: float64


In [128]:
#Plurality Vote (Top-N voting)
(result.iloc[:, :-2] ==1).sum(axis=1).sort_values(ascending=False)

llm
gemini-2.0-flash            4
gpt-4o-mini                 2
deepseek-chat               0
llama-3.3-70b-versatile     0
claude-3-7-sonnet-latest    0
llama3                      0
dtype: int64

In [None]:
Have to discount Ollama as less trustworthy (e.g. smaller model). So lean on median rank.