In [1]:
import os
import json
from typing import List, Literal, Optional, Text, Union, Dict, Any
from IPython.display import Image, display
from langchain_core.runnables.graph_mermaid import MermaidDrawMethod
from agents.agents_modules.workflow import build_agent_workflow
from agents.dataloader import load_dataset_by_name, extract_example
from agents.utilities.agent_utils import save_result_to_json
from agents.llm_model import UnifiedModel, model_name
from agents.agent_prompts import END_TO_END_GENERATION_PROMPT, input_prompt, CONTENT_SELECTION_PROMPT

  from .autonotebook import tqdm as notebook_tqdm


In [3]:
provider = "openai" #ollama, openai, hf, aixplain
process_flow = build_agent_workflow(provider=provider)
# display(Image(process_flow.get_graph(xray=True).draw_mermaid_png()))

In [16]:
# Exclude turku_hockey and conversational_weather
# totto, webnlg, dart, rotowire, sportsett_basketball, mlb
name = "webnlg"  
num = 1750 # Sample id
set = "test"

data = load_dataset_by_name(name)
dataset = data[set]
sample = extract_example(name, data[set][num])
data = sample.get('input', '')
ground_truth = sample.get('references', "")
target = sample.get('target', "")

print(f"Size: {len(dataset)}")
print("========== INPUT DATA ==========")
print(f"{data}\n")
print("========== TARGET (if any) ==========")
print(f"{target}\n")
print("========== GROUND TRUTH REFERENCES ==========")
print(f"{ground_truth}")

Loading dataset: webnlg
Size: 1779
['Mexico | currency | Mexican_peso', 'Mexico | demonym | Mexicans', 'Bionico | course | Dessert', 'Bionico | ingredient | Granola', 'Dessert | dishVariation | Sandesh_(confectionery)', 'Bionico | country | Mexico']

Mexicans inhabit Mexico, where the currency is the peso. Bionico is a food found in Mexico that requires granola as one of its ingredients, and is a dessert like Sandesh.




In [10]:
conf = model_name.get(provider.lower())
llm = UnifiedModel(provider=provider, **conf).model_(CONTENT_SELECTION_PROMPT)
CS_prediction = llm.invoke({'input': data}).content.strip()
print(CS_prediction)

[
  "Location (Hypermarcas): Brazil",
  "Product (Hypermarcas): Drugs",
  "Founding Date (Hypermarcas): 2001-01-01"
]


In [11]:
query = f"""You are an agent designed to generate text from data for a data-to-text natural language generation. You can be provided data in the form of xml, table, meaning representations, graphs etc. 
Your task is to generate the appropriate text given the data information without omitting any field or adding extra information in essence called hallucination.
Here is the data, generate text using the provided data:
{CS_prediction}"""

initial_state = {
    "user_prompt": query,
    "raw_data": data,
    "history_of_steps": [],
    "final_response": "",
    "next_agent": "",
    "next_agent_payload": "",
    "current_step": 0,
    "iteration_count": 0,
    "max_iteration": 100,
}

state = process_flow.invoke(initial_state, config={"recursion_limit": initial_state["max_iteration"]})
prediction = state['final_response']



[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3mAction:
```
{
  "action": "Final Answer",
  "action_input": [
    "Location (Hypermarcas): Brazil",
    "Founding Date (Hypermarcas): 2001-01-01",
    "Product (Hypermarcas): Drugs"
  ]
}
```[0m

[1m> Finished chain.[0m


[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3mAction:
```
{
  "action": "Final Answer",
  "action_input": "<paragraph>\n<snt>\nLocation (Hypermarcas): Brazil  \nFounding Date (Hypermarcas): 2001-01-01\n</snt>\n<snt>\nProduct (Hypermarcas): Drugs\n</snt>\n</paragraph>"
}
```[0m

[1m> Finished chain.[0m


[1m> Entering new AgentExecutor chain...[0m
[32;1m[1;3m{
  "action": "Final Answer",
  "action_input": "Hypermarcas is located in Brazil and was founded on January 1, 2001. The company produces drugs."
}[0m

[1m> Finished chain.[0m


In [12]:
save_result_to_json(state, dataset_folder=f"{name}", filename=f"{name}_{num}.json")
# save_result_to_json(state, filename=f"webnlg_long.json")

[SAVED] Agent result saved to: results/webnlg/webnlg_1002.json


In [13]:
print(prediction)

Hypermarcas is located in Brazil and was founded on January 1, 2001. The company produces drugs.


In [27]:
ground_truth

''

In [28]:
target

"If Celtics fans still had any doubts about what kind of player Jeff Green is , those doubts had to have been put to rest after Monday night . Unfortunately for the Celts , despite Green 's career night and an excellent team - wide performance in the absence of Kevin Garnett , the opponent was the Miami Heat , who came in winners of 22 consecutive games . And LeBron James was n't ready to let that streak end . James had 37 points , 12 assists , seven rebounds , two steals and two blocks to lead the Heat , who trailed by 17 in the first half and 13 in the fourth quarter , to a 105 - 103 win , their 23rd straight . James untied the game by burying a 20 - footer with 10 seconds left to seal it . Green had a career high 43 points , 38 in the first three quarters , on an amazing 14 - of - 21 shooting , and added seven rebounds , four blocks and a couple of steals . It was the highest scoring regular season game from any Celtic since 2007 , the highest scoring game against the Heat all year 

In [None]:
# from agents.evaluator import evaluate_single

# scores = evaluate_single(target, prediction, data)
# for metric, score in scores.items():
#     print(f"{metric}: {score:.4f}")

In [None]:
# FINISHING TO COMPUTE BLEU...
#   BLEU    BLEU NLTK
# ------  -----------
#  27.18         0.26

#   chrF++
# --------
#     0.61

#   TER
# -----
#  0.71

In [None]:
# data = """[['Lagos', 'areaLand', '999600000.0'],
#  ['Lagos', 'areaMetro', '2706700000.0'],
#  ['Lagos', 'areaUrban', '907000000.0'],
#  ['Lagos', 'populationMetro', '21320000'],
#  ['Lagos', 'populationMetroDensity', '7759.0'],
#  ['Lagos', 'areaMetro', '2706.7'],
#  ['Lagos', 'areaTotal', '1171.28'],
#  ['Lagos', 'areaUrban', '907.0'],
#  ['Lagos', 'populationDensity', '6871.0'],
#  ['Lagos', 'populationMetroDensity', '7759.0'],
#  ['Lagos', 'areaCode', '010'],
#  ['Lagos', 'areaTotal', '1171280000.0'],
#  ['Lagos', 'demonym', 'Lagosian'],
#  ['Lagos', 'motto', 'Èkó ò ní bàjé o!'],
#  ['Lagos', 'populationDensity', '6871.0'],
#  ['Lagos', 'populationTotal', '8048430'],
#  ['Lagos', 'utcOffset', '+1'],
#  ['Lagos', 'country', 'Nigeria'],
#  ['Lagos', 'founder', 'Awori_tribe'],
#  ['Lagos', 'timeZone', 'West_Africa_Time'],
#  ['Benfruit_Plant', 'location', 'Lagos'],
#  ['Nigerian_Institute_of_Management', 'location', 'Lagos'],
#  ['Before_30', 'location', 'Lagos'],
#  ['Reduction_of_Lagos', 'place', 'Lagos'],
#  ["Cait_O'Riordan", 'birthPlace', 'Lagos'],
#  ['Candy_Augustine_Agbane', 'birthPlace', 'Lagos'],
#  ['Carol_King_(actress)', 'birthPlace', 'Lagos'],
#  ['Amusa_Shittu', 'deathPlace', 'Lagos'],
#  ['Samuel_Ajayi_Crowther', 'deathPlace', 'Lagos'],
#  ['Samuel_Johnson_(Nigerian_historian)', 'deathPlace', 'Lagos'],
#  ['Dobox', 'foundationPlace', 'Lagos'],
#  ['IROKO_Partners', 'foundationPlace', 'Lagos'],
#  ['Isaac_Fola-Alade', 'significantBuilding', 'Lagos'],
#  ['Olufemi_Majekodunmi', 'significantBuilding', 'Lagos'],
#  ['Cally_Air', 'targetAirport', 'Lagos'],
#  ['Ibom_Air', 'targetAirport', 'Lagos'],
#  ['Max_Air', 'targetAirport', 'Lagos'],
#  ['Isaac_Fola-Alade', 'significantProject', 'Lagos'],
#  ['ADC_Airlines', 'hubAirport', 'Lagos'],
#  ['Babington_Macaulay_Junior_Seminary', 'state', 'Lagos'],
#  ['Government_College_Ikorodu', 'state', 'Lagos'],
#  ['Ebun_Oluwa_Pro_Veritas', 'state', 'Lagos'],
#  ['Colonial_Nigeria', 'capital', 'Lagos'],
#  ['Federation_of_Nigeria', 'capital', 'Lagos'],
#  ['Lagos_Colony', 'capital', 'Lagos'],
#  ['African_Union', 'largestCity', 'Lagos'],
#  ['Nigeria', 'largestCity', 'Lagos'],
#  ['Military_dictatorship_in_Nigeria', 'largestCity', 'Lagos'],
#  ['Volkswagen_Beetle', 'assembly', 'Lagos'],
#  ['Volkswagen_Jetta_(A2)', 'assembly', 'Lagos'],
#  ['Isuzu_Forward', 'assembly', 'Lagos'],
#  ['Juliet_Ibrahim', 'award', 'Lagos'],
#  ['Benin-Ore-Shagamu_Expressway', 'city', 'Lagos'],
#  ['Encomium_Magazine', 'city', 'Lagos'],
#  ['Bogobiri_House', 'city', 'Lagos'],
#  ['Sub-Saharan_Africa', 'country', 'Lagos'],
#  ['Desmond_Majekodunmi', 'education', 'Lagos'],
#  ['Buchi_Emecheta', 'education', 'Lagos'],
#  ['Habeeb_Okunola', 'education', 'Lagos'],
#  ['MFM_F.C.', 'ground', 'Lagos'],
#  ['ACB_Lagos_F.C.', 'ground', 'Lagos'],
#  ['COD_United_F.C.', 'ground', 'Lagos'],
#  ['Caritas_Communications', 'headquarter', 'Lagos'],
#  ['Nigerian_Accounting_Association', 'headquarter', 'Lagos'],
#  ['Merchant_Express_Aviation', 'headquarter', 'Lagos'],
#  ['Sasha_P', 'hometown', 'Lagos'],
#  ['Davina_Oriakhi', 'hometown', 'Lagos'],
#  ['VanJess', 'hometown', 'Lagos'],
#  ['Olu_Amoda', 'knownFor', 'Lagos'],
#  ['Toki_Mabogunje', 'knownFor', 'Lagos'],
#  ['United_Africa_Company_of_Nigeria', 'locationCity', 'Lagos'],
#  ['Promasidor_Nigeria', 'locationCity', 'Lagos'],
#  ['G-Worldwide_Entertainment', 'locationCity', 'Lagos'],
#  ['LADOL', 'locationCountry', 'Lagos'],
#  ['Atlantic_Ocean', 'nearestCity', 'Lagos'],
#  ['Lagos_Lagoon', 'nearestCity', 'Lagos'],
#  ['Let_Me_Roll_It', 'recordedIn', 'Lagos'],
#  ['Mamunia', 'recordedIn', 'Lagos'],
#  ['Band_on_the_Run_(song)', 'recordedIn', 'Lagos'],
#  ['First_Consultant_Hospital', 'region', 'Lagos'],
#  ['National_Orthopaedic_Hospital,_Igbobi', 'region', 'Lagos'],
#  ['GIG_Mobility', 'regionServed', 'Lagos'],
#  ['TEDxLagos', 'regionServed', 'Lagos'],
#  ['Patrick_Utomi', 'residence', 'Lagos'],
#  ['Dakuku_Peterside', 'residence', 'Lagos'],
#  ['Mudashiru_Obasa', 'residence', 'Lagos']]"""

In [17]:
# lagos_86 = """Lagos is located in Nigeria, and its inhabitants are known as Lagosians. The city's motto is 'Èkó ò ní bàjé o!' and it was founded by the Awori tribe. Lagos operates on West Africa Time, with a UTC offset of +1. The land area of Lagos covers 999.6 square kilometers, with the urban area measuring 907 square kilometers, while the metropolitan area spans 2,706.7 square kilometers. The total area of the city is 1,171.28 square kilometers. Lagos has a population of 8,048,430, with a population density of 6,871 people per square kilometer, while the metropolitan population reaches 21,320,000 with a metro density of 7,759 per square kilometer. The city's area code is 010.

# The Atlantic Ocean and Lagos Lagoon are closest to Lagos. It is recognized as the largest city within the African Union, Nigeria, and during the Military dictatorship in Nigeria. Lagos has also served as the capital of Colonial Nigeria, the Federation of Nigeria, and the Lagos Colony. In Sub-Saharan Africa, Lagos is counted among its prominent cities.

# Numerous institutions and organizations are based in Lagos, such as Benfruit Plant, the Nigerian Institute of Management, and Before 30. The city is also the foundation place of Dobox and IROKO Partners. Companies like Caritas Communications, the Nigerian Accounting Association, and Merchant Express Aviation have their headquarters in Lagos. The United Africa Company of Nigeria, Promasidor Nigeria, and G-Worldwide Entertainment are all located in the city. LADOL is also situated in Lagos. Lagos serves as the region for GIG Mobility and TEDxLagos, as well as for First Consultant Hospital and National Orthopaedic Hospital, Igbobi. Notable locations like the Benin-Ore-Shagamu Expressway, Encomium Magazine, and Bogobiri House are found here, and educational institutions such as Babington Macaulay Junior Seminary, Government College Ikorodu, and Ebun Oluwa Pro Veritas are located within Lagos State. Lagos is home to the grounds of MFM F.C., ACB Lagos F.C., and COD United F.C., and is associated with the assembly of vehicles like the Volkswagen Beetle, Volkswagen Jetta (A2), and Isuzu Forward. The city is known for significant buildings designed by Isaac Fola-Alade and Olufemi Majekodunmi, as well as a significant project by Isaac Fola-Alade. Music recordings such as 'Let Me Roll It', 'Mamunia', and 'Band on the Run' were recorded in Lagos. The city is notable for the historic Reduction of Lagos, and features as a target airport for airlines including Cally Air, Ibom Air, and Max Air, while also serving as a hub for ADC Airlines.

# Lagos is the hometown of personalities such as Sasha P, Davina Oriakhi, and VanJess. It is likewise the birthplace of Cait O'Riordan, Candy Augustine Agbane, and Carol King the actress. The city is where figures like Amusa Shittu, Samuel Ajayi Crowther, and Samuel Johnson, the Nigerian historian, passed away. It is home to residents including Patrick Utomi, Dakuku Peterside, and Mudashiru Obasa. Individuals such as Olu Amoda and Toki Mabogunje are known for their connections to Lagos. The city is where Desmond Majekodunmi, Buchi Emecheta, and Habeeb Okunola received their education, and Juliet Ibrahim is among those who have received awards in Lagos."""

# print(lagos_86)

In [18]:
# import json
# from pathlib import Path

# src_path = Path("results/webnlg.json")
# dst_path = Path("results/webnlg.txt")

# with src_path.open(encoding="utf-8") as src, dst_path.open("w", encoding="utf-8") as dst:
#     for lineno, line in enumerate(src, 1):
#         line = line.strip()
#         if not line:          # skip blank lines
#             continue
#         try:
#             obj = json.loads(line)
#         except json.JSONDecodeError as exc:
#             print(f"⚠️  skip line {lineno}: {exc}")
#             continue

#         # Normalise whitespace: collapse any CR/LF runs to a single space
#         pred = " ".join(obj.get("prediction", "").split())
#         if pred:
#             dst.write(pred + "\n")

# print(f"Finished – cleaned predictions saved to {dst_path}")


In [20]:
# import os
# from typing import List, Union
# from agents.dataloader import load_dataset_by_name, extract_example

# # ── CONFIG ────────────────────────────────────────────────────────────────
# name = "webnlg"   # dataset name understood by load_dataset_by_name()
# split = "test"    # which split to use (train / valid / test)
# outdir = "references"  # where to write reference0 … reference3
# num_reference_files = 4
# # ──────────────────────────────────────────────────────────────────────────


# def to_fixed_list(target: Union[str, List[str]]) -> List[str]:
#     """Return a list of exactly `num_reference_files` strings."""
#     if isinstance(target, str):
#         target_list = [target]
#     else:
#         target_list = list(target)
#     padded = (target_list + [""] * num_reference_files)[:num_reference_files]
#     return [t.strip() for t in padded]


# data = load_dataset_by_name(name)
# dataset = data[split]

# os.makedirs(outdir, exist_ok=True)
# files = [open(os.path.join(outdir, f"reference{i}"), "w", encoding="utf-8")
#          for i in range(num_reference_files)]

# try:
#     for i, raw in enumerate(dataset):
#         sample = extract_example(name, raw)
#         targets = to_fixed_list(sample.get("target", ""))
#         for fh, t in zip(files, targets):
#             fh.write(t + "\n")
#         if (i + 1) % 500 == 0:
#             print(f"Processed {i + 1}/{len(dataset)} samples …")
# finally:
#     for fh in files:
#         fh.close()

# print("✅ Done – reference files saved to:")
# print("\n".join(os.path.join(outdir, f"reference{i}") for i in range(num_reference_files)))
