In [1]:
import torch
from transformers import AutoModelForCausalLM, AutoTokenizer
from transformers import BitsAndBytesConfig
from peft import LoraConfig, get_peft_model

device = 'cuda:0' if torch.cuda.is_available() else 'cpu'

print(f'device: {device}')

tokenizer = AutoTokenizer.from_pretrained('../Meta-Llama-3-8B-Instruct')
model = AutoModelForCausalLM.from_pretrained('../Meta-Llama-3-8B-Instruct', torch_dtype=torch.float16, device_map={"": 0})
# tokenizer.pad_token = tokenizer.eos_token
# tokenizer.pad_token_id = tokenizer.eos_token_id
# model.config.use_cache = False
# model.config.pretraining_tp = 1
# model.generation_config.pad_token_id = tokenizer.pad_token_id
# model.config.pad_token_id = tokenizer.pad_token_id

from transformers import pipeline
from langchain_huggingface import HuggingFacePipeline

pipe = pipeline('text-generation', model=model, tokenizer=tokenizer, max_new_tokens=1024, return_full_text=False)
llm = HuggingFacePipeline(pipeline=pipe)


Special tokens have been added in the vocabulary, make sure the associated word embeddings are fine-tuned or trained.


device: cuda:0


Loading checkpoint shards:   0%|          | 0/4 [00:00<?, ?it/s]

In [2]:
from langchain.agents.output_parsers import ReActSingleInputOutputParser


In [3]:
from chain_items import PromptValueToChatML, ChatTemplateApplier
from chain_items.static import generate_history_passthrough, generate_gemma2_chat_prompt_template, generate_tools  

tools = generate_tools()
agent = generate_history_passthrough() | generate_gemma2_chat_prompt_template(tools) | PromptValueToChatML() | ChatTemplateApplier(tokenizer=tokenizer) | llm | ReActSingleInputOutputParser()


In [4]:
print(
  agent.invoke({
    'input': 'who is president of south korea now?',
    # 'input': 'please give me a address of myongji university yongin campus.',
    'intermediate_steps': [
      # (AgentAction('searxng', 'president of south korea', 'Thought: I should search for the recent web results to find the answer.\n\nAction: searxng\n\nAction Input: "current president of south korea"'), 'yoon seok yeol')
    ]
  })
)


Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.


input messages=[HumanMessage(content='Answer the following questions as best you can. You have access to the following tools:\n\nsearxng(*args, **kargs) - Search searxng for recent web results\n\nUse only one of following formats:\n\nThought: you should always think about what to do\nAction: the action to take, should be one of [searxng]\nAction Input: the input to the action\n\nor\n\nThought: I now know the final answer\nFinal Answer: the final answer to the original input question\n\nBegin!\n\nQuestion: who is president of south korea now?')]
tool='searxng' tool_input='who is the current president of south korea' log='Thought: I should search for the most recent information about the President of South Korea.\n\nAction: searxng\n\nAction Input: "who is the current president of south korea"'


In [5]:
from langchain.agents import AgentExecutor

agent_executor = AgentExecutor(agent=agent, tools=tools, verbose=True, handle_parsing_errors=True, max_iterations=3, return_intermediate_steps=True)


In [6]:
res = agent_executor.invoke(input={'input': 'what is reinforcement learning?'})

print(f'output: {res['output']}')


Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.




[1m> Entering new AgentExecutor chain...[0m
input messages=[HumanMessage(content='Answer the following questions as best you can. You have access to the following tools:\n\nsearxng(*args, **kargs) - Search searxng for recent web results\n\nUse only one of following formats:\n\nThought: you should always think about what to do\nAction: the action to take, should be one of [searxng]\nAction Input: the input to the action\n\nor\n\nThought: I now know the final answer\nFinal Answer: the final answer to the original input question\n\nBegin!\n\nQuestion: what is reinforcement learning?')]
[32;1m[1;3mThought: I should search for recent web results on "reinforcement learning"

Action: searxng

Action Input: ["reinforcement learning"]

(Note: I'll wait for the results from searxng before providing a further response)[0m

Setting `pad_token_id` to `eos_token_id`:128001 for open-end generation.


[36;1m[1;3m... reinforcement learning with human feedback (RLHF) scenarios. The labeled data can be used to further train the LLM to improve its performance. Evaluate ...



Despite the feature's presence, users were advised by others to wait for further development before diving in. Unsloth Achieves Significant Downloads: The Unsloth team announced hitting 500k monthly model downloads on Hugging Face, thanking the community for the widespread support and usage of Unsloth's finetuning framework. The necessity of ...

The reinforcement learning (RL) algorithms can all cope with delays to important reward values (such as those from winning or losing a game). Bigger delays - e.g. thousands of $0$ reward steps between more meaningful values - are often associated with harder problems to solve, so the simpler algorithms might not be practical.

Jun 29, 2024 — RLHF Training Puzzles Users: Conversations about Reinforcement Learning from Human Feedback (RLHF) revealed confusion regarding its