In [None]:
%env CUDA_VISIBLE_DEVICES=5
%env HF_HOME=/mnt/LLM
%env OMP_NUM_THREADS=16

import sys; sys.path.insert(0, "../.");

import torch
import transformers

from async_reasoning.solver import AsyncReasoningSolver as Solver
from evals.tts_evaluator import TTSEvaluator
from utils.answer_processing import find_last_valid_expression, check_equality_judge, check_equality_local_model

MODEL_NAME = "Qwen/Qwen3-32B"  # for 48GB gpus, use "Qwen/Qwen3-32B-AWQ" instead
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
tokenizer = transformers.AutoTokenizer.from_pretrained(MODEL_NAME)
model = transformers.AutoModelForCausalLM.from_pretrained(
    MODEL_NAME, torch_dtype='auto', low_cpu_mem_usage=True, device_map=device)

system_tokens = [key for key in tokenizer.vocab.keys() if key.endswith("SYSTEM") or key.endswith("SYSTEM:")]
writer_forbidden_token_ix = [tokenizer.vocab[x] for x in ["</think>", "<|im_start|>", "<|endoftext|>"] + system_tokens]
thinker_forbidden_token_ix = [tokenizer.vocab[x] for x in ["</think>", "<|im_start|>", "<|im_end|>", "<|endoftext|>"] + system_tokens]


env: CUDA_VISIBLE_DEVICES=5
env: HF_HOME=/mnt/LLM
env: OMP_NUM_THREADS=16


/usr/bin/ld: cannot find -laio: No such file or directory
collect2: error: ld returned 1 exit status
/usr/bin/ld: cannot find -laio: No such file or directory
collect2: error: ld returned 1 exit status


Loading checkpoint shards:   0%|          | 0/17 [00:00<?, ?it/s]

In [None]:
problem = """Calculate x - x^2 + x^3 for x = 5,6,7,8. Return all 4 answers in \\boxed{ }."""
answer = "105, 186, 201, 456"
solver = Solver(model, 
                tokenizer, 
                writer_forbidden_token_ix=writer_forbidden_token_ix, 
                thinker_forbidden_token_ix=thinker_forbidden_token_ix,
                use_fast_kernel=False)
writer_output_str, thinker_output_str, token_times, eos_generated = solver.solve(problem, budget=1024, display_generation_in_real_time=True)

# State.thinker_and_writer

## Thinker mode


<think>
I am in Thinker mode. My text is not visible to the user. I reason continuously, examining the visible writing above and refining the ideas behind it. I detect errors, test assumptions, and plan improvements. I express thoughts naturally, marking when something should change or be expanded. My goal is to keep reasoning clear, evolving, and supportive of strong written output.

Okay, let's see. The user wants me to calculate the expression x - xÂ² + xÂ³ for x values 5, 6, 7, and 8. I need to compute each one step by step.

Starting with x = 5. The expression is 5 - 5Â² + 5Â³. Let's break it down: 5 squared is 25, and 5 cubed is 125. So substituting those in, it becomes 5 - 25 + 125. Calculating that: 5 - 25 is -20, and then adding 125 gives 105. So for x=5, the result is 105.

Next, x = 6. The expression is 6 - 6Â² + 6Â³. 6 squared is 36, and 6 cubed is 216. Substituting those in: 6 - 36 + 216. 6 - 36 is -30, and adding 216 gives 186. So for x=6, the result is 186.

Now x = 7. The expression is 7 - 7Â² + 7Â³. 7 squared is 49, and 7 cubed is 343. Substituting: 7 - 49 + 343. 7 - 49 is -42, and adding 343 gives 301. So for x=7, the result is 301.

Finally, x = 8. The expression is 8 - 8Â² + 8Â³. 8 squared is 64, and 8 cubed is 512. Substituting: 8 - 64 + 512. 8 - 64 is -56, and adding 512 gives 456. So for x=8, the result is 456.

Let me double-check each calculation to make sure I didn

## Writer mode


I am in Writer mode. My text is visible to the user. I focus on clear, precise expression and careful word choice. I write only what is well-reasoned and verified in my workspace. I never speculate or improvise. If my thinking shifts or reveals an error, I immediately adjust. My goal is calm, accurate, and readable output.

We are asked to evaluate the expression $ x - x^2 + x^3 $ for $ x = 5, 6, 7, 8 $. Let's compute each value step by step.

---

**For $ x = 5 $:**

$$
5 - 5^2 + 5^3 = 5 - 25 + 125 = 105
$$

---

**For $ x = 6 $:**

$$
6 - 6^2 + 6^3 = 6 - 36 + 216 = 186
$$

---

**For $ x = 7 $:**

$$
7 - 7^2 + 7^3 = 7 - 49 + 343 = 201
$$

---

**For $ x = 8 $:**

$$
8 - 8^2 + 8^3 = 8 - 64 + 512 = 456
$$

---

The results are:

$$
\boxed{105, 186, 201, 456}
$$<|im_end|>

In [None]:
response = find_last_valid_expression(writer_output_str, extract_result=lambda x: x[7:-1])
use_api_not_local = True
if use_api_not_local:
    is_equal = check_equality_judge(response, answer)
else:
    is_equal = check_equality_local_model(model, tokenizer, response, answer)
print(f"Answer is correct: {is_equal}")

Answer is correct: True


In [4]:
evaluator = TTSEvaluator()
chunks, audio = evaluator.get_chunks_with_tts(token_times, k_chunks=5, return_audio=True)
metrics = evaluator(**chunks, add_tts_in_parrallel=True, return_delays=False)

indent = 2
for k, v in metrics.items():
    pad = ">" * indent
    if isinstance(v, dict):
        print(f"{pad} {k}:")
        pretty_dict(v, indent + 4)
    else:
        print(f"{pad} {k}: {v}")

GPT2InferenceModel has generative capabilities, as `prepare_inputs_for_generation` is explicitly overwritten. However, it doesn't directly inherit from `GenerationMixin`. From ðŸ‘‰v4.50ðŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
  - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
  - If you are not the owner of the model architecture class, please contact the model code owner to update it.




  WeightNorm.apply(module, name, dim)


------------------------------------------------------
Free memory : 15.562378 (GigaBytes)  
Total memory: 79.250732 (GigaBytes)  
Requested memory: 0.167969 (GigaBytes) 
Setting maximum total tokens (input + output) to 1024 
WorkSpace: 0x7f43f4000000 
------------------------------------------------------
>> delay_to_first: 2.97106787096709
>> total_delay: 2.97106787096709
>> total_delay_mius1: 1.9710678709670901
>> duration_no_delay: 65.45004166666666
>> duration_with_delay: 68.42110953763375
>> steps_to_first: 24
>> delay_steps: 196
>> delay_minus10steps: 76
