In [21]:
!pip -q install transformers datasets accelerate evaluate

huggingface/tokenizers: The current process just got forked, after parallelism has already been used. Disabling parallelism to avoid deadlocks...
	- Avoid using `tokenizers` before the fork if possible
	- Explicitly set the environment variable TOKENIZERS_PARALLELISM=(true | false)


In [22]:
import os
os.environ["WANDB_DISABLED"] = "true"   # prevents wandb prompts in Colab

import torch

print("CUDA:", torch.cuda.is_available())
if torch.cuda.is_available():
    print("GPU:", torch.cuda.get_device_name(0))

CUDA: True
GPU: Tesla P100-PCIE-16GB


In [23]:
from datasets import load_dataset

DATA_PATH = "/kaggle/input/bitcoin-data/bitcoin_dataset.jsonl"  # if you're in this environment
# In Colab, you will likely upload it, then use: DATA_PATH = "/content/bitcoin_dev_gpt2_dataset.jsonl"

ds = load_dataset("json", data_files=DATA_PATH, split="train")
print(ds)
print(ds[0])

Dataset({
    features: ['instruction', 'input', 'response'],
    num_rows: 7165
})
{'instruction': 'Summarize the following Bitcoin/Lightning/Liquid/Taproot Assets notes in 5-7 bullet points.', 'input': ' Foundations of the Lightning Network The scalability challenge and off-chain payment networks:  Limitations of Bitcoin Layer 1:  Block size: Limited to ~1 MB (up to ~4 MB effective with SegWit).  Block interval: A new block is mined every ~10 minutes.  Throughput: This results in an average of 3–7 transactions per second (tps), far lower than centralized payment systems such as Visa (~20,000+ tps).  These parameters ensure that running a Bitcoin node is possible for individuals with modest hardware, but they also create a bottleneck for scaling to global payment use.  Why not increase block size?  A natural suggestion is to raise the block size, allowing more transactions per block. However, larger blocks would:  Increase bandwidth and storage requirements for nodes, harder for ordin

In [24]:
ds = ds.shuffle(seed=42)
splits = ds.train_test_split(test_size=0.02, seed=42)   # 2% validation
train_ds = splits["train"]
eval_ds  = splits["test"]

print("Train:", len(train_ds), "Eval:", len(eval_ds))

Train: 7021 Eval: 144


In [25]:
def format_example(ex):
    instr = (ex.get("instruction") or "").strip()
    inp   = (ex.get("input") or "").strip()
    resp  = (ex.get("response") or "").strip()

    # This format is what GPT-2 learns to mimic.
    # The "###" headers make the structure explicit and consistent.
    if inp:
        text = f"### Instruction:\n{instr}\n\n### Input:\n{inp}\n\n### Response:\n{resp}\n"
    else:
        text = f"### Instruction:\n{instr}\n\n### Response:\n{resp}\n"

    return {"text": text}

train_ds = train_ds.map(format_example, remove_columns=train_ds.column_names)
eval_ds  = eval_ds.map(format_example, remove_columns=eval_ds.column_names)

print(train_ds[0]["text"][:600])

### Instruction:
Explain these notes using one simple example (invent small numbers if needed) to show the flow.

### Input:
TapTweak — What It Is and How It Works  TapTweak is the cryptographic operation that combines a user’s internal public key with the Merkle root of all TapScript branches.  It creates a single new key — the Taproot output key (P(cid:0)ₐ(cid:0)ᵣₒₒ(cid:0)) — that commits to both the key path and script paths simultaneously.  This tweak ensures that:  all spending conditions (scripts) are cryptographically bound to the output, and  the Taproot address still looks like a simp


In [26]:
from transformers import AutoTokenizer, AutoModelForCausalLM

MODEL_NAME = "gpt2"  # base GPT-2 (124M)

tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME)

# GPT-2 has no pad token by default. Trainer needs padding for batches.
tokenizer.pad_token = tokenizer.eos_token

model = AutoModelForCausalLM.from_pretrained(MODEL_NAME)
model.config.pad_token_id = tokenizer.eos_token_id

print("Loaded:", MODEL_NAME)

Loaded: gpt2


In [27]:
MAX_LEN = 768  

def tokenize(batch):
    return tokenizer(
        batch["text"],
        truncation=True,
        max_length=MAX_LEN,
        padding="max_length",
    )

train_tok = train_ds.map(tokenize, batched=True, remove_columns=["text"])
eval_tok  = eval_ds.map(tokenize, batched=True, remove_columns=["text"])

print(train_tok[0].keys())

Map:   0%|          | 0/144 [00:00<?, ? examples/s]

dict_keys(['input_ids', 'attention_mask'])


In [28]:
from transformers import DataCollatorForLanguageModeling

data_collator = DataCollatorForLanguageModeling(
    tokenizer=tokenizer,
    mlm=False  # IMPORTANT: GPT-2 is NOT masked LM
)

In [29]:
from transformers import TrainingArguments, Trainer

args = TrainingArguments(
    output_dir="gpt2-bitcoin-dev",
    overwrite_output_dir=True,

    # training
    num_train_epochs=2,
    per_device_train_batch_size=4,     # ↑ from 2
    gradient_accumulation_steps=4,     # effective batch = 16
    learning_rate=5e-5,
    weight_decay=0.01,

    # eval/save/logging
    eval_strategy="steps",
    eval_steps=200,
    save_strategy="steps",
    save_steps=200,
    logging_steps=50,
    save_total_limit=2,

    # performance
    fp16=True,                         # P100 handles fp16 very well
    report_to="none",
)

trainer = Trainer(
    model=model,
    args=args,
    train_dataset=train_tok,
    eval_dataset=eval_tok,
    data_collator=data_collator,
)

trainer.train()


Step,Training Loss,Validation Loss
200,1.8302,1.56139
400,1.485,1.226752
600,1.2187,1.016203
800,1.1938,0.92388


TrainOutput(global_step=878, training_loss=1.5329748818434452, metrics={'train_runtime': 1859.2002, 'train_samples_per_second': 7.553, 'train_steps_per_second': 0.472, 'total_flos': 5503594070016000.0, 'train_loss': 1.5329748818434452, 'epoch': 2.0})

In [30]:
save_dir = "gpt2-bitcoin-dev-final"
trainer.save_model(save_dir)
tokenizer.save_pretrained(save_dir)

print("Saved to:", save_dir)

Saved to: gpt2-bitcoin-dev-final


In [31]:
import torch
from transformers import pipeline, StoppingCriteria, StoppingCriteriaList

class StopOnSubstrings(StoppingCriteria):
    def __init__(self, tokenizer, stop_strings):
        self.tokenizer = tokenizer
        self.stop_ids = [tokenizer.encode(s, add_special_tokens=False) for s in stop_strings]

    def __call__(self, input_ids, scores, **kwargs):
        for stop in self.stop_ids:
            if len(input_ids[0]) >= len(stop) and input_ids[0][-len(stop):].tolist() == stop:
                return True
        return False


gen = pipeline(
    "text-generation",
    model=save_dir,
    tokenizer=save_dir,
    device=0 if torch.cuda.is_available() else -1
)

prompt = """### Instruction:
What is Bitcoin?

### Response:
"""

stopping_criteria = StoppingCriteriaList([
    StopOnSubstrings(
        gen.tokenizer,
        stop_strings=["### Instruction:", "### Input:"]
    )
])

out = gen(
    prompt,
    max_new_tokens=160,
    do_sample=False,
    no_repeat_ngram_size=4,
    eos_token_id=gen.tokenizer.eos_token_id,
    pad_token_id=gen.tokenizer.eos_token_id,
    stopping_criteria=stopping_criteria
)

print(out[0]["generated_text"])


Device set to use cuda:0
The following generation flags are not valid and may be ignored: ['temperature']. Set `TRANSFORMERS_VERBOSITY=info` for more details.


### Instruction:
What is Bitcoin?

### Response:
Bitcoin is a distributed, digital money system built on cryptography and cryptography principles.  It is not pegged to any single Bitcoin output — it is a cryptographic, multi-asset system pegged to a Bitcoin UTXO.  It uses Proof-of-Work (PoW) mining to ensure that every transaction is valid, traceable, and verifiable.  Bitcoin’s Proof-of–Work (PoS) system ensures that transactions are not created by anyone else but by the hash of a public key, meaning every transaction is independently valid.  Bitcoin uses Proof-Of-Work (Proof-of-Reserves) to ensure that transactions are valid, traceability, and verifiability, not by anyone else.  Bitcoin Core Concepts  Bitcoin Core is a Bitcoin-native


In [32]:
device = 0 if torch.cuda.is_available() else -1

# Base GPT-2
base_gen = pipeline(
    "text-generation",
    model="gpt2",
    tokenizer="gpt2",
    device=device
)

base_stop = StoppingCriteriaList([
    StopOnSubstrings(base_gen.tokenizer, ["### Instruction:", "### Input:"])
])

# Fine-tuned GPT-2
ft_gen = pipeline(
    "text-generation",
    model=save_dir,
    tokenizer=save_dir,
    device=device
)

ft_stop = StoppingCriteriaList([
    StopOnSubstrings(ft_gen.tokenizer, ["### Instruction:", "### Input:"])
])

Device set to use cuda:0
Device set to use cuda:0


In [33]:
def compare(prompt, max_new_tokens=180):
    print("=" * 90)
    print("PROMPT:")
    print(prompt)
    print("=" * 90)

    base_out = base_gen(
        prompt,
        max_new_tokens=max_new_tokens,
        do_sample=True,
        temperature=0.7,
        top_p=0.9,
    )[0]["generated_text"]

    ft_out = ft_gen(
        prompt,
        max_new_tokens=max_new_tokens,
        do_sample=True,
        temperature=0.7,
        top_p=0.9,
    )[0]["generated_text"]

    print("\n--- BASE GPT-2 OUTPUT ---\n")
    print(base_out)

    print("\n--- FINE-TUNED GPT-2 OUTPUT ---\n")
    print(ft_out)

In [34]:
prompt1 = """### Instruction:
What is Bitcoin?

### Response:
"""
compare(prompt1)

Setting `pad_token_id` to `eos_token_id`:50256 for open-end generation.


PROMPT:
### Instruction:
What is Bitcoin?

### Response:


--- BASE GPT-2 OUTPUT ---

### Instruction:
What is Bitcoin?

### Response:

Bitcoin is a digital currency. It is an alternative to the traditional currency. It is decentralized and can be used to create any kind of digital currency that can be used for any purpose. It is based on Bitcoin and is a digital currency. The primary purpose of Bitcoin is to make money. The main purpose of Bitcoin is to make money and to be able to buy and sell anything. It is a decentralized currency that can be used to purchase anything.

### Response:

Bitcoin is a digital currency. It is a digital currency. It is a decentralized currency that can be used to purchase anything. It is a decentralized currency that can be used to purchase anything.

### Response:

Bitcoin is a digital currency. It is a digital currency. It is a decentralized currency that can be used to purchase anything. It is a decentralized currency that can be used to purchase any

In [35]:
prompt2 = """### Instruction:
Explain how a Bitcoin transaction is verified.

### Response:
"""
compare(prompt2)

Setting `pad_token_id` to `eos_token_id`:50256 for open-end generation.


PROMPT:
### Instruction:
Explain how a Bitcoin transaction is verified.

### Response:


--- BASE GPT-2 OUTPUT ---

### Instruction:
Explain how a Bitcoin transaction is verified.

### Response:

If you have any questions or concerns about our service or our product, please contact us at support@bitcoin.com.

--- FINE-TUNED GPT-2 OUTPUT ---

### Instruction:
Explain how a Bitcoin transaction is verified.

### Response:
Verification relies on comparing a Taproot output with the witness pubkey, ensuring the proof is cryptographically secure.  If the signature matches the revealed Taproot public key, the witness pubkey is valid.  If the signature does not match, the proof is invalid.  Key Derivation  Each output key has a unique cryptographic identifier derived from the hash of the previous output key: H_pubkey → private key derived from previous H_pubkey.  The resulting key pair can be forged using:  A Schnorr signature: H_SchnorrSignature = H_pubkey + e × G.  Derivation using a pairwise

In [36]:
prompt3 = """### Instruction:
Explain what an HTLC is and why it is needed in the Lightning Network.

### Response:
"""

compare(prompt3)

Setting `pad_token_id` to `eos_token_id`:50256 for open-end generation.


PROMPT:
### Instruction:
Explain what an HTLC is and why it is needed in the Lightning Network.

### Response:


--- BASE GPT-2 OUTPUT ---

### Instruction:
Explain what an HTLC is and why it is needed in the Lightning Network.

### Response:

We have received this question from an engineer with the Linux Foundation.

We would like to send an email to you with further information.

The Linux Foundation has the following guidelines on how to respond to questions about the Lightning Network.

1. If you have an answer to a question you would like to have answered, please send us an email.

2. If you have an answer to a question you would like to have answered, please send us an email.

3. If you have an answer to a question you would like to have answered, please send us an email.

4. If you have an answer to a question you would like to have answered, please send us an email.

5. If you have an answer to a question you would like to have answered, please send us an email.

6. If you

---

In [39]:
prompt4 = """### Instruction:
What is taproot assets?

### Response:
"""
compare(prompt4)

Setting `pad_token_id` to `eos_token_id`:50256 for open-end generation.


PROMPT:
### Instruction:
What is taproot assets?

### Response:


--- BASE GPT-2 OUTPUT ---

### Instruction:
What is taproot assets?

### Response:

I have a small set of taproot assets. I have not used them to generate a program, but it is my personal favorite tool.

### Usage:

### Usage:

import taproot

import taproot.common.common.main

import taproot.common.common.tapp

import taproot.common.common.tapp.TappMain

import taproot.common.common.tapp.TappMain

import taproot.common.common.tapp.TappMain.Tapp

import taproot.common.common.tapp.TappMain.TappMain.TappMain

import taproot.common.common.tapp.TappMain.TappMain

import taproot.common.common.tapp.TappMain.T

--- FINE-TUNED GPT-2 OUTPUT ---

### Instruction:
What is taproot assets?

### Response:
Taproot Assets is a Bitcoin-native, proof-of-work implementation built on Merkle roots and Merkle inclusion paths.  It uses Schnorr signatures instead of SHA-256, improves on Schnorr for smaller keys, and reduces the number of checks

In [41]:
prompt6 = """### Instruction:
What is lightning network

### Response:
"""
compare(prompt6)

Setting `pad_token_id` to `eos_token_id`:50256 for open-end generation.


PROMPT:
### Instruction:
What is lightning network

### Response:


--- BASE GPT-2 OUTPUT ---

### Instruction:
What is lightning network

### Response:

What is lightning network

### Example:

### Instruction:

What is lightning network

### Response:

What is lightning network

### Example:

### Instruction:

What is lightning network

### Response:

What is lightning network

### Example:

### Instruction:

What is lightning network

### Response:

What is lightning network

### Example:

### Instruction:

What is lightning network

### Response:

What is lightning network

### Example:

### Instruction:

What is lightning network

### Response:

What is lightning network

### Example:

### Instruction:

What is lightning network

### Response:

What is lightning network

### Example:

### Instruction:

What

--- FINE-TUNED GPT-2 OUTPUT ---

### Instruction:
What is lightning network

### Response:
Lightning Network is a peer-to-peer (P2P) payment network built on Bitcoin Layer 1. 