In [10]:
from transformers import BertForSequenceClassification
print("成功导入 BertForSequenceClassification")

成功导入 BertForSequenceClassification


In [11]:
import pandas as pd
import numpy as np
import jieba
import random
import torch
from sklearn.utils import shuffle
from sklearn.model_selection import train_test_split
from sklearn.metrics import accuracy_score

from transformers import BertTokenizerFast, BertForSequenceClassification, Trainer, TrainingArguments


In [12]:

# 固定随机种子
def set_seed(seed: int):
    random.seed(seed)
    np.random.seed(seed)
    torch.manual_seed(seed)
    torch.cuda.manual_seed_all(seed)

set_seed(123)




In [13]:
# 读取训练数据
train_df = pd.read_csv("train.csv")
train_df = train_df.dropna()
train_df = shuffle(train_df)

In [15]:

# 简单中文分词清洗
stopwords = ['是','的','了','在','和','有','被','这','那','之','更','与','对于','并','我','他','她','它','我们','他们','她们','它们']
punc = r'~`!#$%^&*()_+-=|\';":/.,?><~·！@#￥%……&*（）——+-=“：’；、。，？》《{}'

def cleaning(text):
    cutwords = list(jieba.lcut_for_search(str(text)))
    final_cutwords = ''
    for word in cutwords:
        if word not in stopwords and word not in punc:
            final_cutwords += word + ' '
    return final_cutwords

train_df['title'] = train_df['title'].apply(cleaning)

# 准备数据
model_name = "bert-base-chinese"
max_length = 512
tokenizer = BertTokenizerFast.from_pretrained(model_name, do_lower_case=True)

texts = train_df['title'].tolist()
labels = train_df['label'].tolist()

train_texts, valid_texts, train_labels, valid_labels = train_test_split(texts, labels, test_size=0.2)

train_encodings = tokenizer(train_texts, truncation=True, padding=True, max_length=max_length)
valid_encodings = tokenizer(valid_texts, truncation=True, padding=True, max_length=max_length)

class NewsDataset(torch.utils.data.Dataset):
    def __init__(self, encodings, labels):
        self.encodings = encodings
        self.labels = labels
    def __getitem__(self, idx):
        item = {k: torch.tensor(v[idx]) for k, v in self.encodings.items()}
        item['labels'] = torch.tensor(self.labels[idx], dtype=torch.long)
        return item
    def __len__(self):
        return len(self.labels)

train_dataset = NewsDataset(train_encodings, train_labels)
valid_dataset = NewsDataset(valid_encodings, valid_labels)


In [None]:

# 模型
model = BertForSequenceClassification.from_pretrained(model_name, num_labels=2)

def compute_metrics(pred):
    labels = pred.label_ids
    preds = pred.predictions.argmax(-1)
    acc = accuracy_score(labels, preds)
    return {"accuracy": acc}

training_args = TrainingArguments(
    output_dir='./results',
    num_train_epochs=2,
    per_device_train_batch_size=16,
    per_device_eval_batch_size=32,
    warmup_steps=100,
    logging_dir='./logs',
    load_best_model_at_end=True,
    eval_strategy="steps",  
    logging_steps=368,
    save_steps=368,
)
#现在是最高那个模型的参数版本，result不是
trainer = Trainer(
    model=model,
    args=training_args,
    train_dataset=train_dataset,
    eval_dataset=valid_dataset,
    compute_metrics=compute_metrics,
)

trainer.train()



Some weights of BertForSequenceClassification were not initialized from the model checkpoint at bert-base-chinese and are newly initialized: ['classifier.bias', 'classifier.weight']
You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.


Step,Training Loss,Validation Loss,Accuracy
368,0.4805,0.258739,0.915646
736,0.2043,0.227137,0.936054




TrainOutput(global_step=736, training_loss=0.34238073100214417, metrics={'train_runtime': 6291.5751, 'train_samples_per_second': 0.934, 'train_steps_per_second': 0.117, 'total_flos': 851530152900960.0, 'train_loss': 0.34238073100214417, 'epoch': 2.0})

In [17]:

# 保存模型
model.save_pretrained('./cache/model')
tokenizer.save_pretrained('./cache/tokenizer')


('./cache/tokenizer\\tokenizer_config.json',
 './cache/tokenizer\\special_tokens_map.json',
 './cache/tokenizer\\vocab.txt',
 './cache/tokenizer\\added_tokens.json',
 './cache/tokenizer\\tokenizer.json')

In [18]:
from tqdm import tqdm
import pandas as pd
import numpy as np



In [19]:
# 推理函数：返回真实新闻的概率
def get_prob(text):
    inputs = tokenizer(text, padding=True, truncation=True, max_length=max_length, return_tensors="pt")
    outputs = model(**inputs)
    probs = outputs.logits.softmax(dim=1)
    return float(probs[0][1].cpu().detach().numpy())  # 概率为 label=1 (真实新闻)

# 读取测试集
test_df = pd.read_csv("test.csv")
test_df['title'] = test_df['title'].apply(cleaning)

# 预测 - 增加进度条
tqdm.pandas(desc="Processing")
test_df['prob'] = test_df['title'].progress_apply(get_prob)

# 保存结果
final_df = test_df[['id', 'prob']]
final_df.to_csv("result.csv", index=False)


Processing: 100%|██████████| 1224/1224 [01:44<00:00, 11.72it/s]
