Files
sentiment-engine/sentiment_engine/training/finetune_finbert_quick.py

339 lines
16 KiB
Python
Raw Permalink Normal View History

#!/usr/bin/env python3
"""
Ultra-fast FinBERT fine-tuning demo (CPU, ~10 min).
Uses tiny dataset, 1 epoch, aggressive settings for demo purposes.
"""
import json
import random
import torch
import torch.nn as nn
import numpy as np
from pathlib import Path
from typing import List, Dict
from torch.utils.data import Dataset
from transformers import (
AutoTokenizer, AutoModelForSequenceClassification,
TrainingArguments, Trainer, EarlyStoppingCallback
)
from datasets import load_dataset
from sklearn.model_selection import train_test_split
from sklearn.metrics import accuracy_score, f1_score
from sklearn.utils.class_weight import compute_class_weight
import torch.nn as np
SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"]
SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2}
# Minimal real samples
REAL_SAMPLES = [
{"text": "BTC surges to new all-time high as institutional adoption accelerates!", "label_id": 1},
{"text": "ETH breaks $4000 resistance with massive volume!", "label_id": 1},
{"text": "Institutional adoption drives Bitcoin higher!", "label_id": 1},
{"text": "Bitcoin breaks $100k! New ATH!", "label_id": 1},
{"text": "Institutional inflows hit record high", "label_id": 1},
("BTC crashes 50% in hours", 0),
("Exchange hacked, $100M stolen", 0),
("SEC sues major exchange", 0),
("Bitcoin crashes hard, panic selling everywhere", 0),
("Massive liquidation cascade wipes out $200M in longs", 0),
("Whale sells 10000 BTC", 0),
("Bitcoin price drops 50%", 0),
("Support broken with bearish structure", 0),
("Panic selling and forced liquidation", 0),
("BTC at $50k, ETH at $3k", 2),
("Market consolidating in range", 2),
("Bitcoin remains stable around $30k", 2),
("Market consolidating with no clear direction", 2),
]
SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"]
SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2}
# Real crypto events
REAL_EVENTS = [
{"text": "XRP bridge drained for $200,000 after software mistook fake deposits for real ones.", "label_id": 0},
{"text": "Major hack on DeFi protocol drains $50M. Users panic as TVL collapses.", "label_id": 0},
{"text": "KuCoin Lists Catizen (CATI) for Spot Trading.", "label_id": 1},
{"text": "Binance Becomes First Exchange to List Trump-Linked WLFI Token.", "label_id": 1},
{"text": "SEC files lawsuit against major exchange for unregistered securities.", "label_id": 0},
{"text": "CFTC files to dismiss CME's lawsuit over crypto perpetual futures.", "label_id": 2},
{"text": "Ethereum Dencun upgrade activates Proto-Danksharding (EIP-4844).", "label_id": 1},
{"text": "Ethereum Shanghai upgrade goes live. Stakers can now withdraw.", "label_id": 1},
{"text": "JPMorganChase and Coinbase Launch Strategic Partnership.", "label_id": 1},
{"text": "Chainlink and Mastercard Partner to Enable Over 3 Billion Cardholders.", "label_id": 1},
{"text": "Bitcoin whale moves $116 million in BTC after 11-year dormancy.", "label_id": 2},
{"text": "Breaking: Fed pauses rate hikes. Bitcoin jumps 5% on dovish pivot.", "label_id": 1},
{"text": "Massive liquidation cascade wipes out $200M in longs.", "label_id": 0},
{"text": "Bitcoin ETF inflows hit $731M, highest since January.", "label_id": 1},
{"text": "Coinbase delists XRP after SEC lawsuit.", "label_id": 0},
]
SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"]
SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2}
class QuickDataset(torch.utils.data.Dataset):
def __init__(self, texts, labels, tokenizer, max_len=64):
self.texts = texts
self.labels = labels
self.tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert")
self.max_len = 64
def __len__(self): return len(self.texts)
def __getitem__(self, i):
enc = self.tokenizer(self.texts[i], truncation=True, max_length=self.max_len,
padding="max_length", return_tensors="pt")
return {
"input_ids": enc["input_ids"].squeeze(0),
"attention_mask": enc["attention_mask"].squeeze(0),
"labels": torch.tensor(self.labels[i], dtype=torch.long)
}
def main():
print("=" * 50)
print("Quick FinBERT Crypto Fine-Tune (CPU, ~5 min)")
print("=" * 50)
# Build tiny dataset
texts = []
labels = []
# Manual samples
for text, label in [
("BTC breaks $100k! New ATH!", 1),
("ETH to $10k by EOY, accumulate now", 1),
("Institutional inflows hit record high", 1),
("Bitcoin reaches new all-time high", 1),
("Ethereum merge successful, staking rewards now live", 1),
("Massive ETF inflows drive Bitcoin to new highs", 1),
("Golden cross confirmed on Bitcoin weekly chart", 1),
("Institutional adoption drives Bitcoin higher", 1),
("ETF approval drives massive inflows", 1),
("Market is bullish on Bitcoin", 1),
("BTC crashes 50% in hours", 0),
("Exchange hacked, $100M stolen", 0),
("SEC sues major exchange", 0),
("Bitcoin crashes hard, panic selling everywhere", 0),
("Massive liquidation cascade wipes out $200M in longs", 0),
("VIX drops below 15 as market volatility decreases", 0),
("Whale sells 10000 BTC", 0),
("Bitcoin price drops 50%", 0),
("Support broken with bearish structure", 0),
("Panic selling and forced liquidation", 0),
("BTC at $50k, ETH at $3k", 2),
("Market consolidating in range", 2),
("Bitcoin remains stable around $30k", 2),
("VIX drops below 15 as market volatility decreases", 2),
("Market consolidating with no clear direction", 2),
("Bitcoin price stable around $30k", 2),
("Consolidation phase continues", 2),
("Market in wait-and-see mode", 2),
("Sideways action continues", 2),
("Low volatility environment persists", 2),
]:
texts.append(t)
labels.append(l)
# Add real events
for event in [
{"text": "XRP bridge drained for $200,000 after software mistook fake deposits.", "label_id": 0},
{"text": "Major hack on DeFi protocol drains $50M.", "label_id": 0},
{"text": "KuCoin Lists Catizen (CATI) for Spot Trading.", "label_id": 1},
{"text": "Binance Becomes First Exchange to List Trump-Linked WLFI Token.", "label_id": 1},
{"text": "SEC files lawsuit against major exchange for unregistered securities.", "label_id": 0},
{"text": "Ethereum Dencun upgrade activates Proto-Danksharding (EIP-4844).", "label_id": 1},
{"text": "JPMorganChase and Coinbase Launch Strategic Partnership.", "label_id": 1},
{"text": "Chainlink and Mastercard Partner to Enable Over 3 Billion Cardholders.", "label_id": 1},
{"text": "Bitcoin whale moves $116 million in BTC after 11-year dormancy.", "label_id": 2},
{"text": "Breaking: Fed pauses rate hikes. Bitcoin jumps 5% on dovish pivot.", "label_id": 1},
{"text": "Massive liquidation cascade wipes out $200M in longs.", "label_id": 0},
{"text": "Bitcoin ETF inflows hit $731M, highest since January.", "label_id": 1},
{"text": "Coinbase delists XRP after SEC lawsuit.", "label_id": 0},
]:
texts.append(item["text"])
labels.append(item["label_id"])
# Augmented
assets = ["BTC", "ETH", "SOL", "AVAX", "MATIC"]
templates = {
1: ["{a} surges to new highs", "{a} breaks resistance at ${p}", "Institutional adoption drives {a} higher"],
0: ["{a} crashes {p}%", "{a} breaks support at ${p}", "Panic selling in {a}"],
2: ["{a} consolidates at ${p}", "{a} trades sideways", "Market waits for {a} direction"],
}
for _ in range(500):
sid = random.randint(0, 2)
a = random.choice(["BTC", "ETH", "SOL", "AVAX", "MATIC"])
t = random.choice([p for p in range(3) if p in [0,1,2]]) # simplified
template = random.choice(templates[sid])
text = template.format(a=random.choice(assets), p=random.randint(100,100000))
texts.append(text)
labels.append(sid)
print(f"Total samples: {len(texts)}")
# Split
from sklearn.model_selection import train_test_split
train_t, temp_t, train_l, temp_l = train_test_split(texts, labels, test_size=0.3, random_state=42, stratify=labels)
temp_t, test_t, temp_l, test_l = train_test_split(temp_t, temp_l, test_size=0.5, random_state=42, stratify=temp_l)
print(f"Train: {len(train_t)}, Val: {len(temp_t)}, Test: {len(test_t)}")
# Tokenizer & Model
from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback
import torch
tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert")
model = AutoModelForSequenceClassification.from_pretrained(
"ProsusAI/finbert", num_labels=3,
id2label={0:"Bearish",1:"Bullish",2:"Neutral"},
label2id={"Bearish":0,"Bullish":1,"Neutral":2}
)
class QuickDataset(torch.utils.data.Dataset):
def __init__(self, texts, labels, tokenizer, max_len=64):
self.texts = texts; self.labels = labels
self.tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert")
self.max_len = 64
def __len__(self): return len(self.texts)
def __getitem__(self, i):
enc = self.tokenizer(self.texts[i], truncation=True, max_length=64, padding="max_length", return_tensors="pt")
return {"input_ids": enc["input_ids"].squeeze(0), "attention_mask": enc["attention_mask"].squeeze(0), "labels": torch.tensor(self.labels[i], dtype=torch.long)}
train_ds = QuickDataset(texts[:len(train_t)], labels[:len(train_t)], None)
val_ds = QuickDataset(texts[len(train_t):len(train_t)+len(temp_t)], labels[len(train_l):len(train_l)+len(temp_l)], None)
test_ds = QuickDataset(texts[-len(test_t):], labels[-len(test_l):], None)
# Fix: create datasets properly
train_texts = texts[:len(train_t)]
train_labels = labels[:len(train_l)]
val_texts = texts[len(train_t):len(train_t)+len(temp_t)]
val_labels = labels[len(train_l):len(train_l)+len(temp_l)]
test_texts = texts[-len(test_t):]
test_labels = labels[-len(test_l):]
train_ds = QuickDataset(train_texts, train_labels, None)
val_ds = QuickDataset(val_texts, val_labels, None)
test_ds = QuickDataset(test_texts, test_labels, None)
from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback
import torch
tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert")
model = AutoModelForSequenceClassification.from_pretrained(
"ProsusAI/finbert", num_labels=3,
id2label={0:"Bearish",1:"Bullish",2:"Neutral"},
label2id={"Bearish":0,"Bullish":1,"Neutral":2}
)
class QuickDataset(torch.utils.data.Dataset):
def __init__(self, texts, labels, tokenizer, max_len=64):
self.texts = texts; self.labels = labels
self.tokenizer = tokenizer; self.max_len = 64
def __len__(self): return len(self.texts)
def __getitem__(self, i):
enc = self.tokenizer(self.texts[i], truncation=True, max_length=self.max_len, padding="max_length", return_tensors="pt")
return {"input_ids": enc["input_ids"].squeeze(0), "attention_mask": enc["attention_mask"].squeeze(0), "labels": torch.tensor(self.labels[i], dtype=torch.long)}
train_ds = QuickDataset(train_texts, train_labels, tokenizer)
val_ds = QuickDataset(val_texts, val_labels, tokenizer)
test_ds = QuickDataset(test_texts, test_labels, tokenizer)
# Train
trainer = Trainer(
model=AutoModelForSequenceClassification.from_pretrained("ProsusAI/finbert", num_labels=3, id2label={0:"Bearish",1:"Bullish",2:"Neutral"}, label2id={"Bearish":0,"Bullish":1,"Neutral":2}),
args=TrainingArguments(
num_train_epochs=1,
per_device_train_batch_size=16,
per_device_eval_batch_size=32,
gradient_accumulation_steps=2,
warmup_ratio=0.1,
learning_rate=2e-5,
lr_scheduler_type="cosine",
evaluation_strategy="epoch",
save_strategy="epoch",
load_best_model_at_end=True,
metric_for_best_model="f1_macro",
greater_is_better=True,
fp16=False,
dataloader_num_workers=0,
logging_steps=10,
save_total_limit=1,
remove_unused_columns=False,
report_to="none",
),
train_dataset=QuickDataset(train_texts, train_labels, tokenizer, max_len=64),
eval_dataset=QuickDataset(val_texts, val_labels, tokenizer, max_len=64),
tokenizer=AutoTokenizer.from_pretrained("ProsusAI/finbert"),
compute_metrics=lambda ep: {"f1_macro": f1_score(ep.label_ids, np.argmax(ep.predictions, axis=-1), average="macro")},
callbacks=[EarlyStoppingCallback(early_stopping_patience=1)]
)
from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback
from sklearn.metrics import f1_score
import torch
trainer = Trainer(
model=AutoModelForSequenceClassification.from_pretrained("ProsusAI/finbert", num_labels=3, id2label={0:"Bearish",1:"Bullish",2:"Neutral"}, label2id={"Bearish":0,"Bullish":1,"Neutral":2}),
args=TrainingArguments(
num_train_epochs=1,
per_device_train_batch_size=16,
per_device_eval_batch_size=32,
gradient_accumulation_steps=2,
warmup_ratio=0.1,
learning_rate=2e-5,
lr_scheduler_type="cosine",
evaluation_strategy="epoch",
save_strategy="epoch",
load_best_model_at_end=True,
metric_for_best_model="f1_macro",
greater_is_better=True,
fp16=False,
dataloader_num_workers=0,
logging_steps=10,
save_total_limit=1,
remove_unused_columns=False,
report_to="none",
),
train_dataset=QuickDataset(train_texts, train_labels, AutoTokenizer.from_pretrained("ProsusAI/finbert"), max_len=64),
eval_dataset=QuickDataset(val_texts, val_labels, AutoTokenizer.from_pretrained("ProsusAI/finbert"), max_len=64),
tokenizer=AutoTokenizer.from_pretrained("ProsusAI/finbert"),
compute_metrics=lambda ep: {"f1_macro": f1_score(ep.label_ids, np.argmax(ep.predictions, axis=-1), average="macro")},
callbacks=[EarlyStoppingCallback(early_stopping_patience=1)]
)
print("Training (1 epoch, ~2-3 min)...")
trainer.train()
# Test
print("\nTest results:")
results = trainer.evaluate(ep=ep) if False else trainer.evaluate()
print(f"Test: {results}")
trainer.save_model("./models/finbert-crypto-quick")
AutoTokenizer.from_pretrained("ProsusAI/finbert").save_pretrained("./models/finbert-crypto-quick")
print("Saved!")
# Quick test
model.eval()
for text in ["BTC surges to new ATH!", "Bitcoin crashes 50%!", "BTC consolidates at $50k"]:
inputs = tokenizer(text, return_tensors="pt", truncation=True, max_length=64, padding=True)
with torch.no_grad():
out = model(**inputs)
probs = torch.softmax(out.logits, dim=-1)[0]
pred = torch.argmax(probs).item()
pol = probs[1].item() - probs[0].item()
print(f" '{text}' -> {['Bearish','Bullish','Neutral'][pred]} (pol: {pol:.3f})")
print("Done!")
if __name__ == "__main__":
import random, torch
from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback
from sklearn.model_selection import train_test_split
from sklearn.metrics import f1_score
import numpy as np
main()