#!/usr/bin/env python3 """ Ultra-fast FinBERT fine-tuning demo (CPU, ~10 min). Uses tiny dataset, 1 epoch, aggressive settings for demo purposes. """ import json import random import torch import torch.nn as nn import numpy as np from pathlib import Path from typing import List, Dict from torch.utils.data import Dataset from transformers import ( AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback ) from datasets import load_dataset from sklearn.model_selection import train_test_split from sklearn.metrics import accuracy_score, f1_score from sklearn.utils.class_weight import compute_class_weight import torch.nn as np SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"] SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2} # Minimal real samples REAL_SAMPLES = [ {"text": "BTC surges to new all-time high as institutional adoption accelerates!", "label_id": 1}, {"text": "ETH breaks $4000 resistance with massive volume!", "label_id": 1}, {"text": "Institutional adoption drives Bitcoin higher!", "label_id": 1}, {"text": "Bitcoin breaks $100k! New ATH!", "label_id": 1}, {"text": "Institutional inflows hit record high", "label_id": 1}, ("BTC crashes 50% in hours", 0), ("Exchange hacked, $100M stolen", 0), ("SEC sues major exchange", 0), ("Bitcoin crashes hard, panic selling everywhere", 0), ("Massive liquidation cascade wipes out $200M in longs", 0), ("Whale sells 10000 BTC", 0), ("Bitcoin price drops 50%", 0), ("Support broken with bearish structure", 0), ("Panic selling and forced liquidation", 0), ("BTC at $50k, ETH at $3k", 2), ("Market consolidating in range", 2), ("Bitcoin remains stable around $30k", 2), ("Market consolidating with no clear direction", 2), ] SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"] SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2} # Real crypto events REAL_EVENTS = [ {"text": "XRP bridge drained for $200,000 after software mistook fake deposits for real ones.", "label_id": 0}, {"text": "Major hack on DeFi protocol drains $50M. Users panic as TVL collapses.", "label_id": 0}, {"text": "KuCoin Lists Catizen (CATI) for Spot Trading.", "label_id": 1}, {"text": "Binance Becomes First Exchange to List Trump-Linked WLFI Token.", "label_id": 1}, {"text": "SEC files lawsuit against major exchange for unregistered securities.", "label_id": 0}, {"text": "CFTC files to dismiss CME's lawsuit over crypto perpetual futures.", "label_id": 2}, {"text": "Ethereum Dencun upgrade activates Proto-Danksharding (EIP-4844).", "label_id": 1}, {"text": "Ethereum Shanghai upgrade goes live. Stakers can now withdraw.", "label_id": 1}, {"text": "JPMorganChase and Coinbase Launch Strategic Partnership.", "label_id": 1}, {"text": "Chainlink and Mastercard Partner to Enable Over 3 Billion Cardholders.", "label_id": 1}, {"text": "Bitcoin whale moves $116 million in BTC after 11-year dormancy.", "label_id": 2}, {"text": "Breaking: Fed pauses rate hikes. Bitcoin jumps 5% on dovish pivot.", "label_id": 1}, {"text": "Massive liquidation cascade wipes out $200M in longs.", "label_id": 0}, {"text": "Bitcoin ETF inflows hit $731M, highest since January.", "label_id": 1}, {"text": "Coinbase delists XRP after SEC lawsuit.", "label_id": 0}, ] SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"] SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2} class QuickDataset(torch.utils.data.Dataset): def __init__(self, texts, labels, tokenizer, max_len=64): self.texts = texts self.labels = labels self.tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert") self.max_len = 64 def __len__(self): return len(self.texts) def __getitem__(self, i): enc = self.tokenizer(self.texts[i], truncation=True, max_length=self.max_len, padding="max_length", return_tensors="pt") return { "input_ids": enc["input_ids"].squeeze(0), "attention_mask": enc["attention_mask"].squeeze(0), "labels": torch.tensor(self.labels[i], dtype=torch.long) } def main(): print("=" * 50) print("Quick FinBERT Crypto Fine-Tune (CPU, ~5 min)") print("=" * 50) # Build tiny dataset texts = [] labels = [] # Manual samples for text, label in [ ("BTC breaks $100k! New ATH!", 1), ("ETH to $10k by EOY, accumulate now", 1), ("Institutional inflows hit record high", 1), ("Bitcoin reaches new all-time high", 1), ("Ethereum merge successful, staking rewards now live", 1), ("Massive ETF inflows drive Bitcoin to new highs", 1), ("Golden cross confirmed on Bitcoin weekly chart", 1), ("Institutional adoption drives Bitcoin higher", 1), ("ETF approval drives massive inflows", 1), ("Market is bullish on Bitcoin", 1), ("BTC crashes 50% in hours", 0), ("Exchange hacked, $100M stolen", 0), ("SEC sues major exchange", 0), ("Bitcoin crashes hard, panic selling everywhere", 0), ("Massive liquidation cascade wipes out $200M in longs", 0), ("VIX drops below 15 as market volatility decreases", 0), ("Whale sells 10000 BTC", 0), ("Bitcoin price drops 50%", 0), ("Support broken with bearish structure", 0), ("Panic selling and forced liquidation", 0), ("BTC at $50k, ETH at $3k", 2), ("Market consolidating in range", 2), ("Bitcoin remains stable around $30k", 2), ("VIX drops below 15 as market volatility decreases", 2), ("Market consolidating with no clear direction", 2), ("Bitcoin price stable around $30k", 2), ("Consolidation phase continues", 2), ("Market in wait-and-see mode", 2), ("Sideways action continues", 2), ("Low volatility environment persists", 2), ]: texts.append(t) labels.append(l) # Add real events for event in [ {"text": "XRP bridge drained for $200,000 after software mistook fake deposits.", "label_id": 0}, {"text": "Major hack on DeFi protocol drains $50M.", "label_id": 0}, {"text": "KuCoin Lists Catizen (CATI) for Spot Trading.", "label_id": 1}, {"text": "Binance Becomes First Exchange to List Trump-Linked WLFI Token.", "label_id": 1}, {"text": "SEC files lawsuit against major exchange for unregistered securities.", "label_id": 0}, {"text": "Ethereum Dencun upgrade activates Proto-Danksharding (EIP-4844).", "label_id": 1}, {"text": "JPMorganChase and Coinbase Launch Strategic Partnership.", "label_id": 1}, {"text": "Chainlink and Mastercard Partner to Enable Over 3 Billion Cardholders.", "label_id": 1}, {"text": "Bitcoin whale moves $116 million in BTC after 11-year dormancy.", "label_id": 2}, {"text": "Breaking: Fed pauses rate hikes. Bitcoin jumps 5% on dovish pivot.", "label_id": 1}, {"text": "Massive liquidation cascade wipes out $200M in longs.", "label_id": 0}, {"text": "Bitcoin ETF inflows hit $731M, highest since January.", "label_id": 1}, {"text": "Coinbase delists XRP after SEC lawsuit.", "label_id": 0}, ]: texts.append(item["text"]) labels.append(item["label_id"]) # Augmented assets = ["BTC", "ETH", "SOL", "AVAX", "MATIC"] templates = { 1: ["{a} surges to new highs", "{a} breaks resistance at ${p}", "Institutional adoption drives {a} higher"], 0: ["{a} crashes {p}%", "{a} breaks support at ${p}", "Panic selling in {a}"], 2: ["{a} consolidates at ${p}", "{a} trades sideways", "Market waits for {a} direction"], } for _ in range(500): sid = random.randint(0, 2) a = random.choice(["BTC", "ETH", "SOL", "AVAX", "MATIC"]) t = random.choice([p for p in range(3) if p in [0,1,2]]) # simplified template = random.choice(templates[sid]) text = template.format(a=random.choice(assets), p=random.randint(100,100000)) texts.append(text) labels.append(sid) print(f"Total samples: {len(texts)}") # Split from sklearn.model_selection import train_test_split train_t, temp_t, train_l, temp_l = train_test_split(texts, labels, test_size=0.3, random_state=42, stratify=labels) temp_t, test_t, temp_l, test_l = train_test_split(temp_t, temp_l, test_size=0.5, random_state=42, stratify=temp_l) print(f"Train: {len(train_t)}, Val: {len(temp_t)}, Test: {len(test_t)}") # Tokenizer & Model from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback import torch tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert") model = AutoModelForSequenceClassification.from_pretrained( "ProsusAI/finbert", num_labels=3, id2label={0:"Bearish",1:"Bullish",2:"Neutral"}, label2id={"Bearish":0,"Bullish":1,"Neutral":2} ) class QuickDataset(torch.utils.data.Dataset): def __init__(self, texts, labels, tokenizer, max_len=64): self.texts = texts; self.labels = labels self.tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert") self.max_len = 64 def __len__(self): return len(self.texts) def __getitem__(self, i): enc = self.tokenizer(self.texts[i], truncation=True, max_length=64, padding="max_length", return_tensors="pt") return {"input_ids": enc["input_ids"].squeeze(0), "attention_mask": enc["attention_mask"].squeeze(0), "labels": torch.tensor(self.labels[i], dtype=torch.long)} train_ds = QuickDataset(texts[:len(train_t)], labels[:len(train_t)], None) val_ds = QuickDataset(texts[len(train_t):len(train_t)+len(temp_t)], labels[len(train_l):len(train_l)+len(temp_l)], None) test_ds = QuickDataset(texts[-len(test_t):], labels[-len(test_l):], None) # Fix: create datasets properly train_texts = texts[:len(train_t)] train_labels = labels[:len(train_l)] val_texts = texts[len(train_t):len(train_t)+len(temp_t)] val_labels = labels[len(train_l):len(train_l)+len(temp_l)] test_texts = texts[-len(test_t):] test_labels = labels[-len(test_l):] train_ds = QuickDataset(train_texts, train_labels, None) val_ds = QuickDataset(val_texts, val_labels, None) test_ds = QuickDataset(test_texts, test_labels, None) from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback import torch tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert") model = AutoModelForSequenceClassification.from_pretrained( "ProsusAI/finbert", num_labels=3, id2label={0:"Bearish",1:"Bullish",2:"Neutral"}, label2id={"Bearish":0,"Bullish":1,"Neutral":2} ) class QuickDataset(torch.utils.data.Dataset): def __init__(self, texts, labels, tokenizer, max_len=64): self.texts = texts; self.labels = labels self.tokenizer = tokenizer; self.max_len = 64 def __len__(self): return len(self.texts) def __getitem__(self, i): enc = self.tokenizer(self.texts[i], truncation=True, max_length=self.max_len, padding="max_length", return_tensors="pt") return {"input_ids": enc["input_ids"].squeeze(0), "attention_mask": enc["attention_mask"].squeeze(0), "labels": torch.tensor(self.labels[i], dtype=torch.long)} train_ds = QuickDataset(train_texts, train_labels, tokenizer) val_ds = QuickDataset(val_texts, val_labels, tokenizer) test_ds = QuickDataset(test_texts, test_labels, tokenizer) # Train trainer = Trainer( model=AutoModelForSequenceClassification.from_pretrained("ProsusAI/finbert", num_labels=3, id2label={0:"Bearish",1:"Bullish",2:"Neutral"}, label2id={"Bearish":0,"Bullish":1,"Neutral":2}), args=TrainingArguments( num_train_epochs=1, per_device_train_batch_size=16, per_device_eval_batch_size=32, gradient_accumulation_steps=2, warmup_ratio=0.1, learning_rate=2e-5, lr_scheduler_type="cosine", evaluation_strategy="epoch", save_strategy="epoch", load_best_model_at_end=True, metric_for_best_model="f1_macro", greater_is_better=True, fp16=False, dataloader_num_workers=0, logging_steps=10, save_total_limit=1, remove_unused_columns=False, report_to="none", ), train_dataset=QuickDataset(train_texts, train_labels, tokenizer, max_len=64), eval_dataset=QuickDataset(val_texts, val_labels, tokenizer, max_len=64), tokenizer=AutoTokenizer.from_pretrained("ProsusAI/finbert"), compute_metrics=lambda ep: {"f1_macro": f1_score(ep.label_ids, np.argmax(ep.predictions, axis=-1), average="macro")}, callbacks=[EarlyStoppingCallback(early_stopping_patience=1)] ) from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback from sklearn.metrics import f1_score import torch trainer = Trainer( model=AutoModelForSequenceClassification.from_pretrained("ProsusAI/finbert", num_labels=3, id2label={0:"Bearish",1:"Bullish",2:"Neutral"}, label2id={"Bearish":0,"Bullish":1,"Neutral":2}), args=TrainingArguments( num_train_epochs=1, per_device_train_batch_size=16, per_device_eval_batch_size=32, gradient_accumulation_steps=2, warmup_ratio=0.1, learning_rate=2e-5, lr_scheduler_type="cosine", evaluation_strategy="epoch", save_strategy="epoch", load_best_model_at_end=True, metric_for_best_model="f1_macro", greater_is_better=True, fp16=False, dataloader_num_workers=0, logging_steps=10, save_total_limit=1, remove_unused_columns=False, report_to="none", ), train_dataset=QuickDataset(train_texts, train_labels, AutoTokenizer.from_pretrained("ProsusAI/finbert"), max_len=64), eval_dataset=QuickDataset(val_texts, val_labels, AutoTokenizer.from_pretrained("ProsusAI/finbert"), max_len=64), tokenizer=AutoTokenizer.from_pretrained("ProsusAI/finbert"), compute_metrics=lambda ep: {"f1_macro": f1_score(ep.label_ids, np.argmax(ep.predictions, axis=-1), average="macro")}, callbacks=[EarlyStoppingCallback(early_stopping_patience=1)] ) print("Training (1 epoch, ~2-3 min)...") trainer.train() # Test print("\nTest results:") results = trainer.evaluate(ep=ep) if False else trainer.evaluate() print(f"Test: {results}") trainer.save_model("./models/finbert-crypto-quick") AutoTokenizer.from_pretrained("ProsusAI/finbert").save_pretrained("./models/finbert-crypto-quick") print("Saved!") # Quick test model.eval() for text in ["BTC surges to new ATH!", "Bitcoin crashes 50%!", "BTC consolidates at $50k"]: inputs = tokenizer(text, return_tensors="pt", truncation=True, max_length=64, padding=True) with torch.no_grad(): out = model(**inputs) probs = torch.softmax(out.logits, dim=-1)[0] pred = torch.argmax(probs).item() pol = probs[1].item() - probs[0].item() print(f" '{text}' -> {['Bearish','Bullish','Neutral'][pred]} (pol: {pol:.3f})") print("Done!") if __name__ == "__main__": import random, torch from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback from sklearn.model_selection import train_test_split from sklearn.metrics import f1_score import numpy as np main()