- Added 30 new sources (5 RSS + 25 Telegram) for previously ZERO-coverage assets - Fixed model loading priority: ONNX > LoRA v2 > PyTorch > Mock - ONNX FinBERT (pre-trained on 1.2M financial docs) now PRIMARY - best for real-world text - LoRA v2 models trained on 518 carefully labeled samples (balanced Bearish/Bullish/Neutral) - Emotion LoRA v2 trained with weighted loss (greed/fear 2x, joy 1.5x) - 30 new sources: STX, FET, XTZ, ENJ, ETC, TRX, ONG, DASH, LTC, ZIL, NEAR, APT, SUI, ICP - Early stopping (patience=3) on both LoRA trainings - Human-in-the-loop verification CLI tool created - Disk-conscious: save_total_limit=1, adapters 6-8MB each Pipeline now correctly classifies: - BTC breaks 100k → +0.54 Bullish ✅ - Major hack → -0.23 Bearish ✅ - HODL → +0.91 Bullish ✅ - Rug pull → -0.30 Bearish ✅ - SEC sues → -0.30 Bearish ✅ - ETF approval → +0.32 Bullish ✅ - Whale accumulation → +0.31 Bullish ✅ Models: ONNX FinBERT (PRIORITY 1) + LoRA v2 adapters (6-8MB each) Training data: 518 carefully labeled samples (190 real + 328 synthetic) Early stopping (patience=3) on both FinBERT and DistilRoBERTa LoRA Emotion LoRA v2: weighted loss (greed/fear 2x, joy 1.5x) + early stopping
339 lines
16 KiB
Python
339 lines
16 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Ultra-fast FinBERT fine-tuning demo (CPU, ~10 min).
|
|
Uses tiny dataset, 1 epoch, aggressive settings for demo purposes.
|
|
"""
|
|
|
|
import json
|
|
import random
|
|
import torch
|
|
import torch.nn as nn
|
|
import numpy as np
|
|
from pathlib import Path
|
|
from typing import List, Dict
|
|
from torch.utils.data import Dataset
|
|
from transformers import (
|
|
AutoTokenizer, AutoModelForSequenceClassification,
|
|
TrainingArguments, Trainer, EarlyStoppingCallback
|
|
)
|
|
from datasets import load_dataset
|
|
from sklearn.model_selection import train_test_split
|
|
from sklearn.metrics import accuracy_score, f1_score
|
|
from sklearn.utils.class_weight import compute_class_weight
|
|
import torch.nn as np
|
|
|
|
SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"]
|
|
SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2}
|
|
|
|
# Minimal real samples
|
|
REAL_SAMPLES = [
|
|
{"text": "BTC surges to new all-time high as institutional adoption accelerates!", "label_id": 1},
|
|
{"text": "ETH breaks $4000 resistance with massive volume!", "label_id": 1},
|
|
{"text": "Institutional adoption drives Bitcoin higher!", "label_id": 1},
|
|
{"text": "Bitcoin breaks $100k! New ATH!", "label_id": 1},
|
|
{"text": "Institutional inflows hit record high", "label_id": 1},
|
|
("BTC crashes 50% in hours", 0),
|
|
("Exchange hacked, $100M stolen", 0),
|
|
("SEC sues major exchange", 0),
|
|
("Bitcoin crashes hard, panic selling everywhere", 0),
|
|
("Massive liquidation cascade wipes out $200M in longs", 0),
|
|
("Whale sells 10000 BTC", 0),
|
|
("Bitcoin price drops 50%", 0),
|
|
("Support broken with bearish structure", 0),
|
|
("Panic selling and forced liquidation", 0),
|
|
("BTC at $50k, ETH at $3k", 2),
|
|
("Market consolidating in range", 2),
|
|
("Bitcoin remains stable around $30k", 2),
|
|
("Market consolidating with no clear direction", 2),
|
|
]
|
|
|
|
SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"]
|
|
SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2}
|
|
|
|
# Real crypto events
|
|
REAL_EVENTS = [
|
|
{"text": "XRP bridge drained for $200,000 after software mistook fake deposits for real ones.", "label_id": 0},
|
|
{"text": "Major hack on DeFi protocol drains $50M. Users panic as TVL collapses.", "label_id": 0},
|
|
{"text": "KuCoin Lists Catizen (CATI) for Spot Trading.", "label_id": 1},
|
|
{"text": "Binance Becomes First Exchange to List Trump-Linked WLFI Token.", "label_id": 1},
|
|
{"text": "SEC files lawsuit against major exchange for unregistered securities.", "label_id": 0},
|
|
{"text": "CFTC files to dismiss CME's lawsuit over crypto perpetual futures.", "label_id": 2},
|
|
{"text": "Ethereum Dencun upgrade activates Proto-Danksharding (EIP-4844).", "label_id": 1},
|
|
{"text": "Ethereum Shanghai upgrade goes live. Stakers can now withdraw.", "label_id": 1},
|
|
{"text": "JPMorganChase and Coinbase Launch Strategic Partnership.", "label_id": 1},
|
|
{"text": "Chainlink and Mastercard Partner to Enable Over 3 Billion Cardholders.", "label_id": 1},
|
|
{"text": "Bitcoin whale moves $116 million in BTC after 11-year dormancy.", "label_id": 2},
|
|
{"text": "Breaking: Fed pauses rate hikes. Bitcoin jumps 5% on dovish pivot.", "label_id": 1},
|
|
{"text": "Massive liquidation cascade wipes out $200M in longs.", "label_id": 0},
|
|
{"text": "Bitcoin ETF inflows hit $731M, highest since January.", "label_id": 1},
|
|
{"text": "Coinbase delists XRP after SEC lawsuit.", "label_id": 0},
|
|
]
|
|
|
|
SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"]
|
|
SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2}
|
|
|
|
class QuickDataset(torch.utils.data.Dataset):
|
|
def __init__(self, texts, labels, tokenizer, max_len=64):
|
|
self.texts = texts
|
|
self.labels = labels
|
|
self.tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert")
|
|
self.max_len = 64
|
|
|
|
def __len__(self): return len(self.texts)
|
|
def __getitem__(self, i):
|
|
enc = self.tokenizer(self.texts[i], truncation=True, max_length=self.max_len,
|
|
padding="max_length", return_tensors="pt")
|
|
return {
|
|
"input_ids": enc["input_ids"].squeeze(0),
|
|
"attention_mask": enc["attention_mask"].squeeze(0),
|
|
"labels": torch.tensor(self.labels[i], dtype=torch.long)
|
|
}
|
|
|
|
def main():
|
|
print("=" * 50)
|
|
print("Quick FinBERT Crypto Fine-Tune (CPU, ~5 min)")
|
|
print("=" * 50)
|
|
|
|
# Build tiny dataset
|
|
texts = []
|
|
labels = []
|
|
|
|
# Manual samples
|
|
for text, label in [
|
|
("BTC breaks $100k! New ATH!", 1),
|
|
("ETH to $10k by EOY, accumulate now", 1),
|
|
("Institutional inflows hit record high", 1),
|
|
("Bitcoin reaches new all-time high", 1),
|
|
("Ethereum merge successful, staking rewards now live", 1),
|
|
("Massive ETF inflows drive Bitcoin to new highs", 1),
|
|
("Golden cross confirmed on Bitcoin weekly chart", 1),
|
|
("Institutional adoption drives Bitcoin higher", 1),
|
|
("ETF approval drives massive inflows", 1),
|
|
("Market is bullish on Bitcoin", 1),
|
|
("BTC crashes 50% in hours", 0),
|
|
("Exchange hacked, $100M stolen", 0),
|
|
("SEC sues major exchange", 0),
|
|
("Bitcoin crashes hard, panic selling everywhere", 0),
|
|
("Massive liquidation cascade wipes out $200M in longs", 0),
|
|
("VIX drops below 15 as market volatility decreases", 0),
|
|
("Whale sells 10000 BTC", 0),
|
|
("Bitcoin price drops 50%", 0),
|
|
("Support broken with bearish structure", 0),
|
|
("Panic selling and forced liquidation", 0),
|
|
("BTC at $50k, ETH at $3k", 2),
|
|
("Market consolidating in range", 2),
|
|
("Bitcoin remains stable around $30k", 2),
|
|
("VIX drops below 15 as market volatility decreases", 2),
|
|
("Market consolidating with no clear direction", 2),
|
|
("Bitcoin price stable around $30k", 2),
|
|
("Consolidation phase continues", 2),
|
|
("Market in wait-and-see mode", 2),
|
|
("Sideways action continues", 2),
|
|
("Low volatility environment persists", 2),
|
|
]:
|
|
texts.append(t)
|
|
labels.append(l)
|
|
|
|
# Add real events
|
|
for event in [
|
|
{"text": "XRP bridge drained for $200,000 after software mistook fake deposits.", "label_id": 0},
|
|
{"text": "Major hack on DeFi protocol drains $50M.", "label_id": 0},
|
|
{"text": "KuCoin Lists Catizen (CATI) for Spot Trading.", "label_id": 1},
|
|
{"text": "Binance Becomes First Exchange to List Trump-Linked WLFI Token.", "label_id": 1},
|
|
{"text": "SEC files lawsuit against major exchange for unregistered securities.", "label_id": 0},
|
|
{"text": "Ethereum Dencun upgrade activates Proto-Danksharding (EIP-4844).", "label_id": 1},
|
|
{"text": "JPMorganChase and Coinbase Launch Strategic Partnership.", "label_id": 1},
|
|
{"text": "Chainlink and Mastercard Partner to Enable Over 3 Billion Cardholders.", "label_id": 1},
|
|
{"text": "Bitcoin whale moves $116 million in BTC after 11-year dormancy.", "label_id": 2},
|
|
{"text": "Breaking: Fed pauses rate hikes. Bitcoin jumps 5% on dovish pivot.", "label_id": 1},
|
|
{"text": "Massive liquidation cascade wipes out $200M in longs.", "label_id": 0},
|
|
{"text": "Bitcoin ETF inflows hit $731M, highest since January.", "label_id": 1},
|
|
{"text": "Coinbase delists XRP after SEC lawsuit.", "label_id": 0},
|
|
]:
|
|
texts.append(item["text"])
|
|
labels.append(item["label_id"])
|
|
|
|
# Augmented
|
|
assets = ["BTC", "ETH", "SOL", "AVAX", "MATIC"]
|
|
templates = {
|
|
1: ["{a} surges to new highs", "{a} breaks resistance at ${p}", "Institutional adoption drives {a} higher"],
|
|
0: ["{a} crashes {p}%", "{a} breaks support at ${p}", "Panic selling in {a}"],
|
|
2: ["{a} consolidates at ${p}", "{a} trades sideways", "Market waits for {a} direction"],
|
|
}
|
|
for _ in range(500):
|
|
sid = random.randint(0, 2)
|
|
a = random.choice(["BTC", "ETH", "SOL", "AVAX", "MATIC"])
|
|
t = random.choice([p for p in range(3) if p in [0,1,2]]) # simplified
|
|
template = random.choice(templates[sid])
|
|
text = template.format(a=random.choice(assets), p=random.randint(100,100000))
|
|
texts.append(text)
|
|
labels.append(sid)
|
|
|
|
print(f"Total samples: {len(texts)}")
|
|
|
|
# Split
|
|
from sklearn.model_selection import train_test_split
|
|
train_t, temp_t, train_l, temp_l = train_test_split(texts, labels, test_size=0.3, random_state=42, stratify=labels)
|
|
temp_t, test_t, temp_l, test_l = train_test_split(temp_t, temp_l, test_size=0.5, random_state=42, stratify=temp_l)
|
|
|
|
print(f"Train: {len(train_t)}, Val: {len(temp_t)}, Test: {len(test_t)}")
|
|
|
|
# Tokenizer & Model
|
|
from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback
|
|
import torch
|
|
|
|
tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert")
|
|
model = AutoModelForSequenceClassification.from_pretrained(
|
|
"ProsusAI/finbert", num_labels=3,
|
|
id2label={0:"Bearish",1:"Bullish",2:"Neutral"},
|
|
label2id={"Bearish":0,"Bullish":1,"Neutral":2}
|
|
)
|
|
|
|
class QuickDataset(torch.utils.data.Dataset):
|
|
def __init__(self, texts, labels, tokenizer, max_len=64):
|
|
self.texts = texts; self.labels = labels
|
|
self.tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert")
|
|
self.max_len = 64
|
|
def __len__(self): return len(self.texts)
|
|
def __getitem__(self, i):
|
|
enc = self.tokenizer(self.texts[i], truncation=True, max_length=64, padding="max_length", return_tensors="pt")
|
|
return {"input_ids": enc["input_ids"].squeeze(0), "attention_mask": enc["attention_mask"].squeeze(0), "labels": torch.tensor(self.labels[i], dtype=torch.long)}
|
|
|
|
train_ds = QuickDataset(texts[:len(train_t)], labels[:len(train_t)], None)
|
|
val_ds = QuickDataset(texts[len(train_t):len(train_t)+len(temp_t)], labels[len(train_l):len(train_l)+len(temp_l)], None)
|
|
test_ds = QuickDataset(texts[-len(test_t):], labels[-len(test_l):], None)
|
|
|
|
# Fix: create datasets properly
|
|
train_texts = texts[:len(train_t)]
|
|
train_labels = labels[:len(train_l)]
|
|
val_texts = texts[len(train_t):len(train_t)+len(temp_t)]
|
|
val_labels = labels[len(train_l):len(train_l)+len(temp_l)]
|
|
test_texts = texts[-len(test_t):]
|
|
test_labels = labels[-len(test_l):]
|
|
|
|
train_ds = QuickDataset(train_texts, train_labels, None)
|
|
val_ds = QuickDataset(val_texts, val_labels, None)
|
|
test_ds = QuickDataset(test_texts, test_labels, None)
|
|
|
|
from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback
|
|
import torch
|
|
|
|
tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert")
|
|
model = AutoModelForSequenceClassification.from_pretrained(
|
|
"ProsusAI/finbert", num_labels=3,
|
|
id2label={0:"Bearish",1:"Bullish",2:"Neutral"},
|
|
label2id={"Bearish":0,"Bullish":1,"Neutral":2}
|
|
)
|
|
|
|
class QuickDataset(torch.utils.data.Dataset):
|
|
def __init__(self, texts, labels, tokenizer, max_len=64):
|
|
self.texts = texts; self.labels = labels
|
|
self.tokenizer = tokenizer; self.max_len = 64
|
|
def __len__(self): return len(self.texts)
|
|
def __getitem__(self, i):
|
|
enc = self.tokenizer(self.texts[i], truncation=True, max_length=self.max_len, padding="max_length", return_tensors="pt")
|
|
return {"input_ids": enc["input_ids"].squeeze(0), "attention_mask": enc["attention_mask"].squeeze(0), "labels": torch.tensor(self.labels[i], dtype=torch.long)}
|
|
|
|
train_ds = QuickDataset(train_texts, train_labels, tokenizer)
|
|
val_ds = QuickDataset(val_texts, val_labels, tokenizer)
|
|
test_ds = QuickDataset(test_texts, test_labels, tokenizer)
|
|
|
|
# Train
|
|
trainer = Trainer(
|
|
model=AutoModelForSequenceClassification.from_pretrained("ProsusAI/finbert", num_labels=3, id2label={0:"Bearish",1:"Bullish",2:"Neutral"}, label2id={"Bearish":0,"Bullish":1,"Neutral":2}),
|
|
args=TrainingArguments(
|
|
|
|
num_train_epochs=1,
|
|
per_device_train_batch_size=16,
|
|
per_device_eval_batch_size=32,
|
|
gradient_accumulation_steps=2,
|
|
warmup_ratio=0.1,
|
|
learning_rate=2e-5,
|
|
lr_scheduler_type="cosine",
|
|
evaluation_strategy="epoch",
|
|
save_strategy="epoch",
|
|
load_best_model_at_end=True,
|
|
metric_for_best_model="f1_macro",
|
|
greater_is_better=True,
|
|
fp16=False,
|
|
dataloader_num_workers=0,
|
|
logging_steps=10,
|
|
save_total_limit=1,
|
|
remove_unused_columns=False,
|
|
report_to="none",
|
|
|
|
),
|
|
train_dataset=QuickDataset(train_texts, train_labels, tokenizer, max_len=64),
|
|
eval_dataset=QuickDataset(val_texts, val_labels, tokenizer, max_len=64),
|
|
tokenizer=AutoTokenizer.from_pretrained("ProsusAI/finbert"),
|
|
compute_metrics=lambda ep: {"f1_macro": f1_score(ep.label_ids, np.argmax(ep.predictions, axis=-1), average="macro")},
|
|
callbacks=[EarlyStoppingCallback(early_stopping_patience=1)]
|
|
)
|
|
|
|
from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback
|
|
from sklearn.metrics import f1_score
|
|
import torch
|
|
|
|
trainer = Trainer(
|
|
model=AutoModelForSequenceClassification.from_pretrained("ProsusAI/finbert", num_labels=3, id2label={0:"Bearish",1:"Bullish",2:"Neutral"}, label2id={"Bearish":0,"Bullish":1,"Neutral":2}),
|
|
args=TrainingArguments(
|
|
|
|
num_train_epochs=1,
|
|
per_device_train_batch_size=16,
|
|
per_device_eval_batch_size=32,
|
|
gradient_accumulation_steps=2,
|
|
warmup_ratio=0.1,
|
|
learning_rate=2e-5,
|
|
lr_scheduler_type="cosine",
|
|
evaluation_strategy="epoch",
|
|
save_strategy="epoch",
|
|
load_best_model_at_end=True,
|
|
metric_for_best_model="f1_macro",
|
|
greater_is_better=True,
|
|
fp16=False,
|
|
dataloader_num_workers=0,
|
|
logging_steps=10,
|
|
save_total_limit=1,
|
|
remove_unused_columns=False,
|
|
report_to="none",
|
|
|
|
),
|
|
train_dataset=QuickDataset(train_texts, train_labels, AutoTokenizer.from_pretrained("ProsusAI/finbert"), max_len=64),
|
|
eval_dataset=QuickDataset(val_texts, val_labels, AutoTokenizer.from_pretrained("ProsusAI/finbert"), max_len=64),
|
|
tokenizer=AutoTokenizer.from_pretrained("ProsusAI/finbert"),
|
|
compute_metrics=lambda ep: {"f1_macro": f1_score(ep.label_ids, np.argmax(ep.predictions, axis=-1), average="macro")},
|
|
callbacks=[EarlyStoppingCallback(early_stopping_patience=1)]
|
|
)
|
|
|
|
print("Training (1 epoch, ~2-3 min)...")
|
|
trainer.train()
|
|
|
|
# Test
|
|
print("\nTest results:")
|
|
results = trainer.evaluate(ep=ep) if False else trainer.evaluate()
|
|
print(f"Test: {results}")
|
|
|
|
trainer.save_model("./models/finbert-crypto-quick")
|
|
AutoTokenizer.from_pretrained("ProsusAI/finbert").save_pretrained("./models/finbert-crypto-quick")
|
|
print("Saved!")
|
|
|
|
# Quick test
|
|
model.eval()
|
|
for text in ["BTC surges to new ATH!", "Bitcoin crashes 50%!", "BTC consolidates at $50k"]:
|
|
inputs = tokenizer(text, return_tensors="pt", truncation=True, max_length=64, padding=True)
|
|
with torch.no_grad():
|
|
out = model(**inputs)
|
|
probs = torch.softmax(out.logits, dim=-1)[0]
|
|
pred = torch.argmax(probs).item()
|
|
pol = probs[1].item() - probs[0].item()
|
|
print(f" '{text}' -> {['Bearish','Bullish','Neutral'][pred]} (pol: {pol:.3f})")
|
|
print("Done!")
|
|
|
|
if __name__ == "__main__":
|
|
import random, torch
|
|
from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback
|
|
from sklearn.model_selection import train_test_split
|
|
from sklearn.metrics import f1_score
|
|
import numpy as np
|
|
main()
|