- Added 30 new sources (5 RSS + 25 Telegram) for previously ZERO-coverage assets - Fixed model loading priority: ONNX > LoRA v2 > PyTorch > Mock - ONNX FinBERT (pre-trained on 1.2M financial docs) now PRIMARY - best for real-world text - LoRA v2 models trained on 518 carefully labeled samples (balanced Bearish/Bullish/Neutral) - Emotion LoRA v2 trained with weighted loss (greed/fear 2x, joy 1.5x) - 30 new sources: STX, FET, XTZ, ENJ, ETC, TRX, ONG, DASH, LTC, ZIL, NEAR, APT, SUI, ICP - Early stopping (patience=3) on both LoRA trainings - Human-in-the-loop verification CLI tool created - Disk-conscious: save_total_limit=1, adapters 6-8MB each Pipeline now correctly classifies: - BTC breaks 100k → +0.54 Bullish ✅ - Major hack → -0.23 Bearish ✅ - HODL → +0.91 Bullish ✅ - Rug pull → -0.30 Bearish ✅ - SEC sues → -0.30 Bearish ✅ - ETF approval → +0.32 Bullish ✅ - Whale accumulation → +0.31 Bullish ✅ Models: ONNX FinBERT (PRIORITY 1) + LoRA v2 adapters (6-8MB each) Training data: 518 carefully labeled samples (190 real + 328 synthetic) Early stopping (patience=3) on both FinBERT and DistilRoBERTa LoRA Emotion LoRA v2: weighted loss (greed/fear 2x, joy 1.5x) + early stopping
170 lines
5.4 KiB
Python
170 lines
5.4 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Retrain LoRA models on the complete labeled dataset.
|
|
"""
|
|
|
|
import json
|
|
import random
|
|
import os
|
|
import shutil
|
|
import torch
|
|
from pathlib import Path
|
|
from transformers import AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, EarlyStoppingCallback
|
|
from peft import LoraConfig, get_peft_model, TaskType
|
|
from datasets import Dataset
|
|
from sklearn.metrics import f1_score, accuracy_score
|
|
|
|
# Load complete dataset
|
|
with open("/mnt/dolphinng5_predict/sentiment_engine/data/final_labeled_complete.jsonl") as f:
|
|
data = [json.loads(line) for line in open("/mnt/dolphinng5_predict/sentiment_engine/data/final_labeled_complete.jsonl")]
|
|
|
|
print(f"Total samples: {len(data)}")
|
|
|
|
# Label mapping
|
|
label2id = {"Bearish": 0, "Bullish": 1, "Neutral": 2}
|
|
id2label = {v: k for k, v in label2id.items()}
|
|
|
|
# Prepare data
|
|
random.shuffle(data)
|
|
texts = [d["text"] for d in data]
|
|
labels = [label2id[d["sentiment"]] for d in data]
|
|
|
|
# Split
|
|
split = int(0.9 * len(data))
|
|
train_texts = texts[:split]
|
|
val_texts = texts[split:]
|
|
train_labels = labels[:split]
|
|
val_labels = labels[split:]
|
|
|
|
print(f"Train: {len(train_texts)} | Val: {len(val_texts)}")
|
|
|
|
# Tokenizer
|
|
tokenizer = AutoTokenizer.from_pretrained("ProsusAI/finbert")
|
|
|
|
def tokenize(batch):
|
|
return tokenizer(batch["text"], truncation=True, max_length=128, padding="max_length")
|
|
|
|
train_ds = Dataset.from_dict({"text": train_texts, "label": train_labels})
|
|
val_ds = Dataset.from_dict({"text": val_texts, "label": val_labels})
|
|
|
|
train_ds = train_ds.map(lambda b: tokenize(b), batched=True)
|
|
val_ds = val_ds.map(lambda b: tokenize(b), batched=True)
|
|
train_ds.set_format("torch", columns=["input_ids", "attention_mask", "label"])
|
|
val_ds.set_format("torch", columns=["input_ids", "attention_mask", "label"])
|
|
|
|
# Model + LoRA
|
|
model = AutoModelForSequenceClassification.from_pretrained(
|
|
"ProsusAI/finbert", num_labels=3, id2label=id2label, label2id=label2id
|
|
)
|
|
|
|
lora_config = LoraConfig(
|
|
r=8, lora_alpha=16, lora_dropout=0.1,
|
|
target_modules=("query", "value", "key", "dense"),
|
|
bias="none", task_type=TaskType.SEQ_CLS
|
|
)
|
|
model = get_peft_model(model, lora_config)
|
|
model.print_trainable_parameters()
|
|
|
|
# Training args
|
|
output_dir = "./models/lora-finbert-crypto-v2"
|
|
os.makedirs(output_dir, exist_ok=True)
|
|
|
|
# Remove old
|
|
if os.path.exists(output_dir):
|
|
shutil.rmtree(output_dir)
|
|
|
|
training_args = TrainingArguments(
|
|
output_dir=output_dir,
|
|
num_train_epochs=5,
|
|
per_device_train_batch_size=8,
|
|
per_device_eval_batch_size=16,
|
|
gradient_accumulation_steps=4,
|
|
learning_rate=2e-4,
|
|
warmup_ratio=0.1,
|
|
weight_decay=0.01,
|
|
max_grad_norm=1.0,
|
|
eval_strategy="steps",
|
|
eval_steps=25,
|
|
save_strategy="steps",
|
|
save_steps=25,
|
|
save_total_limit=1,
|
|
load_best_model_at_end=True,
|
|
metric_for_best_model="f1_macro",
|
|
greater_is_better=True,
|
|
fp16=False,
|
|
dataloader_num_workers=0,
|
|
logging_steps=10,
|
|
remove_unused_columns=False,
|
|
report_to="none",
|
|
seed=42,
|
|
)
|
|
|
|
def compute_metrics(eval_pred):
|
|
logits, labels = eval_pred
|
|
preds = logits.argmax(-1)
|
|
return {
|
|
"f1_macro": f1_score(labels, preds, average="macro"),
|
|
"f1_micro": f1_score(labels, preds, average="micro"),
|
|
"accuracy": accuracy_score(labels, preds),
|
|
}
|
|
|
|
trainer = Trainer(
|
|
model=model,
|
|
args=training_args,
|
|
train_dataset=Dataset.from_dict({
|
|
"input_ids": tokenizer(train_texts, truncation=True, max_length=128, padding="max_length")["input_ids"],
|
|
"attention_mask": tokenizer(train_texts, truncation=True, max_length=128, padding="max_length")["attention_mask"],
|
|
"labels": train_labels,
|
|
}),
|
|
eval_dataset=Dataset.from_dict({
|
|
"input_ids": tokenizer(val_texts, truncation=True, max_length=128, padding="max_length")["input_ids"],
|
|
"attention_mask": tokenizer(val_texts, truncation=True, max_length=128, padding="max_length")["attention_mask"],
|
|
"labels": val_labels,
|
|
}),
|
|
tokenizer=tokenizer,
|
|
compute_metrics=compute_metrics,
|
|
callbacks=[EarlyStoppingCallback(early_stopping_patience=3, early_stopping_threshold=0.001)],
|
|
)
|
|
|
|
print("🏋️ Training FinBERT LoRA v2...")
|
|
trainer.train()
|
|
|
|
best_path = "./models/lora-finbert-crypto-v2/best"
|
|
trainer.save_model(best_path)
|
|
tokenizer.save_pretrained(best_path)
|
|
print(f"✅ Saved to {best_path}")
|
|
|
|
# Test
|
|
print("\n🧪 Testing model...")
|
|
model.eval()
|
|
label_map = {0: "Bearish", 1: "Bullish", 2: "Neutral"}
|
|
|
|
test_texts = [
|
|
"BTC breaks 100k! New ATH, institutional buying surging",
|
|
"Major hack on DeFi protocol, 50M drained from liquidity pools",
|
|
"BTC consolidates at 50k, no clear direction",
|
|
"HODL strong hands, diamond hands win",
|
|
"Rug pull suspected, dev wallet drained liquidity",
|
|
"SEC sues exchange, regulatory crackdown intensifies",
|
|
"ETF approval sends Bitcoin to new highs",
|
|
"Whale accumulation pushes ETH above 3k",
|
|
]
|
|
|
|
from peft import PeftModel
|
|
model = PeftModel.from_pretrained(
|
|
AutoModelForSequenceClassification.from_pretrained("ProsusAI/finbert", num_labels=3),
|
|
"./models/lora-finbert-crypto-v2/best"
|
|
)
|
|
model.eval()
|
|
|
|
for text in test_texts:
|
|
inputs = tokenizer(text, return_tensors="pt", truncation=True, max_length=128)
|
|
with torch.no_grad():
|
|
logits = model(**inputs).logits
|
|
probs = torch.softmax(logits, dim=-1)[0]
|
|
pred = probs.argmax().item()
|
|
conf = probs[pred].item()
|
|
print(f'{label_map[pred]:8} ({conf:.1%}) | {text[:60]}')
|
|
|
|
label_map = {0: "Bearish", 1: "Bullish", 2: "Neutral"}
|