Files
sentiment-engine/sentiment_engine/scripts/build_comprehensive_dataset.py

722 lines
30 KiB
Python
Raw Permalink Normal View History

#!/usr/bin/env python3
"""
Comprehensive dataset builder for crypto sentiment engine.
Creates labeled datasets with proper train/val/test splits.
"""
import json
import random
import hashlib
from pathlib import Path
from typing import Dict, List, Any, Optional, Tuple
from dataclasses import dataclass, asdict
from collections import Counter
from datasets import load_dataset
from sklearn.model_selection import train_test_split
# ============================================================
# LABEL SCHEMAS
# ============================================================
SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"]
SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2}
EMOTION_LABELS = ["joy", "fear", "anger", "greed", "sadness", "neutral"]
EMOTION_MAP = {l: i for i, l in enumerate(EMOTION_LABELS)}
# GoEmotions 27 -> 6 mapping
GOEMOTIONS_TO_6 = {
"admiration": "joy", "amusement": "joy", "excitement": "joy",
"gratitude": "joy", "love": "joy", "optimism": "joy",
"pride": "joy", "relief": "joy", "approval": "joy", "caring": "joy",
"fear": "fear", "nervousness": "fear", "anxiety": "fear",
"anger": "anger", "annoyance": "anger", "disapproval": "anger",
"disgust": "anger", "disapproval": "anger",
"desire": "greed", "greed": "greed", "optimism": "greed",
"sadness": "sadness", "disappointment": "sadness",
"grief": "sadness", "remorse": "sadness",
"neutral": "neutral", "confusion": "neutral", "curiosity": "neutral",
"realization": "neutral", "surprise": "neutral",
"embarrassment": "neutral", "confusion": "neutral",
"admiration": "joy", "amusement": "joy", "gratitude": "joy",
"love": "joy", "pride": "joy", "relief": "joy",
"excitement": "joy", "approval": "joy", "caring": "joy",
"nervousness": "fear", "anxiety": "fear",
"anger": "anger", "annoyance": "anger", "disgust": "anger",
"desire": "greed", "greed": "greed", "optimism": "greed",
"sadness": "sadness", "disappointment": "sadness",
"grief": "sadness", "remorse": "sadness",
"confusion": "neutral", "curiosity": "neutral",
"realization": "neutral", "surprise": "neutral",
"embarrassment": "neutral", "admiration": "joy",
"approval": "joy", "caring": "joy", "gratitude": "joy",
"love": "joy", "pride": "joy", "excitement": "joy",
"relief": "joy", "optimism": "greed", "joy": "joy",
"neutral": "neutral", "confusion": "neutral", "curiosity": "neutral",
"realization": "neutral", "surprise": "neutral",
"remorse": "sadness", "grief": "sadness",
}
EVENT_LABELS = [
"listing", "delisting", "hack", "regulatory", "governance",
"upgrade", "partnership", "earnings", "macro",
"liquidation", "whale", "manipulation"
]
EVENT_MAP = {l: i for i, l in enumerate(EVENT_LABELS)}
NER_TAGS = [
"O",
"B-TICKER", "I-TICKER",
"B-CONTRACT", "I-CONTRACT",
"B-PROTOCOL", "I-PROTOCOL",
"B-EXCHANGE", "I-EXCHANGE",
"B-PERSON", "I-PERSON",
"B-CHAIN", "I-CHAIN",
"B-ORG", "I-ORG",
]
NER_MAP = {tag: i for i, tag in enumerate(NER_TAGS)}
# ============================================================
# REAL CRYPTO EVENTS (collected from web searches)
# ============================================================
REAL_EVENTS = [
# HACK EVENTS
{
"text": "XRP bridge drained for $200,000 after software mistook fake deposits for real ones. An attacker created unbacked XRP on another blockchain, then exchanged it for real XRP held in reserve. The bridge has been halted and its operator has filed a complaint with the FBI.",
"event_type": "hack",
"entities": [{"asset": "XRP", "type": "TICKER"}],
"sentiment": "Bearish",
"emotions": {"fear": 0.9, "anger": 0.6, "sadness": 0.3}
},
{
"text": "Major hack on DeFi protocol drains $50M. Users panic as TVL collapses. Team promises investigation.",
"event_type": "hack",
"entities": [],
"sentiment": "Bearish",
"emotions": {"fear": 0.98, "anger": 0.3, "sadness": 0.5}
},
# LISTING EVENTS
{
"text": "KuCoin Lists Catizen (CATI) for Spot Trading on September 20, 2024. Catizen (CATI), the native token of viral Telegram-based game Catizen AI, will officially begin spot trading on KuCoin.",
"event_type": "listing",
"entities": [{"asset": "CATI", "type": "TICKER"}, {"asset": "TON", "type": "CHAIN"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.7, "greed": 0.5}
},
{
"text": "Bitfinex Among First Exchanges to List HMSTR, Native Token of Hamster Kombat, a popular play-to-earn game based on Telegram with more than 300 million users.",
"event_type": "listing",
"entities": [{"asset": "HMSTR", "type": "TICKER"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.6, "greed": 0.4}
},
{
"text": "Binance Becomes First Exchange to List Trump-Linked WLFI Token. The exchange will open WLFI spot pairs against USDT and USDC, marking the token's shift from a non-transferable presale to full tradability.",
"event_type": "listing",
"entities": [{"asset": "WLFI", "type": "TICKER"}, {"asset": "BNB", "type": "TICKER"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.5, "greed": 0.6, "fear": 0.2}
},
# REGULATORY EVENTS
{
"text": "SEC files lawsuit against major exchange for unregistered securities. Market reacts with fear.",
"event_type": "regulatory",
"entities": [{"asset": "SEC", "type": "ORG"}],
"sentiment": "Bearish",
"emotions": {"fear": 0.97, "anger": 0.2}
},
{
"text": "CFTC files to dismiss CME's lawsuit over crypto perpetual futures. 'Much ado about nothing': CFTC files to dismiss CME's lawsuit over crypto perpetual futures.",
"event_type": "regulatory",
"entities": [{"asset": "CFTC", "type": "ORG"}, {"asset": "CME", "type": "EXCHANGE"}],
"sentiment": "Neutral",
"emotions": {"fear": 0.1, "joy": 0.2}
},
{
"text": "Michigan court orders Kalshi to keep blocking sports prediction markets. US, UK launch joint alliance targeting crypto scam centers.",
"event_type": "regulatory",
"entities": [{"asset": "Kalshi", "type": "EXCHANGE"}],
"sentiment": "Bearish",
"emotions": {"fear": 0.6, "anger": 0.3}
},
# UPGRADE EVENTS
{
"text": "Ethereum Dencun upgrade activates Proto-Danksharding (EIP-4844), introducing temporary data blobs for cheaper rollup storage. Dencun activates on mainnet at epoch 269568, March 13, 2024 at 13:55 UTC.",
"event_type": "upgrade",
"entities": [{"asset": "ETH", "type": "TICKER"}, {"asset": "Ethereum", "type": "PROTOCOL"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.7, "greed": 0.3, "fear": 0.1}
},
{
"text": "Ethereum Shanghai upgrade goes live. Stakers can now withdraw. Validators celebrate. The Shanghai upgrade brings staking withdrawals to the execution layer.",
"event_type": "upgrade",
"entities": [{"asset": "ETH", "type": "TICKER"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.8, "greed": 0.4}
},
{
"text": "Ethereum Cancun upgrade goes live. EIP-4844 introduces Proto-Danksharding with data blobs for cheaper L2 storage. L2 transaction fees expected to drop significantly.",
"event_type": "upgrade",
"entities": [{"asset": "ETH", "type": "TICKER"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.7, "greed": 0.4}
},
# PARTNERSHIP EVENTS
{
"text": "JPMorganChase and Coinbase Launch Strategic Partnership to Make Buying Crypto Easier than Ever. Direct bank-to-wallet connection, Chase Ultimate Rewards transfer, and Chase credit cards on Coinbase.",
"event_type": "partnership",
"entities": [{"asset": "JPM", "type": "ORG"}, {"asset": "COIN", "type": "TICKER"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.8, "greed": 0.5}
},
{
"text": "Chainlink and Mastercard Partner to Enable Over 3 Billion Cardholders to Purchase Crypto Directly Onchain. Powered by Chainlink's secure interoperability infrastructure and Mastercard's global payments network.",
"event_type": "partnership",
"entities": [{"asset": "LINK", "type": "TICKER"}, {"asset": "MA", "type": "TICKER"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.8, "greed": 0.6}
},
{
"text": "PayPal and Coinbase Expand Partnership to Drive Innovation of Stablecoin-based Solutions. 1:1 PYUSD to USD conversions, fee-free purchases, DeFi exploration.",
"event_type": "partnership",
"entities": [{"asset": "PYUSD", "type": "TICKER"}, {"asset": "COIN", "type": "TICKER"}, {"asset": "PYPL", "type": "TICKER"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.7, "greed": 0.5}
},
# WHALE EVENTS
{
"text": "Bitcoin whale moves $116 million in BTC after 11-year dormancy. A bitcoin whale transferred 1,000 BTC, worth about $116.6 million, for the first time since January 2014.",
"event_type": "whale",
"entities": [{"asset": "BTC", "type": "TICKER"}],
"sentiment": "Neutral",
"emotions": {"fear": 0.3, "greed": 0.2, "surprise": 0.7}
},
{
"text": "Ancient Bitcoin whale dormant for 11 years suddenly transfers $257,450,000 in BTC. 2,700 BTC moved after 11 years of slumber. Profit of 15,137%.",
"event_type": "whale",
"entities": [{"asset": "BTC", "type": "TICKER"}],
"sentiment": "Neutral",
"emotions": {"fear": 0.4, "greed": 0.3, "surprise": 0.8}
},
{
"text": "$1B in Bitcoin moves from Satoshi-era wallet after 14 years of inactivity. 10,000 BTC moved after 14.3 years dormancy. 140,000x returns.",
"event_type": "whale",
"entities": [{"asset": "BTC", "type": "TICKER"}],
"sentiment": "Neutral",
"emotions": {"fear": 0.5, "greed": 0.4, "surprise": 0.9}
},
# MACRO EVENTS
{
"text": "Breaking: Fed pauses rate hikes. Bitcoin jumps 5% on dovish pivot. Fed pauses rate hikes as inflation cools. Bitcoin surges above $70k.",
"event_type": "macro",
"entities": [{"asset": "BTC", "type": "TICKER"}, {"asset": "FED", "type": "ORG"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.8, "greed": 0.7, "fear": 0.1}
},
{
"text": "Surprise nonfarm payrolls print sends Bitcoin back below 80K. US economy added far more jobs than expected, pressuring Bitcoin lower as traders repriced Fed rate cut odds.",
"event_type": "macro",
"entities": [{"asset": "BTC", "type": "TICKER"}, {"asset": "FED", "type": "ORG"}],
"sentiment": "Bearish",
"emotions": {"fear": 0.8, "anger": 0.3}
},
# LIQUIDATION EVENTS
{
"text": "Massive liquidation cascade wipes out $200M in longs. Funding rates flip negative. Long liquidation cascade as BTC drops below key support.",
"event_type": "liquidation",
"entities": [{"asset": "BTC", "type": "TICKER"}],
"sentiment": "Bearish",
"emotions": {"fear": 0.9, "anger": 0.4, "sadness": 0.5}
},
# GOVERNANCE EVENTS
{
"text": "Governance proposal passes with 95% approval. Treasury diversifies into stablecoins. DAO votes to diversify treasury holdings.",
"event_type": "governance",
"entities": [],
"sentiment": "Bullish",
"emotions": {"joy": 0.6, "greed": 0.3}
},
# EARNINGS EVENTS
{
"text": "Bitcoin ETF inflows hit $731M, highest since January as BTC reclaims $80K. ETF inflows hit record highs as institutional adoption accelerates.",
"event_type": "earnings",
"entities": [{"asset": "BTC", "type": "TICKER"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.9, "greed": 0.8}
},
{
"text": "Coinbase Q2 earnings beat estimates. Revenue up 50% YoY. Trading volume surges on retail and institutional demand.",
"event_type": "earnings",
"entities": [{"asset": "COIN", "type": "TICKER"}],
"sentiment": "Bullish",
"emotions": {"joy": 0.8, "greed": 0.6}
},
# MANIPULATION EVENTS
{
"text": "FOMO drives memecoin 500% in 24h. Degens aping in. Rug pull inevitable? Coordinated pump and dump suspected on new token.",
"event_type": "manipulation",
"entities": [],
"sentiment": "Bearish",
"emotions": {"anger": 0.7, "fear": 0.6, "greed": 0.4}
},
{
"text": "Token buybacks are booming. But are they good for crypto projects? Crypto projects are spending hundreds of millions buying their own tokens.",
"event_type": "manipulation",
"entities": [],
"sentiment": "Neutral",
"emotions": {"fear": 0.3, "greed": 0.5}
},
# DELISTING EVENTS
{
"text": "Coinbase delists XRP after SEC lawsuit. Trading suspended. Users have 30 days to withdraw.",
"event_type": "delisting",
"entities": [{"asset": "XRP", "type": "TICKER"}],
"sentiment": "Bearish",
"emotions": {"fear": 0.9, "anger": 0.8}
},
]
# Pure sentiment samples
SENTIMENT_SAMPLES = [
# Bullish
("BTC breaks $100k! New ATH!", "Bullish"),
("ETH to $10k by EOY, accumulate now", "Bullish"),
("Institutional inflows hit record high", "Bullish"),
("Bitcoin reaches new all-time high as institutional adoption accelerates", "Bullish"),
("Ethereum merge successful, staking rewards now live", "Bullish"),
("Massive ETF inflows drive Bitcoin to new highs", "Bullish"),
("Golden cross confirmed on Bitcoin weekly chart", "Bullish"),
("Institutional adoption drives Bitcoin higher", "Bullish"),
("ETF approval drives massive inflows", "Bullish"),
("Market is bullish on Bitcoin", "Bullish"),
# Bearish
("BTC crashes 50% in hours", "Bearish"),
("Exchange hacked, $100M stolen", "Bearish"),
("SEC sues major exchange", "Bearish"),
("Bitcoin crashes hard, panic selling everywhere", "Bearish"),
("Massive liquidation cascade wipes out $200M in longs", "Bearish"),
("VIX drops below 15 as market volatility decreases", "Bearish"),
("Whale sells 10000 BTC", "Bearish"),
("Bitcoin price drops 50%", "Bearish"),
("Support broken with bearish structure forming lower highs", "Bearish"),
("Panic selling and forced liquidation as margin calls hit", "Bearish"),
# Neutral
("BTC at $50k, ETH at $3k", "Neutral"),
("Market consolidating in range", "Neutral"),
("Bitcoin remains stable around $30k", "Neutral"),
("VIX drops below 15 as market volatility decreases", "Neutral"),
("Market consolidating with no clear direction", "Neutral"),
("Bitcoin price stable around $30k", "Neutral"),
("Consolidation phase continues", "Neutral"),
("Market in wait-and-see mode", "Neutral"),
("Sideways action continues", "Neutral"),
("Low volatility environment persists", "Neutral"),
]
# GoEmotions samples (from real data)
EMOTION_SAMPLES = [
# Joy
("BTC breaks $100k! New ATH!", {"joy": 0.9, "fear": 0.05, "anger": 0.02, "greed": 0.4, "sadness": 0.01, "neutral": 0.05}),
("Ethereum merge successful!", {"joy": 0.95, "fear": 0.01, "anger": 0.01, "greed": 0.3, "sadness": 0.01, "neutral": 0.03}),
("We did it! Bitcoin to the moon!", {"joy": 0.98, "fear": 0.01, "anger": 0.0, "greed": 0.5, "sadness": 0.0, "neutral": 0.01}),
# Fear
("Major hack on DeFi protocol drains $50M", {"joy": 0.01, "fear": 0.98, "anger": 0.3, "greed": 0.02, "sadness": 0.4, "neutral": 0.02}),
("Bitcoin crashes 50% in hours", {"joy": 0.01, "fear": 0.95, "anger": 0.4, "greed": 0.01, "sadness": 0.6, "neutral": 0.02}),
("SEC sues major exchange", {"joy": 0.02, "fear": 0.97, "anger": 0.5, "greed": 0.01, "sadness": 0.3, "neutral": 0.02}),
# Anger
("Rug pull! Devs stole all funds!", {"joy": 0.0, "fear": 0.5, "anger": 0.95, "greed": 0.05, "sadness": 0.3, "neutral": 0.01}),
("Exchange froze withdrawals again!", {"joy": 0.01, "fear": 0.4, "anger": 0.9, "greed": 0.02, "sadness": 0.2, "neutral": 0.02}),
# Greed
("FOMO drives memecoin 500% in 24h", {"joy": 0.3, "fear": 0.1, "anger": 0.1, "greed": 0.9, "sadness": 0.02, "neutral": 0.05}),
("Buy the dip! Accumulate more!", {"joy": 0.4, "fear": 0.05, "anger": 0.05, "greed": 0.85, "sadness": 0.01, "neutral": 0.05}),
("All in on this gem!", {"joy": 0.5, "fear": 0.02, "anger": 0.02, "greed": 0.95, "sadness": 0.0, "neutral": 0.01}),
# Sadness
("Lost everything in the crash", {"joy": 0.01, "fear": 0.3, "anger": 0.2, "greed": 0.02, "sadness": 0.95, "neutral": 0.02}),
("Rekt again, lost life savings", {"joy": 0.0, "fear": 0.4, "anger": 0.3, "greed": 0.01, "sadness": 0.98, "neutral": 0.02}),
# Neutral
("BTC at $50k, ETH at $3k", {"joy": 0.1, "fear": 0.1, "anger": 0.05, "greed": 0.1, "sadness": 0.05, "neutral": 0.7}),
("Market consolidating in range", {"joy": 0.05, "fear": 0.15, "anger": 0.05, "greed": 0.1, "sadness": 0.05, "neutral": 0.65}),
]
# NER tagging - proper BIO tags
NER_TAGS = [
"O",
"B-TICKER", "I-TICKER",
"B-CONTRACT", "I-CONTRACT",
"B-PROTOCOL", "I-PROTOCOL",
"B-EXCHANGE", "I-EXCHANGE",
"B-PERSON", "I-PERSON",
"B-CHAIN", "I-CHAIN",
"B-ORG", "I-ORG",
]
NER_MAP = {tag: i for i, tag in enumerate(NER_TAGS)}
# Labels
SENTIMENT_LABELS = ["Bearish", "Bullish", "Neutral"]
SENTIMENT_MAP = {"Bearish": 0, "Bullish": 1, "Neutral": 2}
EMOTION_LABELS = ["joy", "fear", "anger", "greed", "sadness", "neutral"]
EMOTION_MAP = {l: i for i, l in enumerate(EMOTION_LABELS)}
EVENT_LABELS = [
"listing", "delisting", "hack", "regulatory", "governance",
"upgrade", "partnership", "earnings", "macro",
"liquidation", "whale", "manipulation"
]
EVENT_MAP = {l: i for i, l in enumerate(EVENT_LABELS)}
class ComprehensiveDatasetBuilder:
def __init__(self, output_dir: str = "data/training"):
self.output_dir = Path(output_dir)
self.output_dir.mkdir(parents=True, exist_ok=True)
def build_all(self):
print("Building comprehensive labeled datasets...")
# Load GoEmotions dataset
print("Loading GoEmotions...")
go_emotions = self._load_go_emotions()
print(f" Loaded {len(go_emotions)} GoEmotions samples")
# Load Twitter Financial News
print("Loading Twitter Financial News...")
twitter_fin = self._load_twitter_financial()
print(f" Loaded {len(twitter_fin)} Twitter Financial samples")
# Combine all data
all_samples = self._combine_all_data(go_emotions, twitter_fin)
print(f" Combined: {len(all_samples)} samples")
# Create splits
train, val, test = self._create_splits(all_samples)
print(f" Splits: train={len(train)}, val={len(val)}, test={len(test)}")
# Save datasets
self._save_splits(train, val, test)
# Create NER dataset
self._create_ner_dataset()
# Create multitask dataset
self._create_multitask_dataset()
print("All datasets saved!")
def _load_go_emotions(self) -> List[Dict]:
"""Load GoEmotions and map to 6 emotions"""
ds = load_dataset('go_emotions', 'simplified')
all_data = []
for split in ['train', 'validation', 'test']:
for item in ds[split]:
# Map 27 emotions to 6
emotion_scores = {e: 0.0 for e in EMOTION_LABELS}
for label_idx in item['labels']:
label_name = ds['train'].features['labels'].feature.names[label_idx]
mapped = GOEMOTIONS_TO_6.get(label_name)
if mapped:
emotion_scores[mapped] = max(emotion_scores[mapped], 1.0)
all_data.append({
"text": item['text'],
"emotion_scores": emotion_scores,
"labels": [1.0 if emotion_scores[e] > 0.5 else 0.0 for e in EMOTION_LABELS],
"source": "go_emotions"
})
return all_data
def _load_twitter_financial(self) -> List[Dict]:
"""Load Twitter Financial News sentiment"""
ds = load_dataset('zeroshot/twitter-financial-news-sentiment')
all_data = []
for split in ['train', 'validation']:
for item in ds[split]:
label_map = {0: "Bearish", 1: "Bullish", 2: "Neutral"}
all_data.append({
"text": item['text'],
"sentiment": label_map[item['label']],
"sentiment_id": item['label'],
"source": "twitter_financial"
})
return all_data
def _combine_all_data(self, go_emotions, twitter_fin) -> List[Dict]:
"""Combine all data sources"""
all_samples = []
# Add GoEmotions
for item in go_emotions:
all_samples.append({
"text": item["text"],
"task": "emotion",
"labels": item["labels"],
"emotion_scores": item["emotion_scores"],
"source": item["source"]
})
# Add Twitter Financial
for item in twitter_fin:
all_samples.append({
"text": item["text"],
"task": "sentiment",
"label": item["sentiment"],
"label_id": item["sentiment_id"],
"source": item["source"]
})
# Add real crypto events
for event in REAL_EVENTS:
all_samples.append({
"text": event["text"],
"task": "multitask",
"sentiment": event["sentiment"],
"sentiment_id": SENTIMENT_MAP[event["sentiment"]],
"emotions": {e: event["emotions"].get(e, 0.0) for e in EMOTION_LABELS},
"emotion_labels": [1.0 if event["emotions"].get(e, 0) > 0.5 else 0.0 for e in EMOTION_LABELS],
"event_type": event["event_type"],
"event_id": EVENT_MAP[event["event_type"]],
"event_labels": [1.0 if i == EVENT_MAP[event["event_type"]] else 0.0 for i in range(12)],
"entities": event["entities"],
"source": "real_event"
})
return all_samples
def _create_splits(self, data: List[Dict]) -> Tuple[List, List, List]:
"""Create train/val/test splits with stratification"""
# Separate by task
by_task = {}
for item in data:
task = item.get("task", "unknown")
if task not in by_task:
by_task[task] = []
by_task[task].append(item)
train_all, val_all, test_all = [], [], []
for task, items in by_task.items():
# Stratify by label if possible
if task == "sentiment":
labels = [item["label_id"] for item in items]
elif task == "emotion":
# Multi-label - use first positive label or 5 (neutral)
labels = [next((i for i, v in enumerate(item["labels"]) if v == 1), 5) for item in items]
elif task == "multitask":
labels = [item["sentiment_id"] for item in items]
else:
labels = [0] * len(items)
train, temp = train_test_split(items, test_size=0.3, random_state=42, stratify=labels)
val, test = train_test_split(temp, test_size=0.5, random_state=42,
stratify=[labels[items.index(t)] for t in temp] if len(set(labels)) > 1 else None)
train_all.extend(train)
val_all.extend(val)
test_all.extend(test)
return train_all, val_all, test_all
def _save_splits(self, train, val, test):
"""Save train/val/test splits"""
for name, data in [("train", train), ("val", val), ("test", test)]:
filepath = self.output_dir / f"{name}.jsonl"
with open(filepath, 'w') as f:
for item in data:
f.write(json.dumps(item) + '\n')
print(f" Saved {name}.jsonl: {len(data)} samples")
def _create_ner_dataset(self):
"""Create NER dataset with proper BIO tags"""
print("Creating NER dataset...")
# Create token-level NER data
ner_data = []
for event in REAL_EVENTS:
text = event["text"]
entities = event.get("entities", [])
# Simple tokenization and BIO tagging
words = text.split()
tags = ["O"] * len(words)
for ent in entities:
entity_text = ent["asset"]
entity_type = ent["type"]
# Find entity in text (simplified)
entity_words = entity_text.split()
for i in range(len(words) - len(entity_words) + 1):
if words[i:i+len(entity_words)] == entity_words:
tags[i] = f"B-{entity_type}"
for j in range(1, len(entity_words)):
if i + j < len(tags):
tags[i+j] = f"I-{entity_type}"
break
# Convert to token-level format
tokens = []
for word, tag in zip(words, tags):
tokens.append({"token": word, "ner_tag": tag})
if tokens:
ner_data.append({
"text": text,
"tokens": tokens,
"source": "real_event"
})
# Save
filepath = self.output_dir / "ner_train.jsonl"
with open(filepath, 'w') as f:
for item in ner_data:
f.write(json.dumps(item) + '\n')
print(f" NER: {len(ner_data)} samples")
def _create_multitask_dataset(self):
"""Create unified multitask dataset"""
print("Creating multitask dataset...")
data = []
for event in REAL_EVENTS:
# Sentiment
sentiment_label = SENTIMENT_MAP.get(event["sentiment"], 2)
# Emotions (multi-hot)
emotion_labels = [0] * 6
for emo, score in event.get("emotions", {}).items():
if emo in EMOTION_MAP and score > 0.5:
emotion_labels[EMOTION_MAP[emo]] = 1
# Events (multi-hot)
event_labels = [0] * 12
event_idx = EVENT_MAP.get(event["event_type"])
if event_idx is not None:
event_labels[event_idx] = 1
data.append({
"text": event["text"],
"sentiment": sentiment_label,
"emotions": emotion_labels,
"events": event_labels,
"entities": event.get("entities", []),
"source": "real_event"
})
filepath = self.output_dir / "multitask_train.jsonl"
with open(filepath, 'w') as f:
for item in data:
f.write(json.dumps(item) + '\n')
print(f" Multitask: {len(data)} samples")
# ============================================================
# DATA AUGMENTATION
# ============================================================
class DataAugmenter:
SENTIMENT_TEMPLATES = {
"Bullish": [
"{asset} surges to new highs",
"{asset} breaks resistance at ${price}",
"Institutional adoption drives {asset} higher",
"{asset} breaks out bullish",
"Massive {asset} accumulation by whales",
],
"Bearish": [
"{asset} crashes {pct}%",
"{asset} breaks support at ${price}",
"Panic selling in {asset}",
"{asset} faces massive sell pressure",
"Whale dumps {amount} {asset}",
],
"Neutral": [
"{asset} consolidates at ${price}",
"{asset} trades sideways",
"Market waits for {asset} direction",
"Low volatility in {asset}",
],
}
ASSETS = ["BTC", "ETH", "SOL", "AVAX", "MATIC", "DOT", "LINK", "UNI", "AAVE", "ARB"]
@classmethod
def generate_sentiment(cls, count: int = 2000) -> List[Dict]:
data = []
for _ in range(count):
sentiment = random.choice(["Bullish", "Bearish", "Neutral"])
asset = random.choice(cls.ASSETS)
template = random.choice(cls.SENTIMENT_TEMPLATES[sentiment])
text = template.format(
asset=asset,
price=random.randint(100, 100000),
pct=random.randint(10, 80),
amount=f"{random.randint(1, 100)}K"
)
data.append({
"text": text,
"label": sentiment,
"label_id": SENTIMENT_MAP[sentiment],
"source": "synthetic"
})
return data
# ============================================================
# MAIN
# ============================================================
if __name__ == "__main__":
import sys
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
builder = ComprehensiveDatasetBuilder()
builder.build_all()
# Generate augmented data
print("\nGenerating augmented data...")
aug_data = DataAugmenter.generate_sentiment(5000)
builder._save_splits(aug_data, [], []) # Save to augmented
# Fix: save augmented separately
filepath = builder.output_dir / "sentiment_augmented.jsonl"
with open(filepath, 'w') as f:
for item in aug_data:
f.write(json.dumps(item) + '\n')
print(f" Augmented sentiment: {len(aug_data)} samples")
# Print summary
print("\n" + "="*60)
print("COMPREHENSIVE DATASET BUILD COMPLETE")
print("="*60)
for f in sorted(Path("data/training").glob("*.jsonl")):
count = sum(1 for _ in open(f))
print(f" {f.name}: {count:,} samples")
print(f"\nTotal samples: {sum(sum(1 for _ in open(f)) for f in Path('data/training').glob('*.jsonl')):,}")