Files
sentiment-engine/sentiment_engine/create_final_labeled_set.py

73 lines
2.4 KiB
Python
Raw Permalink Normal View History

#!/usr/bin/env python3
"""
Create final high-quality labeled dataset by combining all sources carefully.
"""
import json
from pathlib import Path
# Load all existing labeled data
all_labeled = []
for fname in [
"labeled_verified.jsonl",
"labeled_expanded.jsonl",
"labeled_large.jsonl",
"labeled_output.jsonl",
"labeled_real_world.jsonl",
]:
path = Path(f"/mnt/dolphinng5_predict/sentiment_engine/data/{fname}")
if path.exists():
with open(path) as f:
for line in f:
try:
item = json.loads(line.strip())
labels = item.get("labels", {})
text = item.get("text", item.get("raw_text", ""))
if text and labels.get("sentiment"):
all_labeled.append({
"text": text,
"sentiment": labels["sentiment"],
"event_type": labels.get("event_type", "unknown"),
"entities": labels.get("entities", []),
"source": "verified" if fname == "labeled_verified.jsonl" else "expanded" if "expanded" in fname else "large" if fname == "labeled_large.jsonl" else "output" if fname == "labeled_output.jsonl" else "real_world",
})
except Exception as e:
pass
# Deduplicate
seen = set()
unique = []
for d in all_labeled:
h = hash(d["text"][:200])
if h not in seen:
seen.add(h)
unique.append(d)
print(f"Total unique labeled samples: {len(unique)}")
# Sentiment distribution
from collections import Counter
sent_dist = Counter(d["sentiment"] for d in unique)
print(f"Sentiment distribution: {dict(sent_dist)}")
# Save final dataset
with open("/mnt/dolphinng5_predict/sentiment_engine/data/final_labeled_set.jsonl", "w") as f:
for d in unique:
json.dump(d, f)
f.write("\n")
print(f"\nSaved {len(unique)} samples to final_labeled_set.jsonl")
# Show sentiment distribution
from collections import Counter
sent = Counter(d["sentiment"] for d in unique)
print(f"\nSentiment: {dict(sent)}")
# Show some samples per class
for s in ["Bullish", "Bearish", "Neutral"]:
samples = [d for d in unique if d["sentiment"] == s]
print(f"\n{s} ({len(samples)} samples):")
for d in samples[:3]:
print(f" {d['text'][:100]}...")