73 lines
2.4 KiB
Python
73 lines
2.4 KiB
Python
|
|
#!/usr/bin/env python3
|
||
|
|
"""
|
||
|
|
Create final high-quality labeled dataset by combining all sources carefully.
|
||
|
|
"""
|
||
|
|
|
||
|
|
import json
|
||
|
|
from pathlib import Path
|
||
|
|
|
||
|
|
# Load all existing labeled data
|
||
|
|
all_labeled = []
|
||
|
|
|
||
|
|
for fname in [
|
||
|
|
"labeled_verified.jsonl",
|
||
|
|
"labeled_expanded.jsonl",
|
||
|
|
"labeled_large.jsonl",
|
||
|
|
"labeled_output.jsonl",
|
||
|
|
"labeled_real_world.jsonl",
|
||
|
|
]:
|
||
|
|
path = Path(f"/mnt/dolphinng5_predict/sentiment_engine/data/{fname}")
|
||
|
|
if path.exists():
|
||
|
|
with open(path) as f:
|
||
|
|
for line in f:
|
||
|
|
try:
|
||
|
|
item = json.loads(line.strip())
|
||
|
|
labels = item.get("labels", {})
|
||
|
|
text = item.get("text", item.get("raw_text", ""))
|
||
|
|
if text and labels.get("sentiment"):
|
||
|
|
all_labeled.append({
|
||
|
|
"text": text,
|
||
|
|
"sentiment": labels["sentiment"],
|
||
|
|
"event_type": labels.get("event_type", "unknown"),
|
||
|
|
"entities": labels.get("entities", []),
|
||
|
|
"source": "verified" if fname == "labeled_verified.jsonl" else "expanded" if "expanded" in fname else "large" if fname == "labeled_large.jsonl" else "output" if fname == "labeled_output.jsonl" else "real_world",
|
||
|
|
})
|
||
|
|
except Exception as e:
|
||
|
|
pass
|
||
|
|
|
||
|
|
# Deduplicate
|
||
|
|
seen = set()
|
||
|
|
unique = []
|
||
|
|
for d in all_labeled:
|
||
|
|
h = hash(d["text"][:200])
|
||
|
|
if h not in seen:
|
||
|
|
seen.add(h)
|
||
|
|
unique.append(d)
|
||
|
|
|
||
|
|
print(f"Total unique labeled samples: {len(unique)}")
|
||
|
|
|
||
|
|
# Sentiment distribution
|
||
|
|
from collections import Counter
|
||
|
|
sent_dist = Counter(d["sentiment"] for d in unique)
|
||
|
|
print(f"Sentiment distribution: {dict(sent_dist)}")
|
||
|
|
|
||
|
|
# Save final dataset
|
||
|
|
with open("/mnt/dolphinng5_predict/sentiment_engine/data/final_labeled_set.jsonl", "w") as f:
|
||
|
|
for d in unique:
|
||
|
|
json.dump(d, f)
|
||
|
|
f.write("\n")
|
||
|
|
|
||
|
|
print(f"\nSaved {len(unique)} samples to final_labeled_set.jsonl")
|
||
|
|
|
||
|
|
# Show sentiment distribution
|
||
|
|
from collections import Counter
|
||
|
|
sent = Counter(d["sentiment"] for d in unique)
|
||
|
|
print(f"\nSentiment: {dict(sent)}")
|
||
|
|
|
||
|
|
# Show some samples per class
|
||
|
|
for s in ["Bullish", "Bearish", "Neutral"]:
|
||
|
|
samples = [d for d in unique if d["sentiment"] == s]
|
||
|
|
print(f"\n{s} ({len(samples)} samples):")
|
||
|
|
for d in samples[:3]:
|
||
|
|
print(f" {d['text'][:100]}...")
|