#!/usr/bin/env python3 """ Create final high-quality labeled dataset by combining all sources carefully. """ import json from pathlib import Path # Load all existing labeled data all_labeled = [] for fname in [ "labeled_verified.jsonl", "labeled_expanded.jsonl", "labeled_large.jsonl", "labeled_output.jsonl", "labeled_real_world.jsonl", ]: path = Path(f"/mnt/dolphinng5_predict/sentiment_engine/data/{fname}") if path.exists(): with open(path) as f: for line in f: try: item = json.loads(line.strip()) labels = item.get("labels", {}) text = item.get("text", item.get("raw_text", "")) if text and labels.get("sentiment"): all_labeled.append({ "text": text, "sentiment": labels["sentiment"], "event_type": labels.get("event_type", "unknown"), "entities": labels.get("entities", []), "source": "verified" if fname == "labeled_verified.jsonl" else "expanded" if "expanded" in fname else "large" if fname == "labeled_large.jsonl" else "output" if fname == "labeled_output.jsonl" else "real_world", }) except Exception as e: pass # Deduplicate seen = set() unique = [] for d in all_labeled: h = hash(d["text"][:200]) if h not in seen: seen.add(h) unique.append(d) print(f"Total unique labeled samples: {len(unique)}") # Sentiment distribution from collections import Counter sent_dist = Counter(d["sentiment"] for d in unique) print(f"Sentiment distribution: {dict(sent_dist)}") # Save final dataset with open("/mnt/dolphinng5_predict/sentiment_engine/data/final_labeled_set.jsonl", "w") as f: for d in unique: json.dump(d, f) f.write("\n") print(f"\nSaved {len(unique)} samples to final_labeled_set.jsonl") # Show sentiment distribution from collections import Counter sent = Counter(d["sentiment"] for d in unique) print(f"\nSentiment: {dict(sent)}") # Show some samples per class for s in ["Bullish", "Bearish", "Neutral"]: samples = [d for d in unique if d["sentiment"] == s] print(f"\n{s} ({len(samples)} samples):") for d in samples[:3]: print(f" {d['text'][:100]}...")