58 lines
1.7 KiB
Python
58 lines
1.7 KiB
Python
|
|
#!/usr/bin/env python3
|
||
|
|
"""
|
||
|
|
Create final high-quality labeled dataset.
|
||
|
|
"""
|
||
|
|
|
||
|
|
import json
|
||
|
|
from pathlib import Path
|
||
|
|
from collections import Counter
|
||
|
|
|
||
|
|
all_labeled = []
|
||
|
|
|
||
|
|
for fname in [
|
||
|
|
"labeled_verified.jsonl",
|
||
|
|
"labeled_expanded.jsonl",
|
||
|
|
"labeled_large.jsonl",
|
||
|
|
"labeled_output.jsonl",
|
||
|
|
"labeled_real_world.jsonl",
|
||
|
|
]:
|
||
|
|
path = Path(f"/mnt/dolphinng5_predict/sentiment_engine/data/{fname}")
|
||
|
|
if path.exists():
|
||
|
|
with open(path) as f:
|
||
|
|
for line in f:
|
||
|
|
try:
|
||
|
|
item = json.loads(line.strip())
|
||
|
|
labels = item.get("labels", {})
|
||
|
|
text = item.get("text", item.get("raw_text", ""))
|
||
|
|
if text and labels.get("sentiment"):
|
||
|
|
all_labeled.append({
|
||
|
|
"text": text,
|
||
|
|
"sentiment": labels["sentiment"],
|
||
|
|
"event_type": labels.get("event_type", "unknown"),
|
||
|
|
"entities": labels.get("entities", []),
|
||
|
|
"source": fname,
|
||
|
|
})
|
||
|
|
except Exception as e:
|
||
|
|
pass
|
||
|
|
|
||
|
|
# Deduplicate
|
||
|
|
seen = set()
|
||
|
|
unique = []
|
||
|
|
for d in all_labeled:
|
||
|
|
h = hash(d["text"][:200])
|
||
|
|
if h not in seen:
|
||
|
|
seen.add(h)
|
||
|
|
unique.append(d)
|
||
|
|
|
||
|
|
print(f"Total unique labeled samples: {len(unique)}")
|
||
|
|
sent_dist = Counter(d["sentiment"] for d in unique)
|
||
|
|
print(f"Sentiment distribution: {dict(sent_dist)}")
|
||
|
|
|
||
|
|
# Save
|
||
|
|
with open("/mnt/dolphinng5_predict/sentiment_engine/data/final_labeled_set.jsonl", "w") as f:
|
||
|
|
for d in unique:
|
||
|
|
json.dump(d, f)
|
||
|
|
f.write("\n")
|
||
|
|
|
||
|
|
print(f"\nSaved {len(unique)} samples to final_labeled_set.jsonl")
|