Files
sentiment-engine/sentiment_engine/label_samples.py

143 lines
4.0 KiB
Python
Raw Permalink Normal View History

#!/usr/bin/env python3
"""
Human-in-the-loop labeling tool for real-world crypto samples.
Creates high-quality labeled dataset for LoRA retraining.
"""
import json
import sys
from pathlib import Path
from datetime import datetime
SAMPLE_FILE = Path("/mnt/dolphinng5_predict/sentiment_engine/data/real_world_samples.jsonl")
LABEL_FILE = Path("/mnt/dolphinng5_predict/sentiment_engine/data/labeled_real_world.jsonl")
SENTIMENT_LABELS = {
'b': 'Bearish',
'r': 'Bullish', # 'r' for bullish (green/up)
'n': 'Neutral',
}
EMOTION_LABELS = {
'g': 'greed',
'f': 'fear',
'j': 'joy',
'a': 'anger',
's': 'sadness',
'n': 'neutral',
}
def load_samples():
samples = []
with open(SAMPLE_FILE) as f:
for line in f:
samples.append(json.loads(line.strip()))
return samples
def load_existing_labels():
labeled = set()
if LABEL_FILE.exists():
with open(LABEL_FILE) as f:
for line in f:
item = json.loads(line.strip())
labeled.add(item.get('text', '')[:100]) # Use first 100 chars as key
return labeled
def save_label(item, sentiment, emotions, confidence, notes=''):
record = {
**item,
'labels': {
'sentiment': sentiment,
'emotions': emotions,
'confidence': confidence,
},
'human_labeled': True,
'labeled_at': datetime.now().isoformat(),
'notes': notes,
}
with open(LABEL_FILE, 'a') as f:
f.write(json.dumps(record) + '\n')
def clear_screen():
print('\033[2J\033[H', end='')
def print_header(idx, total, item):
clear_screen()
print('=' * 70)
print(f'LABELING: {idx+1}/{total} | Source: {item["source_id"]}')
print('=' * 70)
print(f'\nTITLE: {item["title"]}')
print(f'\nTEXT: {item["text"][:400]}...')
print(f'\nURL: {item["url"]}')
print()
def get_sentiment():
print('SENTIMENT:')
print(' [B] Bearish - [R] Bullish - [N] Neutral')
while True:
choice = input(' Choice [B/R/N]: ').strip().lower()
if choice in SENTIMENT_LABELS:
return SENTIMENT_LABELS[choice]
print(' Invalid. Use B, R, or N')
def get_emotions():
print('\nEMOTIONS (multi-select, comma-separated):')
print(' [G] Greed [F] Fear [J] Joy [A] Anger [S] Sadness [N] Neutral')
while True:
choice = input(' Emotions [G,F,J,A,S,N]: ').strip().lower()
if not choice:
return []
selected = []
for c in choice.replace(' ', '').split(','):
if c in EMOTION_LABELS:
selected.append(EMOTION_LABELS[c])
if selected:
return selected
print(' Invalid. Use G,F,J,A,S,N')
def get_confidence():
while True:
try:
conf = float(input('\nConfidence [0.0-1.0]: ').strip())
if 0 <= conf <= 1:
return conf
print(' Must be between 0 and 1')
except ValueError:
print(' Invalid number')
def main():
samples = load_samples()
labeled_keys = load_existing_labels()
# Filter unlabeled
unlabeled = []
for item in samples:
key = item['text'][:100]
if key not in labeled_keys:
unlabeled.append(item)
print(f'Total: {len(samples)} | Already labeled: {len(samples)-len(unlabeled)} | Remaining: {len(unlabeled)}')
if not unlabeled:
print('All samples labeled!')
return
input('Press Enter to start labeling...')
for idx, item in enumerate(unlabeled):
print_header(idx, len(unlabeled), item)
sentiment = get_sentiment()
emotions = get_emotions()
confidence = get_confidence()
notes = input('\nNotes (optional): ').strip()
save_label(item, sentiment, emotions, confidence, notes)
print(f'\n✅ Saved as {sentiment} | {emotions} | conf={confidence}')
input('Press Enter for next...')
print('\n🎉 All samples labeled!')
print(f'Labels saved to: {LABEL_FILE}')
if __name__ == '__main__':
main()