#!/usr/bin/env python3 """ Human-in-the-loop labeling tool for real-world crypto samples. Creates high-quality labeled dataset for LoRA retraining. """ import json import sys from pathlib import Path from datetime import datetime SAMPLE_FILE = Path("/mnt/dolphinng5_predict/sentiment_engine/data/real_world_samples.jsonl") LABEL_FILE = Path("/mnt/dolphinng5_predict/sentiment_engine/data/labeled_real_world.jsonl") SENTIMENT_LABELS = { 'b': 'Bearish', 'r': 'Bullish', # 'r' for bullish (green/up) 'n': 'Neutral', } EMOTION_LABELS = { 'g': 'greed', 'f': 'fear', 'j': 'joy', 'a': 'anger', 's': 'sadness', 'n': 'neutral', } def load_samples(): samples = [] with open(SAMPLE_FILE) as f: for line in f: samples.append(json.loads(line.strip())) return samples def load_existing_labels(): labeled = set() if LABEL_FILE.exists(): with open(LABEL_FILE) as f: for line in f: item = json.loads(line.strip()) labeled.add(item.get('text', '')[:100]) # Use first 100 chars as key return labeled def save_label(item, sentiment, emotions, confidence, notes=''): record = { **item, 'labels': { 'sentiment': sentiment, 'emotions': emotions, 'confidence': confidence, }, 'human_labeled': True, 'labeled_at': datetime.now().isoformat(), 'notes': notes, } with open(LABEL_FILE, 'a') as f: f.write(json.dumps(record) + '\n') def clear_screen(): print('\033[2J\033[H', end='') def print_header(idx, total, item): clear_screen() print('=' * 70) print(f'LABELING: {idx+1}/{total} | Source: {item["source_id"]}') print('=' * 70) print(f'\nTITLE: {item["title"]}') print(f'\nTEXT: {item["text"][:400]}...') print(f'\nURL: {item["url"]}') print() def get_sentiment(): print('SENTIMENT:') print(' [B] Bearish - [R] Bullish - [N] Neutral') while True: choice = input(' Choice [B/R/N]: ').strip().lower() if choice in SENTIMENT_LABELS: return SENTIMENT_LABELS[choice] print(' Invalid. Use B, R, or N') def get_emotions(): print('\nEMOTIONS (multi-select, comma-separated):') print(' [G] Greed [F] Fear [J] Joy [A] Anger [S] Sadness [N] Neutral') while True: choice = input(' Emotions [G,F,J,A,S,N]: ').strip().lower() if not choice: return [] selected = [] for c in choice.replace(' ', '').split(','): if c in EMOTION_LABELS: selected.append(EMOTION_LABELS[c]) if selected: return selected print(' Invalid. Use G,F,J,A,S,N') def get_confidence(): while True: try: conf = float(input('\nConfidence [0.0-1.0]: ').strip()) if 0 <= conf <= 1: return conf print(' Must be between 0 and 1') except ValueError: print(' Invalid number') def main(): samples = load_samples() labeled_keys = load_existing_labels() # Filter unlabeled unlabeled = [] for item in samples: key = item['text'][:100] if key not in labeled_keys: unlabeled.append(item) print(f'Total: {len(samples)} | Already labeled: {len(samples)-len(unlabeled)} | Remaining: {len(unlabeled)}') if not unlabeled: print('All samples labeled!') return input('Press Enter to start labeling...') for idx, item in enumerate(unlabeled): print_header(idx, len(unlabeled), item) sentiment = get_sentiment() emotions = get_emotions() confidence = get_confidence() notes = input('\nNotes (optional): ').strip() save_label(item, sentiment, emotions, confidence, notes) print(f'\nāœ… Saved as {sentiment} | {emotions} | conf={confidence}') input('Press Enter for next...') print('\nšŸŽ‰ All samples labeled!') print(f'Labels saved to: {LABEL_FILE}') if __name__ == '__main__': main()