#!/usr/bin/env python3
"""
Анализ PII данных из tblEMEntity всех баз
"""
import json
from pathlib import Path
from collections import defaultdict

DUMPS_DIR = Path('/root/dumps/50.21.183.111')
ANALYSIS_DIR = Path('/root/ir-assessment/redteam/irelydata/analysis')

print("=== АНАЛИЗ PII ДАННЫХ ===\n")

total_entities = 0
with_email = 0
with_phone = 0
with_tax_id = 0
domains = defaultdict(int)

db_stats = []

for db_dir in DUMPS_DIR.iterdir():
    if not db_dir.is_dir():
        continue
    
    db_name = db_dir.name
    entity_file = db_dir / 'tblEMEntity.dat'
    
    if not entity_file.exists():
        continue
    
    print(f"Анализ {db_name}...")
    
    db_total = 0
    db_email = 0
    db_phone = 0
    db_tax = 0
    
    with open(entity_file, 'r', encoding='utf-8', errors='ignore') as f:
        for line in f:
            if not line.strip() or line.startswith('intEntityId'):
                continue
            
            db_total += 1
            
            # Парсим строку (pipe-separated)
            parts = line.strip().split('|')
            
            # Ищем email (обычно в поле strEmail)
            if len(parts) > 2:
                email = parts[2] if len(parts) > 2 else ''
                phone = parts[3] if len(parts) > 3 else ''
                tax_id = parts[4] if len(parts) > 4 else ''
                
                if email and '@' in email:
                    db_email += 1
                    domain = email.split('@')[1].lower()
                    domains[domain] += 1
                
                if phone and phone.strip():
                    db_phone += 1
                
                if tax_id and tax_id.strip():
                    db_tax += 1
    
    db_stats.append({
        'db': db_name,
        'total': db_total,
        'with_email': db_email,
        'with_phone': db_phone,
        'with_tax_id': db_tax
    })
    
    total_entities += db_total
    with_email += db_email
    with_phone += db_phone
    with_tax_id += db_tax

# Топ доменов
top_domains = sorted(domains.items(), key=lambda x: x[1], reverse=True)[:20]

# Сохраняем результаты
results = {
    'total_entities': total_entities,
    'with_email': with_email,
    'with_phone': with_phone,
    'with_tax_id': with_tax_id,
    'top_domains': top_domains,
    'db_stats': db_stats
}

with open(ANALYSIS_DIR / 'pi_analysis_summary.json', 'w') as f:
    json.dump(results, f, indent=2)

print(f"\nВсего entities: {total_entities:,}")
print(f"С email: {with_email:,} ({100*with_email/total_entities:.1f}%)")
print(f"С телефонами: {with_phone:,} ({100*with_phone/total_entities:.1f}%)")
print(f"С налоговыми ID: {with_tax_id:,} ({100*with_tax_id/total_entities:.1f}%)")
print(f"\nТоп-5 доменов:")
for domain, count in top_domains[:5]:
    print(f"  {domain}: {count}")

print(f"\nРезультаты сохранены в pi_analysis_summary.json")
