import re

def sanitize_filename(title: str) -> str:
    """
    Sanitize a string to make it safe for use as a filename.
    Only allows ASCII characters to avoid HTTP header encoding issues.
    
    Args:
        title: The original title string
        
    Returns:
        str: A sanitized filename-safe string with only ASCII characters
    """
    # First, try to transliterate non-ASCII characters to ASCII equivalents
    # For Hebrew and other non-Latin scripts, this will help preserve some meaning
    sanitized = title
    
    # Replace invalid filename characters with underscores
    sanitized = re.sub(r'[<>:"/\\|?*]', '_', sanitized)
    
    # Keep only ASCII alphanumeric characters, spaces, hyphens, underscores, and dots
    # This will replace Hebrew characters and other Unicode characters with underscores
    sanitized = re.sub(r'[^a-zA-Z0-9\s\-_.]', '_', sanitized)
    
    # Replace multiple spaces/underscores with single ones and strip
    sanitized = re.sub(r'\s+', '_', sanitized)
    sanitized = re.sub(r'_{2,}', '_', sanitized)
    sanitized = sanitized.strip('_')
    
    # Limit length to avoid filesystem issues
    if len(sanitized) > 100:
        sanitized = sanitized[:100]
    
    # Ensure it's not empty and doesn't start/end with problematic characters
    if not sanitized or sanitized == '_' or not sanitized.replace('_', ''):
        sanitized = 'Untitled'
    
    return sanitized

def test_unicode_filename():
    """Test that Hebrew and other Unicode characters are properly handled."""
    
    # Test cases
    test_cases = [
        ("נוהל קליטת עובד", "Hebrew"),
        ("Data Processing Addendum [DPA]", "English with brackets"),
        ("文档处理", "Chinese"),
        ("Документ", "Russian"),
        ("Test/File<Name>", "Invalid characters"),
        ("", "Empty string"),
        ("___", "Only underscores"),
        ("Mixed עברית and English", "Mixed Hebrew and English")
    ]
    
    print("=" * 60)
    print("UNICODE FILENAME SANITIZATION TEST")
    print("=" * 60)
    
    for original, description in test_cases:
        sanitized = sanitize_filename(original)
        
        # Test if the result is ASCII-safe
        try:
            sanitized.encode('ascii')
            ascii_safe = "✅ ASCII-safe"
        except UnicodeEncodeError:
            ascii_safe = "❌ NOT ASCII-safe"
        
        # Test if it can be used in HTTP headers (latin-1 encoding)
        try:
            f'attachment; filename="{sanitized}"'.encode('latin-1')
            http_safe = "✅ HTTP-safe"
        except UnicodeEncodeError:
            http_safe = "❌ NOT HTTP-safe"
        
        print(f"Original: '{original}' ({description})")
        print(f"Result:   '{sanitized}' ({ascii_safe}, {http_safe})")
        print()

if __name__ == '__main__':
    test_unicode_filename()
