import os
import json
import logging
from dotenv import load_dotenv
from notion_client import Client
from notion_client.helpers import collect_paginated_api

# Load environment variables
load_dotenv()

# Set up logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)

def debug_table_data(page_id: str):
    """Debug the table data from a specific Notion page."""
    notion_token = os.getenv('NOTION_TOKEN')
    if not notion_token:
        logger.error("NOTION_TOKEN environment variable is not set!")
        return
    
    client = Client(auth=notion_token)
    
    try:
        # Clean page_id (remove dashes if present)
        clean_page_id = page_id.replace('-', '')
        
        logger.info(f"🔍 Fetching blocks from page: {clean_page_id}")
        
        # Get all blocks from the page
        blocks = collect_paginated_api(
            client.blocks.children.list,
            block_id=clean_page_id
        )
        
        logger.info(f"📊 Found {len(blocks)} total blocks")
        
        # Find table blocks
        table_blocks = [block for block in blocks if block.get('type') == 'table']
        logger.info(f"📋 Found {len(table_blocks)} table blocks")
        
        # Find link_to_page blocks
        link_blocks = [block for block in blocks if block.get('type') == 'link_to_page']
        logger.info(f"🔗 Found {len(link_blocks)} link_to_page blocks")
        
        # Examine linked pages for tables
        for i, link_block in enumerate(link_blocks):
            logger.info(f"\n🔗 Examining link_to_page {i+1}:")
            link_data = link_block['link_to_page']
            
            if link_data['type'] == 'page_id':
                linked_page_id = link_data['page_id']
                logger.info(f"Linked page ID: {linked_page_id}")
                
                try:
                    # Get blocks from linked page
                    linked_blocks = collect_paginated_api(
                        client.blocks.children.list,
                        block_id=linked_page_id
                    )
                    
                    logger.info(f"📊 Found {len(linked_blocks)} blocks in linked page")
                    
                    # Find table blocks in linked page
                    linked_table_blocks = [block for block in linked_blocks if block.get('type') == 'table']
                    logger.info(f"📋 Found {len(linked_table_blocks)} table blocks in linked page")
                    
                    # Examine tables in linked page
                    for table_idx, table_block in enumerate(linked_table_blocks):
                        logger.info(f"\n🔍 Examining table {table_idx+1} from linked page:")
                        logger.info(f"Table ID: {table_block['id']}")
                        
                        # Get table rows
                        table_rows = collect_paginated_api(
                            client.blocks.children.list,
                            block_id=table_block['id']
                        )
                        
                        logger.info(f"Found {len(table_rows)} rows in table")
                        
                        # Print first few rows to understand the structure
                        for row_idx, row in enumerate(table_rows[:2]):  # Just first 2 rows
                            if row['type'] == 'table_row':
                                cells = row['table_row']['cells']
                                logger.info(f"\nRow {row_idx + 1} has {len(cells)} cells:")
                                
                                for cell_idx, cell in enumerate(cells):
                                    logger.info(f"  Cell {cell_idx + 1}:")
                                    
                                    # Extract text like the current code does
                                    text_parts = []
                                    for text_obj in cell:
                                        content = text_obj.get('plain_text', '')
                                        text_parts.append(content)
                                    
                                    full_text = ''.join(text_parts)
                                    logger.info(f"    Extracted text: '{full_text}'")
                                    logger.info(f"    Contains newlines: {chr(10) in full_text}")
                                    if chr(10) in full_text:
                                        lines = full_text.split('\n')
                                        logger.info(f"    Lines: {lines}")
                                        logger.info(f"    Line count: {len(lines)}")
                                    logger.info("")
                        
                        if len(table_rows) > 2:
                            logger.info(f"... and {len(table_rows) - 2} more rows")
                            
                except Exception as e:
                    logger.error(f"❌ Error examining linked page {linked_page_id}: {e}")
        
        # Also examine direct table blocks if any
        for i, table_block in enumerate(table_blocks):
            logger.info(f"\n🔍 Examining direct table {i+1}:")
            logger.info(f"Table ID: {table_block['id']}")
            
            # Get table rows
            table_rows = collect_paginated_api(
                client.blocks.children.list,
                block_id=table_block['id']
            )
            
            logger.info(f"Found {len(table_rows)} rows in table")
            
            # Print first few rows to understand the structure
            for row_idx, row in enumerate(table_rows[:3]):  # Just first 3 rows
                if row['type'] == 'table_row':
                    cells = row['table_row']['cells']
                    logger.info(f"\nRow {row_idx + 1} has {len(cells)} cells:")
                    
                    for cell_idx, cell in enumerate(cells):
                        logger.info(f"  Cell {cell_idx + 1}:")
                        logger.info(f"    Raw cell data: {json.dumps(cell, indent=6)}")
                        
                        # Extract text like the current code does
                        text_parts = []
                        for text_obj in cell:
                            content = text_obj.get('plain_text', '')
                            text_parts.append(content)
                        
                        full_text = ''.join(text_parts)
                        logger.info(f"    Extracted text: '{full_text}'")
                        logger.info(f"    Contains newlines: {chr(10) in full_text}")
                        if chr(10) in full_text:
                            lines = full_text.split('\n')
                            logger.info(f"    Lines: {lines}")
                        logger.info("")
            
            if len(table_rows) > 3:
                logger.info(f"... and {len(table_rows) - 3} more rows")
    
    except Exception as e:
        logger.error(f"❌ Error debugging table data: {e}")
        import traceback
        traceback.print_exc()

def main():
    """Main function to debug table data."""
    print("=" * 60)
    print("TABLE DEBUG TOOL")
    print("=" * 60)
    
    # The page ID provided by the user
    page_id = "2735bd136a84809cafeee7057d8bf99b"
    
    print(f"Debugging table data for page: {page_id}")
    debug_table_data(page_id)

if __name__ == '__main__':
    main()
