import os
import logging
from typing import List, Dict, Any, Optional
from notion_client import Client
from notion_client.helpers import collect_paginated_api

# Set up logging with colors
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)

class NotionToMarkdown:
    def __init__(self, token: str):
        self.client = Client(auth=token)
    
    def get_page_content_as_markdown(self, page_id: str) -> tuple[str, str]:
        """
        Fetch all content from a Notion page and convert it to Markdown format.
        
        Args:
            page_id: The ID of the Notion page to export
            
        Returns:
            tuple[str, str]: A tuple containing (page_title, markdown_content)
        """
        try:
            # Get page info first
            page = self.client.pages.retrieve(page_id)
            
            # Extract page title
            page_title = "Untitled"
            properties = page.get('properties', {})
            
            # Look for any property with type 'title' (could be named 'title', 'Name', 'Doc name', etc.)
            for prop_name, prop_value in properties.items():
                if prop_value.get('type') == 'title':
                    if prop_value.get('title') and len(prop_value['title']) > 0:
                        raw_title = prop_value['title'][0].get('plain_text', 'Untitled').strip()
                        # Clean up multiple spaces
                        page_title = ' '.join(raw_title.split())
                        logger.info(f"Found title from property '{prop_name}': {page_title}")
                        break
            
            # Get all blocks from the page
            blocks = collect_paginated_api(
                self.client.blocks.children.list,
                block_id=page_id
            )
            
            logger.info(f"Retrieved {len(blocks)} blocks from page")
            
            # If page title is "Untitled", try to get a better title from the first content block
            if page_title == "Untitled" and blocks:
                for block in blocks:
                    block_type = block.get('type')
                    title_from_content = None
                    
                    if block_type == 'heading_1':
                        title_from_content = self._extract_rich_text(block['heading_1']['rich_text'])
                    elif block_type == 'heading_2':
                        title_from_content = self._extract_rich_text(block['heading_2']['rich_text'])
                    elif block_type == 'heading_3':
                        title_from_content = self._extract_rich_text(block['heading_3']['rich_text'])
                    elif block_type == 'paragraph':
                        text = self._extract_rich_text(block['paragraph']['rich_text'])
                        # Only use paragraph text if it's not too long and looks like a title
                        if text and len(text.strip()) < 100 and '\n' not in text:
                            title_from_content = text
                    
                    if title_from_content and title_from_content.strip():
                        page_title = title_from_content.strip()
                        logger.info(f"Using first content block as title: {page_title}")
                        break
            
            logger.info(f"Retrieved page: {page_title}")
            
            # Convert blocks to markdown
            markdown_content = self._convert_blocks_to_markdown(blocks)
            
            # Prepend the page title as H1 heading and metadata to the markdown content
            if page_title and page_title != "Untitled":
                # Extract specific properties for metadata
                metadata_lines = []
                
                # Helper function to format property values
                def format_property_for_markdown(prop_data):
                    prop_type = prop_data.get('type')
                    
                    if prop_type == 'people':
                        people = prop_data.get('people', [])
                        if people:
                            names = [person.get('name', 'Unknown') for person in people]
                            return ', '.join(names)
                    
                    elif prop_type == 'date':
                        date_data = prop_data.get('date')
                        if date_data:
                            return date_data.get('start', '')
                    
                    elif prop_type == 'status':
                        status_data = prop_data.get('status')
                        if status_data:
                            return status_data.get('name', '')
                    
                    elif prop_type == 'select':
                        select_data = prop_data.get('select')
                        if select_data:
                            return select_data.get('name', '')
                    
                    return ''
                
                # Extract the specific properties requested
                desired_properties = ['Approval', 'Approval Date', 'Protocol Type', 'Document Status']
                
                for prop_name in desired_properties:
                    if prop_name in properties:
                        prop_value = format_property_for_markdown(properties[prop_name])
                        if prop_value:
                            metadata_lines.append(f"**{prop_name}**: {prop_value}")
                
                # Build the final markdown with H1, metadata, and content
                header_parts = [f"# {page_title}"]
                
                if metadata_lines:
                    header_parts.append('')  # Empty line after H1
                    header_parts.extend(metadata_lines)
                    header_parts.append('')  # Empty line after metadata
                
                header_content = '\n'.join(header_parts)
                markdown_content = f"{header_content}\n{markdown_content}"
            
            return page_title, markdown_content
            
        except Exception as e:
            logger.error(f"Error retrieving page content: {str(e)}")
            raise
    
    def _convert_blocks_to_markdown(self, blocks: List[Dict[str, Any]], process_linked_pages: bool = True, nesting_level: int = 0) -> str:
        """
        Convert a list of Notion blocks to Markdown format.
        
        Args:
            blocks: List of Notion block objects
            process_linked_pages: Whether to embed content from linked pages (prevents recursion)
            nesting_level: Current nesting level for numbering styles (0=top level)
            
        Returns:
            str: Markdown representation of the blocks
        """
        markdown_lines = []
        numbered_list_counter = 1
        last_block_type = None
        
        for block in blocks:
            block_type = block.get('type')
            
            # Handle list separation to prevent different list types from merging
            if last_block_type is not None:
                if (last_block_type in ['bulleted_list_item', 'to_do'] and block_type == 'numbered_list_item') or \
                   (last_block_type == 'numbered_list_item' and block_type in ['bulleted_list_item', 'to_do']):
                    # Add invisible paragraph to forcefully break list context for markdown parser
                    markdown_lines.append('')
                    markdown_lines.append('&nbsp;')  # Non-breaking space creates invisible paragraph
                    markdown_lines.append('')
            
            if block_type == 'paragraph':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_paragraph(block))
            elif block_type == 'heading_1':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_heading(block, 1))
            elif block_type == 'heading_2':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_heading(block, 2))
            elif block_type == 'heading_3':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_heading(block, 3))
            elif block_type == 'bulleted_list_item':
                # Reset numbered list counter only when transitioning from numbered to bulleted
                if last_block_type == 'numbered_list_item':
                    numbered_list_counter = 1
                
                # Special handling: if this bulleted item is followed by numbered items,
                # convert it to a paragraph to avoid list context interference
                next_block_is_numbered = False
                current_index = blocks.index(block) if block in blocks else -1
                if current_index >= 0 and current_index < len(blocks) - 1:
                    next_block = blocks[current_index + 1]
                    if next_block.get('type') == 'numbered_list_item':
                        next_block_is_numbered = True
                
                if next_block_is_numbered:
                    # Convert to plain paragraph to avoid list context interference
                    text = self._extract_rich_text(block['bulleted_list_item']['rich_text'])
                    markdown_lines.append(text)  # Plain paragraph, no bullet symbol
                else:
                    markdown_lines.append(self._convert_bulleted_list_item(block))
            elif block_type == 'numbered_list_item':
                # Reset numbered list counter only when transitioning from non-numbered list types
                if last_block_type not in ['numbered_list_item']:
                    numbered_list_counter = 1
                markdown_lines.append(self._convert_numbered_list_item(block, numbered_list_counter, nesting_level))
                numbered_list_counter += 1
            elif block_type == 'to_do':
                # Reset numbered list counter only when transitioning from numbered to to-do
                if last_block_type == 'numbered_list_item':
                    numbered_list_counter = 1
                markdown_lines.append(self._convert_todo(block))
            elif block_type == 'toggle':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_toggle(block))
            elif block_type == 'quote':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_quote(block))
            elif block_type == 'code':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_code(block))
            elif block_type == 'callout':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_callout(block))
            elif block_type == 'divider':
                numbered_list_counter = 1
                markdown_lines.append('---')
            elif block_type == 'image':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_image(block))
            elif block_type == 'file':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_file(block))
            elif block_type == 'bookmark':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_bookmark(block))
            elif block_type == 'embed':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_embed(block))
            elif block_type == 'table':
                numbered_list_counter = 1
                markdown_lines.append(self._convert_table(block))
            elif block_type == 'table_row':
                # Skip table_row blocks as they are processed as part of the table
                continue
            elif block_type == 'link_to_page':
                numbered_list_counter = 1
                if process_linked_pages:
                    linked_content = self._convert_link_to_page(block)
                    if linked_content:
                        markdown_lines.append(linked_content)
                # If process_linked_pages is False, skip the link entirely (no output)
            else:
                # For unsupported block types, add a comment
                numbered_list_counter = 1
                markdown_lines.append(f'<!-- Unsupported block type: {block_type} -->')
            
            last_block_type = block_type
            
            # Handle child blocks if they exist
            if block.get('has_children', False):
                try:
                    child_blocks = collect_paginated_api(
                        self.client.blocks.children.list,
                        block_id=block['id']
                    )
                    child_markdown = self._convert_blocks_to_markdown(child_blocks, process_linked_pages, nesting_level + 1)
                    if child_markdown.strip():
                        # Determine proper indentation for child content based on parent block type
                        if block_type == 'numbered_list_item':
                            # For numbered lists, indent with 4 spaces to align with content
                            indent = '    '
                        elif block_type == 'bulleted_list_item':
                            # For bulleted lists, indent with 4 spaces to align with content  
                            indent = '    '
                        else:
                            # For other blocks, use 2 spaces
                            indent = '  '
                        
                        indented_child = '\n'.join([indent + line for line in child_markdown.split('\n') if line.strip()])
                        markdown_lines.append(indented_child)
                except Exception as e:
                    logger.warning(f"Error retrieving child blocks for {block['id']}: {str(e)}")
        
        # Join lines, preserving intentionally added empty lines but removing truly empty content
        result_lines = []
        for line in markdown_lines:
            if line.strip() or line == '':  # Keep empty separator lines
                result_lines.append(line)
        
        return '\n\n'.join(result_lines)
    
    def _extract_rich_text(self, rich_text_array: List[Dict[str, Any]]) -> str:
        """
        Extract plain text from Notion rich text array and apply basic formatting.
        
        Args:
            rich_text_array: Array of rich text objects from Notion
            
        Returns:
            str: Formatted text string
        """
        if not rich_text_array:
            return ""
        
        result = []
        for text_obj in rich_text_array:
            content = text_obj.get('plain_text', '')
            annotations = text_obj.get('annotations', {})
            
            # Apply formatting based on annotations
            if annotations.get('bold'):
                content = f'**{content}**'
            if annotations.get('italic'):
                content = f'*{content}*'
            if annotations.get('strikethrough'):
                content = f'~~{content}~~'
            if annotations.get('code'):
                content = f'`{content}`'
            if text_obj.get('href'):
                content = f'[{content}]({text_obj["href"]})'
            
            result.append(content)
        
        return ''.join(result)
    
    def _convert_paragraph(self, block: Dict[str, Any]) -> str:
        """Convert paragraph block to markdown."""
        return self._extract_rich_text(block['paragraph']['rich_text'])
    
    def _convert_heading(self, block: Dict[str, Any], level: int) -> str:
        """Convert heading block to markdown."""
        heading_text = self._extract_rich_text(block[f'heading_{level}']['rich_text'])
        return f"{'#' * level} {heading_text}"
    
    def _convert_bulleted_list_item(self, block: Dict[str, Any]) -> str:
        """Convert bulleted list item to markdown."""
        text = self._extract_rich_text(block['bulleted_list_item']['rich_text'])
        return f"- {text}"
    
    def _convert_numbered_list_item(self, block: Dict[str, Any], list_number: int, nesting_level: int = 0) -> str:
        """Convert numbered list item to markdown with proper numbered lists for HTML conversion."""
        text = self._extract_rich_text(block['numbered_list_item']['rich_text'])
        
        # Use numbers for all levels to ensure proper HTML conversion by markdown library
        # The markdown library only recognizes numbered lists (1. 2. 3.) as ordered lists
        # Letters (a. b. c.) and roman numerals (i. ii. iii.) are treated as plain text
        return f"{list_number}. {text}"
    
    def _convert_todo(self, block: Dict[str, Any]) -> str:
        """Convert to-do block to markdown."""
        text = self._extract_rich_text(block['to_do']['rich_text'])
        checked = block['to_do'].get('checked', False)
        checkbox = '[x]' if checked else '[ ]'
        return f"- {checkbox} {text}"
    
    def _convert_toggle(self, block: Dict[str, Any]) -> str:
        """Convert toggle block to markdown."""
        text = self._extract_rich_text(block['toggle']['rich_text'])
        return f"<details><summary>{text}</summary></details>"
    
    def _convert_quote(self, block: Dict[str, Any]) -> str:
        """Convert quote block to markdown."""
        text = self._extract_rich_text(block['quote']['rich_text'])
        return f"> {text}"
    
    def _convert_code(self, block: Dict[str, Any]) -> str:
        """Convert code block to markdown."""
        code_text = self._extract_rich_text(block['code']['rich_text'])
        language = block['code'].get('language', '')
        return f"```{language}\n{code_text}\n```"
    
    def _convert_callout(self, block: Dict[str, Any]) -> str:
        """Convert callout block to markdown."""
        text = self._extract_rich_text(block['callout']['rich_text'])
        icon = block['callout'].get('icon', {})
        icon_text = ''
        if icon.get('type') == 'emoji':
            icon_text = icon.get('emoji', '')
        return f"> {icon_text} {text}"
    
    def _convert_image(self, block: Dict[str, Any]) -> str:
        """Convert image block to markdown."""
        image_data = block['image']
        if image_data['type'] == 'external':
            url = image_data['external']['url']
        elif image_data['type'] == 'file':
            url = image_data['file']['url']
        else:
            return "<!-- Image block (unsupported type) -->"
        
        caption = self._extract_rich_text(image_data.get('caption', []))
        alt_text = caption if caption else "Image"
        return f"![{alt_text}]({url})"
    
    def _convert_file(self, block: Dict[str, Any]) -> str:
        """Convert file block to markdown."""
        file_data = block['file']
        if file_data['type'] == 'external':
            url = file_data['external']['url']
        elif file_data['type'] == 'file':
            url = file_data['file']['url']
        else:
            return "<!-- File block (unsupported type) -->"
        
        name = file_data.get('name', 'File')
        return f"[{name}]({url})"
    
    def _convert_bookmark(self, block: Dict[str, Any]) -> str:
        """Convert bookmark block to markdown."""
        url = block['bookmark']['url']
        caption = self._extract_rich_text(block['bookmark'].get('caption', []))
        link_text = caption if caption else url
        return f"[{link_text}]({url})"
    
    def _convert_embed(self, block: Dict[str, Any]) -> str:
        """Convert embed block to markdown."""
        url = block['embed']['url']
        return f"[Embedded content]({url})"
    
    def _convert_table(self, block: Dict[str, Any]) -> str:
        """Convert table block to markdown."""
        logger.info(f"Converting table block: {block['id']}")
        try:
            # Get table rows
            table_rows = collect_paginated_api(
                self.client.blocks.children.list,
                block_id=block['id']
            )
            
            if not table_rows:
                logger.warning("Empty table found")
                return "<!-- Empty table -->"
            
            logger.info(f"Found {len(table_rows)} rows in table")
            
            markdown_rows = []
            for i, row in enumerate(table_rows):
                if row['type'] == 'table_row':
                    cells = row['table_row']['cells']
                    # Extract text from cells, using a space for empty cells to maintain table structure
                    cell_contents = []
                    for cell in cells:
                        text = self._extract_rich_text(cell)
                        
                        # Handle newlines in table cells by replacing them with HTML <br> tags
                        # This prevents newlines from breaking the markdown table structure
                        if text:
                            # Replace newlines with <br> tags, and handle multiple consecutive newlines
                            text = text.replace('\n', '<br>')
                            # Clean up multiple consecutive <br> tags (from empty lines)
                            import re
                            text = re.sub(r'(<br>){2,}', '<br><br>', text)
                        
                        # Use a space for empty cells to maintain column structure
                        cell_contents.append(text if text else ' ')
                    
                    row_text = ' | '.join(cell_contents)
                    markdown_rows.append(f"| {row_text} |")
                    
                    # Add header separator after first row
                    if i == 0:
                        separator = ' | '.join(['---'] * len(cells))
                        markdown_rows.append(f"| {separator} |")
            
            logger.info("Successfully converted table to markdown")
            return '\n'.join(markdown_rows)
            
        except Exception as e:
            logger.warning(f"Error converting table: {str(e)}")
            return "<!-- Table conversion error -->"
    
    def _convert_link_to_page(self, block: Dict[str, Any]) -> str:
        """Convert link_to_page block to markdown by embedding the linked page content."""
        try:
            link_to_page_data = block['link_to_page']
            page_id = None
            
            # Extract page_id based on the type of link
            if link_to_page_data['type'] == 'page_id':
                page_id = link_to_page_data['page_id']
            elif link_to_page_data['type'] == 'database_id':
                # For database links, we'll just return a placeholder since we can't easily embed database content
                logger.warning(f"Database link encountered in link_to_page block: {link_to_page_data['database_id']}")
                return "<!-- Linked database content not embedded -->"
            
            if not page_id:
                logger.warning("No page_id found in link_to_page block")
                return ""
            
            logger.info(f"Embedding content from linked page: {page_id}")
            
            # Get the blocks from the linked page
            linked_blocks = collect_paginated_api(
                self.client.blocks.children.list,
                block_id=page_id
            )
            
            if not linked_blocks:
                logger.warning(f"No content found in linked page: {page_id}")
                return ""
            
            # Convert the linked page blocks to markdown, but disable further link processing
            linked_markdown = self._convert_blocks_to_markdown(linked_blocks, process_linked_pages=False, nesting_level=0)
            
            if linked_markdown.strip():
                # Process the linked content: indent all lines and demote heading levels
                processed_markdown = self._process_linked_content(linked_markdown)
                logger.info(f"Successfully embedded content from linked page: {page_id}")
                return processed_markdown
            else:
                return ""
                
        except Exception as e:
            logger.warning(f"Error converting link_to_page block: {str(e)}")
            return ""
    
    def _process_linked_content(self, markdown_content: str) -> str:
        """
        Process linked page content by wrapping in blockquote and demoting heading levels.
        
        Args:
            markdown_content: The raw markdown content from the linked page
            
        Returns:
            str: Processed markdown wrapped in blockquote with demoted headings
        """
        import re
        
        lines = markdown_content.split('\n')
        processed_lines = []
        
        for line in lines:
            # Handle empty lines - add > to maintain blockquote context
            if line.strip() == '':
                processed_lines.append('>')
                continue
            
            # Demote heading levels: # -> ##, ## -> ###, ### -> ####, etc.
            # This regex matches headings at the start of the line
            heading_match = re.match(r'^(#{1,6})\s+(.*)$', line)
            if heading_match:
                current_hashes = heading_match.group(1)
                heading_text = heading_match.group(2)
                # Add one more # to demote the heading level
                new_hashes = current_hashes + '#'
                # Ensure we don't exceed 6 levels (maximum in markdown)
                if len(new_hashes) > 6:
                    new_hashes = '######'
                line = f"{new_hashes} {heading_text}"
            
            # Wrap the line in blockquote format
            blockquote_line = '> ' + line
            processed_lines.append(blockquote_line)
        
        return '\n'.join(processed_lines)


def export_notion_page_to_markdown(page_id: str, token: Optional[str] = None) -> tuple[str, str]:
    """
    Export a Notion page to Markdown format.
    
    Args:
        page_id: The ID of the Notion page to export
        token: Notion API token (if not provided, will use NOTION_TOKEN env var)
        
    Returns:
        tuple[str, str]: A tuple containing (page_title, markdown_content)
    """
    if not token:
        token = os.getenv('NOTION_TOKEN')
        if not token:
            raise ValueError("Notion token not provided and NOTION_TOKEN environment variable not set")
    
    converter = NotionToMarkdown(token)
    return converter.get_page_content_as_markdown(page_id)
