import os
from dotenv import load_dotenv
from notion_to_md import export_notion_page_to_markdown
from md_to_pdf import convert_markdown_to_pdf
import logging

# Set up logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger(__name__)

def test_new_page_numbering():
    """
    Test the numbering issue with the new page ID.
    Page ID: 2735bd13-6a84-809c-afee-e7057d8bf99b
    """
    # Load environment variables
    load_dotenv()
    
    page_id = "2735bd13-6a84-809c-afee-e7057d8bf99b"
    
    try:
        logger.info(f"Testing numbering for page: {page_id}")
        
        # Export the page content
        page_title, markdown_content = export_notion_page_to_markdown(page_id)
        logger.info(f"Retrieved page: {page_title}")
        logger.info(f"Content length: {len(markdown_content)} characters")
        
        # Save the markdown output for inspection
        markdown_filename = "new_page_markdown_output.md"
        with open(markdown_filename, 'w', encoding='utf-8') as f:
            f.write(markdown_content)
        logger.info(f"Markdown saved to: {markdown_filename}")
        
        # Convert to PDF
        pdf_content, pdf_filename = convert_markdown_to_pdf(markdown_content, page_title)
        
        # Save PDF to file
        with open(pdf_filename, 'wb') as f:
            f.write(pdf_content)
        logger.info(f"PDF saved to: {pdf_filename}")
        
        # Also save the HTML intermediate step for debugging
        import markdown
        md = markdown.Markdown(extensions=['tables', 'fenced_code', 'toc'])
        html_content = md.convert(markdown_content)
        
        with open("new_page_debug_html_output.html", 'w', encoding='utf-8') as f:
            f.write(html_content)
        logger.info("HTML debug output saved to: new_page_debug_html_output.html")
        
        # Analyze the numbering issue
        logger.info("\n=== ANALYZING NUMBERING ISSUE ===")
        
        # Look for numbered list items in the markdown
        lines = markdown_content.split('\n')
        numbered_items = []
        for i, line in enumerate(lines):
            stripped = line.strip()
            if stripped and stripped[0].isdigit() and '. ' in stripped:
                numbered_items.append((i+1, stripped[:50] + '...' if len(stripped) > 50 else stripped))
        
        if numbered_items:
            logger.info(f"Found {len(numbered_items)} numbered items in markdown:")
            for line_num, item in numbered_items[:10]:  # Show first 10 items
                logger.info(f"  Line {line_num}: {item}")
        else:
            logger.warning("No numbered items found in markdown!")
        
        # Look for the HTML structure
        if "As a responsible, forward-looking business" in html_content:
            start_index = html_content.find("As a responsible, forward-looking business")
            section_start = max(0, start_index - 100)
            section_end = min(len(html_content), start_index + 1000)
            html_section = html_content[section_start:section_end]
            
            logger.info("HTML section around first item:")
            print("\n" + "="*50)
            print("HTML OUTPUT:")
            print("="*50)
            print(html_section)
            print("="*50)
        
        return markdown_content, pdf_filename
        
    except Exception as e:
        logger.error(f"Error testing new page numbering: {str(e)}")
        raise

if __name__ == "__main__":
    test_new_page_numbering()
