#!/usr/bin/env python

import os
import sys
import html
import re
import logging
from bs4 import BeautifulSoup

# Set up logging
logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s")

def get_content_stats(file_path, root_dir):
    """HTML फाइल से शब्दों, कैरेक्टर और अन्य जानकारी निकालता है, साथ में फोल्डर का नाम भी जोड़ता है।"""

    # Skip search.html
    if os.path.basename(file_path) == "search.html":
        logging.info(f"Skipping {file_path}")
        return None

    content = None
    for encoding in ['utf-8', 'latin-1', 'iso-8859-1']:
        try:
            with open(file_path, 'r', encoding=encoding) as file:
                content = file.read()
            break
        except UnicodeDecodeError:
            continue
    
    if content is None:
        logging.warning(f"Could not decode {file_path}. Skipping this file.")
        return None
    
    try:
        soup = BeautifulSoup(content, 'html.parser')

        # Handle missing <title>
        title_tag = soup.title
        title = title_tag.string.strip() if title_tag and title_tag.string else "Untitled"
        
        # Get content from p and div tags
        text_content = " ".join(tag.get_text(strip=True) for tag in soup.find_all(['p', 'div']))
        
        # Calculate stats
        words = len(text_content.split())
        chars_with_spaces = len(text_content)
        chars_without_spaces = len(re.sub(r'\s+', '', text_content))
        bytes_size = os.path.getsize(file_path)
        
        # Extract folder name (excluding root_dir)
        relative_path = os.path.relpath(file_path, root_dir)
        folder_name = os.path.dirname(relative_path)
        if folder_name == "":
            folder_name = "Root"

        return {
            'title': html.escape(title),
            'words': words,
            'chars_with_spaces': chars_with_spaces,
            'chars_no_spaces': chars_without_spaces,
            'bytes': bytes_size,
            'filename': html.escape(os.path.basename(file_path)),
            'folder': html.escape(folder_name)
        }
    except Exception as e:
        logging.error(f"Error processing {file_path}: {str(e)}")
        return None

def generate_stats_html(root_dir):
    """पूरे डायरेक्टरी से HTML फाइलों का स्टैट निकालकर stats.html में सेव करता है।"""

    html_content = """
    <!DOCTYPE html>
    <html lang="hi">
    <head>
        <meta charset="UTF-8">
        <meta name="viewport" content="width=device-width, initial-scale=1.0">
        <title>Website Statistics</title>
        <style>
            body { font-family: Arial, sans-serif; margin: 20px; }
            table { border-collapse: collapse; width: 100%; }
            th, td { border: 1px solid black; padding: 8px; text-align: left; }
            th { background-color: #f2f2f2; }
        </style>
    </head>
    <body>
        <h1>वेबसाइट सांख्यिकी</h1>
        <table>
            <tr>
                <th>फ़ोल्डर</th>
                <th>फ़ाइल नाम</th>
                <th>शीर्षक</th>
                <th>शब्द</th>
                <th>कैरेक्टर (स्पेस सहित)</th>
                <th>कैरेक्टर (स्पेस रहित)</th>
                <th>बाइट्स</th>
            </tr>
    """
    
    # Process all HTML files
    processed_files = 0
    for root, dirs, files in os.walk(root_dir):
        for file in files:
            if file.endswith('.html'):
                file_path = os.path.join(root, file)
                stats = get_content_stats(file_path, root_dir)
                if stats:
                    html_content += f"""
                        <tr>
                            <td>{stats['folder']}</td>
                            <td>{stats['filename']}</td>
                            <td>{stats['title']}</td>
                            <td>{stats['words']}</td>
                            <td>{stats['chars_with_spaces']}</td>
                            <td>{stats['chars_no_spaces']}</td>
                            <td>{stats['bytes']}</td>
                        </tr>
                    """
                    processed_files += 1
    
    html_content += """
        </table>
    </body>
    </html>
    """
    
    # Save the output to stats.html
    output_path = os.path.join(root_dir, 'stats.html')
    with open(output_path, 'w', encoding='utf-8') as f:
        f.write(html_content)
    
    logging.info(f"Stats successfully generated in {output_path}. Processed {processed_files} files.")

if __name__ == "__main__":
    try:
        directory = sys.argv[1] if len(sys.argv) > 1 else '.'
        generate_stats_html(directory)
    except Exception as e:
        logging.error(f"An error occurred: {str(e)}")
