import os
import re
from collections import defaultdict
from bs4 import BeautifulSoup
import time

start = time.time()

def search_and_count_terms_in_files(root_folder, text_terms, tag_terms, output_file):
    # Compile the regular expressions for the text terms
    text_patterns = [re.compile(term) for term in text_terms]

    # Dictionary to store the counts of each text term
    text_term_counts = defaultdict(int)

    # Dictionary to store the counts of each tag term
    tag_term_counts = defaultdict(int)

    # Dictionary to store the counts of each text term in each file
    file_text_term_counts = defaultdict(lambda: defaultdict(int))

    # Dictionary to store the counts of each tag term in each file
    file_tag_term_counts = defaultdict(lambda: defaultdict(int))

    # Walk through the directory
    for dirpath, dirnames, filenames in os.walk(root_folder):
        for filename in filenames:
            if filename.endswith('.html'):
                file_path = os.path.join(dirpath, filename)
                try:
                    with open(file_path, 'r', encoding='utf-8') as file:
                        content = file.read()
                        soup = BeautifulSoup(content, 'html.parser')
                        p_tags = soup.find_all('p')

                        # Count text terms
                        for p in p_tags:
                            p_text = p.get_text()
                            for pattern in text_patterns:
                                count = len(pattern.findall(p_text))
                                if count > 0:
                                    text_term_counts[pattern.pattern] += count
                                    file_text_term_counts[pattern.pattern][file_path] += count

                        # Count tag terms
                        for tag in tag_terms:
                            tag_count = len(soup.find_all(tag))
                            if tag_count > 0:
                                tag_term_counts[tag] += tag_count
                                file_tag_term_counts[tag][file_path] += tag_count

                except Exception as e:
                    print(f"Error reading file {file_path}: {e}")

    # Open the output file in write mode
    with open(output_file, 'w', encoding='utf-8') as html_file:
        html_file.write("<html><body>\n")
        html_file.write("<h1>Search and Count Results</h1>\n")

        # Write the total counts of each text term to the HTML file
        html_file.write("<h2>Text Terms:</h2>\n")
        for term, count in text_term_counts.items():
            html_file.write(f"<h3>'{term}': {count}</h3>\n")
            sorted_files = sorted(file_text_term_counts[term].items(), key=lambda x: x[1], reverse=True)
            html_file.write("<ul>\n")
            for file_path, file_count in sorted_files:
                html_file.write(f"<li>Found {file_count} times in file: <a href='{file_path}' target='_blank'>{file_path}</a></li>\n")
            html_file.write("</ul>\n")

        # Write the total counts of each tag term to the HTML file
        html_file.write("<h2>Tag Terms:</h2>\n")
        for term, count in tag_term_counts.items():
            html_file.write(f"<h3>'{term}': {count}</h3>\n")
            sorted_files = sorted(file_tag_term_counts[term].items(), key=lambda x: x[1], reverse=True)
            html_file.write("<ul>\n")
            for file_path, file_count in sorted_files:
                html_file.write(f"<li>Found {file_count} times in file: <a href='{file_path}' target='_blank'>{file_path}</a></li>\n")
            html_file.write("</ul>\n")

        html_file.write("</body></html>\n")

# Define the text terms to search for
text_terms_to_search = ["वन्दनीया माताजी", "माताजी", "पूज्य गुरुदेव", "गुरुदेव", "पूज्य गुरुजी", "पूज्य गुरुजी", "शान्तिकुञ्ज", "क्रान्तिधर्मी", "गायत्री साधना", "आत्मबोध", "तत्त्वबोध", "गायत्री मंत्र", "युगऋषि", "युग ऋषि", "ऋषि", "युग निर्माण", "विचार क्रान्ति", "महाकाल", "बलिवैश्व"]

# Define the tag terms to search for
tag_terms_to_search = ["div", "p", "br", "a"]

# Define the root folder of the static website
root_folder = './'

# Define the output HTML file
output_file = 'find-count-words-lechat-mistral.html'

# Call the function to search for and count the terms
search_and_count_terms_in_files(root_folder, text_terms_to_search, tag_terms_to_search, output_file)

print("Output file is saved as find-count-words-lechat-mistral.html")

end = time.time()
print("Elapsed (with compilation) = %s" % (end - start))
