import os
from bs4 import BeautifulSoup
import time

start = time.time()

# Define the directory path to the website
website_dir = './'

# Take the phrase to search for as input from the user
phrase_to_search = input("Enter the phrase to search for: ")

# List of files to exclude from the search
excluded_files = {'searched_phrase.html', 'search_sentences_title.html', 'search_sentences_total.html'}

# Walk through the directory and its subdirectories
found_files = []
for root, dirs, files in os.walk(website_dir):
    for file in files:
        if file.endswith('.html') and file not in excluded_files:
            file_path = os.path.join(root, file)

            # Open the HTML file and read its contents
            with open(file_path, 'r', encoding='utf-8') as f:
                html_content = f.read()

            # Parse the HTML content using BeautifulSoup
            soup = BeautifulSoup(html_content, 'html.parser')

            # Search for the phrase within the parsed HTML content
            if phrase_to_search in soup.text:
                # Extract the title of the page
                title = soup.find('title').text if soup.find('title') else file

                # Count the occurrences of the phrase
                count = soup.text.count(phrase_to_search)

                # Extract the text surrounding the searched phrase
                text = soup.text
                start_idx = 0
                for _ in range(count):
                    start_idx = text.find(phrase_to_search, start_idx)
                    if start_idx != -1:
                        before_text = text[:start_idx].rsplit('।', 1)[-1].rsplit('?', 1)[-1].strip()
                        after_idx = text[start_idx + len(phrase_to_search):].find('।')
                        if after_idx == -1:
                            after_idx = text[start_idx + len(phrase_to_search):].find('?')
                            if after_idx == -1:
                                after_text = text[start_idx + len(phrase_to_search):]
                            else:
                                after_text = text[start_idx + len(phrase_to_search):start_idx + len(phrase_to_search) + after_idx + 1].strip()
                        else:
                            after_text = text[start_idx + len(phrase_to_search):start_idx + len(phrase_to_search) + after_idx + 1].strip()
                        found_files.append((file_path, title, before_text, ' ' + phrase_to_search + ' ', after_text))
                        start_idx += len(phrase_to_search)

if found_files:
    with open('searched_phrase.html', 'w', encoding='utf-8') as f:
        f.write('<html><body>')
        f.write('<h1>Found "' + phrase_to_search + '" in the following files:</h1>')
        f.write('<ul>')
        for file_path, title, before_text, phrase, after_text in found_files:
            f.write('<li><a href="' + file_path + '"  target="_blank">' + title + '</a> - ' + before_text + '<strong>' + phrase + '</strong>' + after_text + '</li>')
        f.write('</ul>')
        f.write('</body></html>')

    print("Result saved to searched_phrase.html")
else:
    print("'" + phrase_to_search + "' not found in any files.")

end = time.time()
print("Elapsed (with compilation) = %s" % (end - start))
