import os
from bs4 import BeautifulSoup
import time

start = time.time()

# Define the directory path to the website
website_dir = './'

# Define the phrase to search for
phrase_to_search = 'नारी का'

# Walk through the directory and its subdirectories
found_files = []
for root, dirs, files in os.walk(website_dir):
    for file in files:
        if file.endswith('.html'):
            file_path = os.path.join(root, file)

            # Open the HTML file and read its contents
            with open(file_path, 'r', encoding='utf-8') as f:
                html_content = f.read()

            # Parse the HTML content using BeautifulSoup
            soup = BeautifulSoup(html_content, 'html.parser')

            # Search for the phrase within the parsed HTML content
            if phrase_to_search in soup.text:
                # Extract the title of the page
                title = soup.find('title').text if soup.find('title') else file

                # Extract the text surrounding the searched phrase
                text = soup.text
                start_idx = text.find(phrase_to_search)
                if start_idx!= -1:
                    before_text = text[:start_idx].rsplit('।', 1)[-1].rsplit('?', 1)[-1].strip()
                    after_idx = text[start_idx+len(phrase_to_search):].find('।')
                    if after_idx == -1:
                        after_idx = text[start_idx+len(phrase_to_search):].find('?')
                        if after_idx == -1:
                            after_text = text[start_idx+len(phrase_to_search):]
                        else:
                            after_text = text[start_idx+len(phrase_to_search):start_idx+len(phrase_to_search)+after_idx+1].strip()
                    else:
                        after_text = text[start_idx+len(phrase_to_search):start_idx+len(phrase_to_search)+after_idx+1].strip()
                    found_files.append((file_path, title, before_text,' ' + phrase_to_search +' ', after_text))

if found_files:
    with open('searched_phrase.html', 'w', encoding='utf-8') as f:
        f.write('<html><body>')
        f.write('<h1>Found "' + phrase_to_search + '" in the following files:</h1>')
        f.write('<ul>')
        for file_path, title, before_text, phrase, after_text in found_files:
            f.write('<li><a href="' + file_path + '"  target="_blank">' + title + '</a> -'+ before_text +'<strong>' + phrase + '</strong>'+ after_text + '</li>')
        f.write('</ul>')
        f.write('</body></html>')

    print("Result saved to searched_phrase.html")
else:
    print("'" + phrase_to_search + "' not found in any files.")


end = time.time()
print("Elapsed (with compilation) = %s" % (end - start))

    
