import os
import time
import itertools
from bs4 import BeautifulSoup

start = time.time()

def search_and_highlight(root_folder, search_phrase, output_file):
    highlight_style = "background-color: yellow; font-weight: bold;"
    output_soup = BeautifulSoup(features="html.parser")
    output_body = output_soup.new_tag("body")
    output_soup.append(output_body)

    results = {}
    search_phrase_words = search_phrase.split()
    search_phrase_permutations = set(itertools.permutations(search_phrase_words))

    for dirpath, _, filenames in os.walk(root_folder):
        for filename in filenames:
            if filename.endswith('.html'):
                file_path = os.path.join(dirpath, filename)

                try:
                    with open(file_path, 'r', encoding='utf-8') as file:
                        content = file.read()
                except Exception as e:
                    print(f"Error reading file {file_path}: {e}")
                    continue

                soup = BeautifulSoup(content, 'html.parser')
                body = soup.find('body')
                title_tag = soup.find('title')
                page_title = title_tag.string if title_tag else "No Title"

                if body:
                    snippets = set()
                    for tag in body.find_all(['p', 'div']):
                        text = tag.get_text()
                        sentences = [s.strip() for s in text.split('।') if s.strip()]

                        for i, sentence in enumerate(sentences):
                            for permutation in search_phrase_permutations:
                                search_phrase_variant = ' '.join(permutation)
                                if search_phrase_variant in sentence:
                                    # Determine the range of sentences to include
                                    start_index = max(0, i - 1)  # Include one sentence before the matched sentence
                                    end_index = min(len(sentences), i + 3)  # Include two sentences after the matched sentence


                                    # Combine the sentences
                                    context = '। '.join(sentences[start_index:end_index])

                                    # Highlight the search phrase in the context
                                    highlighted_context = context.replace(search_phrase_variant,
                                                                        f'<span style="{highlight_style}">{search_phrase_variant}</span>')
                                    snippets.add(highlighted_context)

                    if snippets:
                        if file_path not in results:
                            results[file_path] = (page_title, snippets)
                        else:
                            results[file_path][1].update(snippets)

    for file_path, (page_title, snippets) in results.items():
        link = output_soup.new_tag("a", href=f"file://{os.path.abspath(file_path)}")
        link.string = page_title
        output_body.append(link)
        output_body.append(output_soup.new_tag("br"))
        for snippet in snippets:
            output_body.append(BeautifulSoup(snippet, 'html.parser'))
            output_body.append(output_soup.new_tag("br"))
        output_body.append(output_soup.new_tag("hr"))

    with open(output_file, 'w', encoding='utf-8') as file:
        file.write(str(output_soup))
    print(f"Output saved to: {output_file}")

# Example usage
root_folder = '../'
search_phrase = 'पृथ्वी'
output_file = 'search-phrase.html'
search_and_highlight(root_folder, search_phrase, output_file)

end = time.time()
print("Elapsed (with compilation) = %s" % (end - start))
