import os
import re
import time
from bs4 import BeautifulSoup
from concurrent.futures import ThreadPoolExecutor

def process_file(file_path, words_set, output_dir, results, phrase):
    try:
        with open(file_path, 'r') as f:
            content = f.read()
            soup = BeautifulSoup(content, 'html.parser')
            title = soup.find('title')
            title_text = title.text.strip() if title else os.path.basename(file_path)
            title_text = title_text.replace('(गायत्री परिवार - Gayatri Pariwar)', '')
            p_elements = soup.find_all('p')
            for p in p_elements:
                p_text = p.get_text()
                sentences = re.split(r'([।?॥])', p_text)
                for i in range(0, len(sentences), 2):
                    sentence = sentences[i]
                    delimiter = sentences[i + 1] if i + 1 < len(sentences) else ''
                    if all(word in sentence.lower() for word in phrase.lower().split()):
                        start_idx = max(0, i - 4)  # Include 2-3 more sentences before
                        end_idx = min(len(sentences), i + 5)  # Include 2-3 more sentences after
                        context_sentences = sentences[start_idx:end_idx]
                        bolded_sentence = ' '.join(context_sentences[0::2]) + ' '.join(context_sentences[1::2])
                        for word in phrase.lower().split():
                            bolded_sentence = re.sub(r'(' + re.escape(word) + r')', r'<b style="background-color: #ffffb0; color: #0000FF; font-weight: bold">\1</b>', bolded_sentence, flags=re.IGNORECASE)
                        results.append((title_text, os.path.relpath(file_path, output_dir), bolded_sentence, len([w for w in words_set if w in bolded_sentence]), p_text))
    except Exception as e:
        print(f"Error reading {file_path}: {e}")

def search_words(root_dir, words, output_file, phrase):
    if os.path.exists(output_file):
        os.remove(output_file)
    html_parts = [
        "<html><body><h1>Search Results</h1>",
        "<meta http-equiv='content-type' content='text/html; charset=UTF-8'>",
        "<meta charset='utf-8'>",
        f"<script>function copySentence(event) {{navigator.clipboard.writeText('{phrase}');}}</script>",
        "<div id='google_translate_element'></div>",
        "<script type='text/javascript'>function googleTranslateElementInit() {new google.translate.TranslateElement({pageLanguage: 'hi', layout: google.translate.TranslateElement.InlineLayout.SIMPLE}, 'google_translate_element');}</script>",
        "<script type='text/javascript' src='//translate.google.com/translate_a/element.js?cb=googleTranslateElementInit'></script>",
        "<ul>"
    ]
    words_set = set(word.strip().lower() for word in words.split(','))
    results = []
    output_dir = os.path.dirname(__file__)
    search_dir = os.path.join(output_dir)
    if not os.path.exists(search_dir):
        os.makedirs(search_dir)

    with ThreadPoolExecutor(max_workers=4) as executor:
        futures = []
        for root, _, files in os.walk(root_dir):
            for file in files:
                if file.endswith('.html'):
                    file_path = os.path.join(root, file)
                    futures.append(executor.submit(process_file, file_path, words_set, output_dir, results, phrase))
        for future in futures:
            future.result()

    # Prioritize exact matches
    exact_matches = sorted(
        (result for result in results if all(word in result[2].lower() for word in phrase.lower().split())),
        key=lambda x: tuple(x[2].lower().find(word) for word in words_set)
    )

    # Prioritize nearby matches within the same sentence
    nearby_matches = sorted(
        (result for result in results if not all(word in result[2].lower() for word in phrase.lower().split()) and all(word in result[2].lower() for word in words_set)),
        key=lambda x: min(x[2].lower().find(word) for word in words_set)
    )

    for title_text, relative_path, snippet, _, p_text in exact_matches + nearby_matches:
        title = snippet + ' ' + p_text[p_text.find(snippet) + len(snippet):].strip()
        stop_chars = ['।', '?', '॥', '.']
        title_end_index = len(snippet) + 115
        for char in stop_chars:
            index = title[:title_end_index].rfind(char)
            if index != -1:
                title_end_index = index + 1
                break
        title = title[:title_end_index].strip()
        title = re.sub('<.*?>', '', title)
        html_parts.append(f"<li><a href='{relative_path}' target='_blank' onclick='copySentence(event)'>{title_text}</a> - {snippet}</li>")

    html_parts.append("</ul></body></html>")
    output_path = os.path.join(search_dir, output_file)
    with open(output_path, 'w') as f:
        f.write(''.join(html_parts))
    print(''.join(html_parts))

def sort_html_file(input_file_name, output_file_name, phrase):
    try:
        with open(input_file_name, 'r', encoding='utf-8', errors='ignore') as f:
            html = f.read()
            print("HTML loaded successfully")

        # Find all links and descriptions
        links_and_descriptions = re.findall(r'<li>.*?<\/li>', html, re.DOTALL)
        print("Links and descriptions found:", len(links_and_descriptions))

        # Extract link and description from each item
        extracted_links_and_descriptions = []
        for item in links_and_descriptions:
            match = re.search(r'href=[\'"]([^\'"]+)[\'"]', item)
            if match:
                link = match.group(1)
                match = re.search(r'>[^<]+?<\/a>', item)
                if match:
                    title = match.group(0)[1:-4]
                    description = re.sub(r'<.*?>', '', item).strip().split('-', 1)[1].strip()
                    extracted_links_and_descriptions.append((link, title, description))

        print("Extracted links and descriptions:", len(extracted_links_and_descriptions))

        # Sort links and descriptions based on the presence of the phrase
        phrase_links_and_descriptions = [x for x in extracted_links_and_descriptions if all(word in x[2].lower() for word in phrase.lower().split())]

        sorted_phrase_links_and_descriptions = sorted(phrase_links_and_descriptions, key=lambda x: sum(x[2].lower().find(word) for word in phrase.lower().split()))

        sorted_links_and_descriptions = []
        words = phrase.split()
        added_descriptions = set()

        for link, title, description in sorted_phrase_links_and_descriptions:
            # Highlight each word if they appear in any combination within the same sentence
            for word in phrase.lower().split():
                description = re.sub(r'(' + re.escape(word) + r')', r'<b style="background-color: #ffffb0; color: #0000FF; font-weight: bold">\1</b>', description, flags=re.IGNORECASE)

            description_words = description.split()
            description_hash = tuple(sorted(description_words[:7]))  # Use the first 7 words as a hash
            if description_hash not in added_descriptions:
                sorted_links_and_descriptions.append((link, title, description))
                added_descriptions.add(description_hash)

        print("Links and descriptions sorted")

        # Create the sorted HTML
        sorted_html = '<html><body><h1>Search Results</h1><ul>'
        sorted_html += '<meta http-equiv="content-type" content="text/html; charset=UTF-8">'
        sorted_html += '<meta charset="utf-8">'
        sorted_html += f"<script>function copySentence(event) {{navigator.clipboard.writeText('{phrase}');}}</script>"
        sorted_html += '<div id="google_translate_element"></div>'
        sorted_html += '<script type="text/javascript">function googleTranslateElementInit() {new google.translate.TranslateElement({pageLanguage: "hi", layout: google.translate.TranslateElement.InlineLayout.SIMPLE}, "google_translate_element");}</script>'
        sorted_html += '<script type="text/javascript" src="https://translate.google.com/translate_a/element.js?cb=googleTranslateElementInit"></script>'
        sorted_html += '<ul>'
        for link, title, description in sorted_links_and_descriptions:
            sorted_html += f'<li><a href="{link}" target="_blank" onclick="copySentence(event)">{title}</a> - {description}</li>'
        sorted_html += '</ul></body></html>'
        print("Sorted HTML created")

        # Save the sorted HTML to the output file
        with open(output_file_name, 'w', encoding='utf-8', errors='ignore') as f:
            f.write(sorted_html)
            print("Sorted HTML saved to output file as", output_file_name)

    except Exception as e:
        print("An error occurred:", str(e))

def get_user_input():
    return input("Enter the phrase to search: ")

# Usage example
start = time.time()
root_dir = '../'  # Adjust as needed
phrase = get_user_input()
words = ','.join(phrase.split())
output_file = 'search_multiple_words.html'
search_words(root_dir, words, output_file, phrase)
sort_html_file(output_file, 'sorted_search_results.html', phrase)

# Create the final HTML file
with open('search-phrase.html', 'w', encoding='utf-8') as f:
    f.write('<html><body>')
    f.write('<h3>आपकी महत्त्वपूर्ण जिज्ञासाएँ - पूज्य गुरुसत्ता के समाधान</h3>')
    f.write("<meta http-equiv='content-type' content='text/html; charset=UTF-8'><meta charset='utf-8'><script>function copySentence(event) {navigator.clipboard.writeText(event.target.textContent);}</script><div id='google_translate_element'></div><script type='text/javascript'>function googleTranslateElementInit() {new google.translate.TranslateElement({pageLanguage: 'hi', layout: google.translate.TranslateElement.InlineLayout.SIMPLE}, 'google_translate_element');}</script><script type='text/javascript' src='https://translate.google.com/translate_a/element.js?cb=googleTranslateElementInit'></script><ul>")
    f.write(f'<li><a href="sorted_search_results.html" target="_blank" onclick="copySentence(event)">{phrase}</a></li>')
    f.write('</ul></body></html>')

print('search-phrase.html created successfully')

end = time.time()
print("Elapsed (with compilation) = %s" % (end - start))
