import re

def sort_html_file(input_file_name, output_file_name, phrase):
    try:
        with open(input_file_name, 'r', encoding='utf-8', errors='ignore') as f:
            html = f.read()
            print("HTML loaded successfully")

        # Find all links and descriptions
        links_and_descriptions = re.findall(r'<li>.*?<\/li>', html, re.DOTALL)
        print("Links and descriptions found:", len(links_and_descriptions))

        # Extract link and description from each item
        extracted_links_and_descriptions = []
        for item in links_and_descriptions:
            match = re.search(r'href=[\'"]([^\'"]+)[\'"]', item)
            if match:
                link = match.group(1)
                match = re.search(r'>[^<]+?<\/a>', item)
                if match:
                    title = match.group(0)[1:-4]
                    description = re.sub(r'<.*?>', '', item).strip().split('-', 1)[1].strip()
                    extracted_links_and_descriptions.append((link, title, description))

        print("Extracted links and descriptions:", len(extracted_links_and_descriptions))

        # Sort links and descriptions based on the presence of the phrase
        phrase_links_and_descriptions = [x for x in extracted_links_and_descriptions if phrase.lower() in x[2].lower()]
        other_links_and_descriptions = [x for x in extracted_links_and_descriptions if phrase.lower() not in x[2].lower()]

        sorted_phrase_links_and_descriptions = sorted(phrase_links_and_descriptions, key=lambda x: x[2].lower().find(phrase.lower()))
        sorted_other_links_and_descriptions = sorted(other_links_and_descriptions, key=lambda x: x[2].lower().find(phrase.lower()))

        sorted_links_and_descriptions = []
        words = phrase.split()
        added_descriptions = set()

        for link, title, description in sorted_phrase_links_and_descriptions + sorted_other_links_and_descriptions:
            if phrase.lower() in description.lower():
                description = re.sub(r'(' + re.escape(phrase) + r')', r'<b style="background-color: #ffffb0; color: #0000FF; font-weight: bold">\1</b>', description, flags=re.IGNORECASE)
            for word in words:
                if phrase.lower() not in description.lower():
                    description = re.sub(r'(' + re.escape(word) + r')', r'<b>\1</b>', description, flags=re.IGNORECASE)
            
            description_words = description.split()
            description_hash = tuple(sorted(description_words[:7]))  # Use the first 7 words as a hash
            if description_hash not in added_descriptions:
                sorted_links_and_descriptions.append((link, title, description))
                added_descriptions.add(description_hash)

        print("Links and descriptions sorted")

        # Create the sorted HTML
        sorted_html = '<html><body><h1>Search Results</h1><ul>'
        sorted_html = '<html><body><h1>Search Results</h1>'
        sorted_html += '<meta http-equiv="content-type" content="text/html; charset=UTF-8">'
        sorted_html += '<meta charset="utf-8">'
        sorted_html += '<script>function copySentence(event) {navigator.clipboard.writeText(event.target.textContent);}</script>'
        sorted_html += '<div id="google_translate_element"></div>'
        sorted_html += '<script type="text/javascript">function googleTranslateElementInit() {new google.translate.TranslateElement({pageLanguage: "hi", layout: google.translate.TranslateElement.InlineLayout.SIMPLE}, "google_translate_element");}</script>'
        sorted_html += '<script type="text/javascript" src="https://translate.google.com/translate_a/element.js?cb=googleTranslateElementInit"></script>'
        sorted_html += '<ul>'
        for link, title, description in sorted_links_and_descriptions[:38]:
            sorted_html += f'<li><a href="{link}" target="_blank">{title}</a> - {description}</li>'
        sorted_html += '</ul></body></html>'
        print("Sorted HTML created")

        # Save the sorted HTML to the output file
        with open(output_file_name, 'w', encoding='utf-8', errors='ignore') as f:
            f.write(sorted_html)
            print("Sorted HTML saved to output file as",output_file_name)

    except Exception as e:
        print("An error occurred:", str(e))

# Usage
input_file_name ='search_multiple_words.html'
output_file_name ='adhyatm-kya-hai.html'
phrase = 'अध्यात्म क्या है'
# sort_html_file(input_file_name, output_file_name, phrase)
