#!/usr/bin/env python

import os
import time
from bs4 import BeautifulSoup, NavigableString, Tag

start = time.time()

def find_replace(root_dir, find_text, replace_text, output_file):
    find_words = [word.strip() for word in find_text.split(',')]
    replace_words = [replace_text.strip()] * len(find_words)  # Ensure replace_text is used as is for each find_word

    script_dir = os.path.dirname(os.path.abspath(__file__))
    html = "<html><body><h1>Files Modified during find-replace operation</h1><ul>"

    for root, dirs, files in os.walk(root_dir):
        # Skip the root directory where the script is placed
        if root == script_dir:
            continue

        for file in files:
            if file.endswith('.html'):
                file_path = os.path.join(root, file)
                # Skip files in the root directory
                if root == root_dir:
                    continue
                try:
                    with open(file_path, 'r', encoding='utf-8') as f:
                        content = f.read()
                    soup = BeautifulSoup(content, 'html.parser')
                    modified = False
                    modified_words = []

                    for find_word, replace_word in zip(find_words, replace_words):
                        # Find all <div> tags
                        div_tags = soup.find_all('div')
                        for div in div_tags:
                            # Find all <p> tags within the <div> tag
                            p_tags = div.find_all('p')
                            for p in p_tags:
                                # Concatenate the text content of all child nodes within the <p> tag
                                p_text = ''.join(str(child) for child in p.children if isinstance(child, (NavigableString, Tag)))
                                # Find exact match of find_word in the text content of the <p> tag
                                if find_word in p_text:
                                    print(f"Found '{find_word}' in {file_path}")
                                    # Replace find_word with the given HTML anchor tag
                                    new_p_text = p_text.replace(find_word, replace_word)
                                    # Replace the text content of the <p> tag
                                    p.clear()
                                    p.append(BeautifulSoup(new_p_text, 'html.parser'))
                                    modified = True
                                    modified_words.append(f"{find_word} -> {replace_word}")

                        # Find all heading tags
                        heading_tags = ['h1', 'h2', 'h3', 'h4', 'h5', 'h6']
                        for tag in heading_tags:
                            for heading in soup.find_all(tag):
                                heading_text = ''.join(str(child) for child in heading.children if isinstance(child, (NavigableString, Tag)))
                                # Find exact match of find_word in the text content of the heading tag
                                if find_word in heading_text:
                                    print(f"Found '{find_word}' in {file_path}")
                                    # Replace find_word with the given HTML anchor tag
                                    new_heading_text = heading_text.replace(find_word, replace_word)
                                    # Replace the text content of the heading tag
                                    heading.clear()
                                    heading.append(BeautifulSoup(new_heading_text, 'html.parser'))
                                    modified = True
                                    modified_words.append(f"{find_word} -> {replace_word}")

                    # Exclude searching within <title> tags
                    title_tag = soup.find('title')
                    if title_tag:
                        title_tag.replace_with(title_tag)

                    if modified:
                        with open(file_path, 'w', encoding='utf-8') as f:
                            f.write(str(soup))
                        print(f"Modified {file_path}")
                        html += f"<li><a href='./{file_path}'>{file}</a> - Modified words: {', '.join(modified_words)}</li>"
                except Exception as e:
                    print(f"Error reading {file_path}: {e}")

    html += "</ul></body></html>"
    with open(output_file, 'w', encoding='utf-8') as f:
        f.write(html)
    print("See modified files list in", output_file)

root_dir = './'
# Example: Plain text to HTML anchor tag
find_text = '''गुरुदेव'''
replace_text = '''<a href="../folders-special/hamari-vasiyat-aur-virasat.html" target="blank" title="पूज्य गुरुदेव— मैं व्यक्ति नहीं विचार हूँ।.....हम व्यक्ति के रूप में कब से खत्म हो गए। हम एक व्यक्ति हैं? नहीं हैं। हम कोई व्यक्ति नहीं हैं। हम एक सिद्धांत हैं, आदर्श हैं, हम एक दिशा हैं, हम एक प्रेरणा हैं।">पूज्य गुरुदेव</a>'''

output_file = 'find-replace-single-word-without-root-files-conditional.html'
find_replace(root_dir, find_text, replace_text, output_file)

end = time.time()
print("Elapsed (with compilation) = %s" % (end - start))
