#!/usr/bin/env python

import os
import re
import time
from bs4 import BeautifulSoup, NavigableString, Tag

start = time.time()

def find_replace(root_dir, find_text, replace_text, output_file):
    find_words = [word.strip() for word in find_text.split(',')]
    replace_words = [word.strip() for word in replace_text.split(',')]
    if len(find_words) != len(replace_words):
        print("Error: Number of find words and replace words must be the same. Using only matching pairs.")
        min_len = min(len(find_words), len(replace_words))
        find_words = find_words[:min_len]
        replace_words = replace_words[:min_len]

    script_dir = os.path.dirname(os.path.abspath(__file__))
    html = "<html><body><h1>Files Modified during find-replace operation</h1><ul>"

    for root, dirs, files in os.walk(root_dir):
        # Skip the root directory where the script is placed
        if root == script_dir:
            continue

        for file in files:
            if file.endswith('.html'):
                file_path = os.path.join(root, file)
                # Skip files in the root directory
                if root == root_dir:
                    continue
                try:
                    with open(file_path, 'r', encoding='utf-8') as f:
                        content = f.read()
                    soup = BeautifulSoup(content, 'html.parser')
                    modified = False
                    modified_words = []

                    for find_word, replace_word in zip(find_words, replace_words):
                        # Find all <div> tags
                        div_tags = soup.find_all('div')
                        for div in div_tags:
                            # Find all <p> tags within the <div> tag
                            p_tags = div.find_all('p')
                            for p in p_tags:
                                # Concatenate the text content of all child nodes within the <p> tag
                                p_text = ''.join(str(child) for child in p.children if isinstance(child, (NavigableString, Tag)))
                                # Find all <a> tags within the <p> tag and check if find_word is within any of them
                                a_tags = p.find_all('a')
                                skip_replace = any(find_word in a.get_text() or find_word in a.get('title', '') for a in a_tags)

                                # Check if find_word matches the entire content of an <a> tag
                                allow_replace = any(str(a) == find_word for a in a_tags)

                                if allow_replace:
                                    print(f"Found '{find_word}' as full <a> tag in {file_path}")
                                    for a in a_tags:
                                        if str(a) == find_word:
                                            a.replace_with(replace_word)
                                    modified = True
                                    modified_words.append(f"{find_word} -> {replace_word}")
                                elif not skip_replace and find_word in p_text:
                                    print(f"Found '{find_word}' in {file_path}")
                                    # Replace find_word in the text content of the <p> tag
                                    new_p_text = p_text.replace(find_word, replace_word)
                                    # Replace the text content of the <p> tag
                                    p.clear()
                                    p.append(BeautifulSoup(new_p_text, 'html.parser'))
                                    modified = True
                                    modified_words.append(f"{find_word} -> {replace_word}")

                        # Find all heading tags
                        heading_tags = ['h1', 'h2', 'h3', 'h4', 'h5', 'h6']
                        for tag in heading_tags:
                            for heading in soup.find_all(tag):
                                heading_text = ''.join(str(child) for child in heading.children if isinstance(child, (NavigableString, Tag)))
                                if find_word in heading_text:
                                    print(f"Found '{find_word}' in {file_path}")
                                    # Replace find_word in the text content of the heading tag
                                    new_heading_text = heading_text.replace(find_word, replace_word)
                                    # Replace the text content of the heading tag
                                    heading.clear()
                                    heading.append(BeautifulSoup(new_heading_text, 'html.parser'))
                                    modified = True
                                    modified_words.append(f"{find_word} -> {replace_word}")

                    # Exclude searching within <title> tags
                    title_tag = soup.find('title')
                    if title_tag:
                        title_tag.replace_with(title_tag)

                    if modified:
                        with open(file_path, 'w', encoding='utf-8') as f:
                            f.write(str(soup))
                        print(f"Modified {file_path}")
                        html += f"<li><a href='./{file_path}'>{file}</a> - Modified words: {', '.join(modified_words)}</li>"
                except Exception as e:
                    print(f"Error reading {file_path}: {e}")

    html += "</ul></body></html>"
    with open(output_file, 'w', encoding='utf-8') as f:
        f.write(html)
    print("See modified files list in", output_file)

root_dir = './'
# Example: Plain text to plain text
# find_text = '''गुरुदेव'''
# replace_text = '''पूज्य गुरुदेव'''

# Example: HTML anchor tag to plain text
find_text = '''<a href="../folders-special/hamari-vasiyat-aur-virasat.html" target="blank" title="पूज्य गुरुदेव— मैं व्यक्ति नहीं विचार हूँ।.....हम व्यक्ति के रूप में कब से खत्म हो गए। हम एक व्यक्ति हैं? नहीं हैं। हम कोई व्यक्ति नहीं हैं। हम एक सिद्धांत हैं, आदर्श हैं, हम एक दिशा हैं, हम एक प्रेरणा हैं।">पूज्य गुरुदेव</a>'''
replace_text = '''पूज्य गुरुदेव'''

output_file = 'find-replace-single-word-without-root-files-conditional.html'
find_replace(root_dir, find_text, replace_text, output_file)

end = time.time()
print("Elapsed (with compilation) = %s" % (end - start))
