import os
import re
from bs4 import BeautifulSoup

def extract_audio_urls_from_file(input_file, base_url, output_file):
    try:
        # Read the contents of the input file
        with open(input_file, 'r') as file:
            content = file.read()

        audio_files = []

        # If it's an HTML file, we use BeautifulSoup to parse and extract the <a> tags
        if input_file.endswith('.html'):
            soup = BeautifulSoup(content, 'html.parser')

            # Find all <a> tags with href that end in .mp3 or .ogg
            audio_files = [a['href'] for a in soup.find_all('a', href=True) if a['href'].endswith(('.mp3', '.ogg'))]

        # If it's a TXT file, we use a regular expression to find URLs
        elif input_file.endswith('.txt'):
            # Find all URLs ending in .mp3 or .ogg using regex
            audio_files = re.findall(r'(https?://[^\s]+(?:\.mp3|\.ogg))', content)

        # Prepend the base URL to each file if the links are not complete URLs
        full_audio_urls = [url if url.startswith('http') else f"{base_url}/{url}" for url in audio_files]

        # Write the audio URLs to the output file
        with open(output_file, 'w') as out_file:
            for url in full_audio_urls:
                out_file.write(f"{url},\n")

        print(f"Audio URLs have been extracted and saved to {output_file}")

    except Exception as e:
        print(f"An error occurred: {e}")


# Example usage
input_file_path = 'geetvmataji.html'  # Replace with the path to your input HTML or TXT file
base_url = 'https://geocities.ws/brijesh/amritvachan-audio/aj-pravachan'  # Replace with your base URL
output_file_path = 'output_audio_urls.txt'  # Path where the audio URLs will be saved

extract_audio_urls_from_file(input_file_path, base_url, output_file_path)
