PrathamOrgAI/ReadNet
ReadNet Dataset Description ReadNet is an audio dataset collected from more than 2,00,000 children in the age group of 5-16 years in Hindi and Marathi language. The dataset consists of audio files in the wav format where children read out ASER Samples which consists of letters, words, stories and paragraphs in their native language. This dataset is a subset of the larger dataset which consists of an estimated 2500 hours of data. This dataset consists of ~87 hours… See the full description on the dataset page: https://huggingface.co/datasets/PrathamOrgAI/ReadNet.
0125
1import os2import pandas as pd3import requests4df=pd.read_csv("MR_Words.csv")5 6def download_audio_from_dataframe(df, url_column, folder_path):7 for index, row in df.iterrows():8 audio_url = row[url_column]9 if pd.notna(audio_url) and isinstance(audio_url, str):10 file_name = os.path.basename(audio_url)11 12 try:13 # Create the folder if it doesn't exist14 os.makedirs(folder_path, exist_ok=True)15 16 # Combine folder path and file name to get the full path17 full_path = os.path.join(folder_path, file_name)18 19 # Send a GET request to the URL20 response = requests.get(audio_url)21 22 # Check if the request was successful (status code 200)23 if response.status_code == 200:24 # Open a local file and write the content of the response25 with open(full_path, 'wb') as audio_file:26 audio_file.write(response.content)27 print(f"Audio file downloaded successfully to {full_path}")28 else:29 print(f"Failed to download audio file. Status code: {response.status_code}")30 31 except Exception as e:32 print(f"Error: {e}")33 print(file_name)34 continue35 36# Example usage37# Assuming you have a DataFrame named 'df' with a column 'audio_links' containing URLs with filenames38 39 40# Specify the column name41url_column_name = 'recordingserverurl'42 43# Specify the download folder44download_folder = 'path_to/Marathi/words'45 46# Call the function to download audio files47download_audio_from_dataframe(df, url_column_name, download_folder)48 49 