luulinh90s/Tabular-LLM-Study-Debugging
0
1import os2import re3from pathlib import Path4from bs4 import BeautifulSoup5 6 7def process_html_file(file_path, output_path):8 with open(file_path, 'r', encoding='utf-8') as file:9 content = file.read()10 11 soup = BeautifulSoup(content, 'html.parser')12 13 # Find the Statement line14 statement_tag = soup.find(lambda tag: tag.name == "h3" and tag.find("span", string="Statement:"))15 16 if statement_tag:17 # Extract the text content18 statement_text = statement_tag.get_text(strip=True)19 20 # Remove "in the table:" and everything after it21 new_statement = re.sub(r'\s*in the table:.*$', '', statement_text, flags=re.DOTALL)22 23 # Reconstruct the h3 tag with the modified content24 new_h3 = soup.new_tag('h3')25 new_span = soup.new_tag('span')26 new_span.string = 'Statement:'27 new_h3.append(new_span)28 new_h3.append(f" {new_statement}")29 30 # Replace the old h3 tag with the new one31 statement_tag.replace_with(new_h3)32 33 # Write the modified content34 with open(output_path, 'w', encoding='utf-8') as file:35 file.write(str(soup))36 37 38def process_directory(input_dir, output_dir):39 subfolders = ['TP', 'TN', 'FP', 'FN']40 41 for subfolder in subfolders:42 input_subfolder = Path(input_dir) / subfolder43 output_subfolder = Path(output_dir) / subfolder44 45 if not input_subfolder.exists():46 print(f"Warning: {input_subfolder} does not exist. Skipping.")47 continue48 49 output_subfolder.mkdir(parents=True, exist_ok=True)50 51 for file in input_subfolder.glob('*.html'):52 output_file = output_subfolder / file.name53 process_html_file(file, output_file)54 print(f"Processed: {file} -> {output_file}")55 56 57# Define input and output directories58input_directory = "htmls_DATER_mod"59output_directory = "htmls_DATER_mod2"60 61# Process the files62process_directory(input_directory, output_directory)63 64print("Processing complete. Modified files are in the output directory.")