CedRuiz/data_ClassificationModel
Dataset Card for "data_ClassificationModel"--- dataset_info: features: - name: brands dtype: string - name: categories dtype: string - name: code dtype: string - name: languages_tags dtype: string - name: last_modified_t dtype: int64 - name: product_name_de dtype: string - name: quantity dtype: string - name: index_level_0 dtype: int64 splits: - name: train num_bytes: 231023 num_examples: 673… See the full description on the dataset page: https://huggingface.co/datasets/CedRuiz/data_ClassificationModel.
0133
1import os2import pyarrow as pa3import pyarrow.parquet as pq4import json5 6TEXT_DIR = "./test.txt ( Data GVK)"7OUTPUT_FILE = "data/test_large2.parquet"8BATCH_SIZE = 50009 10# Initialize a Parquet writer with schema11schema = pa.schema([12 ("filename", pa.string()),13 ("full_text", pa.string())14])15writer = pq.ParquetWriter(OUTPUT_FILE, schema)16 17buffer = []18processed = 019skipped = 020 21# === Loop through all .txt files ===22for i, filename in enumerate(os.listdir(TEXT_DIR), start=1):23 if not filename.endswith(".txt"):24 continue25 26 file_path = os.path.join(TEXT_DIR, filename)27 28 # Try to parse the .txt as JSON29 try:30 with open(file_path, "r", encoding="utf-8") as f:31 data = json.load(f)32 except json.JSONDecodeError:33 print(f"Skipping {filename} (invalid JSON format)")34 skipped += 135 continue36 37 # Extract the text field from the OCR response38 try:39 full_text = data["responses"][0]["textAnnotations"][0]["description"]40 except (KeyError, IndexError, TypeError):41 full_text = ""42 43 buffer.append({"filename": filename, "full_text": full_text})44 processed += 145 46 # Write periodically in chunks to avoid high memory use47 if i % BATCH_SIZE == 0:48 table = pa.Table.from_pylist(buffer, schema=schema)49 writer.write_table(table)50 print(f"Processed and wrote {processed} files so far...")51 buffer = []52 53# === Final flush ===54if buffer:55 table = pa.Table.from_pylist(buffer, schema=schema)56 writer.write_table(table)57 58writer.close()59 60print("Done.")61print(f"Saved Parquet file: {OUTPUT_FILE}")62print(f"Total processed: {processed}")63print(f"Skipped (invalid JSON): {skipped}")