Team Ai
Datasetpublic

CedRuiz/data_ClassificationModel

Dataset Card for "data_ClassificationModel"--- dataset_info: features: - name: brands dtype: string - name: categories dtype: string - name: code dtype: string - name: languages_tags dtype: string - name: last_modified_t dtype: int64 - name: product_name_de dtype: string - name: quantity dtype: string - name: index_level_0 dtype: int64 splits: - name: train num_bytes: 231023 num_examples: 673… See the full description on the dataset page: https://huggingface.co/datasets/CedRuiz/data_ClassificationModel.

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes133downloads
streamToParquet.py63 linesDownload Raw Back to root
1import os2import pyarrow as pa3import pyarrow.parquet as pq4import json5 6TEXT_DIR = "./test.txt ( Data GVK)"7OUTPUT_FILE = "data/test_large2.parquet"8BATCH_SIZE = 50009 10# Initialize a Parquet writer with schema11schema = pa.schema([12    ("filename", pa.string()),13    ("full_text", pa.string())14])15writer = pq.ParquetWriter(OUTPUT_FILE, schema)16 17buffer = []18processed = 019skipped = 020 21# === Loop through all .txt files ===22for i, filename in enumerate(os.listdir(TEXT_DIR), start=1):23    if not filename.endswith(".txt"):24        continue25 26    file_path = os.path.join(TEXT_DIR, filename)27 28    # Try to parse the .txt as JSON29    try:30        with open(file_path, "r", encoding="utf-8") as f:31            data = json.load(f)32    except json.JSONDecodeError:33        print(f"Skipping {filename} (invalid JSON format)")34        skipped += 135        continue36 37    # Extract the text field from the OCR response38    try:39        full_text = data["responses"][0]["textAnnotations"][0]["description"]40    except (KeyError, IndexError, TypeError):41        full_text = ""42 43    buffer.append({"filename": filename, "full_text": full_text})44    processed += 145 46    # Write periodically in chunks to avoid high memory use47    if i % BATCH_SIZE == 0:48        table = pa.Table.from_pylist(buffer, schema=schema)49        writer.write_table(table)50        print(f"Processed and wrote {processed} files so far...")51        buffer = []52 53# === Final flush ===54if buffer:55    table = pa.Table.from_pylist(buffer, schema=schema)56    writer.write_table(table)57 58writer.close()59 60print("Done.")61print(f"Saved Parquet file: {OUTPUT_FILE}")62print(f"Total processed: {processed}")63print(f"Skipped (invalid JSON): {skipped}")