Team Ai
Datasetpublic

CedRuiz/data_ClassificationModel

Dataset Card for "data_ClassificationModel"--- dataset_info: features: - name: brands dtype: string - name: categories dtype: string - name: code dtype: string - name: languages_tags dtype: string - name: last_modified_t dtype: int64 - name: product_name_de dtype: string - name: quantity dtype: string - name: index_level_0 dtype: int64 splits: - name: train num_bytes: 231023 num_examples: 673… See the full description on the dataset page: https://huggingface.co/datasets/CedRuiz/data_ClassificationModel.

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes133downloads
parset_pipeline.py343 linesDownload Raw Back to root
1import os2import json3import re4import pyarrow as pa5import pyarrow.parquet as pq6from datasets import load_dataset7import glob8import re9import unicodedata10import datetime11 12SKIP_KEYWORDS = set([13    # Receipt metadata / math14    "summe","zwischensumme","gesamt","gesamtbetrag","zwischenbetrag",15    "brutto","netto","steuer","steuerbetrag","ust","mwst","mehrwertsteuer",16    "betrag","wechselgeld","rabatt","abzug","rundung","rabattsumme",17    "beleg","bon","kassenbon","kundenbeleg","händlerbeleg",18    "datum","uhr","uhrzeit","zeit","tag","monat","jahr","wochentag",19    "mo","di","mi","do","fr","sa","so","montag","dienstag","mittwoch",20    "donnerstag","freitag","samstag","sonntag",21    "kasse","kassierer","bed","bediener","mitarbeiter","filiale","markt",22    "markt-nr","marktnr","marktnummer","seriennummer","server","druck",23    "druckzeit","belegdruck","intern","vorgang","vorgangsnummer",24    "transaktion","transaktionsnummer","trace","pos","terminal","terminal-id",25    # Supermarkets / chains26    "rewe","edeka","aldi","lidl","netto","penny","kaufland","rossmann","dm",27    "real","metro","nahkauf","globus","tegut","marktkauf","nah und gut",28    "billa","spar","hofer","famila","toom","hit","müller","mueller",29    # Payment / finance30    "zahlung","kartenzahlung","barzahlung","kontaktlos","pin",31    "visa","mastercard","american express","amex","diners","maestro",32    "girocard","ec","ec-cash","debit","credit","kreditkarte",33    "auth","authorisierung","autorisierung","genehmigung",34    "autorisierung erfolgt","zahlung erfolgt","zahlung genehmigt",35    "referenz","reference","ref","receipt","paypal","klarna","apple pay","google pay",36    "postbank","sparkasse","commerzbank","volksbank","raiffeisenbank",37    # Legal / tax / invoice38    "ust-","ust","ustnr","ustid","uid","uidnr","umsatzsteuer",39    "steuer-nr","steuernummer","umsatzsteuer-id","rnr","rechnung",40    "kundennr","kundennummer","kunden-nr",41    # Contact & address42    "tel","telefon","fax","email","mail","straße","strasse","str.","weg","platz",43    "allee","ring","gasse","ufer","hof","chaussee","pl.","strs","chaus.",44    "berlin","hamburg","münchen","muenchen","frankfurt","dortmund",45    "köln","koeln","hannover","stuttgart","bremen","leipzig",46    "nürnberg","nuernberg","dresden","bochum","wuppertal",47    # Marketing / promo / generic info48    "produkt","produkte","sortiment","aktion","aktionsangebot","aktionspreis",49    "angebot","angebote","angebotspreis","sonderpreis","rabattaktion",50    "discount","preis","sparpreis","gilt","gültig","cashback","bonusprogramm",51    "bonus","coupon","coupons","einlösen","einlösbar","einlösungen",52    "einkaufswert","ersparnis","verkauf","rückgabe","abwicklung","kontakt",53    "werbung","homepage","website","webseite","www.","online","shop","shoppe",54    "besuche uns","besuch uns","folge uns","wir sind jetzt auf",55    "instagram","facebook","tiktok","twitter","youtube","payback","payback pay",56    "punkte","punktestand","punktezahl","kundenkarte","treuepunkte","gutschein",57    # Recruitment / jobs58    "möchtest","moechtest","bewirb","bewerbe","bewerbung","karriere","team",59    "arbeit","arbeiten","stelle","jobs","job","werde teil","teil unseres teams",60    "jetzt bewerben","join","jobportal","karriereportal",61    # Customer info / FAQ62    "sie haben fragen","fragen","antwort","antworten","hilfe",63    "kundenservice","kundendienst","reklamation","service",64    "bitte beachten sie","beachten sie","wir auch","erfahre mehr","mehr erfahren",65    # Additives / labeling / packaging66    "zusatz","zusatzstoff","zusatzstoffe","künstlich","kuenstlich","künstlichen",67    "farb","farbstoff","farbstoffen","konservierungsstoffe",68    "geschmacksverstärker","aroma","zutaten","inhaltstoffe","verbrauch",69    "verbraucher","aufbewahrung","rückseite","gebrauchsanweisung",70    "erzeugnis","produktinformationen","hinweis","hinweise","serviervorschlag",71    "informationen","auf basis","auf basis von",72    # Deposit / refund73    "pfand","pfand euro","pfand eur","mehrweg","einweg","pfandrückgabe","pfandrueckgabe",74    # Locations / malls / travel promos75    "einkaufszentrum","einkaufs-zentrum","center","city center","city-center",76    "shopping","passage","galerie","marktzentrum","marktcenter",77    "einkaufspark","aez","amper-einkaufs-zentrum","amper","reisen","rewe reisen",78    # Random / technical79    "hash","token","gerät","software","system","signatur","zähler",80    "vorgang abgeschlossen","verbindung","netzwerk",81    # Small units / quantity tokens not as products82    "stk","stck","stück","stueck","kg","g","ml","l","cl","packung","flasche",83    # Phrases you flagged84    "app joker","karte 0","ready to","to go"85])86 87# === CONFIG ===88TEXT_DIR = "./test.txt ( Data GVK)"# folder containing .txt JSON files89run_id = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")90TEMP_OUTPUT_FILE = "data/test_reformatted.parquet"91FINAL_OUTPUT_FILE = "data/test_aligned.parquet"92BATCH_SIZE = 100093REPO_ID = "CedRuiz/data_ClassificationModel"94DATA_DIR = "data"95PROCESSED_LOG = "processed_files.txt"96 97# === Define schema matching your model dataset ===98# schema = pa.schema([99#     ("product_name_de", pa.string()),100#     ("categories", pa.string())101# ])102# writer = pq.ParquetWriter(TEMP_OUTPUT_FILE, schema)103 104if os.path.exists(PROCESSED_LOG):105    with open(PROCESSED_LOG, "r", encoding="utf-8") as f:106        processed_files = set(f.read().splitlines())107else:108    processed_files = set()109 110def extract_products(full_text: str):111    """112    Return a list of product-like lines from OCR text.113    Aggressively filters out receipt metadata, addresses, payment, promos, recruitment,114    social, legal/tax, random tokens, and non-product phrases.115    """116    unique_products = set(buffer_entry["product_name_de"] for buffer_entry in buffer)117    WHITELIST_CUES = {118        "bio", "vegan", "vegetarisch", "glutenfrei", "laktosefrei",119        "light", "zero", "protein", "fit", "natural", "classic",120        "fresh", "pure", "organic", "soja", "hafer", "reis",121        "mandel", "oat", "coconut", "ohne", "zuckerfrei", "free",122        "superfood", "plantbased", "pflanzlich", "green", "eco"123    }124 125    # Normalize Unicode (ä→ä, etc.), keep original for output later126    text_norm = unicodedata.normalize("NFKC", full_text)127    lines = text_norm.split("\n")128    products = []129 130 131    # Precompiled regexes132    re_has_letters = re.compile(r"[A-Za-zÄÖÜäöüß]")133    re_price = re.compile(r"\b\d+[.,]\d{2}\b")134    re_price_trailer = re.compile(r"\b\d+[.,]\d{2}\s*[A-Z]?\b")135    re_pct = re.compile(r"\b\d{1,2}\s*%\b")136 137    # Phones / addresses138    re_phone = re.compile(r"(tel|telefon|fax)\s*[:\.]?\s*\d", re.IGNORECASE)139    re_phone_plain = re.compile(r"\b\d{3,4}[-/ ]?\d{3,}\b")140    re_postal = re.compile(r"\b\d{4,5}\b")141    re_street = re.compile(r"\b(str|straße|strasse|weg|platz|allee|ring|gasse|ufer|chaussee)\b", re.IGNORECASE)142    re_cityline = re.compile(143        r"\b(berlin|hamburg|m(ü|u)nchen|frankfurt|dortmund|k(ö|oe)ln|hannover|stuttgart|bremen|leipzig|n(ü|ue)rnberg|dresden|bochum|wuppertal)\b",144        re.IGNORECASE)145 146    # Payments/brands/online/social147    re_cardbrands = re.compile(148        r"(american\s*express|amex|paypal|klarna|apple\s*pay|google\s*pay|visa|mastercard|maestro|girocard|ec[- ]?cash)",149        re.IGNORECASE)150    re_payment_words = re.compile(r"(kartenzahlung|barzahlung|kontaktlos|kreditkarte|debit|credit)", re.IGNORECASE)151    re_url = re.compile(r"(https?://|www\.)", re.IGNORECASE)152    re_social = re.compile(r"(instagram|facebook|tiktok|twitter|youtube)", re.IGNORECASE)153 154    # Legal/tax/IDs/random155    re_taxid = re.compile(r"\b(de|at|ch)?\s*[-–]?\s*\d{8,12}\b", re.IGNORECASE)156    re_iban = re.compile(r"\b[A-Z]{2}\d{2}[A-Z0-9]{10,30}\b")157    re_random_token = re.compile(r"\b[a-zA-Z0-9]{10,}\b")158 159    # Recruitment / promo / mall160    re_recruit = re.compile(161        r"(bewerb|möchtest|moechtest|arbeiten|teil\s+unser(es|s)?\s+teams|wir\s+sind|jetzt\s+bewerben)", re.IGNORECASE)162    re_promo = re.compile(r"(discount|preis|angebot|rabatt|aktionspreis|preiswert|sparpreis)", re.IGNORECASE)163    re_mall = re.compile(r"(einkaufs.?zentrum|center|galerie|passage|shopping|aez|amper)", re.IGNORECASE)164 165    PHRASE_SKIP = [166        "auf basis von", "foto von", "app joker", "karte 0", "payback pay",167        "rewe reisen", "bitte beachten sie", "wir auch", "erfahre mehr", "mehr erfahren"168    ]169    # Many receipts are shouty; allow uppercase, but we still need a product-like shape.170    for raw in lines:171        original_line = raw.strip()172        if not original_line:173            continue174 175        # Quick reject: must contain a letter176        if not re_has_letters.search(original_line):177            continue178 179        low = original_line.lower()180        # Keyword skip (broad)181        if any(k in low for k in SKIP_KEYWORDS):182            if not any(w in low for w in WHITELIST_CUES):183                continue184 185 186        if any(phrase in low for phrase in PHRASE_SKIP):187            continue188        # Numeric-word artifacts like "1701 Whr", "App Joker 25"189        if re.search(r"\b\d+\s*[A-Za-zÄÖÜäöüß]{2,}\b", original_line):190            continue  # numbers attached to text tokens (1701 Whr, App Joker)191 192        # Lines that start or end with a lone number or short token like "Karte 0"193        if re.match(r"^(karte|app|foto)\s*\d+\b", low):194            continue195        if re.search(r"\b\d+\s*(karte|app|foto)\b", low):196            continue197 198        # Phone / address / postal199        if re_phone.search(low) or re_phone_plain.search(original_line):200            continue201        if re_postal.search(original_line) or re_street.search(low) or re_cityline.search(low):202            continue203 204        # Payment, brands, URLs, social, recruitment, promo, mall205        if re_cardbrands.search(low) or re_payment_words.search(low):206            continue207        if re_url.search(low) or re_social.search(low):208            continue209        if re_recruit.search(low) or re_promo.search(low) or re_mall.search(low):210            continue211 212        # Legal/tax IDs / IBAN / random hashes213        if re_taxid.search(low) or re_iban.search(original_line) or re_random_token.search(original_line):214            continue215 216        # Strip obvious numeric/price trailers (don't reject the line; clean it)217        cleaned = re_price_trailer.sub("", original_line)218        cleaned = re_price.sub("", cleaned)219        cleaned = re_pct.sub("", cleaned)220 221        # Remove lingering non-alnum (keep spaces, dots, hyphens)222        cleaned = re.sub(r"[^A-Za-zÄÖÜäöüß0-9\s\-.]", "", cleaned).strip()223 224        # Heuristics: keep lines that look like short product names225        #  - Between 2 and 6 tokens typically works well for receipts226        tokens = cleaned.split()227        if not (1 <= len(tokens) <= 9):228            continue229 230        # Require at least one uppercase (helps skip purely descriptive sentences)231        if not re.search(r"[A-ZÄÖÜ]", cleaned):232            continue233 234        # Avoid generic one-word leftovers like "Produkt", "Sortiment"235        if len("".join(tokens)) <= 3:236            continue237 238        # Final tiny blacklist on line shape239        if cleaned.lower() in {"produkt", "sortiment"}:240            continue241 242        # Deduplicate consecutive duplicate tokens (e.g., "VEG VEG SALAMI")243        dedup = []244        for t in tokens:245            if not dedup or t != dedup[-1]:246                dedup.append(t)247        cleaned = " ".join(dedup)248 249        # Guard against very long char length (often sentences)250        if len(cleaned) > 48:251            continue252 253        if any(w in cleaned.lower() for w in WHITELIST_CUES) and len(tokens) >= 2:254            products.append(cleaned)255            continue256 257        # Looks like a product258        products.append(cleaned)259 260    print(f"Unique product names extracted: {len(unique_products)}")261    return products262 263 264 265buffer = []266processed = 0267skipped = 0268new_files = []269 270# === Loop through all .txt files ===271for i, filename in enumerate(os.listdir(TEXT_DIR), start=1):272    if not filename.endswith(".txt") or filename in processed_files:273        continue274 275    file_path = os.path.join(TEXT_DIR, filename)276    try:277        with open(file_path, "r", encoding="utf-8") as f:278            data = json.load(f)279        full_text = data["responses"][0]["textAnnotations"][0]["description"]280        print(f"[{i}] {filename} — length of OCR text: {len(full_text)}")281    except Exception:282        print(f"Skipping {filename} (invalid JSON)")283        skipped += 1284        continue285 286    products = extract_products(full_text)287    for product in products:288        buffer.append({"product_name_de": product, "categories": None})289        processed += 1290 291    processed_files.add(filename)292    with open(PROCESSED_LOG, "a", encoding="utf-8") as f:293        f.write(filename + "\n")294 295    # Write in batches to save memory296    if i % BATCH_SIZE == 0:297        table = pa.Table.from_pylist(buffer)298        pq.write_table(table, f"data/temp_batch_{i}.parquet")299        buffer = []300        print(f"Processed {i} files, total {processed} products...")301 302# Final flush303# === Write new batch only if new files found ===304# Final flush305if buffer:306 307    run_id = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")308 309    batch_id = len([f for f in os.listdir(DATA_DIR) if f.startswith("parsed_batch_")]) + 1310    batch_file = os.path.join(DATA_DIR, f"parsed_batch_{batch_id}.parquet")311 312    table = pa.Table.from_pylist(buffer)313    pq.write_table(table, batch_file)314 315    print(f"Saved new batch file: {batch_file} ({len(buffer)} products)")316    print(f"File size: {os.path.getsize(batch_file)} bytes")317 318    if os.path.getsize(batch_file) < 100:319        raise ValueError(f"Parquet file {batch_file} seems too small — check data creation.")320 321    # Upload merged dataset322    dataset = load_dataset(323        "parquet",324        data_files={325            "train": "data/train-00000-of-00001.parquet",326            "test": batch_file327        }328    )329 330    dataset.push_to_hub(REPO_ID)331    print(f"Uploaded updated dataset with new batch file: {batch_file}")332 333    # Update processed log334    with open(PROCESSED_LOG, "w", encoding="utf-8") as f:335        f.write("\n".join(sorted(processed_files)))336else:337    print("No new .txt files found to process.")338 339 340print(f"Finished extracting OCR products.")341print(f"Total processed: {processed} entries, skipped: {skipped}")342 343