CedRuiz/data_ClassificationModel
Dataset Card for "data_ClassificationModel"--- dataset_info: features: - name: brands dtype: string - name: categories dtype: string - name: code dtype: string - name: languages_tags dtype: string - name: last_modified_t dtype: int64 - name: product_name_de dtype: string - name: quantity dtype: string - name: index_level_0 dtype: int64 splits: - name: train num_bytes: 231023 num_examples: 673… See the full description on the dataset page: https://huggingface.co/datasets/CedRuiz/data_ClassificationModel.
0133
1import os2import json3import re4import pyarrow as pa5import pyarrow.parquet as pq6from datasets import load_dataset7import glob8import re9import unicodedata10import datetime11 12SKIP_KEYWORDS = set([13 # Receipt metadata / math14 "summe","zwischensumme","gesamt","gesamtbetrag","zwischenbetrag",15 "brutto","netto","steuer","steuerbetrag","ust","mwst","mehrwertsteuer",16 "betrag","wechselgeld","rabatt","abzug","rundung","rabattsumme",17 "beleg","bon","kassenbon","kundenbeleg","händlerbeleg",18 "datum","uhr","uhrzeit","zeit","tag","monat","jahr","wochentag",19 "mo","di","mi","do","fr","sa","so","montag","dienstag","mittwoch",20 "donnerstag","freitag","samstag","sonntag",21 "kasse","kassierer","bed","bediener","mitarbeiter","filiale","markt",22 "markt-nr","marktnr","marktnummer","seriennummer","server","druck",23 "druckzeit","belegdruck","intern","vorgang","vorgangsnummer",24 "transaktion","transaktionsnummer","trace","pos","terminal","terminal-id",25 # Supermarkets / chains26 "rewe","edeka","aldi","lidl","netto","penny","kaufland","rossmann","dm",27 "real","metro","nahkauf","globus","tegut","marktkauf","nah und gut",28 "billa","spar","hofer","famila","toom","hit","müller","mueller",29 # Payment / finance30 "zahlung","kartenzahlung","barzahlung","kontaktlos","pin",31 "visa","mastercard","american express","amex","diners","maestro",32 "girocard","ec","ec-cash","debit","credit","kreditkarte",33 "auth","authorisierung","autorisierung","genehmigung",34 "autorisierung erfolgt","zahlung erfolgt","zahlung genehmigt",35 "referenz","reference","ref","receipt","paypal","klarna","apple pay","google pay",36 "postbank","sparkasse","commerzbank","volksbank","raiffeisenbank",37 # Legal / tax / invoice38 "ust-","ust","ustnr","ustid","uid","uidnr","umsatzsteuer",39 "steuer-nr","steuernummer","umsatzsteuer-id","rnr","rechnung",40 "kundennr","kundennummer","kunden-nr",41 # Contact & address42 "tel","telefon","fax","email","mail","straße","strasse","str.","weg","platz",43 "allee","ring","gasse","ufer","hof","chaussee","pl.","strs","chaus.",44 "berlin","hamburg","münchen","muenchen","frankfurt","dortmund",45 "köln","koeln","hannover","stuttgart","bremen","leipzig",46 "nürnberg","nuernberg","dresden","bochum","wuppertal",47 # Marketing / promo / generic info48 "produkt","produkte","sortiment","aktion","aktionsangebot","aktionspreis",49 "angebot","angebote","angebotspreis","sonderpreis","rabattaktion",50 "discount","preis","sparpreis","gilt","gültig","cashback","bonusprogramm",51 "bonus","coupon","coupons","einlösen","einlösbar","einlösungen",52 "einkaufswert","ersparnis","verkauf","rückgabe","abwicklung","kontakt",53 "werbung","homepage","website","webseite","www.","online","shop","shoppe",54 "besuche uns","besuch uns","folge uns","wir sind jetzt auf",55 "instagram","facebook","tiktok","twitter","youtube","payback","payback pay",56 "punkte","punktestand","punktezahl","kundenkarte","treuepunkte","gutschein",57 # Recruitment / jobs58 "möchtest","moechtest","bewirb","bewerbe","bewerbung","karriere","team",59 "arbeit","arbeiten","stelle","jobs","job","werde teil","teil unseres teams",60 "jetzt bewerben","join","jobportal","karriereportal",61 # Customer info / FAQ62 "sie haben fragen","fragen","antwort","antworten","hilfe",63 "kundenservice","kundendienst","reklamation","service",64 "bitte beachten sie","beachten sie","wir auch","erfahre mehr","mehr erfahren",65 # Additives / labeling / packaging66 "zusatz","zusatzstoff","zusatzstoffe","künstlich","kuenstlich","künstlichen",67 "farb","farbstoff","farbstoffen","konservierungsstoffe",68 "geschmacksverstärker","aroma","zutaten","inhaltstoffe","verbrauch",69 "verbraucher","aufbewahrung","rückseite","gebrauchsanweisung",70 "erzeugnis","produktinformationen","hinweis","hinweise","serviervorschlag",71 "informationen","auf basis","auf basis von",72 # Deposit / refund73 "pfand","pfand euro","pfand eur","mehrweg","einweg","pfandrückgabe","pfandrueckgabe",74 # Locations / malls / travel promos75 "einkaufszentrum","einkaufs-zentrum","center","city center","city-center",76 "shopping","passage","galerie","marktzentrum","marktcenter",77 "einkaufspark","aez","amper-einkaufs-zentrum","amper","reisen","rewe reisen",78 # Random / technical79 "hash","token","gerät","software","system","signatur","zähler",80 "vorgang abgeschlossen","verbindung","netzwerk",81 # Small units / quantity tokens not as products82 "stk","stck","stück","stueck","kg","g","ml","l","cl","packung","flasche",83 # Phrases you flagged84 "app joker","karte 0","ready to","to go"85])86 87# === CONFIG ===88TEXT_DIR = "./test.txt ( Data GVK)"# folder containing .txt JSON files89run_id = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")90TEMP_OUTPUT_FILE = "data/test_reformatted.parquet"91FINAL_OUTPUT_FILE = "data/test_aligned.parquet"92BATCH_SIZE = 100093REPO_ID = "CedRuiz/data_ClassificationModel"94DATA_DIR = "data"95PROCESSED_LOG = "processed_files.txt"96 97# === Define schema matching your model dataset ===98# schema = pa.schema([99# ("product_name_de", pa.string()),100# ("categories", pa.string())101# ])102# writer = pq.ParquetWriter(TEMP_OUTPUT_FILE, schema)103 104if os.path.exists(PROCESSED_LOG):105 with open(PROCESSED_LOG, "r", encoding="utf-8") as f:106 processed_files = set(f.read().splitlines())107else:108 processed_files = set()109 110def extract_products(full_text: str):111 """112 Return a list of product-like lines from OCR text.113 Aggressively filters out receipt metadata, addresses, payment, promos, recruitment,114 social, legal/tax, random tokens, and non-product phrases.115 """116 unique_products = set(buffer_entry["product_name_de"] for buffer_entry in buffer)117 WHITELIST_CUES = {118 "bio", "vegan", "vegetarisch", "glutenfrei", "laktosefrei",119 "light", "zero", "protein", "fit", "natural", "classic",120 "fresh", "pure", "organic", "soja", "hafer", "reis",121 "mandel", "oat", "coconut", "ohne", "zuckerfrei", "free",122 "superfood", "plantbased", "pflanzlich", "green", "eco"123 }124 125 # Normalize Unicode (ä→ä, etc.), keep original for output later126 text_norm = unicodedata.normalize("NFKC", full_text)127 lines = text_norm.split("\n")128 products = []129 130 131 # Precompiled regexes132 re_has_letters = re.compile(r"[A-Za-zÄÖÜäöüß]")133 re_price = re.compile(r"\b\d+[.,]\d{2}\b")134 re_price_trailer = re.compile(r"\b\d+[.,]\d{2}\s*[A-Z]?\b")135 re_pct = re.compile(r"\b\d{1,2}\s*%\b")136 137 # Phones / addresses138 re_phone = re.compile(r"(tel|telefon|fax)\s*[:\.]?\s*\d", re.IGNORECASE)139 re_phone_plain = re.compile(r"\b\d{3,4}[-/ ]?\d{3,}\b")140 re_postal = re.compile(r"\b\d{4,5}\b")141 re_street = re.compile(r"\b(str|straße|strasse|weg|platz|allee|ring|gasse|ufer|chaussee)\b", re.IGNORECASE)142 re_cityline = re.compile(143 r"\b(berlin|hamburg|m(ü|u)nchen|frankfurt|dortmund|k(ö|oe)ln|hannover|stuttgart|bremen|leipzig|n(ü|ue)rnberg|dresden|bochum|wuppertal)\b",144 re.IGNORECASE)145 146 # Payments/brands/online/social147 re_cardbrands = re.compile(148 r"(american\s*express|amex|paypal|klarna|apple\s*pay|google\s*pay|visa|mastercard|maestro|girocard|ec[- ]?cash)",149 re.IGNORECASE)150 re_payment_words = re.compile(r"(kartenzahlung|barzahlung|kontaktlos|kreditkarte|debit|credit)", re.IGNORECASE)151 re_url = re.compile(r"(https?://|www\.)", re.IGNORECASE)152 re_social = re.compile(r"(instagram|facebook|tiktok|twitter|youtube)", re.IGNORECASE)153 154 # Legal/tax/IDs/random155 re_taxid = re.compile(r"\b(de|at|ch)?\s*[-–]?\s*\d{8,12}\b", re.IGNORECASE)156 re_iban = re.compile(r"\b[A-Z]{2}\d{2}[A-Z0-9]{10,30}\b")157 re_random_token = re.compile(r"\b[a-zA-Z0-9]{10,}\b")158 159 # Recruitment / promo / mall160 re_recruit = re.compile(161 r"(bewerb|möchtest|moechtest|arbeiten|teil\s+unser(es|s)?\s+teams|wir\s+sind|jetzt\s+bewerben)", re.IGNORECASE)162 re_promo = re.compile(r"(discount|preis|angebot|rabatt|aktionspreis|preiswert|sparpreis)", re.IGNORECASE)163 re_mall = re.compile(r"(einkaufs.?zentrum|center|galerie|passage|shopping|aez|amper)", re.IGNORECASE)164 165 PHRASE_SKIP = [166 "auf basis von", "foto von", "app joker", "karte 0", "payback pay",167 "rewe reisen", "bitte beachten sie", "wir auch", "erfahre mehr", "mehr erfahren"168 ]169 # Many receipts are shouty; allow uppercase, but we still need a product-like shape.170 for raw in lines:171 original_line = raw.strip()172 if not original_line:173 continue174 175 # Quick reject: must contain a letter176 if not re_has_letters.search(original_line):177 continue178 179 low = original_line.lower()180 # Keyword skip (broad)181 if any(k in low for k in SKIP_KEYWORDS):182 if not any(w in low for w in WHITELIST_CUES):183 continue184 185 186 if any(phrase in low for phrase in PHRASE_SKIP):187 continue188 # Numeric-word artifacts like "1701 Whr", "App Joker 25"189 if re.search(r"\b\d+\s*[A-Za-zÄÖÜäöüß]{2,}\b", original_line):190 continue # numbers attached to text tokens (1701 Whr, App Joker)191 192 # Lines that start or end with a lone number or short token like "Karte 0"193 if re.match(r"^(karte|app|foto)\s*\d+\b", low):194 continue195 if re.search(r"\b\d+\s*(karte|app|foto)\b", low):196 continue197 198 # Phone / address / postal199 if re_phone.search(low) or re_phone_plain.search(original_line):200 continue201 if re_postal.search(original_line) or re_street.search(low) or re_cityline.search(low):202 continue203 204 # Payment, brands, URLs, social, recruitment, promo, mall205 if re_cardbrands.search(low) or re_payment_words.search(low):206 continue207 if re_url.search(low) or re_social.search(low):208 continue209 if re_recruit.search(low) or re_promo.search(low) or re_mall.search(low):210 continue211 212 # Legal/tax IDs / IBAN / random hashes213 if re_taxid.search(low) or re_iban.search(original_line) or re_random_token.search(original_line):214 continue215 216 # Strip obvious numeric/price trailers (don't reject the line; clean it)217 cleaned = re_price_trailer.sub("", original_line)218 cleaned = re_price.sub("", cleaned)219 cleaned = re_pct.sub("", cleaned)220 221 # Remove lingering non-alnum (keep spaces, dots, hyphens)222 cleaned = re.sub(r"[^A-Za-zÄÖÜäöüß0-9\s\-.]", "", cleaned).strip()223 224 # Heuristics: keep lines that look like short product names225 # - Between 2 and 6 tokens typically works well for receipts226 tokens = cleaned.split()227 if not (1 <= len(tokens) <= 9):228 continue229 230 # Require at least one uppercase (helps skip purely descriptive sentences)231 if not re.search(r"[A-ZÄÖÜ]", cleaned):232 continue233 234 # Avoid generic one-word leftovers like "Produkt", "Sortiment"235 if len("".join(tokens)) <= 3:236 continue237 238 # Final tiny blacklist on line shape239 if cleaned.lower() in {"produkt", "sortiment"}:240 continue241 242 # Deduplicate consecutive duplicate tokens (e.g., "VEG VEG SALAMI")243 dedup = []244 for t in tokens:245 if not dedup or t != dedup[-1]:246 dedup.append(t)247 cleaned = " ".join(dedup)248 249 # Guard against very long char length (often sentences)250 if len(cleaned) > 48:251 continue252 253 if any(w in cleaned.lower() for w in WHITELIST_CUES) and len(tokens) >= 2:254 products.append(cleaned)255 continue256 257 # Looks like a product258 products.append(cleaned)259 260 print(f"Unique product names extracted: {len(unique_products)}")261 return products262 263 264 265buffer = []266processed = 0267skipped = 0268new_files = []269 270# === Loop through all .txt files ===271for i, filename in enumerate(os.listdir(TEXT_DIR), start=1):272 if not filename.endswith(".txt") or filename in processed_files:273 continue274 275 file_path = os.path.join(TEXT_DIR, filename)276 try:277 with open(file_path, "r", encoding="utf-8") as f:278 data = json.load(f)279 full_text = data["responses"][0]["textAnnotations"][0]["description"]280 print(f"[{i}] {filename} — length of OCR text: {len(full_text)}")281 except Exception:282 print(f"Skipping {filename} (invalid JSON)")283 skipped += 1284 continue285 286 products = extract_products(full_text)287 for product in products:288 buffer.append({"product_name_de": product, "categories": None})289 processed += 1290 291 processed_files.add(filename)292 with open(PROCESSED_LOG, "a", encoding="utf-8") as f:293 f.write(filename + "\n")294 295 # Write in batches to save memory296 if i % BATCH_SIZE == 0:297 table = pa.Table.from_pylist(buffer)298 pq.write_table(table, f"data/temp_batch_{i}.parquet")299 buffer = []300 print(f"Processed {i} files, total {processed} products...")301 302# Final flush303# === Write new batch only if new files found ===304# Final flush305if buffer:306 307 run_id = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")308 309 batch_id = len([f for f in os.listdir(DATA_DIR) if f.startswith("parsed_batch_")]) + 1310 batch_file = os.path.join(DATA_DIR, f"parsed_batch_{batch_id}.parquet")311 312 table = pa.Table.from_pylist(buffer)313 pq.write_table(table, batch_file)314 315 print(f"Saved new batch file: {batch_file} ({len(buffer)} products)")316 print(f"File size: {os.path.getsize(batch_file)} bytes")317 318 if os.path.getsize(batch_file) < 100:319 raise ValueError(f"Parquet file {batch_file} seems too small — check data creation.")320 321 # Upload merged dataset322 dataset = load_dataset(323 "parquet",324 data_files={325 "train": "data/train-00000-of-00001.parquet",326 "test": batch_file327 }328 )329 330 dataset.push_to_hub(REPO_ID)331 print(f"Uploaded updated dataset with new batch file: {batch_file}")332 333 # Update processed log334 with open(PROCESSED_LOG, "w", encoding="utf-8") as f:335 f.write("\n".join(sorted(processed_files)))336else:337 print("No new .txt files found to process.")338 339 340print(f"Finished extracting OCR products.")341print(f"Total processed: {processed} entries, skipped: {skipped}")342 343 