fabelouz/interactive-dashboard
1
1import numpy as np2import pandas as pd3from scipy.interpolate import CubicSpline4import random5 6DATA_AMOUNT = 100007WAVE_HEIGHT_NOISE = 0.3 # how much wave height and depth will differ from each other at random8DEFECT_STRENGTH = 2 # how strong a defect is (higher = bigger defect)9UNCERATAIN_STRENGTH = 1 # a small defect that can be hard to see because of WAVE_HEIGHT_NOISE - maybe not defect but acceptable deviation10DEFECT_RATIO = 0.05 # 0.2 = (5%), 0.5 = (50%), 1 = (100%)11POINTS_PER_SIGNAL = 2000 # number of points per wave12 13rows = []14 15for group_id in range(DATA_AMOUNT):16 # Define target heights17 signal = [base + random.uniform(-WAVE_HEIGHT_NOISE, WAVE_HEIGHT_NOISE) for base in [1, -1] * 20]18 label = True # default label19 status = "Normal" # default status20 21 # DEFECT22 if random.random() < DEFECT_RATIO:23 idx = random.randrange(len(signal))24 signal[idx] += DEFECT_STRENGTH * (1 if signal[idx] >= 0 else -1)25 label = False26 status = "Defect"27 if random.random() < DEFECT_RATIO: # 5% chance to be wrongly classified as normal28 label = True29 30 # UNCERTAIN (small defect)31 elif random.random() < DEFECT_RATIO:32 idx = random.randrange(len(signal))33 signal[idx] += UNCERATAIN_STRENGTH * (1 if signal[idx] >= 0 else -1)34 label = random.choice([False, True])35 status = "Uncertain"36 37 # MISSCLASSIFIED38 elif random.random() < DEFECT_RATIO:39 label = not label40 status = "Missclassified"41 42 # Scale amplitude43 scale = 0.244 signal = [h * scale for h in signal]45 46 # Setup spline interpolation47 num_points = len(signal)48 extrema_x = np.linspace(0, num_points - 1, num_points)49 cs = CubicSpline(extrema_x, signal, bc_type="natural")50 51 # Dense sampling52 x_dense = np.linspace(0, num_points - 1, POINTS_PER_SIGNAL)53 y_dense = cs(x_dense)54 55 # Add rows (group, x, y, label, status)56 for x, y in zip(x_dense, y_dense):57 rows.append((group_id, x, y, label, status))58 59# Build DataFrame60df = pd.DataFrame(rows, columns=["group", "col1", "col2", "label", "status"])61 62# Save to parquet63df.to_parquet("wave_dataset.parquet", index=False)64 65print("Saved parquet with shape:", df.shape)66 