Team Ai
Datasetpublic

SparseWake/sparsewake

SparseWake SparseWake is a synthetic benchmark for sparse temporal hydrodynamic sensing. ICLR 2027 release The expanded release adds controlled multi-source mixtures and common-prior nearest-source tasks, with complete core data banks, reference checkpoints, a small review supplement, and reproduction code with a frozen wake-library input. Download release iclr2027-v1.0rc2 The version page lists the three archives, exact sizes, checksums, extraction instructions… See the full description on the dataset page: https://huggingface.co/datasets/SparseWake/sparsewake.

sourceHugging Facecc-by-4.0updated 14d agoView on Hugging Face
0likes275downloads
features.py76 linesDownload Raw Back to sparsewake
1from __future__ import annotations2 3import numpy as np4 5 6def raw_features(x: np.ndarray, sensor_indices: list[int] | None = None) -> np.ndarray:7    if sensor_indices is not None:8        x = x[:, sensor_indices, :]9    return x.astype(np.float32)10 11 12def raw_norm_features(x: np.ndarray, sensor_indices: list[int] | None = None, eps: float = 1e-8) -> np.ndarray:13    x = raw_features(x, sensor_indices)14    scale = np.mean(np.linalg.norm(x[:, :, :2], axis=2), axis=1, keepdims=True)15    scale = np.maximum(scale, eps).astype(np.float32)16    norm = x / scale[:, None, :]17    return np.concatenate(18        [19            x.reshape(x.shape[0], -1),20            norm.reshape(norm.shape[0], -1),21            scale,22        ],23        axis=1,24    ).astype(np.float32)25 26 27def make_temporal_windows(28    per_sample_features: np.ndarray,29    phase_id: np.ndarray,30    pose_id: np.ndarray,31    history: int,32) -> tuple[np.ndarray, np.ndarray]:33    phase_values = np.sort(np.unique(phase_id))34    pose_values = np.sort(np.unique(pose_id))35    phase_to_i = {int(v): i for i, v in enumerate(phase_values)}36    pose_to_i = {int(v): i for i, v in enumerate(pose_values)}37    n_phase = len(phase_values)38    n_pose = len(pose_values)39    feature_shape = per_sample_features.shape[1:]40    grid = np.zeros((n_phase, n_pose) + feature_shape, dtype=np.float32)41    valid = np.zeros((n_phase, n_pose), dtype=bool)42    sample_index = np.full((n_phase, n_pose), -1, dtype=np.int64)43    for i, (phase, pose) in enumerate(zip(phase_id, pose_id)):44        pi = phase_to_i[int(phase)]45        qi = pose_to_i[int(pose)]46        grid[pi, qi] = per_sample_features[i]47        valid[pi, qi] = True48        sample_index[pi, qi] = i49 50    windows = []51    indices = []52    for pi in range(n_phase):53        start = max(0, pi - history + 1)54        hist = grid[start : pi + 1]55        if hist.shape[0] < history:56            pad = np.repeat(hist[:1], history - hist.shape[0], axis=0)57            hist = np.concatenate([pad, hist], axis=0)58        for qi in range(n_pose):59            if valid[pi, qi]:60                windows.append(hist[:, qi].reshape(-1))61                indices.append(sample_index[pi, qi])62    return np.asarray(windows, dtype=np.float32), np.asarray(indices, dtype=np.int64)63 64 65def build_design_matrix(data: dict[str, np.ndarray], feature_set: str = "raw_norm", history: int = 24) -> tuple[np.ndarray, np.ndarray]:66    if feature_set == "raw":67        per_sample = raw_features(data["input"])68    elif feature_set == "raw_norm":69        per_sample = raw_norm_features(data["input"])70    else:71        raise ValueError(f"Unsupported feature_set: {feature_set}")72    phase_id = np.asarray(data.get("groups")).reshape(-1)73    pose_id = np.asarray(data["pose_id"]).reshape(-1)74    x, idx = make_temporal_windows(per_sample, phase_id, pose_id, history)75    return x, idx76