Source code for practice_sdoml.dataset

"""
Data ingestion and preprocessing module with train/test partitioning support.
"""
from pathlib import Path
from typing import Tuple
import pandas as pd
import numpy as np
import torch
from torch.utils.data import Dataset, DataLoader
from sklearn.model_selection import train_test_split


[docs] class DiabetesDataset(Dataset): """ PyTorch Dataset for diabetes_risk.csv supporting stratified train/test splits. :param csv_path: Optional explicit path to the CSV file. :type csv_path: Path | str | None :param split: Target data partition ('train', 'test', or 'all'). :type split: str :param test_size: Fraction of samples allocated to the test subset. :type test_size: float :param random_state: Random seed for deterministic reproducibility. :type random_state: int """ def __init__( self, csv_path: Path | str | None = None, split: str = "train", test_size: float = 0.2, random_state: int = 42, ): if csv_path is None: # 1. Search inside installed package data directory (PyPI runtime) pkg_data_path = Path(__file__).resolve().parent / "data" / "diabetes_risk.csv" # 2. Search inside repository root data/raw directory (local runtime) repo_data_path = Path(__file__).resolve().parents[1] / "data" / "raw" / "diabetes_risk.csv" if pkg_data_path.exists(): csv_path = pkg_data_path elif repo_data_path.exists(): csv_path = repo_data_path else: csv_path = None if csv_path is not None and Path(csv_path).exists(): # Load real dataset from resolved file path df = pd.read_csv(csv_path) X_df = df.iloc[:, :-1] y_series = df.iloc[:, -1] else: # Defensive fallback: generate synthetic dataset to prevent crashes np.random.seed(random_state) synthetic_size = 600 feature_names = [f"Feature_{i}" for i in range(10)] X_df = pd.DataFrame( np.random.randn(synthetic_size, 10).astype(np.float32), columns=feature_names, ) y_series = pd.Series(np.random.choice([0, 1], size=synthetic_size)) # Apply one-hot encoding to categorical features X_df = pd.get_dummies(X_df, drop_first=True) X_all = X_df.values.astype(np.float32) # Convert categorical target labels to zero-indexed integer encodings y_cat = y_series.astype("category") y_all = y_cat.cat.codes.values.astype(np.int64) self.num_features = X_all.shape[1] self.num_classes = len(y_cat.cat.categories) # Perform stratified train/test split if requested if split in ["train", "test"]: X_train, X_test, y_train, y_test = train_test_split( X_all, y_all, test_size=test_size, random_state=random_state, stratify=y_all, ) if split == "train": self.X = torch.tensor(X_train, dtype=torch.float32) self.y = torch.tensor(y_train, dtype=torch.long) else: self.X = torch.tensor(X_test, dtype=torch.float32) self.y = torch.tensor(y_test, dtype=torch.long) else: self.X = torch.tensor(X_all, dtype=torch.float32) self.y = torch.tensor(y_all, dtype=torch.long) def __len__(self) -> int: """Returns the total number of samples in the active partition.""" return len(self.y) def __getitem__(self, idx: int) -> Tuple[torch.Tensor, torch.Tensor]: """ Retrieves feature and label tensors for a given sample index. :param idx: Sample index. :type idx: int :return: Tuple containing (features_tensor, label_tensor). :rtype: Tuple[torch.Tensor, torch.Tensor] """ return self.X[idx], self.y[idx]
[docs] def get_dataloader( batch_size: int = 32, shuffle: bool = True, split: str = "train", ) -> Tuple[DataLoader, int, int]: """ Factory helper function to instantiate DiabetesDataset and build a DataLoader. :param batch_size: Number of samples per batch. :type batch_size: int :param shuffle: Whether to shuffle samples at every epoch. :type shuffle: bool :param split: Partition identifier ('train', 'test', or 'all'). :type split: str :return: Tuple containing (DataLoader, num_features, num_classes). :rtype: Tuple[DataLoader, int, int] """ dataset = DiabetesDataset(split=split) loader = DataLoader(dataset, batch_size=batch_size, shuffle=shuffle) return loader, dataset.num_features, dataset.num_classes