{
  "acquisition": {
    "acquired_at": "2026-06-27T23:15:06+00:00",
    "dedupe_fingerprint": "0fe8708f4399152b661f2a9a287612eb2d18c0c0",
    "dropped_feature_columns_for_cap": [],
    "fetched_columns": 16,
    "fetched_rows": 690,
    "openml_data_id": 29,
    "openml_name": "credit-approval",
    "row_policy": "kept_all",
    "source_dataset_id": "29",
    "source_dataset_name": "credit-approval",
    "source_platform": "openml",
    "source_resource_id": null,
    "source_rows_after_target_cleanup": 690,
    "source_target_column_original": "A15",
    "source_url": "https://www.openml.org/d/29",
    "staged_feature_columns": 15,
    "staged_rows": 690,
    "staging_path": "/home/ubuntu/hyper/data/tabular/real_regression_sources/openml-credit-approval-29-target-a15.csv",
    "target_profile": {
      "column": "A15",
      "finite_rows": 690,
      "integer_like": true,
      "maximum": 100000.0,
      "minimum": 0.0,
      "missing_fraction": 0.0,
      "numeric_valid_fraction": 1.0,
      "unique_values": 240
    },
    "tool": "models.tools.acquire_real_regression_datasets"
  },
  "counts": {
    "dropped_feature_count": 0,
    "n_features": 15,
    "original_columns": 16,
    "original_rows": 690,
    "rows_after_target_cleanup": 690,
    "selected_rows": 690,
    "strata_count": 42,
    "test_rows": 115,
    "train_rows": 575
  },
  "created_at": "2026-06-27T23:15:06+00:00",
  "dataset_hash": "b5beff4a53fa12c0",
  "dedupe_fingerprint": "0fe8708f4399152b661f2a9a287612eb2d18c0c0",
  "dropped_feature_columns": [],
  "feature_columns": [
    "A1",
    "A2",
    "A3",
    "A4",
    "A5",
    "A6",
    "A7",
    "A8",
    "A9",
    "A10",
    "A11",
    "A12",
    "A13",
    "A14",
    "class"
  ],
  "feature_encodings": {
    "A1": {
      "impute_value": 1.0,
      "kind": "numeric"
    },
    "A10": {
      "impute_value": 0.0,
      "kind": "numeric"
    },
    "A11": {
      "impute_value": 0.0,
      "kind": "numeric"
    },
    "A12": {
      "impute_value": 0.0,
      "kind": "numeric"
    },
    "A13": {
      "impute_value": 0.0,
      "kind": "numeric"
    },
    "A14": {
      "impute_value": 160.0,
      "kind": "numeric"
    },
    "A2": {
      "impute_value": 28.46,
      "kind": "numeric"
    },
    "A3": {
      "impute_value": 2.75,
      "kind": "numeric"
    },
    "A4": {
      "impute_value": 1.0,
      "kind": "numeric"
    },
    "A5": {
      "impute_value": 0.0,
      "kind": "numeric"
    },
    "A6": {
      "impute_value": 6.0,
      "kind": "numeric"
    },
    "A7": {
      "impute_value": 7.0,
      "kind": "numeric"
    },
    "A8": {
      "impute_value": 1.0,
      "kind": "numeric"
    },
    "A9": {
      "impute_value": 1.0,
      "kind": "numeric"
    },
    "class": {
      "impute_value": 1.0,
      "kind": "numeric"
    }
  },
  "feature_scaling": {
    "constant_columns": [],
    "max": [
      1.0,
      80.25,
      28.0,
      2.0,
      2.0,
      13.0,
      8.0,
      28.5,
      1.0,
      1.0,
      67.0,
      1.0,
      2.0,
      2000.0,
      1.0
    ],
    "method": "minmax over cleaned source rows before sampling",
    "min": [
      0.0,
      13.75,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0
    ],
    "range": [
      0.0,
      1.0
    ]
  },
  "hash_algorithm": "sha256:first16",
  "headers_removed_from_csv": true,
  "n_features": 15,
  "origin": "real",
  "original_headers": [
    "A1",
    "A2",
    "A3",
    "A4",
    "A5",
    "A6",
    "A7",
    "A8",
    "A9",
    "A10",
    "A11",
    "A12",
    "A13",
    "A14",
    "class",
    "target"
  ],
  "output_headers": [
    "A1",
    "A2",
    "A3",
    "A4",
    "A5",
    "A6",
    "A7",
    "A8",
    "A9",
    "A10",
    "A11",
    "A12",
    "A13",
    "A14",
    "class",
    "target"
  ],
  "retrieved_at": "2026-06-27T23:15:06+00:00",
  "sampling": {
    "final_max_rows": 2000,
    "method": "quota stratified by target distribution plus high-variance feature quantile bins",
    "minimum_final_rows": 101,
    "minimum_test_to_train_ratio_when_small": 0.2,
    "seed": 3029063751,
    "source_read_max_rows": null,
    "test_target_rows": 1000,
    "train_target_rows": 1000
  },
  "sha256": "b5beff4a53fa12c03a4958b92d0757b9fc5031ec90e1bc5f9d51edfe4e832623",
  "source_dataset_id": "29",
  "source_dataset_name": "credit-approval",
  "source_family": "credit-approval",
  "source_file": "openml-credit-approval-29-target-a15.csv",
  "source_license": null,
  "source_modified_at": "2026-06-27T23:15:06+00:00",
  "source_path": "/home/ubuntu/hyper/data/tabular/real_regression_sources/openml-credit-approval-29-target-a15.csv",
  "source_platform": "openml",
  "source_real_data": true,
  "source_resource_id": null,
  "source_rows": 690,
  "source_size_bytes": 48564,
  "source_target_column_original": "A15",
  "source_url": "https://www.openml.org/d/29",
  "staged_rows": 690,
  "staging_path": "/home/ubuntu/hyper/data/tabular/real_regression_sources/openml-credit-approval-29-target-a15.csv",
  "target_column": "target",
  "target_generation": "original_numeric_column",
  "target_is_final_column": true,
  "target_names": [
    "target"
  ],
  "target_scaling": {
    "max": 100000.0,
    "min": 0.0,
    "mode": "regression",
    "range": [
      0.0,
      1.0
    ]
  },
  "target_source_encoding": {
    "kind": "numeric"
  },
  "task_type": "regression",
  "test_rows": 115,
  "train_rows": 575
}