{
  "acquisition": {
    "acquired_at": "2026-06-26T09:19:27+00:00",
    "fetched_feature_columns": 21,
    "fetched_rows": 5956,
    "openml_data_id": 47152,
    "openml_name": "aztrees4",
    "openml_url": "https://www.openml.org/d/47152",
    "row_policy": "stratified_cap_to_final_max",
    "source": "openml",
    "source_rows_after_target_cleanup": 5956,
    "stage_path": "/home/ubuntu/hyper/data/tabular/aztrees4.csv",
    "staged_feature_columns": 21,
    "staged_rows": 2000,
    "tool": "models.tools.acquire_real_binary_datasets"
  },
  "counts": {
    "dropped_feature_count": 0,
    "n_features": 21,
    "original_columns": 22,
    "original_rows": 2000,
    "rows_after_target_cleanup": 2000,
    "selected_rows": 2000,
    "strata_count": 39,
    "test_rows": 1000,
    "train_rows": 1000
  },
  "created_at": "2026-06-26T09:19:27+00:00",
  "dataset_hash": "bde2040d63cb1d9b",
  "dropped_feature_columns": [],
  "feature_columns": [
    "SAMPLE_1",
    "SAMPLE_2",
    "SAMPLE_3",
    "SAMPLE_4",
    "SAMPLE_5",
    "SAMPLE_6",
    "SAMPLE_7",
    "SAMPLE_8",
    "SAMPLE_9",
    "SAMPLE_10",
    "SAMPLE_11",
    "SAMPLE_12",
    "SAMPLE_13",
    "SAMPLE_14",
    "SAMPLE_15",
    "SAMPLE_16",
    "SAMPLE_17",
    "SAMPLE_18",
    "SAMPLE_19",
    "SAMPLE_20",
    "SAMPLE_21"
  ],
  "feature_encodings": {
    "SAMPLE_1": {
      "impute_value": 1471.0,
      "kind": "numeric"
    },
    "SAMPLE_10": {
      "impute_value": 2046.5,
      "kind": "numeric"
    },
    "SAMPLE_11": {
      "impute_value": 84.0,
      "kind": "numeric"
    },
    "SAMPLE_12": {
      "impute_value": 14.0,
      "kind": "numeric"
    },
    "SAMPLE_13": {
      "impute_value": -13.0,
      "kind": "numeric"
    },
    "SAMPLE_14": {
      "impute_value": 317.0,
      "kind": "numeric"
    },
    "SAMPLE_15": {
      "impute_value": 87.0,
      "kind": "numeric"
    },
    "SAMPLE_16": {
      "impute_value": 75.0,
      "kind": "numeric"
    },
    "SAMPLE_17": {
      "impute_value": 48.0,
      "kind": "numeric"
    },
    "SAMPLE_18": {
      "impute_value": 0.0,
      "kind": "numeric"
    },
    "SAMPLE_19": {
      "impute_value": -4.0,
      "kind": "numeric"
    },
    "SAMPLE_2": {
      "impute_value": 1662.0,
      "kind": "numeric"
    },
    "SAMPLE_20": {
      "impute_value": -26.5,
      "kind": "numeric"
    },
    "SAMPLE_21": {
      "impute_value": 146.0,
      "kind": "numeric"
    },
    "SAMPLE_3": {
      "impute_value": 1608.0,
      "kind": "numeric"
    },
    "SAMPLE_4": {
      "impute_value": 1902.0,
      "kind": "numeric"
    },
    "SAMPLE_5": {
      "impute_value": 2568.0,
      "kind": "numeric"
    },
    "SAMPLE_6": {
      "impute_value": 2760.0,
      "kind": "numeric"
    },
    "SAMPLE_7": {
      "impute_value": 2785.5,
      "kind": "numeric"
    },
    "SAMPLE_8": {
      "impute_value": 2930.5,
      "kind": "numeric"
    },
    "SAMPLE_9": {
      "impute_value": 2459.0,
      "kind": "numeric"
    }
  },
  "feature_scaling": {
    "constant_columns": [],
    "max": [
      8728.0,
      8784.0,
      8780.0,
      8621.0,
      8538.0,
      8688.0,
      8504.0,
      8448.0,
      8959.0,
      9621.0,
      682.0,
      404.0,
      446.0,
      32767.0,
      22760.0,
      32767.0,
      6302.0,
      1459.0,
      2196.0,
      2880.0,
      3831.0
    ],
    "method": "minmax over cleaned source rows before sampling",
    "min": [
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      0.0,
      -164.0,
      -355.0,
      -533.0,
      17.0,
      7.0,
      0.0,
      1.0,
      -1770.0,
      -2124.0,
      -2280.0,
      0.0
    ],
    "range": [
      0.0,
      1.0
    ]
  },
  "hash_algorithm": "sha256:first16",
  "headers_removed_from_csv": true,
  "n_features": 21,
  "origin": "real",
  "original_headers": [
    "SAMPLE_1",
    "SAMPLE_2",
    "SAMPLE_3",
    "SAMPLE_4",
    "SAMPLE_5",
    "SAMPLE_6",
    "SAMPLE_7",
    "SAMPLE_8",
    "SAMPLE_9",
    "SAMPLE_10",
    "SAMPLE_11",
    "SAMPLE_12",
    "SAMPLE_13",
    "SAMPLE_14",
    "SAMPLE_15",
    "SAMPLE_16",
    "SAMPLE_17",
    "SAMPLE_18",
    "SAMPLE_19",
    "SAMPLE_20",
    "SAMPLE_21",
    "target"
  ],
  "output_headers": [
    "SAMPLE_1",
    "SAMPLE_2",
    "SAMPLE_3",
    "SAMPLE_4",
    "SAMPLE_5",
    "SAMPLE_6",
    "SAMPLE_7",
    "SAMPLE_8",
    "SAMPLE_9",
    "SAMPLE_10",
    "SAMPLE_11",
    "SAMPLE_12",
    "SAMPLE_13",
    "SAMPLE_14",
    "SAMPLE_15",
    "SAMPLE_16",
    "SAMPLE_17",
    "SAMPLE_18",
    "SAMPLE_19",
    "SAMPLE_20",
    "SAMPLE_21",
    "target"
  ],
  "sampling": {
    "final_max_rows": 2000,
    "method": "quota stratified by target distribution plus high-variance feature quantile bins",
    "minimum_final_rows": 101,
    "minimum_test_to_train_ratio_when_small": 0.2,
    "seed": 3168215808,
    "source_read_max_rows": null,
    "test_target_rows": 1000,
    "train_target_rows": 1000
  },
  "sha256": "bde2040d63cb1d9b829e88762443a747481fa50ea02b56484a9d2d09633cfdc9",
  "source": "openml",
  "source_file": "aztrees4.csv",
  "source_modified_at": "2026-06-26T09:19:27+00:00",
  "source_openml_data_id": 47152,
  "source_openml_name": "aztrees4",
  "source_path": "/home/ubuntu/hyper/data/tabular/aztrees4.csv",
  "source_real_data": true,
  "source_size_bytes": 269769,
  "source_url": "https://www.openml.org/d/47152",
  "target_column": "target",
  "target_is_final_column": true,
  "target_names": [
    "target"
  ],
  "target_scaling": {
    "class_to_code": {
      "0.0": 0,
      "1.0": 1
    },
    "classes": [
      0.0,
      1.0
    ],
    "mode": "classification",
    "n_classes": 2,
    "range": [
      0.0,
      1.0
    ],
    "scaled_code_formula": "code / max(1, n_classes - 1)"
  },
  "target_source_encoding": {
    "kind": "numeric"
  },
  "task_type": "binary",
  "test_rows": 1000,
  "train_rows": 1000
}