Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 7 additions & 3 deletions benchmark/tabular/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -31,7 +31,9 @@ ______________________________________________________________________
- **`KumoTabular`:**

```bash
python -m benchmark.tabular.tabarena.main --model kumo-tabular
python -m benchmark.tabular.tabarena.main --model kumo-tabular-large
python -m benchmark.tabular.tabarena.main --model kumo-tabular-medium
python -m benchmark.tabular.tabarena.main --model kumo-tabular-small
```

- **`TabFM`:**
Expand All @@ -44,7 +46,7 @@ Pass a dataset name to run only that TabArena dataset:

```bash
python -m benchmark.tabular.tabarena.main \
--model kumo-tabular \
--model kumo-tabular-large \
--dataset blood-transfusion-service-center
```

Expand All @@ -71,7 +73,9 @@ ______________________________________________________________________
- **`KumoTabular`:**

```bash
python -m benchmark.tabular.beyondarena.main --model kumo-tabular
python -m benchmark.tabular.beyondarena.main --model kumo-tabular-large
python -m benchmark.tabular.beyondarena.main --model kumo-tabular-medium
python -m benchmark.tabular.beyondarena.main --model kumo-tabular-small
```

- **`TabFM`:**
Expand Down
64 changes: 53 additions & 11 deletions benchmark/tabular/model.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@
from sdm.processing.execution import RecipeExecution

Task = Literal["classification", "regression"]
KumoTabularSize = Literal["small", "medium", "large"]


class SDMModel(AbstractTorchModel, abc.ABC):
Expand Down Expand Up @@ -280,14 +281,22 @@ def _create_model(
return sdm.models.TabICLv2(task=task, device=device)


def _load_kumo_network(*, task: str, device: torch.device) -> torch.nn.Module:
return sdm.models.KumoTabular(task=task, device=device).models[task]
def _load_kumo_network(
*,
task: str,
size: KumoTabularSize,
device: torch.device,
) -> torch.nn.Module:
return sdm.models.KumoTabular(
task=task,
size=size,
device=device,
).models[task]


class SDMKumoTabularModel(SDMModel):
ag_key = "SDM-KUMO-TABULAR"
ag_name = "SDMKumoTabular"
default_num_estimators = 8
size: ClassVar[KumoTabularSize]
default_num_estimators = 16
autocast_dtype = torch.float16
# Bagged children are fit one at a time in this process, so they share the
# pretrained network of their task through AutoGluon's registry.
Expand All @@ -296,7 +305,7 @@ class SDMKumoTabularModel(SDMModel):
}
shared_weights: ClassVar[SharedWeights] = SharedWeights(
loader="benchmark.tabular.model:_load_kumo_network",
key=("task",),
key=("task", "size"),
)

@classmethod
Expand All @@ -311,17 +320,23 @@ def warmup(
) -> None:
warmup_torch(cuda=None if num_gpus is None else num_gpus > 0)

@staticmethod
@classmethod
def _create_model(
cls,
task: Task,
device: torch.device,
) -> sdm.models.KumoTabular:
model = sdm.models.KumoTabular(
task=task,
size=cls.size,
pretrained=False,
device="meta",
)
model.models[task] = _load_kumo_network(task=task, device=device)
model.models[task] = _load_kumo_network(
task=task,
size=cls.size,
device=device,
)
return model

# AutoGluon does not look inside the served model for the shared network,
Expand All @@ -346,6 +361,7 @@ def __setstate__(self, state: dict[str, Any]) -> None:
for task in served.models:
served.models[task] = _load_kumo_network(
task=task,
size=self.size,
device=self._device,
)

Expand All @@ -369,6 +385,24 @@ def _create_recipe(self) -> sdm.Recipe:
return recipe


class SDMKumoTabularSmallModel(SDMKumoTabularModel):
ag_key = "SDM-KUMO-TABULAR-SMALL"
ag_name = "SDMKumoTabularSmall"
size = "small"


class SDMKumoTabularMediumModel(SDMKumoTabularModel):
ag_key = "SDM-KUMO-TABULAR-MEDIUM"
ag_name = "SDMKumoTabularMedium"
size = "medium"


class SDMKumoTabularLargeModel(SDMKumoTabularModel):
ag_key = "SDM-KUMO-TABULAR-LARGE"
ag_name = "SDMKumoTabularLarge"
size = "large"


class SDMTabFMModel(SDMModel):
ag_key = "SDM-TABFM"
ag_name = "SDMTabFM"
Expand Down Expand Up @@ -406,9 +440,17 @@ def beyondarena_method_name(self) -> str:
name="TabICLv2",
model_cls=SDMTabICLv2Model,
),
"kumo-tabular": ModelConfig(
name="KumoTabular",
model_cls=SDMKumoTabularModel,
"kumo-tabular-small": ModelConfig(
name="KumoTabular-Small",
model_cls=SDMKumoTabularSmallModel,
),
"kumo-tabular-medium": ModelConfig(
name="KumoTabular-Medium",
model_cls=SDMKumoTabularMediumModel,
),
"kumo-tabular-large": ModelConfig(
name="KumoTabular-Large",
model_cls=SDMKumoTabularLargeModel,
),
"tabfm": ModelConfig(
name="TabFM",
Expand Down
10 changes: 9 additions & 1 deletion benchmark/tabular/scoringbench/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,15 @@ pip install -r ScoringBench/requirements.txt
```bash
python -m benchmark.tabular.scoringbench.main \
--scoringbench-path /path/to/ScoringBench \
--model kumo-tabular
--model kumo-tabular-large

python -m benchmark.tabular.scoringbench.main \
--scoringbench-path /path/to/ScoringBench \
--model kumo-tabular-medium

python -m benchmark.tabular.scoringbench.main \
--scoringbench-path /path/to/ScoringBench \
--model kumo-tabular-small
```

Pass `--dataset cpu_act` to run one dataset, `--dataset-index 0` to select by validated index, or `--lite` to use two folds.
Expand Down
9 changes: 7 additions & 2 deletions benchmark/tabular/scoringbench/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,12 @@
from typing import Any

BENCHMARK_DIR = Path(__file__).parent.parent
MODEL_NAMES = ("tabiclv2", "kumo-tabular")
MODEL_NAMES = (
"tabiclv2",
"kumo-tabular-small",
"kumo-tabular-medium",
"kumo-tabular-large",
)


def _parser() -> argparse.ArgumentParser:
Expand All @@ -25,7 +30,7 @@ def _parser() -> argparse.ArgumentParser:
parser.add_argument(
"--model",
choices=MODEL_NAMES,
default="kumo-tabular",
default="kumo-tabular-large",
)
selection = parser.add_mutually_exclusive_group()
selection.add_argument("--dataset")
Expand Down
50 changes: 39 additions & 11 deletions benchmark/tabular/scoringbench/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,8 @@
import abc
from collections.abc import Callable
from dataclasses import dataclass
from typing import ClassVar
from functools import partial
from typing import ClassVar, Literal

import numpy as np
import pandas as pd
Expand All @@ -35,8 +36,11 @@ def _create_tabiclv2(device: torch.device) -> sdm.models.TabICLv2:
return sdm.models.TabICLv2(task="regression", device=device)


def _create_kumo_tabular(device: torch.device) -> sdm.models.KumoTabular:
return sdm.models.KumoTabular(task="regression", device=device)
def _create_kumo_tabular(
device: torch.device,
size: Literal["small", "medium", "large"],
) -> sdm.models.KumoTabular:
return sdm.models.KumoTabular(task="regression", size=size, device=device)


MODEL_CONFIGS = {
Expand All @@ -47,12 +51,26 @@ def _create_kumo_tabular(device: torch.device) -> sdm.models.KumoTabular:
autocast_dtype=torch.float16,
num_estimators=8,
),
"kumo-tabular": ModelConfig(
name="KumoTabular",
method="sdm_kumo_tabular",
factory=_create_kumo_tabular,
"kumo-tabular-small": ModelConfig(
name="KumoTabular-Small",
method="sdm_kumo_tabular_small",
factory=partial(_create_kumo_tabular, size="small"),
autocast_dtype=torch.float16,
num_estimators=8,
num_estimators=16,
),
"kumo-tabular-medium": ModelConfig(
name="KumoTabular-Medium",
method="sdm_kumo_tabular_medium",
factory=partial(_create_kumo_tabular, size="medium"),
autocast_dtype=torch.float16,
num_estimators=16,
),
"kumo-tabular-large": ModelConfig(
name="KumoTabular-Large",
method="sdm_kumo_tabular_large",
factory=partial(_create_kumo_tabular, size="large"),
autocast_dtype=torch.float16,
num_estimators=16,
),
}

Expand Down Expand Up @@ -137,11 +155,21 @@ class SDMTabICLv2Wrapper(SDMQuantileWrapper):
config = MODEL_CONFIGS["tabiclv2"]


class SDMKumoTabularWrapper(SDMQuantileWrapper):
config = MODEL_CONFIGS["kumo-tabular"]
class SDMKumoTabularSmallWrapper(SDMQuantileWrapper):
config = MODEL_CONFIGS["kumo-tabular-small"]


class SDMKumoTabularMediumWrapper(SDMQuantileWrapper):
config = MODEL_CONFIGS["kumo-tabular-medium"]


class SDMKumoTabularLargeWrapper(SDMQuantileWrapper):
config = MODEL_CONFIGS["kumo-tabular-large"]


WRAPPERS: dict[str, type[SDMQuantileWrapper]] = {
"tabiclv2": SDMTabICLv2Wrapper,
"kumo-tabular": SDMKumoTabularWrapper,
"kumo-tabular-small": SDMKumoTabularSmallWrapper,
"kumo-tabular-medium": SDMKumoTabularMediumWrapper,
"kumo-tabular-large": SDMKumoTabularLargeWrapper,
}
6 changes: 4 additions & 2 deletions benchmark/tabular/talent/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,9 @@ Download and extract the datasets from the [official TALENT dataset page](https:
- **`KumoTabular`:**

```bash
python -m benchmark.tabular.talent.main --model kumo-tabular --dataset-path /path/to/talent/data
python -m benchmark.tabular.talent.main --model kumo-tabular-large --dataset-path /path/to/talent/data
python -m benchmark.tabular.talent.main --model kumo-tabular-medium --dataset-path /path/to/talent/data
python -m benchmark.tabular.talent.main --model kumo-tabular-small --dataset-path /path/to/talent/data
```

- **`TabFM`:**
Expand All @@ -41,7 +43,7 @@ Pass a dataset name to run only that TALENT dataset:

```bash
python -m benchmark.tabular.talent.main \
--model kumo-tabular \
--model kumo-tabular-large \
--dataset-path /path/to/talent/data \
--dataset Bank_Customer_Churn_Dataset
```
Expand Down
27 changes: 21 additions & 6 deletions benchmark/tabular/talent/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@
import time
from collections.abc import Callable, Sequence
from dataclasses import dataclass
from functools import lru_cache
from functools import lru_cache, partial
from typing import Any, Literal, cast

import numpy as np
Expand Down Expand Up @@ -59,8 +59,9 @@ def _create_tabiclv2(
def _create_kumo_tabular(
task: Task,
device: torch.device,
size: Literal["small", "medium", "large"],
) -> sdm.models.KumoTabular:
return sdm.models.KumoTabular(task=task, device=device)
return sdm.models.KumoTabular(task=task, size=size, device=device)


@lru_cache(maxsize=1)
Expand All @@ -82,10 +83,24 @@ def _create_tabfm(
num_estimators=8,
autocast_dtype=torch.float16,
),
"kumo-tabular": ModelConfig(
name="KumoTabular",
factory=_create_kumo_tabular,
num_estimators=8,
"kumo-tabular-small": ModelConfig(
name="KumoTabular-Small",
factory=partial(_create_kumo_tabular, size="small"),
num_estimators=16,
autocast_dtype=torch.float16,
max_classes=10,
),
"kumo-tabular-medium": ModelConfig(
name="KumoTabular-Medium",
factory=partial(_create_kumo_tabular, size="medium"),
num_estimators=16,
autocast_dtype=torch.float16,
max_classes=10,
),
"kumo-tabular-large": ModelConfig(
name="KumoTabular-Large",
factory=partial(_create_kumo_tabular, size="large"),
num_estimators=16,
autocast_dtype=torch.float16,
max_classes=10,
),
Expand Down
Loading