diff --git a/README.md b/README.md index 2cd2685f..787dc791 100644 --- a/README.md +++ b/README.md @@ -112,29 +112,33 @@ When `run_cluster = BashSLURM` or `run_cluster = BashLSF`, config-driven runs wa Dry-run a config to inspect the resolved phase calls: ```bash -python run.py -c run_configs/uci_binary_hcc.cfg --dry_run +python run.py -c run_configs/local/uci_binary_hcc.cfg --dry_run ``` Run the configured pipeline: ```bash -python run.py -c run_configs/uci_binary_hcc.cfg +python run.py -c run_configs/local/uci_binary_hcc.cfg ``` Useful controls: ```bash -python run.py -c run_configs/uci_binary_hcc.cfg --start_at p4 -python run.py -c run_configs/uci_binary_hcc.cfg --stop_after p8 -python run.py -c run_configs/uci_binary_hcc.cfg --only p6,p8,p11 -python run.py -c run_configs/uci_binary_hcc.cfg --skip p3,p4 +python run.py -c run_configs/local/uci_binary_hcc.cfg --start_at p4 +python run.py -c run_configs/local/uci_binary_hcc.cfg --stop_after p8 +python run.py -c run_configs/local/uci_binary_hcc.cfg --only p6,p8,p11 +python run.py -c run_configs/local/uci_binary_hcc.cfg --skip p3,p4 ``` -Example configs are included for the three UCI demos: +Example configs are included for the three UCI demos and are organized by run environment: -- `run_configs/uci_binary_hcc.cfg` -- `run_configs/uci_multiclass_student.cfg` -- `run_configs/uci_regression_auto_mpg.cfg` +- `run_configs/local/uci_binary_hcc.cfg` +- `run_configs/local/uci_multiclass_student.cfg` +- `run_configs/local/uci_regression_auto_mpg.cfg` +- `run_configs/hpc/cedars_slurm_hcc.cfg` +- `run_configs/hpc/upenn_lsf_hcc.cfg` + +The original top-level demo configs are still kept for backward compatibility. Phase 10 runs only when replication paths are configured, unless it is explicitly enabled. Phase 7 is automatically skipped for continuous/regression runs because the current ensemble registry is classification-only. diff --git a/docs/source/changelog.md b/docs/source/changelog.md index 1da5ea88..4f1fa582 100644 --- a/docs/source/changelog.md +++ b/docs/source/changelog.md @@ -5,6 +5,36 @@ Older public release notes are based on the [GitHub Releases](https://github.com/UrbsLab/STREAMLINE/releases) entries, with minor wording cleanup for readability. +## v1.0.1 - Bug Fix Release + +STREAMLINE v1.0.1 is a focused bug-fix release for regression reporting, +regression EDA robustness, and multiclass weighted feature-importance +visualization. + +### Fixed + +* Fixed Phase 1 regression EDA so skewness and kurtosis calculations do not + crash on mixed-type, missing, infinite, or nonnumeric continuous outcome + values. +* Fixed Phase 11 report task detection so explicit saved or CLI `outcome_type` + values are respected before falling back to dataset inference. +* Fixed regression reports for low-cardinality continuous outcomes that could + otherwise be misidentified as multiclass based on `ClassCounts.csv`. +* Fixed regression report performance tables so both display-style metric names + and snake_case metric keys are recognized. +* Fixed multiclass weighted composite feature-importance plots so the + balanced-accuracy no-skill baseline is `1 / number_of_classes` rather than + always `0.5`. + +### Added + +* Added an HPC and cluster-running documentation page with Conda setup notes, + `tmux` workflow basics, SLURM/LSF monitoring commands, scheduler config + explanations, and recovery/rerun guidance. +* Added a UPenn/LSF HCC demo config template alongside the existing + Cedars/SLURM template and updated README, installation, running, and + parameter documentation to point users to both HPC examples. + ## v1.0.0 - Main Release STREAMLINE v1.0.0 is a major reorganization and expansion of STREAMLINE into a diff --git a/docs/source/conf.py b/docs/source/conf.py index 00d124e6..8cf95e28 100644 --- a/docs/source/conf.py +++ b/docs/source/conf.py @@ -9,7 +9,7 @@ project = "STREAMLINE" copyright = "2026, Ryan Urbanowicz, Harsh Bandhey" author = "Ryan Urbanowicz, Harsh Bandhey" -release = "1.0.0" +release = "1.0.1" extensions = [ "sphinx.ext.autodoc", diff --git a/docs/source/development.md b/docs/source/development.md index f4ab825b..fe791fbd 100644 --- a/docs/source/development.md +++ b/docs/source/development.md @@ -43,7 +43,7 @@ Prefer the existing registry patterns when adding new components: ## Tests The default pytest configuration collects only the current main end-to-end -tests. Legacy and phase-level subtests were removed from the maintained v1.0.0 +tests. Legacy and phase-level subtests were removed from the maintained v1.0.1 test path so routine testing stays focused on the binary, multiclass, and regression demo pipelines. diff --git a/docs/source/hpc.md b/docs/source/hpc.md new file mode 100644 index 00000000..e8cbe169 --- /dev/null +++ b/docs/source/hpc.md @@ -0,0 +1,215 @@ +# HPC and Cluster Runs + +STREAMLINE can run small examples on a laptop, but paper-scale runs often need a +cluster. The cluster path is still the same pipeline: edit a `.cfg`, dry-run it, +then launch the config runner. The difference is that selected phases submit +many scheduler jobs through SLURM or LSF and the config runner waits for those +jobs to finish before moving to the next phase. + +## When To Use Each Execution Mode + +| Mode | Best use | +| --- | --- | +| `Serial` | Debugging, small demos, and first config checks. | +| `Parallel` | A single machine or one allocated compute node using joblib multiprocessing. | +| `Local` | A local Dask cluster on one machine. | +| `BashSLURM` | HPC systems that submit jobs with `sbatch`. | +| `BashLSF` | HPC systems that submit jobs with `bsub`. | +| Named Dask cluster | Site-specific Dask jobqueue execution when configured by the user/site. | + +Use a scheduler mode for long P4/P6/P8/P10/P11-style workloads or any analysis +that would be inappropriate to run directly on a login node. Use `Parallel` only +inside an interactive allocation or on a machine where it is acceptable to use +multiple local cores. + +## Included HPC Config Templates + +HPC configs live in `run_configs/hpc/`. + +| Config | Scheduler | Intended starting point | +| --- | --- | --- | +| `run_configs/hpc/cedars_slurm_hcc.cfg` | SLURM | Cedars/Sinai-style SLURM clusters using `run_cluster = BashSLURM`. | +| `run_configs/hpc/upenn_lsf_hcc.cfg` | LSF | UPenn/I2C2-style LSF clusters using `run_cluster = BashLSF`. | + +Both templates run the HCC binary demo by default. Copy one of them before using +it for a real project and edit at least `output_path`, `experiment_name`, +`data_path`, `queue`, `reserved_memory`, model list, and modeling budget. + +## Basic Cluster Setup + +From a login node: + +```bash +ssh @ +git clone --single-branch https://github.com/UrbsLab/STREAMLINE.git +cd STREAMLINE +conda create -n streamline python=3.11 pip +conda activate streamline +pip install -r requirements.txt +python run.py --help +``` + +Many clusters require modules before Conda, Python, or compiled libraries are +available. If your site uses modules, load the same modules before installation +and before running STREAMLINE jobs. Also make sure the repository, data, and +`output_path` are on a filesystem visible to compute nodes. + +## Conda Installation Quickstart + +If Conda is already available on the cluster, either directly or through a +module, create a dedicated STREAMLINE environment from the repository root: + +```bash +module load anaconda # omit or change this if your cluster uses a different module name +conda create -n streamline python=3.11 pip +conda activate streamline +pip install -r requirements.txt +``` + +If Conda is not available, install Miniconda in your home or project space using +your cluster's approved download method: + +```bash +mkdir -p ~/miniconda3 +curl -L https://repo.anaconda.com/miniconda/Miniconda3-latest-Linux-x86_64.sh -o /tmp/miniconda.sh +bash /tmp/miniconda.sh -b -p ~/miniconda3 +source ~/miniconda3/etc/profile.d/conda.sh +conda create -n streamline python=3.11 pip +conda activate streamline +pip install -r requirements.txt +``` + +Some HPC systems block outbound internet from compute nodes. In that case, +install packages from the login node, a site Conda mirror, or an administrator +provided module/wheelhouse, then run STREAMLINE from the same environment. + +## Use tmux For Long Runs + +The config runner is the phase orchestrator. Scheduler jobs can keep running if +your SSH connection drops, but the runner may stop waiting and the next phases +may not launch. Use `tmux` or `screen` for long runs. + +```bash +tmux new -s streamline +conda activate streamline +python run.py -c run_configs/hpc/cedars_slurm_hcc.cfg --dry_run +python run.py -c run_configs/hpc/cedars_slurm_hcc.cfg +``` + +Useful `tmux` commands: + +```bash +# Detach from the session without stopping STREAMLINE: +Ctrl-b, then d + +# List sessions: +tmux ls + +# Reattach later: +tmux attach -t streamline + +# Kill the session after the run is done: +tmux kill-session -t streamline +``` + +The same pattern works for the UPenn LSF template: + +```bash +tmux new -s streamline +conda activate streamline +python run.py -c run_configs/hpc/upenn_lsf_hcc.cfg --dry_run +python run.py -c run_configs/hpc/upenn_lsf_hcc.cfg +``` + +## Scheduler Settings In Configs + +The core cluster settings live in the `[run]` section: + +```ini +run_cluster = BashSLURM +wait_for_cluster_completion = True +cluster_phase_timeout = 86400 +cluster_phase_poll_interval = 30 +queue = defq +reserved_memory = 4 +``` + +For UPenn/LSF, the same fields look like: + +```ini +run_cluster = BashLSF +queue = i2c2_normal +reserved_memory = 4 +``` + +`queue` maps to the scheduler queue or partition. `reserved_memory` is the memory +request in GB used when STREAMLINE writes scheduler scripts. The exact queue +names and memory limits are site-specific, so treat the included values as +starting points. + +`wait_for_cluster_completion = True` tells the config runner to wait for +STREAMLINE completion markers in `jobsCompleted/` before it starts the next +phase. This is important because later phases depend on files written by earlier +scheduler jobs. + +## Monitoring Jobs + +STREAMLINE writes scheduler scripts to the experiment `jobs/` folder and +stdout/stderr files to `logs/`. + +Common SLURM commands: + +```bash +squeue -u $USER +sacct -j +scancel +``` + +Common LSF commands: + +```bash +bjobs +bjobs -l +bkill +``` + +If a phase appears stuck, check the scheduler first, then inspect +`//logs/` and the `jobsCompleted/` markers. + +## Recovery And Reruns + +Use a dry run before every large launch: + +```bash +python run.py -c run_configs/hpc/cedars_slurm_hcc.cfg --dry_run +``` + +If one phase fails, restart from that phase instead of repeating the full run: + +```bash +python run.py -c run_configs/hpc/cedars_slurm_hcc.cfg --start_at p6 +python run.py -c run_configs/hpc/cedars_slurm_hcc.cfg --only p8,p11 +``` + +Phase 6 reruns and overwrites requested model jobs by default. For recovery, +set `skip_completed_models = True` in `[p6]` or pass +`--skip_completed_models 1` to the P6 CLI. That runs missing or failed model/CV +jobs while leaving completed model jobs in place. + +If the config runner times out while scheduler jobs are still queued or running, +increase `cluster_phase_timeout` and rerun from the interrupted phase after +checking the logs. + +## Practical HPC Checklist + +Before a paper-scale cluster run: + +* Confirm the config with `--dry_run`. +* Use absolute paths for project data and outputs when running outside the repo. +* Keep `output_path` on shared storage visible to login and compute nodes. +* Start from small `models`, `n_trials`, `timeout`, and `n_splits` values. +* Use `tmux` or `screen` for any run that may outlive an SSH session. +* Confirm the Conda environment is available on compute nodes. +* Check `logs/` and `jobsCompleted/` before restarting a failed phase. +* Use `skip_completed_models = True` only for Phase 6 recovery runs where you do + not want to overwrite completed model artifacts. diff --git a/docs/source/index.rst b/docs/source/index.rst index 515e3b47..0852bbb2 100644 --- a/docs/source/index.rst +++ b/docs/source/index.rst @@ -7,16 +7,16 @@ Overview -------------------------------------- STREAMLINE is an end-to-end automated machine learning pipeline for -supervised tabular data. The v1.0.0 main release supports binary classification, +supervised tabular data. The v1.0.1 release supports binary classification, multiclass classification, and regression, with integrated data processing, imputation, scaling, feature learning, feature importance, feature selection, model training, classification ensembles, summary statistics, dataset comparison, replication, and PDF reporting. -The schematic below summarizes the STREAMLINE v1.0.0 workflow. +The schematic below summarizes the STREAMLINE v1.0.1 workflow. .. image:: pictures/STREAMLINE_v3_paper_new_lightcolor.png - :alt: STREAMLINE v1.0.0 automated machine learning pipeline overview + :alt: STREAMLINE v1.0.1 automated machine learning pipeline overview :width: 100% The repository is organized around eleven explicit phases: @@ -80,8 +80,8 @@ For most users, the easiest local route is: conda create -n streamline python=3.11 pip conda activate streamline pip install -r requirements.txt - python run.py -c run_configs/uci_binary_hcc.cfg --dry_run - python run.py -c run_configs/uci_binary_hcc.cfg + python run.py -c run_configs/local/uci_binary_hcc.cfg --dry_run + python run.py -c run_configs/local/uci_binary_hcc.cfg The notebooks expose the same major settings as the config files and are a better starting point for interactive tutorials, Colab demos, and custom data @@ -93,6 +93,7 @@ How This Documentation Is Organized * Use :doc:`install` to prepare a local environment. * Use :doc:`data` to format custom datasets and understand the included UCI demos. * Use :doc:`running` for notebooks, config-driven runs, and phase-by-phase CLI commands. +* Use :doc:`hpc` for SLURM/LSF configs, tmux basics, scheduler monitoring, and cluster recovery. * Use :doc:`parameters` when editing ``.cfg`` files or command-line calls. * Use :doc:`output` to navigate experiment folders and reports. * Use :doc:`pipeline` for a phase-by-phase explanation of what STREAMLINE does. @@ -101,7 +102,7 @@ How This Documentation Is Organized Version History -------------------------------------- -This site documents the STREAMLINE v1.0.0 main release. See +This site documents the STREAMLINE v1.0.1 release. See :doc:`changelog` for dated release entries and notable changes. Current Scope @@ -143,6 +144,7 @@ questions, contact Harsh Bandhey at ``harsh.bandhey@cshs.org``. install tabpfn_token running + hpc parameters model_params_json output diff --git a/docs/source/install.md b/docs/source/install.md index b566fbaf..b6c7cbf8 100644 --- a/docs/source/install.md +++ b/docs/source/install.md @@ -92,6 +92,16 @@ depending on phase support. For long runs, use a persistent terminal session such as `tmux` or `screen` so orchestration is not interrupted if your SSH connection drops. +Use the scheduler templates in `run_configs/hpc/` as starting points: + +```bash +python run.py -c run_configs/hpc/cedars_slurm_hcc.cfg --dry_run +python run.py -c run_configs/hpc/upenn_lsf_hcc.cfg --dry_run +``` + +See [HPC and Cluster Runs](hpc.md) for tmux basics, SLURM/LSF monitoring +commands, scheduler config fields, and recovery notes. + ## Known Installation Issues Some modeling and reporting packages include compiled dependencies. On macOS, diff --git a/docs/source/parameters.md b/docs/source/parameters.md index a98f0970..0adbdeb8 100644 --- a/docs/source/parameters.md +++ b/docs/source/parameters.md @@ -1,143 +1,388 @@ # Run Parameters -STREAMLINE parameters can be supplied through notebooks, `.cfg` files, or -phase CLI flags. The `.cfg` names intentionally match the command-line names -where possible. +STREAMLINE parameters can be supplied through `.cfg` files, notebooks, or phase +CLI flags. The recommended full-pipeline path is the config runner: -## Shared Run Parameters +```bash +python run.py -c run_configs/local/uci_binary_hcc.cfg --dry_run +python run.py -c run_configs/local/uci_binary_hcc.cfg +``` + +The `.cfg` parameter names intentionally match the command-line names wherever +possible. This page has two layers: a short set of essential parameters that +most users need to set correctly, followed by a more exhaustive phase-by-phase +reference. Essential parameters are repeated in the exhaustive reference so the +reader can either skim from the top or look up a single phase later. + +Use these terms consistently when reading the tables: + +* **Task type** means the supervised learning problem: `Binary`, `Multiclass`, + or `Continuous`. This is controlled by `outcome_type`. +* **Execution mode** means where jobs run: `Serial`, `Parallel`, `Local`, + `BashSLURM`, `BashLSF`, or a named Dask cluster. This is controlled by + `run_cluster`. +* **Run path** means how STREAMLINE is launched: a notebook, the full config + runner, or an individual phase CLI. +* **Report mode** means whether P11 creates a standard training/CV report or a + replication report. + +Later phases can load values saved by earlier phases in `metadata.pickle` and +`run_commands.pickle`. When a default below says it comes from metadata, the +owning phase is named where possible. For example, outcome labels, feature +types, CV counts, and one-hot settings are saved by P1; imputation, scaling, +and SMOTE settings are saved by P2; feature-learning settings are saved by P3; +feature-importance settings are saved by P4; modeling settings are saved by P6. +Explicit `.cfg` or CLI values override remembered values. + +## Essential Parameters To Run STREAMLINE -| Parameter | Typical value | Used by | Description | +These are the parameters most users should understand before starting a run. +Start from one of the included files in `run_configs/local/` or +`run_configs/hpc/`, change these values, and use `--dry_run` to inspect the +resolved phase calls before launching a full analysis. + +### Every Config Run + +These parameters define the run itself. They belong in `[run]` for config files, +or must be repeated on each individual phase CLI when running phases manually. + +| Parameter | Default value | Where to set it | Description | | --- | --- | --- | --- | -| `output_path` | `out` | all phases | Parent folder for experiment outputs. | -| `experiment_name` | `UCIHCCPipeline` | all phases | Experiment folder name. | -| `outcome_label` | `Class`, `MPG` | P1, P6, P8, P9, P11 | Outcome column. | -| `outcome_type` | `Binary`, `Multiclass`, `Continuous` | P1, P6, P8, P9, P11 | Learning task type. | -| `instance_label` | `InstanceID` | P1, P6-P11 | Optional row identifier column. | -| `n_splits` | `3`, `5`, `10` | CV-aware phases | Number of CV folds. | -| `run_cluster` | `Serial`, `Local`, `Parallel`, `BashSLURM`, `BashLSF` | all phases | Execution mode. `Local` uses a local Dask cluster; `Parallel` uses local joblib parallelism. | -| `wait_for_cluster_completion` | `True` | config runner | For `BashSLURM`/`BashLSF` full-pipeline runs, wait for STREAMLINE completion markers before starting the next phase. | -| `cluster_phase_timeout` | `86400` | config runner | Maximum seconds to wait for a scheduler-submitted phase. | -| `cluster_phase_poll_interval` | `30` | config runner | Seconds between completion-marker checks. | -| `random_state` | `42` | stochastic phases | Seed for reproducibility. | - -## Phase Toggles - -The `[phases]` section controls which phases run: - -```ini -[phases] -phase_order = p1,p2,p3,p4,p5,p6,p7,p8,p9,p10,p11 -do_p1 = True -do_p2 = True -do_p3 = True -do_p4 = True -do_p5 = True -do_p6 = True -do_p7 = True -do_p8 = True -do_p9 = True -do_p10 = True -do_p11 = True -``` +| `output_path` | Required | `[run]`; every phase CLI | Parent folder where STREAMLINE writes the experiment output. | +| `experiment_name` | Required | `[run]`; every phase CLI | Name of the experiment folder created under `output_path`. | +| `outcome_label` | `Class` | `[run]`, `[p1]`; repeated by later phase CLIs when needed | Outcome column in the input data. P1 records it for later phases. | +| `outcome_type` | P1 can infer; later phases should be explicit | `[run]`, `[p1]`, `[p6]`, `[p8]`, `[p9]`, `[p11]` | Task type: `Binary`, `Multiclass`, or `Continuous`. Set this explicitly for paper or benchmark runs. | +| `instance_label` | `None` | `[run]`, `[p1]`; repeated by later phase CLIs when needed | Optional row identifier column. P1 records it and excludes it from modeling. | +| `n_splits` | `10` | `[run]`; CV-aware phase CLIs | Number of cross-validation folds. Demo configs use `3` for speed. | +| `run_cluster` | `Serial` | `[run]`; every phase CLI | Execution mode. Use `Parallel` for local joblib multiprocessing, `Local` for local Dask, or `BashSLURM`/`BashLSF` for scheduler submission. | +| `random_state` | Phase-specific, often `None` or `0` | `[run]`; stochastic phase CLIs | Seed for reproducible CV partitioning, imputation/SMOTE, feature learning, modeling, and ensembles. | +| `phase_order` | `p1,p2,p3,p4,p5,p6,p7,p8,p9,p10,p11` | `[phases]` | Ordered list of phases for the config runner. | +| `do_p1` through `do_p11` | `True` unless disabled | `[phases]` | Phase toggles. P10 also requires replication paths; P7 is skipped for continuous outcomes. | + +### Phase-Specific Essentials + +These parameters decide what data are analyzed, how features are handled, which +models run, and which reports are produced. They are grouped by the phase that +first uses or remembers them. -The runner also accepts old-style broad flags such as `do_till_report`. +| Parameter | Default value | Where to set it | Description | +| --- | --- | --- | --- | +| `data_path` | Required for P1 | `[p1]` | Folder containing one or more input `.csv`, `.tsv`, or `.txt` datasets. | +| `categorical_features` | `None` | `[p1]` | Optional file listing categorical feature names. Recommended when feature types matter. | +| `quantitative_features` | `None` | `[p1]` | Optional file listing quantitative feature names. Recommended with `categorical_features`. | +| `ignore_features` | `None` | `[p1]` | Optional file or list of feature names to exclude before modeling. | +| `partition_method` | `Stratified` | `[p1]` | CV strategy. Continuous outcomes are forced to `Random`. | +| `one_hot_encoding` | `True` | `[p1]` | Expand non-binary categorical features in P1. If `False`, P6 only allows native-categorical models unless that guard is disabled. | +| `scale_data` | P2 remembered value, fallback `True` | `[p2]`, `[p8]` | Applies scaling in P2 and records whether scaled data was used in summary/reporting. | +| `impute_data` | P2 remembered value, fallback `True` | `[p2]` | Enables missing-value imputation for CV train/test folds. | +| `smote` | P2 remembered value, fallback `False` | `[p2]` | Enables classification-only training-fold oversampling after imputation and scaling. | +| `models` | All available non-excluded P6 models | `[p6]` | Model IDs to train, such as `NB,LR,DT,RF,CGB,HEROS,ExSTraCS`. Demo configs use small model lists for speed. | +| `model_params_json` | `None` | `[p6]` | Optional JSON or Python-literal dictionary of model-specific overrides. See [Model Parameter JSON](model_params_json.md). | +| `scoring_metric` | `balanced_accuracy` | `[p6]`, `[p8]` | Primary modeling/evaluation metric. Use `explained_variance` for regression unless intentionally changing the regression metric. | +| `metric_direction` | `maximize` | `[p6]` | Optuna optimization direction. Use `minimize` only for loss/error metrics where lower is better. | +| `n_trials` | `200` | `[p6]` | Maximum Optuna trials per model/CV job. | +| `timeout` | `900` | `[p6]` | Maximum Optuna time budget in seconds per model/CV job. | +| `training_subsample` | `0` | `[p6]` | Optional cap on training rows for models that explicitly allow subsampling. `0` disables it. | +| `skip_completed_models` | `False` | `[p6]` | When `True`, P6 runs only missing or failed model/CV jobs. Default behavior reruns requested model jobs and overwrites artifacts. | +| `rep_data_path` | Required for P10 | `[p10]` | Folder containing replication/external-validation datasets. | +| `dataset_for_rep` | Required for P10 | `[p10]` | Original training dataset path used to identify the trained dataset output folder. | +| `report_modes` | `standard` for P11; demo configs use `standard,replication` | `[p11]` | Report types generated by the config runner. | + +CLI-only controls are prefixed with `--` in the exhaustive reference. They are +not written into `.cfg` files. Use them to choose a config file, dry-run a +config, run only part of the phase order, list registry methods, or control +saved run-command reuse. + +### Config Template Folders + +The `run_configs/` directory is organized by execution environment: + +| Folder | Use case | Included examples | +| --- | --- | --- | +| `run_configs/local/` | Local serial, local joblib `Parallel`, and local Dask `Local` runs. | Binary HCC, multiclass student dropout, and regression Auto MPG demo configs. | +| `run_configs/hpc/` | Scheduler-oriented templates that use `BashSLURM`, `BashLSF`, or site-specific cluster settings. | `cedars_slurm_hcc.cfg` for Cedars/SLURM and `upenn_lsf_hcc.cfg` for UPenn/LSF starting points. | -## P1 Data Process +The original top-level demo configs are still kept for backward compatibility, +but new examples should point users to the environment-specific subfolders. -| Parameter | Default or example | Description | +For cluster-specific workflow details, including `tmux`, SLURM/LSF monitoring, +and recovery after interrupted runs, see [HPC and Cluster Runs](hpc.md). + +## Full Parameter Reference By Phase + +Defaults below are the current runner defaults when a parameter is omitted. +Where noted, later phases may load the value from experiment metadata saved by +earlier phases. + +### Shared Run Parameters + +| Parameter | Default value | Description | | --- | --- | --- | -| `data_path` | `data/UCIBinaryClassification` | Folder containing one or more input datasets. | -| `categorical_features` | `data/UCIFeatureTypes/hcc_survival_categorical_features.csv` | Optional feature-name file. | -| `quantitative_features` | `data/UCIFeatureTypes/hcc_survival_quantitative_features.csv` | Optional feature-name file. | -| `ignore_features` | empty | Optional feature-name file/list to drop. | -| `partition_method` | `Stratified` or `Random` | CV partitioning strategy. | -| `categorical_cutoff` | `10` | Inference threshold when feature type files are absent. | -| `one_hot_encoding` | `True` | Expand categorical features in P1. | -| `force` | `False` | Overwrite existing phase outputs. | +| `output_path` | Required | Parent folder for experiment outputs. | +| `experiment_name` | Required | Experiment folder name under `output_path`. | +| `outcome_label` | `Class` | Outcome column. Passed to P1 and reused by modeling, evaluation, comparison, replication, and reporting. | +| `outcome_type` | `None` in P1; later phases use metadata when possible | Learning task type: `Binary`, `Multiclass`, or `Continuous`. | +| `instance_label` | `None` | Optional row identifier column. | +| `n_splits` | `10` | Number of CV folds for CV-aware phases. | +| `run_cluster` | `Serial` | Execution mode. `Parallel` uses local joblib multiprocessing; `Local` uses a local Dask cluster; `BashSLURM` and `BashLSF` submit scheduler scripts. | +| `queue` | `defq` | Scheduler queue/partition for BashSLURM, BashLSF, or named cluster execution. | +| `reserved_memory` | `4` | Memory request in GB for submitted cluster jobs. | +| `random_state` | `None` in P1 and P6; metadata or `0` in several later phases | Seed for stochastic steps. Set explicitly for reproducible paper runs. | +| `wait_for_cluster_completion` | `True` for BashSLURM/BashLSF config runs | Makes the config runner wait for submitted cluster jobs to write completion markers before starting the next phase. | +| `cluster_phase_timeout` | `86400` | Maximum seconds to wait for a submitted cluster phase. | +| `cluster_phase_poll_interval` | `30` | Seconds between completion-marker checks for submitted cluster phases. | + +### Config Runner Controls -## P2 Impute And Scale +These are command-line controls for `python run.py -c ...`, not `.cfg` keys. -| Parameter | Default or example | Description | +| Parameter | Default value | Description | | --- | --- | --- | -| `imputer_id` | phase default | Registry imputer. | -| `scaler_id` | phase default | Registry scaler. | -| `smote` | `False` | Apply training-fold oversampling after imputation/scaling. | -| `smote_method` | `auto` | Use `SMOTENC` when categorical features are present, otherwise `SMOTE`. | +| `--config`, `-c` | Required | Path to a STREAMLINE `.cfg` or `.ini` file. | +| `--dry_run` | `False` | Print resolved phase runner calls without running phases. | +| `--start_at` | `None` | Start at a phase alias such as `p4` or `p6_modeling`. | +| `--stop_after` | `None` | Stop after a phase alias such as `p8` or `p11`. | +| `--only` | `None` | Run only a comma-separated set of phase aliases. | +| `--skip` | `None` | Skip a comma-separated set of phase aliases. | +| `--log_level` | `INFO` | Python logging level for the config runner. | -## P3 Feature Learning +### Phase Toggles -| Parameter | Default or example | Description | +| Parameter | Default value | Description | | --- | --- | --- | -| `learner_id` | `pca` | Feature learner registry ID. | -| `learner_params` | `{}` | JSON/Python-literal dictionary of learner parameters. | -| `keep_original_features` | `True` | Keep input features alongside learned features. | +| `phase_order` | `p1,p2,p3,p4,p5,p6,p7,p8,p9,p10,p11` | Phase order used by the config runner. | +| `do_p1` | `True` | Run P1 data exploration and processing. | +| `do_p2` | `True` | Run P2 imputation, scaling, and optional SMOTE. | +| `do_p3` | `True` | Run P3 feature learning. | +| `do_p4` | `True` | Run P4 feature importance. | +| `do_p5` | `True` | Run P5 feature selection. | +| `do_p6` | `True` | Run P6 modeling. | +| `do_p7` | `True` | Run P7 ensembles. P7 is skipped automatically for continuous outcomes. | +| `do_p8` | `True` | Run P8 summary statistics and plots. | +| `do_p9` | `True` | Run P9 dataset comparison. P9 skips itself when fewer than two datasets are available. | +| `do_p10` | `True` when replication paths are configured | Run P10 replication or external validation. | +| `do_p11` | `True` | Run P11 reporting. | +| `enabled` | `True` | Per-phase override for disabling an individual phase section, commonly used as `enabled = False` in `[p7]` for regression configs. | +| `do_all` | Not set | Old-style broad toggle that enables or disables all phases when present. | +| `do_till_report` | Not set | Old-style broad toggle for running phases through the standard report path. | -## P4 Feature Importance +### P1 Data Process -| Parameter | Default or example | Description | +| Parameter | Default value | Description | | --- | --- | --- | -| `models` | all registered methods | Feature-importance methods to run. | -| `models_params` | method dictionary | Per-method parameter dictionary. STREAMLINE injects ReBATE `categorical_features` from saved feature-type artifacts. | -| `instance_subset` | not used unless provided | Optional sampling limit for expensive methods. | +| `data_path` | Required | Folder containing raw input datasets, or omitted only when importing prebuilt CV datasets. | +| `exclude_eda_output` | `None` | Optional list of EDA outputs to skip, such as `describe_csv` or `correlation`. | +| `match_label` | `None` | Optional column label used when matching or harmonizing datasets. | +| `ignore_features` | `None` | Optional file or list of feature names to exclude. | +| `categorical_features` | `None` | Optional feature-name file for categorical variables. | +| `quantitative_features` | `None` | Optional feature-name file for quantitative variables. | +| `top_features` | `20` | Number of top features shown in applicable P1 summaries. | +| `categorical_cutoff` | `10` | If feature-type files are absent, features with at most this many unique values may be treated as categorical. | +| `sig_cutoff` | `0.05` | Statistical significance threshold used in P1 analyses. | +| `featureeng_missingness` | `0.5` | Missingness threshold for creating missingness indicator features. | +| `cleaning_missingness` | `0.5` | Missingness threshold for removing high-missingness features or instances. | +| `correlation_removal_threshold` | `1.0` | Correlation threshold for removing highly correlated features. `1.0` effectively disables correlation removal. | +| `partition_method` | `Stratified` | CV partitioning strategy. Continuous outcomes are forced to `Random`. | +| `show_plots` | `False` | Display P1 plots interactively. Usually `False` for batch runs. | +| `one_hot_encoding` | `True` | Expand non-binary categorical features during P1 processing. | +| `cv_provided` | `False` | Import existing CV train/test files instead of creating CV splits from raw datasets. | +| `cv_input_root` | `None` | Root folder containing prebuilt `/CVDatasets` folders when `cv_provided=True`. | +| `enable_plots` | `False` | Master toggle for optional P1 plot generation. | +| `plot_missingness` | `False` | Generate missingness plots. | +| `plot_class_counts` | `False` | Generate outcome/class count plots. | +| `plot_correlation` | `False` | Generate correlation plots. | +| `correlation_plot_max_features` | `200` | Maximum number of features included in correlation plots. | +| `plot_univariate` | `False` | Generate univariate feature analysis plots. | +| `univariate_top_k` | `20` | Number of top univariate features to display. | +| `plot_anomalies` | `False` | Generate anomaly/outlier plots when available. | +| `force` | `False` | Overwrite existing P1 outputs. Demo configs set this to `True` for easy reruns. | -## P5 Feature Selection +### P2 Impute, Scale, And Balance -| Parameter | Default or example | Description | +| Parameter | Default value | Description | | --- | --- | --- | +| `scale_data` | P2 saved metadata, fallback `True` | Scale features using the selected scaler. | +| `impute_data` | P2 saved metadata, fallback `True` | Impute missing feature values. | +| `multi_impute` | P2 saved metadata, fallback `False` | Use multivariate imputation for quantitative features when supported. | +| `overwrite_cv` | `True` | Rewrite CV train/test files with P2 outputs. | +| `outcome_label` | P1 saved metadata, fallback `Class` | Outcome column. | +| `outcome_type` | P1 saved metadata, fallback `None` | Learning task type. | +| `instance_label` | P1 saved metadata, fallback `None` | Optional row identifier column. | +| `random_state` | P1/P2 saved metadata, fallback `0` | Seed for stochastic imputers or SMOTE. | +| `imputer_id` | P2 saved metadata, fallback `None` | Registry imputer ID. `None` uses the phase default. | +| `imputer_params` | P2 saved metadata, fallback `{}` | Dictionary of imputer parameters. | +| `scaler_id` | P2 saved metadata, fallback `None` | Registry scaler ID. `None` uses the phase default. | +| `scaler_params` | P2 saved metadata, fallback `{}` | Dictionary of scaler parameters. | +| `smote` | P2 saved metadata, fallback `False` | Apply classification-only oversampling to training folds after imputation and scaling. | +| `smote_method` | P2 saved metadata, fallback `auto` | `auto`, `smote`, or `smotenc`. `auto` uses SMOTENC when categorical features are present. | +| `smote_sampling_strategy` | P2 saved metadata, fallback `auto` | Sampling strategy passed to imbalanced-learn. | +| `smote_k_neighbors` | P2 saved metadata, fallback `5` | Neighbor count passed to SMOTE or SMOTENC. | +| `--list-imputers` | `False` | CLI-only utility: list discovered imputer registry IDs and exit. | +| `--list-scalers` | `False` | CLI-only utility: list discovered scaler registry IDs and exit. | + +### P3 Feature Learning + +| Parameter | Default value | Description | +| --- | --- | --- | +| `learner_id` | P3 saved metadata, fallback `pca` | Feature learner registry ID. | +| `learner_params` | P3 saved metadata, fallback `{}` | Dictionary of learner parameters. | +| `feature_namespace` | P3 saved metadata, fallback `FL_PCA` | Prefix/namespace for learned feature names. | +| `keep_original_features` | P3 saved metadata, fallback `True` | Keep original features alongside learned features. | +| `overwrite_cv` | `True` | Rewrite CV train/test files with P3 outputs. | +| `outcome_label` | P1 saved metadata, fallback `Class` | Outcome column. | +| `instance_label` | P1 saved metadata, fallback `None` | Optional row identifier column. | +| `random_state` | P1/P3 saved metadata, fallback `0` | Seed for stochastic learners. | +| `--list-learners` | `False` | CLI-only utility: list discovered feature-learning registry IDs and exit. | + +### P4 Feature Importance + +| Parameter | Default value | Description | +| --- | --- | --- | +| `models` | P4 saved metadata, fallback all registered FI methods | Feature-importance methods to run, such as `mutualinformation,multiswrfdb`. | +| `models_params` | P4 saved metadata, fallback ReBATE `n_jobs=1` defaults where applicable | Per-method parameter dictionary. STREAMLINE injects saved categorical feature indexes for ReBATE methods. | +| `top_k` | P4 saved metadata, fallback `None` | Optional top-k selector control for model-specific selected outputs. | +| `threshold` | P4 saved metadata, fallback `None` | Optional score threshold for model-specific selected outputs. | +| `keep_original_features` | P4 saved metadata, fallback `False` | Keep original features in selected-output artifacts when generated. | +| `overwrite_cv` | `True` | Overwrite P4 model-specific outputs. Shared CV files are not mutated by P4. | +| `outcome_label` | P1 saved metadata, fallback `Class` | Outcome column. | +| `outcome_type` | P1 saved metadata, fallback `None` | Learning task type passed to compatible FI methods. | +| `instance_label` | P1 saved metadata, fallback `None` | Optional row identifier column. | +| `random_state` | P1/P4 saved metadata, fallback `0` | Seed for stochastic FI methods. | +| `instance_subset` | P4 saved metadata, fallback `None` | Optional row cap for expensive FI methods. No subsampling is used when `None`. | +| `--list-models` | `False` | CLI-only utility: list discovered feature-importance methods and exit. | + +### P5 Feature Selection + +| Parameter | Default value | Description | +| --- | --- | --- | +| `algorithms` | `auto` | FI algorithms considered by the selector. `auto` discovers completed P4 outputs. | +| `n_splits` | `10` | Number of CV folds expected in FI outputs. Usually inherited from `[run]`. | +| `outcome_label` | P1 saved metadata, fallback `Class` | Outcome column. | +| `instance_label` | P1 saved metadata, fallback `None` | Optional row identifier column. | +| `max_features_to_keep` | `2000` | Upper bound on selected features after combining FI rankings. | +| `filter_poor_features` | `True` | Remove features with consistently poor or zero FI evidence. | +| `overwrite_cv` | `False` | Overwrite P5 selected CV outputs. | | `selector_id` | `default` | Feature selector registry ID. | -| `algorithms` | `auto` | Feature-importance methods considered by selector logic. | -| `top_features` | `20` | Number of features to keep when applicable. | - -## P6 Modeling - -| Parameter | Default or example | Description | -| --- | --- | --- | -| `outcome_type` | `Binary`, `Multiclass`, `Continuous` | Modeling task. `model_type` is still accepted as a backward-compatible alias. | -| `models` | `NB,LR,DT` | Model registry IDs. | -| `model_params_json` | `None` | Optional JSON mapping model IDs to constructor/model overrides. See [Model Parameter JSON](model_params_json.md) for HEROS, ExSTraCS, CLI, cfg, and notebook examples. | -| `scoring_metric` | `balanced_accuracy`, `explained_variance` | Optuna/evaluation metric. | -| `metric_direction` | `maximize` or `minimize` | Optimization direction. | -| `n_trials` | `200` | Optuna trial budget. | -| `timeout` | `900` | Optuna time budget in seconds. | -| `training_subsample` | `0` | Optional training subset size for models that set `subsampling_allowed=True`, including ANN, SVM, KNN, XGB, and HEROS. Classification subsampling is class-balanced by default with imbalanced-learn `RandomUnderSampler(sampling_strategy="auto")`; model wrappers can internally set `subsampling_strategy` to `stratified` for scikit-learn `StratifiedShuffleSplit` or `random`, or set `undersampling_strategy` to another imbalanced-learn string. | -| `calibrate` | `0` or `1` | Classification calibration toggle. | -| `skip_completed_models` | `False` | Skip completed P6 model/CV jobs and run only failed or missing jobs. When `False`, P6 reruns the requested model jobs and overwrites existing model artifacts. | -| `bypass_one_hot_for_native_models` | `True` | Allow native categorical model path. | -| `native_categorical_models` | `CGB,ExSTraCS` | Models allowed when P1 did not one-hot encode. | +| `selector_params` | `{}` | Dictionary of selector parameters. | +| `export_scores` | `True` | Write feature-selection score summaries. | +| `top_features` | `20` | Number of top features shown in P5 plots/summaries. | +| `show_plots` | `False` | Display P5 plots interactively. | +| `strict_discovery` | `False` | Require all expected CV FI files for an algorithm during `auto` discovery. | +| `--list-algorithms` | `False` | CLI-only utility: list available/discovered FI algorithms and exit. | + +### P6 Modeling + +| Parameter | Default value | Description | +| --- | --- | --- | +| `outcome_type` | `None`; resolved from `model_type`, otherwise `Binary` | Modeling task: `Binary`, `Multiclass`, or `Continuous`. The config runner fills this from `[run]` when available. | +| `model_type` | `None` | Backward-compatible alias for `outcome_type`; prefer `outcome_type` in new configs. | +| `models` | All available non-excluded models for the task | Model registry IDs. eLCS is excluded from default discovery. | +| `model_params_json` | `None` | Optional JSON or Python-literal mapping of model IDs to parameter overrides. See [Model Parameter JSON](model_params_json.md). | +| `calibrate` | `False` | Enable probability calibration for classification models. | +| `calibrate_method` | `sigmoid` | Calibration method, usually `sigmoid` or `isotonic`. | +| `calibrate_cv` | `5` | Internal CV folds used for calibration. | +| `scoring_metric` | `balanced_accuracy` | Optuna/evaluation metric. Regression configs should use a regression metric such as `explained_variance`. | +| `metric_direction` | `maximize` | Optuna optimization direction. | +| `n_trials` | `200` | Maximum Optuna trials per model/CV job. | +| `timeout` | `900` | Maximum Optuna seconds per model/CV job. | +| `training_subsample` | `0` | Optional training subset size for models with `subsampling_allowed=True`; `0` disables subsampling. | +| `uniform_fi` | `False` | Use uniform permutation FI handling when supported. | +| `save_plot` | `False` | Save model-level plots generated during modeling. | +| `skip_completed_models` | `False` | When `True`, run only failed or missing model/CV jobs. When `False`, rerun requested jobs and overwrite artifacts. | +| `bypass_one_hot_for_native_models` | `True` | Allow the native categorical model path when P1 was run with `one_hot_encoding=False`. | +| `native_categorical_models` | `CGB,ExSTraCS` | Allowed native-categorical model IDs when one-hot encoding is bypassed. | +| `--list_models` | `False` | CLI-only utility: list default model IDs for the selected task and exit. | +| `--list_models_all` | `False` | CLI-only utility: list all registered model IDs for all tasks and exit. | P6 records Optuna trial accounting in model outputs so reports can show how many trials actually ran within the requested budget. By default, P6 reruns the requested model/CV jobs and overwrites existing model artifacts. Use `skip_completed_models = True` in a config file, or -`--skip_completed_models 1` on the P6 CLI, when you want recovery behavior -that skips completed `job_model_*` markers and runs only failed or missing -jobs. +`--skip_completed_models 1` on the P6 CLI, when you want recovery behavior that +skips completed `job_model_*` markers and runs only failed or missing jobs. -## P7 Ensembles +### P7 Ensembles -P7 is classification-only in the current codebase. +P7 is classification-only. The config runner skips P7 automatically for +continuous outcomes. -| Parameter | Default or example | Description | +| Parameter | Default value | Description | | --- | --- | --- | | `ensembles` | `hard_voting,soft_voting,stack_lr` | Ensemble registry IDs. | -| `base_models` | `NB,LR,DT` | Base model predictions to combine. | -| `meta_train_source` | `train` | Source for stacking meta-training. | +| `base_models` | `None` | Base model predictions to combine. `None` lets P7 discover compatible model outputs. | +| `meta_train_source` | `train` | Source for stacking meta-training data: `train` or `test`. | +| `calibrate` | `False` | Enable calibration for ensemble probabilities when supported. | +| `calibrate_method` | `sigmoid` | Calibration method, usually `sigmoid` or `isotonic`. | +| `calibrate_cv` | `5` | Internal CV folds used for calibration. | +| `random_state` | `0` | Seed for stochastic ensemble behavior. | +| `--list_ensembles` | `False` | CLI-only utility: list ensemble registry IDs and exit. | -## P8 To P11 +### P8 Summary Statistics -| Phase | Key parameters | Notes | +| Parameter | Default value | Description | | --- | --- | --- | -| P8 Summary | `scoring_metric`, `metric_weight`, `top_features`, `include_ensembles` | Aggregates model, ensemble, and feature outputs. | -| P9 Compare | `sig_cutoff`, `show_plots` | Compares datasets in an experiment. | -| P10 Replication | `rep_data_path`, `dataset_for_rep`, `show_plots` | Applies trained workflows to external data. | -| P11 Reporting | `report_modes`, `report_mode`, `make_pdf`, `enable_plots`, `reuse_existing_figures` | Builds standard and replication reports. | +| `outcome_type` | P1/P6 saved metadata, fallback `Binary` | Learning task used to choose classification or regression summaries. | +| `scoring_metric` | `balanced_accuracy` | Primary metric label used in summaries. | +| `metric_weight` | `balanced_accuracy` | Metric used to weight composite model FI plots. Continuous outcomes default to `explained_variance` if an incompatible metric is supplied. | +| `top_features` | `40` | Number of top features shown in composite FI visualizations. | +| `sig_cutoff` | `0.05` | Statistical significance threshold for comparisons. | +| `scale_data` | `True` | Metadata/reporting flag indicating whether scaled data are being summarized. | +| `exclude_plots` | `None` or empty string | Comma-separated plots to skip, such as `plot_ROC,plot_PRC,plot_FI_box,plot_metric_boxplots`. | +| `show_plots` | `False` | Display P8 plots interactively. | +| `include_ensembles` | `True` | Include P7 ensemble outputs in summaries when present. | +| `multiclass_average` | `micro` | Multiclass averaging mode for ROC/PRC summaries: `micro` or `macro`. | -## Saved Run Command Controls +### P9 Compare Datasets -All phase CLIs support: +| Parameter | Default value | Description | +| --- | --- | --- | +| `outcome_label` | `Class` | Outcome column. | +| `outcome_type` | `Binary` | Learning task type. | +| `instance_label` | `None` | Optional row identifier column. | +| `sig_cutoff` | `0.05` | Statistical significance threshold for between-dataset comparisons. | +| `show_plots` | `False` | Display P9 plots interactively. | + +P9 compares datasets within the same experiment and writes a skipped marker when +fewer than two dataset folders with `CVDatasets/` are present. + +### P10 Replication + +| Parameter | Default value | Description | +| --- | --- | --- | +| `rep_data_path` | Required | Folder containing external replication datasets. | +| `dataset_for_rep` | Required | Original training dataset path used to identify the trained dataset output folder. | +| `outcome_label` | P1 saved metadata | Optional override for the outcome column. | +| `instance_label` | P1 saved metadata | Optional override for the row identifier column. | +| `match_label` | `None` | Optional label used to match or harmonize replication inputs. | +| `exclude_plots` | `None` | Comma-separated plots to skip, such as `plot_ROC`, `plot_PRC`, `plot_metric_boxplots`, `plot_FI_box`, or `feature_correlations`. | +| `show_plots` | `False` | Display replication plots interactively. | + +### P11 Reporting + +| Parameter | Default value | Description | +| --- | --- | --- | +| `experiment_path` | Required unless `output_path` and `experiment_name` are provided | Direct path to the experiment output folder. | +| `output_path` | Required unless `experiment_path` is provided | Parent output folder. | +| `experiment_name` | Required unless `experiment_path` is provided | Experiment folder name. | +| `reporting_dir` | `None` | Optional directory for report artifacts. `None` uses the standard experiment reporting folders. | +| `report_modes` | `standard` in config runner unless set | Config-runner convenience parameter for generating multiple report modes, such as `standard,replication`. | +| `report_mode` | `standard` | Single report mode: `standard` or `replication`. | +| `outcome_label` | `Class` in runner, often loaded from metadata/report data | Outcome column used in report labels. | +| `outcome_type` | `Binary` in runner, often loaded from metadata/report data | Learning task type used in report labels and metric filtering. | +| `instance_label` | `None` | Optional row identifier column. | +| `make_pdf` | `True` | Export a PDF report. | +| `enable_plots` | `True` | Generate missing report plots when possible. | +| `reuse_existing_figures` | `True` | Reuse existing report figure PNGs when available. Set to `False` to regenerate report figures. | + +### Saved Run Command Controls + +All phase CLIs support these run-command controls. + +| Flag | Default value | Description | +| --- | --- | --- | +| `--ignore_saved_run_command` | `False` | Ignore `run_commands.pickle` for this run. | +| `--no_update_saved_run_command` | `False` | Do not update `run_commands.pickle` after the run. | -| Flag | Behavior | -| --- | --- | -| `--ignore_saved_run_command` | Ignore `run_commands.pickle` for this run. | -| `--no_update_saved_run_command` | Do not update `run_commands.pickle` after the run. | +Use these flags when you want to run a phase with explicit command-line values +instead of reusing arguments saved from a previous run. diff --git a/docs/source/pipeline.md b/docs/source/pipeline.md index 6fcb2560..fcf0e008 100644 --- a/docs/source/pipeline.md +++ b/docs/source/pipeline.md @@ -6,7 +6,7 @@ STREAMLINE run is called an **experiment**. Each experiment can contain one or more datasets, and each dataset is processed through cross-validation folds so that model evaluation stays separated from model training. -The current v1.0.0 pipeline has eleven phases. P1-P8 are the core training and +The current v1.0.1 pipeline has eleven phases. P1-P8 are the core training and summary workflow, P9 compares multiple datasets inside an experiment, P10 applies trained workflows to external replication data, and P11 produces PDF reports. @@ -140,7 +140,7 @@ P7 builds classification ensembles from P6 base model predictions. Current ensemble methods include hard voting, soft voting, and logistic-regression stacking. -P7 is classification-only in the current v1.0.0 implementation. Regression +P7 is classification-only in the current v1.0.1 implementation. Regression workflows should skip P7 and continue from P6 to P8. ## P8: Summary Statistics @@ -189,17 +189,17 @@ debugging report content without parsing the PDF. from a `.cfg` file: ```bash -python run.py -c run_configs/uci_binary_hcc.cfg --dry_run -python run.py -c run_configs/uci_binary_hcc.cfg +python run.py -c run_configs/local/uci_binary_hcc.cfg --dry_run +python run.py -c run_configs/local/uci_binary_hcc.cfg ``` Useful partial-run controls: ```bash -python run.py -c run_configs/uci_binary_hcc.cfg --start_at p4 -python run.py -c run_configs/uci_binary_hcc.cfg --stop_after p8 -python run.py -c run_configs/uci_binary_hcc.cfg --only p6,p8,p11 -python run.py -c run_configs/uci_binary_hcc.cfg --skip p3,p4 +python run.py -c run_configs/local/uci_binary_hcc.cfg --start_at p4 +python run.py -c run_configs/local/uci_binary_hcc.cfg --stop_after p8 +python run.py -c run_configs/local/uci_binary_hcc.cfg --only p6,p8,p11 +python run.py -c run_configs/local/uci_binary_hcc.cfg --skip p3,p4 ``` ## Saved Run Commands diff --git a/docs/source/running.md b/docs/source/running.md index 7e296371..857dfe1a 100644 --- a/docs/source/running.md +++ b/docs/source/running.md @@ -16,11 +16,11 @@ settings, phase toggles, and phase-specific parameters in one editable file. | --- | --- | | Conference tutorial or first demo | Google Colab notebook | | Interactive local exploration | `STREAMLINE_Notebook.ipynb` | -| Reproducible full pipeline run | `python run.py -c run_configs/.cfg` | +| Reproducible full pipeline run | `python run.py -c run_configs/local/.cfg` | | Debugging one phase | Phase CLI command | | Faster local execution without Dask | `run_cluster = Parallel` | | Local Dask execution | `run_cluster = Local` | -| HPC execution | `BashSLURM`, `BashLSF`, or a site-specific Dask cluster setting | +| HPC execution | `run_configs/hpc/.cfg` with `BashSLURM`, `BashLSF`, or a site-specific Dask cluster setting | ## Google Colab @@ -48,20 +48,20 @@ or custom dataset is run. Dry-run a config first: ```bash -python run.py -c run_configs/uci_binary_hcc.cfg --dry_run +python run.py -c run_configs/local/uci_binary_hcc.cfg --dry_run ``` Run a full binary demo: ```bash -python run.py -c run_configs/uci_binary_hcc.cfg +python run.py -c run_configs/local/uci_binary_hcc.cfg ``` Run the multiclass and regression demos: ```bash -python run.py -c run_configs/uci_multiclass_student.cfg -python run.py -c run_configs/uci_regression_auto_mpg.cfg +python run.py -c run_configs/local/uci_multiclass_student.cfg +python run.py -c run_configs/local/uci_regression_auto_mpg.cfg ``` The included configs are designed as reproducible examples. For a short @@ -71,10 +71,10 @@ smaller modeling budget and shows plots by default. Partial-run examples: ```bash -python run.py -c run_configs/uci_binary_hcc.cfg --start_at p4 -python run.py -c run_configs/uci_binary_hcc.cfg --stop_after p8 -python run.py -c run_configs/uci_binary_hcc.cfg --only p6,p8,p11 -python run.py -c run_configs/uci_binary_hcc.cfg --skip p3,p4 +python run.py -c run_configs/local/uci_binary_hcc.cfg --start_at p4 +python run.py -c run_configs/local/uci_binary_hcc.cfg --stop_after p8 +python run.py -c run_configs/local/uci_binary_hcc.cfg --only p6,p8,p11 +python run.py -c run_configs/local/uci_binary_hcc.cfg --skip p3,p4 ``` ## Config File Layout @@ -117,7 +117,8 @@ timeout = 900 ``` Use `run_cluster = Local` for a local Dask cluster, or `run_cluster = Parallel` -for local joblib parallelism without Dask. +for local joblib parallelism without Dask. Use scheduler configs for cluster +runs rather than changing a local demo config in place. Use `model_params_json` when a model needs specific Phase 6 wrapper settings, such as HEROS `pop_size` or ExSTraCS `N`. See @@ -130,14 +131,23 @@ has finished writing `CVDatasets`. The wait can be adjusted with `wait_for_cluster_completion`, `cluster_phase_timeout`, and `cluster_phase_poll_interval` in the `[run]` section. +For long HPC runs, launch the config runner inside `tmux` or `screen` so a lost +SSH connection does not stop the orchestration process. See +[HPC and Cluster Runs](hpc.md) for Cedars SLURM and UPenn LSF templates, +scheduler monitoring commands, and recovery patterns. + `Parallel` and Dask-backed runs show progress when `tqdm`/Dask progress support is available. Set `STREAMLINE_PROGRESS=0` to disable these progress displays. -Use the included configs as templates: +Use the included configs as templates. Local demo configs live in +`run_configs/local/`, and scheduler/HPC templates live in `run_configs/hpc/`. +The original top-level demo config paths are kept for backward compatibility. -* `run_configs/uci_binary_hcc.cfg` -* `run_configs/uci_multiclass_student.cfg` -* `run_configs/uci_regression_auto_mpg.cfg` +* `run_configs/local/uci_binary_hcc.cfg` +* `run_configs/local/uci_multiclass_student.cfg` +* `run_configs/local/uci_regression_auto_mpg.cfg` +* `run_configs/hpc/cedars_slurm_hcc.cfg` +* `run_configs/hpc/upenn_lsf_hcc.cfg` ## Rerunning Phases diff --git a/docs/source/tips.md b/docs/source/tips.md index 0d3eeb5e..a804aaa1 100644 --- a/docs/source/tips.md +++ b/docs/source/tips.md @@ -5,7 +5,7 @@ Before launching a full config, inspect the resolved phase calls: ```bash -python run.py -c run_configs/uci_binary_hcc.cfg --dry_run +python run.py -c run_configs/local/uci_binary_hcc.cfg --dry_run ``` This catches most path, phase toggle, and parameter-name mistakes early. diff --git a/run_configs/hpc/cedars_slurm_hcc.cfg b/run_configs/hpc/cedars_slurm_hcc.cfg new file mode 100644 index 00000000..e8ccedab --- /dev/null +++ b/run_configs/hpc/cedars_slurm_hcc.cfg @@ -0,0 +1,160 @@ +[run] +# Cedars / SLURM-oriented template. Edit output_path, queue, reserved_memory, +# and any account/site scheduler settings before launching on a cluster. +output_path = out +experiment_name = UCIHCCPipeline_CedarsSLURM +outcome_label = Class +outcome_type = Binary +instance_label = InstanceID +n_splits = 3 +run_cluster = BashSLURM +wait_for_cluster_completion = True +cluster_phase_timeout = 86400 +cluster_phase_poll_interval = 30 +queue = defq +reserved_memory = 4 +random_state = 42 + +[phases] +phase_order = p1,p2,p3,p4,p5,p6,p7,p8,p9,p10,p11 +do_p1 = True +do_p2 = True +do_p3 = True +do_p4 = True +do_p5 = True +do_p6 = True +do_p7 = True +do_p8 = True +do_p9 = True +do_p10 = True +do_p11 = True + +[p1] +data_path = data/UCIBinaryClassification +exclude_eda_output = None +match_label = None +ignore_features = None +categorical_features = data/UCIFeatureTypes/hcc_survival_categorical_features.csv +quantitative_features = data/UCIFeatureTypes/hcc_survival_quantitative_features.csv +top_features = 20 +categorical_cutoff = 10 +sig_cutoff = 0.05 +featureeng_missingness = 0.5 +cleaning_missingness = 0.5 +correlation_removal_threshold = 1.0 +partition_method = Stratified +show_plots = False +one_hot_encoding = True +cv_provided = False +cv_input_root = None +enable_plots = False +plot_missingness = False +plot_class_counts = False +plot_correlation = False +correlation_plot_max_features = 200 +plot_univariate = False +univariate_top_k = 20 +plot_anomalies = False +force = True + +[p2] +scale_data = True +impute_data = True +multi_impute = False +overwrite_cv = True +imputer_id = None +imputer_params = {} +scaler_id = None +scaler_params = {} +smote = False +smote_method = auto +smote_sampling_strategy = auto +smote_k_neighbors = 5 + +[p3] +learner_id = pca +learner_params = {} +feature_namespace = FL_PCA +keep_original_features = True +overwrite_cv = True + +[p4] +models = mutualinformation,multiswrfdb +models_params = {'mutualinformation': {'outcome_type': 'Binary'}, 'multiswrfdb': {'n_jobs': 1}} +top_k = None +threshold = None +keep_original_features = False +overwrite_cv = True +instance_subset = None + +[p5] +algorithms = auto +n_splits = 3 +max_features_to_keep = 2000 +filter_poor_features = True +overwrite_cv = False +selector_id = default +selector_params = {} +export_scores = True +top_features = 20 +show_plots = False +strict_discovery = False + +[p6] +outcome_type = Binary +model_type = None +models = NB,LR,DT +model_params_json = None +calibrate = False +calibrate_method = sigmoid +calibrate_cv = 5 +scoring_metric = balanced_accuracy +metric_direction = maximize +n_trials = 200 +timeout = 900 +training_subsample = 0 +uniform_fi = False +save_plot = False +skip_completed_models = False +bypass_one_hot_for_native_models = False +native_categorical_models = CGB,ExSTraCS + +[p7] +ensembles = hard_voting,soft_voting,stack_lr +base_models = NB,LR,DT +meta_train_source = train +calibrate = 0 +calibrate_method = sigmoid +calibrate_cv = 5 + +[p8] +scoring_metric = balanced_accuracy +metric_weight = balanced_accuracy +top_features = 40 +sig_cutoff = 0.05 +scale_data = True +exclude_plots = None +show_plots = False +include_ensembles = True +multiclass_average = micro + +[p9] +sig_cutoff = 0.05 +show_plots = False + +[p10] +rep_data_path = data/UCIRepBinaryClassification +dataset_for_rep = data/UCIBinaryClassification/hcc_survival.csv +match_label = None +exclude_plots = None +show_plots = False + +[p11] +report_modes = standard,replication +reporting_dir = None +outcome_label = Class +outcome_type = Binary +instance_label = InstanceID +make_pdf = True +enable_plots = True +reuse_existing_figures = True diff --git a/run_configs/hpc/upenn_lsf_hcc.cfg b/run_configs/hpc/upenn_lsf_hcc.cfg new file mode 100644 index 00000000..0deec383 --- /dev/null +++ b/run_configs/hpc/upenn_lsf_hcc.cfg @@ -0,0 +1,160 @@ +[run] +# UPenn / LSF-oriented template. Edit output_path, queue, reserved_memory, +# and any account/site scheduler settings before launching on a cluster. +output_path = out +experiment_name = UCIHCCPipeline_UPennLSF +outcome_label = Class +outcome_type = Binary +instance_label = InstanceID +n_splits = 3 +run_cluster = BashLSF +wait_for_cluster_completion = True +cluster_phase_timeout = 86400 +cluster_phase_poll_interval = 30 +queue = i2c2_normal +reserved_memory = 4 +random_state = 42 + +[phases] +phase_order = p1,p2,p3,p4,p5,p6,p7,p8,p9,p10,p11 +do_p1 = True +do_p2 = True +do_p3 = True +do_p4 = True +do_p5 = True +do_p6 = True +do_p7 = True +do_p8 = True +do_p9 = True +do_p10 = True +do_p11 = True + +[p1] +data_path = data/UCIBinaryClassification +exclude_eda_output = None +match_label = None +ignore_features = None +categorical_features = data/UCIFeatureTypes/hcc_survival_categorical_features.csv +quantitative_features = data/UCIFeatureTypes/hcc_survival_quantitative_features.csv +top_features = 20 +categorical_cutoff = 10 +sig_cutoff = 0.05 +featureeng_missingness = 0.5 +cleaning_missingness = 0.5 +correlation_removal_threshold = 1.0 +partition_method = Stratified +show_plots = False +one_hot_encoding = True +cv_provided = False +cv_input_root = None +enable_plots = False +plot_missingness = False +plot_class_counts = False +plot_correlation = False +correlation_plot_max_features = 200 +plot_univariate = False +univariate_top_k = 20 +plot_anomalies = False +force = True + +[p2] +scale_data = True +impute_data = True +multi_impute = False +overwrite_cv = True +imputer_id = None +imputer_params = {} +scaler_id = None +scaler_params = {} +smote = False +smote_method = auto +smote_sampling_strategy = auto +smote_k_neighbors = 5 + +[p3] +learner_id = pca +learner_params = {} +feature_namespace = FL_PCA +keep_original_features = True +overwrite_cv = True + +[p4] +models = mutualinformation,multiswrfdb +models_params = {'mutualinformation': {'outcome_type': 'Binary'}, 'multiswrfdb': {'n_jobs': 1}} +top_k = None +threshold = None +keep_original_features = False +overwrite_cv = True +instance_subset = None + +[p5] +algorithms = auto +n_splits = 3 +max_features_to_keep = 2000 +filter_poor_features = True +overwrite_cv = False +selector_id = default +selector_params = {} +export_scores = True +top_features = 20 +show_plots = False +strict_discovery = False + +[p6] +outcome_type = Binary +model_type = None +models = NB,LR,DT +model_params_json = None +calibrate = False +calibrate_method = sigmoid +calibrate_cv = 5 +scoring_metric = balanced_accuracy +metric_direction = maximize +n_trials = 200 +timeout = 900 +training_subsample = 0 +uniform_fi = False +save_plot = False +skip_completed_models = False +bypass_one_hot_for_native_models = False +native_categorical_models = CGB,ExSTraCS + +[p7] +ensembles = hard_voting,soft_voting,stack_lr +base_models = NB,LR,DT +meta_train_source = train +calibrate = 0 +calibrate_method = sigmoid +calibrate_cv = 5 + +[p8] +scoring_metric = balanced_accuracy +metric_weight = balanced_accuracy +top_features = 40 +sig_cutoff = 0.05 +scale_data = True +exclude_plots = None +show_plots = False +include_ensembles = True +multiclass_average = micro + +[p9] +sig_cutoff = 0.05 +show_plots = False + +[p10] +rep_data_path = data/UCIRepBinaryClassification +dataset_for_rep = data/UCIBinaryClassification/hcc_survival.csv +match_label = None +exclude_plots = None +show_plots = False + +[p11] +report_modes = standard,replication +reporting_dir = None +outcome_label = Class +outcome_type = Binary +instance_label = InstanceID +make_pdf = True +enable_plots = True +reuse_existing_figures = True diff --git a/run_configs/local/uci_binary_hcc.cfg b/run_configs/local/uci_binary_hcc.cfg new file mode 100644 index 00000000..d7bed967 --- /dev/null +++ b/run_configs/local/uci_binary_hcc.cfg @@ -0,0 +1,159 @@ +[run] +output_path = out +experiment_name = UCIHCCPipeline +outcome_label = Class +outcome_type = Binary +instance_label = InstanceID +n_splits = 3 +# Options: Serial, Local (Dask), Parallel (joblib), BashSLURM, BashLSF, or a named Dask cluster. +run_cluster = Serial +wait_for_cluster_completion = True +cluster_phase_timeout = 86400 +cluster_phase_poll_interval = 30 +queue = defq +reserved_memory = 4 +random_state = 42 + +[phases] +phase_order = p1,p2,p3,p4,p5,p6,p7,p8,p9,p10,p11 +do_p1 = True +do_p2 = True +do_p3 = True +do_p4 = True +do_p5 = True +do_p6 = True +do_p7 = True +do_p8 = True +do_p9 = True +do_p10 = True +do_p11 = True + +[p1] +data_path = data/UCIBinaryClassification +exclude_eda_output = None +match_label = None +ignore_features = None +categorical_features = data/UCIFeatureTypes/hcc_survival_categorical_features.csv +quantitative_features = data/UCIFeatureTypes/hcc_survival_quantitative_features.csv +top_features = 20 +categorical_cutoff = 10 +sig_cutoff = 0.05 +featureeng_missingness = 0.5 +cleaning_missingness = 0.5 +correlation_removal_threshold = 1.0 +partition_method = Stratified +show_plots = False +one_hot_encoding = True +cv_provided = False +cv_input_root = None +enable_plots = False +plot_missingness = False +plot_class_counts = False +plot_correlation = False +correlation_plot_max_features = 200 +plot_univariate = False +univariate_top_k = 20 +plot_anomalies = False +force = True + +[p2] +scale_data = True +impute_data = True +multi_impute = False +overwrite_cv = True +imputer_id = None +imputer_params = {} +scaler_id = None +scaler_params = {} +smote = False +smote_method = auto +smote_sampling_strategy = auto +smote_k_neighbors = 5 + +[p3] +learner_id = pca +learner_params = {} +feature_namespace = FL_PCA +keep_original_features = True +overwrite_cv = True + +[p4] +models = mutualinformation,multiswrfdb +models_params = {'mutualinformation': {'outcome_type': 'Binary'}, 'multiswrfdb': {'n_jobs': 1}} +top_k = None +threshold = None +keep_original_features = False +overwrite_cv = True +instance_subset = None + +[p5] +algorithms = auto +n_splits = 3 +max_features_to_keep = 2000 +filter_poor_features = True +overwrite_cv = False +selector_id = default +selector_params = {} +export_scores = True +top_features = 20 +show_plots = False +strict_discovery = False + +[p6] +outcome_type = Binary +model_type = None +models = NB,LR,DT +model_params_json = None +calibrate = False +calibrate_method = sigmoid +calibrate_cv = 5 +scoring_metric = balanced_accuracy +metric_direction = maximize +n_trials = 200 +timeout = 900 +training_subsample = 0 +uniform_fi = False +save_plot = False +skip_completed_models = False +bypass_one_hot_for_native_models = False +native_categorical_models = CGB,ExSTraCS + +[p7] +ensembles = hard_voting,soft_voting,stack_lr +base_models = NB,LR,DT +meta_train_source = train +calibrate = 0 +calibrate_method = sigmoid +calibrate_cv = 5 + +[p8] +scoring_metric = balanced_accuracy +metric_weight = balanced_accuracy +top_features = 40 +sig_cutoff = 0.05 +scale_data = True +exclude_plots = None +show_plots = False +include_ensembles = True +multiclass_average = micro + +[p9] +sig_cutoff = 0.05 +show_plots = False + +[p10] +rep_data_path = data/UCIRepBinaryClassification +dataset_for_rep = data/UCIBinaryClassification/hcc_survival.csv +match_label = None +exclude_plots = None +show_plots = False + +[p11] +report_modes = standard,replication +reporting_dir = None +outcome_label = Class +outcome_type = Binary +instance_label = InstanceID +make_pdf = True +enable_plots = True +reuse_existing_figures = True diff --git a/run_configs/local/uci_multiclass_student.cfg b/run_configs/local/uci_multiclass_student.cfg new file mode 100644 index 00000000..d43c33df --- /dev/null +++ b/run_configs/local/uci_multiclass_student.cfg @@ -0,0 +1,159 @@ +[run] +output_path = out +experiment_name = UCIStudentPipeline +outcome_label = Class +outcome_type = Multiclass +instance_label = InstanceID +n_splits = 3 +# Options: Serial, Local (Dask), Parallel (joblib), BashSLURM, BashLSF, or a named Dask cluster. +run_cluster = Serial +wait_for_cluster_completion = True +cluster_phase_timeout = 86400 +cluster_phase_poll_interval = 30 +queue = defq +reserved_memory = 4 +random_state = 42 + +[phases] +phase_order = p1,p2,p3,p4,p5,p6,p7,p8,p9,p10,p11 +do_p1 = True +do_p2 = True +do_p3 = True +do_p4 = True +do_p5 = True +do_p6 = True +do_p7 = True +do_p8 = True +do_p9 = True +do_p10 = True +do_p11 = True + +[p1] +data_path = data/UCIMulticlassClassification +exclude_eda_output = None +match_label = None +ignore_features = None +categorical_features = data/UCIFeatureTypes/student_dropout_categorical_features.csv +quantitative_features = data/UCIFeatureTypes/student_dropout_quantitative_features.csv +top_features = 20 +categorical_cutoff = 10 +sig_cutoff = 0.05 +featureeng_missingness = 0.5 +cleaning_missingness = 0.5 +correlation_removal_threshold = 1.0 +partition_method = Stratified +show_plots = False +one_hot_encoding = True +cv_provided = False +cv_input_root = None +enable_plots = False +plot_missingness = False +plot_class_counts = False +plot_correlation = False +correlation_plot_max_features = 200 +plot_univariate = False +univariate_top_k = 20 +plot_anomalies = False +force = True + +[p2] +scale_data = True +impute_data = True +multi_impute = False +overwrite_cv = True +imputer_id = None +imputer_params = {} +scaler_id = None +scaler_params = {} +smote = False +smote_method = auto +smote_sampling_strategy = auto +smote_k_neighbors = 5 + +[p3] +learner_id = pca +learner_params = {} +feature_namespace = FL_PCA +keep_original_features = True +overwrite_cv = True + +[p4] +models = mutualinformation,multiswrfdb +models_params = {'mutualinformation': {'outcome_type': 'Multiclass'}, 'multiswrfdb': {'n_jobs': 1}} +top_k = None +threshold = None +keep_original_features = False +overwrite_cv = True +instance_subset = 1000 + +[p5] +algorithms = auto +n_splits = 3 +max_features_to_keep = 2000 +filter_poor_features = True +overwrite_cv = False +selector_id = default +selector_params = {} +export_scores = True +top_features = 20 +show_plots = False +strict_discovery = False + +[p6] +outcome_type = Multiclass +model_type = None +models = NB,LR,DT +model_params_json = None +calibrate = False +calibrate_method = sigmoid +calibrate_cv = 5 +scoring_metric = balanced_accuracy +metric_direction = maximize +n_trials = 200 +timeout = 900 +training_subsample = 0 +uniform_fi = False +save_plot = False +skip_completed_models = False +bypass_one_hot_for_native_models = False +native_categorical_models = CGB,ExSTraCS + +[p7] +ensembles = hard_voting,soft_voting,stack_lr +base_models = NB,LR,DT +meta_train_source = train +calibrate = 0 +calibrate_method = sigmoid +calibrate_cv = 5 + +[p8] +scoring_metric = balanced_accuracy +metric_weight = balanced_accuracy +top_features = 40 +sig_cutoff = 0.05 +scale_data = True +exclude_plots = None +show_plots = False +include_ensembles = True +multiclass_average = micro + +[p9] +sig_cutoff = 0.05 +show_plots = False + +[p10] +rep_data_path = data/UCIRepMulticlassClassification +dataset_for_rep = data/UCIMulticlassClassification/student_dropout_academic_success.csv +match_label = None +exclude_plots = None +show_plots = False + +[p11] +report_modes = standard,replication +reporting_dir = None +outcome_label = Class +outcome_type = Multiclass +instance_label = InstanceID +make_pdf = True +enable_plots = True +reuse_existing_figures = True diff --git a/run_configs/local/uci_regression_auto_mpg.cfg b/run_configs/local/uci_regression_auto_mpg.cfg new file mode 100644 index 00000000..fa1fe832 --- /dev/null +++ b/run_configs/local/uci_regression_auto_mpg.cfg @@ -0,0 +1,160 @@ +[run] +output_path = out +experiment_name = UCIAutoMPGPipeline +outcome_label = MPG +outcome_type = Continuous +instance_label = InstanceID +n_splits = 3 +# Options: Serial, Local (Dask), Parallel (joblib), BashSLURM, BashLSF, or a named Dask cluster. +run_cluster = Serial +wait_for_cluster_completion = True +cluster_phase_timeout = 86400 +cluster_phase_poll_interval = 30 +queue = defq +reserved_memory = 4 +random_state = 42 + +[phases] +phase_order = p1,p2,p3,p4,p5,p6,p7,p8,p9,p10,p11 +do_p1 = True +do_p2 = True +do_p3 = True +do_p4 = True +do_p5 = True +do_p6 = True +do_p7 = False +do_p8 = True +do_p9 = True +do_p10 = True +do_p11 = True + +[p1] +data_path = data/UCIRegression +exclude_eda_output = None +match_label = None +ignore_features = None +categorical_features = data/UCIFeatureTypes/auto_mpg_categorical_features.csv +quantitative_features = data/UCIFeatureTypes/auto_mpg_quantitative_features.csv +top_features = 20 +categorical_cutoff = 10 +sig_cutoff = 0.05 +featureeng_missingness = 0.5 +cleaning_missingness = 0.5 +correlation_removal_threshold = 1.0 +partition_method = Random +show_plots = False +one_hot_encoding = True +cv_provided = False +cv_input_root = None +enable_plots = False +plot_missingness = False +plot_class_counts = False +plot_correlation = False +correlation_plot_max_features = 200 +plot_univariate = False +univariate_top_k = 20 +plot_anomalies = False +force = True + +[p2] +scale_data = True +impute_data = True +multi_impute = False +overwrite_cv = True +imputer_id = None +imputer_params = {} +scaler_id = None +scaler_params = {} +smote = False +smote_method = auto +smote_sampling_strategy = auto +smote_k_neighbors = 5 + +[p3] +learner_id = pca +learner_params = {} +feature_namespace = FL_PCA +keep_original_features = True +overwrite_cv = True + +[p4] +models = mutualinformation,multiswrfdb +models_params = {'mutualinformation': {'outcome_type': 'Continuous'}, 'multiswrfdb': {'n_jobs': 1}} +top_k = None +threshold = None +keep_original_features = False +overwrite_cv = True +instance_subset = None + +[p5] +algorithms = auto +n_splits = 3 +max_features_to_keep = 2000 +filter_poor_features = True +overwrite_cv = False +selector_id = default +selector_params = {} +export_scores = True +top_features = 20 +show_plots = False +strict_discovery = False + +[p6] +outcome_type = Continuous +model_type = None +models = LR,RF +model_params_json = None +calibrate = False +calibrate_method = sigmoid +calibrate_cv = 5 +scoring_metric = explained_variance +metric_direction = maximize +n_trials = 200 +timeout = 900 +training_subsample = 0 +uniform_fi = False +save_plot = False +skip_completed_models = False +bypass_one_hot_for_native_models = False +native_categorical_models = CGB,ExSTraCS + +[p7] +enabled = False +ensembles = hard_voting,soft_voting,stack_lr +base_models = LR,RF +meta_train_source = train +calibrate = 0 +calibrate_method = sigmoid +calibrate_cv = 5 + +[p8] +scoring_metric = explained_variance +metric_weight = explained_variance +top_features = 40 +sig_cutoff = 0.05 +scale_data = True +exclude_plots = None +show_plots = False +include_ensembles = False +multiclass_average = micro + +[p9] +sig_cutoff = 0.05 +show_plots = False + +[p10] +rep_data_path = data/UCIRepRegression +dataset_for_rep = data/UCIRegression/auto_mpg.csv +match_label = None +exclude_plots = None +show_plots = False + +[p11] +report_modes = standard,replication +reporting_dir = None +outcome_label = MPG +outcome_type = Continuous +instance_label = InstanceID +make_pdf = True +enable_plots = True +reuse_existing_figures = True diff --git a/sample_runcommands.txt b/sample_runcommands.txt index ca73c6e3..b05ceffa 100644 --- a/sample_runcommands.txt +++ b/sample_runcommands.txt @@ -27,35 +27,36 @@ # These run the same P1-P11 runner classes used by the phase # commands below, with shared and phase-specific arguments loaded # from editable .cfg files. Use --dry_run first to inspect resolved calls. +# Local examples live under run_configs/local/. HPC templates live under run_configs/hpc/. python run.py \ - -c run_configs/uci_binary_hcc.cfg \ + -c run_configs/local/uci_binary_hcc.cfg \ --dry_run python run.py \ - -c run_configs/uci_binary_hcc.cfg + -c run_configs/local/uci_binary_hcc.cfg python run.py \ - -c run_configs/uci_multiclass_student.cfg + -c run_configs/local/uci_multiclass_student.cfg python run.py \ - -c run_configs/uci_regression_auto_mpg.cfg + -c run_configs/local/uci_regression_auto_mpg.cfg # Useful partial-run controls: python run.py \ - -c run_configs/uci_binary_hcc.cfg \ + -c run_configs/local/uci_binary_hcc.cfg \ --start_at p4 python run.py \ - -c run_configs/uci_binary_hcc.cfg \ + -c run_configs/local/uci_binary_hcc.cfg \ --stop_after p8 python run.py \ - -c run_configs/uci_binary_hcc.cfg \ + -c run_configs/local/uci_binary_hcc.cfg \ --only p6,p8,p11 python run.py \ - -c run_configs/uci_binary_hcc.cfg \ + -c run_configs/local/uci_binary_hcc.cfg \ --skip p3,p4 diff --git a/streamline/__init__.py b/streamline/__init__.py index c948509a..b14db20a 100644 --- a/streamline/__init__.py +++ b/streamline/__init__.py @@ -3,6 +3,6 @@ Pipeline for Supervised Learning in Tabular Classification and Regression Data """ -__version__ = "1.0.0" +__version__ = "1.0.1" __author__ = 'Harsh Bandhey and Ryan Urbanowicz' __credits__ = 'UrbsLabs' diff --git a/streamline/p11_reporting/reporting.py b/streamline/p11_reporting/reporting.py index 89cc87cb..f6311205 100644 --- a/streamline/p11_reporting/reporting.py +++ b/streamline/p11_reporting/reporting.py @@ -1229,9 +1229,22 @@ def _detect_task_from_train(self, ds_dir: Path, metadata: Dict[str, Any]) -> str def _detect_task_type(self, ds_dir: Path, metadata: Dict[str, Any]) -> str: # Rule priority: - # 1) ClassCounts.csv if clearly binary. - # 2) ClassCounts with high-cardinality numeric labels can indicate regression. + # 1) Explicit run metadata/CLI outcome type. + # 2) ClassCounts.csv when no explicit task type was saved. # 3) Otherwise infer from *_Train.csv target values. + outcome_type = str( + self.outcome_type + or metadata.get("Outcome Type") + or metadata.get("outcome_type") + or "" + ).strip().lower() + if outcome_type in {"continuous", "regression", "numeric", "real", "float"}: + return "Regression" + if outcome_type in {"binary", "binary classification"}: + return "Binary Classification" + if outcome_type in {"multiclass", "multiclass classification", "multi-class"}: + return "Multiclass Classification" + cc = self._read_csv_table(ds_dir / "exploratory" / "ClassCounts.csv") if cc and cc.rows: label_col = cc.columns[0] @@ -2692,11 +2705,29 @@ def as_map(table: Optional[TableData], alg_label_hint: str) -> Tuple[str, Dict[s ordered_algs = list(m_map.keys()) ordered_ens = list(em_map.keys()) + all_rows = list(m_map.values()) + list(em_map.values()) + + def metric_value(row: Dict[str, str], metric: str) -> str: + if metric in row: + return row.get(metric, "") + + json_key = METRIC_JSON_KEYS.get(metric) + if json_key and json_key in row: + return row.get(json_key, "") + + normalized_metric = metric.strip().lower().replace(" ", "_").replace("-", "_") + normalized_json_key = str(json_key or "").strip().lower() + for col, val in row.items(): + normalized_col = col.strip().lower().replace(" ", "_").replace("-", "_") + if normalized_col in {normalized_metric, normalized_json_key}: + return val + return "" + available_metrics: List[str] = [] for metric in metrics: present = False - for row in list(m_map.values()) + list(em_map.values()): - if metric in row: + for row in all_rows: + if metric_value(row, metric) != "": present = True break if present: @@ -2709,8 +2740,8 @@ def as_map(table: Optional[TableData], alg_label_hint: str) -> Tuple[str, Dict[s row = [alg] raw_means[alg] = {} for metric in available_metrics: - mval = _safe_float(m_map.get(alg, {}).get(metric, "")) - sval = _safe_float(s_map.get(alg, {}).get(metric, "")) + mval = _safe_float(metric_value(m_map.get(alg, {}), metric)) + sval = _safe_float(metric_value(s_map.get(alg, {}), metric)) if mval is None: row.append("") continue @@ -2726,8 +2757,8 @@ def as_map(table: Optional[TableData], alg_label_hint: str) -> Tuple[str, Dict[s row = [label] raw_means[label] = {} for metric in available_metrics: - mval = _safe_float(em_map.get(ens, {}).get(metric, "")) - sval = _safe_float(es_map.get(ens, {}).get(metric, "")) + mval = _safe_float(metric_value(em_map.get(ens, {}), metric)) + sval = _safe_float(metric_value(es_map.get(ens, {}), metric)) if mval is None: row.append("") continue @@ -2739,6 +2770,8 @@ def as_map(table: Optional[TableData], alg_label_hint: str) -> Tuple[str, Dict[s mean_rows.append(row) mean_columns = ["Algorithm"] + available_metrics + base_mean_rows = mean_rows[: len(ordered_algs)] + ensemble_mean_rows = mean_rows[len(ordered_algs) :] # Highlight best mean per metric with tie handling at 3 decimals. mean_highlight_cells: Set[Tuple[int, int]] = set() @@ -2764,7 +2797,7 @@ def as_map(table: Optional[TableData], alg_label_hint: str) -> Tuple[str, Dict[s row = [alg] median_raw[alg] = {} for metric in available_metrics: - val = _safe_float(md_map.get(alg, {}).get(metric, "")) + val = _safe_float(metric_value(md_map.get(alg, {}), metric)) if val is None: row.append("") else: @@ -2776,7 +2809,7 @@ def as_map(table: Optional[TableData], alg_label_hint: str) -> Tuple[str, Dict[s row = [label] median_raw[label] = {} for metric in available_metrics: - val = _safe_float(ed_map.get(ens, {}).get(metric, "")) + val = _safe_float(metric_value(ed_map.get(ens, {}), metric)) if val is None: row.append("") else: @@ -2785,6 +2818,8 @@ def as_map(table: Optional[TableData], alg_label_hint: str) -> Tuple[str, Dict[s median_rows.append(row) median_columns = ["Algorithm"] + available_metrics + base_median_rows = median_rows[: len(ordered_algs)] + ensemble_median_rows = median_rows[len(ordered_algs) :] median_highlight_cells: Set[Tuple[int, int]] = set() for c_idx, metric in enumerate(available_metrics, start=1): scored: List[Tuple[int, float]] = [] @@ -2801,6 +2836,27 @@ def as_map(table: Optional[TableData], alg_label_hint: str) -> Tuple[str, Dict[s if v == best_val: median_highlight_cells.add((r_idx, c_idx)) + def highlight_cells_for_rows( + rows_to_score: List[List[str]], + raw_values: Dict[str, Dict[str, float]], + ) -> List[Tuple[int, int]]: + highlight_cells: Set[Tuple[int, int]] = set() + for c_idx, metric in enumerate(available_metrics, start=1): + scored: List[Tuple[int, float]] = [] + for r_idx, row in enumerate(rows_to_score, start=1): + raw_name = row[0] if row else "" + val = raw_values.get(raw_name, {}).get(metric) + if val is not None: + scored.append((r_idx, round(val, 3))) + if not scored: + continue + higher = METRIC_DIRECTION_HIGHER_IS_BETTER.get(metric, True) + best_val = max(v for _, v in scored) if higher else min(v for _, v in scored) + for r_idx, v in scored: + if v == best_val: + highlight_cells.add((r_idx, c_idx)) + return [(r, c) for r, c in sorted(highlight_cells)] + return { "mean_columns": mean_columns, "mean_rows": mean_rows, @@ -2810,6 +2866,18 @@ def as_map(table: Optional[TableData], alg_label_hint: str) -> Tuple[str, Dict[s "median_rows": median_rows, "median_highlight_cells": [(r, c) for r, c in sorted(median_highlight_cells)], "median_bold_cells": [(r, c) for r, c in sorted(median_highlight_cells)], + "model_mean_columns": mean_columns, + "model_mean_rows": base_mean_rows, + "model_mean_highlight_cells": highlight_cells_for_rows(base_mean_rows, raw_means), + "model_median_columns": median_columns, + "model_median_rows": base_median_rows, + "model_median_highlight_cells": highlight_cells_for_rows(base_median_rows, median_raw), + "ensemble_mean_columns": mean_columns, + "ensemble_mean_rows": ensemble_mean_rows, + "ensemble_mean_highlight_cells": highlight_cells_for_rows(ensemble_mean_rows, raw_means), + "ensemble_median_columns": median_columns, + "ensemble_median_rows": ensemble_median_rows, + "ensemble_median_highlight_cells": highlight_cells_for_rows(ensemble_median_rows, median_raw), } def _resolve_dataset_images( @@ -3661,18 +3729,50 @@ def _render_global_summary(self, pdf: _StreamlinePDF, report_data: Dict[str, Any + 1 ) - def _render_dataset_header(self, pdf: _StreamlinePDF, ds: Dict[str, Any], section_title: str): + def _render_dataset_header( + self, + pdf: _StreamlinePDF, + ds: Dict[str, Any], + section_title: str, + *, + primary: bool = False, + ): pdf.add_page() - pdf.set_font("Times", "B", 11) - _pdf_cell(pdf, 190, 6, section_title, border=1, align="L", new_line=True) - pdf.set_font("Times", "B", 10) - pdf.set_fill_color(235, 238, 242) - _pdf_cell(pdf, 190, 6.5, f"{ds.get('dataset_id')} | Dataset: {ds.get('dataset_name')}", border=1, align="L", fill=True, new_line=True) + if primary: + pdf.set_font("Times", "B", 16) + pdf.set_fill_color(218, 226, 238) + _pdf_cell( + pdf, + 190, + 12.5, + f"{ds.get('dataset_id')}: {ds.get('dataset_name')}", + border=1, + align="L", + fill=True, + new_line=True, + ) + pdf.set_font("Times", "B", 10) + _pdf_cell(pdf, 190, 6, section_title, border=1, align="L", new_line=True) + else: + pdf.set_font("Times", "B", 11) + _pdf_cell(pdf, 190, 6, section_title, border=1, align="L", new_line=True) + pdf.set_font("Times", "B", 9) + pdf.set_fill_color(235, 238, 242) + _pdf_cell( + pdf, + 190, + 5.5, + f"{ds.get('dataset_id')} | Dataset: {ds.get('dataset_name')}", + border=1, + align="L", + fill=True, + new_line=True, + ) pdf.set_font("Times", "", 7.5) _pdf_cell(pdf, 190, 5, f"Dataset Path: {ds.get('dataset_path')}", border=1, align="L", new_line=True) def _render_dataset_eda_page(self, pdf: _StreamlinePDF, ds: Dict[str, Any]): - self._render_dataset_header(pdf, ds, "EDA and Feature Engineering") + self._render_dataset_header(pdf, ds, "EDA and Feature Engineering", primary=True) y_start = 34.0 uv = ds.get("tables", {}).get("univariate_top10", {}) @@ -3885,23 +3985,23 @@ def _render_performance_page(self, pdf: _StreamlinePDF, ds: Dict[str, Any]): self._render_dataset_header(pdf, ds, self.performance_page_title()) perf = ds.get("performance", {}) figs = ds.get("figures", {}) - mean_cols = perf.get("mean_columns", []) - mean_rows = perf.get("mean_rows", []) + mean_cols = perf.get("model_mean_columns", perf.get("mean_columns", [])) + mean_rows = perf.get("model_mean_rows", perf.get("mean_rows", [])) mean_highlight = set( (int(r), int(c)) - for r, c in perf.get("mean_highlight_cells", perf.get("mean_bold_cells", [])) + for r, c in perf.get("model_mean_highlight_cells", perf.get("mean_highlight_cells", perf.get("mean_bold_cells", []))) ) - med_cols = perf.get("median_columns", []) - med_rows = perf.get("median_rows", []) + med_cols = perf.get("model_median_columns", perf.get("median_columns", [])) + med_rows = perf.get("model_median_rows", perf.get("median_rows", [])) med_highlight = set( (int(r), int(c)) - for r, c in perf.get("median_highlight_cells", perf.get("median_bold_cells", [])) + for r, c in perf.get("model_median_highlight_cells", perf.get("median_highlight_cells", perf.get("median_bold_cells", []))) ) y = 34.0 pdf.set_xy(10, y) pdf.set_font("Times", "B", 9) - _pdf_cell(pdf, 190, 5, "Model and Ensemble Performance (Mean +/- SD; gray = best/tied metric)", border=1, align="L", new_line=True) + _pdf_cell(pdf, 190, 5, "Core Algorithm Performance (Mean +/- SD; gray = best/tied metric)", border=1, align="L", new_line=True) y = self._render_table( pdf, x=10, @@ -3918,7 +4018,7 @@ def _render_performance_page(self, pdf: _StreamlinePDF, ds: Dict[str, Any]): y += 2 pdf.set_xy(10, y) pdf.set_font("Times", "B", 9) - _pdf_cell(pdf, 190, 5, "Model and Ensemble Performance (Median; gray = best/tied metric)", border=1, align="L", new_line=True) + _pdf_cell(pdf, 190, 5, "Core Algorithm Performance (Median; gray = best/tied metric)", border=1, align="L", new_line=True) y = self._render_table( pdf, x=10, @@ -3932,50 +4032,122 @@ def _render_performance_page(self, pdf: _StreamlinePDF, ds: Dict[str, Any]): max_first_col_width=46.0, ) - y = max(y + 2, 178) metric = ds.get("performance_distribution_metric") or "" distribution_title = f"{metric} Distribution by Algorithm" if metric else "Performance Distribution by Algorithm" - composite_path = figs.get("composite_feature_scores") distribution_path = figs.get("performance_distribution") - if composite_path and distribution_path: + if distribution_path: + y = max(y + 3, 172) + if y > 214: + self._render_dataset_header(pdf, ds, f"{self.performance_page_title()} (continued)") + y = 34.0 self._draw_image_panel( pdf, x=10, y=y, - w=94, - h=78, - title="Permutation Feature Importance (Composite)", - img_path=composite_path, - ) - self._draw_image_panel( - pdf, - x=106, - y=y, - w=94, - h=78, + w=190, + h=72, title=distribution_title, img_path=distribution_path, ) - elif composite_path: + + def _render_composite_feature_importance_page(self, pdf: _StreamlinePDF, ds: Dict[str, Any]): + figs = ds.get("figures", {}) + composite_path = figs.get("composite_feature_scores") + if not composite_path: + return + self._render_dataset_header(pdf, ds, "Composite Feature Importance") + self._draw_image_panel( + pdf, + x=10, + y=34, + w=190, + h=205, + title="Composite Permutation Feature Importance", + img_path=composite_path, + ) + + def _render_ensemble_page(self, pdf: _StreamlinePDF, ds: Dict[str, Any]): + perf = ds.get("performance", {}) + figs = ds.get("figures", {}) + mean_rows = perf.get("ensemble_mean_rows", []) + med_rows = perf.get("ensemble_median_rows", []) + has_ensemble_table = bool(mean_rows or med_rows) + has_ensemble_figures = bool(figs.get("ensembles_roc") or figs.get("ensembles_prc")) + if not has_ensemble_table and not has_ensemble_figures: + return + + self._render_dataset_header(pdf, ds, "Ensemble Performance and Evaluation") + y = 34.0 + + mean_cols = perf.get("ensemble_mean_columns", []) + mean_highlight = set( + (int(r), int(c)) + for r, c in perf.get("ensemble_mean_highlight_cells", []) + ) + if mean_cols and mean_rows: + pdf.set_xy(10, y) + pdf.set_font("Times", "B", 9) + _pdf_cell(pdf, 190, 5, "Ensemble Performance (Mean +/- SD; gray = best/tied metric)", border=1, align="L", new_line=True) + y = self._render_table( + pdf, + x=10, + y=pdf.get_y(), + width=190, + columns=mean_cols, + rows=mean_rows, + font_size=5.7, + row_h=3.5, + shade_cells=mean_highlight, + max_first_col_width=52.0, + ) + 2 + + med_cols = perf.get("ensemble_median_columns", []) + med_highlight = set( + (int(r), int(c)) + for r, c in perf.get("ensemble_median_highlight_cells", []) + ) + if med_cols and med_rows: + pdf.set_xy(10, y) + pdf.set_font("Times", "B", 9) + _pdf_cell(pdf, 190, 5, "Ensemble Performance (Median; gray = best/tied metric)", border=1, align="L", new_line=True) + y = self._render_table( + pdf, + x=10, + y=pdf.get_y(), + width=190, + columns=med_cols, + rows=med_rows, + font_size=5.7, + row_h=3.5, + shade_cells=med_highlight, + max_first_col_width=52.0, + ) + 3 + + if ds.get("task_type") == "Regression": + return + + if has_ensemble_figures: + if y > 140: + self._render_dataset_header(pdf, ds, "Ensemble Performance and Evaluation (continued)") + y = 34.0 self._draw_image_panel( pdf, x=10, y=y, - w=190, - h=78, - title="Permutation Feature Importance (Composite)", - img_path=composite_path, + w=94, + h=112, + title="ROC Summary (Ensembles)", + img_path=figs.get("ensembles_roc"), ) - else: self._draw_image_panel( pdf, - x=10, + x=106, y=y, - w=190, - h=78, - title=distribution_title, - img_path=distribution_path, + w=94, + h=112, + title="PRC Summary (Ensembles)", + img_path=figs.get("ensembles_prc"), ) def _render_evaluation_page(self, pdf: _StreamlinePDF, ds: Dict[str, Any]): @@ -4018,7 +4190,7 @@ def _render_evaluation_page(self, pdf: _StreamlinePDF, ds: Dict[str, Any]): x=10, y=34, w=94, - h=104, + h=126, title="ROC Summary (Base Models)", img_path=figs.get("models_roc"), ) @@ -4027,28 +4199,10 @@ def _render_evaluation_page(self, pdf: _StreamlinePDF, ds: Dict[str, Any]): x=106, y=34, w=94, - h=104, + h=126, title="PRC Summary (Base Models)", img_path=figs.get("models_prc"), ) - self._draw_image_panel( - pdf, - x=10, - y=142, - w=94, - h=104, - title="ROC Summary (Ensembles)", - img_path=figs.get("ensembles_roc"), - ) - self._draw_image_panel( - pdf, - x=106, - y=142, - w=94, - h=104, - title="PRC Summary (Ensembles)", - img_path=figs.get("ensembles_prc"), - ) def _render_runtime_page(self, pdf: _StreamlinePDF, ds: Dict[str, Any]): self._render_dataset_header(pdf, ds, "Runtime Summary") @@ -4179,7 +4333,9 @@ def _render_pdf(self, report_data: Dict[str, Any]): if not is_replication_report: self._render_feature_learning_page(pdf, ds) self._render_performance_page(pdf, ds) + self._render_composite_feature_importance_page(pdf, ds) self._render_evaluation_page(pdf, ds) + self._render_ensemble_page(pdf, ds) self._render_runtime_page(pdf, ds) dc = report_data.get("dataset_comparisons", {}) diff --git a/streamline/p1_data_process/data_process.py b/streamline/p1_data_process/data_process.py index cfdb3ace..f829e63c 100644 --- a/streamline/p1_data_process/data_process.py +++ b/streamline/p1_data_process/data_process.py @@ -620,8 +620,14 @@ def counts_summary(self, total_missing=None, save=True, replicate=False, initial df_value_counts = pd.DataFrame(class_counts).reset_index() df_value_counts.columns = ['Top Occurring Values', 'Counts'] class_counts.to_csv(os.path.join(out_dir, 'ClassCounts.csv'), header=['Count'], index_label='Label') - logging.info("Skewness: %s", str(skew(self.data[self.outcome_label]))) - logging.info("Kurtosis: %s", str(kurtosis(self.data[self.outcome_label]))) + outcome_values = pd.to_numeric(self.data[self.outcome_label], errors='coerce') + outcome_values = outcome_values.replace([np.inf, -np.inf], np.nan).dropna().to_numpy(dtype=float) + if outcome_values.size: + logging.info("Skewness: %s", str(skew(outcome_values))) + logging.info("Kurtosis: %s", str(kurtosis(outcome_values))) + else: + logging.info("Skewness: not available (no numeric outcome values)") + logging.info("Kurtosis: not available (no numeric outcome values)") if not replicate: logging.info("Categorical: %s", self.categorical_features) diff --git a/streamline/p8_summary_statistics/statistics.py b/streamline/p8_summary_statistics/statistics.py index cde5aa20..f98fc110 100644 --- a/streamline/p8_summary_statistics/statistics.py +++ b/streamline/p8_summary_statistics/statistics.py @@ -706,10 +706,12 @@ def fi_stats(self, metric_dict, ave_or_median='median'): if self.outcome_type == "Continuous" and self.metric_weight == "Explained Variance": fi_weight_mode = "explained_variance" + skill_score_baseline = self.feature_importance_skill_baseline() weighted_lists, weights = weight_fi( med_metric_list=med_metric_list, top_fi_med_norm_list=top_fi_med_norm_list, weight_mode=fi_weight_mode, + skill_score_baseline=skill_score_baseline, ) # Generate Normalized and Weighted Composite FI plot @@ -754,6 +756,61 @@ def fi_stats(self, metric_dict, ave_or_median='median'): # all_feature_list_to_viz, 'Norm_Frac_Weight', # 'Normalized, Fractionated, and Weighted Feature Importance') + def feature_importance_skill_baseline(self) -> float: + """ + Return the no-skill baseline used for balanced-accuracy FI weighting. + + Binary balanced accuracy has a 0.5 no-skill baseline. For multiclass, + balanced accuracy should be compared against 1 / number_of_classes. + """ + if self.outcome_type != "Multiclass": + return 0.5 + + labels = set() + + def label_key(value): + if pd.isna(value): + return None + text = str(value).strip() + if text == "": + return None + try: + return str(float(text)) + except (TypeError, ValueError): + return text + + class_counts_path = Path(self.full_path) / "exploratory" / "ClassCounts.csv" + if class_counts_path.exists(): + try: + class_counts = pd.read_csv(class_counts_path) + if not class_counts.empty: + label_col = class_counts.columns[0] + for value in class_counts[label_col].dropna().tolist(): + key = label_key(value) + if key is not None: + labels.add(key) + except Exception as exc: + logging.warning("Could not read ClassCounts.csv for FI weighting baseline: %r", exc) + + cv_dir = Path(self.full_path) / "CVDatasets" + for path in sorted(cv_dir.glob(f"{self.data_name}_CV_*_*.csv")): + try: + df = pd.read_csv(path, usecols=[self.outcome_label]) + except Exception: + continue + for value in df[self.outcome_label].dropna().tolist(): + key = label_key(value) + if key is not None: + labels.add(key) + + if len(labels) >= 2: + return 1.0 / float(len(labels)) + + logging.warning( + "Could not infer multiclass class count for FI weighting baseline; falling back to 0.5" + ) + return 0.5 + def preparation(self): """ Creates directory for all results files, decodes included ML modeling diff --git a/streamline/p8_summary_statistics/utils/fi_core.py b/streamline/p8_summary_statistics/utils/fi_core.py index 31562c56..1784172e 100644 --- a/streamline/p8_summary_statistics/utils/fi_core.py +++ b/streamline/p8_summary_statistics/utils/fi_core.py @@ -231,13 +231,16 @@ def weight_fi( med_metric_list: List[float], top_fi_med_norm_list: List[List[float]], weight_mode: str = "balanced_accuracy", + skill_score_baseline: float = 0.5, ) -> Tuple[List[List[float]], List[float]]: """ Weight normalized FI scores by algorithm performance. Supported modes: - - balanced_accuracy: any metric <= 0.5 is treated as 0; remaining values - are linearly scaled from [0.5, 1] -> [0, 1]. + - balanced_accuracy: any metric <= skill_score_baseline is treated as 0; + remaining values are linearly scaled from [skill_score_baseline, 1] -> [0, 1]. + For binary classification this baseline is 0.5; for multiclass it should + be 1 / number_of_classes. - explained_variance: any metric <= 0 is treated as 0; positive values are used directly and clipped to [0, 1]. """ @@ -254,16 +257,13 @@ def weight_fi( else: weights.append(v) else: - # Legacy/default behavior for balanced_accuracy-style weighting. - for i, v in enumerate(metrics): - if v <= 0.5: - metrics[i] = 0.0 - + # Balanced-accuracy-style weighting above the no-skill baseline. + baseline = min(max(float(skill_score_baseline), 0.0), 0.999999) for v in metrics: - if v == 0: + if v <= baseline: weights.append(0.0) else: - weights.append((v - 0.5) / 0.5) + weights.append(min(max((v - baseline) / (1.0 - baseline), 0.0), 1.0)) # Weight normalized FI weighted_lists: List[List[float]] = []