Genetic Matcher

The GeneticMatcher can be used to optimize any function of the baseline covariates, both linear and non-linear. In this demo notebook, we show how to call the matcher in the PyBalance library, including an example of a non-linear balance function.

[1]:
import logging
logging.basicConfig(
    format="%(levelname)-4s [%(filename)s:%(lineno)d] %(message)s",
    level='INFO',
)

from pybalance.sim import generate_toy_dataset
from pybalance.utils import (
    BetaBalance,
    BetaSquaredBalance,
    BetaXBalance,
    BetaMaxBalance,
    GammaBalance,
    GammaSquaredBalance,
    GammaXBalance,
    GammaXTreeBalance,
    MatchingData
)
from pybalance.genetic import GeneticMatcher, get_global_defaults
from pybalance.visualization import (
    plot_numeric_features,
    plot_categoric_features,
    plot_binary_features,
    plot_per_feature_loss,
)

time_limit = 120
[2]:
m = generate_toy_dataset()
m
[2]:
Headers Numeric:
['age', 'height', 'weight']

Headers Categoric:
['gender', 'haircolor', 'country', 'binary_0', 'binary_1', 'binary_2', 'binary_3']

Populations
['pool', 'target']
age height weight gender haircolor country population binary_0 binary_1 binary_2 binary_3 patient_id
0 62.511573 190.229250 105.165097 0.0 2 3 pool 0 0 0 0 0
1 68.505065 161.121236 95.001474 0.0 1 1 pool 1 0 1 0 1
2 50.071384 162.325356 84.290576 1.0 0 5 pool 0 0 1 1 2
3 44.423692 150.948096 82.031381 1.0 2 2 pool 0 0 0 1 3
4 41.695052 132.952651 54.857540 0.0 1 3 pool 0 0 1 1 4
... ... ... ... ... ... ... ... ... ... ... ... ...
995 21.474205 168.602546 70.342128 0.0 2 5 target 0 0 0 1 10995
996 40.643320 188.188724 61.611744 0.0 2 4 target 1 0 0 1 10996
997 29.472765 161.408162 57.214095 0.0 0 1 target 0 1 1 1 10997
998 41.291949 150.968833 91.270798 0.0 0 3 target 0 0 0 0 10998
999 67.530294 155.124741 56.196505 1.0 0 1 target 1 0 0 0 10999

11000 rows × 12 columns

Optimize Beta (Mean Absolute SMD)

[3]:
objective = beta = BetaBalance(m)
matcher = matcher_beta = GeneticMatcher(
    matching_data = m,
    objective = objective,
    log_every = 1000,
    n_generations = 5000,
    time_limit = time_limit
)
matcher.get_params()
INFO [matcher.py:125] cpu
[3]:
{'objective': 'beta',
 'candidate_population_size': 1000,
 'n_candidate_populations': 1024,
 'n_keep_best': 256,
 'n_voting_populations': 256,
 'n_mutation': 256,
 'n_generations': 5000,
 'n_iter_no_change': 100,
 'time_limit': 120,
 'max_batch_size_gb': 2,
 'seed': 1234,
 'verbose': True,
 'log_every': 1000,
 'initialization': {'benchmarks': {'propensity': 'include'},
  'sampling': {'propensity': 1.0, 'uniform': 1.0}}}
[4]:
matcher_beta.match()
INFO [initialization.py:31] Optimizing balance with genetic algorithm ...
INFO [initialization.py:32] Initial balance scores:
INFO [initialization.py:37]     beta:   0.233
INFO [initialization.py:38] Initializing candidate populations ...
INFO [initialization.py:86] Computing PROPENSITY 1-1 matching method ...
INFO [matcher.py:179] Training model SGDClassifier (iter 1/50, 0.000 min) ...
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: SGDClassifier
INFO [matcher.py:142]   * alpha: 1.5074398973827774
INFO [matcher.py:142]   * class_weight: None
INFO [matcher.py:142]   * early_stopping: True
INFO [matcher.py:142]   * fit_intercept: True
INFO [matcher.py:142]   * loss: log_loss
INFO [matcher.py:142]   * max_iter: 1500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:143]   Score (beta): 0.0525
INFO [matcher.py:144]   Solution time: 0.001 min
INFO [matcher.py:179] Training model LogisticRegression (iter 2/50, 0.001 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 0.05835496346821344
INFO [matcher.py:142]   * fit_intercept: True
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (beta): 0.0291
INFO [matcher.py:144]   Solution time: 0.003 min
INFO [matcher.py:179] Training model LogisticRegression (iter 3/50, 0.003 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 4/50, 0.016 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 5/50, 0.017 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 6/50, 0.028 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 7/50, 0.029 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 8/50, 0.030 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 9/50, 0.031 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 10/50, 0.039 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 11/50, 0.040 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 12/50, 0.042 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 13/50, 0.043 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 14/50, 0.043 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 15/50, 0.045 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 16/50, 0.045 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 17/50, 0.046 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 18/50, 0.047 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 2.3905570899706423
INFO [matcher.py:142]   * fit_intercept: True
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (beta): 0.0289
INFO [matcher.py:144]   Solution time: 0.049 min
INFO [matcher.py:179] Training model LogisticRegression (iter 19/50, 0.049 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 20/50, 0.055 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 21/50, 0.056 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 22/50, 0.057 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 23/50, 0.058 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 24/50, 0.059 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 25/50, 0.060 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 26/50, 0.061 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 0.2699411413616818
INFO [matcher.py:142]   * fit_intercept: True
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l1
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (beta): 0.0288
INFO [matcher.py:144]   Solution time: 0.062 min
INFO [matcher.py:179] Training model LogisticRegression (iter 27/50, 0.062 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 28/50, 0.063 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 29/50, 0.073 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 30/50, 0.074 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 31/50, 0.076 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 32/50, 0.077 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 33/50, 0.078 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 34/50, 0.079 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 35/50, 0.080 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 36/50, 0.093 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 37/50, 0.096 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 0.5868985298319505
INFO [matcher.py:142]   * fit_intercept: False
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l1
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (beta): 0.0249
INFO [matcher.py:144]   Solution time: 0.109 min
INFO [matcher.py:179] Training model SGDClassifier (iter 38/50, 0.109 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 39/50, 0.110 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 40/50, 0.111 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 41/50, 0.112 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 42/50, 0.114 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 43/50, 0.120 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 44/50, 0.133 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 45/50, 0.134 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 46/50, 0.135 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 47/50, 0.148 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 48/50, 0.161 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 49/50, 0.162 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 50/50, 0.163 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 0.5868985298319505
INFO [matcher.py:142]   * fit_intercept: False
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l1
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (beta): 0.0249
INFO [matcher.py:144]   Solution time: 0.109 min
INFO [initialization.py:66]     beta:   0.025
INFO [initialization.py:71]     Included in initial population.

INFO [initialization.py:132] Sampling 512 candidate populations according to PROPENSITY distribution ...

INFO [initialization.py:132] Sampling 511 candidate populations according to UNIFORM distribution ...

INFO [logger.py:34] Generation 0
INFO [logger.py:35]     remaining patients: 10000
INFO [logger.py:36]     elapsed time: 0.18 min
INFO [logger.py:45]     best beta: 0.02494      worst beta: 0.25118
INFO [logger.py:34] Generation 1000
INFO [logger.py:35]     remaining patients: 7482
INFO [logger.py:36]     elapsed time: 1.63 min
INFO [logger.py:45]     best beta: 0.01146      worst beta: 0.24292
INFO [matcher.py:211] Time limit exceeded. Stopping.
[4]:
Headers Numeric:
['age', 'height', 'weight']

Headers Categoric:
['gender', 'haircolor', 'country', 'binary_0', 'binary_1', 'binary_2', 'binary_3']

Populations
['pool', 'target']
age height weight gender haircolor country population binary_0 binary_1 binary_2 binary_3 patient_id
0 55.261578 139.396134 94.438359 0.0 2 2 target 0 0 1 1 10000
1 63.113091 165.563337 67.433016 1.0 2 2 target 0 1 1 0 10001
2 58.232216 160.859857 71.915385 1.0 0 2 target 0 0 0 0 10002
3 58.996941 140.357415 115.606615 1.0 0 3 target 1 1 0 0 10003
4 36.850195 189.983706 53.000581 0.0 2 5 target 0 0 0 0 10004
... ... ... ... ... ... ... ... ... ... ... ... ...
5044 42.548928 129.729442 94.445375 1.0 1 2 pool 0 0 0 0 5044
1144 29.400226 167.737236 76.118095 1.0 0 4 pool 0 1 0 1 1144
5314 50.104985 163.663484 85.785445 1.0 2 4 pool 0 1 0 0 5314
2174 54.372402 149.801277 92.946485 0.0 1 2 pool 0 0 0 0 2174
8610 72.912497 185.908237 96.553338 0.0 1 4 pool 1 1 1 1 8610

2000 rows × 12 columns

[5]:
%matplotlib inline

match = matcher_beta.get_best_match()
m_data = m.copy().get_population('pool')
m_data.loc[:, 'population'] = m_data['population'] + ' (prematch)'
match.append(m_data)
fig = plot_per_feature_loss(match, beta, 'target', debin=False)
fig = plot_numeric_features(match, hue_order=['pool (prematch)', 'pool', 'target', ])
fig = plot_categoric_features(match,  hue_order=['pool (prematch)', 'pool', 'target'])
../_images/demos_matching_04_genetic_6_0.png
../_images/demos_matching_04_genetic_6_1.png
../_images/demos_matching_04_genetic_6_2.png

Optimize Beta^2

[6]:
objective = beta2 = BetaSquaredBalance(m)
matcher = matcher_beta2 = GeneticMatcher(
    matching_data = m,
    objective = objective,
    log_every = 1000,
    n_generations = 5000,
    time_limit = time_limit
)
matcher.get_params()
INFO [matcher.py:125] cpu
[6]:
{'objective': 'beta_squared',
 'candidate_population_size': 1000,
 'n_candidate_populations': 1024,
 'n_keep_best': 256,
 'n_voting_populations': 256,
 'n_mutation': 256,
 'n_generations': 5000,
 'n_iter_no_change': 100,
 'time_limit': 120,
 'max_batch_size_gb': 2,
 'seed': 1234,
 'verbose': True,
 'log_every': 1000,
 'initialization': {'benchmarks': {'propensity': 'include'},
  'sampling': {'propensity': 1.0, 'uniform': 1.0}}}
[7]:
match = matcher.match()
INFO [initialization.py:31] Optimizing balance with genetic algorithm ...
INFO [initialization.py:32] Initial balance scores:
INFO [initialization.py:37]     beta_squared:   0.263
INFO [initialization.py:38] Initializing candidate populations ...
INFO [initialization.py:86] Computing PROPENSITY 1-1 matching method ...
INFO [matcher.py:179] Training model SGDClassifier (iter 1/50, 0.000 min) ...
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: SGDClassifier
INFO [matcher.py:142]   * alpha: 1.5074398973827774
INFO [matcher.py:142]   * class_weight: None
INFO [matcher.py:142]   * early_stopping: True
INFO [matcher.py:142]   * fit_intercept: True
INFO [matcher.py:142]   * loss: log_loss
INFO [matcher.py:142]   * max_iter: 1500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:143]   Score (beta_squared): 0.0603
INFO [matcher.py:144]   Solution time: 0.001 min
INFO [matcher.py:179] Training model LogisticRegression (iter 2/50, 0.001 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 0.05835496346821344
INFO [matcher.py:142]   * fit_intercept: True
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (beta_squared): 0.0374
INFO [matcher.py:144]   Solution time: 0.002 min
INFO [matcher.py:179] Training model LogisticRegression (iter 3/50, 0.002 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 16.16555309446666
INFO [matcher.py:142]   * fit_intercept: False
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l1
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (beta_squared): 0.0357
INFO [matcher.py:144]   Solution time: 0.015 min
INFO [matcher.py:179] Training model SGDClassifier (iter 4/50, 0.015 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 5/50, 0.016 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 54.02072493419677
INFO [matcher.py:142]   * fit_intercept: False
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (beta_squared): 0.0347
INFO [matcher.py:144]   Solution time: 0.027 min
INFO [matcher.py:179] Training model LogisticRegression (iter 6/50, 0.027 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 7/50, 0.028 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 8/50, 0.029 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 9/50, 0.030 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 10/50, 0.038 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 11/50, 0.039 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 12/50, 0.041 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 13/50, 0.041 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 14/50, 0.042 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 15/50, 0.043 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 16/50, 0.044 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 17/50, 0.045 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 18/50, 0.046 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 19/50, 0.048 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 20/50, 0.053 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 21/50, 0.054 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 22/50, 0.055 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 23/50, 0.056 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 24/50, 0.057 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 25/50, 0.059 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 26/50, 0.059 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 27/50, 0.061 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 28/50, 0.062 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 29/50, 0.072 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 30/50, 0.073 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 31/50, 0.074 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 32/50, 0.075 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 33/50, 0.076 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 34/50, 0.077 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 35/50, 0.078 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 36/50, 0.091 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 37/50, 0.094 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 0.5868985298319505
INFO [matcher.py:142]   * fit_intercept: False
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l1
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (beta_squared): 0.0311
INFO [matcher.py:144]   Solution time: 0.107 min
INFO [matcher.py:179] Training model SGDClassifier (iter 38/50, 0.107 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 39/50, 0.108 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 40/50, 0.109 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 41/50, 0.110 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 42/50, 0.111 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 43/50, 0.118 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 44/50, 0.130 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 45/50, 0.131 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 46/50, 0.132 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 47/50, 0.145 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 48/50, 0.158 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 49/50, 0.159 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 50/50, 0.160 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 0.5868985298319505
INFO [matcher.py:142]   * fit_intercept: False
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l1
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (beta_squared): 0.0311
INFO [matcher.py:144]   Solution time: 0.107 min
INFO [initialization.py:66]     beta_squared:   0.031
INFO [initialization.py:71]     Included in initial population.

INFO [initialization.py:132] Sampling 512 candidate populations according to PROPENSITY distribution ...

INFO [initialization.py:132] Sampling 511 candidate populations according to UNIFORM distribution ...

INFO [logger.py:34] Generation 0
INFO [logger.py:35]     remaining patients: 10000
INFO [logger.py:36]     elapsed time: 0.18 min
INFO [logger.py:45]     best beta_squared: 0.03106      worst beta_squared: 0.28163
INFO [matcher.py:203] No improvement in last 100 iterations. Stopping.
[8]:
%matplotlib inline

match = matcher_beta2.get_best_match()
m_data = m.copy().get_population('pool')
m_data.loc[:, 'population'] = m_data['population'] + ' (prematch)'
match.append(m_data)
fig = plot_per_feature_loss(match, beta, 'target', debin=False)
fig = plot_numeric_features(match, hue_order=['pool (prematch)', 'pool', 'target', ])
fig = plot_categoric_features(match,  hue_order=['pool (prematch)', 'pool', 'target'])
../_images/demos_matching_04_genetic_10_0.png
../_images/demos_matching_04_genetic_10_1.png
../_images/demos_matching_04_genetic_10_2.png

Optimize Gamma (Area Between CDFs)

[9]:
objective = gamma = GammaBalance(m, feature_weights={'age':2})
matcher = matcher_gamma = GeneticMatcher(
    matching_data = m,
    objective = objective,
    log_every = 1000,
    n_generations = 5000,
    time_limit = time_limit
)
matcher.get_params()
INFO [preprocess.py:340] Discretized age with bins [18.05, 27.54, 37.04, 46.53, 56.02, 65.51, 75.0].
INFO [preprocess.py:340] Discretized height with bins [125.01, 136.68, 148.34, 160.01, 171.67, 183.34, 195.0].
INFO [preprocess.py:340] Discretized weight with bins [50.0, 61.67, 73.33, 85.0, 96.66, 108.33, 120.0].
INFO [matcher.py:125] cpu
[9]:
{'objective': 'gamma',
 'candidate_population_size': 1000,
 'n_candidate_populations': 1024,
 'n_keep_best': 256,
 'n_voting_populations': 256,
 'n_mutation': 256,
 'n_generations': 5000,
 'n_iter_no_change': 100,
 'time_limit': 120,
 'max_batch_size_gb': 2,
 'seed': 1234,
 'verbose': True,
 'log_every': 1000,
 'initialization': {'benchmarks': {'propensity': 'include'},
  'sampling': {'propensity': 1.0, 'uniform': 1.0}}}
[10]:
match = matcher.match()
INFO [initialization.py:31] Optimizing balance with genetic algorithm ...
INFO [initialization.py:32] Initial balance scores:
INFO [initialization.py:37]     gamma:  0.217
INFO [initialization.py:38] Initializing candidate populations ...
INFO [initialization.py:86] Computing PROPENSITY 1-1 matching method ...
INFO [matcher.py:179] Training model SGDClassifier (iter 1/50, 0.000 min) ...
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: SGDClassifier
INFO [matcher.py:142]   * alpha: 1.5074398973827774
INFO [matcher.py:142]   * class_weight: None
INFO [matcher.py:142]   * early_stopping: True
INFO [matcher.py:142]   * fit_intercept: True
INFO [matcher.py:142]   * loss: log_loss
INFO [matcher.py:142]   * max_iter: 1500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:143]   Score (gamma): 0.1083
INFO [matcher.py:144]   Solution time: 0.002 min
INFO [matcher.py:179] Training model LogisticRegression (iter 2/50, 0.002 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 0.05835496346821344
INFO [matcher.py:142]   * fit_intercept: True
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (gamma): 0.0391
INFO [matcher.py:144]   Solution time: 0.003 min
INFO [matcher.py:179] Training model LogisticRegression (iter 3/50, 0.003 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 16.16555309446666
INFO [matcher.py:142]   * fit_intercept: False
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l1
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (gamma): 0.0376
INFO [matcher.py:144]   Solution time: 0.023 min
INFO [matcher.py:179] Training model SGDClassifier (iter 4/50, 0.024 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 5/50, 0.025 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 54.02072493419677
INFO [matcher.py:142]   * fit_intercept: False
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (gamma): 0.0337
INFO [matcher.py:144]   Solution time: 0.040 min
INFO [matcher.py:179] Training model LogisticRegression (iter 6/50, 0.040 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 7/50, 0.042 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 8/50, 0.043 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 9/50, 0.045 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 13.179630432958701
INFO [matcher.py:142]   * fit_intercept: False
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (gamma): 0.0269
INFO [matcher.py:144]   Solution time: 0.060 min
INFO [matcher.py:179] Training model LogisticRegression (iter 10/50, 0.060 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 11/50, 0.062 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 12/50, 0.063 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 13/50, 0.065 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 14/50, 0.066 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 15/50, 0.067 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 16/50, 0.068 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 17/50, 0.069 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 18/50, 0.071 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 19/50, 0.074 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 20/50, 0.089 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 21/50, 0.090 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 22/50, 0.092 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 23/50, 0.093 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 24/50, 0.094 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 25/50, 0.096 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 26/50, 0.098 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 27/50, 0.100 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 28/50, 0.102 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 29/50, 0.121 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 30/50, 0.122 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 31/50, 0.125 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 32/50, 0.126 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 33/50, 0.127 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 34/50, 0.129 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 35/50, 0.130 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 36/50, 0.150 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 37/50, 0.157 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 38/50, 0.177 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 39/50, 0.178 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 40/50, 0.180 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 41/50, 0.181 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 42/50, 0.183 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 43/50, 0.198 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 44/50, 0.218 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 45/50, 0.220 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 46/50, 0.221 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model LogisticRegression (iter 47/50, 0.240 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1407: UserWarning: Inconsistent values: penalty=l1 with l1_ratio=0.0. penalty is deprecated. Please use l1_ratio only.
  warnings.warn(
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_sag.py:348: ConvergenceWarning: The max_iter was reached which means the coef_ did not converge
  warnings.warn(
INFO [matcher.py:179] Training model SGDClassifier (iter 48/50, 0.260 min) ...
INFO [matcher.py:179] Training model SGDClassifier (iter 49/50, 0.262 min) ...
INFO [matcher.py:179] Training model LogisticRegression (iter 50/50, 0.263 min) ...
/Users/gmema/src/matching-fork/venv/lib/python3.12/site-packages/sklearn/linear_model/_logistic.py:1381: FutureWarning: 'penalty' was deprecated in version 1.8 and will be removed in 1.10. To avoid this warning, leave 'penalty' set to its default value and use 'l1_ratio' or 'C' instead. Use l1_ratio=0 instead of penalty='l2', l1_ratio=1 instead of penalty='l1', l1_ratio set to a float between 0 and 1 instead of penalty='elasticnet', and C=np.inf instead of penalty=None.
  warnings.warn(
INFO [matcher.py:139] Best propensity score match found:
INFO [matcher.py:140]   Model: LogisticRegression
INFO [matcher.py:142]   * C: 13.179630432958701
INFO [matcher.py:142]   * fit_intercept: False
INFO [matcher.py:142]   * max_iter: 500
INFO [matcher.py:142]   * penalty: l2
INFO [matcher.py:142]   * solver: saga
INFO [matcher.py:143]   Score (gamma): 0.0269
INFO [matcher.py:144]   Solution time: 0.060 min
INFO [initialization.py:66]     gamma:  0.027
INFO [initialization.py:71]     Included in initial population.

INFO [initialization.py:132] Sampling 512 candidate populations according to PROPENSITY distribution ...

INFO [initialization.py:132] Sampling 511 candidate populations according to UNIFORM distribution ...

INFO [logger.py:34] Generation 0
INFO [logger.py:35]     remaining patients: 10000
INFO [logger.py:36]     elapsed time: 0.28 min
INFO [logger.py:45]     best gamma: 0.02688     worst gamma: 0.23595
INFO [logger.py:34] Generation 1000
INFO [logger.py:35]     remaining patients: 7355
INFO [logger.py:36]     elapsed time: 1.80 min
INFO [logger.py:45]     best gamma: 0.01164     worst gamma: 0.22939
INFO [matcher.py:211] Time limit exceeded. Stopping.
[11]:
%matplotlib inline

match = matcher.get_best_match()
m_data = m.copy().get_population('pool')
m_data.loc[:, 'population'] = m_data['population'] + ' (prematch)'
match.append(m_data)
fig = plot_per_feature_loss(match, gamma, 'target', debin=False)
fig = plot_numeric_features(match, hue_order=['pool (prematch)', 'pool', 'target', ])
fig = plot_categoric_features(match,  hue_order=['pool (prematch)', 'pool', 'target'])
../_images/demos_matching_04_genetic_14_0.png
../_images/demos_matching_04_genetic_14_1.png
../_images/demos_matching_04_genetic_14_2.png