(model)
| 869 | |
| 870 | |
| 871 | def run_validation_weights(model): |
| 872 | from sklearn.datasets import make_hastie_10_2 |
| 873 | |
| 874 | # prepare training and test data |
| 875 | X, y = make_hastie_10_2(n_samples=2000, random_state=42) |
| 876 | labels, y = np.unique(y, return_inverse=True) |
| 877 | X_train, X_test = X[:1600], X[1600:] |
| 878 | y_train, y_test = y[:1600], y[1600:] |
| 879 | |
| 880 | # instantiate model |
| 881 | param_dist = { |
| 882 | "objective": "binary:logistic", |
| 883 | "n_estimators": 2, |
| 884 | "random_state": 123, |
| 885 | } |
| 886 | clf = model(**param_dist) |
| 887 | |
| 888 | # train it using instance weights only in the training set |
| 889 | weights_train = np.random.choice([1, 2], len(X_train)) |
| 890 | clf.set_params(eval_metric="logloss") |
| 891 | clf.fit( |
| 892 | X_train, |
| 893 | y_train, |
| 894 | sample_weight=weights_train, |
| 895 | eval_set=[(X_test, y_test)], |
| 896 | verbose=False, |
| 897 | ) |
| 898 | # evaluate logloss metric on test set *without* using weights |
| 899 | evals_result_without_weights = clf.evals_result() |
| 900 | logloss_without_weights = evals_result_without_weights["validation_0"]["logloss"] |
| 901 | |
| 902 | # now use weights for the test set |
| 903 | np.random.seed(0) |
| 904 | weights_test = np.random.choice([1, 2], len(X_test)) |
| 905 | clf.set_params(eval_metric="logloss") |
| 906 | clf.fit( |
| 907 | X_train, |
| 908 | y_train, |
| 909 | sample_weight=weights_train, |
| 910 | eval_set=[(X_test, y_test)], |
| 911 | sample_weight_eval_set=[weights_test], |
| 912 | verbose=False, |
| 913 | ) |
| 914 | evals_result_with_weights = clf.evals_result() |
| 915 | logloss_with_weights = evals_result_with_weights["validation_0"]["logloss"] |
| 916 | |
| 917 | # check that the logloss in the test set is actually different when using |
| 918 | # weights than when not using them |
| 919 | assert all((logloss_with_weights[i] != logloss_without_weights[i] for i in [0, 1])) |
| 920 | |
| 921 | with pytest.raises(ValueError): |
| 922 | # length of eval set and sample weight doesn't match. |
| 923 | clf.fit( |
| 924 | X_train, |
| 925 | y_train, |
| 926 | sample_weight=weights_train, |
| 927 | eval_set=[(X_train, y_train), (X_test, y_test)], |
| 928 | sample_weight_eval_set=[weights_train], |
no test coverage detected