Skip to content

Commit ddb8e04

Browse files
committed
Update dependencies and adapt DecisionTree rules to new format
1 parent 122d437 commit ddb8e04

3 files changed

Lines changed: 35 additions & 27 deletions

File tree

‎pyproject.toml‎

Lines changed: 8 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
11
[tool.poetry]
22
name = "pix-framework"
3-
version = "0.13.10"
3+
version = "0.13.11"
44
description = "Process Improvement Explorer Framework contains process discovery and improvement modules of the Process Improvement Explorer project."
55
authors = [
66
"David Chapela de la Campa <david.chapela.delacampa@gmail.com>",
@@ -15,16 +15,16 @@ build-backend = "poetry.core.masonry.api"
1515

1616
[tool.poetry.dependencies]
1717
python = ">=3.9, <3.12"
18-
pandas = "^2.0.1"
19-
scipy = "^1.10.1"
20-
pytz = "^2023.3"
18+
pandas = "^2.2.1"
19+
scipy = "^1.13.0"
20+
pytz = "^2024.1"
2121
networkx = "^3.1"
2222
wittgenstein = "^0.3.4"
2323
pandasql = "^0.7.3"
24-
scikit-learn = "^1.3.0"
25-
polars = "^0.18.15"
26-
pyarrow = "^12.0.1"
27-
lxml = "^4.9.3"
24+
scikit-learn = "^1.4.2"
25+
polars = "^0.20.19"
26+
pyarrow = "^15.0.2"
27+
lxml = "^5.2.1"
2828

2929
[tool.poetry.group.dev.dependencies]
3030
pytest = "^7.3.1"

‎src/pix_framework/discovery/prioritization/rules.py‎

Lines changed: 18 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -69,22 +69,25 @@ def _get_rules(data: pd.DataFrame, outcome: str) -> list:
6969
"""
7070
# Discover 5 times and get the one with more confidence
7171
best_confidence = 0
72+
best_support = 0
7273
best_rules = []
73-
for i in range(5):
74+
for i in range(3):
7475
# Train new model to extract 1 rule
7576
new_model = DecisionTreeClassifier()
7677
new_model.fit(data[[column for column in data.columns if column is not outcome]], data[outcome])
77-
best_rules = _tree_to_best_rules(new_model, [column for column in data.columns if column is not outcome])
78+
rules = _tree_to_best_rules(new_model, [column for column in data.columns if column is not outcome])
7879
# If any rule has been discovered
79-
if len(best_rules) > 0:
80+
if len(rules) > 0:
8081
# Measure confidence
81-
predictions = _predict(best_rules, data.drop([outcome], axis=1))
82+
predictions = _predict(rules, data.drop([outcome], axis=1))
8283
true_positives = [p and a for (p, a) in zip(predictions, data[outcome])]
8384
confidence = sum(true_positives) / sum(predictions)
85+
support = sum(true_positives) / len(data)
8486
# Retain if it's better than the previous one
85-
if confidence > best_confidence:
87+
if confidence > best_confidence or (confidence == best_confidence and support > best_support):
8688
best_confidence = confidence
87-
best_rules = best_rules
89+
best_support = support
90+
best_rules = rules
8891
# Return the best one, or None if no rules found in any iteration
8992
return best_rules
9093

@@ -114,7 +117,7 @@ def _tree_to_best_rules(tree, feature_names) -> list:
114117
else:
115118
# Leaf node
116119
current_impurity = tree_.impurity[current_node]
117-
current_sample_sizes = tree_.value[current_node][0] # Number of positive samples
120+
current_sample_sizes = tree_.value[current_node][0] * tree_.n_node_samples[current_node] # #PositiveSamples
118121
# If it is the best leaf node, save it
119122
if current_sample_sizes[0] < current_sample_sizes[1] and ( # Less samples with negative outcome
120123
current_impurity < best_rule["impurity"]
@@ -139,8 +142,14 @@ def _summarize_rules(rules: list) -> list:
139142
'attribute': attribute,
140143
'comparison': 'in',
141144
'value': "({},{}]".format(
142-
max([rule['value'] for rule in rules if rule['attribute'] == attribute and rule['comparison'] == ">"]),
143-
min([rule['value'] for rule in rules if rule['attribute'] == attribute and rule['comparison'] == "<="])
145+
max([
146+
rule['value'] for rule in rules
147+
if rule['attribute'] == attribute and rule['comparison'] == ">"
148+
]),
149+
min([
150+
rule['value'] for rule in rules
151+
if rule['attribute'] == attribute and rule['comparison'] == "<="
152+
])
144153
)
145154
}]
146155
else:

‎tests/pix_framework/discovery/prioritization/test_prioritization_rules.py‎

Lines changed: 9 additions & 10 deletions
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,5 @@
1-
from pathlib import Path
2-
31
import pandas as pd
2+
43
from pix_framework.discovery.prioritization.rules import (
54
_reverse_one_hot_encoding,
65
discover_prioritization_rules,
@@ -63,14 +62,14 @@ def test_discover_prioritization_rules_with_extra_attribute():
6362
prioritization_rules = discover_prioritization_rules(prioritizations, "outcome")
6463
# Assert the rules
6564
assert (
66-
sort_rules(prioritization_rules)
67-
== sort_rules(prioritization_rules)
68-
== sort_rules(
69-
[
70-
{"priority_level": 1, "rules": [[{"attribute": "loan_amount", "comparison": ">", "value": "750.0"}]]},
71-
{"priority_level": 2, "rules": [[{"attribute": "loan_amount", "comparison": ">", "value": "300.0"}]]},
72-
]
73-
)
65+
sort_rules(prioritization_rules)
66+
== sort_rules(prioritization_rules)
67+
== sort_rules(
68+
[
69+
{"priority_level": 1, "rules": [[{"attribute": "loan_amount", "comparison": ">", "value": "750.0"}]]},
70+
{"priority_level": 2, "rules": [[{"attribute": "loan_amount", "comparison": ">", "value": "300.0"}]]},
71+
]
72+
)
7473
)
7574

7675

0 commit comments

Comments
 (0)