Compare commits

...

25 commits

Author SHA1 Message Date
KhalimCK
d4836e02cb
Merge pull request #92 from Hestia-Homes/carbon-dev-model
Carbon dev model
2024-01-18 10:36:46 +00:00
Michael Duong
9b29e838af update requirements for dvc 2024-01-17 23:45:07 +00:00
Michael Duong
79a55ba8b5 train 600 second model on new data 2024-01-17 23:35:50 +00:00
Michael Duong
e78a4bb30e Merge branch 'carbon-dev' of github.com:Hestia-Homes/ML into carbon-dev-model 2024-01-17 23:12:26 +00:00
Michael Duong
ae53499742 add keep only non negative carbon change to carbon model 2023-12-22 09:51:57 +00:00
Github-Bot
db29bece80 Update Registry 2023-11-28 15:27:34 +00:00
Github-Bot
65335468b4 Update Registry 2023-11-28 15:26:50 +00:00
quandanrepo
53afbd26d8
Merge pull request #88 from Hestia-Homes/carbon-dev-model
Carbon dev model
2023-11-28 15:26:04 +00:00
Michael Duong
718003b3d9 Merge branch 'carbon-dev' of github.com:Hestia-Homes/ML into carbon-dev-model 2023-11-28 15:14:09 +00:00
Michael Duong
888bfc30c6 Merge branch 'master' of github.com:Hestia-Homes/ML into carbon-dev-model 2023-11-28 15:13:50 +00:00
Michael Duong
2b1e8b912b restrict dataset 2023-11-28 15:13:42 +00:00
Github-Bot
62f2f83b0a Update Registry 2023-11-27 19:22:00 +00:00
Github-Bot
03322a13e7 Update Registry 2023-11-27 19:21:22 +00:00
KhalimCK
5f3d9efa92
Merge pull request #85 from Hestia-Homes/carbon-dev-model
Carbon dev model
2023-11-27 19:20:40 +00:00
Michael Duong
f29d6af6a2 change readme 2023-11-27 19:13:23 +00:00
Michael Duong
7afc4b06b2 Merge branch 'master' of github.com:Hestia-Homes/ML into carbon-dev-model 2023-11-27 19:12:40 +00:00
Michael Duong
217fb3dca8 add inference speed check 2023-11-27 18:52:47 +00:00
Michael Duong
9a04ffde3b Merge branch 'master' of github.com:Hestia-Homes/ML into carbon-dev-model 2023-11-27 18:30:10 +00:00
Michael Duong
e6c7b2f58c Merge branch 'carbon-dev' of github.com:Hestia-Homes/ML into carbon-dev-model 2023-10-12 08:39:24 +00:00
Michael Duong
f2cc32f4b4 using good model 4000s 2023-10-12 08:38:55 +00:00
Github-Bot
2f9092f447 Update Registry 2023-10-11 15:48:52 +00:00
Github-Bot
bb2db16f61 Update Registry 2023-10-11 15:48:04 +00:00
quandanrepo
5aaebd7f44
Merge pull request #71 from Hestia-Homes/carbon-dev-model
400 second model
2023-10-11 16:47:13 +01:00
Michael Duong
680e879503 400 second model 2023-10-11 15:38:55 +00:00
Michael Duong
f4e91162ec initial model 2023-10-11 13:23:54 +00:00
9 changed files with 116 additions and 58 deletions

View file

@ -8,17 +8,25 @@
"active": true "active": true
}, },
"sap": { "sap": {
"version": "v0.1.0", "version": "v0.2.6",
"stage": { "stage": {
"dev": "v0.1.0" "dev": "v0.2.6"
}, },
"registered": true, "registered": true,
"active": true "active": true
}, },
"heat": { "heat": {
"version": "v0.0.1", "version": "v0.2.0",
"stage": { "stage": {
"dev": "v0.0.1" "dev": "v0.2.0"
},
"registered": true,
"active": true
},
"carbon": {
"version": "v0.2.0",
"stage": {
"dev": "v0.2.0"
}, },
"registered": true, "registered": true,
"active": true "active": true

View file

@ -1,3 +1,3 @@
# The generic reproducible ML-pipeline # The generic reproducible ML-pipeline!
Pipeline required to build a model to produce an output, that gets hashed via DVC Pipeline required to build a model to produce an output, that gets hashed via DVC

View file

@ -13,7 +13,7 @@ default:
output_filepath: ./data/model/allmodels/ output_filepath: ./data/model/allmodels/
problem_type: regression problem_type: regression
eval_metric: mean_squared_error #mean_absolute_error eval_metric: mean_squared_error #mean_absolute_error
time_limit: 4000 time_limit: 600
presets: medium_quality presets: medium_quality
excluded_model_types: ['KNN', 'RF'] excluded_model_types: ['KNN', 'RF']
infer_limit: 0.05 infer_limit: 0.05

View file

@ -9,15 +9,56 @@ Business Logic dict + functions
def remove_starting_columns(df): def remove_starting_columns(df):
keep_column_index = [ keep_column_index = [
False if col_name.endswith("_STARTING") else True False if col_name.endswith("_starting") else True
for col_name in list(df.columns) for col_name in list(df.columns)
] ]
keep_columns = df.columns[keep_column_index].to_list() keep_columns = df.columns[keep_column_index].to_list()
keep_columns.append("SAP_STARTING") keep_columns.append("sap_starting")
df = df[keep_columns] df = df[keep_columns]
return df return df
def keep_negative_heat_change(df):
df = df[df["heat_demand_change"] < 0]
return df
def keep_non_negative_carbon_ending(df):
df = df[df["carbon_ending"] > 0]
return df
def keep_negative_carbon_change(df):
df = df[df["carbon_change"] < 0]
return df
# TODO: Move to ETL pipeline
def remove_unreasonable_habitable_rooms(df):
"""
Assumption is that proportion of floor area to habitable rooms should be at least 6.5m2
"""
minimum_room_size_index = (
df["total_floor_area_ending"] / df["number_habitable_rooms"] >= 6.5
)
df = df[minimum_room_size_index]
return df
def remove_top_1_percent_heat_demand(df):
# threshold_value = df.describe(percentiles=[0.99])['HEAT_DEMAND_STARTING']['99%']
threshold_value = 860
df = df[df["heat_demand_starting"] < threshold_value]
return df
def remove_top_1_percent_carbon(df):
# threshold_value = df.describe(percentiles=[0.99])['CARBON_STARTING']['99%']
threshold_value = 18
df = df[df["carbon_starting"] < threshold_value]
return df
# def keep_ending_columns(df): # def keep_ending_columns(df):
# ending_column_index = [ col_name.endswith("_ENDING") for col_name in list(df.columns)] # ending_column_index = [ col_name.endswith("_ENDING") for col_name in list(df.columns)]
# keep_columns = df.columns[ending_column_index].to_list() # keep_columns = df.columns[ending_column_index].to_list()
@ -27,6 +68,12 @@ def remove_starting_columns(df):
# return df # return df
business_logic = { business_logic = {
"remove_unreasonable_habitable_rooms": remove_unreasonable_habitable_rooms,
"keep_negative_heat_change": keep_negative_heat_change,
"keep_negative_carbon_change": keep_negative_carbon_change,
"remove_top_1_percent_heat_demand": remove_top_1_percent_heat_demand,
"remove_top_1_percent_carbon": remove_top_1_percent_carbon,
"keep_non_negative_carbon_ending": keep_non_negative_carbon_ending
# "remove_starting_columns": remove_starting_columns # "remove_starting_columns": remove_starting_columns
# "keep_ENDING_COLUMNS": keep_ending_columns # "keep_ENDING_COLUMNS": keep_ending_columns
} }

View file

@ -5,17 +5,18 @@ import pandas as pd
def clip_predictions_to_minimum_value( def clip_predictions_to_minimum_value(
data: pd.DataFrame, predictions: pd.Series, minimum_value: int = 1 data: pd.DataFrame,
predictions: pd.Series,
) -> pd.Series: ) -> pd.Series:
series_name = predictions.name series_name = predictions.name
predictions.name = "predictions" predictions.name = "predictions"
predictions_df = pd.concat([data, predictions], axis=1) predictions_df = pd.concat([data, predictions], axis=1)
# We expect all prediction to be atleast one point improvement # We expect all prediction to be atleast one point improvement
replace_index = predictions_df["SAP_STARTING"] + 1 > predictions_df["predictions"] replace_index = predictions_df["predictions"] > predictions_df["carbon_starting"]
predictions_df.loc[replace_index, "predictions"] = ( predictions_df.loc[replace_index, "predictions"] = predictions_df.loc[
predictions_df.loc[replace_index, "SAP_STARTING"] + minimum_value replace_index, "carbon_starting"
) ]
predictions_new = predictions_df["predictions"] predictions_new = predictions_df["predictions"]
predictions_new.name = series_name predictions_new.name = series_name

View file

@ -21,7 +21,7 @@ default:
# data_filepath: s3://retrofit-data-dev/sap_change_model/dataset_with_differencing.parquet # data_filepath: s3://retrofit-data-dev/sap_change_model/dataset_with_differencing.parquet
# data_filepath: s3://retrofit-data-dev/sap_change_model/floor_area_clean_test.parquet # data_filepath: s3://retrofit-data-dev/sap_change_model/floor_area_clean_test.parquet
# data_filepath: s3://retrofit-data-dev/sap_change_model/dataset_without_differencing.parquet # data_filepath: s3://retrofit-data-dev/sap_change_model/dataset_without_differencing.parquet
data_filepath: s3://retrofit-data-dev/sap_change_model/dataset_test.parquet data_filepath: s3://retrofit-data-dev/sap_change_model/dataset.parquet
train_proportion: 0.9 train_proportion: 0.9
output_train_filepath: ./data/prepared_data/train.parquet output_train_filepath: ./data/prepared_data/train.parquet
output_test_filepath: ./data/prepared_data/test.parquet output_test_filepath: ./data/prepared_data/test.parquet
@ -31,9 +31,9 @@ default:
feature_processor_config: feature_processor_config:
subsample_amount: null subsample_amount: null
subsample_seed: 0 subsample_seed: 0
target: SAP_ENDING target: carbon_ending
identifier_columns: ["UPRN"] identifier_columns: ["uprn"]
drop_columns: ["HEAT_DEMAND_CHANGE", "CARBON_CHANGE", "RDSAP_CHANGE", "HEAT_DEMAND_ENDING", "CARBON_ENDING"] drop_columns: ["heat_demand_change", "carbon_change", "rdsap_change", "heat_demand_ending", "sap_ending"]
# retain_features: ["SAP_STARTING", "TOTAL_FLOOR_AREA_DIFF"] # retain_features: ["SAP_STARTING", "TOTAL_FLOOR_AREA_DIFF"]
retain_features: null retain_features: null

View file

@ -5,20 +5,20 @@ stages:
deps: deps:
- path: 1_prepare_data.py - path: 1_prepare_data.py
hash: md5 hash: md5
md5: c9f030df733e318b80d1fa91b7732f79 md5: 896d3d88a4a9f68d174efe71dc089517
size: 5132 size: 4222
params: params:
configs/settings.yaml: configs/settings.yaml:
default.feature_processor.feature_processor_config.drop_columns: default.feature_processor.feature_processor_config.drop_columns:
- HEAT_DEMAND_CHANGE - heat_demand_change
- CARBON_CHANGE - carbon_change
- RDSAP_CHANGE - rdsap_change
- HEAT_DEMAND_ENDING - heat_demand_ending
- CARBON_ENDING - sap_ending
default.feature_processor.feature_processor_config.retain_features: default.feature_processor.feature_processor_config.retain_features:
default.feature_processor.feature_processor_config.subsample_amount: default.feature_processor.feature_processor_config.subsample_amount:
default.feature_processor.feature_processor_config.subsample_seed: 0 default.feature_processor.feature_processor_config.subsample_seed: 0
default.feature_processor.feature_processor_config.target: SAP_ENDING default.feature_processor.feature_processor_config.target: carbon_ending
default.feature_processor.feature_processor_type: dataframe default.feature_processor.feature_processor_type: dataframe
default.prepare_data.data_filepath: s3://retrofit-data-dev/sap_change_model/dataset.parquet default.prepare_data.data_filepath: s3://retrofit-data-dev/sap_change_model/dataset.parquet
default.prepare_data.input_dataclient_type: aws-s3 default.prepare_data.input_dataclient_type: aws-s3
@ -29,20 +29,20 @@ stages:
outs: outs:
- path: data/prepared_data/ - path: data/prepared_data/
hash: md5 hash: md5
md5: 9ce5c45722da7fc40491b5a4d00daf9e.dir md5: 70d79ba4a6f0648439dc55031c944d47.dir
size: 33881619 size: 32673907
nfiles: 2 nfiles: 2
build_model: build_model:
cmd: python 2_build_model.py cmd: python 2_build_model.py
deps: deps:
- path: 2_build_model.py - path: 2_build_model.py
hash: md5 hash: md5
md5: 84699d208874c52accaff61c6af9bb0a md5: b824822475c222521516493e68eef9c5
size: 5359 size: 4149
- path: data/prepared_data - path: data/prepared_data
hash: md5 hash: md5
md5: 9ce5c45722da7fc40491b5a4d00daf9e.dir md5: 70d79ba4a6f0648439dc55031c944d47.dir
size: 33881619 size: 32673907
nfiles: 2 nfiles: 2
params: params:
configs/build_model.yaml: configs/build_model.yaml:
@ -58,37 +58,39 @@ stages:
output_filepath: ./data/model/allmodels/ output_filepath: ./data/model/allmodels/
problem_type: regression problem_type: regression
eval_metric: mean_squared_error eval_metric: mean_squared_error
time_limit: 4000 time_limit: 600
presets: medium_quality presets: medium_quality
excluded_model_types: excluded_model_types:
- KNN - KNN
- RF - RF
infer_limit: 0.05
infer_limit_batch_size: 10000
outs: outs:
- path: data/model/ - path: data/model/
hash: md5 hash: md5
md5: 7bb5156243b4db39349e80a01ffecde4.dir md5: 2fc9223da8b72e61d81f06665e75019e.dir
size: 473398662 size: 324532985
nfiles: 27 nfiles: 27
- path: metrics/fit_metrics.json - path: metrics/fit_metrics.json
hash: md5 hash: md5
md5: 2bb16ac67de8778fbc08171d562b34d5 md5: 7d2f226251ce6f8e92af73d50dadb890
size: 184 size: 228
generate_predictions: generate_predictions:
cmd: python 3_generate_predictions.py cmd: python 3_generate_predictions.py
deps: deps:
- path: 3_generate_predictions.py - path: 3_generate_predictions.py
hash: md5 hash: md5
md5: 5ef2856a5a977304f1ec01f9b4205262 md5: 0a70ad4dfe99414a75d1261c75a177b9
size: 3028 size: 2464
- path: data/model - path: data/model
hash: md5 hash: md5
md5: 7bb5156243b4db39349e80a01ffecde4.dir md5: 2fc9223da8b72e61d81f06665e75019e.dir
size: 473398662 size: 324532985
nfiles: 27 nfiles: 27
- path: data/prepared_data - path: data/prepared_data
hash: md5 hash: md5
md5: 9ce5c45722da7fc40491b5a4d00daf9e.dir md5: 70d79ba4a6f0648439dc55031c944d47.dir
size: 33881619 size: 32673907
nfiles: 2 nfiles: 2
params: params:
configs/settings.yaml: configs/settings.yaml:
@ -100,25 +102,25 @@ stages:
outs: outs:
- path: data/predictions/ - path: data/predictions/
hash: md5 hash: md5
md5: 0bb3cf991906953def81c8204cdcfaf0.dir md5: 8bfc33c14aba5abf5ac4bdba32ff3c4c.dir
size: 374532 size: 412880
nfiles: 1 nfiles: 1
generate_metrics: generate_metrics:
cmd: python 4_generate_metrics.py cmd: python 4_generate_metrics.py
deps: deps:
- path: 4_generate_metrics.py - path: 4_generate_metrics.py
hash: md5 hash: md5
md5: 2c9fb78955a8c19cff0a098976f81d1b md5: d09a80dd55f1f69e2a832b1991b3c406
size: 4487 size: 3485
- path: data/predictions - path: data/predictions
hash: md5 hash: md5
md5: 0bb3cf991906953def81c8204cdcfaf0.dir md5: 8bfc33c14aba5abf5ac4bdba32ff3c4c.dir
size: 374532 size: 412880
nfiles: 1 nfiles: 1
- path: data/prepared_data - path: data/prepared_data
hash: md5 hash: md5
md5: 9ce5c45722da7fc40491b5a4d00daf9e.dir md5: 70d79ba4a6f0648439dc55031c944d47.dir
size: 33881619 size: 32673907
nfiles: 2 nfiles: 2
params: params:
configs/settings.yaml: configs/settings.yaml:
@ -128,15 +130,15 @@ stages:
outs: outs:
- path: metrics/metrics.json - path: metrics/metrics.json
hash: md5 hash: md5
md5: 2e13ae67759a64261d03224f1c0d4bf4 md5: 9a0b57244dfdbd6dab0392a4fd618123
size: 185 size: 225
startup_cleanup: startup_cleanup:
cmd: python 0_startup_cleanup.py cmd: python 0_startup_cleanup.py
deps: deps:
- path: 0_startup_cleanup.py - path: 0_startup_cleanup.py
hash: md5 hash: md5
md5: fbb7e3b1b98b517c870f3e1df3e7f695 md5: b1b12f6b6393fbf8b83d23684df0a3d4
size: 1676 size: 1220
params: params:
configs/settings.yaml: configs/settings.yaml:
default.startup_cleanup.artefacts: ./data default.startup_cleanup.artefacts: ./data

View file

@ -1,4 +1,4 @@
dvc==3.18.0 dvc==3.36.0
dvc-s3==2.23.0 dvc-s3==3.0.1
gto==1.0.4 gto==1.6.1
pyOpenSSL==23.2.0 pyOpenSSL==23.3.0