From 7ccb0ca3eb2d548e9759eb09aa711df47b1c0908 Mon Sep 17 00:00:00 2001 From: T-Aji Date: Fri, 23 Aug 2024 11:46:44 +0100 Subject: removed duplicate functions --- tests/test_fact_sales_order.py | 85 ++++++++++++++++++++++++++++++++++-------- 1 file changed, 69 insertions(+), 16 deletions(-) (limited to 'tests') diff --git a/tests/test_fact_sales_order.py b/tests/test_fact_sales_order.py index 82845d7..ca53faa 100644 --- a/tests/test_fact_sales_order.py +++ b/tests/test_fact_sales_order.py @@ -1,5 +1,6 @@ -from src.fact_sales_order import create_dim_design, create_dim_staff, create_dim_currency +from src.dataframes import create_dim_design, create_dim_staff, create_dim_payment_type, create_dim_counterparty, create_dim_currency import pandas as pd +from unittest.mock import patch class TestCreateDimDesign: def test_dim_design_returns_dataframe(self): @@ -36,22 +37,74 @@ class TestCreateDimStaff: expected_d = {"staff_id": ["Hello", "Bye"], "first_name": ["Hello", "Bye"], "last_name": ["Hello", "Bye"], "department_name": ["Hello", "Bye"], "location": ["Hello", "Bye"], "email_address": ["Hello", "Bye"]} expected_df = pd.DataFrame(data=expected_d) expected_result = expected_df.copy() - assert result.equals(expected_result) + assert result.equals(expected_result) -class TestCreateDimCurrency: - def test_dim_currency_returns_dataframe(self): - d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"]} - test_df = {"currency": pd.DataFrame(data=d)} - result = create_dim_currency(test_df) - assert isinstance(result, pd.DataFrame) - - def test_dim_currency_returns_columns_and_values(self): - d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"]} - test_df = {"currency": pd.DataFrame(data=d)} - result = create_dim_currency(test_df) - expected_d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"], "currency_name": ["US Dollar", "Euro", "Pound"]} +class TestCreatePaymentType: + def test_create_dim_payment_type_returns_correct_columns_and_values(self): + d = {"payment_type_id": ["Hello", "Bye"], "payment_type_name": ["Hello", "Bye"]} + test_df = {"payment_type": pd.DataFrame(data=d)} + result = create_dim_payment_type(test_df) + expected_columns = ["payment_type_id", "payment_type_name"] + expected_d = {"payment_type_id": ["Hello", "Bye"], "payment_type_name": ["Hello", "Bye"]} expected_df = pd.DataFrame(data=expected_d) - expected_result = expected_df.copy() - assert result.equals(expected_result) + assert isinstance(result, pd.DataFrame) + assert list(result.columns) == expected_columns + assert result.equals(expected_df) + +class TestCreateDimCounterparty: + def test_create_dim_counterparty_type_returns_correct_columns_and_values(self): + data_d = {"counterparty_id": ["Hello", "Bye"], + "counterparty_legal_name": ["Hello", "Bye"], + "counterparty_legal_address_line_1": ["Hello", "Bye"], + } + data_a = {"address_id": + "address", + } + test_df = {"address": pd.DataFrame(data=data_a)} + test_df = {} + result = create_dim_counterparty(test_df) + + expected_columns = ["counterparty_id", + "counterparty_legal_name", + "counterparty_legal_address_line_1", + "counterparty_legal_address_line_2", + "counterparty_legal_district", + "counterparty_legal_city", + "counterparty_legal_postal_code", + "counterparty_legal_postal_code", + "counterparty_legal_phone_number"] + expected_d = {"counterparty_id": ["Hello", "Bye"], + "counterparty_legal_name": ["Hello", "Bye"], + "counterparty_legal_address_line_1": ["Hello", "Bye"], + "counterparty_legal_address_line_2": ["Hello", "Bye"], + "counterparty_legal_district": ["Hello", "Bye"], + "counterparty_legal_city": ["Hello", "Bye"], + "counterparty_legal_postal_code": ["Hello", "Bye"], + "counterparty_legal_postal_code": ["Hello", "Bye"], + "counterparty_legal_phone_number": ["Hello", "Bye"]} + expected_df = pd.DataFrame(data=expected_d) + assert isinstance(result, pd.DataFrame) + assert list(result.columns) == expected_columns + assert result.equals(expected_df) + +# # figuring out how to mock currency scraper functiom +# class TestCreateDimCurrency: +# @patch("src.dataframes.scrape_currency_names") +# def test_dim_currency_returns_columns_and_values(self): +# d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"]} +# test_df = {"currency": pd.DataFrame(data=d)} +# result = create_dim_currency(test_df) +# expected_d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"], "currency_name": ["US Dollar", "Euro", "Pound"]} +# expected_df = pd.DataFrame(data=expected_d) +# expected_result = expected_df.copy() +# assert result.equals(expected_result) + +# def test_dim_currency_returns_dataframe(self): +# d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"]} +# test_df = {"currency": pd.DataFrame(data=d)} +# result = create_dim_currency(test_df) +# assert isinstance(result, pd.DataFrame) + + \ No newline at end of file -- cgit v1.2.3 From 821e241c925e682845e02e9609ba3a2c758966d8 Mon Sep 17 00:00:00 2001 From: Ang Bel Date: Fri, 23 Aug 2024 17:09:27 +0100 Subject: tests: additional tests written (pass) for dim tables transformation. Fact transformation functions not yet tested --- src/dataframes.py | 30 ++++++----- tests/test_fact_sales_order.py | 113 ++++++++++++++++++++++++----------------- 2 files changed, 82 insertions(+), 61 deletions(-) (limited to 'tests') diff --git a/src/dataframes.py b/src/dataframes.py index 042c8aa..7d10aa7 100644 --- a/src/dataframes.py +++ b/src/dataframes.py @@ -81,28 +81,28 @@ def create_fact_payment(dict_of_df): ]] return fact_payment +#test passed def create_dim_transaction(dict_of_df): - df_transaction = dict_of_df["transaction"].drop(labels=['created_at', 'last_updated'], axis=1).set_index('transaction_id') - dim_transaction = df_transaction.loc[:, ["payment_type_id", "payment_type_name"]] - return dim_transaction + df_transaction = dict_of_df["transaction"].drop(labels=['created_at', 'last_updated'], axis=1) + return df_transaction -## dim_location from address --> drops 2 columns +#test passed def create_dim_location(dict_of_df): - df_loc = dict_of_df['address'].drop(labels=['created_at', 'last_updated'], axis=1).rename(columns={'address_id': 'location_id'}).set_index('location_id') + df_loc = dict_of_df['address'].drop(labels=['created_at', 'last_updated'], axis=1).rename(columns={'address_id': 'location_id'}) return df_loc -## dim_counterparty from address and counterparty + def create_dim_counterparty(dict_of_df): df_prefixed_address = dict_of_df['address'].add_prefix('counterparty_legal_', axis=1) df_cp = pd.merge(dict_of_df['counterparty'], df_prefixed_address, left_on="legal_address_id", - right_on="address_id", - how="outer").set_index('counterparty_id') + right_on="counterparty_legal_address_id", + how="outer") + df_cp.drop(columns=["legal_address_id","counterparty_legal_address_id"],inplace=True) return df_cp - -## dim_date from purchase_order +#test passed def create_dim_date(dict_of_df): fact_dfs = [create_fact_payment(dict_of_df), create_fact_purchase_orders(dict_of_df), create_fact_sales_order(dict_of_df)] date_col_names = [col_name for col_name in list(fact_dfs[0].columns) if 'date' in col_name] @@ -119,9 +119,10 @@ def create_dim_date(dict_of_df): df_date['day_of_week'] = df_date['date_id'].dt.dayofweek df_date['day_name'] = df_date['date_id'].dt.day_name() df_date['month_name'] = df_date['date_id'].dt.month_name() - df_date['quarter'] = df_date['date_id'].dt.quarter #By default, the DataFrame index is not included when uploading to RDS. We are not setting indexes to retain the column information - return + df_date['quarter'] = df_date['date_id'].dt.quarter + return df_date +#tests passed def scrape_currency_names(): response = requests.get('https://www.xe.com/currency/').content soup = BeautifulSoup(response,'html.parser') @@ -130,11 +131,12 @@ def scrape_currency_names(): df_cur = sr.str.split(pat=" - ",expand=True).rename({0:'currency_code',1:'currency_name'},axis=1) return df_cur +#tests passed def create_dim_currency(dict_of_df,names=scrape_currency_names()): df_cur = dict_of_df['currency'].drop(labels=['created_at', 'last_updated'], axis=1) - dim_cur = pd.merge(df_cur,names,left_on='currency_code',right_on='currency_code',how='inner').set_index('currency_id') - print(dim_cur) + dim_cur = pd.merge(df_cur,names,left_on='currency_code',right_on='currency_code',how='inner') return dim_cur + #tests passed def create_dim_payment_type(dict_of_df): df_payment_type = dict_of_df["payment_type"] diff --git a/tests/test_fact_sales_order.py b/tests/test_fact_sales_order.py index ca53faa..f0796eb 100644 --- a/tests/test_fact_sales_order.py +++ b/tests/test_fact_sales_order.py @@ -1,6 +1,7 @@ -from src.dataframes import create_dim_design, create_dim_staff, create_dim_payment_type, create_dim_counterparty, create_dim_currency +from src.dataframes import * import pandas as pd from unittest.mock import patch +from datetime import datetime as dt class TestCreateDimDesign: def test_dim_design_returns_dataframe(self): @@ -52,59 +53,77 @@ class TestCreatePaymentType: assert result.equals(expected_df) class TestCreateDimCounterparty: - def test_create_dim_counterparty_type_returns_correct_columns_and_values(self): - data_d = {"counterparty_id": ["Hello", "Bye"], + + def test_create_dim_counterparty_type_returns_correct_columns_and_object(self): + data_l = pd.DataFrame(data={"counterparty_id": ["Hello", "Bye"], "counterparty_legal_name": ["Hello", "Bye"], - "counterparty_legal_address_line_1": ["Hello", "Bye"], - } - data_a = {"address_id": - "address", - } - test_df = {"address": pd.DataFrame(data=data_a)} - test_df = {} + "commercial_contact": ["Hello", "Bye"], + "legal_address_id": ["bond street", "regent street"]}) + data_a = pd.DataFrame(data={"address_id":["bond street", "regent street"], + "postcode":[98365,93753]}) + test_df = {"address": data_a,"counterparty":data_l} result = create_dim_counterparty(test_df) - expected_columns = ["counterparty_id", - "counterparty_legal_name", - "counterparty_legal_address_line_1", - "counterparty_legal_address_line_2", - "counterparty_legal_district", - "counterparty_legal_city", - "counterparty_legal_postal_code", - "counterparty_legal_postal_code", - "counterparty_legal_phone_number"] - expected_d = {"counterparty_id": ["Hello", "Bye"], - "counterparty_legal_name": ["Hello", "Bye"], - "counterparty_legal_address_line_1": ["Hello", "Bye"], - "counterparty_legal_address_line_2": ["Hello", "Bye"], - "counterparty_legal_district": ["Hello", "Bye"], - "counterparty_legal_city": ["Hello", "Bye"], - "counterparty_legal_postal_code": ["Hello", "Bye"], - "counterparty_legal_postal_code": ["Hello", "Bye"], - "counterparty_legal_phone_number": ["Hello", "Bye"]} - expected_df = pd.DataFrame(data=expected_d) + expected_columns = ["counterparty_id", "counterparty_legal_name", + "commercial_contact", "counterparty_legal_postcode"] + print(data_l) + print(data_a) assert isinstance(result, pd.DataFrame) assert list(result.columns) == expected_columns - assert result.equals(expected_df) -# # figuring out how to mock currency scraper functiom -# class TestCreateDimCurrency: -# @patch("src.dataframes.scrape_currency_names") -# def test_dim_currency_returns_columns_and_values(self): -# d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"]} -# test_df = {"currency": pd.DataFrame(data=d)} -# result = create_dim_currency(test_df) -# expected_d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"], "currency_name": ["US Dollar", "Euro", "Pound"]} -# expected_df = pd.DataFrame(data=expected_d) -# expected_result = expected_df.copy() -# assert result.equals(expected_result) +class TestCreateDimCurrency: + + def test_dim_currency_returns_columns_and_values(self): + nones = [None,None,None] + d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"],"created_at":nones,"last_updated":nones} + test_df = {"currency": pd.DataFrame(data=d)} + scraper_output = pd.DataFrame({"currency_code":["RUS","USD","PHP","GBP","EUR"],"currency_name":["Rubble","US Dollar","Peso","Pound","Euro"]}) + result = create_dim_currency(test_df,names=scraper_output).sort_values(by="currency_code",axis=0) + expected_d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"], "currency_name": ["US Dollar", "Euro", "Pound"]} + expected_df = pd.DataFrame(data=expected_d).sort_values(by="currency_code",axis=0) + assert isinstance(result, pd.DataFrame) + assert result.equals(expected_df) -# def test_dim_currency_returns_dataframe(self): -# d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"]} -# test_df = {"currency": pd.DataFrame(data=d)} -# result = create_dim_currency(test_df) -# assert isinstance(result, pd.DataFrame) + def test_scrape_currency_names_returns_dataframe_with_correct_collumns(self): + result = scrape_currency_names() + assert isinstance(result,pd.DataFrame) + assert list(result.columns) == ['currency_code', 'currency_name'] + +class TestCreateDimDate: + + def test_returns_required_columns(self): + df_one = pd.DataFrame(data={'updated_date':dt(2020, 5, 17),'created_date':dt(2021, 5, 13),'not_dat':None},index=[0]) + df_two = pd.DataFrame(data={'updated_date':dt(2020, 5, 17),'created_date':dt(2021, 9, 13)},index=[0]) + df_three = pd.DataFrame(data={'updated_date':dt(2022, 5, 17),'created_date':dt(2023, 5, 13)},index=[0]) + expected_df = pd.DataFrame(data= + [[dt(2020,5,17),2020,5,17,6,'Sunday','May',2], + [dt(2021,5,13),2021,5,13,3,'Thursday','May',2], + [dt(2021,9,13),2021,9,13,0,'Monday','September',3], + [dt(2022,5,17),2022,5,17,1,'Tuesday','May',2], + [dt(2023,5,13),2023,5,13,5,'Saturday','May',2]], + columns=['date_id','year','month','day','day_of_week','day_name','month_name','quarter']) + with patch("src.dataframes.create_fact_payment") as mock_fp: + with patch("src.dataframes.create_fact_purchase_orders") as mock_fpo: + with patch("src.dataframes.create_fact_sales_order") as mock_fso: + mock_fp.return_value = df_one + mock_fpo.return_value = df_two + mock_fso.return_value = df_three + result = create_dim_date({'dum':0}) + result.reset_index(inplace=True,drop=True) + assert result.eq(expected_df, axis="columns").all(axis=None) - +class TestCreateDimLocation: + def test_returns_correct_columns_lo(self): + dict_df = {'address':pd.DataFrame(data=[['some_time','some_other_time',1,'SE18 9QO']], + columns=['created_at','last_updated','address_id','postal_code'])} + result = create_dim_location(dict_df) + assert list(result.columns) == ['location_id','postal_code'] + +class TestCreateDimTransaction: + def test_returns_correct_columns_tr(self): + dict_df = {'transaction':pd.DataFrame(data=[['some_time','some_other_time',1,'SE18 9QO']], + columns=['created_at','last_updated','transaction_id','some_other_id'])} + result = create_dim_transaction(dict_df) + assert list(result.columns) == ['transaction_id','some_other_id'] \ No newline at end of file -- cgit v1.2.3 From 30525f27ba1d20c65216cbe58a62953b8f1fe947 Mon Sep 17 00:00:00 2001 From: "deepsource-autofix[bot]" <62050782+deepsource-autofix[bot]@users.noreply.github.com> Date: Fri, 23 Aug 2024 16:11:04 +0000 Subject: style: format code with Autopep8, Black and Ruff Formatter This commit fixes the style issues introduced in 821e241 according to the output from Autopep8, Black and Ruff Formatter. Details: https://github.com/ajschofield/de-project-bentley/pull/96 --- src/dataframes.py | 250 +++++++++++++++++++++++++---------------- tests/test_fact_sales_order.py | 235 ++++++++++++++++++++++++++++---------- 2 files changed, 330 insertions(+), 155 deletions(-) (limited to 'tests') diff --git a/src/dataframes.py b/src/dataframes.py index 7d10aa7..737ee2a 100644 --- a/src/dataframes.py +++ b/src/dataframes.py @@ -2,7 +2,7 @@ import pandas as pd from bs4 import BeautifulSoup import requests -#Table names: +# Table names: # fact_sales_order # fact_purchase_orders # fact_payment @@ -16,7 +16,6 @@ import requests # dim_counterparty - def create_fact_sales_order(dict_of_df): df_sales = dict_of_df["sales_order"] df_sales.index.name = "sales_record_id" @@ -24,36 +23,46 @@ def create_fact_sales_order(dict_of_df): df_sales["created_time"] = pd.to_datetime(df_sales["created_at"]).dt.time df_sales["last_updated_date"] = pd.to_datetime(df_sales["last_updated"]).dt.date df_sales["last_updated_time"] = pd.to_datetime(df_sales["last_updated"]).dt.time - fact_sales_order = df_sales.loc[:,[ - "sales_record_id", - "sales_order_id", - "created_date", - "created_time", - "last_updated_date", - "last_updated_time", - "sales_staff_id", - "counterparty_id", - "units_sold", - "unit_price", - "currency_id", - "design_id", - "agreed_payment_date", - "agreed_delivery_date", - "agreed_delivery_location_id" - ]] + fact_sales_order = df_sales.loc[ + :, + [ + "sales_record_id", + "sales_order_id", + "created_date", + "created_time", + "last_updated_date", + "last_updated_time", + "sales_staff_id", + "counterparty_id", + "units_sold", + "unit_price", + "currency_id", + "design_id", + "agreed_payment_date", + "agreed_delivery_date", + "agreed_delivery_location_id", + ], + ] return fact_sales_order -## fact_purchase_order from purchase_order + +# fact_purchase_order from purchase_order + + def create_fact_purchase_orders(dict_of_df): - df_po = dict_of_df['purchase_order'] - df_po.index.name = 'purchase_record_id' - df_po['created_date'] = df_po['created_at'].date() - df_po['created_time'] = df_po['created_at'].dt.time - df_po['last_updated_date'] = df_po['last_updated_at'].date() - df_po['last_updated_time'] = df_po['last_updated_at'].dt.time - df_po['agreed_delivery_date'] = pd.to_datetime(df_po['agreed_delivery_date'],format="%Y-%m-%d") - df_po['agreed_payment_date'] = pd.to_datetime(df_po['agreed_payment_date'],format="%Y-%m-%d") - df_po.drop(labels=['created_at','last_updated_at'],axis=1,inplace=True) + df_po = dict_of_df["purchase_order"] + df_po.index.name = "purchase_record_id" + df_po["created_date"] = df_po["created_at"].date() + df_po["created_time"] = df_po["created_at"].dt.time + df_po["last_updated_date"] = df_po["last_updated_at"].date() + df_po["last_updated_time"] = df_po["last_updated_at"].dt.time + df_po["agreed_delivery_date"] = pd.to_datetime( + df_po["agreed_delivery_date"], format="%Y-%m-%d" + ) + df_po["agreed_payment_date"] = pd.to_datetime( + df_po["agreed_payment_date"], format="%Y-%m-%d" + ) + df_po.drop(labels=["created_at", "last_updated_at"], axis=1, inplace=True) return df_po @@ -64,109 +73,158 @@ def create_fact_payment(dict_of_df): df_payment["created_time"] = pd.to_datetime(df_payment["created_at"]).dt.time df_payment["last_updated_date"] = pd.to_datetime(df_payment["last_updated"]).dt.date df_payment["last_updated_time"] = pd.to_datetime(df_payment["last_updated"]).dt.time - fact_payment = df_payment.loc[:,[ - "payment_record_id", - "payment_id", - "created_date", - "created_time", - "last_updated_date", - "last_updated_time", - "transaction_id", - "counterparty_id", - "payment_amount", - "currency_id", - "payment_type_id", - "paid", - "payment_date" - ]] + fact_payment = df_payment.loc[ + :, + [ + "payment_record_id", + "payment_id", + "created_date", + "created_time", + "last_updated_date", + "last_updated_time", + "transaction_id", + "counterparty_id", + "payment_amount", + "currency_id", + "payment_type_id", + "paid", + "payment_date", + ], + ] return fact_payment -#test passed + +# test passed + + def create_dim_transaction(dict_of_df): - df_transaction = dict_of_df["transaction"].drop(labels=['created_at', 'last_updated'], axis=1) + df_transaction = dict_of_df["transaction"].drop( + labels=["created_at", "last_updated"], axis=1 + ) return df_transaction -#test passed + +# test passed + + def create_dim_location(dict_of_df): - df_loc = dict_of_df['address'].drop(labels=['created_at', 'last_updated'], axis=1).rename(columns={'address_id': 'location_id'}) + df_loc = ( + dict_of_df["address"] + .drop(labels=["created_at", "last_updated"], axis=1) + .rename(columns={"address_id": "location_id"}) + ) return df_loc def create_dim_counterparty(dict_of_df): - df_prefixed_address = dict_of_df['address'].add_prefix('counterparty_legal_', axis=1) - df_cp = pd.merge(dict_of_df['counterparty'], - df_prefixed_address, - left_on="legal_address_id", - right_on="counterparty_legal_address_id", - how="outer") - df_cp.drop(columns=["legal_address_id","counterparty_legal_address_id"],inplace=True) + df_prefixed_address = dict_of_df["address"].add_prefix( + "counterparty_legal_", axis=1 + ) + df_cp = pd.merge( + dict_of_df["counterparty"], + df_prefixed_address, + left_on="legal_address_id", + right_on="counterparty_legal_address_id", + how="outer", + ) + df_cp.drop( + columns=["legal_address_id", "counterparty_legal_address_id"], inplace=True + ) return df_cp -#test passed + +# test passed + + def create_dim_date(dict_of_df): - fact_dfs = [create_fact_payment(dict_of_df), create_fact_purchase_orders(dict_of_df), create_fact_sales_order(dict_of_df)] - date_col_names = [col_name for col_name in list(fact_dfs[0].columns) if 'date' in col_name] + fact_dfs = [ + create_fact_payment(dict_of_df), + create_fact_purchase_orders(dict_of_df), + create_fact_sales_order(dict_of_df), + ] + date_col_names = [ + col_name for col_name in list(fact_dfs[0].columns) if "date" in col_name + ] list_of_date_columns = [] for df in fact_dfs: for col in date_col_names: list_of_date_columns.append(df[col]) - sr_date = pd.array(pd.concat(list_of_date_columns),dtype='datetime64[ns]') - df_date = pd.DataFrame(data=sr_date,columns=['date_id']) + sr_date = pd.array(pd.concat(list_of_date_columns), dtype="datetime64[ns]") + df_date = pd.DataFrame(data=sr_date, columns=["date_id"]) df_date.drop_duplicates(inplace=True) - df_date['year'] = df_date['date_id'].dt.year - df_date['month'] = df_date['date_id'].dt.month - df_date['day'] = df_date['date_id'].dt.day - df_date['day_of_week'] = df_date['date_id'].dt.dayofweek - df_date['day_name'] = df_date['date_id'].dt.day_name() - df_date['month_name'] = df_date['date_id'].dt.month_name() - df_date['quarter'] = df_date['date_id'].dt.quarter + df_date["year"] = df_date["date_id"].dt.year + df_date["month"] = df_date["date_id"].dt.month + df_date["day"] = df_date["date_id"].dt.day + df_date["day_of_week"] = df_date["date_id"].dt.dayofweek + df_date["day_name"] = df_date["date_id"].dt.day_name() + df_date["month_name"] = df_date["date_id"].dt.month_name() + df_date["quarter"] = df_date["date_id"].dt.quarter return df_date -#tests passed -def scrape_currency_names(): - response = requests.get('https://www.xe.com/currency/').content - soup = BeautifulSoup(response,'html.parser') - currency = [item.text for item in soup.findAll('a', attrs={'class' : "sc-299dec64-6 fZPTSw"})] - sr = pd.Series(currency) - df_cur = sr.str.split(pat=" - ",expand=True).rename({0:'currency_code',1:'currency_name'},axis=1) - return df_cur - -#tests passed -def create_dim_currency(dict_of_df,names=scrape_currency_names()): - df_cur = dict_of_df['currency'].drop(labels=['created_at', 'last_updated'], axis=1) - dim_cur = pd.merge(df_cur,names,left_on='currency_code',right_on='currency_code',how='inner') - return dim_cur -#tests passed -def create_dim_payment_type(dict_of_df): - df_payment_type = dict_of_df["payment_type"] - dim_payment_type = df_payment_type.loc[:, ["payment_type_id", "payment_type_name"]] - return dim_payment_type +# tests passed -#tests passed -def create_dim_design(dict_of_df): - df_design = dict_of_df["design"] - dim_design = df_design.loc[:, ["design_id", "design_name", "file_name", "file_location"]] - return dim_design -#tests passed -def create_dim_staff(dict_of_df): - staff_department = pd.merge(dict_of_df["staff"], dict_of_df["department"], on='department_id', how="left") - dim_staff = staff_department.loc[:, ['staff_id', 'first_name', 'last_name', 'department_name', 'location', 'email_address']] - return dim_staff +def scrape_currency_names(): + response = requests.get("https://www.xe.com/currency/").content + soup = BeautifulSoup(response, "html.parser") + currency = [ + item.text for item in soup.findAll("a", attrs={"class": "sc-299dec64-6 fZPTSw"}) + ] + sr = pd.Series(currency) + df_cur = sr.str.split(pat=" - ", expand=True).rename( + {0: "currency_code", 1: "currency_name"}, axis=1 + ) + return df_cur +# tests passed +def create_dim_currency(dict_of_df, names=scrape_currency_names()): + df_cur = dict_of_df["currency"].drop(labels=["created_at", "last_updated"], axis=1) + dim_cur = pd.merge( + df_cur, names, left_on="currency_code", right_on="currency_code", how="inner" + ) + return dim_cur +# tests passed +def create_dim_payment_type(dict_of_df): + df_payment_type = dict_of_df["payment_type"] + dim_payment_type = df_payment_type.loc[:, ["payment_type_id", "payment_type_name"]] + return dim_payment_type +# tests passed +def create_dim_design(dict_of_df): + df_design = dict_of_df["design"] + dim_design = df_design.loc[ + :, ["design_id", "design_name", "file_name", "file_location"] + ] + return dim_design +# tests passed +def create_dim_staff(dict_of_df): + staff_department = pd.merge( + dict_of_df["staff"], dict_of_df["department"], on="department_id", how="left" + ) + dim_staff = staff_department.loc[ + :, + [ + "staff_id", + "first_name", + "last_name", + "department_name", + "location", + "email_address", + ], + ] + return dim_staff diff --git a/tests/test_fact_sales_order.py b/tests/test_fact_sales_order.py index f0796eb..a245379 100644 --- a/tests/test_fact_sales_order.py +++ b/tests/test_fact_sales_order.py @@ -3,42 +3,88 @@ import pandas as pd from unittest.mock import patch from datetime import datetime as dt + class TestCreateDimDesign: def test_dim_design_returns_dataframe(self): - d = {"test": ["Hello", "Bye"], "design_id": ["Hello", "Bye"], "design_name": ["Hello", "Bye"], - "file_name": ["Hello", "Bye"], "file_location": ["Hello", "Bye"], "Hello": ["Hello", "Bye"]} + d = { + "test": ["Hello", "Bye"], + "design_id": ["Hello", "Bye"], + "design_name": ["Hello", "Bye"], + "file_name": ["Hello", "Bye"], + "file_location": ["Hello", "Bye"], + "Hello": ["Hello", "Bye"], + } test_df = {"design": pd.DataFrame(data=d)} result = create_dim_design(test_df) assert isinstance(result, pd.DataFrame) def test_dim_design_returns_correct_columns_and_values(self): - d = {"test": ["Hello", "Bye"], "design_id": ["Hello", "Bye"], "design_name": ["Hello", "Bye"], - "file_name": ["Hello", "Bye"], "file_location": ["Hello", "Bye"], "Hello": ["Hello", "Bye"]} + d = { + "test": ["Hello", "Bye"], + "design_id": ["Hello", "Bye"], + "design_name": ["Hello", "Bye"], + "file_name": ["Hello", "Bye"], + "file_location": ["Hello", "Bye"], + "Hello": ["Hello", "Bye"], + } test_df = {"design": pd.DataFrame(data=d)} result = create_dim_design(test_df) - d2 = {"design_id": ["Hello", "Bye"], "design_name": ["Hello", "Bye"], "file_name": ["Hello", "Bye"], - "file_location": ["Hello", "Bye"]} + d2 = { + "design_id": ["Hello", "Bye"], + "design_name": ["Hello", "Bye"], + "file_name": ["Hello", "Bye"], + "file_location": ["Hello", "Bye"], + } expected_df = pd.DataFrame(data=d2) expected_result = expected_df.copy() assert result.equals(expected_result) + class TestCreateDimStaff: def test_dim_staff_returns_dataframe(self): - d = {"staff_id": ["Hello", "Bye"], "first_name": ["Hello", "Bye"], "last_name": ["Hello", "Bye"], "department_id": ["Hello", "Bye"]} - d2 = {"department_name": ["Hello", "Bye"], "location": ["Hello", "Bye"], "email_address": ["Hello", "Bye"], "department_id": ["Hello", "Bye"]} + d = { + "staff_id": ["Hello", "Bye"], + "first_name": ["Hello", "Bye"], + "last_name": ["Hello", "Bye"], + "department_id": ["Hello", "Bye"], + } + d2 = { + "department_name": ["Hello", "Bye"], + "location": ["Hello", "Bye"], + "email_address": ["Hello", "Bye"], + "department_id": ["Hello", "Bye"], + } test_df = {"staff": pd.DataFrame(data=d), "department": pd.DataFrame(data=d2)} result = create_dim_staff(test_df) - assert isinstance(result, pd.DataFrame) + assert isinstance(result, pd.DataFrame) def test_dim_staff_returns_correct_columns_and_values(self): - d = {"staff_id": ["Hello", "Bye"], "first_name": ["Hello", "Bye"], "last_name": ["Hello", "Bye"], "department_id": ["Hello", "Bye"]} - d2 = {"department_name": ["Hello", "Bye"], "location": ["Hello", "Bye"], "email_address": ["Hello", "Bye"], "department_id": ["Hello", "Bye"]} + d = { + "staff_id": ["Hello", "Bye"], + "first_name": ["Hello", "Bye"], + "last_name": ["Hello", "Bye"], + "department_id": ["Hello", "Bye"], + } + d2 = { + "department_name": ["Hello", "Bye"], + "location": ["Hello", "Bye"], + "email_address": ["Hello", "Bye"], + "department_id": ["Hello", "Bye"], + } test_df = {"staff": pd.DataFrame(data=d), "department": pd.DataFrame(data=d2)} result = create_dim_staff(test_df) - expected_d = {"staff_id": ["Hello", "Bye"], "first_name": ["Hello", "Bye"], "last_name": ["Hello", "Bye"], "department_name": ["Hello", "Bye"], "location": ["Hello", "Bye"], "email_address": ["Hello", "Bye"]} + expected_d = { + "staff_id": ["Hello", "Bye"], + "first_name": ["Hello", "Bye"], + "last_name": ["Hello", "Bye"], + "department_name": ["Hello", "Bye"], + "location": ["Hello", "Bye"], + "email_address": ["Hello", "Bye"], + } expected_df = pd.DataFrame(data=expected_d) expected_result = expected_df.copy() - assert result.equals(expected_result) + assert result.equals(expected_result) + class TestCreatePaymentType: def test_create_dim_payment_type_returns_correct_columns_and_values(self): @@ -46,84 +92,155 @@ class TestCreatePaymentType: test_df = {"payment_type": pd.DataFrame(data=d)} result = create_dim_payment_type(test_df) expected_columns = ["payment_type_id", "payment_type_name"] - expected_d = {"payment_type_id": ["Hello", "Bye"], "payment_type_name": ["Hello", "Bye"]} + expected_d = { + "payment_type_id": ["Hello", "Bye"], + "payment_type_name": ["Hello", "Bye"], + } expected_df = pd.DataFrame(data=expected_d) assert isinstance(result, pd.DataFrame) assert list(result.columns) == expected_columns assert result.equals(expected_df) + class TestCreateDimCounterparty: - def test_create_dim_counterparty_type_returns_correct_columns_and_object(self): - data_l = pd.DataFrame(data={"counterparty_id": ["Hello", "Bye"], - "counterparty_legal_name": ["Hello", "Bye"], - "commercial_contact": ["Hello", "Bye"], - "legal_address_id": ["bond street", "regent street"]}) - data_a = pd.DataFrame(data={"address_id":["bond street", "regent street"], - "postcode":[98365,93753]}) - test_df = {"address": data_a,"counterparty":data_l} + data_l = pd.DataFrame( + data={ + "counterparty_id": ["Hello", "Bye"], + "counterparty_legal_name": ["Hello", "Bye"], + "commercial_contact": ["Hello", "Bye"], + "legal_address_id": ["bond street", "regent street"], + } + ) + data_a = pd.DataFrame( + data={ + "address_id": ["bond street", "regent street"], + "postcode": [98365, 93753], + } + ) + test_df = {"address": data_a, "counterparty": data_l} result = create_dim_counterparty(test_df) - expected_columns = ["counterparty_id", "counterparty_legal_name", - "commercial_contact", "counterparty_legal_postcode"] + expected_columns = [ + "counterparty_id", + "counterparty_legal_name", + "commercial_contact", + "counterparty_legal_postcode", + ] print(data_l) print(data_a) assert isinstance(result, pd.DataFrame) assert list(result.columns) == expected_columns + class TestCreateDimCurrency: - def test_dim_currency_returns_columns_and_values(self): - nones = [None,None,None] - d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"],"created_at":nones,"last_updated":nones} + nones = [None, None, None] + d = { + "currency_id": [1, 2, 3], + "currency_code": ["USD", "EUR", "GBP"], + "created_at": nones, + "last_updated": nones, + } test_df = {"currency": pd.DataFrame(data=d)} - scraper_output = pd.DataFrame({"currency_code":["RUS","USD","PHP","GBP","EUR"],"currency_name":["Rubble","US Dollar","Peso","Pound","Euro"]}) - result = create_dim_currency(test_df,names=scraper_output).sort_values(by="currency_code",axis=0) - expected_d = {"currency_id": [1, 2, 3], "currency_code": ["USD", "EUR", "GBP"], "currency_name": ["US Dollar", "Euro", "Pound"]} - expected_df = pd.DataFrame(data=expected_d).sort_values(by="currency_code",axis=0) - assert isinstance(result, pd.DataFrame) - assert result.equals(expected_df) + scraper_output = pd.DataFrame( + { + "currency_code": ["RUS", "USD", "PHP", "GBP", "EUR"], + "currency_name": ["Rubble", "US Dollar", "Peso", "Pound", "Euro"], + } + ) + result = create_dim_currency(test_df, names=scraper_output).sort_values( + by="currency_code", axis=0 + ) + expected_d = { + "currency_id": [1, 2, 3], + "currency_code": ["USD", "EUR", "GBP"], + "currency_name": ["US Dollar", "Euro", "Pound"], + } + expected_df = pd.DataFrame(data=expected_d).sort_values( + by="currency_code", axis=0 + ) + assert isinstance(result, pd.DataFrame) + assert result.equals(expected_df) def test_scrape_currency_names_returns_dataframe_with_correct_collumns(self): result = scrape_currency_names() - assert isinstance(result,pd.DataFrame) - assert list(result.columns) == ['currency_code', 'currency_name'] + assert isinstance(result, pd.DataFrame) + assert list(result.columns) == ["currency_code", "currency_name"] -class TestCreateDimDate: +class TestCreateDimDate: def test_returns_required_columns(self): - df_one = pd.DataFrame(data={'updated_date':dt(2020, 5, 17),'created_date':dt(2021, 5, 13),'not_dat':None},index=[0]) - df_two = pd.DataFrame(data={'updated_date':dt(2020, 5, 17),'created_date':dt(2021, 9, 13)},index=[0]) - df_three = pd.DataFrame(data={'updated_date':dt(2022, 5, 17),'created_date':dt(2023, 5, 13)},index=[0]) - expected_df = pd.DataFrame(data= - [[dt(2020,5,17),2020,5,17,6,'Sunday','May',2], - [dt(2021,5,13),2021,5,13,3,'Thursday','May',2], - [dt(2021,9,13),2021,9,13,0,'Monday','September',3], - [dt(2022,5,17),2022,5,17,1,'Tuesday','May',2], - [dt(2023,5,13),2023,5,13,5,'Saturday','May',2]], - columns=['date_id','year','month','day','day_of_week','day_name','month_name','quarter']) + df_one = pd.DataFrame( + data={ + "updated_date": dt(2020, 5, 17), + "created_date": dt(2021, 5, 13), + "not_dat": None, + }, + index=[0], + ) + df_two = pd.DataFrame( + data={"updated_date": dt(2020, 5, 17), "created_date": dt(2021, 9, 13)}, + index=[0], + ) + df_three = pd.DataFrame( + data={"updated_date": dt(2022, 5, 17), "created_date": dt(2023, 5, 13)}, + index=[0], + ) + expected_df = pd.DataFrame( + data=[ + [dt(2020, 5, 17), 2020, 5, 17, 6, "Sunday", "May", 2], + [dt(2021, 5, 13), 2021, 5, 13, 3, "Thursday", "May", 2], + [dt(2021, 9, 13), 2021, 9, 13, 0, "Monday", "September", 3], + [dt(2022, 5, 17), 2022, 5, 17, 1, "Tuesday", "May", 2], + [dt(2023, 5, 13), 2023, 5, 13, 5, "Saturday", "May", 2], + ], + columns=[ + "date_id", + "year", + "month", + "day", + "day_of_week", + "day_name", + "month_name", + "quarter", + ], + ) with patch("src.dataframes.create_fact_payment") as mock_fp: with patch("src.dataframes.create_fact_purchase_orders") as mock_fpo: with patch("src.dataframes.create_fact_sales_order") as mock_fso: mock_fp.return_value = df_one mock_fpo.return_value = df_two mock_fso.return_value = df_three - result = create_dim_date({'dum':0}) - result.reset_index(inplace=True,drop=True) + result = create_dim_date({"dum": 0}) + result.reset_index(inplace=True, drop=True) assert result.eq(expected_df, axis="columns").all(axis=None) - -class TestCreateDimLocation: + +class TestCreateDimLocation: def test_returns_correct_columns_lo(self): - dict_df = {'address':pd.DataFrame(data=[['some_time','some_other_time',1,'SE18 9QO']], - columns=['created_at','last_updated','address_id','postal_code'])} + dict_df = { + "address": pd.DataFrame( + data=[["some_time", "some_other_time", 1, "SE18 9QO"]], + columns=["created_at", "last_updated", "address_id", "postal_code"], + ) + } result = create_dim_location(dict_df) - assert list(result.columns) == ['location_id','postal_code'] - + assert list(result.columns) == ["location_id", "postal_code"] + + class TestCreateDimTransaction: - def test_returns_correct_columns_tr(self): - dict_df = {'transaction':pd.DataFrame(data=[['some_time','some_other_time',1,'SE18 9QO']], - columns=['created_at','last_updated','transaction_id','some_other_id'])} + def test_returns_correct_columns_tr(self): + dict_df = { + "transaction": pd.DataFrame( + data=[["some_time", "some_other_time", 1, "SE18 9QO"]], + columns=[ + "created_at", + "last_updated", + "transaction_id", + "some_other_id", + ], + ) + } result = create_dim_transaction(dict_df) - assert list(result.columns) == ['transaction_id','some_other_id'] - \ No newline at end of file + assert list(result.columns) == ["transaction_id", "some_other_id"] -- cgit v1.2.3 From 843471508b150f505c2b8921d175c8f9b781bf48 Mon Sep 17 00:00:00 2001 From: "deepsource-autofix[bot]" <62050782+deepsource-autofix[bot]@users.noreply.github.com> Date: Fri, 23 Aug 2024 16:25:59 +0000 Subject: style: format code with Autopep8, Black and Ruff Formatter This commit fixes the style issues introduced in 8f75a47 according to the output from Autopep8, Black and Ruff Formatter. Details: https://github.com/ajschofield/de-project-bentley/pull/96 --- src/dataframes.py | 76 +++++++++++++++++++++++------------------- tests/test_fact_sales_order.py | 3 -- 2 files changed, 41 insertions(+), 38 deletions(-) (limited to 'tests') diff --git a/src/dataframes.py b/src/dataframes.py index fc84f48..f2cae5d 100644 --- a/src/dataframes.py +++ b/src/dataframes.py @@ -16,14 +16,15 @@ import requests # dim_counterparty - def create_fact_sales_order(dict_of_df): df_sales = dict_of_df["sales_order"] df_sales.index.name = "sales_record_id" df_sales["created_date"] = pd.to_datetime(df_sales["created_at"]).dt.date df_sales["created_time"] = pd.to_datetime(df_sales["created_at"]).dt.time - df_sales["last_updated_date"] = pd.to_datetime(df_sales["last_updated"]).dt.date - df_sales["last_updated_time"] = pd.to_datetime(df_sales["last_updated"]).dt.time + df_sales["last_updated_date"] = pd.to_datetime( + df_sales["last_updated"]).dt.date + df_sales["last_updated_time"] = pd.to_datetime( + df_sales["last_updated"]).dt.time fact_sales_order = df_sales.loc[ :, [ @@ -70,10 +71,14 @@ def create_fact_purchase_orders(dict_of_df): def create_fact_payment(dict_of_df): df_payment = dict_of_df["payment"] df_payment.index.name = "payment_record_id" - df_payment["created_date"] = pd.to_datetime(df_payment["created_at"]).dt.date - df_payment["created_time"] = pd.to_datetime(df_payment["created_at"]).dt.time - df_payment["last_updated_date"] = pd.to_datetime(df_payment["last_updated"]).dt.date - df_payment["last_updated_time"] = pd.to_datetime(df_payment["last_updated"]).dt.time + df_payment["created_date"] = pd.to_datetime( + df_payment["created_at"]).dt.date + df_payment["created_time"] = pd.to_datetime( + df_payment["created_at"]).dt.time + df_payment["last_updated_date"] = pd.to_datetime( + df_payment["last_updated"]).dt.date + df_payment["last_updated_time"] = pd.to_datetime( + df_payment["last_updated"]).dt.time fact_payment = df_payment.loc[ :, [ @@ -95,7 +100,6 @@ def create_fact_payment(dict_of_df): return fact_payment - # test passed @@ -117,10 +121,10 @@ def create_dim_location(dict_of_df): def create_dim_counterparty(dict_of_df): - df_prefixed_address = dict_of_df["address"].add_prefix( + df_prefixed_address=dict_of_df["address"].add_prefix( "counterparty_legal_", axis=1 ) - df_cp = pd.merge( + df_cp=pd.merge( dict_of_df["counterparty"], df_prefixed_address, left_on="legal_address_id", @@ -137,40 +141,40 @@ def create_dim_counterparty(dict_of_df): def create_dim_date(dict_of_df): - fact_dfs = [ + fact_dfs=[ create_fact_payment(dict_of_df), create_fact_purchase_orders(dict_of_df), create_fact_sales_order(dict_of_df), ] - date_col_names = [ + date_col_names=[ col_name for col_name in list(fact_dfs[0].columns) if "date" in col_name ] - list_of_date_columns = [] + list_of_date_columns=[] for df in fact_dfs: for col in date_col_names: list_of_date_columns.append(df[col]) - sr_date = pd.array(pd.concat(list_of_date_columns), dtype="datetime64[ns]") - df_date = pd.DataFrame(data=sr_date, columns=["date_id"]) + sr_date=pd.array(pd.concat(list_of_date_columns), dtype="datetime64[ns]") + df_date=pd.DataFrame(data=sr_date, columns=["date_id"]) df_date.drop_duplicates(inplace=True) - df_date["year"] = df_date["date_id"].dt.year - df_date["month"] = df_date["date_id"].dt.month - df_date["day"] = df_date["date_id"].dt.day - df_date["day_of_week"] = df_date["date_id"].dt.dayofweek - df_date["day_name"] = df_date["date_id"].dt.day_name() - df_date["month_name"] = df_date["date_id"].dt.month_name() - df_date["quarter"] = df_date["date_id"].dt.quarter + df_date["year"]=df_date["date_id"].dt.year + df_date["month"]=df_date["date_id"].dt.month + df_date["day"]=df_date["date_id"].dt.day + df_date["day_of_week"]=df_date["date_id"].dt.dayofweek + df_date["day_name"]=df_date["date_id"].dt.day_name() + df_date["month_name"]=df_date["date_id"].dt.month_name() + df_date["quarter"]=df_date["date_id"].dt.quarter return df_date # tests passed def scrape_currency_names(): - response = requests.get("https://www.xe.com/currency/").content - soup = BeautifulSoup(response, "html.parser") - currency = [ + response=requests.get("https://www.xe.com/currency/").content + soup=BeautifulSoup(response, "html.parser") + currency=[ item.text for item in soup.findAll("a", attrs={"class": "sc-299dec64-6 fZPTSw"}) ] - sr = pd.Series(currency) - df_cur = sr.str.split(pat=" - ", expand=True).rename( + sr=pd.Series(currency) + df_cur=sr.str.split(pat=" - ", expand=True).rename( {0: "currency_code", 1: "currency_name"}, axis=1 ) return df_cur @@ -179,8 +183,9 @@ def scrape_currency_names(): def create_dim_currency(dict_of_df, names=scrape_currency_names()): - df_cur = dict_of_df["currency"].drop(labels=["created_at", "last_updated"], axis=1) - dim_cur = pd.merge( + df_cur=dict_of_df["currency"].drop( + labels=["created_at", "last_updated"], axis=1) + dim_cur=pd.merge( df_cur, names, left_on="currency_code", right_on="currency_code", how="inner" ) return dim_cur @@ -189,8 +194,9 @@ def create_dim_currency(dict_of_df, names=scrape_currency_names()): # tests passed def create_dim_payment_type(dict_of_df): - df_payment_type = dict_of_df["payment_type"] - dim_payment_type = df_payment_type.loc[:, ["payment_type_id", "payment_type_name"]] + df_payment_type=dict_of_df["payment_type"] + dim_payment_type=df_payment_type.loc[:, [ + "payment_type_id", "payment_type_name"]] return dim_payment_type @@ -199,8 +205,8 @@ def create_dim_payment_type(dict_of_df): def create_dim_design(dict_of_df): - df_design = dict_of_df["design"] - dim_design = df_design.loc[ + df_design=dict_of_df["design"] + dim_design=df_design.loc[ :, ["design_id", "design_name", "file_name", "file_location"] ] return dim_design @@ -210,10 +216,10 @@ def create_dim_design(dict_of_df): # tests passed def create_dim_staff(dict_of_df): - staff_department = pd.merge( + staff_department=pd.merge( dict_of_df["staff"], dict_of_df["department"], on="department_id", how="left" ) - dim_staff = staff_department.loc[ + dim_staff=staff_department.loc[ :, [ "staff_id", diff --git a/tests/test_fact_sales_order.py b/tests/test_fact_sales_order.py index 77395a1..a245379 100644 --- a/tests/test_fact_sales_order.py +++ b/tests/test_fact_sales_order.py @@ -4,7 +4,6 @@ from unittest.mock import patch from datetime import datetime as dt - class TestCreateDimDesign: def test_dim_design_returns_dataframe(self): d = { @@ -135,7 +134,6 @@ class TestCreateDimCounterparty: class TestCreateDimCurrency: - def test_dim_currency_returns_columns_and_values(self): nones = [None, None, None] d = { @@ -246,4 +244,3 @@ class TestCreateDimTransaction: } result = create_dim_transaction(dict_df) assert list(result.columns) == ["transaction_id", "some_other_id"] - -- cgit v1.2.3