diff options
| author | Ang Bel <anzelikabelotelova@Anzelikas-MacBook-Air.local> | 2024-08-27 17:00:04 +0100 |
|---|---|---|
| committer | Ellie <ecsymonds@gmail.com> | 2024-08-28 09:12:00 +0100 |
| commit | efab1eccd4e2f0a8069ff4f1c968807a9c1ce05f (patch) | |
| tree | 3d3a41536f00df517f598e04cb94cc080cb3f105 /src/transform_lambda.py | |
| parent | 8588d4b318d7732d33a59bc6c8b93870310668c5 (diff) | |
| download | de-project-bentley-efab1eccd4e2f0a8069ff4f1c968807a9c1ce05f.tar.gz de-project-bentley-efab1eccd4e2f0a8069ff4f1c968807a9c1ce05f.zip | |
test: transform refactoring - it now loads parquet files into s3 bucket
Diffstat (limited to 'src/transform_lambda.py')
| -rw-r--r-- | src/transform_lambda.py | 6 |
1 files changed, 3 insertions, 3 deletions
diff --git a/src/transform_lambda.py b/src/transform_lambda.py index 2cd9272..ccf90e5 100644 --- a/src/transform_lambda.py +++ b/src/transform_lambda.py @@ -117,7 +117,7 @@ def process_to_parquet_and_upload_to_s3( parquet_file = df.to_parquet( f"{table_name}.parquet", engine="pyarrow" ) # or fastparquet - client.upload_file(parquet_file, bucket, f"{table_name}.parquet") + client.upload_file(f"{table_name}.parquet", bucket, f"{table_name}.parquet") #changed parquet_file variable to the file name status["uploaded"].append(table_name) for table_name, df in mutable_df_dict.items(): @@ -127,7 +127,7 @@ def process_to_parquet_and_upload_to_s3( parquet_file = df.to_parquet( f"{table_name}.parquet", engine="pyarrow" ) # or fastparquet - client.upload_file(parquet_file, bucket, s3_key) + client.upload_file(f"{table_name}.parquet", bucket, s3_key) status["uploaded"].append(table_name) return status @@ -203,7 +203,7 @@ def list_existing_s3_files(bucket_name, client=boto3.client("s3")): existing_files = [obj["Key"] for obj in response["Contents"]] else: logger.error("The bucket is empty") - return None + return [] #changed from None to [] so it is an iterable except ClientError as e: logger.error(f"Error listing S3 objects: {e}") |
