From 072cd59276ab4456e1c0b1931e8fe231f78b7a2e Mon Sep 17 00:00:00 2001 From: Lord_Devi Date: Mon, 16 Sep 2024 06:19:53 -0400 Subject: [PATCH] Added fix-reviews-in-business-name.py. --- bin/process-stage-6.py | 7 +- bin/stage-6/fix-reviews-in-business-name.py | 115 ++++++++++++++++++++ 2 files changed, 117 insertions(+), 5 deletions(-) create mode 100755 bin/stage-6/fix-reviews-in-business-name.py diff --git a/bin/process-stage-6.py b/bin/process-stage-6.py index 50ae64e..b111f93 100755 --- a/bin/process-stage-6.py +++ b/bin/process-stage-6.py @@ -1,6 +1,5 @@ #!/usr/bin/env python # Script Name: process-stage-6 - import os import subprocess from pyfiglet import Figlet @@ -41,7 +40,8 @@ def run_scripts(): os.path.join(BIN_STAGE_6_DIR, "uniqify-pass-2.py"), os.path.join(BIN_STAGE_6_DIR, "services-gbp-reputation-management.py"), os.path.join(BIN_STAGE_6_DIR, "services-gbp-optimization.py"), - os.path.join(BIN_STAGE_6_DIR, "services-needs-website.py") + os.path.join(BIN_STAGE_6_DIR, "services-needs-website.py"), + os.path.join(BIN_STAGE_6_DIR, "fix-reviews-in-business-name.py") ] # Run scripts that do not require arguments @@ -59,14 +59,11 @@ def main(): figlet = Figlet(font='slant') script_name = "process-stage-6".replace("-", " ").title() print(figlet.renderText(script_name)) - print("Running scripts for Stage 6...\n") run_scripts() - final_spinner = Halo(spinner='dots', color='green') final_spinner.start() final_spinner.succeed("All scripts completed.") if __name__ == '__main__': main() - diff --git a/bin/stage-6/fix-reviews-in-business-name.py b/bin/stage-6/fix-reviews-in-business-name.py new file mode 100755 index 0000000..d18ce83 --- /dev/null +++ b/bin/stage-6/fix-reviews-in-business-name.py @@ -0,0 +1,115 @@ +#!/usr/bin/env python +# Script Name: fix-reviews-in-business-name + +import os +import csv +import re +from collections import defaultdict +from pyfiglet import Figlet +from halo import Halo + +def find_project_root(current_path): + """Find the project root by locating the 'bin' directory.""" + while current_path != os.path.dirname(current_path): + if os.path.basename(current_path) == 'bin': + return os.path.dirname(current_path) + current_path = os.path.dirname(current_path) + raise FileNotFoundError("Could not find 'bin' directory in the path hierarchy.") + +# Define the paths based on the project details +script_path = os.path.abspath(__file__) +PROJECT_ROOT = find_project_root(script_path) +STAGE_6_DIRECTORY = os.path.join(PROJECT_ROOT, 'current-data', '.data', 'stage-6') +INPUT_CSV = os.path.join(STAGE_6_DIRECTORY, '20-services-needs-website.csv') +OUTPUT_CSV = os.path.join(STAGE_6_DIRECTORY, '21-fix-reviews-in-business-name.csv') + +def is_review_rating(name): + """Check if the given string matches the review rating pattern.""" + return re.match(r'^[1-5]\.[0-9]$', name) is not None + +def get_most_common_name(names): + """Return the most common non-empty name from the list.""" + name_counts = defaultdict(int) + for name in names: + if name.strip() and not is_review_rating(name.strip()): + name_counts[name.strip()] += 1 + return max(name_counts, key=name_counts.get) if name_counts else '' + +def fix_business_names(): + """Fix business names that are actually review ratings.""" + uuid_to_rows = defaultdict(list) + category_count = defaultdict(int) + municipality_count = defaultdict(int) + unknown_count = 0 + fixed_count = 0 + + # Read the input CSV and group rows by Business UUID + with open(INPUT_CSV, 'r') as infile: + reader = csv.DictReader(infile) + headers = reader.fieldnames + rows = list(reader) + + for row in rows: + business_uuid = row['Business UUID'].strip() + if business_uuid: + uuid_to_rows[business_uuid].append(row) + + # Process each group of rows with the same Business UUID + for uuid, grouped_rows in uuid_to_rows.items(): + business_names = [row['Business Name'] for row in grouped_rows if not is_review_rating(row['Business Name'])] + root_domains = [row['Root Domain'] for row in grouped_rows if row['Root Domain'].strip()] + categories = [row['GBP Business Category'] for row in grouped_rows if row['GBP Business Category'].strip()] + municipalities = [row['GBP Location Municipality'] for row in grouped_rows if row['GBP Location Municipality'].strip()] + + new_name = '' + if business_names: + new_name = get_most_common_name(business_names) + elif root_domains: + new_name = root_domains[0] + elif categories: + category = get_most_common_name(categories) + category_count[category] += 1 + new_name = f"Unknown {category} {category_count[category]}" + elif municipalities: + municipality = get_most_common_name(municipalities) + municipality_count[municipality] += 1 + new_name = f"Unknown {municipality} Business {municipality_count[municipality]}" + else: + unknown_count += 1 + new_name = f"Unknown Business {unknown_count}" + + for row in grouped_rows: + if is_review_rating(row['Business Name']): + row['Business Name'] = new_name + fixed_count += 1 + + # Write to the output CSV + with open(OUTPUT_CSV, 'w', newline='') as outfile: + writer = csv.DictWriter(outfile, fieldnames=headers) + writer.writeheader() + writer.writerows(rows) + + return len(rows), fixed_count + +def main(): + figlet = Figlet(font='slant') + script_name = "Fix Reviews in Business Name" + print(figlet.renderText(script_name)) + + # Check if the output file already exists and delete it if it does + if os.path.exists(OUTPUT_CSV): + os.remove(OUTPUT_CSV) + print(f"Existing output file {OUTPUT_CSV} found and deleted.") + + spinner = Halo(text='Fixing reviews in business names', spinner='dots') + spinner.start() + total_rows, fixed_count = fix_business_names() + spinner.succeed('Business names fixed successfully.') + + print(f"Processed input CSV: {INPUT_CSV}") + print(f"Output CSV: {OUTPUT_CSV}") + print(f"Total rows processed: {total_rows}") + print(f"Total business names fixed: {fixed_count}") + +if __name__ == '__main__': + main()