import pandas as pd
import numpy as np
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from sklearn.metrics import mean_squared_error
from sklearn.linear_model import LinearRegression

# Generate synthetic data
np.random.seed(42)

# =============================================================================
# random: This is a submodule in NumPy that provides functions for generating 
# pseudo-random numbers.
# 
# seed(42): The number 42 is an arbitrary choice for the seed.it means that every 
# time you run the program, you will get the same sequence of random numbers.
# This is useful for debugging and ensuring that your results are reproducible
# =============================================================================

# Generate random customer data
# =============================================================================
# I will be using backslash (\) quite a few times to specify that the code is 
# continuing in the next line
# =============================================================================

customer_data = pd.DataFrame({
    'customer_id': np.arange(1, 501),
    'customer_age': np.random.randint(18, 65, size=500),
    'customer_gender': np.random.choice(['Male', 'Female'], size=500),
    'customer_location': np.random.choice(['Urban', 'Suburban', 'Rural'], \
                                          size=500)
})

# Generate random product data
product_data = pd.DataFrame({
    'product_id': np.arange(1, 11),
    'product_category': np.random.choice(['Electronics', 'Clothing', 'Home', \
                                          'Sports'], size=10),
    'product_brand': np.random.choice(['Brand_A', 'Brand_B', 'Brand_C'], size=10),
    'product_price': np.random.uniform(50, 500, size=10),
    'min_price': np.random.uniform(40,4900, size=10),
    'cm': np.random.uniform(50,500, size = 10)/1000
})

# Generate random sales data
sales_data = pd.DataFrame({
    'customer_id': np.random.choice(np.arange(1, 501), size=1000),
    'product_id': np.random.choice(np.arange(1, 11), size=1000),
    'sales_quantity': np.random.randint(1, 10, size=1000)
})

# Merge datasets to consolidate all the data and join on primary keys 
all_data = pd.merge(customer_data, sales_data, on="customer_id")
all_data = pd.merge(all_data, product_data, on="product_id")