Kiuyha's picture
Upload app.py with huggingface_hub
e356cf8 verified
Raw
History Blame
8.6 kB
import streamlit as st
import pandas as pd
import numpy as np
import joblib
import plotly.express as px
import re
import string
import nltk
import os
from datetime import datetime
from nltk.corpus import stopwords
st.set_page_config(page_title="Tokopedia Project Dashboard", layout="wide")
ARTIFACTS_DIR = 'artifacts'
@st.cache_resource
def load_resources():
try:
nltk.download('stopwords')
indo_stopwords = set(stopwords.words('indonesian'))
custom_stopwords = {
'yg', 'dg', 'dgn', 'ny', 'nya', 'kalo', 'klo', 'gan', 'sis', 'kak', 'bro',
'bosh', 'boss', 'nomor', 'wa', 'hub', 'cod', 'ready', 'stock', 'stok',
'promo', 'murah', 'diskon', 'termurah', 'terlaris', 'bestseller'
}
final_stopwords = indo_stopwords.union(custom_stopwords)
# Load core artifacts
vectorizer = joblib.load(os.path.join(ARTIFACTS_DIR, 'tfidf_vectorizer.pkl'))
# Load Price Artifacts
price_preprocess = joblib.load(os.path.join(ARTIFACTS_DIR, 'price_preprocessor.pkl'))
price_cols = joblib.load(os.path.join(ARTIFACTS_DIR, 'price_columns.pkl'))
# Load Models
price_models = {
"Random Forest": joblib.load(os.path.join(ARTIFACTS_DIR, 'Random_Forest_price.pkl')),
"XGBoost": joblib.load(os.path.join(ARTIFACTS_DIR, 'XGBoost_price.pkl')),
"MLP": joblib.load(os.path.join(ARTIFACTS_DIR, 'MLP_price.pkl'))
}
return vectorizer, price_preprocess, price_cols, price_models, final_stopwords
except Exception as e:
st.error(f"Error loading resources: {e}")
return None, None, None, None, None
(vectorizer, price_preprocess, price_cols,
price_models, final_stopwords) = load_resources()
@st.cache_data
def get_categories():
columns = ['Other']
if price_cols:
cat_cols = [c for c in price_cols if c.startswith('cat_')]
columns = [c.replace('cat_', '') for c in cat_cols]
return sorted(columns)
def clean_text(text):
if not isinstance(text, str): return ""
text = text.lower()
text = re.sub(r'[^a-z0-9\s]', ' ', text)
text = re.sub(r'\s+', ' ', text).strip()
words = text.split()
return " ".join([w for w in words if w not in final_stopwords])
def get_date_features(date_obj, prefix):
return {
f"{prefix}_dayofweek": date_obj.weekday(),
f"{prefix}_month": date_obj.month,
f"{prefix}_is_weekend": 1 if date_obj.weekday() >= 5 else 0
}
def prepare_raw_dataframe(user_input, target_columns, vectorizer):
input_df = pd.DataFrame(0, index=[0], columns=target_columns)
# Text
combined_text = clean_text(user_input.get('product_name', '')) + " " + clean_text(user_input.get('description', ''))
tfidf_matrix = vectorizer.transform([combined_text])
tfidf_df = pd.DataFrame(tfidf_matrix.toarray(), columns=[f"word_{w}" for w in vectorizer.get_feature_names_out()])
common_text_cols = input_df.columns.intersection(tfidf_df.columns)
input_df[common_text_cols] = tfidf_df[common_text_cols]
# Categories
cat_col = f"cat_{user_input.get('category', 'Other')}"
if cat_col in input_df.columns:
input_df[cat_col] = 1
elif 'cat_Other' in input_df.columns:
input_df['cat_Other'] = 1
# Numerical
for key, value in user_input.items():
if key in input_df.columns:
input_df[key] = value
return input_df
def load_and_plot_importance(model_name, target_name):
filename = f"importance_data_{target_name}_{model_name.replace(' ', '_')}.csv"
path = os.path.join(ARTIFACTS_DIR, filename)
if os.path.exists(path):
df = pd.read_csv(path)
fig = px.bar(
df.head(15), x='Importance', y='Feature', orientation='h', error_x='Std',
title=f"Feature Importance ({model_name}) - {target_name.title()}", color='Importance'
)
fig.update_layout(yaxis=dict(autorange="reversed"))
st.plotly_chart(fig)
st.title("🛍️ Advanced E-Commerce Predictor")
with st.sidebar:
st.header("1. Product Info")
p_name = st.text_input("Product Name", "Samsung Galaxy S24 Ultra")
p_desc = st.text_area("Description", "Original SEIN, Garansi Resmi 1 Tahun")
cat_options = get_categories()
category = st.selectbox("Category", cat_options)
condition = st.radio("Condition", ["New", "Used"])
is_regular = st.checkbox("Regular Merchant?", value=True)
is_discount = st.checkbox("Is Discounted?", value=False)
st.header("2. Physical & Stock")
col1, col2 = st.columns(2)
with col1:
weight = st.number_input("Weight (g)", value=500)
min_order = st.number_input("Min Order", value=1)
with col2:
max_order = st.number_input("Max Order", value=100)
stock = st.number_input("Stock", value=50)
sold_input = st.number_input("Sold (Current)", value=0)
st.header("3. Media")
vid_count = st.number_input("Video Count", 0)
img_count = st.number_input("Image Count", 1)
with st.expander("Detailed Ratings", expanded=True):
c1, c2, c3, c4, c5 = st.columns(5)
r5 = c1.number_input("5 Star Count", value=10)
r4 = c2.number_input("4 Star Count", value=2)
r3 = c3.number_input("3 Star Count", value=0)
r2 = c4.number_input("2 Star Count", value=0)
r1 = c5.number_input("1 Star Count", value=0)
total_rating = r5 + r4 + r3 + r2 + r1
c_qual, c_srv, c_ship = st.columns(3)
rev_qual = c_qual.number_input("Reviews: Quality", 0)
rev_srv = c_srv.number_input("Reviews: Service", 0)
rev_ship = c_ship.number_input("Reviews: Shipping", 0)
with st.expander("Shipping & Shop Metadata", expanded=False):
c1, c2 = st.columns(2)
with c1:
ship_sameday = st.checkbox("Same Day", True)
ship_reg = st.checkbox("Regular", True)
ship_cargo = st.checkbox("Cargo", False)
ship_eco = st.checkbox("Economy", True)
with c2:
shop_rating = st.slider("Shop Rating Score", 0.0, 5.0, 4.8)
shop_age = st.number_input("Shop Age (Days)", 365)
shop_pop = st.number_input("Shop City Popularity", 0.0)
date_listing = st.date_input("Listing Created", datetime.now())
date_shop_open = st.date_input("Shop Open Since", datetime(2020, 1, 1))
feat_created = get_date_features(date_listing, "created_at")
feat_shop = get_date_features(date_shop_open, "shop_open_since")
listing_age_calc = (datetime.now().date() - date_listing).days
total_shipping = sum([ship_sameday, ship_reg, ship_cargo, ship_eco])
base_input = {
'product_name': p_name, 'description': p_desc, 'category': category,
'condition_encoded': 1 if condition == 'New' else 0,
'is_regular_merchant': 1 if is_regular else 0,
'is_discount': 1 if is_discount else 0,
'weight_grams': weight,
'min_order': min_order, 'max_order': max_order,
'stock': stock,
'video_count': vid_count, 'image_count': img_count,
'can_shipping_sameday': 1 if ship_sameday else 0,
'can_shipping_regular': 1 if ship_reg else 0,
'can_shipping_cargo': 1 if ship_cargo else 0,
'can_shipping_economy': 1 if ship_eco else 0,
'total_shipping_types': total_shipping,
'shop_rating_score': shop_rating,
'shop_age_days': shop_age,
'shop_city_popularity': shop_pop,
'listing_age_days': listing_age_calc,
'product_rating_5_star_count': r5,
'product_rating_4_star_count': r4,
'product_rating_3_star_count': r3,
'product_rating_2_star_count': r2,
'product_rating_1_star_count': r1,
'total_product_rating': total_rating,
'total_review_about_kualitas': rev_qual,
'total_review_about_pelayanan': rev_srv,
'total_review_about_pengiriman': rev_ship,
**feat_created, **feat_shop
}
if sold_input:
base_input['sold'] = sold_input
# Log Transforms (Stock & Sold)
base_input['Log_stock'] = np.log1p(stock)
base_input['Log_sold'] = np.log1p(sold_input)
st.divider()
st.subheader("💰 Price Prediction")
if price_models:
model_name = st.selectbox("Select Price Model", list(price_models.keys()), key="p_model")
if st.button("Predict Price", type="primary"):
raw_df = prepare_raw_dataframe(base_input, price_cols, vectorizer)
X_scaled = price_preprocess.transform(raw_df)
log_pred = price_models[model_name].predict(X_scaled)[0]
price_pred = np.expm1(log_pred)
st.success(f"Estimated Price: **Rp {price_pred:,.0f}**")
st.markdown("### Feature Importance")
load_and_plot_importance(model_name, "price")
else:
st.error("Price models not loaded.")