# -*- coding: utf-8 -*-
"""SIC 33

Automatically generated by Colab.

Original file is located at
    https://colab.research.google.com/drive/16pnMBfxzZHVqM1rDM0u1FjUI49WDukUj

#Stock Price Movement Predictor
## Random Forest Classifier for Next-Day Stock Direction **bold text**

This notebook builds a complete machine learning pipeline to predict whether a stock’s next-day closing price will move up or down using the “World Stock Prices – Daily Updating” dataset from Kaggle. The data is cleaned, time-sorted, and processed separately for each company to maintain accurate time-series behavior.

A wide range of financial indicators is engineered, including price range, intraday change, moving averages, volume trends, daily returns, volatility, RSI, MACD, Bollinger Bands, and momentum/ROC features. The target label marks whether the next day’s closing price is higher than the current day’s.

After feature creation, the data is split using stratified sampling and scaled using StandardScaler. A regularized Random Forest Classifier is trained with constraints such as limited depth, minimum samples per split/leaf, feature sampling, and subsampling to prevent overfitting.

Model performance is evaluated using accuracy, classification reports, confusion matrices, and ROC-AUC. Feature importance plots, ROC curves, and prediction samples are generated for deeper insight.

The notebook also performs company-level analysis, reporting accuracy, confidence, and the model’s latest prediction for each stock. Overall, the project provides a practical stock-movement prediction pipeline with clear evaluation and extendable structure for future improvements.

###The team members for Group 33 are:

Dina Abdelnaser Mohammed

Nicelin Printo

THARISH J

Mariam lulu mohammed

Raia Mohammed Gueye

Naman Chawla

#Library import
"""

import pandas as pd
import numpy as np
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from sklearn.ensemble import RandomForestClassifier
from sklearn.metrics import accuracy_score, classification_report, confusion_matrix, roc_auc_score
import matplotlib.pyplot as plt
import seaborn as sns
from datetime import datetime
import warnings
warnings.filterwarnings('ignore')

"""#Download and explore data"""

print("=" * 60)
print("STOCK PRICE MOVEMENT PREDICTOR")
print("=" * 60)

data = pd.read_csv("World-Stock-Prices-Dataset.csv")

print(f"\n✓ Dataset loaded: {data.shape[0]} rows, {data.shape[1]} columns")
print(f"✓ Date range: {data['Date'].min()} to {data['Date'].max()}")

data['Date'] = pd.to_datetime(data['Date'])
data = data.sort_values(['Brand_Name', 'Date']).reset_index(drop=True)

print(f"\n  Companies in dataset: {data['Brand_Name'].nunique()}")
print("\nCompanies breakdown:")
print(data['Brand_Name'].value_counts())

"""#Function to create properties for each company"""

def create_features_by_company(df):
    features_list = []

    for company in df['Brand_Name'].unique():
        company_data = df[df['Brand_Name'] == company].copy()
        company_data = company_data.sort_values('Date').reset_index(drop=True)

        if len(company_data) < 20:
            continue

        company_data['Price_Range'] = company_data['High'] - company_data['Low']
        company_data['Price_Change'] = company_data['Close'] - company_data['Open']
        company_data['Volume_MA_5'] = company_data['Volume'].rolling(window=5, min_periods=1).mean()

        company_data['Daily_Return'] = company_data['Close'].pct_change() * 100
        company_data['MA_5'] = company_data['Close'].rolling(window=5, min_periods=1).mean()
        company_data['MA_10'] = company_data['Close'].rolling(window=10, min_periods=1).mean()

        company_data['Volatility'] = company_data['Daily_Return'].rolling(window=5, min_periods=1).std()

        delta = company_data['Close'].diff()
        gain = (delta.where(delta > 0, 0)).rolling(window=14, min_periods=1).mean()
        loss = (-delta.where(delta < 0, 0)).rolling(window=14, min_periods=1).mean()
        rs = gain / (loss + 1e-10)
        company_data['RSI'] = 100 - (100 / (1 + rs))

        company_data['EMA_12'] = company_data['Close'].ewm(span=12, adjust=False).mean()
        company_data['EMA_26'] = company_data['Close'].ewm(span=26, adjust=False).mean()

        company_data['MACD'] = company_data['EMA_12'] - company_data['EMA_26']
        company_data['Signal'] = company_data['MACD'].ewm(span=9, adjust=False).mean()

        bb_mid = company_data['Close'].rolling(window=20, min_periods=1).mean()
        bb_std = company_data['Close'].rolling(window=20, min_periods=1).std()
        company_data['BB_Upper'] = bb_mid + 2 * bb_std
        company_data['BB_Lower'] = bb_mid - 2 * bb_std

        company_data['Momentum_10'] = company_data['Close'] - company_data['Close'].shift(10)
        company_data['ROC_10'] = company_data['Close'].pct_change(periods=10) * 100

        company_data['Target'] = (company_data['Close'].shift(-1) > company_data['Close']).astype(int)
        company_data = company_data[:-1]

        company_data = company_data.replace([np.inf, -np.inf], np.nan).fillna(0)

        features_list.append(company_data)

    result = pd.concat(features_list, ignore_index=True)
    return result

"""#Data processing and property creation"""

data_processed = create_features_by_company(data)
print(f"\n Features created for {len(data_processed)} data points")
print(f" Target distribution:\n{data_processed['Target'].value_counts()}")

"""#Data preparation for training"""

feature_cols = [
    'Open', 'High', 'Low', 'Close', 'Volume',
    'Price_Range', 'Price_Change', 'Volume_MA_5',
    'Daily_Return', 'MA_5', 'MA_10', 'Volatility', 'RSI',
    'EMA_12', 'EMA_26', 'MACD', 'Signal',
    'BB_Upper', 'BB_Lower', 'Momentum_10', 'ROC_10',
    'Dividends', 'Stock Splits', 'Capital Gains'
]

X = data_processed[feature_cols].copy()
X = X.replace([np.inf, -np.inf], np.nan).fillna(0)

y = data_processed['Target'].copy()

X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.2, random_state=42, stratify=y
)

scaler = StandardScaler()
X_train_scaled = scaler.fit_transform(X_train)
X_test_scaled = scaler.transform(X_test)

print("Total samples after feature engineering:", len(X))
print("Training set size:", len(X_train))
print("Test set size:", len(X_test))

"""#Random Forest Model Training"""

rf = RandomForestClassifier(
    n_estimators=250,
    max_depth=10,
    min_samples_split=5,
    min_samples_leaf=2,
    max_features='sqrt',
    max_samples=0.9,
    class_weight='balanced_subsample',
    random_state=42,
    n_jobs=-1
)

rf.fit(X_train_scaled, y_train)

"""#Model evaluation"""

y_train_pred = rf.predict(X_train_scaled)
y_test_pred = rf.predict(X_test_scaled)
y_test_pred_proba = rf.predict_proba(X_test_scaled)[:, 1]

train_accuracy = accuracy_score(y_train, y_train_pred)
test_accuracy = accuracy_score(y_test, y_test_pred)
overfitting_gap = train_accuracy - test_accuracy

roc_auc = roc_auc_score(y_test, y_test_pred_proba)

print(classification_report(y_test, y_test_pred, target_names=['DOWN ↓', 'UP ↑']))

cm = confusion_matrix(y_test, y_test_pred)
print(cm)

"""## SAMPLE PREDICTIONS TABLE"""

sample_indices = np.random.choice(len(X_test), min(10, len(X_test)), replace=False)

rows = []
for idx in sample_indices:
    actual_label = "UP" if y_test.iloc[idx] == 1 else "DOWN"
    predicted_label = "UP" if y_test_pred[idx] == 1 else "DOWN"
    confidence = float(y_test_pred_proba[idx])
    correct = "YES" if predicted_label == actual_label else "NO"

    rows.append({
        "Actual": actual_label,
        "Predicted": predicted_label,
        "Confidence": f"{confidence:.2%}",
        "Correct": correct
    })

sample_df = pd.DataFrame(rows, columns=["Actual", "Predicted", "Confidence", "Correct"])

print("\nSample Predictions (10 examples from test set):")
print(sample_df.to_string(index=False))

"""#Extracting the importance of the characteristics"""

feature_importance_df = pd.DataFrame({
    'Feature': feature_cols,
    'Importance': rf.feature_importances_
}).sort_values('Importance', ascending=False)

fig, ax = plt.subplots(figsize=(12, 6))
ax.barh(feature_importance_df['Feature'][:10], feature_importance_df['Importance'][:10])
ax.set_xlabel('Importance Score')
ax.set_title('Top 10 Feature Importance - Random Forest')
ax.invert_yaxis()
plt.tight_layout()
plt.savefig('feature_importance.png', dpi=100, bbox_inches='tight')

"""#Model performance analysis for each company"""

X_full = X.copy()
X_full_scaled = scaler.transform(X_full)
data_processed['Prediction'] = rf.predict(X_full_scaled)
data_processed['Prediction_Probability'] = rf.predict_proba(X_full_scaled)[:, 1]

company_analysis = []
for company in data_processed['Brand_Name'].unique():
    company_mask = data_processed['Brand_Name'] == company
    company_data = data_processed[company_mask]

    if len(company_data) > 0:
        accuracy_company = accuracy_score(
            company_data['Target'],
            company_data['Prediction']
        )

        avg_prob = company_data['Prediction_Probability'].mean()

        company_analysis.append({
            'Company': company,
            'Samples': len(company_data),
            'Accuracy': accuracy_company,
            'Avg_Confidence': avg_prob,
            'Latest_Close': company_data['Close'].iloc[-1],
            'Latest_Prediction': 'UP ↑' if company_data['Prediction'].iloc[-1] == 1 else 'DOWN ↓'
        })

company_df = pd.DataFrame(company_analysis).sort_values('Accuracy', ascending=False)
print(company_df.head(15)[['Company', 'Samples', 'Accuracy', 'Avg_Confidence', 'Latest_Prediction']])

"""#Accuracy per Company Bar Chart"""

plt.figure(figsize=(12, 6))
sns.barplot(data=company_df, x='Company', y='Accuracy')
plt.xticks(rotation=90)
plt.title('Accuracy per Company')
plt.tight_layout()
plt.savefig('company_accuracy.png', dpi=100)

"""#Confusion Matrix"""

fig, ax = plt.subplots(figsize=(8, 6))
sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', ax=ax,
            xticklabels=['DOWN ↓', 'UP ↑'], yticklabels=['DOWN ↓', 'UP ↑'])
ax.set_ylabel('True Label')
ax.set_xlabel('Predicted Label')
ax.set_title('Confusion Matrix - Stock Price Prediction')
plt.tight_layout()
plt.savefig('confusion_matrix.png', dpi=100, bbox_inches='tight')

"""#ROC Curve"""

from sklearn.metrics import roc_curve

fpr, tpr, thresholds = roc_curve(y_test, y_test_pred_proba)

plt.figure(figsize=(8, 6))
plt.plot(fpr, tpr, label=f'AUC = {roc_auc:.3f}')
plt.plot([0, 1], [0, 1], linestyle='--')
plt.xlabel('False Positive Rate')
plt.ylabel('True Positive Rate')
plt.title('ROC Curve')
plt.legend()
plt.tight_layout()
plt.savefig('roc_curve.png', dpi=100)