Skip to content

Commit fa96913

Browse files
authored
PR #97: Titanic Survival Prediction
Add Titanic Survival Prediction project Merge pull request #97 from zain-cs/titanic-survival-prediction
2 parents d2d8ab0 + c87bf10 commit fa96913

6 files changed

Lines changed: 181 additions & 0 deletions

File tree

Lines changed: 38 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,38 @@
1+
# Titanic Survival Predictor 🚢
2+
3+
A machine learning model that predicts whether a Titanic passenger would have survived, based on features like age, gender, class, and fare.
4+
5+
## Results
6+
| Model | Accuracy |
7+
|---|---|
8+
| Logistic Regression | 81.01% |
9+
| Random Forest | **82.12%** |
10+
11+
## Project Structure
12+
```
13+
titanic-survival-predictor/
14+
├── data/ # Raw dataset
15+
├── explore.py # Data exploration & visualization
16+
├── preprocess.py # Data cleaning & feature engineering
17+
├── train.py # Model training & evaluation
18+
├── predict.py # Make predictions on new passengers
19+
└── requirements.txt # Dependencies
20+
```
21+
22+
## Setup
23+
```bash
24+
pip install -r requirements.txt
25+
```
26+
## Dataset
27+
Download `titanic.csv` from [here](https://github.com/datasciencedojo/datasets/blob/master/Titanic.csv) and place it inside the `data/` folder.
28+
29+
## Usage
30+
```bash
31+
python predict.py
32+
```
33+
34+
## What I learned
35+
- Data cleaning and handling missing values
36+
- Feature engineering and label encoding
37+
- Training and comparing ML models
38+
- Evaluating with confusion matrix and classification report
Lines changed: 32 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,32 @@
1+
import pandas as pd
2+
import matplotlib.pyplot as plt
3+
import seaborn as sns
4+
5+
# Load the data
6+
df = pd.read_csv('data/titanic.csv')
7+
8+
# --- Basic overview ---
9+
print("Shape:", df.shape)
10+
print("\nFirst 5 rows:")
11+
print(df.head())
12+
print("\nColumn info:")
13+
print(df.info())
14+
print("\nMissing values:")
15+
print(df.isnull().sum())
16+
17+
# --- Survival rate by key features ---
18+
fig, axes = plt.subplots(1, 3, figsize=(14, 4))
19+
20+
sns.barplot(x='Sex', y='Survived', data=df, ax=axes[0])
21+
axes[0].set_title('Survival by Sex')
22+
23+
sns.barplot(x='Pclass', y='Survived', data=df, ax=axes[1])
24+
axes[1].set_title('Survival by Passenger Class')
25+
26+
sns.histplot(data=df, x='Age', hue='Survived', bins=30, ax=axes[2])
27+
axes[2].set_title('Survival by Age')
28+
29+
plt.tight_layout()
30+
plt.savefig('exploration.png')
31+
plt.show()
32+
print("\nChart saved as exploration.png")
Lines changed: 23 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,23 @@
1+
import pandas as pd
2+
import joblib
3+
4+
# Load the saved model
5+
model = joblib.load('models/titanic_model.pkl')
6+
7+
# Example passenger (you can change these values)
8+
passenger = pd.DataFrame([{
9+
'Pclass': 3, # 1=First, 2=Second, 3=Third class
10+
'Sex': 1, # 1=Male, 0=Female
11+
'Age': 22,
12+
'SibSp': 1, # siblings/spouses aboard
13+
'Parch': 0, # parents/children aboard
14+
'Fare': 7.25,
15+
'Embarked': 2 # 0=Cherbourg, 1=Queenstown, 2=Southampton
16+
}])
17+
18+
result = model.predict(passenger)[0]
19+
probability = model.predict_proba(passenger)[0]
20+
21+
print(f"Prediction: {'Survived' if result == 1 else 'Did not survive'}")
22+
print(f"Survival probability: {probability[1]:.2%}")
23+
print(f"Death probability: {probability[0]:.2%}")
Lines changed: 30 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,30 @@
1+
import pandas as pd
2+
from sklearn.preprocessing import LabelEncoder
3+
4+
def preprocess(df):
5+
# Drop columns that are useless for prediction
6+
df = df.drop(columns=['PassengerId', 'Name', 'Ticket', 'Cabin'])
7+
8+
# Fill missing Age with median (more robust than mean)
9+
df['Age'] = df['Age'].fillna(df['Age'].median())
10+
11+
# Fill missing Embarked with most common value
12+
df['Embarked'] = df['Embarked'].fillna(df['Embarked'].mode()[0])
13+
14+
# Convert Sex and Embarked from text to numbers
15+
# Machine learning models only understand numbers, not "male"/"female"
16+
le = LabelEncoder()
17+
df['Sex'] = le.fit_transform(df['Sex']) # male=1, female=0
18+
df['Embarked'] = le.fit_transform(df['Embarked']) # S=2, C=0, Q=1
19+
20+
return df
21+
22+
if __name__ == '__main__':
23+
df = pd.read_csv('data/titanic.csv')
24+
df = preprocess(df)
25+
26+
print("Cleaned shape:", df.shape)
27+
print("\nMissing values after cleaning:")
28+
print(df.isnull().sum())
29+
print("\nFirst 5 rows after cleaning:")
30+
print(df.head())
Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,6 @@
1+
pandas
2+
numpy
3+
scikit-learn
4+
matplotlib
5+
seaborn
6+
joblib
Lines changed: 52 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,52 @@
1+
import pandas as pd
2+
from sklearn.model_selection import train_test_split
3+
from sklearn.linear_model import LogisticRegression
4+
from sklearn.ensemble import RandomForestClassifier
5+
from sklearn.metrics import accuracy_score, classification_report, confusion_matrix
6+
import joblib
7+
import os
8+
from preprocess import preprocess
9+
10+
# Load and preprocess
11+
df = pd.read_csv('data/titanic.csv')
12+
df = preprocess(df)
13+
14+
# Split features and target
15+
X = df.drop(columns=['Survived']) # everything except what we're predicting
16+
y = df['Survived'] # what we're predicting
17+
18+
# 80% for training, 20% for testing
19+
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
20+
21+
# --- Model 1: Logistic Regression ---
22+
lr = LogisticRegression(max_iter=200)
23+
lr.fit(X_train, y_train)
24+
lr_preds = lr.predict(X_test)
25+
lr_acc = accuracy_score(y_test, lr_preds)
26+
27+
# --- Model 2: Random Forest ---
28+
rf = RandomForestClassifier(n_estimators=100, random_state=42)
29+
rf.fit(X_train, y_train)
30+
rf_preds = rf.predict(X_test)
31+
rf_acc = accuracy_score(y_test, rf_preds)
32+
33+
# --- Compare ---
34+
print(f"Logistic Regression Accuracy: {lr_acc:.2%}")
35+
print(f"Random Forest Accuracy: {rf_acc:.2%}")
36+
37+
# Pick the better model
38+
best_model = rf if rf_acc >= lr_acc else lr
39+
best_name = "Random Forest" if rf_acc >= lr_acc else "Logistic Regression"
40+
print(f"\nBest model: {best_name}")
41+
42+
# --- Detailed report for best model ---
43+
best_preds = rf_preds if rf_acc >= lr_acc else lr_preds
44+
print("\nClassification Report:")
45+
print(classification_report(y_test, best_preds))
46+
print("Confusion Matrix:")
47+
print(confusion_matrix(y_test, best_preds))
48+
49+
# --- Save the best model ---
50+
os.makedirs('models', exist_ok=True)
51+
joblib.dump(best_model, 'models/titanic_model.pkl')
52+
print("\nModel saved to models/titanic_model.pkl")

0 commit comments

Comments
 (0)