Turning Learners Into Developers
Codekilla
CODEKILLA
// projects.cyber-security-ai

10 Best Cyber Security with AI Projects

Hand-picked and ordered easiest → hardest — each with complete code and expected output. Build these to turn lessons into a portfolio.

Back to Cyber Security with AI course
// solution.code
security
# AI-Powered Phishing Email Detector
# Uses TfidfVectorizer and RandomForest for email classification

import re
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.ensemble import RandomForestClassifier
from sklearn.model_selection import train_test_split
from sklearn.metrics import classification_report, accuracy_score
import numpy as np

# Sample training data: (email_text, is_phishing)
training_data = [
    ("Congratulations! You've won $1,000,000. Click here: http://suspicious-site.xyz", 1),
    ("Your account has been compromised. Verify immediately at http://paypal-verify.tk", 1),
    ("URGENT: Your bank account will be closed. Update info at http://bank-update.ml", 1),
    ("Dear user, click here to claim your prize from Nigerian prince", 1),
    ("Meeting scheduled for tomorrow at 2 PM. Please confirm your attendance.", 0),
    ("Here are the quarterly reports you requested. Best regards, Finance Team", 0),
    ("Your Amazon order #12345 has shipped. Track at amazon.com/orders", 0),
    ("Team lunch next Friday at the usual place. See you there!", 0),
    ("Weekly security update: All systems operating normally", 0),
    ("Project deadline reminder: Submit by end of week", 0)
]

# Feature extraction function
def extract_features(email):
    features = {
        'urgency_words': len(re.findall(r'urgent|immediate|now|act|verify|confirm', email.lower())),
        'suspicious_urls': len(re.findall(r'http://[\w\.-]+\.(tk|ml|ga|xyz)', email.lower())),
        'money_mentions': len(re.findall(r'\$[\d,]+|prize|won|claim', email.lower())),
        'text': email
    }
    return features

# Prepare data
X_text = [item[0] for item in training_data]
y = [item[1] for item in training_data]

# Vectorize email text
vectorizer = TfidfVectorizer(max_features=50, stop_words='english')
X_vectorized = vectorizer.fit_transform(X_text)

# Train model
clf = RandomForestClassifier(n_estimators=100, random_state=42)
clf.fit(X_vectorized, y)

# Test emails
test_emails = [
    "URGENT: Verify your PayPal account at http://paypal-secure.tk or lose access",
    "Hi team, the meeting notes from yesterday are attached. Thanks!"
]

print("=== AI-Powered Phishing Email Detector ===")
print("\nModel trained on", len(training_data), "samples\n")

for i, email in enumerate(test_emails, 1):
    X_test = vectorizer.transform([email])
    prediction = clf.predict(X_test)[0]
    probability = clf.predict_proba(X_test)[0]
    
    print(f"Email {i}: {email[:60]}...")
    print(f"Prediction: {'PHISHING' if prediction == 1 else 'LEGITIMATE'}")
    print(f"Confidence: {max(probability)*100:.1f}%")
    print()

print("Detection Complete!")
output
=== AI-Powered Phishing Email Detector ===

Model trained on 10 samples

Email 1: URGENT: Verify your PayPal account at http://paypal-secur...
Prediction: PHISHING
Confidence: 90.0%

Email 2: Hi team, the meeting notes from yesterday are attached. ...
Prediction: LEGITIMATE
Confidence: 100.0%

Detection Complete!