// projects.cyber-security-ai
10 Best Cyber Security with AI Projects
Hand-picked and ordered easiest → hardest — each with complete code and expected output. Build these to turn lessons into a portfolio.
Back to Cyber Security with AI course
// solution.code
security
# AI-Powered Phishing Email Detector
# Uses TfidfVectorizer and RandomForest for email classification
import re
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.ensemble import RandomForestClassifier
from sklearn.model_selection import train_test_split
from sklearn.metrics import classification_report, accuracy_score
import numpy as np
# Sample training data: (email_text, is_phishing)
training_data = [
("Congratulations! You've won $1,000,000. Click here: http://suspicious-site.xyz", 1),
("Your account has been compromised. Verify immediately at http://paypal-verify.tk", 1),
("URGENT: Your bank account will be closed. Update info at http://bank-update.ml", 1),
("Dear user, click here to claim your prize from Nigerian prince", 1),
("Meeting scheduled for tomorrow at 2 PM. Please confirm your attendance.", 0),
("Here are the quarterly reports you requested. Best regards, Finance Team", 0),
("Your Amazon order #12345 has shipped. Track at amazon.com/orders", 0),
("Team lunch next Friday at the usual place. See you there!", 0),
("Weekly security update: All systems operating normally", 0),
("Project deadline reminder: Submit by end of week", 0)
]
# Feature extraction function
def extract_features(email):
features = {
'urgency_words': len(re.findall(r'urgent|immediate|now|act|verify|confirm', email.lower())),
'suspicious_urls': len(re.findall(r'http://[\w\.-]+\.(tk|ml|ga|xyz)', email.lower())),
'money_mentions': len(re.findall(r'\$[\d,]+|prize|won|claim', email.lower())),
'text': email
}
return features
# Prepare data
X_text = [item[0] for item in training_data]
y = [item[1] for item in training_data]
# Vectorize email text
vectorizer = TfidfVectorizer(max_features=50, stop_words='english')
X_vectorized = vectorizer.fit_transform(X_text)
# Train model
clf = RandomForestClassifier(n_estimators=100, random_state=42)
clf.fit(X_vectorized, y)
# Test emails
test_emails = [
"URGENT: Verify your PayPal account at http://paypal-secure.tk or lose access",
"Hi team, the meeting notes from yesterday are attached. Thanks!"
]
print("=== AI-Powered Phishing Email Detector ===")
print("\nModel trained on", len(training_data), "samples\n")
for i, email in enumerate(test_emails, 1):
X_test = vectorizer.transform([email])
prediction = clf.predict(X_test)[0]
probability = clf.predict_proba(X_test)[0]
print(f"Email {i}: {email[:60]}...")
print(f"Prediction: {'PHISHING' if prediction == 1 else 'LEGITIMATE'}")
print(f"Confidence: {max(probability)*100:.1f}%")
print()
print("Detection Complete!") output
=== AI-Powered Phishing Email Detector === Model trained on 10 samples Email 1: URGENT: Verify your PayPal account at http://paypal-secur... Prediction: PHISHING Confidence: 90.0% Email 2: Hi team, the meeting notes from yesterday are attached. ... Prediction: LEGITIMATE Confidence: 100.0% Detection Complete!
