Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -57,7 +57,10 @@ MAX_CSV_ROWS=100000 # 100k rows max
ENABLE_MALWARE_SCAN=false # Enable ClamAV scanning
CLAMAV_HOST=localhost
CLAMAV_PORT=3310
<<<<<<< Updated upstream

# Archival Job Schedule (Cron expression)
# Default is '0 3 * * *' (run once daily at 3:00 AM)
ARCHIVAL_CRON_SCHEDULE=0 3 * * *
=======
>>>>>>> Stashed changes
5 changes: 4 additions & 1 deletion backend/.env.example
Original file line number Diff line number Diff line change
Expand Up @@ -109,6 +109,7 @@ PREDICT_RATE_LIMIT=50 per minute
VITE_API_URI=http://localhost:3000/api
VITE_ML_API_URI=http://localhost:5000/predict

<<<<<<< Updated upstream
# --------------------------------------------
# GROQ API (AI Chat)
# --------------------------------------------
Expand All @@ -119,4 +120,6 @@ GROQ_API_KEY=your_groq_api_key_here
GROQ_MODELS=["llama-3.1-8b-instant","llama-3.1-70b-versatile","mixtral-8x7b-32768","gemma2-9b-it"]

# OR single model (backward compatible)
# GROQ_MODEL=llama-3.1-8b-instant
# GROQ_MODEL=llama-3.1-8b-instant
=======
>>>>>>> Stashed changes
90 changes: 90 additions & 0 deletions backend/adversarial_training.py
Original file line number Diff line number Diff line change
Expand Up @@ -72,6 +72,16 @@ def __init__(self):

# Noise patterns
self.noise_chars = ['.', '!', '?', ',', ';', ':', ' ']
<<<<<<< HEAD
=======

# Spam trigger patterns
self.spam_triggers = [
'urgent', 'free', 'claim', 'prize', 'winner', 'congratulations',
'limited time', 'act now', 'exclusive', 'guaranteed', 'money back',
'cash', 'bonus', 'credit', 'loan', 'investment', 'profit'
]
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda


# Spam trigger patterns
Expand All @@ -97,9 +107,16 @@ def synonym_replacement(self, text, intensity=0.4):
word_lower = word.lower().strip('.,!?')
if word_lower in self.synonyms and random.random() < intensity:
new_word = random.choice(self.synonyms[word_lower])
<<<<<<< Updated upstream

# Preserve punctuation

=======
<<<<<<< HEAD
=======
# Preserve punctuation
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda
>>>>>>> Stashed changes
punct = word[-1] if word[-1] in '.,!?' else ''
result.append(new_word + punct)
else:
Expand All @@ -118,12 +135,17 @@ def noise_injection(self, text, intensity=0.2):
result.insert(pos, char)
return ''.join(result)

<<<<<<< HEAD
def generate_variants(self, text, num_variants=5):
"""Generate multiple adversarial variants"""
variants = []
for _ in range(num_variants):
variant = text
<<<<<<< Updated upstream

=======
=======
>>>>>>> Stashed changes
def sentence_rephrasing(self, text):
"""Simple sentence rephrasing (rule-based)"""
# This is a simple version - can be enhanced with LLM
Expand All @@ -146,6 +168,10 @@ def generate_variants(self, text, num_variants=5):
for _ in range(num_variants):
variant = text
# Apply random transformations
<<<<<<< Updated upstream
=======
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda
>>>>>>> Stashed changes
transformations = random.sample([
('char', 0.2 + random.random() * 0.3),
('synonym', 0.2 + random.random() * 0.3),
Expand All @@ -160,12 +186,21 @@ def generate_variants(self, text, num_variants=5):
elif transform_type == 'noise':
variant = self.noise_injection(variant, intensity)

<<<<<<< Updated upstream

=======
<<<<<<< HEAD
=======
>>>>>>> Stashed changes
# Sometimes rephrase
if random.random() < 0.3:
variant = self.sentence_rephrasing(variant)

<<<<<<< Updated upstream

=======
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda
>>>>>>> Stashed changes
variants.append(variant)

return variants
Expand All @@ -176,6 +211,7 @@ def generate_variants(self, text, num_variants=5):
# ============================================

def load_datasets():
<<<<<<< HEAD
"""Load dataset"""
if not os.path.exists(DATASET_PATH):
print(f"❌ Dataset not found at {DATASET_PATH}")
Expand Down Expand Up @@ -207,7 +243,11 @@ def create_sample_dataset():
]
df = pd.DataFrame(samples, columns=['text', 'label'])
print(f"βœ… Created sample dataset with {len(df)} samples")
<<<<<<< Updated upstream

=======
=======
>>>>>>> Stashed changes
"""Load and combine email + SMS datasets"""
data = []

Expand Down Expand Up @@ -272,7 +312,11 @@ def create_sample_dataset():
print(f" Spam: {len(df[df['label']=='spam'])}")
print(f" Ham: {len(df[df['label']=='ham'])}")

<<<<<<< Updated upstream

=======
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda
>>>>>>> Stashed changes
return df


Expand All @@ -284,7 +328,14 @@ def train_adversarial_model(df):
"""Train model with adversarial examples"""

print("\nπŸ”„ Generating adversarial examples...")
<<<<<<< Updated upstream

=======
<<<<<<< HEAD
=======

>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda
>>>>>>> Stashed changes
augmentor = AdversarialAugmentor()

# Separate spam and ham
Expand Down Expand Up @@ -336,7 +387,14 @@ def train_adversarial_model(df):
('lr', LogisticRegression(class_weight='balanced', max_iter=1000, random_state=42))
]

<<<<<<< Updated upstream

=======
<<<<<<< HEAD
=======
# Train each model
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda
>>>>>>> Stashed changes
trained_models = {}
for name, model in models:
print(f" Training {name}...")
Expand Down Expand Up @@ -367,17 +425,31 @@ def train_adversarial_model(df):

def main():
print("=" * 60)
<<<<<<< HEAD
print("πŸ›‘οΈ Adversarial Training for Spam Detection")
<<<<<<< Updated upstream

print("πŸš€ Adversarial Training for Spam Detection")
=======
=======
print("πŸš€ Adversarial Training for Spam Detection")
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda
>>>>>>> Stashed changes
print("=" * 60)

# Load datasets
df = load_datasets()
if df is None:
<<<<<<< HEAD
print("\n❌ No dataset available")
<<<<<<< Updated upstream

print("\n❌ Please ensure dataset.csv exists in backend/")
=======
=======
print("\n❌ Please ensure dataset.csv exists in backend/")
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda
>>>>>>> Stashed changes
return

# Train model
Expand All @@ -390,24 +462,42 @@ def main():
print(f"πŸ’Ύ Saving vectorizer to {VECTORIZER_OUTPUT_PATH}")
pickle.dump(vectorizer, open(VECTORIZER_OUTPUT_PATH, 'wb'))

<<<<<<< Updated upstream

# Save label encoder (for compatibility)

=======
<<<<<<< HEAD
# Save label encoder
=======
# Save label encoder (for compatibility)
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda
>>>>>>> Stashed changes
from sklearn.preprocessing import LabelEncoder
le = LabelEncoder()
le.fit(['ham', 'spam'])
pickle.dump(le, open(LABEL_ENCODER_PATH, 'wb'))

<<<<<<< HEAD
print("\nβœ… Adversarial training complete!")
<<<<<<< Updated upstream
print("\nβœ… Training complete!")

=======
=======
print("\nβœ… Training complete!")
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda
>>>>>>> Stashed changes
print(f" Model saved: {MODEL_OUTPUT_PATH}")
print(f" Vectorizer saved: {VECTORIZER_OUTPUT_PATH}")
print("\nπŸ“ To use the robust model, update your .env:")
print(f" MODEL_PATH={MODEL_OUTPUT_PATH}")
print(f" VECTORIZER_PATH={VECTORIZER_OUTPUT_PATH}")
<<<<<<< HEAD
print(f" CONFIDENCE_THRESHOLD=0.6")
print(f" FLAG_LOW_CONFIDENCE=true")
=======
>>>>>>> 42c19e1f454b08381ffc8132bf9ae9f6a57dabda



Expand Down
Loading
Loading