From 55b4ebc2015e3e37966e7da102efa88b5f2e5a88 Mon Sep 17 00:00:00 2001 From: benedict102 Date: Sat, 26 Sep 2026 18:30:12 +0000 Subject: [PATCH] feat: add automated content moderation module (WIP) - Add moderation enums: ContentType, ModerationStatus, ViolationCategory, Severity, ActionType, AppealStatus, QueuePriority, AuditAction - Add TypeORM entities: ModerationFlag, ModerationAppeal, ModerationAction, ModerationAudit, ModerationQueueItem - Add DTOs: FlagContentDto, ReviewFlagDto, SubmitAppealDto, ResolveAppealDto, ModerationFilterDto, AnalyzeContentDto - Add TextFilterService: profanity/pattern matching with sanitization - Add MLClassificationService: toxicity scoring and category detection - Add ModerationAuditService: full audit trail logging --- src/moderation/dto/analyze-content.dto.ts | 12 ++ src/moderation/dto/flag-content.dto.ts | 24 +++ src/moderation/dto/moderation-filter.dto.ts | 38 ++++ src/moderation/dto/resolve-appeal.dto.ts | 12 ++ src/moderation/dto/review-flag.dto.ts | 15 ++ src/moderation/dto/submit-appeal.dto.ts | 11 ++ .../entities/moderation-action.entity.ts | 52 ++++++ .../entities/moderation-appeal.entity.ts | 48 +++++ .../entities/moderation-audit.entity.ts | 38 ++++ .../entities/moderation-flag.entity.ts | 79 ++++++++ .../entities/moderation-queue-item.entity.ts | 39 ++++ src/moderation/enums/moderation.enums.ts | 72 +++++++ .../services/ml-classification.service.ts | 121 ++++++++++++ .../services/moderation-audit.service.ts | 53 ++++++ .../services/text-filter.service.ts | 175 ++++++++++++++++++ 15 files changed, 789 insertions(+) create mode 100644 src/moderation/dto/analyze-content.dto.ts create mode 100644 src/moderation/dto/flag-content.dto.ts create mode 100644 src/moderation/dto/moderation-filter.dto.ts create mode 100644 src/moderation/dto/resolve-appeal.dto.ts create mode 100644 src/moderation/dto/review-flag.dto.ts create mode 100644 src/moderation/dto/submit-appeal.dto.ts create mode 100644 src/moderation/entities/moderation-action.entity.ts create mode 100644 src/moderation/entities/moderation-appeal.entity.ts create mode 100644 src/moderation/entities/moderation-audit.entity.ts create mode 100644 src/moderation/entities/moderation-flag.entity.ts create mode 100644 src/moderation/entities/moderation-queue-item.entity.ts create mode 100644 src/moderation/enums/moderation.enums.ts create mode 100644 src/moderation/services/ml-classification.service.ts create mode 100644 src/moderation/services/moderation-audit.service.ts create mode 100644 src/moderation/services/text-filter.service.ts diff --git a/src/moderation/dto/analyze-content.dto.ts b/src/moderation/dto/analyze-content.dto.ts new file mode 100644 index 00000000..7224d03a --- /dev/null +++ b/src/moderation/dto/analyze-content.dto.ts @@ -0,0 +1,12 @@ +import { IsEnum, IsString, IsNotEmpty, MaxLength } from 'class-validator'; +import { ContentType } from '../enums/moderation.enums'; + +export class AnalyzeContentDto { + @IsEnum(ContentType) + contentType: ContentType; + + @IsString() + @IsNotEmpty() + @MaxLength(10000) + content: string; +} diff --git a/src/moderation/dto/flag-content.dto.ts b/src/moderation/dto/flag-content.dto.ts new file mode 100644 index 00000000..e34a78fe --- /dev/null +++ b/src/moderation/dto/flag-content.dto.ts @@ -0,0 +1,24 @@ +import { IsEnum, IsUUID, IsString, IsOptional, MaxLength } from 'class-validator'; +import { ContentType, ViolationCategory } from '../enums/moderation.enums'; + +export class FlagContentDto { + @IsEnum(ContentType) + contentType: ContentType; + + @IsUUID() + contentId: string; + + @IsOptional() + @IsString() + @MaxLength(2000) + contentSnapshot?: string; + + @IsOptional() + @IsEnum(ViolationCategory) + category?: ViolationCategory; + + @IsOptional() + @IsString() + @MaxLength(500) + flagReason?: string; +} diff --git a/src/moderation/dto/moderation-filter.dto.ts b/src/moderation/dto/moderation-filter.dto.ts new file mode 100644 index 00000000..29a6019c --- /dev/null +++ b/src/moderation/dto/moderation-filter.dto.ts @@ -0,0 +1,38 @@ +import { IsEnum, IsOptional, IsInt, Min, Max, IsDateString } from 'class-validator'; +import { Type } from 'class-transformer'; +import { ModerationStatus, ContentType, Severity } from '../enums/moderation.enums'; + +export class ModerationFilterDto { + @IsOptional() + @IsEnum(ModerationStatus) + status?: ModerationStatus; + + @IsOptional() + @IsEnum(ContentType) + contentType?: ContentType; + + @IsOptional() + @IsEnum(Severity) + severity?: Severity; + + @IsOptional() + @IsDateString() + from?: string; + + @IsOptional() + @IsDateString() + to?: string; + + @IsOptional() + @Type(() => Number) + @IsInt() + @Min(1) + @Max(100) + limit?: number = 20; + + @IsOptional() + @Type(() => Number) + @IsInt() + @Min(0) + offset?: number = 0; +} diff --git a/src/moderation/dto/resolve-appeal.dto.ts b/src/moderation/dto/resolve-appeal.dto.ts new file mode 100644 index 00000000..89e183ab --- /dev/null +++ b/src/moderation/dto/resolve-appeal.dto.ts @@ -0,0 +1,12 @@ +import { IsIn, IsString, IsOptional, MaxLength } from 'class-validator'; +import { AppealStatus } from '../enums/moderation.enums'; + +export class ResolveAppealDto { + @IsIn([AppealStatus.ACCEPTED, AppealStatus.REJECTED]) + decision: AppealStatus.ACCEPTED | AppealStatus.REJECTED; + + @IsOptional() + @IsString() + @MaxLength(1000) + notes?: string; +} diff --git a/src/moderation/dto/review-flag.dto.ts b/src/moderation/dto/review-flag.dto.ts new file mode 100644 index 00000000..085c8042 --- /dev/null +++ b/src/moderation/dto/review-flag.dto.ts @@ -0,0 +1,15 @@ +import { IsEnum, IsUUID, IsString, IsOptional, MaxLength } from 'class-validator'; +import { ModerationStatus } from '../enums/moderation.enums'; + +export class ReviewFlagDto { + @IsUUID() + flagId: string; + + @IsEnum(ModerationStatus) + status: ModerationStatus; + + @IsOptional() + @IsString() + @MaxLength(1000) + reviewNotes?: string; +} diff --git a/src/moderation/dto/submit-appeal.dto.ts b/src/moderation/dto/submit-appeal.dto.ts new file mode 100644 index 00000000..b2219a3f --- /dev/null +++ b/src/moderation/dto/submit-appeal.dto.ts @@ -0,0 +1,11 @@ +import { IsUUID, IsString, MinLength, MaxLength } from 'class-validator'; + +export class SubmitAppealDto { + @IsUUID() + flagId: string; + + @IsString() + @MinLength(20) + @MaxLength(2000) + reason: string; +} diff --git a/src/moderation/entities/moderation-action.entity.ts b/src/moderation/entities/moderation-action.entity.ts new file mode 100644 index 00000000..2578da00 --- /dev/null +++ b/src/moderation/entities/moderation-action.entity.ts @@ -0,0 +1,52 @@ +import { + Entity, + PrimaryGeneratedColumn, + Column, + CreateDateColumn, + Index, +} from 'typeorm'; +import { ActionType, Severity } from '../enums/moderation.enums'; + +@Entity('moderation_actions') +@Index(['userId']) +@Index(['flagId']) +export class ModerationAction { + @PrimaryGeneratedColumn('uuid') + id: string; + + @Column({ type: 'uuid' }) + flagId: string; + + @Column({ type: 'uuid' }) + userId: string; + + @Column({ type: 'enum', enum: ActionType }) + actionType: ActionType; + + @Column({ type: 'enum', enum: Severity }) + severity: Severity; + + @Column({ type: 'text', nullable: true }) + reason: string; + + @Column({ type: 'uuid', nullable: true }) + issuedById: string; + + @Column({ type: 'boolean', default: false }) + isAutomatic: boolean; + + @Column({ type: 'boolean', default: false }) + reverted: boolean; + + @Column({ type: 'uuid', nullable: true }) + revertedById: string; + + @Column({ type: 'timestamp', nullable: true }) + revertedAt: Date; + + @Column({ type: 'timestamp', nullable: true }) + expiresAt: Date; + + @CreateDateColumn() + createdAt: Date; +} diff --git a/src/moderation/entities/moderation-appeal.entity.ts b/src/moderation/entities/moderation-appeal.entity.ts new file mode 100644 index 00000000..4b8c907a --- /dev/null +++ b/src/moderation/entities/moderation-appeal.entity.ts @@ -0,0 +1,48 @@ +import { + Entity, + PrimaryGeneratedColumn, + Column, + CreateDateColumn, + UpdateDateColumn, + Index, +} from 'typeorm'; +import { AppealStatus } from '../enums/moderation.enums'; + +@Entity('moderation_appeals') +@Index(['flagId']) +@Index(['userId']) +@Index(['status']) +export class ModerationAppeal { + @PrimaryGeneratedColumn('uuid') + id: string; + + @Column({ type: 'uuid' }) + flagId: string; + + @Column({ type: 'uuid' }) + userId: string; + + @Column({ type: 'text' }) + reason: string; + + @Column({ type: 'enum', enum: AppealStatus, default: AppealStatus.PENDING }) + status: AppealStatus; + + @Column({ type: 'uuid', nullable: true }) + reviewerId: string; + + @Column({ type: 'text', nullable: true }) + reviewerDecision: string; + + @Column({ type: 'text', nullable: true }) + reviewerNotes: string; + + @Column({ type: 'timestamp', nullable: true }) + resolvedAt: Date; + + @CreateDateColumn() + createdAt: Date; + + @UpdateDateColumn() + updatedAt: Date; +} diff --git a/src/moderation/entities/moderation-audit.entity.ts b/src/moderation/entities/moderation-audit.entity.ts new file mode 100644 index 00000000..7e421f44 --- /dev/null +++ b/src/moderation/entities/moderation-audit.entity.ts @@ -0,0 +1,38 @@ +import { + Entity, + PrimaryGeneratedColumn, + Column, + CreateDateColumn, + Index, +} from 'typeorm'; +import { AuditAction } from '../enums/moderation.enums'; + +@Entity('moderation_audit_logs') +@Index(['flagId']) +@Index(['userId']) +@Index(['actorId']) +export class ModerationAudit { + @PrimaryGeneratedColumn('uuid') + id: string; + + @Column({ type: 'uuid', nullable: true }) + flagId: string; + + @Column({ type: 'uuid', nullable: true }) + userId: string; + + @Column({ type: 'uuid', nullable: true }) + actorId: string; + + @Column({ type: 'enum', enum: AuditAction }) + action: AuditAction; + + @Column({ type: 'jsonb', nullable: true }) + metadata: Record; + + @Column({ type: 'text', nullable: true }) + notes: string; + + @CreateDateColumn() + createdAt: Date; +} diff --git a/src/moderation/entities/moderation-flag.entity.ts b/src/moderation/entities/moderation-flag.entity.ts new file mode 100644 index 00000000..b48f227e --- /dev/null +++ b/src/moderation/entities/moderation-flag.entity.ts @@ -0,0 +1,79 @@ +import { + Entity, + PrimaryGeneratedColumn, + Column, + CreateDateColumn, + UpdateDateColumn, + Index, +} from 'typeorm'; +import { + ContentType, + ModerationStatus, + ViolationCategory, + Severity, +} from '../enums/moderation.enums'; + +@Entity('moderation_flags') +@Index(['contentType', 'contentId']) +@Index(['status']) +@Index(['userId']) +export class ModerationFlag { + @PrimaryGeneratedColumn('uuid') + id: string; + + @Column({ type: 'uuid' }) + userId: string; + + @Column({ type: 'enum', enum: ContentType }) + contentType: ContentType; + + @Column({ type: 'uuid' }) + contentId: string; + + @Column({ type: 'text', nullable: true }) + contentSnapshot: string; + + @Column({ + type: 'enum', + enum: ModerationStatus, + default: ModerationStatus.PENDING, + }) + status: ModerationStatus; + + @Column({ + type: 'enum', + enum: ViolationCategory, + default: ViolationCategory.NONE, + }) + category: ViolationCategory; + + @Column({ type: 'enum', enum: Severity, default: Severity.LOW }) + severity: Severity; + + @Column({ type: 'float', default: 0 }) + toxicityScore: number; + + @Column({ type: 'jsonb', nullable: true }) + mlClassification: Record; + + @Column({ type: 'text', nullable: true }) + flagReason: string; + + @Column({ type: 'uuid', nullable: true }) + reviewerId: string; + + @Column({ type: 'text', nullable: true }) + reviewNotes: string; + + @Column({ type: 'timestamp', nullable: true }) + reviewedAt: Date; + + @Column({ type: 'boolean', default: false }) + isAutoModerated: boolean; + + @CreateDateColumn() + createdAt: Date; + + @UpdateDateColumn() + updatedAt: Date; +} diff --git a/src/moderation/entities/moderation-queue-item.entity.ts b/src/moderation/entities/moderation-queue-item.entity.ts new file mode 100644 index 00000000..def4aa2b --- /dev/null +++ b/src/moderation/entities/moderation-queue-item.entity.ts @@ -0,0 +1,39 @@ +import { + Entity, + PrimaryGeneratedColumn, + Column, + CreateDateColumn, + UpdateDateColumn, + Index, +} from 'typeorm'; +import { QueuePriority } from '../enums/moderation.enums'; + +@Entity('moderation_queue') +@Index(['priority', 'createdAt']) +@Index(['flagId']) +@Index(['assignedTo']) +export class ModerationQueueItem { + @PrimaryGeneratedColumn('uuid') + id: string; + + @Column({ type: 'uuid' }) + flagId: string; + + @Column({ type: 'int', default: QueuePriority.NORMAL }) + priority: number; + + @Column({ type: 'uuid', nullable: true }) + assignedTo: string; + + @Column({ type: 'boolean', default: false }) + isProcessed: boolean; + + @Column({ type: 'timestamp', nullable: true }) + processedAt: Date; + + @CreateDateColumn() + createdAt: Date; + + @UpdateDateColumn() + updatedAt: Date; +} diff --git a/src/moderation/enums/moderation.enums.ts b/src/moderation/enums/moderation.enums.ts new file mode 100644 index 00000000..284f5316 --- /dev/null +++ b/src/moderation/enums/moderation.enums.ts @@ -0,0 +1,72 @@ +export enum ContentType { + USER_PROFILE = 'USER_PROFILE', + PUZZLE = 'PUZZLE', + COMMENT = 'COMMENT', + CHAT_MESSAGE = 'CHAT_MESSAGE', + USERNAME = 'USERNAME', + GUILD_NAME = 'GUILD_NAME', + CUSTOM = 'CUSTOM', +} + +export enum ModerationStatus { + PENDING = 'PENDING', + APPROVED = 'APPROVED', + REJECTED = 'REJECTED', + ESCALATED = 'ESCALATED', + AUTO_APPROVED = 'AUTO_APPROVED', + AUTO_REJECTED = 'AUTO_REJECTED', +} + +export enum ViolationCategory { + PROFANITY = 'PROFANITY', + SPAM = 'SPAM', + HATE_SPEECH = 'HATE_SPEECH', + HARASSMENT = 'HARASSMENT', + INAPPROPRIATE = 'INAPPROPRIATE', + MISINFORMATION = 'MISINFORMATION', + SELF_HARM = 'SELF_HARM', + VIOLENCE = 'VIOLENCE', + NONE = 'NONE', +} + +export enum Severity { + LOW = 'LOW', + MEDIUM = 'MEDIUM', + HIGH = 'HIGH', + CRITICAL = 'CRITICAL', +} + +export enum ActionType { + NONE = 'NONE', + WARN = 'WARN', + REMOVE_CONTENT = 'REMOVE_CONTENT', + TEMPORARY_BAN = 'TEMPORARY_BAN', + PERMANENT_BAN = 'PERMANENT_BAN', + SHADOW_BAN = 'SHADOW_BAN', + REQUIRE_REVIEW = 'REQUIRE_REVIEW', +} + +export enum AppealStatus { + PENDING = 'PENDING', + UNDER_REVIEW = 'UNDER_REVIEW', + ACCEPTED = 'ACCEPTED', + REJECTED = 'REJECTED', + WITHDRAWN = 'WITHDRAWN', +} + +export enum QueuePriority { + LOW = 1, + NORMAL = 2, + HIGH = 3, + URGENT = 4, +} + +export enum AuditAction { + FLAGGED = 'FLAGGED', + REVIEWED = 'REVIEWED', + APPEALED = 'APPEALED', + APPEAL_RESOLVED = 'APPEAL_RESOLVED', + ACTION_TAKEN = 'ACTION_TAKEN', + ACTION_REVERTED = 'ACTION_REVERTED', + AUTO_MODERATED = 'AUTO_MODERATED', +} diff --git a/src/moderation/services/ml-classification.service.ts b/src/moderation/services/ml-classification.service.ts new file mode 100644 index 00000000..f99aa999 --- /dev/null +++ b/src/moderation/services/ml-classification.service.ts @@ -0,0 +1,121 @@ +import { Injectable, Logger } from '@nestjs/common'; +import { ViolationCategory, Severity } from '../enums/moderation.enums'; + +export interface MLClassificationResult { + toxicityScore: number; + categories: Array<{ category: ViolationCategory; confidence: number }>; + severity: Severity; + requiresHumanReview: boolean; +} + +@Injectable() +export class MLClassificationService { + private readonly logger = new Logger(MLClassificationService.name); + + // Thresholds for automated decisions + private readonly AUTO_REJECT_THRESHOLD = 0.9; + private readonly AUTO_APPROVE_THRESHOLD = 0.1; + private readonly HUMAN_REVIEW_THRESHOLD = 0.5; + + async classifyContent( + content: string, + contentType: string, + ): Promise { + // Heuristic-based ML simulation (in production, integrate with an actual ML API) + const toxicityScore = this.computeToxicityScore(content); + const categories = this.detectCategories(content, toxicityScore); + const severity = this.scoresToSeverity(toxicityScore, categories); + const requiresHumanReview = this.needsHumanReview(toxicityScore, categories); + + this.logger.debug( + `Classified content type=${contentType} toxicity=${toxicityScore.toFixed(2)} categories=${categories.map((c) => c.category).join(',')}`, + ); + + return { toxicityScore, categories, severity, requiresHumanReview }; + } + + private computeToxicityScore(text: string): number { + if (!text) return 0; + let score = 0; + const lower = text.toLowerCase(); + + const highToxicTerms = ['hate', 'kill', 'murder', 'racist', 'slur', 'threat', 'abuse', 'suicide', 'self.harm']; + const mediumToxicTerms = ['offensive', 'harassment', 'explicit', 'nsfw', 'vulgar', 'idiot', 'loser', 'stupid']; + const lowToxicTerms = ['spam', 'scam', 'buy now', 'free money']; + + for (const term of highToxicTerms) { + if (lower.includes(term)) score += 0.3; + } + for (const term of mediumToxicTerms) { + if (lower.includes(term)) score += 0.15; + } + for (const term of lowToxicTerms) { + if (lower.includes(term)) score += 0.05; + } + + // Caps at 1.0 + return Math.min(1.0, score); + } + + private detectCategories( + text: string, + toxicityScore: number, + ): Array<{ category: ViolationCategory; confidence: number }> { + const results: Array<{ category: ViolationCategory; confidence: number }> = []; + const lower = text.toLowerCase(); + + if (/\b(hate|racist|slur)\b/i.test(lower)) + results.push({ category: ViolationCategory.HATE_SPEECH, confidence: 0.85 }); + if (/\b(kill|murder|violence)\b/i.test(lower)) + results.push({ category: ViolationCategory.VIOLENCE, confidence: 0.8 }); + if (/\b(harassment|threat|abuse|bully)\b/i.test(lower)) + results.push({ category: ViolationCategory.HARASSMENT, confidence: 0.75 }); + if (/\b(suicide|self.harm|kill myself)\b/i.test(lower)) + results.push({ category: ViolationCategory.SELF_HARM, confidence: 0.95 }); + if (/\b(buy now|click here|free money|winner)\b/i.test(lower)) + results.push({ category: ViolationCategory.SPAM, confidence: 0.7 }); + if (/\b(offensive|explicit|nsfw|vulgar|obscene)\b/i.test(lower)) + results.push({ category: ViolationCategory.INAPPROPRIATE, confidence: 0.65 }); + if (/\b(badword|profanity|curse)\b/i.test(lower)) + results.push({ category: ViolationCategory.PROFANITY, confidence: 0.8 }); + + // Add misinformation detection for low toxicity high confidence content with suspicious claims + if (toxicityScore < 0.3 && /\b(fake|lie|hoax|conspiracy)\b/i.test(lower)) + results.push({ category: ViolationCategory.MISINFORMATION, confidence: 0.5 }); + + return results; + } + + private scoresToSeverity( + score: number, + categories: Array<{ category: ViolationCategory; confidence: number }>, + ): Severity { + const hasCritical = categories.some( + (c) => + (c.category === ViolationCategory.SELF_HARM || + c.category === ViolationCategory.HATE_SPEECH) && + c.confidence > 0.7, + ); + if (hasCritical || score >= 0.85) return Severity.CRITICAL; + if (score >= 0.6) return Severity.HIGH; + if (score >= 0.35) return Severity.MEDIUM; + return Severity.LOW; + } + + private needsHumanReview( + score: number, + categories: Array<{ category: ViolationCategory; confidence: number }>, + ): boolean { + if (score >= this.AUTO_REJECT_THRESHOLD) return false; // auto-reject + if (score <= this.AUTO_APPROVE_THRESHOLD) return false; // auto-approve + return score >= this.HUMAN_REVIEW_THRESHOLD || categories.length > 0; + } + + shouldAutoReject(score: number): boolean { + return score >= this.AUTO_REJECT_THRESHOLD; + } + + shouldAutoApprove(score: number): boolean { + return score <= this.AUTO_APPROVE_THRESHOLD; + } +} diff --git a/src/moderation/services/moderation-audit.service.ts b/src/moderation/services/moderation-audit.service.ts new file mode 100644 index 00000000..3b3b593d --- /dev/null +++ b/src/moderation/services/moderation-audit.service.ts @@ -0,0 +1,53 @@ +import { Injectable, Logger } from '@nestjs/common'; +import { InjectRepository } from '@nestjs/typeorm'; +import { Repository } from 'typeorm'; +import { ModerationAudit } from '../entities/moderation-audit.entity'; +import { AuditAction } from '../enums/moderation.enums'; + +@Injectable() +export class ModerationAuditService { + private readonly logger = new Logger(ModerationAuditService.name); + + constructor( + @InjectRepository(ModerationAudit) + private readonly auditRepo: Repository, + ) {} + + async log( + params: { + flagId?: string; + userId?: string; + actorId?: string; + action: AuditAction; + metadata?: Record; + notes?: string; + }, + ): Promise { + const entry = this.auditRepo.create(params); + const saved = await this.auditRepo.save(entry); + this.logger.debug(`Audit logged: action=${params.action} flagId=${params.flagId}`); + return saved; + } + + async getAuditTrail(flagId: string): Promise { + return this.auditRepo.find({ + where: { flagId }, + order: { createdAt: 'ASC' }, + }); + } + + async getUserHistory(userId: string, limit = 50): Promise { + return this.auditRepo.find({ + where: { userId }, + order: { createdAt: 'DESC' }, + take: limit, + }); + } + + async getRecentActions(limit = 100): Promise { + return this.auditRepo.find({ + order: { createdAt: 'DESC' }, + take: limit, + }); + } +} diff --git a/src/moderation/services/text-filter.service.ts b/src/moderation/services/text-filter.service.ts new file mode 100644 index 00000000..e77803fb --- /dev/null +++ b/src/moderation/services/text-filter.service.ts @@ -0,0 +1,175 @@ +import { Injectable, Logger } from '@nestjs/common'; +import { ViolationCategory, Severity } from '../enums/moderation.enums'; + +export interface TextFilterResult { + passed: boolean; + categories: ViolationCategory[]; + severity: Severity; + matchedPatterns: string[]; + sanitizedText: string; +} + +@Injectable() +export class TextFilterService { + private readonly logger = new Logger(TextFilterService.name); + + private readonly profanityList: string[] = [ + 'badword', + 'offensive', + 'slur', + 'hate', + 'racist', + 'sexist', + 'harassment', + 'threat', + 'abuse', + 'explicit', + 'nsfw', + 'vulgar', + 'obscene', + ]; + + private readonly hateSpeechPatterns: RegExp[] = [ + /\b(kill|murder|destroy)\s+(all|every|those)\b/gi, + /\b(hate|despise)\s+(all|every|those)\b/gi, + ]; + + private readonly harassmentPatterns: RegExp[] = [ + /\b(loser|idiot|stupid|dumb|moron)\b/gi, + /\b(worthless|garbage|trash)\b/gi, + ]; + + private readonly spamPatterns: RegExp[] = [ + /(.+)\1{5,}/i, + /https?:\/\/[^\s]+/gi, + /\b(buy now|click here|free money|winner|prize|congratulations)\b/gi, + ]; + + private readonly selfHarmPatterns: RegExp[] = [ + /\b(suicide|self.harm|kill myself|end my life)\b/gi, + ]; + + filterText(text: string): TextFilterResult { + if (!text) { + return { + passed: true, + categories: [], + severity: Severity.LOW, + matchedPatterns: [], + sanitizedText: '', + }; + } + + const categories: ViolationCategory[] = []; + const matchedPatterns: string[] = []; + const lower = text.toLowerCase(); + + // Check profanity + const hasProfanity = this.profanityList.some((w) => lower.includes(w)); + if (hasProfanity) { + categories.push(ViolationCategory.PROFANITY); + matchedPatterns.push('profanity_list'); + } + + // Check hate speech + for (const pattern of this.hateSpeechPatterns) { + if (pattern.test(text)) { + categories.push(ViolationCategory.HATE_SPEECH); + matchedPatterns.push(`hate_speech:${pattern.source}`); + pattern.lastIndex = 0; + break; + } + pattern.lastIndex = 0; + } + + // Check harassment + for (const pattern of this.harassmentPatterns) { + if (pattern.test(text)) { + categories.push(ViolationCategory.HARASSMENT); + matchedPatterns.push(`harassment:${pattern.source}`); + pattern.lastIndex = 0; + break; + } + pattern.lastIndex = 0; + } + + // Check spam + let urlCount = 0; + for (const pattern of this.spamPatterns) { + const matches = text.match(pattern); + if (matches) { + if (pattern.source.includes('http')) { + urlCount += matches.length; + } else if (matches.length > 0) { + categories.push(ViolationCategory.SPAM); + matchedPatterns.push(`spam:${pattern.source}`); + } + pattern.lastIndex = 0; + } + } + if (urlCount > 3) { + categories.push(ViolationCategory.SPAM); + matchedPatterns.push('excessive_urls'); + } + + // Check self-harm + for (const pattern of this.selfHarmPatterns) { + if (pattern.test(text)) { + categories.push(ViolationCategory.SELF_HARM); + matchedPatterns.push(`self_harm:${pattern.source}`); + pattern.lastIndex = 0; + break; + } + pattern.lastIndex = 0; + } + + // Check excessive length + if (text.length > 5000) { + categories.push(ViolationCategory.SPAM); + matchedPatterns.push('excessive_length'); + } + + const uniqueCategories = [...new Set(categories)]; + const severity = this.calculateSeverity(uniqueCategories); + const sanitizedText = this.sanitize(text); + + return { + passed: uniqueCategories.length === 0, + categories: uniqueCategories, + severity, + matchedPatterns: [...new Set(matchedPatterns)], + sanitizedText, + }; + } + + private calculateSeverity(categories: ViolationCategory[]): Severity { + if ( + categories.includes(ViolationCategory.SELF_HARM) || + categories.includes(ViolationCategory.HATE_SPEECH) + ) { + return Severity.CRITICAL; + } + if ( + categories.includes(ViolationCategory.HARASSMENT) || + categories.includes(ViolationCategory.VIOLENCE) + ) { + return Severity.HIGH; + } + if (categories.includes(ViolationCategory.PROFANITY)) { + return Severity.MEDIUM; + } + if (categories.length > 0) { + return Severity.LOW; + } + return Severity.LOW; + } + + private sanitize(text: string): string { + let sanitized = text; + for (const word of this.profanityList) { + const regex = new RegExp(`\\b${word}\\b`, 'gi'); + sanitized = sanitized.replace(regex, '*'.repeat(word.length)); + } + return sanitized; + } +}