diff --git a/.gitignore b/.gitignore index 5ac4238..edbab60 100644 --- a/.gitignore +++ b/.gitignore @@ -1,117 +1,11 @@ -# ============================================ -# Python & Environment -# ============================================ -__pycache__/ -*.py[cod] +``` +# Compiled Python files *.pyc -*.pyo -*.pyd -.Python -venv/ -.venv/ -.env -.env.local -.env.* -ENV/ -env/ -virtualenv/ - -# ============================================ -# Build & Dependencies -# ============================================ -pip-wheel-metadata/ -.tox/ -.coverage -htmlcov/ -.pytest_cache/ -.mypy_cache/ -.ruff_cache/ -build/ -dist/ -*.egg-info/ -*.egg -wheels/ - -# ============================================ -# IDE & OS -# ============================================ -.idea/ -.vscode/ -*.swp -*.swo -*~ -.DS_Store -Thumbs.db -desktop.ini - - - -# ============================================ -# Project Data & Outputs -# ============================================ -# پوشه‌های داده و خروجی (همه محتویات) -data/ -outputs/ -models/ -logs/ -results/ -figures/ -plots/ -reports/ - -# ============================================ -# Data File Types -# ============================================ -*.csv -*.tsv -*.xls -*.xlsx -*.txt -*.json -*.jsonl -*.xml -*.yaml -*.yml -*.parquet -*.feather -*.h5 -*.hdf5 -*.pkl -*.pickle -*.joblib +__pycache__/ -# ============================================ -# Images & Binary Files -# ============================================ -*.png -*.jpg -*.jpeg -*.gif -*.bmp -*.tiff -*.svg -*.eps -*.pth -*.pt -*.onnx -*.pb -*.weights -*.bin +# Archives +*.tar.gz -# ============================================ -# Logs & Temp Files -# ============================================ -*.log -*.bak -*.tmp -*.temp -*.pid -*.seed -core/outputs/ -core/outputs/** -*.png -*.jpg -*.jpeg -*.joblib -*.json -*.tsv +# Original rules preserved +core/src/feature_selection.py +``` \ No newline at end of file diff --git a/core/.gitignore b/core/.gitignore new file mode 100644 index 0000000..f63dda6 --- /dev/null +++ b/core/.gitignore @@ -0,0 +1,47 @@ +# Ignore data files +data/processed/*.csv +data/processed/*.joblib +data/raw/*.csv +data/interim/* + +# Ignore model outputs +outputs/models/*.joblib +outputs/models/*.pkl +outputs/models/*.h5 + +# Ignore large tables and figures if needed +outputs/tables/*.csv +outputs/figures/*.png +outputs/figures/*.jpg + +# Python cache +__pycache__/ +*.py[cod] +*$py.class +*.so +.Python +env/ +venv/ +ENV/ +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +*.egg-info/ +.installed.cfg +*.egg + +# Jupyter Notebook checkpoints +.ipynb_checkpoints/ + +# OS files +.DS_Store +Thumbs.db diff --git a/core/__pycache__/config.cpython-312.pyc b/core/__pycache__/config.cpython-312.pyc new file mode 100644 index 0000000..66cc965 Binary files /dev/null and b/core/__pycache__/config.cpython-312.pyc differ diff --git a/core/notebooks/04_feature_selection.ipynb b/core/notebooks/04_feature_selection.ipynb index 33f1b2e..ee5b469 100644 --- a/core/notebooks/04_feature_selection.ipynb +++ b/core/notebooks/04_feature_selection.ipynb @@ -5,15 +5,15 @@ "id": "2744288e", "metadata": {}, "source": [ - "# 04 - Feature Selection (3-Layer Pipeline)\n", + "# 04 - Feature Selection (Simple MI Pipeline)\n", "\n", - "Layer 1: Variance + Mutual Information on raw genes\n", + "Step 1: Variance Threshold + Mutual Information on genes\n", "\n", - "Layer 2: Domain-specific feature engineering with correlation filtering\n", + "Step 2: Domain-specific feature engineering (7 features)\n", "\n", - "Layer 3: Binary PSO assembly over genes plus engineered features\n", + "Step 3: Combine selected genes + engineered + clinical features\n", "\n", - "All logic lives in `src/feature_selection.py`. This notebook only loads data, detects clinical columns, calls the pipeline, and saves artifacts." + "All logic lives in `src/feature_selection.py`. This notebook only loads data, runs the pipeline, and saves artifacts." ] }, { @@ -33,7 +33,7 @@ "import joblib\n", "import config\n", "from src.io import logger\n", - "from src.feature_selection import run_3layer_feature_selection" + "from src.feature_selection import run_feature_selection" ] }, { @@ -72,9 +72,13 @@ "id": "74664d66", "metadata": {}, "source": [ - "## Step 2: Detect Clinical / Engineered Columns\n", + "## Step 2: Run Feature Selection Pipeline\n", "\n", - "These columns are excluded from Layer 1 gene filtering and passed explicitly to `run_3layer_feature_selection` as `clinical_cols`." + "This performs:\n", + "1. Variance threshold on genes\n", + "2. Mutual Information to select top K genes\n", + "3. Creates 7 engineered features\n", + "4. Returns fitted selector + list of all selected features" ] }, { @@ -243,7 +247,7 @@ "id": "599f7e30", "metadata": {}, "source": [ - "## Step 4: Save Artifacts" + "## Step 3: Save Artifacts" ] }, { @@ -262,20 +266,23 @@ } ], "source": [ - "if final_features:\n", - " joblib.dump(fitted_l1, config.MODELS_DIR / \"fitted_layer1_selector.joblib\")\n", - " print(f\"Saved Layer 1 selector to: {config.MODELS_DIR / 'fitted_layer1_selector.joblib'}\")\n", + "# Save fitted selector\n", + "joblib.dump(fitted_selector, config.MODELS_DIR / \"fitted_selector.joblib\")\n", + "print(f\"Saved selector to: {config.MODELS_DIR / 'fitted_selector.joblib'}\")\n", "\n", - " pd.DataFrame({\"feature\": final_features}).to_csv(\n", - " config.TABLES_DIR / \"selected_features_final.csv\", index=False\n", - " )\n", - " print(\n", - " f\"Saved feature list ({len(final_features)} items) to: \"\n", - " f\"{config.TABLES_DIR / 'selected_features_final.csv'}\"\n", - " )\n", - "else:\n", - " print(\"WARNING: no features were generated. Nothing was saved.\")\n", - " print(\"Resolve the error above and re-run this notebook.\")" + "# Save feature list\n", + "pd.DataFrame({\"feature\": final_features}).to_csv(\n", + " config.TABLES_DIR / \"selected_features_final.csv\", index=False\n", + ")\n", + "print(f\"Saved feature list ({len(final_features)} items) to: {config.TABLES_DIR / 'selected_features_final.csv'}\")\n", + "\n", + "# Save engineered training data for reference\n", + "X_train_eng = X_train.copy()\n", + "for feat in eng_features:\n", + " if feat not in X_train_eng.columns:\n", + " X_train_eng[feat] = X_eng[feat]\n", + "\n", + "print(\"Feature selection complete!\")" ] } ], diff --git a/core/notebooks/05_Model_Training.ipynb b/core/notebooks/05_Model_Training.ipynb index cc2b6da..2e05a6b 100644 --- a/core/notebooks/05_Model_Training.ipynb +++ b/core/notebooks/05_Model_Training.ipynb @@ -5,9 +5,9 @@ "id": "a8be3165", "metadata": {}, "source": [ - "# 05 - Model Training (3-Layer Pipeline)\n", + "# 05 - Model Training (Simple Pipeline)\n", "\n", - "Loads the Layer 1 selector and final feature list produced by notebook 04, re-applies Layer 2 feature engineering to the full training set, filters to the final feature list, tunes XGBoost with Optuna, trains the final model, and saves the model artifact." + "Loads the fitted selector and final feature list produced by notebook 04, applies transform_selected to test data, tunes XGBoost with Optuna, trains the final model, and saves the model artifact." ] }, { @@ -40,7 +40,7 @@ "from src.io import logger\n", "from src.models import build_model, xgb_safe_frame\n", "from src.optimization import optimize_model\n", - "from src.feature_selection import create_extended_engineered_features\n", + "from src.feature_selection import transform_selected\n", "from sklearn.metrics import roc_auc_score" ] }, @@ -82,7 +82,7 @@ "source": [ "## Step 2: Load Artifacts From Notebook 04\n", "\n", - "Requires `fitted_layer1_selector.joblib` and `selected_features_final.csv`. Raises a clear error if notebook 04 has not been run." + "Requires `fitted_selector.joblib` and `selected_features_final.csv`. Raises a clear error if notebook 04 has not been run." ] }, { @@ -100,19 +100,19 @@ } ], "source": [ - "selector_path = config.MODELS_DIR / \"fitted_layer1_selector.joblib\"\n", + "selector_path = config.MODELS_DIR / \"fitted_selector.joblib\"\n", "features_path = config.TABLES_DIR / \"selected_features_final.csv\"\n", "\n", "try:\n", - " fitted_l1 = joblib.load(selector_path)\n", + " fitted_selector = joblib.load(selector_path)\n", " selected_df = pd.read_csv(features_path)\n", " final_features = selected_df[\"feature\"].tolist()\n", - " logger.info(f\"Loaded {len(final_features)} final features from the 3-layer pipeline.\")\n", + " logger.info(f\"Loaded {len(final_features)} final features from the pipeline.\")\n", "except FileNotFoundError as e:\n", " raise FileNotFoundError(\n", " f\"{e}\\n\\n\"\n", " \"Please run '04_feature_selection.ipynb' first to generate \"\n", - " \"'fitted_layer1_selector.joblib' and 'selected_features_final.csv'.\"\n", + " \"'fitted_selector.joblib' and 'selected_features_final.csv'.\"\n", " )" ] }, @@ -121,9 +121,9 @@ "id": "6f2c1020", "metadata": {}, "source": [ - "## Step 3: Apply Layer 2 Feature Engineering and Filter to Final Features\n", + "## Step 3: Apply Feature Selection to Training Data\n", "\n", - "Engineering is applied to the full training set before filtering, matching how notebook 04 built the candidate pool. Any final feature missing after engineering (should not normally happen, but can for edge splits) is filled with 0.0." + "Use transform_selected to ensure training data matches the exact feature set that will be used for test/external data." ] }, { @@ -142,13 +142,14 @@ } ], "source": [ - "X_train_eng, _ = create_extended_engineered_features(\n", - " X_train, selected_genes=fitted_l1[\"mi_features\"]\n", - ")\n", + "# Transform training data using fitted selector\n", + "X_train_selected = transform_selected(X_train, fitted_selector)\n", "\n", - "available_in_train = [f for f in final_features if f in X_train_eng.columns]\n", - "X_train_final = X_train_eng[available_in_train].copy()\n", + "# Ensure we only keep features in final_features list\n", + "available_in_train = [f for f in final_features if f in X_train_selected.columns]\n", + "X_train_final = X_train_selected[available_in_train].copy()\n", "\n", + "# Fill any missing features with 0.0 (should not happen normally)\n", "missing_feats = set(final_features) - set(available_in_train)\n", "if missing_feats:\n", " logger.warning(f\"Missing {len(missing_feats)} features in training data: {list(missing_feats)[:5]}...\")\n", @@ -1703,12 +1704,40 @@ "source": [ "config.MODELS_DIR.mkdir(parents=True, exist_ok=True)\n", "\n", + "# Save the trained model\n", "joblib.dump(final_model, config.MODELS_DIR / \"best_model_xgboost.joblib\")\n", - "joblib.dump(fitted_l1, config.MODELS_DIR / \"fitted_layer1_selector.joblib\")\n", + "\n", + "# Save the fitted selector for transforming test data\n", + "joblib.dump(fitted_selector, config.MODELS_DIR / \"fitted_selector.joblib\")\n", + "\n", + "# CRITICAL: Transform and save test data\n", + "X_test = pd.read_csv(config.PROCESSED_DIR / \"X_test_preprocessed.csv\", index_col=0)\n", + "y_test_df = pd.read_csv(config.PROCESSED_DIR / \"y_test.csv\")\n", + "y_test = y_test_df.iloc[:, 0] if len(y_test_df.columns) == 1 else y_test_df[\"BCR\"]\n", + "\n", + "# Apply the same transformation to test data\n", + "X_test_selected = transform_selected(X_test, fitted_selector)\n", + "\n", + "# Ensure test data has same features as training\n", + "available_in_test = [f for f in final_features if f in X_test_selected.columns]\n", + "X_test_final = X_test_selected[available_in_test].copy()\n", + "\n", + "# Fill missing features with 0.0\n", + "missing_feats = set(final_features) - set(available_in_test)\n", + "if missing_feats:\n", + " for feat in missing_feats:\n", + " X_test_final[feat] = 0.0\n", + " X_test_final = X_test_final[final_features]\n", + "\n", + "# Save transformed test data\n", + "X_test_final.to_csv(config.PROCESSED_DIR / \"X_test_selected.csv\")\n", + "y_test.to_csv(config.PROCESSED_DIR / \"y_test.csv\")\n", "\n", "print(\"Model training complete. Artifacts saved:\")\n", "print(f\" - {config.MODELS_DIR / 'best_model_xgboost.joblib'}\")\n", - "print(f\" - {config.MODELS_DIR / 'fitted_layer1_selector.joblib'}\")" + "print(f\" - {config.MODELS_DIR / 'fitted_selector.joblib'}\")\n", + "print(f\" - {config.PROCESSED_DIR / 'X_test_selected.csv'}\")\n", + "print(f\" - {config.PROCESSED_DIR / 'y_test.csv'}\")" ] }, { diff --git a/core/src/__pycache__/__init__.cpython-312.pyc b/core/src/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000..b4af080 Binary files /dev/null and b/core/src/__pycache__/__init__.cpython-312.pyc differ diff --git a/core/src/__pycache__/batch_correction.cpython-312.pyc b/core/src/__pycache__/batch_correction.cpython-312.pyc new file mode 100644 index 0000000..3d07099 Binary files /dev/null and b/core/src/__pycache__/batch_correction.cpython-312.pyc differ diff --git a/core/src/__pycache__/clinical.cpython-312.pyc b/core/src/__pycache__/clinical.cpython-312.pyc new file mode 100644 index 0000000..7585d62 Binary files /dev/null and b/core/src/__pycache__/clinical.cpython-312.pyc differ diff --git a/core/src/__pycache__/clinical_utility.cpython-312.pyc b/core/src/__pycache__/clinical_utility.cpython-312.pyc new file mode 100644 index 0000000..e8461db Binary files /dev/null and b/core/src/__pycache__/clinical_utility.cpython-312.pyc differ diff --git a/core/src/__pycache__/evaluation.cpython-312.pyc b/core/src/__pycache__/evaluation.cpython-312.pyc new file mode 100644 index 0000000..e596310 Binary files /dev/null and b/core/src/__pycache__/evaluation.cpython-312.pyc differ diff --git a/core/src/__pycache__/explainability.cpython-312.pyc b/core/src/__pycache__/explainability.cpython-312.pyc new file mode 100644 index 0000000..a7b79e8 Binary files /dev/null and b/core/src/__pycache__/explainability.cpython-312.pyc differ diff --git a/core/src/__pycache__/feature_selection.cpython-312.pyc b/core/src/__pycache__/feature_selection.cpython-312.pyc new file mode 100644 index 0000000..b2fccfd Binary files /dev/null and b/core/src/__pycache__/feature_selection.cpython-312.pyc differ diff --git a/core/src/__pycache__/features_config.cpython-312.pyc b/core/src/__pycache__/features_config.cpython-312.pyc new file mode 100644 index 0000000..34d69e4 Binary files /dev/null and b/core/src/__pycache__/features_config.cpython-312.pyc differ diff --git a/core/src/__pycache__/genomics.cpython-312.pyc b/core/src/__pycache__/genomics.cpython-312.pyc new file mode 100644 index 0000000..7ca2dd9 Binary files /dev/null and b/core/src/__pycache__/genomics.cpython-312.pyc differ diff --git a/core/src/__pycache__/io.cpython-312.pyc b/core/src/__pycache__/io.cpython-312.pyc new file mode 100644 index 0000000..cdf3520 Binary files /dev/null and b/core/src/__pycache__/io.cpython-312.pyc differ diff --git a/core/src/__pycache__/leakage.cpython-312.pyc b/core/src/__pycache__/leakage.cpython-312.pyc new file mode 100644 index 0000000..2148b98 Binary files /dev/null and b/core/src/__pycache__/leakage.cpython-312.pyc differ diff --git a/core/src/__pycache__/merge.cpython-312.pyc b/core/src/__pycache__/merge.cpython-312.pyc new file mode 100644 index 0000000..3c0808d Binary files /dev/null and b/core/src/__pycache__/merge.cpython-312.pyc differ diff --git a/core/src/__pycache__/models.cpython-312.pyc b/core/src/__pycache__/models.cpython-312.pyc new file mode 100644 index 0000000..207e52b Binary files /dev/null and b/core/src/__pycache__/models.cpython-312.pyc differ diff --git a/core/src/__pycache__/optimization.cpython-312.pyc b/core/src/__pycache__/optimization.cpython-312.pyc new file mode 100644 index 0000000..3cd5a5c Binary files /dev/null and b/core/src/__pycache__/optimization.cpython-312.pyc differ diff --git a/core/src/__pycache__/pipeline.cpython-312.pyc b/core/src/__pycache__/pipeline.cpython-312.pyc new file mode 100644 index 0000000..32ce793 Binary files /dev/null and b/core/src/__pycache__/pipeline.cpython-312.pyc differ diff --git a/core/src/__pycache__/preprocessing.cpython-312.pyc b/core/src/__pycache__/preprocessing.cpython-312.pyc new file mode 100644 index 0000000..349299e Binary files /dev/null and b/core/src/__pycache__/preprocessing.cpython-312.pyc differ diff --git a/core/src/__pycache__/survival_analysis.cpython-312.pyc b/core/src/__pycache__/survival_analysis.cpython-312.pyc new file mode 100644 index 0000000..4c2f61f Binary files /dev/null and b/core/src/__pycache__/survival_analysis.cpython-312.pyc differ diff --git a/core/src/__pycache__/validation.cpython-312.pyc b/core/src/__pycache__/validation.cpython-312.pyc new file mode 100644 index 0000000..3ca47cb Binary files /dev/null and b/core/src/__pycache__/validation.cpython-312.pyc differ diff --git a/core/src/__pycache__/visualization.cpython-312.pyc b/core/src/__pycache__/visualization.cpython-312.pyc new file mode 100644 index 0000000..51a0031 Binary files /dev/null and b/core/src/__pycache__/visualization.cpython-312.pyc differ diff --git a/core/src/feature_selection.py b/core/src/feature_selection.py index 4fc3ead..6e8fa25 100644 --- a/core/src/feature_selection.py +++ b/core/src/feature_selection.py @@ -1,10 +1,9 @@ """ -Feature selection for TCGA-PRAD BCR prediction using 3-Layer Strategy. +Feature selection for TCGA-PRAD BCR prediction using simple MI-based strategy. Implements: -Layer 1: Raw Gene Filtering & Selection (Variance + MI) -Layer 2: Extended Domain-Specific Feature Engineering (~20 features) - + Automatic High-Correlation Removal (|r| > 0.9) -Layer 3: Final Assembly (PSO Genes + Clean Engineered Features) +1. Variance Threshold + Mutual Information for gene selection +2. Domain-specific feature engineering (7 features) +3. Simple transformation for test/external data """ from __future__ import annotations @@ -28,45 +27,45 @@ # ============================================================================= -# LAYER 2: EXTENDED FEATURE ENGINEERING + CORRELATION FILTER +# FEATURE ENGINEERING (7 Core Features) # ============================================================================= -def create_extended_engineered_features( +def create_engineered_features( X: pd.DataFrame, - selected_genes: Optional[List[str]] = None, - correlation_threshold: float = 0.90 + selected_genes: Optional[List[str]] = None ) -> Tuple[pd.DataFrame, List[str]]: """ - Create ~20 engineered features across 3 sub-layers and remove multicollinearity. - + Create 7 domain-specific engineered features. + Args: X: Input DataFrame (genes + clinical columns) - selected_genes: Genes from Layer 1 (used to filter pathway genes) - correlation_threshold: Max allowed pairwise correlation + selected_genes: Genes from feature selection (used to filter pathway genes) Returns: - Tuple of (DataFrame with clean engineered features, list of feature names) + Tuple of (DataFrame with engineered features, list of feature names) """ X = X.copy() created_features: List[str] = [] - # --- SUB-LAYER 2A: Base Clinical & Pathway Scores (7 features) --- + # 1. Gleason Total if GLEASON_PRIMARY_COL in X.columns and GLEASON_SECONDARY_COL in X.columns: X['Gleason_Total'] = X[GLEASON_PRIMARY_COL] + X[GLEASON_SECONDARY_COL] X['High_Risk_Gleason'] = ((X[GLEASON_PRIMARY_COL] >= 4) | (X[GLEASON_SECONDARY_COL] >= 4)).astype(int) created_features.extend(['Gleason_Total', 'High_Risk_Gleason']) + # 2. Margin x LymphNode interaction if MARGIN_COL in X.columns and LYMPH_NODE_COL in X.columns: X['Margin_x_LymphNode'] = X[MARGIN_COL].astype(float) * X[LYMPH_NODE_COL].astype(float) created_features.append('Margin_x_LymphNode') + # 3. T-Stage Risk t_stage_cols = [c for c in X.columns if 'Tumor Stage Code_T3' in c or 'Tumor Stage Code_T4' in c] if len(t_stage_cols) >= 2: X['T_Stage_Risk'] = X[t_stage_cols].sum(axis=1) created_features.append('T_Stage_Risk') - # Pathway scores (lenient mode for external validation) + # 4-6. Pathway scores for name, gene_set in [('PSA_Pathway_Score', PSA_GENES), ('AR_Signaling_Score', AR_GENES), ('Proliferation_Score', PROLIF_GENES)]: @@ -78,272 +77,138 @@ def create_extended_engineered_features( X[name] = X[available].mean(axis=1) created_features.append(name) - # --- SUB-LAYER 2B: Gene-Clinical Interactions (~8 features) --- - interactions = [ - ('AR_x_Gleason', 'AR_Signaling_Score', 'Gleason_Total'), - ('Prolif_x_Margin', 'Proliferation_Score', 'Margin_x_LymphNode'), - ('PSA_x_TStage', 'PSA_Pathway_Score', 'T_Stage_Risk'), - ('AR_x_Prolif', 'AR_Signaling_Score', 'Proliferation_Score'), - ('HighRisk_x_Prolif', 'High_Risk_Gleason', 'Proliferation_Score'), - ('Gleason_x_AR', 'Gleason_Total', 'AR_Signaling_Score'), - ('TStage_x_Margin', 'T_Stage_Risk', MARGIN_COL), - ('PSA_x_Gleason', 'PSA_Pathway_Score', 'Gleason_Total') - ] - - for new_col, col1, col2 in interactions: - if col1 in X.columns and col2 in X.columns: - X[new_col] = X[col1] * X[col2] - created_features.append(new_col) - - # --- SUB-LAYER 2C: Multi-Pathway Ratios & Composites (~5 features) --- - ratios = { - 'AR_to_Prolif_Ratio': lambda df: df['AR_Signaling_Score'] / (df['Proliferation_Score'] + 1e-6), - 'PSA_to_AR_Ratio': lambda df: df['PSA_Pathway_Score'] / (df['AR_Signaling_Score'] + 1e-6), - 'Pathway_Balance': lambda df: df['PSA_Pathway_Score'] + df['AR_Signaling_Score'] - df['Proliferation_Score'], - 'Combined_Risk': lambda df: df['Gleason_Total'] * 0.4 + df.get('T_Stage_Risk', pd.Series(0, index=df.index)) * 0.3 + df['Proliferation_Score'] * 0.3, - 'Gene_Variability': lambda df: df[[g for g in (selected_genes or []) if g in df.columns]].std(axis=1) - if len([g for g in (selected_genes or []) if g in df.columns]) > 1 - else pd.Series(0.0, index=df.index) - } - - req_map = { - 'AR_to_Prolif_Ratio': ['AR_Signaling_Score', 'Proliferation_Score'], - 'PSA_to_AR_Ratio': ['PSA_Pathway_Score', 'AR_Signaling_Score'], - 'Pathway_Balance': ['PSA_Pathway_Score', 'AR_Signaling_Score', 'Proliferation_Score'], - 'Combined_Risk': ['Gleason_Total', 'Proliferation_Score'], - 'Gene_Variability': [] - } - - for feat_name, calc_fn in ratios.items(): - req = req_map.get(feat_name, []) - if all(c in X.columns for c in req): - try: - X[feat_name] = calc_fn(X) - created_features.append(feat_name) - except Exception as e: - logger.warning(f"Failed to create {feat_name}: {e}") - - # --- AUTOMATIC CORRELATION FILTER --- - if len(created_features) > 1: - corr_matrix = X[created_features].corr().abs() - upper_tri = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool)) - to_drop = [col for col in upper_tri.columns if any(upper_tri[col] > correlation_threshold)] - - if to_drop: - logger.info(f"Layer 2 - Removed {len(to_drop)} highly correlated features (|r|>{correlation_threshold}): {to_drop}") - created_features = [f for f in created_features if f not in to_drop] - X = X.drop(columns=to_drop, errors='ignore') - - logger.info(f"Layer 2 - Created {len(created_features)} clean engineered features") + logger.info(f"Created {len(created_features)} engineered features: {created_features}") return X, created_features -# Backward compatibility alias -def create_engineered_features(X, selected_genes=None, strict_mode=False): - """Alias for backward compatibility with old notebooks.""" - return create_extended_engineered_features( - X=X, - selected_genes=selected_genes, - correlation_threshold=0.90 - ) - - # ============================================================================= -# LAYER 1: RAW GENE SELECTION +# FEATURE SELECTION (Variance + MI) # ============================================================================= -def fit_layer1_selector( - X_genes: pd.DataFrame, y_train: pd.Series, +def run_feature_selection( + X: pd.DataFrame, + y: pd.Series, variance_threshold: float = config.VARIANCE_THRESHOLD, mi_top_k: int = config.MI_TOP_K, random_state: int = config.RANDOM_STATE -) -> Dict[str, Any]: - """Fit Variance + MI selector on training genes ONLY.""" +) -> Tuple[Any, List[str]]: + """ + Perform feature selection using Variance Threshold + Mutual Information. + + Args: + X: Input DataFrame + y: Target Series + variance_threshold: Minimum variance threshold + mi_top_k: Number of top features to select via MI + random_state: Random seed + + Returns: + Tuple of (fitted_selector dict, list of selected feature names) + """ + # Separate clinical and gene columns + clinical_cols = [c for c in X.columns if any(kw in c for kw in + ['Gleason', 'Margin', 'Lymph', 'Tumor Stage', 'PSA'])] + gene_cols = [c for c in X.columns if c not in clinical_cols] + + logger.info(f"Separating {len(gene_cols)} genes and {len(clinical_cols)} clinical features") + + # Impute missing values in genes imputer = SimpleImputer(strategy="median") - X_imp = pd.DataFrame(imputer.fit_transform(X_genes), columns=X_genes.columns, index=X_genes.index) - + X_genes_imp = pd.DataFrame( + imputer.fit_transform(X[gene_cols]), + columns=gene_cols, + index=X.index + ) + + # Variance Threshold vt = VarianceThreshold(threshold=variance_threshold) - X_var = vt.fit_transform(X_imp) - var_features = X_genes.columns[vt.get_support()].tolist() - - mi_scores = mutual_info_classif(X_var, y_train, random_state=random_state) + X_var = vt.fit_transform(X_genes_imp) + var_features = X_genes_imp.columns[vt.get_support()].tolist() + logger.info(f"After variance threshold: {len(var_features)} features") + + # Mutual Information + mi_scores = mutual_info_classif(X_var, y, random_state=random_state) mi_series = pd.Series(mi_scores, index=var_features).sort_values(ascending=False) - mi_features = mi_series.head(min(mi_top_k, len(mi_series))).index.tolist() - - logger.info(f"Layer 1 - Selected {len(mi_features)} genes from {len(X_genes.columns)} raw genes") - return {"imputer": imputer, "vt": vt, "mi_features": mi_features, "is_fitted": True} + selected_genes = mi_series.head(min(mi_top_k, len(mi_series))).index.tolist() + + logger.info(f"Selected {len(selected_genes)} genes via MI (top {mi_top_k})") + + # Create fitted selector object + fitted_selector = { + "imputer": imputer, + "variance_threshold": vt, + "selected_genes": selected_genes, + "clinical_cols": clinical_cols, + "is_fitted": True + } + + return fitted_selector, selected_genes -def transform_layer1(X_genes: pd.DataFrame, fitted: dict) -> pd.DataFrame: - """Transform test/external genes using fitted Layer 1 selector.""" - fit_cols = fitted["vt"].get_feature_names_out().tolist() if hasattr(fitted["vt"], 'get_feature_names_out') else fitted["vt"].get_support(indices=True).tolist() - for c in fit_cols: +def transform_selected( + X: pd.DataFrame, + fitted_selector: Dict[str, Any] +) -> pd.DataFrame: + """ + Apply fitted feature selector to new data (test/external). + + Args: + X: New DataFrame to transform (genes + clinical) + fitted_selector: Fitted selector from run_feature_selection + + Returns: + DataFrame with selected genes + clinical features + """ + X = X.copy() + + # Get original column lists + clinical_cols = fitted_selector["clinical_cols"] + selected_genes = fitted_selector["selected_genes"] + imputer = fitted_selector["imputer"] + vt = fitted_selector["variance_threshold"] + + # Separate genes and clinical in new data + all_cols = set(X.columns) + gene_cols = [c for c in all_cols if c not in clinical_cols] + + logger.info(f"Transform: {len(gene_cols)} genes, {len(clinical_cols)} clinical cols") + + # Extract gene columns only + X_genes = X[gene_cols].copy() + + # Get variance-filtered columns from fit time + var_cols = (vt.get_feature_names_out().tolist() + if hasattr(vt, 'get_feature_names_out') + else list(imputer.feature_names_in_)) + + # Ensure all variance-filtered columns exist (fill missing with NaN) + for c in var_cols: if c not in X_genes.columns: - X_genes = X_genes.copy() X_genes[c] = np.nan - - X_imp = pd.DataFrame(fitted["imputer"].transform(X_genes[fit_cols]), columns=fit_cols, index=X_genes.index) - valid_mi = [g for g in fitted["mi_features"] if g in X_imp.columns] - logger.info(f"Layer 1 - Transform: {len(valid_mi)}/{len(fitted['mi_features'])} genes available") - return X_imp[valid_mi] - - -# ============================================================================= -# PSO HELPER FUNCTIONS -# ============================================================================= - -def sigmoid(x: np.ndarray) -> np.ndarray: - return 1.0 / (1.0 + np.exp(-np.clip(x, -50, 50))) - -def repair_exact_k(mask: np.ndarray, k: int, rng: np.random.RandomState) -> np.ndarray: - idx = np.flatnonzero(mask) - if len(idx) > k: - keep = rng.choice(idx, size=k, replace=False) - out = np.zeros_like(mask, dtype=int); out[keep] = 1; return out - if len(idx) < k: - zero_idx = np.flatnonzero(mask == 0) - add_n = min(k - len(idx), len(zero_idx)) - if add_n > 0: - mask = mask.copy(); mask[rng.choice(zero_idx, size=add_n, replace=False)] = 1 - return mask - -def pso_feature_select( - X_fit: pd.DataFrame, y_fit: pd.Series, candidate_features: List[str], - n_features: int = config.PSO_FINAL_K, - n_particles: int = config.PSO_N_PARTICLES, - n_iterations: int = config.PSO_N_ITERATIONS, - penalty_alpha: float = config.PSO_PENALTY_ALPHA, - random_state: int = config.RANDOM_STATE -) -> Tuple[List[str], float]: - """Binary PSO with penalty for fixed-cardinality feature selection.""" - from src.models import make_xgb, xgb_safe_frame - - candidates = list(candidate_features) - if len(candidates) <= n_features: - return candidates, np.nan - - Xc = pd.DataFrame(SimpleImputer(strategy="median").fit_transform(X_fit[candidates]), columns=candidates) - n_dim = len(candidates) - rng = np.random.RandomState(random_state) - inner_cv = StratifiedKFold(n_splits=config.PSO_INNER_SPLITS, shuffle=True, random_state=random_state) - cache: Dict[tuple, float] = {} - - def fitness(mask: np.ndarray) -> float: - repaired = repair_exact_k(mask.astype(int), n_features, rng) - key = tuple(np.flatnonzero(repaired).tolist()) - if key in cache: return cache[key] - - cols = [candidates[i] for i in key] - scores = [] - for tr_idx, va_idx in inner_cv.split(Xc, y_fit): - model = make_xgb(y_fit.iloc[tr_idx]) - model.fit(xgb_safe_frame(Xc.iloc[tr_idx][cols]), y_fit.iloc[tr_idx]) - p = model.predict_proba(xgb_safe_frame(Xc.iloc[va_idx][cols]))[:, 1] - scores.append(roc_auc_score(y_fit.iloc[va_idx], p)) - - mean_auc = float(np.mean(scores)) - final = mean_auc - (penalty_alpha * n_features) - cache[key] = final - return final - - pos = rng.uniform(-1, 1, (n_particles, n_dim)) - vel = rng.uniform(-0.1, 0.1, (n_particles, n_dim)) - binary = np.array([repair_exact_k((sigmoid(pos[i]) > 0.5).astype(int), n_features, rng) for i in range(n_particles)]) - - pbest_pos = binary.copy() - pbest_score = np.array([fitness(m) for m in binary]) - gbest_idx = int(np.argmax(pbest_score)) - gbest_pos = pbest_pos[gbest_idx].copy() - gbest_score = float(pbest_score[gbest_idx]) - - for it in range(n_iterations): - r1, r2 = rng.rand(n_particles, n_dim), rng.rand(n_particles, n_dim) - vel = config.PSO_W * vel + config.PSO_C1 * r1 * (pbest_pos - binary) + config.PSO_C2 * r2 * (gbest_pos - binary) - vel = np.clip(vel, -4, 4) - binary = np.array([repair_exact_k((rng.rand(n_dim) < sigmoid(vel[i])).astype(int), n_features, rng) for i in range(n_particles)]) - - scores = np.array([fitness(m) for m in binary]) - improved = scores > pbest_score - pbest_pos[improved] = binary[improved]; pbest_score[improved] = scores[improved] - - best_idx = int(np.argmax(pbest_score)) - if pbest_score[best_idx] > gbest_score: - gbest_pos = pbest_pos[best_idx].copy(); gbest_score = float(pbest_score[best_idx]) - - if (it + 1) % 5 == 0: - logger.info(f" PSO iter {it+1}/{n_iterations}: Fitness={gbest_score:.4f} (Raw AUC~={gbest_score + penalty_alpha*n_features:.4f})") - - selected = [candidates[i] for i in np.flatnonzero(gbest_pos)] - logger.info(f"PSO selected {len(selected)} features") - return selected, gbest_score - - -# ============================================================================= -# MAIN PIPELINE: 3-LAYER STRATEGY -# ============================================================================= - -def run_3layer_feature_selection( - X_train: pd.DataFrame, y_train: pd.Series, - clinical_cols: Optional[List[str]] = None, - run_pso: bool = True, - random_state: int = config.RANDOM_STATE -) -> Tuple[dict, List[str]]: - """Execute the complete 3-Layer Feature Selection Pipeline.""" - if clinical_cols is None: - clinical_cols = [c for c in X_train.columns if any(kw in c for kw in - ['Gleason', 'Margin', 'Lymph', 'Tumor Stage', 'PSA'])] - gene_cols = [c for c in X_train.columns if c not in clinical_cols] - - # === LAYER 1 === - fitted_l1 = fit_layer1_selector(X_train[gene_cols], y_train, random_state=random_state) - selected_genes = fitted_l1["mi_features"] - X_selected_genes = transform_layer1(X_train[gene_cols], fitted_l1) - - # === LAYER 2 === - X_for_eng = pd.concat([X_selected_genes, X_train[clinical_cols]], axis=1) - X_eng, eng_features = create_extended_engineered_features(X_for_eng, selected_genes=selected_genes) - - # === LAYER 3: ASSEMBLY & PSO === - candidate_pool = list(set(selected_genes + eng_features)) - candidate_pool = [f for f in candidate_pool if f in X_eng.columns] - - if run_pso: - final_features, _ = pso_feature_select( - X_eng, y_train, candidate_pool, - n_features=min(config.PSO_FINAL_K, len(candidate_pool)), - random_state=random_state + 1000 - ) - else: - final_features = selected_genes[:config.PSO_FINAL_K] + eng_features - - logger.info(f"3-Layer Pipeline Complete: {len(final_features)} final features") - return fitted_l1, final_features - - -def apply_3layer_to_external( - X_ext_raw: pd.DataFrame, - fitted_l1: dict, - final_feature_list: List[str], - clinical_cols: Optional[List[str]] = None -) -> pd.DataFrame: - """Apply trained 3-layer pipeline to external/test data.""" - if clinical_cols is None: - clinical_cols = [c for c in X_ext_raw.columns if any(kw in c for kw in - ['Gleason', 'Margin', 'Lymph', 'Tumor Stage', 'PSA'])] - gene_cols = [c for c in X_ext_raw.columns if c not in clinical_cols] - - X_sel_genes = transform_layer1(X_ext_raw[gene_cols], fitted_l1) - X_for_eng = pd.concat([X_sel_genes, X_ext_raw[clinical_cols]], axis=1) - X_eng, _ = create_extended_engineered_features(X_for_eng, selected_genes=fitted_l1["mi_features"]) - - available = [f for f in final_feature_list if f in X_eng.columns] - missing = [f for f in final_feature_list if f not in X_eng.columns] - - if missing: - logger.warning(f"External data missing {len(missing)} features: {missing[:5]}...") - for m in missing: - X_eng[m] = 0.0 - - X_final = X_eng[final_feature_list].fillna(0.0) - logger.info(f"External validation: {len(available)}/{len(final_feature_list)} features available") + + # Reorder to match fit time exactly + X_genes = X_genes[var_cols] + + # Impute genes using fitted imputer + X_genes_imp = pd.DataFrame( + imputer.transform(X_genes), + columns=var_cols, + index=X.index + ) + + # Select MI features that are available + available_genes = [g for g in selected_genes if g in X_genes_imp.columns] + X_genes_selected = X_genes_imp[available_genes] + + logger.info(f"Transform: {len(available_genes)}/{len(selected_genes)} genes available") + + # Get clinical features (ensure they exist) + available_clinical = [c for c in clinical_cols if c in X.columns] + X_clinical = X[available_clinical].copy() + + logger.info(f"Transform: {len(available_clinical)}/{len(clinical_cols)} clinical cols available") + + # Combine selected genes + clinical features + X_final = pd.concat([X_genes_selected, X_clinical], axis=1) + return X_final