Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
122 changes: 8 additions & 114 deletions .gitignore
Original file line number Diff line number Diff line change
@@ -1,117 +1,11 @@
# ============================================
# Python & Environment
# ============================================
__pycache__/
*.py[cod]
```
# Compiled Python files
*.pyc
*.pyo
*.pyd
.Python
venv/
.venv/
.env
.env.local
.env.*
ENV/
env/
virtualenv/

# ============================================
# Build & Dependencies
# ============================================
pip-wheel-metadata/
.tox/
.coverage
htmlcov/
.pytest_cache/
.mypy_cache/
.ruff_cache/
build/
dist/
*.egg-info/
*.egg
wheels/

# ============================================
# IDE & OS
# ============================================
.idea/
.vscode/
*.swp
*.swo
*~
.DS_Store
Thumbs.db
desktop.ini



# ============================================
# Project Data & Outputs
# ============================================
# پوشه‌های داده و خروجی (همه محتویات)
data/
outputs/
models/
logs/
results/
figures/
plots/
reports/

# ============================================
# Data File Types
# ============================================
*.csv
*.tsv
*.xls
*.xlsx
*.txt
*.json
*.jsonl
*.xml
*.yaml
*.yml
*.parquet
*.feather
*.h5
*.hdf5
*.pkl
*.pickle
*.joblib
__pycache__/

# ============================================
# Images & Binary Files
# ============================================
*.png
*.jpg
*.jpeg
*.gif
*.bmp
*.tiff
*.svg
*.eps
*.pth
*.pt
*.onnx
*.pb
*.weights
*.bin
# Archives
*.tar.gz

# ============================================
# Logs & Temp Files
# ============================================
*.log
*.bak
*.tmp
*.temp
*.pid
*.seed
core/outputs/
core/outputs/**
*.png
*.jpg
*.jpeg
*.joblib
*.json
*.tsv
# Original rules preserved
core/src/feature_selection.py
```
47 changes: 47 additions & 0 deletions core/.gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1,47 @@
# Ignore data files
data/processed/*.csv
data/processed/*.joblib
data/raw/*.csv
data/interim/*

# Ignore model outputs
outputs/models/*.joblib
outputs/models/*.pkl
outputs/models/*.h5

# Ignore large tables and figures if needed
outputs/tables/*.csv
outputs/figures/*.png
outputs/figures/*.jpg

# Python cache
__pycache__/
*.py[cod]
*$py.class
*.so
.Python
env/
venv/
ENV/
build/
develop-eggs/
dist/
downloads/
eggs/
.eggs/
lib/
lib64/
parts/
sdist/
var/
wheels/
*.egg-info/
.installed.cfg
*.egg

# Jupyter Notebook checkpoints
.ipynb_checkpoints/

# OS files
.DS_Store
Thumbs.db
Binary file added core/__pycache__/config.cpython-312.pyc
Binary file not shown.
51 changes: 29 additions & 22 deletions core/notebooks/04_feature_selection.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -5,15 +5,15 @@
"id": "2744288e",
"metadata": {},
"source": [
"# 04 - Feature Selection (3-Layer Pipeline)\n",
"# 04 - Feature Selection (Simple MI Pipeline)\n",
"\n",
"Layer 1: Variance + Mutual Information on raw genes\n",
"Step 1: Variance Threshold + Mutual Information on genes\n",
"\n",
"Layer 2: Domain-specific feature engineering with correlation filtering\n",
"Step 2: Domain-specific feature engineering (7 features)\n",
"\n",
"Layer 3: Binary PSO assembly over genes plus engineered features\n",
"Step 3: Combine selected genes + engineered + clinical features\n",
"\n",
"All logic lives in `src/feature_selection.py`. This notebook only loads data, detects clinical columns, calls the pipeline, and saves artifacts."
"All logic lives in `src/feature_selection.py`. This notebook only loads data, runs the pipeline, and saves artifacts."
]
},
{
Expand All @@ -33,7 +33,7 @@
"import joblib\n",
"import config\n",
"from src.io import logger\n",
"from src.feature_selection import run_3layer_feature_selection"
"from src.feature_selection import run_feature_selection"
]
},
{
Expand Down Expand Up @@ -72,9 +72,13 @@
"id": "74664d66",
"metadata": {},
"source": [
"## Step 2: Detect Clinical / Engineered Columns\n",
"## Step 2: Run Feature Selection Pipeline\n",
"\n",
"These columns are excluded from Layer 1 gene filtering and passed explicitly to `run_3layer_feature_selection` as `clinical_cols`."
"This performs:\n",
"1. Variance threshold on genes\n",
"2. Mutual Information to select top K genes\n",
"3. Creates 7 engineered features\n",
"4. Returns fitted selector + list of all selected features"
]
},
{
Expand Down Expand Up @@ -243,7 +247,7 @@
"id": "599f7e30",
"metadata": {},
"source": [
"## Step 4: Save Artifacts"
"## Step 3: Save Artifacts"
]
},
{
Expand All @@ -262,20 +266,23 @@
}
],
"source": [
"if final_features:\n",
" joblib.dump(fitted_l1, config.MODELS_DIR / \"fitted_layer1_selector.joblib\")\n",
" print(f\"Saved Layer 1 selector to: {config.MODELS_DIR / 'fitted_layer1_selector.joblib'}\")\n",
"# Save fitted selector\n",
"joblib.dump(fitted_selector, config.MODELS_DIR / \"fitted_selector.joblib\")\n",
"print(f\"Saved selector to: {config.MODELS_DIR / 'fitted_selector.joblib'}\")\n",
"\n",
" pd.DataFrame({\"feature\": final_features}).to_csv(\n",
" config.TABLES_DIR / \"selected_features_final.csv\", index=False\n",
" )\n",
" print(\n",
" f\"Saved feature list ({len(final_features)} items) to: \"\n",
" f\"{config.TABLES_DIR / 'selected_features_final.csv'}\"\n",
" )\n",
"else:\n",
" print(\"WARNING: no features were generated. Nothing was saved.\")\n",
" print(\"Resolve the error above and re-run this notebook.\")"
"# Save feature list\n",
"pd.DataFrame({\"feature\": final_features}).to_csv(\n",
" config.TABLES_DIR / \"selected_features_final.csv\", index=False\n",
")\n",
"print(f\"Saved feature list ({len(final_features)} items) to: {config.TABLES_DIR / 'selected_features_final.csv'}\")\n",
"\n",
"# Save engineered training data for reference\n",
"X_train_eng = X_train.copy()\n",
"for feat in eng_features:\n",
" if feat not in X_train_eng.columns:\n",
" X_train_eng[feat] = X_eng[feat]\n",
"\n",
"print(\"Feature selection complete!\")"
]
}
],
Expand Down
63 changes: 46 additions & 17 deletions core/notebooks/05_Model_Training.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -5,9 +5,9 @@
"id": "a8be3165",
"metadata": {},
"source": [
"# 05 - Model Training (3-Layer Pipeline)\n",
"# 05 - Model Training (Simple Pipeline)\n",
"\n",
"Loads the Layer 1 selector and final feature list produced by notebook 04, re-applies Layer 2 feature engineering to the full training set, filters to the final feature list, tunes XGBoost with Optuna, trains the final model, and saves the model artifact."
"Loads the fitted selector and final feature list produced by notebook 04, applies transform_selected to test data, tunes XGBoost with Optuna, trains the final model, and saves the model artifact."
]
},
{
Expand Down Expand Up @@ -40,7 +40,7 @@
"from src.io import logger\n",
"from src.models import build_model, xgb_safe_frame\n",
"from src.optimization import optimize_model\n",
"from src.feature_selection import create_extended_engineered_features\n",
"from src.feature_selection import transform_selected\n",
"from sklearn.metrics import roc_auc_score"
]
},
Expand Down Expand Up @@ -82,7 +82,7 @@
"source": [
"## Step 2: Load Artifacts From Notebook 04\n",
"\n",
"Requires `fitted_layer1_selector.joblib` and `selected_features_final.csv`. Raises a clear error if notebook 04 has not been run."
"Requires `fitted_selector.joblib` and `selected_features_final.csv`. Raises a clear error if notebook 04 has not been run."
]
},
{
Expand All @@ -100,19 +100,19 @@
}
],
"source": [
"selector_path = config.MODELS_DIR / \"fitted_layer1_selector.joblib\"\n",
"selector_path = config.MODELS_DIR / \"fitted_selector.joblib\"\n",
"features_path = config.TABLES_DIR / \"selected_features_final.csv\"\n",
"\n",
"try:\n",
" fitted_l1 = joblib.load(selector_path)\n",
" fitted_selector = joblib.load(selector_path)\n",
" selected_df = pd.read_csv(features_path)\n",
" final_features = selected_df[\"feature\"].tolist()\n",
" logger.info(f\"Loaded {len(final_features)} final features from the 3-layer pipeline.\")\n",
" logger.info(f\"Loaded {len(final_features)} final features from the pipeline.\")\n",
"except FileNotFoundError as e:\n",
" raise FileNotFoundError(\n",
" f\"{e}\\n\\n\"\n",
" \"Please run '04_feature_selection.ipynb' first to generate \"\n",
" \"'fitted_layer1_selector.joblib' and 'selected_features_final.csv'.\"\n",
" \"'fitted_selector.joblib' and 'selected_features_final.csv'.\"\n",
" )"
]
},
Expand All @@ -121,9 +121,9 @@
"id": "6f2c1020",
"metadata": {},
"source": [
"## Step 3: Apply Layer 2 Feature Engineering and Filter to Final Features\n",
"## Step 3: Apply Feature Selection to Training Data\n",
"\n",
"Engineering is applied to the full training set before filtering, matching how notebook 04 built the candidate pool. Any final feature missing after engineering (should not normally happen, but can for edge splits) is filled with 0.0."
"Use transform_selected to ensure training data matches the exact feature set that will be used for test/external data."
]
},
{
Expand All @@ -142,13 +142,14 @@
}
],
"source": [
"X_train_eng, _ = create_extended_engineered_features(\n",
" X_train, selected_genes=fitted_l1[\"mi_features\"]\n",
")\n",
"# Transform training data using fitted selector\n",
"X_train_selected = transform_selected(X_train, fitted_selector)\n",
"\n",
"available_in_train = [f for f in final_features if f in X_train_eng.columns]\n",
"X_train_final = X_train_eng[available_in_train].copy()\n",
"# Ensure we only keep features in final_features list\n",
"available_in_train = [f for f in final_features if f in X_train_selected.columns]\n",
"X_train_final = X_train_selected[available_in_train].copy()\n",
"\n",
"# Fill any missing features with 0.0 (should not happen normally)\n",
"missing_feats = set(final_features) - set(available_in_train)\n",
"if missing_feats:\n",
" logger.warning(f\"Missing {len(missing_feats)} features in training data: {list(missing_feats)[:5]}...\")\n",
Expand Down Expand Up @@ -1703,12 +1704,40 @@
"source": [
"config.MODELS_DIR.mkdir(parents=True, exist_ok=True)\n",
"\n",
"# Save the trained model\n",
"joblib.dump(final_model, config.MODELS_DIR / \"best_model_xgboost.joblib\")\n",
"joblib.dump(fitted_l1, config.MODELS_DIR / \"fitted_layer1_selector.joblib\")\n",
"\n",
"# Save the fitted selector for transforming test data\n",
"joblib.dump(fitted_selector, config.MODELS_DIR / \"fitted_selector.joblib\")\n",
"\n",
"# CRITICAL: Transform and save test data\n",
"X_test = pd.read_csv(config.PROCESSED_DIR / \"X_test_preprocessed.csv\", index_col=0)\n",
"y_test_df = pd.read_csv(config.PROCESSED_DIR / \"y_test.csv\")\n",
"y_test = y_test_df.iloc[:, 0] if len(y_test_df.columns) == 1 else y_test_df[\"BCR\"]\n",
"\n",
"# Apply the same transformation to test data\n",
"X_test_selected = transform_selected(X_test, fitted_selector)\n",
"\n",
"# Ensure test data has same features as training\n",
"available_in_test = [f for f in final_features if f in X_test_selected.columns]\n",
"X_test_final = X_test_selected[available_in_test].copy()\n",
"\n",
"# Fill missing features with 0.0\n",
"missing_feats = set(final_features) - set(available_in_test)\n",
"if missing_feats:\n",
" for feat in missing_feats:\n",
" X_test_final[feat] = 0.0\n",
" X_test_final = X_test_final[final_features]\n",
"\n",
"# Save transformed test data\n",
"X_test_final.to_csv(config.PROCESSED_DIR / \"X_test_selected.csv\")\n",
"y_test.to_csv(config.PROCESSED_DIR / \"y_test.csv\")\n",
"\n",
"print(\"Model training complete. Artifacts saved:\")\n",
"print(f\" - {config.MODELS_DIR / 'best_model_xgboost.joblib'}\")\n",
"print(f\" - {config.MODELS_DIR / 'fitted_layer1_selector.joblib'}\")"
"print(f\" - {config.MODELS_DIR / 'fitted_selector.joblib'}\")\n",
"print(f\" - {config.PROCESSED_DIR / 'X_test_selected.csv'}\")\n",
"print(f\" - {config.PROCESSED_DIR / 'y_test.csv'}\")"
]
},
{
Expand Down
Binary file added core/src/__pycache__/__init__.cpython-312.pyc
Binary file not shown.
Binary file not shown.
Binary file added core/src/__pycache__/clinical.cpython-312.pyc
Binary file not shown.
Binary file not shown.
Binary file added core/src/__pycache__/evaluation.cpython-312.pyc
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file added core/src/__pycache__/genomics.cpython-312.pyc
Binary file not shown.
Binary file added core/src/__pycache__/io.cpython-312.pyc
Binary file not shown.
Binary file added core/src/__pycache__/leakage.cpython-312.pyc
Binary file not shown.
Binary file added core/src/__pycache__/merge.cpython-312.pyc
Binary file not shown.
Binary file added core/src/__pycache__/models.cpython-312.pyc
Binary file not shown.
Binary file added core/src/__pycache__/optimization.cpython-312.pyc
Binary file not shown.
Binary file added core/src/__pycache__/pipeline.cpython-312.pyc
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file added core/src/__pycache__/validation.cpython-312.pyc
Binary file not shown.
Binary file not shown.
Loading