From ff8f35b87e84bc2c6b44c33054f35cb1daf092f4 Mon Sep 17 00:00:00 2001 From: thepineapplepirate Date: Wed, 16 Sep 2026 14:32:52 -0400 Subject: [PATCH 1/2] Modernize QBioCode for the shared Qiskit stack --- MANIFEST.in | 3 +- README.md | 48 ++++- conda-recipe/meta.yaml | 29 +-- .../tutorials/QProfiler/configs/config.yaml | 2 +- .../QProfiler/sc_binary_qprofiler.ipynb | 4 +- .../QPL_example.ipynb | 2 +- .../configs/qpl.yaml | 2 +- .../configs/rf.yaml | 2 +- .../configs/xgb.yaml | 2 +- pyproject.toml | 2 +- qbiocode/apps/sage/sage.py | 8 + qbiocode/embeddings/embed.py | 8 +- requirements-base.txt | 30 --- requirements.txt | 56 ++---- setup.py | 189 +----------------- tutorial/QProfiler/configs/config.yaml | 2 +- tutorial/QProfiler/sc_binary_qprofiler.ipynb | 4 +- .../QPL_example.ipynb | 2 +- .../configs/qpl.yaml | 2 +- .../configs/rf.yaml | 2 +- .../configs/xgb.yaml | 2 +- 21 files changed, 113 insertions(+), 288 deletions(-) delete mode 100644 requirements-base.txt diff --git a/MANIFEST.in b/MANIFEST.in index b240d0a4..14dd2c56 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -8,7 +8,6 @@ include SECURITY.md include SUPPORT.md include CHANGELOG.md include requirements.txt -include requirements-base.txt include .zenodo.json # Include configuration files @@ -35,4 +34,4 @@ global-exclude .git* # Exclude test and development files prune tests prune .pytest_cache -prune archive \ No newline at end of file +prune archive diff --git a/README.md b/README.md index 1c63e9da..8a6847cd 100644 --- a/README.md +++ b/README.md @@ -30,16 +30,45 @@ QBioCode requires Python **3.10 or higher** and has been tested with Python vers #### Install from PyPI (Recommended) ```bash -# Install the latest stable version +# Standard installation: QBioCode's library, applications, quantum, +# machine-learning, visualization, and tutorial runtime dependencies pip install qbiocode -# Install with apps support (QProfiler, QSage) +# Backward-compatible alias; QProfiler and QSage are already included above pip install 'qbiocode[apps]' -# Install with all optional dependencies +# Install everything needed for application use, documentation development, +# testing, linting, formatting, and type checking pip install 'qbiocode[all]' ``` +Use the standard installation when running QBioCode, QProfiler, QSage, its +notebooks, or its library functions. The `apps` extra remains available for +backward compatibility but currently adds no packages beyond the standard +installation. The `all` installation is intended for contributors who need +every optional development tool; it is not required for normal use and is not +needed in runtime images such as the Galaxy interactive tool. + +More focused contributor installations are also available: + +```bash +# Build the Sphinx documentation locally +pip install -e '.[docs]' + +# Run tests, linters, formatters, and type checking +pip install -e '.[dev]' +``` + +Building notebook documentation also requires the Pandoc executable. Install +it with your operating system's package manager, for example +`conda install -c conda-forge pandoc` on macOS or Linux. This is needed only +when rebuilding the documentation, not when running QBioCode or its tutorials. + +A fresh repository clone already includes the prebuilt HTML documentation. +Open `docs/_build/html/index.html` in a browser to view it without installing +the `docs` extra or Pandoc. The current published documentation is also +available from the Documentation link at the top of this README. + #### Install with Conda QBioCode will be available on conda-forge and bioconda after the initial release review process. @@ -78,11 +107,20 @@ source .env/bin/activate # On Windows: .env\Scripts\activate # Install QBioCode in editable mode pip install -e . -# Install with apps support (QProfiler, QSage) +# Backward-compatible alias; applications are part of the standard install pip install -e '.[apps]' + +# Install every optional contributor dependency +pip install -e '.[all]' +``` + +**macOS Users:** XGBoost requires OpenMP. In a Conda or Miniforge +environment, install the cross-platform runtime from conda-forge: +```bash +conda install -c conda-forge llvm-openmp ``` -**macOS Users:** XGBoost requires OpenMP. Install it using Homebrew: +Alternatively, Homebrew users can install it with: ```bash brew install libomp pip install --force-reinstall xgboost diff --git a/conda-recipe/meta.yaml b/conda-recipe/meta.yaml index 0a2dc0b6..4bfed8d2 100644 --- a/conda-recipe/meta.yaml +++ b/conda-recipe/meta.yaml @@ -14,9 +14,9 @@ build: number: 0 script: {{ PYTHON }} -m pip install . -vv entry_points: - - qprofiler=apps.qprofiler.cli:main - - qprofiler-batch=apps.qprofiler.qprofiler_batchmode:main - - qsage=apps.sage.sage:main + - qprofiler=qbiocode.apps.qprofiler.cli:main + - qprofiler-batch=qbiocode.apps.qprofiler.qprofiler_batchmode:main + - qsage=qbiocode.apps.sage.sage:main requirements: host: @@ -27,32 +27,33 @@ requirements: - python >=3.10,<3.13 # Core dependencies - dill - - h5py - hfda - hydra-core - ipykernel - - networkx - numpy - optuna - - pandas + - pandas >=2,<3 # Qiskit ecosystem - - qiskit ==2.2.0 - - qiskit-aer ==0.17.0 + - qiskit ==2.4.2 + - qiskit-aer ==0.17.2 - qiskit-algorithms ==0.4.0 - qiskit-ibm-runtime ==0.44.0 - - qiskit-ibm-transpiler ==0.11.0 - - qiskit-machine-learning ==0.9.0 - - qiskit-nature ==0.7.2 + - qiskit-ibm-transpiler ==0.18.0 + - qiskit-machine-learning ==0.9.1 # ML and data science - scikit-dimension - scikit-learn - scipy - seaborn - - tensorflow - pytorch - tqdm - umap-learn - xgboost + # Single-cell and multi-omics tutorials + - anndata + - python-igraph + - leidenalg + - scanpy # Apps extras - joblib @@ -65,8 +66,8 @@ test: - qbiocode.learning - qbiocode.utils - qbiocode.visualization - - apps.qprofiler - - apps.sage + - qbiocode.apps.qprofiler + - qbiocode.apps.sage commands: - qprofiler --help - qsage --help diff --git a/docs/source/tutorials/QProfiler/configs/config.yaml b/docs/source/tutorials/QProfiler/configs/config.yaml index a12a48fa..ce7ab7fd 100644 --- a/docs/source/tutorials/QProfiler/configs/config.yaml +++ b/docs/source/tutorials/QProfiler/configs/config.yaml @@ -4,7 +4,7 @@ config_file_name: 'basic_config' # specify output directory where input datasets are located -folder_path: 'tutorial/QProfiler/data/ld_data' +folder_path: 'data/ld_data' file_dataset: 'ALL' # or use a list as below, to only select a few datasets # file_dataset: ['file1', 'file2', 'file3', etc] diff --git a/docs/source/tutorials/QProfiler/sc_binary_qprofiler.ipynb b/docs/source/tutorials/QProfiler/sc_binary_qprofiler.ipynb index 0296d07f..dd92f79e 100644 --- a/docs/source/tutorials/QProfiler/sc_binary_qprofiler.ipynb +++ b/docs/source/tutorials/QProfiler/sc_binary_qprofiler.ipynb @@ -1312,9 +1312,9 @@ ], "metadata": { "kernelspec": { - "display_name": "Python (qbc-pkg)", + "display_name": "Python 3", "language": "python", - "name": "qbc-pkg" + "name": "python3" }, "language_info": { "codemirror_mode": { diff --git a/docs/source/tutorials/Quantum_Projection_Learning/QPL_example.ipynb b/docs/source/tutorials/Quantum_Projection_Learning/QPL_example.ipynb index a2cbaa16..8f13c71a 100644 --- a/docs/source/tutorials/Quantum_Projection_Learning/QPL_example.ipynb +++ b/docs/source/tutorials/Quantum_Projection_Learning/QPL_example.ipynb @@ -70,7 +70,7 @@ "\n", "# Import QBioCode\n", "import qbiocode as qbc\n", - "import qprofiler.qprofiler as profiler\n", + "from qbiocode.apps.qprofiler import qprofiler as profiler\n", "\n", "# Set plotting style\n", "sns.set_style('whitegrid')\n", diff --git a/docs/source/tutorials/Quantum_Projection_Learning/configs/qpl.yaml b/docs/source/tutorials/Quantum_Projection_Learning/configs/qpl.yaml index 7113850d..b0570e68 100644 --- a/docs/source/tutorials/Quantum_Projection_Learning/configs/qpl.yaml +++ b/docs/source/tutorials/Quantum_Projection_Learning/configs/qpl.yaml @@ -4,7 +4,7 @@ config_file_name: 'basic_config' # specify output directory where input datasets are located -folder_path: 'tutorial/QProfiler/data/ld_data' +folder_path: 'data/qpl_tutorial_data' file_dataset: 'ALL' # or use a list as below, to only select a few datasets # file_dataset: ['file1', 'file2', 'file3', etc] diff --git a/docs/source/tutorials/Quantum_Projection_Learning/configs/rf.yaml b/docs/source/tutorials/Quantum_Projection_Learning/configs/rf.yaml index 2e340f95..41d9cc92 100644 --- a/docs/source/tutorials/Quantum_Projection_Learning/configs/rf.yaml +++ b/docs/source/tutorials/Quantum_Projection_Learning/configs/rf.yaml @@ -4,7 +4,7 @@ config_file_name: 'basic_config' # specify output directory where input datasets are located -folder_path: 'tutorial/QProfiler/data/ld_data' +folder_path: 'data/qpl_tutorial_data' file_dataset: 'ALL' # or use a list as below, to only select a few datasets # file_dataset: ['file1', 'file2', 'file3', etc] diff --git a/docs/source/tutorials/Quantum_Projection_Learning/configs/xgb.yaml b/docs/source/tutorials/Quantum_Projection_Learning/configs/xgb.yaml index 0c62fa26..dadbb7d8 100644 --- a/docs/source/tutorials/Quantum_Projection_Learning/configs/xgb.yaml +++ b/docs/source/tutorials/Quantum_Projection_Learning/configs/xgb.yaml @@ -4,7 +4,7 @@ config_file_name: 'basic_config' # specify output directory where input datasets are located -folder_path: 'tutorial/QProfiler/data/ld_data' +folder_path: 'data/qpl_tutorial_data' file_dataset: 'ALL' # or use a list as below, to only select a few datasets # file_dataset: ['file1', 'file2', 'file3', etc] diff --git a/pyproject.toml b/pyproject.toml index bec8d006..bfc4f8c1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -127,7 +127,7 @@ exclude = ["tests*", "docs*", "archive*"] [tool.setuptools.dynamic] version = {attr = "qbiocode.version.__version__"} -dependencies = {file = ["requirements-base.txt"]} +dependencies = {file = ["requirements.txt"]} [tool.setuptools.package-data] qbiocode = ["py.typed"] diff --git a/qbiocode/apps/sage/sage.py b/qbiocode/apps/sage/sage.py index 61c951da..7abfc84f 100644 --- a/qbiocode/apps/sage/sage.py +++ b/qbiocode/apps/sage/sage.py @@ -40,6 +40,14 @@ def __init__(self, data_input): self._columns_metrics = ['accuracy', 'f1_score', 'auc'] self._columns_metadata = ['Dataset', 'embeddings','datatype', 'model_embed_datatype', 'iteration', 'model', 'BestParams_GridSearch', 'Model_Parameters'] + # Grid-search and serialized model parameters are optional QProfiler + # outputs. Keep the QSage input schema stable when either feature was + # disabled in the producing QProfiler run. + data_input = data_input.copy() + for column in ('BestParams_GridSearch', 'Model_Parameters'): + if column not in data_input.columns: + data_input[column] = None + self._input_data_features_only = data_input[self._columns_data_features] self._input_data_metrics = data_input[self._columns_metrics] self._input_data_metadata = data_input[self._columns_metadata] diff --git a/qbiocode/embeddings/embed.py b/qbiocode/embeddings/embed.py index 3b6b087d..bf22cf8e 100644 --- a/qbiocode/embeddings/embed.py +++ b/qbiocode/embeddings/embed.py @@ -55,7 +55,7 @@ def pqk( if data_map: # This function ensures that all multiplicative factors of data features inside single qubit gates are 1.0 - def data_map_func(x: np.ndarray) -> float: + def data_map_func(x: np.ndarray): """ Define a function map from R^n to R. @@ -63,10 +63,12 @@ def data_map_func(x: np.ndarray) -> float: x: data Returns: - float: the mapped value + The mapped numeric value or symbolic Qiskit expression. """ coeff = x[0] / 2 if len(x) == 1 else reduce(lambda m, n: (m * n) / 2, x) - return float(coeff) + # Qiskit calls this function with ParameterExpression objects while + # constructing the feature map, so preserve symbolic expressions. + return coeff else: data_map_func = None diff --git a/requirements-base.txt b/requirements-base.txt deleted file mode 100644 index 05e39a91..00000000 --- a/requirements-base.txt +++ /dev/null @@ -1,30 +0,0 @@ -dill -h5py -hfda -hydra-core -ipykernel -networkx -numpy -optuna -pandas -qiskit==2.2.0 -qiskit-aer==0.17.0 -qiskit-algorithms==0.4.0 -qiskit-ibm-runtime==0.44.0 -qiskit-ibm-transpiler==0.11.0 -qiskit-machine-learning==0.9.0 -qiskit-nature==0.7.2 -scikit-dimension -scikit-learn -scipy -seaborn -# Single-cell / multi-omics tutorials (sc-qc preprocessing, QProfiler single-cell) -scanpy -anndata -leidenalg -igraph -tensorflow -torch -tqdm -umap-learn -xgboost diff --git a/requirements.txt b/requirements.txt index ef2f42d5..aff0413b 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,48 +1,34 @@ +# Core scientific and application stack dill -h5py hfda hydra-core ipykernel -networkx -numpy +joblib +matplotlib>=3.9 +numpy>=2,<2.3 optuna -pandas -qiskit==2.2.0 -qiskit-aer==0.17.0 -qiskit-algorithms==0.4.0 -qiskit-ibm-runtime==0.44.0 -qiskit-ibm-transpiler==0.11.0 -qiskit-machine-learning==0.9.0 -qiskit-nature==0.7.2 +pandas>=2,<3 +pyyaml +requests scikit-dimension -scikit-learn -scipy +scikit-learn>=1.6 +scipy>=1.14 seaborn -tensorflow torch -tqdm +tqdm>=4.67.1 umap-learn xgboost -# Single-cell / multi-omics tutorials (sc-qc preprocessing, QProfiler single-cell) -scanpy +# Shared Qiskit stack validated with FCC, tetrahedral, and QTF on Python 3.12 +qiskit==2.4.2 +qiskit-aer==0.17.2 +qiskit-algorithms==0.4.0 +qiskit-ibm-runtime==0.44.0 +qiskit-ibm-transpiler==0.18.0 +qiskit-machine-learning==0.9.1 + +# Single-cell and multi-omics tutorials anndata -leidenalg igraph -# Imported directly by tutorial notebooks (also available transitively; pinned here to be explicit) -matplotlib -pyyaml -requests - -# Documentation -sphinx -sphinx-autodoc-typehints -sphinx_rtd_theme -myst_parser -better_apidoc -nbsphinx -myst_nb -furo -pydata-sphinx-theme -pandoc -sphinx-design +leidenalg +scanpy diff --git a/setup.py b/setup.py index f07b13fb..c80c881f 100644 --- a/setup.py +++ b/setup.py @@ -1,189 +1,10 @@ -""" -QBioCode: Quantum Applications in Healthcare and Life Science Data +"""Compatibility entry point for legacy setuptools invocations. -A comprehensive suite of computational resources supporting quantum machine learning -applications for healthcare and life science (HCLS) data analysis. +All project metadata and dependencies are defined in ``pyproject.toml`` so +PEP 517 builds and legacy setuptools builds share one source of truth. """ -import os -from setuptools import setup, find_packages - -# Read version from version.py -def get_version(): - """Extract version from qbiocode/version.py""" - version_file = os.path.join('qbiocode', 'version.py') - with open(version_file, 'r') as f: - for line in f: - if line.startswith('__version__'): - return line.split('=')[1].strip().strip('"').strip("'") - return '0.0.1' - -def read_file(fname): - """Read file contents from the project root directory.""" - return open(os.path.join(os.path.dirname(__file__), fname), encoding='utf-8').read() - -def read_requirements(): - """ - Read and parse requirements from requirements.txt. - Separates main dependencies from documentation dependencies. - """ - with open('requirements.txt', 'r', encoding='utf-8') as f: - lines = f.readlines() - - requirements = [] - doc_requirements = [] - in_doc_section = False - - for line in lines: - line = line.strip() - # Skip empty lines and comments - if not line or line.startswith('#'): - if '# Documentation' in line: - in_doc_section = True - continue - - # Add to appropriate list - if in_doc_section: - doc_requirements.append(line) - else: - requirements.append(line) - - return requirements, doc_requirements - -# Get requirements -install_requires, docs_require = read_requirements() +from setuptools import setup -# Read long description from README -long_description = read_file('README.md') if os.path.exists('README.md') else '' -setup( - name="qbiocode", - version=get_version(), - - # Author information - author="Bryan Raubenolt, Aritra Bose, Kahn Rhrissorrakrai, Filippo Utro, Akhil Mohan, Daniel Blankenberg, Laxmi Parida", - maintainer="IBM Research", - - # Project description - description=( - "A comprehensive suite of computational resources for quantum computing " - "applications in healthcare and life science data analysis" - ), - long_description=long_description, - long_description_content_type='text/markdown', - - # Project URLs - url="https://github.com/IBM/qbiocode", - project_urls={ - "Documentation": "https://ibm.github.io/QBioCode/", - "Source Code": "https://github.com/IBM/qbiocode", - "Bug Tracker": "https://github.com/IBM/qbiocode/issues", - }, - - # License - license="Apache License 2.0", - - # Package discovery - packages=find_packages(exclude=['tests', 'tests.*', 'docs', 'docs.*', 'archive', 'archive.*']), - - # Include package data - include_package_data=True, - package_data={ - 'qbiocode': ['py.typed'], - 'qbiocode.apps.qprofiler': ['configs/*.yaml'], - }, - - # Dependencies - install_requires=install_requires, - extras_require={ - 'apps': [ - 'hydra-core', - 'joblib', - ], - 'docs': docs_require, - 'dev': docs_require + [ - 'pytest>=7.0', - 'pytest-cov>=4.0', - 'black>=23.0', - 'flake8>=6.0', - 'mypy>=1.0', - 'types-PyYAML', - ], - 'all': docs_require + [ - 'hydra-core', - 'joblib', - 'pytest>=7.0', - 'pytest-cov>=4.0', - 'black>=23.0', - 'flake8>=6.0', - 'mypy>=1.0', - 'types-PyYAML', - ], - }, - - # Python version requirement - python_requires='>=3.10,<3.13', - - # PyPI classifiers - classifiers=[ - # Development status - "Development Status :: 3 - Alpha", - - # Intended audience - "Intended Audience :: Science/Research", - "Intended Audience :: Healthcare Industry", - "Intended Audience :: Developers", - - # Topic - "Topic :: Scientific/Engineering :: Artificial Intelligence", - "Topic :: Scientific/Engineering :: Bio-Informatics", - "Topic :: Scientific/Engineering :: Physics", - "Topic :: Software Development :: Libraries :: Python Modules", - - # License - "License :: OSI Approved :: Apache Software License", - - # Programming language - "Programming Language :: Python :: 3", - "Programming Language :: Python :: 3.10", - "Programming Language :: Python :: 3.11", - "Programming Language :: Python :: 3.12", - "Programming Language :: Python :: 3 :: Only", - - # Operating system - "Operating System :: OS Independent", - - # Natural language - "Natural Language :: English", - ], - - # Keywords for PyPI search - keywords=[ - "quantum machine learning", - "quantum computing", - "qiskit", - "bioinformatics", - "healthcare", - "life sciences", - "omics", - "genomics", - "proteomics", - "metabolomics", - "data complexity", - "model profiling", - "model selection", - "oracle", - ], - - # Console scripts for command-line tools - entry_points={ - 'console_scripts': [ - 'qprofiler=qbiocode.apps.qprofiler.cli:main', - 'qprofiler-batch=qbiocode.apps.qprofiler.qprofiler_batchmode:main', - 'qsage=qbiocode.apps.sage.sage:main', - ], - }, - - # Zip safety - zip_safe=False, -) +setup() diff --git a/tutorial/QProfiler/configs/config.yaml b/tutorial/QProfiler/configs/config.yaml index b110be63..78686280 100644 --- a/tutorial/QProfiler/configs/config.yaml +++ b/tutorial/QProfiler/configs/config.yaml @@ -4,7 +4,7 @@ config_file_name: 'basic_config' # specify output directory where input datasets are located -folder_path: 'tutorial/QProfiler/data/ld_data' +folder_path: 'data/ld_data' file_dataset: 'ALL' # or use a list as below, to only select a few datasets # file_dataset: ['file1', 'file2', 'file3', etc] diff --git a/tutorial/QProfiler/sc_binary_qprofiler.ipynb b/tutorial/QProfiler/sc_binary_qprofiler.ipynb index 9086045a..7476bf2d 100644 --- a/tutorial/QProfiler/sc_binary_qprofiler.ipynb +++ b/tutorial/QProfiler/sc_binary_qprofiler.ipynb @@ -1166,9 +1166,9 @@ ], "metadata": { "kernelspec": { - "display_name": "Python (qbc-pkg)", + "display_name": "Python 3", "language": "python", - "name": "qbc-pkg" + "name": "python3" }, "language_info": { "codemirror_mode": { diff --git a/tutorial/Quantum_Projection_Learning/QPL_example.ipynb b/tutorial/Quantum_Projection_Learning/QPL_example.ipynb index a2cbaa16..8f13c71a 100644 --- a/tutorial/Quantum_Projection_Learning/QPL_example.ipynb +++ b/tutorial/Quantum_Projection_Learning/QPL_example.ipynb @@ -70,7 +70,7 @@ "\n", "# Import QBioCode\n", "import qbiocode as qbc\n", - "import qprofiler.qprofiler as profiler\n", + "from qbiocode.apps.qprofiler import qprofiler as profiler\n", "\n", "# Set plotting style\n", "sns.set_style('whitegrid')\n", diff --git a/tutorial/Quantum_Projection_Learning/configs/qpl.yaml b/tutorial/Quantum_Projection_Learning/configs/qpl.yaml index 7113850d..b0570e68 100644 --- a/tutorial/Quantum_Projection_Learning/configs/qpl.yaml +++ b/tutorial/Quantum_Projection_Learning/configs/qpl.yaml @@ -4,7 +4,7 @@ config_file_name: 'basic_config' # specify output directory where input datasets are located -folder_path: 'tutorial/QProfiler/data/ld_data' +folder_path: 'data/qpl_tutorial_data' file_dataset: 'ALL' # or use a list as below, to only select a few datasets # file_dataset: ['file1', 'file2', 'file3', etc] diff --git a/tutorial/Quantum_Projection_Learning/configs/rf.yaml b/tutorial/Quantum_Projection_Learning/configs/rf.yaml index 2e340f95..41d9cc92 100644 --- a/tutorial/Quantum_Projection_Learning/configs/rf.yaml +++ b/tutorial/Quantum_Projection_Learning/configs/rf.yaml @@ -4,7 +4,7 @@ config_file_name: 'basic_config' # specify output directory where input datasets are located -folder_path: 'tutorial/QProfiler/data/ld_data' +folder_path: 'data/qpl_tutorial_data' file_dataset: 'ALL' # or use a list as below, to only select a few datasets # file_dataset: ['file1', 'file2', 'file3', etc] diff --git a/tutorial/Quantum_Projection_Learning/configs/xgb.yaml b/tutorial/Quantum_Projection_Learning/configs/xgb.yaml index 0c62fa26..dadbb7d8 100644 --- a/tutorial/Quantum_Projection_Learning/configs/xgb.yaml +++ b/tutorial/Quantum_Projection_Learning/configs/xgb.yaml @@ -4,7 +4,7 @@ config_file_name: 'basic_config' # specify output directory where input datasets are located -folder_path: 'tutorial/QProfiler/data/ld_data' +folder_path: 'data/qpl_tutorial_data' file_dataset: 'ALL' # or use a list as below, to only select a few datasets # file_dataset: ['file1', 'file2', 'file3', etc] From f164eeffa43941893f14edb82f7679b91e664488 Mon Sep 17 00:00:00 2001 From: thepineapplepirate Date: Fri, 18 Sep 2026 16:52:33 -0400 Subject: [PATCH 2/2] Make QSage accept raw QProfiler output --- conda-recipe/meta.yaml | 1 + qbiocode/apps/sage/sage.py | 26 +++++- tests/test_sage_contract.py | 35 +++++--- tutorial/QSage/qsage.ipynb | 157 +----------------------------------- 4 files changed, 48 insertions(+), 171 deletions(-) diff --git a/conda-recipe/meta.yaml b/conda-recipe/meta.yaml index 731d8fcb..cbcaba38 100644 --- a/conda-recipe/meta.yaml +++ b/conda-recipe/meta.yaml @@ -28,6 +28,7 @@ requirements: - python >=3.10,<3.13 # Core dependencies - dill + - h5py - hfda - hydra-core - ipykernel diff --git a/qbiocode/apps/sage/sage.py b/qbiocode/apps/sage/sage.py index 0e18ca25..07688f77 100644 --- a/qbiocode/apps/sage/sage.py +++ b/qbiocode/apps/sage/sage.py @@ -46,6 +46,25 @@ def __init__(self, data_input): This function initializes the Sage with the input data frame that contains the data characteristics and performance metrics ''' + # QProfiler already writes the source fields needed to derive QSage's + # bookkeeping metadata. Normalize them here so a raw ModelResults.csv + # can be passed directly, while preserving caller-supplied values. + data_input = data_input.copy() + if 'datatype' not in data_input.columns and 'Dataset' in data_input.columns: + data_input['datatype'] = data_input['Dataset'] + if 'model_embed_datatype' not in data_input.columns: + required = {'model', 'embeddings', 'datatype'} + if required <= set(data_input.columns): + data_input['model_embed_datatype'] = ( + data_input['model'].astype(str) + + '_' + + data_input['embeddings'].fillna('none').astype(str) + + '_' + + data_input['datatype'].astype(str) + ) + if 'iteration' not in data_input.columns: + data_input['iteration'] = 1 + # Detected rather than hardcoded: the complexity block QProfiler writes is # now pyMFE-backed, but the committed benchmark table predates that and # cannot be regenerated from this repository. Reading whichever schema the @@ -88,15 +107,14 @@ def __init__(self, data_input): raise ValueError( f"data_input is missing {len(missing)} required column(s): {missing}. " "QSage trains on a QProfiler results table (ModelResults.csv) with the " - "metadata columns the QSage tutorial adds -- 'datatype', " - "'model_embed_datatype' and 'iteration'. See " - "tutorial/QSage/qsage.ipynb for the exact preparation step." + "metadata columns derived from a QProfiler table -- 'datatype', " + "'model_embed_datatype' and 'iteration' -- could not be created. " + "Check that Dataset, model, and embeddings are present." ) # Grid-search and serialized model parameters are optional QProfiler # outputs. Keep the QSage input schema stable when either feature was # disabled in the producing QProfiler run. - data_input = data_input.copy() for column in ('BestParams_GridSearch', 'Model_Parameters'): if column not in data_input.columns: data_input[column] = None diff --git a/tests/test_sage_contract.py b/tests/test_sage_contract.py index f44663e6..ad276627 100644 --- a/tests/test_sage_contract.py +++ b/tests/test_sage_contract.py @@ -93,7 +93,10 @@ MODELS = ["rf", "svc"] -def results_table(parameter_column="Model_Parameters", n_datasets=6, schema="legacy"): +def results_table( + parameter_column="Model_Parameters", n_datasets=6, schema="legacy", *, + include_derived_metadata=True, +): """A QProfiler-shaped results table with one parameter column, as QProfiler writes.""" rng = np.random.default_rng(0) rows = [] @@ -115,10 +118,11 @@ def results_table(parameter_column="Model_Parameters", n_datasets=6, schema="leg row[parameter_column] = "{}" rows.append(row) frame = pd.DataFrame(rows) - frame["datatype"] = frame["Dataset"] - frame["model_embed_datatype"] = ( - frame["model"] + "_" + frame["embeddings"] + "_" + frame["datatype"] - ) + if include_derived_metadata: + frame["datatype"] = frame["Dataset"] + frame["model_embed_datatype"] = ( + frame["model"] + "_" + frame["embeddings"] + "_" + frame["datatype"] + ) return frame @@ -138,6 +142,18 @@ def test_the_pymfe_schema_is_recognized(self): assert set(PYMFE_FEATURES) <= set(sage._columns_data_features) assert any(c.startswith("mfe.") for c in sage._columns_data_features) + @pytest.mark.parametrize("schema", sorted(SCHEMAS)) + def test_raw_qprofiler_output_gets_derived_metadata(self, schema): + """QSage accepts ModelResults.csv without notebook-only preparation.""" + frame = results_table(schema=schema, include_derived_metadata=False) + sage = _sage.QuantumSage(data_input=frame) + assert sage._input_data_metadata["datatype"].equals(frame["Dataset"]) + expected = ( + frame["model"] + "_" + frame["embeddings"] + "_" + frame["Dataset"] + ) + assert sage._input_data_metadata["model_embed_datatype"].equals(expected) + assert sage._input_data_metadata["iteration"].eq(1).all() + def test_the_committed_benchmark_table_is_still_trainable(self): """The table the QSage tutorial trains on is legacy-schema and unregenerable. @@ -192,13 +208,10 @@ def test_it_accepts_a_table_recording_no_parameters_at_all(schema): @pytest.mark.parametrize("schema", sorted(SCHEMAS)) -def test_a_genuinely_missing_metadata_column_is_named_with_what_to_do(schema): +def test_missing_iteration_defaults_to_first_iteration(schema): frame = results_table(schema=schema).drop(columns=["iteration"]) - with pytest.raises(ValueError) as failure: - _sage.QuantumSage(data_input=frame) - message = str(failure.value) - assert "'iteration'" in message - assert "ModelResults.csv" in message + sage = _sage.QuantumSage(data_input=frame) + assert sage._input_data_metadata["iteration"].eq(1).all() @pytest.mark.parametrize("schema", sorted(SCHEMAS)) diff --git a/tutorial/QSage/qsage.ipynb b/tutorial/QSage/qsage.ipynb index 4ecd14d9..d8c98798 100644 --- a/tutorial/QSage/qsage.ipynb +++ b/tutorial/QSage/qsage.ipynb @@ -360,161 +360,6 @@ "results_df.head()" ] }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 2b. Prepare QProfiler Output for QSage\n", - "\n", - "QSage expects a few additional metadata columns that are not directly output by QProfiler. The cell below adds them automatically:\n", - "\n", - "- **`datatype`** — the file/dataset name (derived from `Dataset`)\n", - "- **`model_embed_datatype`** — a combined identifier string in the format `model_embedding_datatype`\n", - "- **`iteration`** — trial/run index (set to 1 if not present)" - ] - }, - { - "cell_type": "code", - "execution_count": 3, - "metadata": { - "execution": { - "iopub.execute_input": "2026-08-31T17:38:22.331869Z", - "iopub.status.busy": "2026-08-31T17:38:22.331791Z", - "iopub.status.idle": "2026-08-31T17:38:22.336682Z", - "shell.execute_reply": "2026-08-31T17:38:22.336247Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "✓ QProfiler output prepared for QSage\n", - "Added columns: datatype, model_embed_datatype, iteration\n" - ] - }, - { - "data": { - "text/html": [ - "
\n", - "\n", - "\n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - "
Datasetembeddingsmodeldatatypemodel_embed_datatypeiteration
0class_data-1.csvpcadtclass_data-1.csvdt_pca_class_data-1.csv1
1class_data-1.csvpcalrclass_data-1.csvlr_pca_class_data-1.csv1
2class_data-1.csvpcamlpclass_data-1.csvmlp_pca_class_data-1.csv1
3class_data-1.csvpcanbclass_data-1.csvnb_pca_class_data-1.csv1
4class_data-1.csvpcarfclass_data-1.csvrf_pca_class_data-1.csv1
\n", - "
" - ], - "text/plain": [ - " Dataset embeddings model datatype \\\n", - "0 class_data-1.csv pca dt class_data-1.csv \n", - "1 class_data-1.csv pca lr class_data-1.csv \n", - "2 class_data-1.csv pca mlp class_data-1.csv \n", - "3 class_data-1.csv pca nb class_data-1.csv \n", - "4 class_data-1.csv pca rf class_data-1.csv \n", - "\n", - " model_embed_datatype iteration \n", - "0 dt_pca_class_data-1.csv 1 \n", - "1 lr_pca_class_data-1.csv 1 \n", - "2 mlp_pca_class_data-1.csv 1 \n", - "3 nb_pca_class_data-1.csv 1 \n", - "4 rf_pca_class_data-1.csv 1 " - ] - }, - "execution_count": 3, - "metadata": {}, - "output_type": "execute_result" - } - ], - "source": [ - "# Add columns required by QSage that are not directly in QProfiler output\n", - "\n", - "# 'datatype': use the Dataset column as-is\n", - "results_df['datatype'] = results_df['Dataset']\n", - "\n", - "# 'model_embed_datatype': combined identifier used internally by QSage\n", - "results_df['model_embed_datatype'] = (\n", - " results_df['model'] + '_' +\n", - " results_df['embeddings'] + '_' +\n", - " results_df['datatype']\n", - ")\n", - "\n", - "# 'iteration': use existing column if present, otherwise default to 1\n", - "if 'iteration' not in results_df.columns:\n", - " results_df['iteration'] = 1\n", - "\n", - "print(\"✓ QProfiler output prepared for QSage\")\n", - "print(f\"Added columns: datatype, model_embed_datatype, iteration\")\n", - "results_df[['Dataset', 'embeddings', 'model', 'datatype', 'model_embed_datatype', 'iteration']].head()" - ] - }, { "cell_type": "markdown", "metadata": {}, @@ -1134,7 +979,7 @@ "In this tutorial, you learned how to:\n", "\n", "1. ✅ Understand what input data QSage requires and why it must span many datasets\n", - "2. ✅ Prepare a QProfiler results table for QSage (add the required metadata columns)\n", + "2. ✅ Load a QProfiler results table directly; QSage derives its bookkeeping metadata\n", "3. ✅ Initialize QSage with `QuantumSage(data_input=df)`\n", "4. ✅ Train sub-sages with `train_sub_sages()`\n", "5. ✅ Inspect how well each sub-sage fits with `plot_results()`\n",