diff --git a/MANIFEST.in b/MANIFEST.in index 8478b695..e0a699dd 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -64,4 +64,4 @@ prune docs/source/tutorials # Exclude test and development files prune tests prune .pytest_cache -prune archive \ No newline at end of file +prune archive diff --git a/README.md b/README.md index 62652bf9..193ff981 100644 --- a/README.md +++ b/README.md @@ -30,10 +30,11 @@ QBioCode requires Python **3.10 or higher** and has been tested with Python vers #### Install from PyPI (Recommended) ```bash -# Install the latest stable version +# Standard installation: QBioCode's library, applications, quantum, +# machine-learning, visualization, and tutorial runtime dependencies pip install qbiocode -# Install with apps support (QProfiler, QSage) +# Backward-compatible alias; QProfiler and QSage are already included above pip install 'qbiocode[apps]' # Install with QuVINE graph embeddings (quvine_rwr, quvine_dtqw, node2vec, ...) @@ -106,14 +107,20 @@ source .env/bin/activate # On Windows: .env\Scripts\activate # Install QBioCode in editable mode pip install -e . -# Install with apps support (QProfiler, QSage) +# Backward-compatible alias; applications are part of the standard install pip install -e '.[apps]' # Install with QuVINE graph embeddings pip install -e '.[quvine]' ``` -**macOS Users:** XGBoost requires OpenMP. Install it using Homebrew: +**macOS Users:** XGBoost requires OpenMP. In a Conda or Miniforge +environment, install the cross-platform runtime from conda-forge: +```bash +conda install -c conda-forge llvm-openmp +``` + +Alternatively, Homebrew users can install it with: ```bash brew install libomp pip install --force-reinstall xgboost diff --git a/conda-recipe/meta.yaml b/conda-recipe/meta.yaml index 55a90aa2..cbcaba38 100644 --- a/conda-recipe/meta.yaml +++ b/conda-recipe/meta.yaml @@ -35,25 +35,24 @@ requirements: - joblib - matplotlib-base - networkx - - numpy + - numpy >=2,<2.3 - optuna - - pandas + - pandas >=2,<3 - pyyaml # Qiskit ecosystem - - qiskit ==2.2.0 - - qiskit-aer ==0.17.0 + - qiskit ==2.4.2 + - qiskit-aer ==0.17.2 - qiskit-algorithms ==0.4.0 - qiskit-ibm-runtime ==0.44.0 - - qiskit-ibm-transpiler ==0.11.0 - - qiskit-machine-learning ==0.9.0 - - qiskit-nature ==0.7.2 + - qiskit-ibm-transpiler ==0.18.0 + - qiskit-machine-learning ==0.9.1 # ML and data science - scikit-dimension - - scikit-learn - - scipy + - scikit-learn >=1.6 + - scipy >=1.14 - seaborn - pytorch - - tqdm + - tqdm >=4.67.1 - umap-learn - xgboost # catboost is a core dependency alongside xgboost. tabpfn is deliberately diff --git a/qbiocode/apps/sage/sage.py b/qbiocode/apps/sage/sage.py index fe737501..07688f77 100644 --- a/qbiocode/apps/sage/sage.py +++ b/qbiocode/apps/sage/sage.py @@ -46,6 +46,25 @@ def __init__(self, data_input): This function initializes the Sage with the input data frame that contains the data characteristics and performance metrics ''' + # QProfiler already writes the source fields needed to derive QSage's + # bookkeeping metadata. Normalize them here so a raw ModelResults.csv + # can be passed directly, while preserving caller-supplied values. + data_input = data_input.copy() + if 'datatype' not in data_input.columns and 'Dataset' in data_input.columns: + data_input['datatype'] = data_input['Dataset'] + if 'model_embed_datatype' not in data_input.columns: + required = {'model', 'embeddings', 'datatype'} + if required <= set(data_input.columns): + data_input['model_embed_datatype'] = ( + data_input['model'].astype(str) + + '_' + + data_input['embeddings'].fillna('none').astype(str) + + '_' + + data_input['datatype'].astype(str) + ) + if 'iteration' not in data_input.columns: + data_input['iteration'] = 1 + # Detected rather than hardcoded: the complexity block QProfiler writes is # now pyMFE-backed, but the committed benchmark table predates that and # cannot be regenerated from this repository. Reading whichever schema the @@ -88,11 +107,18 @@ def __init__(self, data_input): raise ValueError( f"data_input is missing {len(missing)} required column(s): {missing}. " "QSage trains on a QProfiler results table (ModelResults.csv) with the " - "metadata columns the QSage tutorial adds -- 'datatype', " - "'model_embed_datatype' and 'iteration'. See " - "tutorial/QSage/qsage.ipynb for the exact preparation step." + "metadata columns derived from a QProfiler table -- 'datatype', " + "'model_embed_datatype' and 'iteration' -- could not be created. " + "Check that Dataset, model, and embeddings are present." ) + # Grid-search and serialized model parameters are optional QProfiler + # outputs. Keep the QSage input schema stable when either feature was + # disabled in the producing QProfiler run. + for column in ('BestParams_GridSearch', 'Model_Parameters'): + if column not in data_input.columns: + data_input[column] = None + self._input_data_features_only = data_input[self._columns_data_features] self._input_data_metrics = data_input[self._columns_metrics] self._input_data_metadata = data_input[self._columns_metadata] diff --git a/requirements/requirements-base.txt b/requirements/requirements-base.txt index 57d38203..8a095c67 100644 --- a/requirements/requirements-base.txt +++ b/requirements/requirements-base.txt @@ -33,29 +33,28 @@ hfda hydra-core ipykernel joblib -matplotlib +matplotlib>=3.9 networkx -numpy +numpy>=2,<2.3 optuna -pandas +pandas>=2,<3 # pyMFE supplies the dataset-complexity meta-features used by # qbiocode.evaluation.evaluate (see qbiocode/evaluation/mfe_features.py). # It pulls in `gower`; every other dependency of it is already declared here. pymfe pyyaml -qiskit==2.2.0 -qiskit-aer==0.17.0 +qiskit==2.4.2 +qiskit-aer==0.17.2 qiskit-algorithms==0.4.0 qiskit-ibm-runtime==0.44.0 -qiskit-ibm-transpiler==0.11.0 -qiskit-machine-learning==0.9.0 -qiskit-nature==0.7.2 +qiskit-ibm-transpiler==0.18.0 +qiskit-machine-learning==0.9.1 scikit-dimension -scikit-learn -scipy +scikit-learn>=1.6 +scipy>=1.14 seaborn torch -tqdm +tqdm>=4.67.1 umap-learn xgboost diff --git a/tests/test_sage_contract.py b/tests/test_sage_contract.py index f44663e6..ad276627 100644 --- a/tests/test_sage_contract.py +++ b/tests/test_sage_contract.py @@ -93,7 +93,10 @@ MODELS = ["rf", "svc"] -def results_table(parameter_column="Model_Parameters", n_datasets=6, schema="legacy"): +def results_table( + parameter_column="Model_Parameters", n_datasets=6, schema="legacy", *, + include_derived_metadata=True, +): """A QProfiler-shaped results table with one parameter column, as QProfiler writes.""" rng = np.random.default_rng(0) rows = [] @@ -115,10 +118,11 @@ def results_table(parameter_column="Model_Parameters", n_datasets=6, schema="leg row[parameter_column] = "{}" rows.append(row) frame = pd.DataFrame(rows) - frame["datatype"] = frame["Dataset"] - frame["model_embed_datatype"] = ( - frame["model"] + "_" + frame["embeddings"] + "_" + frame["datatype"] - ) + if include_derived_metadata: + frame["datatype"] = frame["Dataset"] + frame["model_embed_datatype"] = ( + frame["model"] + "_" + frame["embeddings"] + "_" + frame["datatype"] + ) return frame @@ -138,6 +142,18 @@ def test_the_pymfe_schema_is_recognized(self): assert set(PYMFE_FEATURES) <= set(sage._columns_data_features) assert any(c.startswith("mfe.") for c in sage._columns_data_features) + @pytest.mark.parametrize("schema", sorted(SCHEMAS)) + def test_raw_qprofiler_output_gets_derived_metadata(self, schema): + """QSage accepts ModelResults.csv without notebook-only preparation.""" + frame = results_table(schema=schema, include_derived_metadata=False) + sage = _sage.QuantumSage(data_input=frame) + assert sage._input_data_metadata["datatype"].equals(frame["Dataset"]) + expected = ( + frame["model"] + "_" + frame["embeddings"] + "_" + frame["Dataset"] + ) + assert sage._input_data_metadata["model_embed_datatype"].equals(expected) + assert sage._input_data_metadata["iteration"].eq(1).all() + def test_the_committed_benchmark_table_is_still_trainable(self): """The table the QSage tutorial trains on is legacy-schema and unregenerable. @@ -192,13 +208,10 @@ def test_it_accepts_a_table_recording_no_parameters_at_all(schema): @pytest.mark.parametrize("schema", sorted(SCHEMAS)) -def test_a_genuinely_missing_metadata_column_is_named_with_what_to_do(schema): +def test_missing_iteration_defaults_to_first_iteration(schema): frame = results_table(schema=schema).drop(columns=["iteration"]) - with pytest.raises(ValueError) as failure: - _sage.QuantumSage(data_input=frame) - message = str(failure.value) - assert "'iteration'" in message - assert "ModelResults.csv" in message + sage = _sage.QuantumSage(data_input=frame) + assert sage._input_data_metadata["iteration"].eq(1).all() @pytest.mark.parametrize("schema", sorted(SCHEMAS)) diff --git a/tutorial/QSage/qsage.ipynb b/tutorial/QSage/qsage.ipynb index 4ecd14d9..d8c98798 100644 --- a/tutorial/QSage/qsage.ipynb +++ b/tutorial/QSage/qsage.ipynb @@ -360,161 +360,6 @@ "results_df.head()" ] }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## 2b. Prepare QProfiler Output for QSage\n", - "\n", - "QSage expects a few additional metadata columns that are not directly output by QProfiler. The cell below adds them automatically:\n", - "\n", - "- **`datatype`** — the file/dataset name (derived from `Dataset`)\n", - "- **`model_embed_datatype`** — a combined identifier string in the format `model_embedding_datatype`\n", - "- **`iteration`** — trial/run index (set to 1 if not present)" - ] - }, - { - "cell_type": "code", - "execution_count": 3, - "metadata": { - "execution": { - "iopub.execute_input": "2026-08-31T17:38:22.331869Z", - "iopub.status.busy": "2026-08-31T17:38:22.331791Z", - "iopub.status.idle": "2026-08-31T17:38:22.336682Z", - "shell.execute_reply": "2026-08-31T17:38:22.336247Z" - } - }, - "outputs": [ - { - "name": "stdout", - "output_type": "stream", - "text": [ - "✓ QProfiler output prepared for QSage\n", - "Added columns: datatype, model_embed_datatype, iteration\n" - ] - }, - { - "data": { - "text/html": [ - "
\n", - "\n", - "\n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - "
Datasetembeddingsmodeldatatypemodel_embed_datatypeiteration
0class_data-1.csvpcadtclass_data-1.csvdt_pca_class_data-1.csv1
1class_data-1.csvpcalrclass_data-1.csvlr_pca_class_data-1.csv1
2class_data-1.csvpcamlpclass_data-1.csvmlp_pca_class_data-1.csv1
3class_data-1.csvpcanbclass_data-1.csvnb_pca_class_data-1.csv1
4class_data-1.csvpcarfclass_data-1.csvrf_pca_class_data-1.csv1
\n", - "
" - ], - "text/plain": [ - " Dataset embeddings model datatype \\\n", - "0 class_data-1.csv pca dt class_data-1.csv \n", - "1 class_data-1.csv pca lr class_data-1.csv \n", - "2 class_data-1.csv pca mlp class_data-1.csv \n", - "3 class_data-1.csv pca nb class_data-1.csv \n", - "4 class_data-1.csv pca rf class_data-1.csv \n", - "\n", - " model_embed_datatype iteration \n", - "0 dt_pca_class_data-1.csv 1 \n", - "1 lr_pca_class_data-1.csv 1 \n", - "2 mlp_pca_class_data-1.csv 1 \n", - "3 nb_pca_class_data-1.csv 1 \n", - "4 rf_pca_class_data-1.csv 1 " - ] - }, - "execution_count": 3, - "metadata": {}, - "output_type": "execute_result" - } - ], - "source": [ - "# Add columns required by QSage that are not directly in QProfiler output\n", - "\n", - "# 'datatype': use the Dataset column as-is\n", - "results_df['datatype'] = results_df['Dataset']\n", - "\n", - "# 'model_embed_datatype': combined identifier used internally by QSage\n", - "results_df['model_embed_datatype'] = (\n", - " results_df['model'] + '_' +\n", - " results_df['embeddings'] + '_' +\n", - " results_df['datatype']\n", - ")\n", - "\n", - "# 'iteration': use existing column if present, otherwise default to 1\n", - "if 'iteration' not in results_df.columns:\n", - " results_df['iteration'] = 1\n", - "\n", - "print(\"✓ QProfiler output prepared for QSage\")\n", - "print(f\"Added columns: datatype, model_embed_datatype, iteration\")\n", - "results_df[['Dataset', 'embeddings', 'model', 'datatype', 'model_embed_datatype', 'iteration']].head()" - ] - }, { "cell_type": "markdown", "metadata": {}, @@ -1134,7 +979,7 @@ "In this tutorial, you learned how to:\n", "\n", "1. ✅ Understand what input data QSage requires and why it must span many datasets\n", - "2. ✅ Prepare a QProfiler results table for QSage (add the required metadata columns)\n", + "2. ✅ Load a QProfiler results table directly; QSage derives its bookkeeping metadata\n", "3. ✅ Initialize QSage with `QuantumSage(data_input=df)`\n", "4. ✅ Train sub-sages with `train_sub_sages()`\n", "5. ✅ Inspect how well each sub-sage fits with `plot_results()`\n",