Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion MANIFEST.in
Original file line number Diff line number Diff line change
Expand Up @@ -64,4 +64,4 @@ prune docs/source/tutorials
# Exclude test and development files
prune tests
prune .pytest_cache
prune archive
prune archive
15 changes: 11 additions & 4 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -30,10 +30,11 @@ QBioCode requires Python **3.10 or higher** and has been tested with Python vers
#### Install from PyPI (Recommended)

```bash
# Install the latest stable version
# Standard installation: QBioCode's library, applications, quantum,
# machine-learning, visualization, and tutorial runtime dependencies
pip install qbiocode

# Install with apps support (QProfiler, QSage)
# Backward-compatible alias; QProfiler and QSage are already included above
pip install 'qbiocode[apps]'

# Install with QuVINE graph embeddings (quvine_rwr, quvine_dtqw, node2vec, ...)
Expand Down Expand Up @@ -106,14 +107,20 @@ source .env/bin/activate # On Windows: .env\Scripts\activate
# Install QBioCode in editable mode
pip install -e .

# Install with apps support (QProfiler, QSage)
# Backward-compatible alias; applications are part of the standard install
pip install -e '.[apps]'

# Install with QuVINE graph embeddings
pip install -e '.[quvine]'
```

**macOS Users:** XGBoost requires OpenMP. Install it using Homebrew:
**macOS Users:** XGBoost requires OpenMP. In a Conda or Miniforge
environment, install the cross-platform runtime from conda-forge:
```bash
conda install -c conda-forge llvm-openmp
```

Alternatively, Homebrew users can install it with:
```bash
brew install libomp
pip install --force-reinstall xgboost
Expand Down
19 changes: 9 additions & 10 deletions conda-recipe/meta.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -35,25 +35,24 @@ requirements:
- joblib
- matplotlib-base
- networkx
- numpy
- numpy >=2,<2.3
- optuna
- pandas
- pandas >=2,<3
- pyyaml
# Qiskit ecosystem
- qiskit ==2.2.0
- qiskit-aer ==0.17.0
- qiskit ==2.4.2
- qiskit-aer ==0.17.2
- qiskit-algorithms ==0.4.0
- qiskit-ibm-runtime ==0.44.0
- qiskit-ibm-transpiler ==0.11.0
- qiskit-machine-learning ==0.9.0
- qiskit-nature ==0.7.2
- qiskit-ibm-transpiler ==0.18.0
- qiskit-machine-learning ==0.9.1
# ML and data science
- scikit-dimension
- scikit-learn
- scipy
- scikit-learn >=1.6
- scipy >=1.14
- seaborn
- pytorch
- tqdm
- tqdm >=4.67.1
- umap-learn
- xgboost
# catboost is a core dependency alongside xgboost. tabpfn is deliberately
Expand Down
32 changes: 29 additions & 3 deletions qbiocode/apps/sage/sage.py
Original file line number Diff line number Diff line change
Expand Up @@ -46,6 +46,25 @@ def __init__(self, data_input):
This function initializes the Sage with the input data frame that contains the data characteristics and performance metrics
'''

# QProfiler already writes the source fields needed to derive QSage's
# bookkeeping metadata. Normalize them here so a raw ModelResults.csv
# can be passed directly, while preserving caller-supplied values.
data_input = data_input.copy()
if 'datatype' not in data_input.columns and 'Dataset' in data_input.columns:
data_input['datatype'] = data_input['Dataset']
if 'model_embed_datatype' not in data_input.columns:
required = {'model', 'embeddings', 'datatype'}
if required <= set(data_input.columns):
data_input['model_embed_datatype'] = (
data_input['model'].astype(str)
+ '_'
+ data_input['embeddings'].fillna('none').astype(str)
+ '_'
+ data_input['datatype'].astype(str)
)
if 'iteration' not in data_input.columns:
data_input['iteration'] = 1

# Detected rather than hardcoded: the complexity block QProfiler writes is
# now pyMFE-backed, but the committed benchmark table predates that and
# cannot be regenerated from this repository. Reading whichever schema the
Expand Down Expand Up @@ -88,11 +107,18 @@ def __init__(self, data_input):
raise ValueError(
f"data_input is missing {len(missing)} required column(s): {missing}. "
"QSage trains on a QProfiler results table (ModelResults.csv) with the "
"metadata columns the QSage tutorial adds -- 'datatype', "
"'model_embed_datatype' and 'iteration'. See "
"tutorial/QSage/qsage.ipynb for the exact preparation step."
"metadata columns derived from a QProfiler table -- 'datatype', "
"'model_embed_datatype' and 'iteration' -- could not be created. "
"Check that Dataset, model, and embeddings are present."
)

# Grid-search and serialized model parameters are optional QProfiler
# outputs. Keep the QSage input schema stable when either feature was
# disabled in the producing QProfiler run.
for column in ('BestParams_GridSearch', 'Model_Parameters'):
if column not in data_input.columns:
data_input[column] = None

self._input_data_features_only = data_input[self._columns_data_features]
self._input_data_metrics = data_input[self._columns_metrics]
self._input_data_metadata = data_input[self._columns_metadata]
Expand Down
21 changes: 10 additions & 11 deletions requirements/requirements-base.txt
Original file line number Diff line number Diff line change
Expand Up @@ -33,29 +33,28 @@ hfda
hydra-core
ipykernel
joblib
matplotlib
matplotlib>=3.9
networkx
numpy
numpy>=2,<2.3
optuna
pandas
pandas>=2,<3
# pyMFE supplies the dataset-complexity meta-features used by
# qbiocode.evaluation.evaluate (see qbiocode/evaluation/mfe_features.py).
# It pulls in `gower`; every other dependency of it is already declared here.
pymfe
pyyaml
qiskit==2.2.0
qiskit-aer==0.17.0
qiskit==2.4.2
qiskit-aer==0.17.2
qiskit-algorithms==0.4.0
qiskit-ibm-runtime==0.44.0
qiskit-ibm-transpiler==0.11.0
qiskit-machine-learning==0.9.0
qiskit-nature==0.7.2
qiskit-ibm-transpiler==0.18.0
qiskit-machine-learning==0.9.1
scikit-dimension
scikit-learn
scipy
scikit-learn>=1.6
scipy>=1.14
seaborn
torch
tqdm
tqdm>=4.67.1
umap-learn
xgboost

Expand Down
35 changes: 24 additions & 11 deletions tests/test_sage_contract.py
Original file line number Diff line number Diff line change
Expand Up @@ -93,7 +93,10 @@
MODELS = ["rf", "svc"]


def results_table(parameter_column="Model_Parameters", n_datasets=6, schema="legacy"):
def results_table(
parameter_column="Model_Parameters", n_datasets=6, schema="legacy", *,
include_derived_metadata=True,
):
"""A QProfiler-shaped results table with one parameter column, as QProfiler writes."""
rng = np.random.default_rng(0)
rows = []
Expand All @@ -115,10 +118,11 @@ def results_table(parameter_column="Model_Parameters", n_datasets=6, schema="leg
row[parameter_column] = "{}"
rows.append(row)
frame = pd.DataFrame(rows)
frame["datatype"] = frame["Dataset"]
frame["model_embed_datatype"] = (
frame["model"] + "_" + frame["embeddings"] + "_" + frame["datatype"]
)
if include_derived_metadata:
frame["datatype"] = frame["Dataset"]
frame["model_embed_datatype"] = (
frame["model"] + "_" + frame["embeddings"] + "_" + frame["datatype"]
)
return frame


Expand All @@ -138,6 +142,18 @@ def test_the_pymfe_schema_is_recognized(self):
assert set(PYMFE_FEATURES) <= set(sage._columns_data_features)
assert any(c.startswith("mfe.") for c in sage._columns_data_features)

@pytest.mark.parametrize("schema", sorted(SCHEMAS))
def test_raw_qprofiler_output_gets_derived_metadata(self, schema):
"""QSage accepts ModelResults.csv without notebook-only preparation."""
frame = results_table(schema=schema, include_derived_metadata=False)
sage = _sage.QuantumSage(data_input=frame)
assert sage._input_data_metadata["datatype"].equals(frame["Dataset"])
expected = (
frame["model"] + "_" + frame["embeddings"] + "_" + frame["Dataset"]
)
assert sage._input_data_metadata["model_embed_datatype"].equals(expected)
assert sage._input_data_metadata["iteration"].eq(1).all()

def test_the_committed_benchmark_table_is_still_trainable(self):
"""The table the QSage tutorial trains on is legacy-schema and unregenerable.

Expand Down Expand Up @@ -192,13 +208,10 @@ def test_it_accepts_a_table_recording_no_parameters_at_all(schema):


@pytest.mark.parametrize("schema", sorted(SCHEMAS))
def test_a_genuinely_missing_metadata_column_is_named_with_what_to_do(schema):
def test_missing_iteration_defaults_to_first_iteration(schema):
frame = results_table(schema=schema).drop(columns=["iteration"])
with pytest.raises(ValueError) as failure:
_sage.QuantumSage(data_input=frame)
message = str(failure.value)
assert "'iteration'" in message
assert "ModelResults.csv" in message
sage = _sage.QuantumSage(data_input=frame)
assert sage._input_data_metadata["iteration"].eq(1).all()


@pytest.mark.parametrize("schema", sorted(SCHEMAS))
Expand Down
Loading
Loading