Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -67,3 +67,4 @@ Thumbs.db
!training/samples/*.csv
!training/samples/*.md
training/resolve_cache.json
training/profile_index.npz
7 changes: 5 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -80,8 +80,11 @@ is tagged by how it was derived**, so nothing reads as more certain than its sou
acrylamide, ethyl carbamate, furan, and more) and an OAV dosing-balance analysis
that flags the component about to overpower a blend (quantitative when threshold
tables are loaded).
- **Substitution search** — nearest-neighbor lookup over the labeled set for
reformulation and cost-down ("find me a molecule that behaves like this one").
- **Substitutes & structural neighbors** — two nearest-neighbor searches over the whole
molecule universe for reformulation and cost-down: **substitutes** rank by *taste + aroma
profile* match (a molecule that tastes and smells like the target — e.g. ethyl vanillin for
vanillin — regardless of structure), while **structural neighbors** rank by Tanimoto/Morgan
structure similarity (the look-alikes).
- **Flavor Studio** — one hub to pick any mix of everyday **flavors** (banana,
saffron, pumpkin, bubble gum…) *and* **notes** (citrus, floral…) → ranked food-safe
molecules + drop-in swaps. A flavor *is* a set of notes, so they live in one picker.
Expand Down
24 changes: 21 additions & 3 deletions mcp-server/server.py
Original file line number Diff line number Diff line change
Expand Up @@ -96,8 +96,15 @@ def _read_full(molecule: str) -> dict:
return d


def _find_substitutes(molecule: str, k: int) -> dict:
def _find_structural_neighbors(molecule: str, k: int) -> dict:
r = _post("/api/neighbors", {"smiles": molecule, "k": k})
if r.status_code != 200:
return {"error": f"structural-neighbor search failed (HTTP {r.status_code})"}
return _strip(r.json())


def _find_substitutes(molecule: str, k: int) -> dict:
r = _post("/api/substitutes", {"smiles": molecule, "k": k})
if r.status_code != 200:
return {"error": f"substitute search failed (HTTP {r.status_code})"}
return _strip(r.json())
Expand Down Expand Up @@ -253,12 +260,23 @@ def read_full(molecule: str) -> dict:

@mcp.tool()
def find_substitutes(molecule: str, k: int = 8) -> dict:
"""Find the k structurally-nearest molecules (drop-in swaps / reformulation candidates)
to the given molecule, each with its similarity and known tastes.
"""Find the k best SUBSTITUTES — molecules whose predicted taste+aroma PROFILE is closest to
the given molecule (cosine over the head scores). These are the drop-in swaps: a molecule that
tastes and smells like the target, regardless of structure (e.g. ethyl vanillin for vanillin).
Each with its profile_match, known tastes, and aromas.
"""
return _find_substitutes(molecule, k)


@mcp.tool()
def find_structural_neighbors(molecule: str, k: int = 8) -> dict:
"""Find the k STRUCTURAL neighbors — molecules most similar in structure (Tanimoto / Morgan
fingerprint) to the given molecule. Structural look-alikes (contrast find_substitutes, which
matches by taste+aroma profile). Each with its similarity and known tastes.
"""
return _find_structural_neighbors(molecule, k)


@mcp.tool()
def list_stereoisomers(molecule: str) -> dict:
"""List every stereoisomer (R/S centers and E/Z bonds) of a molecule, with any
Expand Down
11 changes: 10 additions & 1 deletion skills/flavormancer/scripts/flavormancer.py
Original file line number Diff line number Diff line change
Expand Up @@ -118,6 +118,12 @@ def cmd_flavor(a):


def cmd_substitutes(a):
# profile-based swaps: closest taste+aroma head-score match (e.g. ethyl vanillin for vanillin)
return _req("/api/substitutes", "POST", {"smiles": a.molecule, "k": a.k})


def cmd_structural_neighbors(a):
# structural look-alikes: Tanimoto / Morgan nearest neighbors
return _req("/api/neighbors", "POST", {"smiles": a.molecule, "k": a.k})


Expand Down Expand Up @@ -188,9 +194,12 @@ def mol(name):
mol("read").set_defaults(fn=cmd_read)
mol("read-full").set_defaults(fn=cmd_read_full)
mol("stereoisomers").set_defaults(fn=cmd_stereoisomers)
s = mol("substitutes")
s = mol("substitutes") # taste+aroma profile match (the drop-in swaps)
s.add_argument("-k", type=int, default=8)
s.set_defaults(fn=cmd_substitutes)
s = mol("structural-neighbors") # Tanimoto structural look-alikes
s.add_argument("-k", type=int, default=8)
s.set_defaults(fn=cmd_structural_neighbors)

s = sub.add_parser("formulate")
s.add_argument("ingredients", nargs="+")
Expand Down
4 changes: 2 additions & 2 deletions tests/test_substitute.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@ def test_substitute_rejects_bad_smiles():


def test_substitute_graceful_without_data(monkeypatch):
monkeypatch.setattr(predict, "_SUB_INDEX", ([], [], [], []))
monkeypatch.setattr(predict, "_SUB_INDEX", ([], [], [], [], None, []))
out = predict.substitute("CCO")
assert out["neighbors"] == []
assert "note" in out
Expand All @@ -26,7 +26,7 @@ def test_substitute_ranks_by_similarity(monkeypatch):
mols = ["CCO", "CCCO", "c1ccccc1"] # ethanol, propanol, benzene
fps = [predict._MORGAN.GetFingerprint(Chem.MolFromSmiles(s)) for s in mols]
canon = [_canon(s) for s in mols]
monkeypatch.setattr(predict, "_SUB_INDEX", (fps, canon, [[], [], []], [[], [], []]))
monkeypatch.setattr(predict, "_SUB_INDEX", (fps, canon, [[], [], []], [[], [], []], None, []))
out = predict.substitute("CCO", k=2)
neighbors = out["neighbors"]
# the query itself is excluded
Expand Down
18 changes: 18 additions & 0 deletions training/app.py
Original file line number Diff line number Diff line change
Expand Up @@ -300,6 +300,24 @@ def api_neighbors(q: Query):
return res


@app.post("/api/substitutes")
def api_substitutes(q: Query):
"""Profile-based substitutes: molecules whose predicted taste+aroma head scores line up
closest with the query — the taste/smell-alikes (vs /api/neighbors' structural look-alikes)."""
smi = _resolve(q.smiles)
if not smi:
return {"substitutes": []}
res = P.substitutes(smi, k=q.k)
for n in res.get("substitutes", []): # same enrichment as neighbors: structure + names + aroma + GRAS
n["svg"] = _svg(n["smiles"], 132, 96)
nm = _names(n["smiles"])
n["name"], n["iupac"] = nm[0], nm[1]
n["aroma"] = _aroma_tags_cheap(n["smiles"], n.pop("aromas", []))
_m = Chem.MolFromSmiles(n["smiles"])
n["gras"] = bool(_m is not None and Chem.MolToInchiKey(_m).split("-")[0] in P._GRAS)
return res


@app.post("/api/names")
def api_names(q: Query):
"""Common (PubChem Title) + IUPAC names for the queried molecule."""
Expand Down
66 changes: 66 additions & 0 deletions training/build_profile_index.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,66 @@
"""build_profile_index.py — precompute the neighbor / substitute reference index.

The profile-based substitutes rank molecules by cosine similarity over their predicted taste +
aroma head SCORES. Computing that 170-head inference over the whole ~8.8k-molecule universe at app
startup is slow (~3 min), so we precompute it here once and cache to profile_index.npz. predict.py
loads it instantly (rebuilding the cheap Morgan fingerprints from SMILES on load).

Output profile_index.npz:
smiles (N,) canonical SMILES, deduped by connectivity skeleton
taste_documented (N,) comma-separated documented tastes ("" if none)
aromas (N,) comma-separated confident aroma heads (score >= 0.5)
profiles (N,D) float32 head-score matrix: taste heads then aroma heads
dims (D,) column labels ("taste:sweet", "aroma:citrus", ...)

Usage: python build_profile_index.py
"""
import numpy as np
import pandas as pd
import predict as P
from rdkit import Chem

SRC = "master_enrichment.parquet"
OUT = "profile_index.npz"


def main():
m = pd.read_parquet(SRC)
taste_heads, aroma_heads = P._profile_heads()
smis, td, feats, seen = [], [], [], set()
for _, r in m.iterrows():
mol = Chem.MolFromSmiles(str(r["smiles"]))
if mol is None:
continue
skel = Chem.MolToInchiKey(mol).split("-")[0]
if skel in seen:
continue
seen.add(skel)
smis.append(Chem.MolToSmiles(mol))
td.append(str(r.get("taste_documented") or ""))
feats.append(P._feat(mol)[0])
X = np.vstack(feats)
print(f"{len(smis)} unique structures — running {len(taste_heads)}+{len(aroma_heads)} heads...", flush=True)
cols, aromas = [], [[] for _ in smis]
for t in taste_heads:
cols.append(P._CLASSIFIERS[t].predict_proba(X)[:, 1])
for a in aroma_heads:
col = P._AROMA_MODELS[a].predict_proba(X)[:, 1]
cols.append(col)
for i in range(len(smis)):
if col[i] >= 0.5:
aromas[i].append(a)
profiles = np.column_stack(cols).astype("float32")
dims = [f"taste:{t}" for t in taste_heads] + [f"aroma:{a}" for a in aroma_heads]
np.savez_compressed(
OUT,
smiles=np.array(smis, dtype=object),
taste_documented=np.array(td, dtype=object),
aromas=np.array([",".join(a) for a in aromas], dtype=object),
profiles=profiles,
dims=np.array(dims, dtype=object),
)
print(f"wrote {OUT}: {profiles.shape[0]} molecules x {profiles.shape[1]} heads")


if __name__ == "__main__":
main()
136 changes: 111 additions & 25 deletions training/predict.py
Original file line number Diff line number Diff line change
Expand Up @@ -239,6 +239,11 @@ def _load_rf(path):
# molecule is in our labeled set, we report the verified fact instead of a guess.
_KNOWN = {} # inchikey -> {taste: 1}
_MASTER = Path("taste_master.parquet")
# The neighbor / substitute reference set: the FULL molecule universe (every structure we know,
# ~8.8k) so structural neighbors and profile substitutes can surface ANY molecule — e.g. ethyl
# vanillin as the top vanillin substitute — not just the taste-labelled subset. Falls back to
# taste_master when the enrichment table hasn't been built yet.
_UNIVERSE = Path("master_enrichment.parquet")
if _MASTER.exists():
import pandas as pd
_m = pd.read_parquet(_MASTER)
Expand Down Expand Up @@ -1061,57 +1066,99 @@ def _taste_profile(out):
# tool (swap an expensive or supply-constrained ingredient for a close analogue,
# with its known tastes shown). This is the clean Track-A core; the product
# (Track B, #22) mirrors it as a pgvector ANN query over the same fingerprints.
_SUB_INDEX = None # lazily built: (fps, smiles, known_tastes, predicted_aromas)
# lazily built: (fps, smiles, known_tastes, predicted_aromas, profiles, profile_dims)
# profiles: an (N x D) float32 matrix of predicted head SCORES — taste heads then aroma heads —
# the "flavor profile" vector used for profile-based substitutes (vs the fingerprint fps used for
# structural neighbors). profile_dims labels the columns.
_SUB_INDEX = None
_SUB_LOCK = _threading.Lock() # guards the one-time index build against concurrent callers


def _profile_heads():
"""The ordered head list backing a flavor-profile vector: taste heads then aroma heads.
Same order is used at index-build and query time so the vectors line up."""
return sorted(_CLASSIFIERS), list(_AROMA_MODELS)


def _build_sub_index():
global _SUB_INDEX
import numpy as np
# Fast path: load the precomputed profile index (build_profile_index.py). The 170-head
# inference over ~8.8k molecules is slow (~3 min); the cache makes startup instant. We only
# rebuild the cheap Morgan fingerprints from SMILES on load.
cache = Path("profile_index.npz")
if cache.exists():
z = np.load(cache, allow_pickle=True)
smis = [str(s) for s in z["smiles"]]
tastes = [[t for t in str(s).split(",") if t.strip()] for s in z["taste_documented"]]
aromas = [[a for a in str(s).split(",") if a.strip()] for s in z["aromas"]]
fps = [_MORGAN.GetFingerprint(Chem.MolFromSmiles(s)) for s in smis]
_SUB_INDEX = (fps, smis, tastes, aromas, z["profiles"], list(z["dims"]))
return
fps, smis, tastes, feats = [], [], [], []
if _MASTER.exists():
import numpy as np
profiles, profile_dims = None, []
src = _UNIVERSE if _UNIVERSE.exists() else _MASTER
if src.exists():
import pandas as pd
m = pd.read_parquet(_MASTER)
m = pd.read_parquet(src)
basic = [t for t in ("sweet", "bitter", "umami", "sour", "salty") if t in m.columns]
has_documented = "taste_documented" in m.columns # enrichment: comma-separated string
seen_skel = set()
for _, r in m.iterrows():
mol = Chem.MolFromSmiles(str(r["smiles"]))
if mol is None:
continue
skel = Chem.MolToInchiKey(mol).split("-")[0]
if skel in seen_skel: # dedupe by connectivity so the universe doesn't repeat a molecule
continue
seen_skel.add(skel)
fps.append(_MORGAN.GetFingerprint(mol))
smis.append(Chem.MolToSmiles(mol))
tastes.append([t for t in basic if r.get(t) == 1])
if has_documented:
tastes.append([t for t in str(r.get("taste_documented") or "").split(",") if t.strip()])
else:
tastes.append([t for t in basic if r.get(t) == 1])
feats.append(_feat(mol)[0])
# predicted aroma descriptors per molecule (batched over the 16 heads) — so the palette
# match can score aroma as well as taste, once, at first use
# One-time batched inference over the whole labeled set. We keep BOTH:
# - the thresholded confident-aroma NAME list per molecule (for display / palette match)
# - the full head-SCORE matrix (taste heads + aroma heads) for profile-based substitutes.
# Kept at n_jobs=1: a single big predict_proba is already vectorized C; joblib fan-out here
# thrashed under concurrency. Reusing this index is what makes the endpoints fast.
aromas = [[] for _ in smis]
if _AROMA_MODELS and feats:
if feats:
X = np.vstack(feats)
# One-time batch over the whole labeled set (~8k rows x 24 heads). Kept at n_jobs=1:
# a single big predict_proba is already vectorized C, and joblib fan-out here just
# thrashed under concurrency. Reusing these aromas is what makes the endpoint fast.
for name, clf in _AROMA_MODELS.items():
col = clf.predict_proba(X)[:, 1]
taste_heads, aroma_heads = _profile_heads()
cols = [_CLASSIFIERS[t].predict_proba(X)[:, 1] for t in taste_heads]
for name in aroma_heads:
col = _AROMA_MODELS[name].predict_proba(X)[:, 1]
cols.append(col)
for i in range(len(smis)):
if col[i] >= 0.5:
aromas[i].append(name)
profile_dims = [f"taste:{t}" for t in taste_heads] + [f"aroma:{a}" for a in aroma_heads]
profiles = np.column_stack(cols).astype("float32") if cols else None
else:
aromas = []
_SUB_INDEX = (fps, smis, tastes, aromas)
_SUB_INDEX = (fps, smis, tastes, aromas, profiles, profile_dims)


def substitute(smiles: str, k: int = 8, min_similarity: float = 0.0) -> dict:
"""Nearest-neighbor substitution: the k labeled molecules most structurally
similar to the query (Tanimoto over Morgan fingerprints), each with its known
tastes. The reformulation / cost-down tool — swap an ingredient for a close
analogue. Returns {'neighbors': [...]} ranked by similarity (self excluded)."""
mol = Chem.MolFromSmiles(smiles)
if mol is None:
return {"error": f"unparseable SMILES: {smiles}"}
def _ensure_sub_index():
if _SUB_INDEX is None: # double-checked lock: a request during the startup build waits for
with _SUB_LOCK: # that one build instead of kicking off a second (which would thrash cores)
if _SUB_INDEX is None:
_build_sub_index()
fps, smis, tastes, _aromas = _SUB_INDEX


def structural_neighbors(smiles: str, k: int = 8, min_similarity: float = 0.0) -> dict:
"""STRUCTURAL neighbors: the k labeled molecules most structurally similar to the query
(Tanimoto over Morgan fingerprints), each with its known tastes. Structural look-alikes —
contrast with profile-based `substitutes` (taste/aroma-alikes). Returns {'neighbors': [...]}
ranked by similarity (self excluded)."""
mol = Chem.MolFromSmiles(smiles)
if mol is None:
return {"error": f"unparseable SMILES: {smiles}"}
_ensure_sub_index()
fps, smis, tastes, _aromas, _profiles, _dims = _SUB_INDEX
if not fps:
return {"neighbors": [], "note": "no reference set loaded (taste_master.parquet absent)"}
q = _MORGAN.GetFingerprint(mol)
Expand All @@ -1138,7 +1185,46 @@ def substitute(smiles: str, k: int = 8, min_similarity: float = 0.0) -> dict:
if len(neighbors) >= k:
break
return {"query": self_smi, "neighbors": neighbors,
"basis": "Tanimoto / Morgan r2 2048-bit over labeled molecules"}
"basis": "structural — Tanimoto / Morgan r2 2048-bit over labeled molecules"}


# back-compat alias: the endpoint / callers historically called this `substitute`
substitute = structural_neighbors


def substitutes(smiles: str, k: int = 8) -> dict:
"""PROFILE-based substitutes: the k molecules whose predicted FLAVOR profile (taste + aroma
head scores) is closest to the query's — the drop-in reformulation list. A molecule that
*tastes and smells* like the target is a likely substitute regardless of its structure, so
this ranks by cosine similarity over the head-score vectors (not fingerprint distance).
Returns {'substitutes': [...]} ranked by profile match (self excluded)."""
import numpy as np
mol = Chem.MolFromSmiles(smiles)
if mol is None:
return {"error": f"unparseable SMILES: {smiles}"}
_ensure_sub_index()
_fps, smis, tastes, aromas, profiles, _dims = _SUB_INDEX
if profiles is None or not len(smis):
return {"substitutes": [], "note": "no reference set / models loaded"}
x = _feat(mol)
taste_heads, aroma_heads = _profile_heads()
qv = np.array([_CLASSIFIERS[t].predict_proba(x)[0, 1] for t in taste_heads]
+ [_AROMA_MODELS[a].predict_proba(x)[0, 1] for a in aroma_heads], dtype="float32")
qn = qv / (float(np.linalg.norm(qv)) + 1e-9)
pn = profiles / (np.linalg.norm(profiles, axis=1, keepdims=True) + 1e-9)
sims = pn @ qn
self_skel = Chem.MolToInchiKey(mol).split("-")[0]
subs = []
for i in np.argsort(-sims):
ni = Chem.MolFromSmiles(smis[i])
if ni is None or Chem.MolToInchiKey(ni).split("-")[0] == self_skel:
continue
subs.append({"smiles": smis[i], "profile_match": round(float(sims[i]), 3),
"known_tastes": tastes[i], "aromas": aromas[i] if i < len(aromas) else []})
if len(subs) >= k:
break
return {"query": Chem.MolToSmiles(mol), "substitutes": subs,
"basis": "profile — cosine over predicted taste + aroma head scores"}


def palette_match(tastes, aromas=None, k=5):
Expand All @@ -1150,7 +1236,7 @@ def palette_match(tastes, aromas=None, k=5):
predicted descriptors."""
if _SUB_INDEX is None:
_build_sub_index()
_, smis, tlist, alist = _SUB_INDEX
_, smis, tlist, alist, _profiles, _dims = _SUB_INDEX
t_target, a_target = set(tastes or []), set(aromas or [])
if not (t_target or a_target) or not smis:
return {"target": {"tastes": sorted(t_target), "aromas": sorted(a_target)}, "matches": []}
Expand Down
Loading
Loading