From 23ce7cba907468c03ccd331b55c356fec3eaa4ec Mon Sep 17 00:00:00 2001 From: robin-atlasbio Date: Wed, 11 Feb 2026 09:52:26 +0000 Subject: [PATCH] Remove gensim dependency text2term only uses two functions from gensim: `strip_non_alphanum` and `strip_multiple_whitespaces`. Both are trivial regex substitutions. Inlining them removes the heavy gensim dependency (which pulls in numpy, scipy, smart-open, etc. and fails to build on Python 3.14 due to native extensions). The inlined implementations are exact equivalents of the gensim originals (gensim's versions just call `utils.to_unicode(s)` first, which is a no-op for str input on Python 3). --- pyproject.toml | 1 - text2term/onto_utils.py | 21 ++++++++++++++++++++- 2 files changed, 20 insertions(+), 2 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index a912a57..17c06c1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -29,7 +29,6 @@ dependencies = [ "argparse~=1.4.0", "pandas~=2.2.3", "numpy~=1.26.4", - "gensim~=4.3.2", "scipy~=1.12.0", "scikit-learn~=1.6.1", "setuptools~=80.8.0", diff --git a/text2term/onto_utils.py b/text2term/onto_utils.py index 61b147b..370af3c 100644 --- a/text2term/onto_utils.py +++ b/text2term/onto_utils.py @@ -1,10 +1,29 @@ +import re import sys import logging import pandas as pd import bioregistry import shortuuid from owlready2 import * -from gensim.parsing import strip_non_alphanum, strip_multiple_whitespaces + +_RE_NONALPHA = re.compile(r"\W", re.UNICODE) +_RE_WHITESPACE = re.compile(r"(\s)+", re.UNICODE) + + +def strip_non_alphanum(s): + """Replace non-alphanumeric characters with spaces. + + Equivalent to gensim.parsing.preprocessing.strip_non_alphanum. + """ + return _RE_NONALPHA.sub(" ", s) + + +def strip_multiple_whitespaces(s): + """Collapse repeating whitespace characters into a single space. + + Equivalent to gensim.parsing.preprocessing.strip_multiple_whitespaces. + """ + return _RE_WHITESPACE.sub(" ", s) BASE_IRI = "https://text2term.utils/"