Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
19 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion .kiro/steering/project-description.md
Original file line number Diff line number Diff line change
Expand Up @@ -36,8 +36,9 @@ The broader open-source community, particularly users of OpenMDAO and multidisci
| `IntVariable` | `problem.py` | Integer design variable |
| `ArrayVariable` | `problem.py` | NumPy array design variable |
| `CategoricalVariable` | `problem.py` | Categorical (discrete set) variable |
| `StringVariable` | `problem.py` | Free-form string variable (allowed as input or response; cannot be an objective/constraint; treated as fixed / ignored during optimization) |

`Variable` is the union type: `FloatVariable | IntVariable | ArrayVariable | CategoricalVariable`
`Variable` is the union type: `FloatVariable | IntVariable | ArrayVariable | CategoricalVariable | StringVariable`

### Evaluator Hierarchy

Expand Down
8 changes: 7 additions & 1 deletion docs/source/demos/evaluator_interface.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,13 @@
"\n",
"2. **`EvaluatorInfo`** — Used for general-purpose evaluators that only need input/output descriptions without optimization semantics. It defines named, typed inputs and outputs but does not include objectives or constraints.\n",
"\n",
"Both approaches use the same variable types (`FloatVariable`, `IntVariable`, `ArrayVariable`, `CategoricalVariable`) to describe individual inputs and outputs with names, bounds, defaults, and other metadata."
"Both approaches use the same variable types (`FloatVariable`, `IntVariable`, `ArrayVariable`, `CategoricalVariable`) to describe individual inputs and outputs with names, bounds, defaults, and other metadata.\n",
"\n",
"```{note}\n",
"A fifth variable type `StringVariable` exists. It is intended to carry non-numeric\n",
"context into an evaluation (for example a mode name, a material label, or a file path). It\n",
"is not something an optimizer can act on, so its use in an `OptProblem` is restricted. See `StringVariable` docstring for more information.\n",
"```"
]
},
{
Expand Down
5 changes: 5 additions & 0 deletions docs/source/reference/data_models.rst
Original file line number Diff line number Diff line change
Expand Up @@ -50,3 +50,8 @@ Variable Types
:members:
:show-inheritance:
:special-members: __init__

.. autoclass:: standard_evaluator.problem.StringVariable
:members:
:show-inheritance:
:special-members: __init__
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@ authors = [
{name = "Anjali Prasad"},
{name = "Tyler Smith"},
{name = "Mikel Woo"},
{name = "Eduardo Ocampo"},
]
requires-python = ">=3.9"
classifiers = [
Expand Down
3 changes: 2 additions & 1 deletion src/standard_evaluator/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@
from .standard_group import StandardGroup
from .utilities import unique_names
from .problem import ArrayVariable, FloatVariable, IntVariable, Variable, MAXINT
from .problem import OptProblem, CategoricalVariable
from .problem import OptProblem, CategoricalVariable, StringVariable
from .evaluator import EvaluatorInfo, GroupInfo, EquationInfo, JoinedInfo
from .converters import evaluator_info_to_opt_problem, opt_problem_to_evaluator_info
from .aviary_encoder import AviaryEncoder
Expand Down Expand Up @@ -39,6 +39,7 @@
"ArrayVariable",
"FloatVariable",
"CategoricalVariable",
"StringVariable",
"IntVariable",
"Variable",
"MAXINT",
Expand Down
6 changes: 4 additions & 2 deletions src/standard_evaluator/evaluators/abstract_evaluator.py
Original file line number Diff line number Diff line change
Expand Up @@ -322,11 +322,13 @@ def __call__(self, sites: pd.DataFrame, **kwargs) -> None:
+ " Check that input is not unrolled."
)

# Check fixed variables
# Check fixed variables. Only variables with numeric bounds where the
# lower bound equals the upper bound are "pinned" fixed variables.
fixed_vars = [
var
for var in self._opt_problem.variables
if np.array_equal(var.bounds[0], var.bounds[1])
if var.bounds is not None
and np.array_equal(var.bounds[0], var.bounds[1])
]
# mask is a NumPy boolean array of length equal to the number of rows in the sites DataFrame, initialized with all True values.
mask = np.ones(len(sites), dtype=bool)
Expand Down
84 changes: 82 additions & 2 deletions src/standard_evaluator/problem.py
Original file line number Diff line number Diff line change
Expand Up @@ -381,6 +381,69 @@ def check_default(self):
raise ValueError(f"Default not in bounds: {default}")
return self

class StringVariable(FloatVariable, validate_assignment=True):
"""Class representing a free-form string variable.

Unlike :class:`CategoricalVariable`, a string variable is not restricted to a
predefined set of allowed values; it can take on any string value. Bounds,
shift, and scale do not apply to string variables and are therefore fixed to
None.

- It may be used as a *variable* (input) or as a *response* (output). For
example, an analysis may produce a unique file name pointing to a file it
generated.
- It can never be named as an objective or a constraint; doing so raises a
``ValueError``.
- As a variable it is always treated as a *fixed* variable: it is passed
through to the evaluator unchanged and is excluded from the optimization
(no gradient/Jacobian row, not counted as a free variable).

Attributes:
default: Default value for this variable.
bounds: Not applicable for string variables. Always ``None``.
shift: Not applicable for string variables. Always ``None``.
scale: Not applicable for string variables. Always ``None``.
class_type: Class marker for identifying the variable type.
"""

default: Optional[str] = Field(
default=None, description="Default value for this variable"
)
bounds: Literal[None] = Field(
default=None,
description="Not applicable for string variables.",
)
shift: Literal[None] = Field(
default=None,
description="Shift value to be used for this variable. Does not make sense for string variables.",
)
scale: Literal[None] = Field(
default=None,
description="Scale value to be used for this variable. Does not make sense for string variables.",
)
units: Optional[str] = None
class_type: Literal["str"] = Field(default="str", description="Class marker")

def calculate_default(self, overwrite: bool = True) -> None:
"""Calculate the default value for string variables.

A free-form string variable has no bounds from which a meaningful
default could be derived, so an existing default is never overwritten.
The ``overwrite`` flag is therefore ignored. If a default is already
set it is kept, and only an unset (``None``) default is filled with an
empty string.

Parameters
----------
overwrite : bool
Accepted for interface consistency with the other variable types,
but ignored for string variables. An existing default is always
preserved.
"""
if self.default is None:
self.default = ""


class ArrayVariable(FloatVariable, validate_assignment=False):
"""Class defining array variables. The underlying data type are NumPy float64 arrays.

Expand Down Expand Up @@ -659,7 +722,7 @@ def serialize_scale(self, scale: NDArray[Shape["*,..."], np.float64]):
return scale

# Define the Union of the different variable types. Note that we use that for responses as well
Variable = Union[FloatVariable, IntVariable, ArrayVariable, CategoricalVariable]
Variable = Union[FloatVariable, IntVariable, ArrayVariable, CategoricalVariable, StringVariable]


from pydantic import BaseModel, Field
Expand Down Expand Up @@ -847,6 +910,18 @@ def check_problem(self):
response_names = self.unroll_names(self.responses)
elements = set(variable_names + response_names)

# A string variable can never be an objective or a constraint.
string_names = {
element.name
for element in list(self.variables) + list(self.responses)
if isinstance(element, StringVariable)
}
for name in list(self.objectives) + list(self.constraints):
if name in string_names:
raise ValueError(
f"{name} is a StringVariable and cannot be used as an objective or constraint."
)

if self.objectives is not None:
# Check if all the objectives are either a variable or response
for name in self.objectives:
Expand Down Expand Up @@ -950,7 +1025,12 @@ def build_maps(self) -> Tuple[pd.DataFrame, pd.DataFrame]:
fixed = []
for var in opt_problem.variables:
n_elements = len(var_map[var_map["name"] == var.name])
is_fixed = np.array_equal(var.bounds[0], var.bounds[1])
# Variables without numeric bounds (StringVariable) cannot be
# perturbed by the optimizer, so they are always treated as fixed.
if var.bounds is None:
is_fixed = True
else:
is_fixed = np.array_equal(var.bounds[0], var.bounds[1])
fixed.extend([is_fixed] * n_elements)
var_map["fixed"] = fixed

Expand Down
14 changes: 8 additions & 6 deletions src/standard_evaluator/utilities/se_arrays.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@


from standard_evaluator.evaluator import EvaluatorInfo
from standard_evaluator.problem import ArrayVariable, CategoricalVariable, FloatVariable, IntVariable, Variable
from standard_evaluator.problem import ArrayVariable, CategoricalVariable, FloatVariable, IntVariable, StringVariable, Variable


def generate_names(name: str, shape: tuple) -> typing.List[str]:
Expand Down Expand Up @@ -72,7 +72,7 @@ def unroll_names_using_variables(variables: typing.List[Variable]) -> typing.Lis
"""
if not isinstance(variables, list):
raise TypeError('Not given a list')
elif np.any([not isinstance(var, (IntVariable, FloatVariable, ArrayVariable, CategoricalVariable)) for var in variables]):
elif np.any([not isinstance(var, (IntVariable, FloatVariable, ArrayVariable, CategoricalVariable, StringVariable)) for var in variables]):
raise TypeError('The list is not a list of Variables')

unrolled_names = []
Expand Down Expand Up @@ -174,7 +174,7 @@ def unroll_data_frame_using_variables(rolled_df: pd.DataFrame, variables: typing
raise TypeError("Input is not a DataFrame")
if not isinstance(variables, list):
raise TypeError('Not given a list')
elif np.any([not isinstance(var, (IntVariable, FloatVariable, ArrayVariable, CategoricalVariable)) for var in variables]):
elif np.any([not isinstance(var, (IntVariable, FloatVariable, ArrayVariable, CategoricalVariable, StringVariable)) for var in variables]):
raise TypeError('The list is not a list of Variables')

# Create a NumPy array that unrolls arrays in the sub DataFrame.
Expand All @@ -188,6 +188,8 @@ def unroll_data_frame_using_variables(rolled_df: pd.DataFrame, variables: typing
for var in variables:
if isinstance(var, CategoricalVariable):
unrolled_df[var.name] = unrolled_df[var.name].astype(pd.CategoricalDtype(var.bounds, ordered=True))
elif isinstance(var, StringVariable):
unrolled_df[var.name] = unrolled_df[var.name].astype('string')
elif isinstance(var, IntVariable):
unrolled_df[var.name] = unrolled_df[var.name].astype('int')
elif isinstance(var, ArrayVariable):
Expand Down Expand Up @@ -272,7 +274,7 @@ def roll_data_frame_using_variables(unrolled_df: pd.DataFrame, variables: typing
raise TypeError("Input is not a DataFrame")
if not isinstance(variables, list):
raise TypeError('Not given a list')
elif np.any([not isinstance(var, (IntVariable, FloatVariable, ArrayVariable, CategoricalVariable)) for var in variables]):
elif np.any([not isinstance(var, (IntVariable, FloatVariable, ArrayVariable, CategoricalVariable, StringVariable)) for var in variables]):
raise TypeError('The list is not a list of Variables')

vars_to_roll = [var for var in variables if isinstance(var, ArrayVariable)]
Expand Down Expand Up @@ -312,7 +314,7 @@ def check_rolled_data_frame_against_variable_shapes(input_df: pd.DataFrame, var
raise TypeError("Input is not a DataFrame")
if not isinstance(variables, list):
raise TypeError('Not given a list')
elif np.any([not isinstance(var, (IntVariable, FloatVariable, ArrayVariable, CategoricalVariable)) for var in variables]):
elif np.any([not isinstance(var, (IntVariable, FloatVariable, ArrayVariable, CategoricalVariable, StringVariable)) for var in variables]):
raise TypeError('The list is not a list of Variables')


Expand Down Expand Up @@ -346,7 +348,7 @@ def get_variable_shape_in_data_frame(input_df:pd.DataFrame, variables: typing.Li
raise TypeError("Input is not a DataFrame")
if not isinstance(variables, list):
raise TypeError('Not given a list')
elif np.any([not isinstance(var, (IntVariable, FloatVariable, ArrayVariable, CategoricalVariable)) for var in variables]):
elif np.any([not isinstance(var, (IntVariable, FloatVariable, ArrayVariable, CategoricalVariable, StringVariable)) for var in variables]):
raise TypeError('The list is not a list of Variables')
vars_names = [var.name for var in variables]
col_names = input_df.columns
Expand Down
8 changes: 5 additions & 3 deletions src/standard_evaluator/utilities/utility.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@


from standard_evaluator.evaluator import EvaluatorInfo
from standard_evaluator.problem import Variable, ArrayVariable, CategoricalVariable, FloatVariable, IntVariable
from standard_evaluator.problem import Variable, ArrayVariable, CategoricalVariable, FloatVariable, IntVariable, StringVariable
from standard_evaluator.problem import OptProblem

def check_prob(prob: dict) -> None:
Expand Down Expand Up @@ -277,6 +277,8 @@ def get_types_from_evaluator_info(my_info: EvaluatorInfo, variables_only: bool =
# Must start with restrictive types first as they are all Float variables
if isinstance(var, CategoricalVariable):
type_info[var.name] = pd.CategoricalDtype(var.bounds, ordered=True)
elif isinstance(var, StringVariable):
type_info[var.name] = "string"
elif isinstance(var, IntVariable):
type_info[var.name] = "Int64"
elif isinstance(var, ArrayVariable):
Expand Down Expand Up @@ -668,7 +670,7 @@ def update_bounds_to_optimizer_space(element: Variable, shift_val, scale_val) ->
shift_val: The shift value (scalar or array).
scale_val: The scale value (scalar or array).
"""
if isinstance(element, CategoricalVariable):
if isinstance(element, (CategoricalVariable, StringVariable)):
element.shift = None
element.scale = None
return
Expand Down Expand Up @@ -705,7 +707,7 @@ def update_bounds_to_design_space(element: Variable, shift_val, scale_val) -> No
shift_val: The original shift value (scalar or array).
scale_val: The original scale value (scalar or array).
"""
if isinstance(element, CategoricalVariable):
if isinstance(element, (CategoricalVariable, StringVariable)):
element.shift = None
element.scale = None
return
Expand Down
65 changes: 65 additions & 0 deletions tests/evaluators/test_string_variable_evaluation.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,65 @@
import pandas as pd
from pandas.testing import assert_frame_equal

from standard_evaluator import EvaluatorInfo, FloatVariable, StringVariable
from standard_evaluator.evaluators import PyEvaluator
from standard_evaluator.utilities import apply_types_from_evaluator_info


def my_example(df: pd.DataFrame) -> None:
"""Modify ``df`` in place."""
df["out"] = df["input"] + df["val"].astype(str)
df["val_out"] = df["val"] * 4.0


def _make_interface() -> EvaluatorInfo:
return EvaluatorInfo(
name="string_example",
inputs=[
StringVariable(name="input"),
FloatVariable(name="val"),
],
outputs=[
StringVariable(name="out"),
FloatVariable(name="val_out"),
],
)


def test_pyevaluator_matches_direct_call_with_string_columns():
interface = _make_interface()

# Call the function directly, then normalize the column dtypes the
# same way the evaluator does (string columns -> pandas "string" extension
# dtype, float columns -> float64), so the comparison is against a
# like-for-like baseline.
direct = pd.DataFrame({"input": ["one", "two"], "val": [3.4, 5.6]})
my_example(direct)
apply_types_from_evaluator_info(direct, interface)

# Wrapped: run the same function through a PyEvaluator.
wrapped = pd.DataFrame({"input": ["one", "two"], "val": [3.4, 5.6]})
evaluator = PyEvaluator(my_example, interface=interface)
evaluator(wrapped)

assert_frame_equal(wrapped, direct)


def test_string_input_and_output_values_are_preserved():
interface = _make_interface()
wrapped = pd.DataFrame({"input": ["one", "two"], "val": [3.4, 5.6]})

evaluator = PyEvaluator(my_example, interface=interface)
evaluator(wrapped)

# The string input column is untouched.
assert list(wrapped["input"]) == ["one", "two"]
# The string output column holds the concatenated values, not NaN.
assert list(wrapped["out"]) == ["one3.4", "two5.6"]
# The float output is computed as expected.
assert list(wrapped["val_out"]) == [13.6, 22.4]

# String columns use the pandas "string" extension dtype (matching the
# convention already used for IntVariable -> "Int64").
assert wrapped["input"].dtype == "string"
assert wrapped["out"].dtype == "string"
Loading
Loading