Skip to content

Image InputsΒΆ

flock.Image fields as DSPy image inputs (#510).

Artifacts carry images as :class:flock.core.image.Image, a base64 data URL. DSPy sends an image to a multimodal model only when the signature types it as dspy.Image; otherwise the model gets the base64 text. The DSPy engine therefore builds, for each input model that contains images, a copy whose Image fields are dspy.Image (at any depth: lists, optionals, nested models) and passes the inputs as instances of that copy.

Images in conversation context are described (πŸ–Ό image/png Β· 12 KB) rather than inlined, so earlier artifacts do not flood the prompt with base64 text.

ClassesΒΆ

Functions:ΒΆ

dspy_input_model ΒΆ

dspy_input_model(model: type[BaseModel]) -> type[BaseModel]

model, or a copy of it with dspy.Image fields if it holds images.

Source code in src/flock/engines/dspy/image_inputs.py
def dspy_input_model(model: type[BaseModel]) -> type[BaseModel]:
    """``model``, or a copy of it with ``dspy.Image`` fields if it holds images."""
    if model in _copies:
        return _copies[model]
    if model in _in_progress:
        return _in_progress[model]  # resolved by model_rebuild below
    if not _contains_image(model):
        return model
    reference = f"{model.__name__}_{id(model):x}"
    _in_progress[model] = typing.ForwardRef(reference)
    try:
        fields: dict[str, Any] = {}
        for name, field in model.model_fields.items():
            field_copy = copy.copy(field)
            field_copy.annotation = _to_dspy(field.annotation)
            fields[name] = (field_copy.annotation, field_copy)
        # Computed values are in the payload (model_dump); keep them as plain
        # fields so the model sees the same data as for a model without images
        for name, computed in model.model_computed_fields.items():
            fields[name] = (_to_dspy(computed.return_type), None)
        copied = create_model(
            model.__name__,
            __config__=model.model_config,
            __doc__=model.__doc__,
            __module__=__name__,
            **fields,
        )
    finally:
        del _in_progress[model]
    _namespace[reference] = copied
    _copies[model] = copied
    if not _in_progress:  # outermost call: every forward reference is known now
        for pending in _copies.values():
            if not pending.__pydantic_complete__:
                pending.model_rebuild(_types_namespace=_namespace)
    return copied

to_dspy_input ΒΆ

to_dspy_input(value: Any, model: type[BaseModel]) -> Any

A validated input as an instance of :func:dspy_input_model.

Source code in src/flock/engines/dspy/image_inputs.py
def to_dspy_input(value: Any, model: type[BaseModel]) -> Any:
    """A validated input as an instance of :func:`dspy_input_model`."""
    target = dspy_input_model(model)
    if target is model or not isinstance(value, model):
        return value
    # Field names, whatever the model's alias settings
    return target.model_validate(value.model_dump(), by_name=True)

describe_context_images ΒΆ

describe_context_images(items: Iterable[Any]) -> list[Any]

Context items with image data replaced by a short description.

Items are artifacts (copied, never modified) or plain data.

Source code in src/flock/engines/dspy/image_inputs.py
def describe_context_images(items: Iterable[Any]) -> list[Any]:
    """Context items with image data replaced by a short description.

    Items are artifacts (copied, never modified) or plain data.
    """
    described = []
    for item in items:
        if isinstance(item, Artifact):
            payload = _describe_images(item.payload)
            if payload != item.payload:
                item = item.model_copy(update={"payload": payload})
            described.append(item)
        else:
            described.append(_describe_images(item))
    return described

ensure_vision_support ΒΆ

ensure_vision_support(model_name: str, inputs: EvalInputs, *, openai_api: bool) -> None

Fail clearly when images go to an OpenAI API model known to be text-only.

Only models of the OpenAI API itself (openai_api) are checked against LiteLLM's model table: elsewhere a model name can be a deployment alias (Azure, gateways, local servers), so the provider decides.

Source code in src/flock/engines/dspy/image_inputs.py
def ensure_vision_support(
    model_name: str, inputs: EvalInputs, *, openai_api: bool
) -> None:
    """Fail clearly when images go to an OpenAI API model known to be text-only.

    Only models of the OpenAI API itself (``openai_api``) are checked against
    LiteLLM's model table: elsewhere a model name can be a deployment alias
    (Azure, gateways, local servers), so the provider decides.
    """
    if not openai_api:
        return
    import litellm

    info = litellm.model_cost.get(model_name.split("/", 1)[-1])
    if info is None or info.get("supports_vision"):
        return
    fields = _image_fields(inputs)  # validates payloads: only for text-only models
    if not fields:
        return
    raise ValueError(
        f"Model {model_name} does not accept images, but the input "
        f"{', '.join(fields)} contains images. Use a model with vision support."
    )