This is an automated email from the ASF dual-hosted git repository.
bamaer pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/hop.git
The following commit(s) were added to refs/heads/main by this push:
new 600288fc4f Fixes #8441 : Add an Embed text transform to turn text into
embedding vectors (#8454)
600288fc4f is described below
commit 600288fc4fb27752a586c19fa44dd88d3f112765
Author: Bart Maertens <[email protected]>
AuthorDate: Mon Sep 21 10:17:52 2026 +0200
Fixes #8441 : Add an Embed text transform to turn text into embedding
vectors (#8454)
Completes the retrieval chain: chunk, embed, store. Nothing in Hop could
call an embedding model, so #8415 and #8422 had no middle step.
AI provider gains model roles. One provider can now serve a chat model, an
embedding model and a reranker from the same endpoint and credentials, with
a models table in its editor. A transform resolves the role it needs rather
than asking the user to pick one.
- AiModelRole, AiProviderModel and AiProvider.resolveModelName(role)
- AiEmbeddingFactory alongside AiChatFactory, for Ollama and OpenAI
compatible providers
- AiChatFactory honours a CHAT row, still falling back to modelName
- Embed text, in plugins/tech/ai next to the metadata type it needs, the
way pgvector, parquet and avro carry their transforms
- a dependencies.xml so the langchain4j libraries are on the class loader
before the first row rather than once another plugin happens to load
- an AI transform category, which did not exist
Fully backward compatible. modelName and temperature keep their meaning, a
provider with no rows behaves exactly as before, and AiChatFactory was the
only production consumer of either field.
Adds an integration test project for the transforms that call an AI model.
They share one Ollama container, and are disabled by default because the
image and model pulls are too expensive for every run.
---
docker/integration-tests/integration-tests-ai.yaml | 53 ++++
.../assets/images/transforms/icons/embedtext.svg | 10 +
docs/hop-user-manual/modules/ROOT/nav.adoc | 1 +
.../ROOT/pages/metadata-types/ai-provider.adoc | 38 +++
.../ROOT/pages/pipeline/transforms/embedtext.adoc | 90 ++++++
.../transform/messages/messages_en_US.properties | 1 +
integration-tests/ai/0001-embed-text.hpl | 264 ++++++++++++++++
integration-tests/ai/README.md | 44 +++
integration-tests/ai/dev-env-config.json | 9 +
integration-tests/ai/disabled.txt | 21 ++
integration-tests/ai/hop-config.json | 290 ++++++++++++++++++
integration-tests/ai/main-0001-embed-text.hwf | 91 ++++++
.../ai/metadata/ai-provider/ollama-embeddings.json | 17 ++
.../metadata/pipeline-run-configuration/local.json | 17 ++
.../metadata/workflow-run-configuration/local.json | 9 +
integration-tests/ai/project-config.json | 20 ++
plugins/tech/ai/pom.xml | 5 +
plugins/tech/ai/src/assembly/assembly.xml | 4 +
.../org/apache/hop/ai/engine/AiChatFactory.java | 3 +-
.../apache/hop/ai/engine/AiEmbeddingFactory.java | 163 ++++++++++
.../org/apache/hop/ai/metadata/AiModelRole.java | 61 ++++
.../org/apache/hop/ai/metadata/AiProvider.java | 35 +++
.../apache/hop/ai/metadata/AiProviderEditor.java | 91 ++++++
.../apache/hop/ai/metadata/AiProviderModel.java | 74 +++++
.../hop/ai/transforms/embedtext/EmbedText.java | 238 +++++++++++++++
.../hop/ai/transforms/embedtext/EmbedTextData.java | 70 +++++
.../ai/transforms/embedtext/EmbedTextDialog.java | 150 +++++++++
.../hop/ai/transforms/embedtext/EmbedTextMeta.java | 340 +++++++++++++++++++++
.../embedtext/EmbedTextOutputFormat.java | 44 +++
.../tech/ai/src/main/resources/dependencies.xml | 25 ++
plugins/tech/ai/src/main/resources/embedtext.svg | 10 +
.../ai/metadata/messages/messages_en_US.properties | 5 +
.../embedtext/messages/messages_en_US.properties | 59 ++++
.../org/apache/hop/ai/metadata/AiProviderTest.java | 67 ++++
.../ai/transforms/embedtext/EmbedTextMetaTest.java | 241 +++++++++++++++
.../hop/ai/transforms/embedtext/EmbedTextTest.java | 206 +++++++++++++
36 files changed, 2865 insertions(+), 1 deletion(-)
diff --git a/docker/integration-tests/integration-tests-ai.yaml
b/docker/integration-tests/integration-tests-ai.yaml
new file mode 100644
index 0000000000..a773f0ff0b
--- /dev/null
+++ b/docker/integration-tests/integration-tests-ai.yaml
@@ -0,0 +1,53 @@
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements. See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership. The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied. See the License for the
+# specific language governing permissions and limitations
+# under the License.
+
+services:
+ integration_test_ai:
+ extends:
+ file: integration-tests-base.yaml
+ service: integration_test
+ depends_on:
+ ollama:
+ condition: service_healthy
+ links:
+ - ollama
+
+ # One Ollama for every AI integration test. Each transform needs a different
kind of model, so
+ # the models are pulled together here rather than standing up a container
per transform. Add a
+ # pull line below when a new AI test needs another model, and add it to the
healthcheck so the
+ # test container waits for it.
+ ollama:
+ # Pin version for CI stability; tests reach it over compose DNS
(ollama:11434).
+ image: ollama/ollama:0.34.0
+ hostname: ollama
+ # The image ships no models, so the server is started and the test model
pulled here. The
+ # healthcheck below is what holds the test container back until that pull
has finished.
+ entrypoint: ["/bin/sh", "-c"]
+ command:
+ - |
+ ollama serve &
+ until ollama list >/dev/null 2>&1; do sleep 1; done
+ ollama pull nomic-embed-text
+ wait
+ healthcheck:
+ # nomic-embed-text is ~274MB and returns 768 dimensions, which the Embed
text test asserts
+ # on. Extend this condition when another model is added above.
+ test: ["CMD-SHELL", "ollama list | grep -q nomic-embed-text"]
+ interval: 10s
+ timeout: 10s
+ retries: 60
+ start_period: 20s
diff --git
a/docs/hop-user-manual/modules/ROOT/assets/images/transforms/icons/embedtext.svg
b/docs/hop-user-manual/modules/ROOT/assets/images/transforms/icons/embedtext.svg
new file mode 100644
index 0000000000..3f2017dc61
--- /dev/null
+++
b/docs/hop-user-manual/modules/ROOT/assets/images/transforms/icons/embedtext.svg
@@ -0,0 +1,10 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<svg width="24" height="24" viewBox="0 0 24 24" fill="none"
xmlns="http://www.w3.org/2000/svg">
+ <line x1="2" y1="6" x2="9" y2="6" stroke="#2A5D8F" stroke-width="1.6"
stroke-linecap="round"/>
+ <line x1="2" y1="10" x2="9" y2="10" stroke="#2A5D8F" stroke-width="1.6"
stroke-linecap="round"/>
+ <line x1="2" y1="14" x2="7" y2="14" stroke="#2A5D8F" stroke-width="1.6"
stroke-linecap="round"/>
+ <path d="M11 10 L15 10 M13.5 8 L15.5 10 L13.5 12" stroke="#2A5D8F"
stroke-width="1.6" stroke-linecap="round" stroke-linejoin="round"/>
+ <circle cx="19" cy="5" r="1.7" stroke="#2A5D8F" stroke-width="1.6"/>
+ <circle cx="19" cy="12" r="1.7" stroke="#2A5D8F" stroke-width="1.6"/>
+ <circle cx="19" cy="19" r="1.7" stroke="#2A5D8F" stroke-width="1.6"/>
+</svg>
diff --git a/docs/hop-user-manual/modules/ROOT/nav.adoc
b/docs/hop-user-manual/modules/ROOT/nav.adoc
index 133debb528..32e5f8ae8d 100644
--- a/docs/hop-user-manual/modules/ROOT/nav.adoc
+++ b/docs/hop-user-manual/modules/ROOT/nav.adoc
@@ -161,6 +161,7 @@ under the License.
*** xref:pipeline/transforms/dynamicsqlrow.adoc[Dynamic SQL row]
*** xref:pipeline/transforms/edi2xml.adoc[Edi to XML]
*** xref:pipeline/transforms/emailinput.adoc[Email Messages Input]
+*** xref:pipeline/transforms/embedtext.adoc[Embed text]
*** xref:pipeline/transforms/enhancedjsonoutput.adoc[Enhanced JSON output]
*** xref:pipeline/transforms/excelinput.adoc[Excel input]
*** xref:pipeline/transforms/excelwriter.adoc[Excel writer]
diff --git
a/docs/hop-user-manual/modules/ROOT/pages/metadata-types/ai-provider.adoc
b/docs/hop-user-manual/modules/ROOT/pages/metadata-types/ai-provider.adoc
index 2ec4e1aa73..4837768823 100644
--- a/docs/hop-user-manual/modules/ROOT/pages/metadata-types/ai-provider.adoc
+++ b/docs/hop-user-manual/modules/ROOT/pages/metadata-types/ai-provider.adoc
@@ -109,6 +109,9 @@ A GitHub Copilot-style OAuth provider is an extension
point; it is not bundled.
|Model name
|Model identifier for the chosen provider. *Refresh models* fills the combo
from the live catalog when the provider supports it. For Hugging Face, a model
id is sent to the inference router; an `http(s)` URL is called as a dedicated
endpoint. If this field is empty, Base URL is used instead.
+|Models per role
+|Optional. One model for each role this provider serves, so a single provider
can back a chat transform and an embedding transform at the same time. See
below.
+
|Timeout (seconds)
|HTTP timeout for completions.
@@ -117,3 +120,38 @@ A GitHub Copilot-style OAuth provider is an extension
point; it is not bundled.
|===
When Language Model Chat points at an AI Provider, extra transform options
(proxy, retries, mock, I/O fields) are not taken from the provider.
+
+== Models per role
+
+A provider endpoint usually serves more than one kind of model. The same
OpenAI key reaches a chat model and an embedding model; the same Ollama server
serves both, and a reranker besides. **Models per role** lets one provider
object cover all of them, so the base URL and the API key are configured once.
+
+Each row pairs a role with a model name:
+
+|===
+|Role |Used by
+
+|`CHAT`
+|Conversation and completion, for example Language Model Chat and the AI
Advisor.
+
+|`EMBEDDING`
+|Turning text into a vector, for example
xref:pipeline/transforms/embedtext.adoc[Embed text].
+
+|`SCORING`
+|Scoring a passage against a query, used by rerankers.
+
+|`IMAGE`
+|Image generation.
+
+|`MODERATION`
+|Content moderation.
+|===
+
+A transform never asks which role to use: it needs one kind of model and looks
up that role. Point a chat transform and an embedding transform at the same
provider and each finds its own model.
+
+Use at most one row per role, since that is what makes the lookup unambiguous.
A transform that needs a different model than the provider's default for its
role can override it on the transform itself.
+
+=== Relationship to Model name
+
+**Model name** above is the chat model, and it stays that way. A provider with
no rows in this table behaves exactly as before: `CHAT` falls back to **Model
name**, and nothing else resolves.
+
+The fallback is deliberately limited to `CHAT`. Handing a chat model to an
embedding endpoint fails inside the provider with a message that is hard to act
on, so a provider with no `EMBEDDING` row reports that plainly instead.
diff --git
a/docs/hop-user-manual/modules/ROOT/pages/pipeline/transforms/embedtext.adoc
b/docs/hop-user-manual/modules/ROOT/pages/pipeline/transforms/embedtext.adoc
new file mode 100644
index 0000000000..8beb2d237a
--- /dev/null
+++ b/docs/hop-user-manual/modules/ROOT/pages/pipeline/transforms/embedtext.adoc
@@ -0,0 +1,90 @@
+////
+Licensed to the Apache Software Foundation (ASF) under one
+or more contributor license agreements. See the NOTICE file
+distributed with this work for additional information
+regarding copyright ownership. The ASF licenses this file
+to you under the Apache License, Version 2.0 (the
+"License"); you may not use this file except in compliance
+with the License. You may obtain a copy of the License at
+ http://www.apache.org/licenses/LICENSE-2.0
+Unless required by applicable law or agreed to in writing,
+software distributed under the License is distributed on an
+"AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+KIND, either express or implied. See the License for the
+specific language governing permissions and limitations
+under the License.
+////
+:documentationPath: /pipeline/transforms/
+:language: en_US
+:description: Turn a text field into an embedding vector using an AI provider.
+
+= image:transforms/icons/embedtext.svg[Embed text transform Icon,
role="image-doc-icon"] Embed text
+
+[%noheader,cols="3a,1a", role="table-no-borders" ]
+|===
+|
+== Description
+
+The Embed text transform turns a text field into an embedding vector by
calling the embedding model of an xref:metadata-types/ai-provider.adoc[AI
provider]. It is the middle step of a retrieval pipeline: chunk the documents,
embed the chunks, then write them to a vector store.
+
+One row in, one row out. The embedding is added to the row, so nothing already
on it is lost.
+
+|
+== Supported Engines
+[%noheader,cols="2,1a",frame=none, role="table-supported-engines"]
+!===
+!Hop Engine! image:check_mark.svg[Supported, 24]
+!Single Threaded! image:check_mark.svg[Supported, 24]
+!Native Spark! image:question_mark.svg[Maybe Supported, 24]
+!Beam Spark! image:question_mark.svg[Maybe Supported, 24]
+!Beam Flink! image:question_mark.svg[Maybe Supported, 24]
+!Beam Dataflow! image:question_mark.svg[Maybe Supported, 24]
+!===
+|===
+
+== Choosing the model
+
+The model comes from the **EMBEDDING** entry of the selected AI provider, so
one provider can serve a chat transform and this transform at the same time,
with the credentials configured once.
+
+The optional **Model** option overrides that for this transform only, which is
useful when one pipeline needs a different model than the rest.
+
+A provider with no EMBEDDING entry is an error rather than a silent fallback
to its chat model: embedding text with a chat model fails at the provider with
a message that is hard to act on.
+
+Ollama and OpenAI compatible providers are supported. The OpenAI path covers
anything that speaks that API, such as Azure OpenAI, vLLM, LM Studio or a
gateway.
+
+== Output format
+
+The embedding is written either as:
+
+* a **String** holding a JSON array, for example `[0.1,-0.2,0.35]`, which
xref:pipeline/transforms/pgvector-upsert.adoc[pgvector upsert] accepts as a
literal, or
+* a field of the **Vector** value type, which avoids rendering the numbers to
text and back.
+
+Vector is the better choice when the Vector value type plugin is installed.
When it is not, the transform falls back to the String form rather than
failing, so a pipeline moved between installations keeps running.
+
+== Options
+
+[options="header"]
+|===
+|Option |Description
+
+|Transform name|Name of the transform, unique within the pipeline.
+|AI provider|The AI provider serving the embedding model.
+|Model|Optional. Overrides the provider's EMBEDDING model for this transform
only. Accepts variables.
+|Input field|Field holding the text to embed. A row whose text is empty is
passed on with empty output fields rather than being dropped, and keeps its
place in the stream.
+|Output field|Field the embedding is written to.
+|Output format|`STRING` for a JSON array, `VECTOR` for a Vector field.
+|Batch size|Rows buffered before the provider is called. Accepts variables.
Rows with empty text are buffered so they keep their place in the stream, but
are not sent, so a batch can send fewer texts than its size.
+|Include model metadata|Adds the model name and the vector width to each row.
+|Model field|Output field for the embedding model name.
+|Dimensions field|Output field for the number of dimensions in the vector.
+|===
+
+== Notes
+
+Embedding calls are billed and rate limited per request, so **Batch size** is
the main throughput control. The provider embeds the texts in a batch in one
call.
+
+Rows leave this transform in the order they arrived, including the ones with
no text to embed, which wait for the batch around them rather than overtaking
it.
+
+That also shapes what happens on failure. The provider answers per batch and
cannot say which text it choked on, so when an error hop is attached every row
that was sent is diverted rather than one row. Rows with empty text were never
sent and are passed on normally. A batch size of 1 gives per-row precision at
the cost of throughput.
+
+**Include model metadata** is worth leaving on for anything that will be
re-indexed later. Knowing which model produced a stored vector is what lets you
re-embed only what a model change actually affects, and mixing vectors from two
models in one index produces silently poor results.
diff --git
a/engine/src/main/resources/org/apache/hop/pipeline/transform/messages/messages_en_US.properties
b/engine/src/main/resources/org/apache/hop/pipeline/transform/messages/messages_en_US.properties
index 0609894b59..1142d5e203 100644
---
a/engine/src/main/resources/org/apache/hop/pipeline/transform/messages/messages_en_US.properties
+++
b/engine/src/main/resources/org/apache/hop/pipeline/transform/messages/messages_en_US.properties
@@ -19,6 +19,7 @@
+BaseTransform.Category.AI=AI
BaseTransform.Category.BigData=Big Data
BaseTransform.Category.Bulk=Bulk loading
BaseTransform.Category.Cryptography=Cryptography
diff --git a/integration-tests/ai/0001-embed-text.hpl
b/integration-tests/ai/0001-embed-text.hpl
new file mode 100644
index 0000000000..cc780df9eb
--- /dev/null
+++ b/integration-tests/ai/0001-embed-text.hpl
@@ -0,0 +1,264 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<!--
+
+Licensed to the Apache Software Foundation (ASF) under one or more
+contributor license agreements. See the NOTICE file distributed with
+this work for additional information regarding copyright ownership.
+The ASF licenses this file to You under the Apache License, Version 2.0
+(the "License"); you may not use this file except in compliance with
+the License. You may obtain a copy of the License at
+
+ http://www.apache.org/licenses/LICENSE-2.0
+
+Unless required by applicable law or agreed to in writing, software
+distributed under the License is distributed on an "AS IS" BASIS,
+WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+See the License for the specific language governing permissions and
+limitations under the License.
+
+-->
+<pipeline>
+ <info>
+ <name>0001-embed-text</name>
+ <name_sync_with_filename>Y</name_sync_with_filename>
+ <description>Embed two chunks with a real Ollama model and assert the
shape of what comes
+ back. The vector values themselves vary, so the test checks the
dimensions, the model name
+ and that an embedding is present rather than comparing
numbers.</description>
+ <extended_description/>
+ <pipeline_version/>
+ <pipeline_type>Normal</pipeline_type>
+ <pipeline_status>0</pipeline_status>
+ <parameters>
+ </parameters>
+ <capture_transform_performance>N</capture_transform_performance>
+
<transform_performance_capturing_delay>1000</transform_performance_capturing_delay>
+
<transform_performance_capturing_size_limit>100</transform_performance_capturing_size_limit>
+ <created_user>-</created_user>
+ <created_date>2026/09/17 12:00:00.000</created_date>
+ <modified_user>-</modified_user>
+ <modified_date>2026/09/17 12:00:00.000</modified_date>
+ <key_for_session_key/>
+ <is_key_private>N</is_key_private>
+ </info>
+ <notepads>
+ </notepads>
+ <order>
+ <hop> <from>chunks</from> <to>Embed text</to> <enabled>Y</enabled> </hop>
+ <hop> <from>Embed text</from> <to>check dimensions</to>
<enabled>Y</enabled> </hop>
+ <hop> <from>check dimensions</from> <to>check model</to>
<enabled>Y</enabled> </hop>
+ <hop> <from>check dimensions</from> <to>Abort</to> <enabled>Y</enabled>
</hop>
+ <hop> <from>check model</from> <to>check embedding</to>
<enabled>Y</enabled> </hop>
+ <hop> <from>check model</from> <to>Abort</to> <enabled>Y</enabled> </hop>
+ <hop> <from>check embedding</from> <to>OUTPUT</to> <enabled>Y</enabled>
</hop>
+ <hop> <from>check embedding</from> <to>Abort</to> <enabled>Y</enabled>
</hop>
+ </order>
+ <transform>
+ <name>chunks</name>
+ <type>DataGrid</type>
+ <description/>
+ <distribute>Y</distribute>
+ <custom_distribution/>
+ <copies>1</copies>
+ <partitioning>
+ <method>none</method>
+ <schema_name/>
+ </partitioning>
+ <data>
+ <line>
+ <item>Apache Hop is a data orchestration platform.</item>
+ </line>
+ <line>
+ <item>Embeddings turn text into vectors for retrieval.</item>
+ </line>
+ </data>
+ <fields>
+ <field>
+ <length>-1</length>
+ <precision>-1</precision>
+ <set_empty_string>N</set_empty_string>
+ <name>chunk_text</name>
+ <type>String</type>
+ </field>
+ </fields>
+ <attributes/>
+ <GUI>
+ <xloc>96</xloc>
+ <yloc>96</yloc>
+ </GUI>
+ </transform>
+ <transform>
+ <name>Embed text</name>
+ <type>EmbedText</type>
+ <description/>
+ <distribute>Y</distribute>
+ <custom_distribution/>
+ <copies>1</copies>
+ <partitioning>
+ <method>none</method>
+ <schema_name/>
+ </partitioning>
+ <ai_provider>ollama-embeddings</ai_provider>
+ <model_name/>
+ <input_field>chunk_text</input_field>
+ <output_field>embedding</output_field>
+ <output_format>STRING</output_format>
+ <batch_size>${EMBED_BATCH_SIZE}</batch_size>
+ <include_model_metadata>Y</include_model_metadata>
+ <model_field>embedding_model</model_field>
+ <dimensions_field>embedding_dimensions</dimensions_field>
+ <attributes/>
+ <GUI>
+ <xloc>256</xloc>
+ <yloc>96</yloc>
+ </GUI>
+ </transform>
+ <transform>
+ <name>check dimensions</name>
+ <type>FilterRows</type>
+ <description/>
+ <distribute>Y</distribute>
+ <custom_distribution/>
+ <copies>1</copies>
+ <partitioning>
+ <method>none</method>
+ <schema_name/>
+ </partitioning>
+ <send_true_to>check model</send_true_to>
+ <send_false_to>Abort</send_false_to>
+ <compare>
+ <condition>
+ <negated>N</negated>
+ <leftvalue>embedding_dimensions</leftvalue>
+ <function>=</function>
+ <rightvalue/>
+ <value>
+ <name>constant</name>
+ <type>Integer</type>
+ <text>768</text>
+ <length>-1</length>
+ <precision>-1</precision>
+ <isnull>N</isnull>
+ <mask>####0;-####0</mask>
+ </value>
+ </condition>
+ </compare>
+ <attributes/>
+ <GUI>
+ <xloc>416</xloc>
+ <yloc>96</yloc>
+ </GUI>
+ </transform>
+ <transform>
+ <name>check model</name>
+ <type>FilterRows</type>
+ <description/>
+ <distribute>Y</distribute>
+ <custom_distribution/>
+ <copies>1</copies>
+ <partitioning>
+ <method>none</method>
+ <schema_name/>
+ </partitioning>
+ <send_true_to>check embedding</send_true_to>
+ <send_false_to>Abort</send_false_to>
+ <compare>
+ <condition>
+ <negated>N</negated>
+ <leftvalue>embedding_model</leftvalue>
+ <function>=</function>
+ <rightvalue/>
+ <value>
+ <name>constant</name>
+ <type>String</type>
+ <text>nomic-embed-text</text>
+ <length>-1</length>
+ <precision>-1</precision>
+ <isnull>N</isnull>
+ <mask>####0;-####0</mask>
+ </value>
+ </condition>
+ </compare>
+ <attributes/>
+ <GUI>
+ <xloc>576</xloc>
+ <yloc>96</yloc>
+ </GUI>
+ </transform>
+ <transform>
+ <name>check embedding</name>
+ <type>FilterRows</type>
+ <description/>
+ <distribute>Y</distribute>
+ <custom_distribution/>
+ <copies>1</copies>
+ <partitioning>
+ <method>none</method>
+ <schema_name/>
+ </partitioning>
+ <send_true_to>OUTPUT</send_true_to>
+ <send_false_to>Abort</send_false_to>
+ <compare>
+ <condition>
+ <negated>N</negated>
+ <leftvalue>embedding</leftvalue>
+ <function>IS NOT NULL</function>
+ <rightvalue/>
+ <value>
+ <name>constant</name>
+ <type>String</type>
+ <text></text>
+ <length>-1</length>
+ <precision>-1</precision>
+ <isnull>N</isnull>
+ <mask>####0;-####0</mask>
+ </value>
+ </condition>
+ </compare>
+ <attributes/>
+ <GUI>
+ <xloc>736</xloc>
+ <yloc>96</yloc>
+ </GUI>
+ </transform>
+ <transform>
+ <name>OUTPUT</name>
+ <type>Dummy</type>
+ <description/>
+ <distribute>Y</distribute>
+ <custom_distribution/>
+ <copies>1</copies>
+ <partitioning>
+ <method>none</method>
+ <schema_name/>
+ </partitioning>
+ <attributes/>
+ <GUI>
+ <xloc>896</xloc>
+ <yloc>96</yloc>
+ </GUI>
+ </transform>
+ <transform>
+ <name>Abort</name>
+ <type>Abort</type>
+ <description/>
+ <distribute>Y</distribute>
+ <custom_distribution/>
+ <copies>1</copies>
+ <partitioning>
+ <method>none</method>
+ <schema_name/>
+ </partitioning>
+ <row_threshold>0</row_threshold>
+ <message>The Language Model Chat transform did not return a usable answer
from Ollama</message>
+ <always_log_rows>Y</always_log_rows>
+ <abort_option>ABORT_WITH_ERROR</abort_option>
+ <attributes/>
+ <GUI>
+ <xloc>608</xloc>
+ <yloc>240</yloc>
+ </GUI>
+ </transform>
+ <transform_error_handling>
+ </transform_error_handling>
+ <attributes/>
+</pipeline>
diff --git a/integration-tests/ai/README.md b/integration-tests/ai/README.md
new file mode 100644
index 0000000000..cf5c67f1ab
--- /dev/null
+++ b/integration-tests/ai/README.md
@@ -0,0 +1,44 @@
+<!--
+Licensed to the Apache Software Foundation (ASF) under one or more
+contributor license agreements. See the NOTICE file distributed with
+this work for additional information regarding copyright ownership.
+The ASF licenses this file to You under the Apache License, Version 2.0
+(the "License"); you may not use this file except in compliance with
+the License. You may obtain a copy of the License at
+
+ http://www.apache.org/licenses/LICENSE-2.0
+
+Unless required by applicable law or agreed to in writing, software
+distributed under the License is distributed on an "AS IS" BASIS,
+WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+See the License for the specific language governing permissions and
+limitations under the License.
+-->
+
+# AI integration tests
+
+Tests for the transforms that call an AI model. They share one Ollama
container, defined in
+`docker/integration-tests/integration-tests-ai.yaml`, because the image is
large and standing up
+one container per transform would pay for it several times over.
+
+Run them with:
+
+ ./run-tests-docker.sh PROJECT_NAME=ai INCLUDE_DISABLED=true
+
+They are disabled by default: the image and the model pulls are too expensive
for every run of the
+standard suite. See `disabled.txt`.
+
+## Adding a test
+
+1. Add a `main-000N-<name>.hwf` and the pipelines it drives. Each `main-*.hwf`
is reported as its
+ own test.
+2. If it needs a model the container does not pull yet, add a `ollama pull`
line to the compose
+ file and extend the healthcheck so the test container waits for it.
+3. Models are reached over compose DNS at `ollama:11434`. The `ai-provider`
metadata objects in
+ `metadata/ai-provider` point there.
+
+## What each test covers
+
+| Test | Covers |
+|---|---|
+| `0001-embed-text` | The Embed text transform against `nomic-embed-text`: the
vector width, the model name on the row, and that an embedding comes back. The
vector values themselves vary per call, so they are not compared. |
diff --git a/integration-tests/ai/dev-env-config.json
b/integration-tests/ai/dev-env-config.json
new file mode 100644
index 0000000000..a93f908b40
--- /dev/null
+++ b/integration-tests/ai/dev-env-config.json
@@ -0,0 +1,9 @@
+{
+ "variables": [
+ {
+ "name": "OLLAMA_BASE_URL",
+ "value": "http://ollama:11434",
+ "description": "Ollama endpoint on the docker integration-test network.
The ai-provider metadata objects point here, so this is for pipelines that need
the URL directly."
+ }
+ ]
+}
diff --git a/integration-tests/ai/disabled.txt
b/integration-tests/ai/disabled.txt
new file mode 100644
index 0000000000..1c8d67b875
--- /dev/null
+++ b/integration-tests/ai/disabled.txt
@@ -0,0 +1,21 @@
+#
+# Licensed to the Apache Software Foundation (ASF) under one or more
+# contributor license agreements. See the NOTICE file distributed with
+# this work for additional information regarding copyright ownership.
+# The ASF licenses this file to You under the Apache License, Version 2.0
+# (the "License"); you may not use this file except in compliance with
+# the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+# Needs the ollama image and a model pull, which the standard suite should not
pay for on every
+# run. Enable it explicitly:
+#
+# ./run-tests-docker.sh PROJECT_NAME=ai INCLUDE_DISABLED=true
+#
diff --git a/integration-tests/ai/hop-config.json
b/integration-tests/ai/hop-config.json
new file mode 100644
index 0000000000..d9e1e6562e
--- /dev/null
+++ b/integration-tests/ai/hop-config.json
@@ -0,0 +1,290 @@
+{
+ "variables": [
+ {
+ "name": "HOP_LENIENT_STRING_TO_NUMBER_CONVERSION",
+ "value": "N",
+ "description": "System wide flag to allow lenient string to number
conversion for backward compatibility. If this setting is set to \"Y\", an
string starting with digits will be converted successfully into a number.
(example: 192.168.1.1 will be converted into 192 or 192.168 or 192168 depending
on the decimal and grouping symbol). The default (N) will be to throw an error
if non-numeric symbols are found in the string."
+ },
+ {
+ "name": "HOP_COMPATIBILITY_DB_IGNORE_TIMEZONE",
+ "value": "N",
+ "description": "System wide flag to ignore timezone while writing
date/timestamp value to the database."
+ },
+ {
+ "name": "HOP_LOG_SIZE_LIMIT",
+ "value": "0",
+ "description": "The log size limit for all pipelines and workflows that
don't have the \"log size limit\" property set in their respective properties."
+ },
+ {
+ "name": "HOP_EMPTY_STRING_DIFFERS_FROM_NULL",
+ "value": "N",
+ "description": "NULL vs Empty String. If this setting is set to Y, an
empty string and null are different. Otherwise they are not."
+ },
+ {
+ "name": "HOP_MAX_LOG_SIZE_IN_LINES",
+ "value": "0",
+ "description": "The maximum number of log lines that are kept internally
by Hop. Set to 0 to keep all rows (default)"
+ },
+ {
+ "name": "HOP_MAX_LOG_TIMEOUT_IN_MINUTES",
+ "value": "1440",
+ "description": "The maximum age (in minutes) of a log line while being
kept internally by Hop. Set to 0 to keep all rows indefinitely (default)"
+ },
+ {
+ "name": "HOP_MAX_WORKFLOW_TRACKER_SIZE",
+ "value": "5000",
+ "description": "The maximum number of workflow trackers kept in memory"
+ },
+ {
+ "name": "HOP_MAX_ACTIONS_LOGGED",
+ "value": "5000",
+ "description": "The maximum number of action results kept in memory for
logging purposes."
+ },
+ {
+ "name": "HOP_MAX_LOGGING_REGISTRY_SIZE",
+ "value": "10000",
+ "description": "The maximum number of logging registry entries kept in
memory for logging purposes."
+ },
+ {
+ "name": "HOP_LOG_TAB_REFRESH_DELAY",
+ "value": "1000",
+ "description": "The hop log tab refresh delay."
+ },
+ {
+ "name": "HOP_LOG_TAB_REFRESH_PERIOD",
+ "value": "1000",
+ "description": "The hop log tab refresh period."
+ },
+ {
+ "name": "HOP_PLUGIN_CLASSES",
+ "value": null,
+ "description": "A comma delimited list of classes to scan for plugin
annotations"
+ },
+ {
+ "name": "HOP_PLUGIN_PACKAGES",
+ "value": null,
+ "description": "A comma delimited list of packages to scan for plugin
annotations (warning: slow!!)"
+ },
+ {
+ "name": "HOP_TRANSFORM_PERFORMANCE_SNAPSHOT_LIMIT",
+ "value": "0",
+ "description": "The maximum number of transform performance snapshots to
keep in memory. Set to 0 to keep all snapshots indefinitely (default)"
+ },
+ {
+ "name": "HOP_ROWSET_GET_TIMEOUT",
+ "value": "50",
+ "description": "The name of the variable that optionally contains an
alternative rowset get timeout (in ms). This only makes a difference for
extremely short lived pipelines."
+ },
+ {
+ "name": "HOP_ROWSET_PUT_TIMEOUT",
+ "value": "50",
+ "description": "The name of the variable that optionally contains an
alternative rowset put timeout (in ms). This only makes a difference for
extremely short lived pipelines."
+ },
+ {
+ "name": "HOP_CORE_TRANSFORMS_FILE",
+ "value": null,
+ "description": "The name of the project variable that will contain the
alternative location of the hop-transforms.xml file. You can use this to
customize the list of available internal transforms outside of the codebase."
+ },
+ {
+ "name": "HOP_CORE_WORKFLOW_ACTIONS_FILE",
+ "value": null,
+ "description": "The name of the project variable that will contain the
alternative location of the hop-workflow-actions.xml file."
+ },
+ {
+ "name": "HOP_SERVER_OBJECT_TIMEOUT_MINUTES",
+ "value": "1440",
+ "description": "This project variable will set a time-out after which
waiting, completed or stopped pipelines and workflows will be automatically
cleaned up. The default value is 1440 (one day)."
+ },
+ {
+ "name": "HOP_PIPELINE_PAN_JVM_EXIT_CODE",
+ "value": null,
+ "description": "Set this variable to an integer that will be returned as
the Pan JVM exit code."
+ },
+ {
+ "name": "HOP_DISABLE_CONSOLE_LOGGING",
+ "value": "N",
+ "description": "Set this variable to Y to disable standard Hop logging
to the console. (stdout)"
+ },
+ {
+ "name": "HOP_REDIRECT_STDERR",
+ "value": "N",
+ "description": "Set this variable to Y to redirect stderr to Hop
logging."
+ },
+ {
+ "name": "HOP_REDIRECT_STDOUT",
+ "value": "N",
+ "description": "Set this variable to Y to redirect stdout to Hop
logging."
+ },
+ {
+ "name": "HOP_DEFAULT_NUMBER_FORMAT",
+ "value": null,
+ "description": "The name of the variable containing an alternative
default number format"
+ },
+ {
+ "name": "HOP_DEFAULT_BIGNUMBER_FORMAT",
+ "value": null,
+ "description": "The name of the variable containing an alternative
default bignumber format"
+ },
+ {
+ "name": "HOP_DEFAULT_INTEGER_FORMAT",
+ "value": null,
+ "description": "The name of the variable containing an alternative
default integer format"
+ },
+ {
+ "name": "HOP_DEFAULT_DATE_FORMAT",
+ "value": null,
+ "description": "The name of the variable containing an alternative
default date format"
+ },
+ {
+ "name": "HOP_DEFAULT_TIMESTAMP_FORMAT",
+ "value": null,
+ "description": "The name of the variable containing an alternative
default timestamp format"
+ },
+ {
+ "name": "HOP_DEFAULT_SERVLET_ENCODING",
+ "value": null,
+ "description": "Defines the default encoding for servlets, leave it
empty to use Java default encoding"
+ },
+ {
+ "name": "HOP_FAIL_ON_LOGGING_ERROR",
+ "value": "N",
+ "description": "Set this variable to Y when you want the
workflow/pipeline fail with an error when the related logging process (e.g. to
a database) fails."
+ },
+ {
+ "name": "HOP_AGGREGATION_MIN_NULL_IS_VALUED",
+ "value": "N",
+ "description": "Set this variable to Y to set the minimum to NULL if
NULL is within an aggregate. Otherwise by default NULL is ignored by the MIN
aggregate and MIN is set to the minimum value that is not NULL. See also the
variable HOP_AGGREGATION_ALL_NULLS_ARE_ZERO."
+ },
+ {
+ "name": "HOP_AGGREGATION_ALL_NULLS_ARE_ZERO",
+ "value": "N",
+ "description": "Set this variable to Y to return 0 when all values
within an aggregate are NULL. Otherwise by default a NULL is returned when all
values are NULL."
+ },
+ {
+ "name": "HOP_COMPATIBILITY_TEXT_FILE_OUTPUT_APPEND_NO_HEADER",
+ "value": "N",
+ "description": "Set this variable to Y for backward compatibility for
the Text File Output transform. Setting this to Ywill add no header row at all
when the append option is enabled, regardless if the file is existing or not."
+ },
+ {
+ "name": "HOP_PASSWORD_ENCODER_PLUGIN",
+ "value": "Hop",
+ "description": "Specifies the password encoder plugin to use by ID (Hop
is the default)."
+ },
+ {
+ "name": "HOP_SYSTEM_HOSTNAME",
+ "value": null,
+ "description": "You can use this variable to speed up hostname lookup.
Hostname lookup is performed by Hop so that it is capable of logging the server
on which a workflow or pipeline is executed."
+ },
+ {
+ "name": "HOP_SERVER_JETTY_ACCEPTORS",
+ "value": null,
+ "description": "A variable to configure jetty option: acceptors for
Carte"
+ },
+ {
+ "name": "HOP_SERVER_JETTY_ACCEPT_QUEUE_SIZE",
+ "value": null,
+ "description": "A variable to configure jetty option: acceptQueueSize
for Carte"
+ },
+ {
+ "name": "HOP_SERVER_JETTY_RES_MAX_IDLE_TIME",
+ "value": null,
+ "description": "A variable to configure jetty option:
lowResourcesMaxIdleTime for Carte"
+ },
+ {
+ "name":
"HOP_COMPATIBILITY_MERGE_ROWS_USE_REFERENCE_STREAM_WHEN_IDENTICAL",
+ "value": "N",
+ "description": "Set this variable to Y for backward compatibility for
the Merge Rows (diff) transform. Setting this to Y will use the data from the
reference stream (instead of the comparison stream) in case the compared rows
are identical."
+ },
+ {
+ "name": "HOP_SPLIT_FIELDS_REMOVE_ENCLOSURE",
+ "value": "false",
+ "description": "Set this variable to false to preserve enclosure symbol
after splitting the string in the Split fields transform. Changing it to true
will remove first and last enclosure symbol from the resulting string chunks."
+ },
+ {
+ "name": "HOP_ALLOW_EMPTY_FIELD_NAMES_AND_TYPES",
+ "value": "false",
+ "description": "Set this variable to TRUE to allow your pipeline to pass
'null' fields and/or empty types."
+ },
+ {
+ "name": "HOP_GLOBAL_LOG_VARIABLES_CLEAR_ON_EXPORT",
+ "value": "false",
+ "description": "Set this variable to false to preserve global log
variables defined in pipeline / workflow Properties -> Log panel. Changing it
to true will clear it when export pipeline / workflow."
+ },
+ {
+ "name": "HOP_FILE_OUTPUT_MAX_STREAM_COUNT",
+ "value": "1024",
+ "description": "This project variable is used by the Text File Output
transform. It defines the max number of simultaneously open files within the
transform. The transform will close/reopen files as necessary to insure the max
is not exceeded"
+ },
+ {
+ "name": "HOP_FILE_OUTPUT_MAX_STREAM_LIFE",
+ "value": "0",
+ "description": "This project variable is used by the Text File Output
transform. It defines the max number of milliseconds between flushes of files
opened by the transform."
+ },
+ {
+ "name": "HOP_USE_NATIVE_FILE_DIALOG",
+ "value": "N",
+ "description": "Set this value to Y if you want to use the system file
open/save dialog when browsing files"
+ },
+ {
+ "name": "HOP_AUTO_CREATE_CONFIG",
+ "value": "Y",
+ "description": "Set this value to N if you don't want to automatically
create a hop configuration file (hop-config.json) when it's missing"
+ }
+ ],
+ "LocaleDefault": "en_BE",
+ "guiProperties": {
+ "FontFixedSize": "13",
+ "MaxUndo": "100",
+ "DarkMode": "Y",
+ "FontNoteSize": "13",
+ "ShowOSLook": "Y",
+ "FontFixedStyle": "0",
+ "FontNoteName": ".AppleSystemUIFont",
+ "FontFixedName": "Monospaced",
+ "FontGraphStyle": "0",
+ "FontDefaultSize": "13",
+ "GraphColorR": "255",
+ "FontGraphSize": "13",
+ "IconSize": "32",
+ "BackgroundColorB": "255",
+ "FontNoteStyle": "0",
+ "FontGraphName": ".AppleSystemUIFont",
+ "FontDefaultName": ".AppleSystemUIFont",
+ "GraphColorG": "255",
+ "UseGlobalFileBookmarks": "Y",
+ "FontDefaultStyle": "0",
+ "GraphColorB": "255",
+ "BackgroundColorR": "255",
+ "BackgroundColorG": "255",
+ "WorkflowDialogStyle": "RESIZE,MAX,MIN",
+ "LineWidth": "1",
+ "ContextDialogShowCategories": "Y"
+ },
+ "projectsConfig": {
+ "enabled": true,
+ "projectMandatory": true,
+ "environmentMandatory": false,
+ "defaultProject": "default",
+ "defaultEnvironment": null,
+ "standardParentProject": "default",
+ "standardProjectsFolder": null,
+ "projectConfigurations": [
+ {
+ "projectName": "default",
+ "projectHome": "${HOP_CONFIG_FOLDER}",
+ "configFilename": "project-config.json"
+ }
+ ],
+ "lifecycleEnvironments": [
+ {
+ "name": "dev",
+ "purpose": "Testing",
+ "projectName": "default",
+ "configurationFiles": [
+ "${PROJECT_HOME}/dev-env-config.json"
+ ]
+ }
+ ],
+ "projectLifecycles": []
+ }
+}
\ No newline at end of file
diff --git a/integration-tests/ai/main-0001-embed-text.hwf
b/integration-tests/ai/main-0001-embed-text.hwf
new file mode 100644
index 0000000000..fff20f97cc
--- /dev/null
+++ b/integration-tests/ai/main-0001-embed-text.hwf
@@ -0,0 +1,91 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<!--
+
+Licensed to the Apache Software Foundation (ASF) under one or more
+contributor license agreements. See the NOTICE file distributed with
+this work for additional information regarding copyright ownership.
+The ASF licenses this file to You under the Apache License, Version 2.0
+(the "License"); you may not use this file except in compliance with
+the License. You may obtain a copy of the License at
+
+ http://www.apache.org/licenses/LICENSE-2.0
+
+Unless required by applicable law or agreed to in writing, software
+distributed under the License is distributed on an "AS IS" BASIS,
+WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+See the License for the specific language governing permissions and
+limitations under the License.
+
+-->
+<workflow>
+ <name>main-0001-embed-text</name>
+ <name_sync_with_filename>Y</name_sync_with_filename>
+ <description>Embed text against a real Ollama embedding model.</description>
+ <extended_description/>
+ <workflow_version/>
+ <created_user>-</created_user>
+ <created_date>2026/09/17 12:00:00.000</created_date>
+ <modified_user>-</modified_user>
+ <modified_date>2026/09/17 12:00:00.000</modified_date>
+ <parameters>
+ </parameters>
+ <actions>
+ <action>
+ <name>Start</name>
+ <description/>
+ <type>SPECIAL</type>
+ <attributes/>
+ <DayOfMonth>1</DayOfMonth>
+ <doNotWaitOnFirstExecution>N</doNotWaitOnFirstExecution>
+ <hour>12</hour>
+ <intervalMinutes>60</intervalMinutes>
+ <intervalSeconds>0</intervalSeconds>
+ <minutes>0</minutes>
+ <repeat>N</repeat>
+ <schedulerType>0</schedulerType>
+ <weekDay>1</weekDay>
+ <parallel>N</parallel>
+ <xloc>64</xloc>
+ <yloc>64</yloc>
+ <attributes_hac/>
+ </action>
+ <action>
+ <name>Embed two chunks</name>
+ <description/>
+ <type>PIPELINE</type>
+ <attributes/>
+ <add_date>N</add_date>
+ <add_time>N</add_time>
+ <clear_files>N</clear_files>
+ <clear_rows>N</clear_rows>
+ <create_parent_folder>N</create_parent_folder>
+ <exec_per_row>N</exec_per_row>
+ <filename>${PROJECT_HOME}/0001-embed-text.hpl</filename>
+ <loglevel>Basic</loglevel>
+ <parameters>
+ <pass_all_parameters>Y</pass_all_parameters>
+ </parameters>
+ <params_from_previous>N</params_from_previous>
+ <run_configuration>local</run_configuration>
+ <set_append_logfile>N</set_append_logfile>
+ <set_logfile>N</set_logfile>
+ <wait_until_finished>Y</wait_until_finished>
+ <parallel>N</parallel>
+ <xloc>256</xloc>
+ <yloc>80</yloc>
+ <attributes_hac/>
+ </action>
+ </actions>
+ <hops>
+ <hop>
+ <from>Start</from>
+ <to>Embed two chunks</to>
+ <enabled>Y</enabled>
+ <evaluation>Y</evaluation>
+ <unconditional>Y</unconditional>
+ </hop>
+ </hops>
+ <notepads>
+ </notepads>
+ <attributes/>
+</workflow>
diff --git a/integration-tests/ai/metadata/ai-provider/ollama-embeddings.json
b/integration-tests/ai/metadata/ai-provider/ollama-embeddings.json
new file mode 100644
index 0000000000..899c4e0ab8
--- /dev/null
+++ b/integration-tests/ai/metadata/ai-provider/ollama-embeddings.json
@@ -0,0 +1,17 @@
+{
+ "modelName": "",
+ "models": [
+ {
+ "role": "EMBEDDING",
+ "model_name": "nomic-embed-text"
+ }
+ ],
+ "baseUrl": "http://ollama:11434",
+ "apiKey": "Encrypted ",
+ "provider": {
+ "ollama": {}
+ },
+ "name": "ollama-embeddings",
+ "temperature": "0.3",
+ "timeoutSeconds": "60"
+}
diff --git
a/integration-tests/ai/metadata/pipeline-run-configuration/local.json
b/integration-tests/ai/metadata/pipeline-run-configuration/local.json
new file mode 100644
index 0000000000..d8e37459d8
--- /dev/null
+++ b/integration-tests/ai/metadata/pipeline-run-configuration/local.json
@@ -0,0 +1,17 @@
+{
+ "engineRunConfiguration": {
+ "Local": {
+ "feedback_size": "50000",
+ "sample_size": "100",
+ "sample_type_in_gui": "Last",
+ "rowset_size": "10000",
+ "safe_mode": false,
+ "show_feedback": false,
+ "topo_sort": false,
+ "gather_metrics": false
+ }
+ },
+ "configurationVariables": [],
+ "name": "local",
+ "description": "Runs your pipelines locally with the standard local Hop
pipeline engine"
+}
diff --git
a/integration-tests/ai/metadata/workflow-run-configuration/local.json
b/integration-tests/ai/metadata/workflow-run-configuration/local.json
new file mode 100644
index 0000000000..ddb388679a
--- /dev/null
+++ b/integration-tests/ai/metadata/workflow-run-configuration/local.json
@@ -0,0 +1,9 @@
+{
+ "engineRunConfiguration": {
+ "Local": {
+ "safe_mode": false
+ }
+ },
+ "name": "local",
+ "description": "Runs your workflows locally with the standard local Hop
workflow engine"
+}
diff --git a/integration-tests/ai/project-config.json
b/integration-tests/ai/project-config.json
new file mode 100644
index 0000000000..2746fef850
--- /dev/null
+++ b/integration-tests/ai/project-config.json
@@ -0,0 +1,20 @@
+{
+ "metadataBaseFolder": "${PROJECT_HOME}/metadata",
+ "unitTestsBasePath": "${PROJECT_HOME}",
+ "dataSetsCsvFolder": "${PROJECT_HOME}/datasets",
+ "enforcingExecutionInHome": true,
+ "config": {
+ "variables": [
+ {
+ "name": "HOP_LICENSE_HEADER_FILE",
+ "value": "${PROJECT_HOME}/../asf-header.txt",
+ "description": "This will automatically serialize the ASF license
header into pipelines and workflows in the integration test projects"
+ },
+ {
+ "name": "EMBED_BATCH_SIZE",
+ "value": "2",
+ "description": "Batch size for the Embed text transform, supplied as a
variable so the test covers the resolvable form of the option."
+ }
+ ]
+ }
+}
diff --git a/plugins/tech/ai/pom.xml b/plugins/tech/ai/pom.xml
index b2b39a23e5..154413e00c 100644
--- a/plugins/tech/ai/pom.xml
+++ b/plugins/tech/ai/pom.xml
@@ -33,6 +33,11 @@
</properties>
<dependencies>
+ <dependency>
+ <groupId>dev.langchain4j</groupId>
+ <artifactId>langchain4j-ollama</artifactId>
+ <version>${langchain4j.version}</version>
+ </dependency>
<!-- Catalog modules are compile-provided. Runtime jars already ship
with Language Model
Chat (classLoaderGroup hop-ai); do not copy a second set into
this plugin. -->
<dependency>
diff --git a/plugins/tech/ai/src/assembly/assembly.xml
b/plugins/tech/ai/src/assembly/assembly.xml
index e5665f8ac1..ba1a4f2489 100644
--- a/plugins/tech/ai/src/assembly/assembly.xml
+++ b/plugins/tech/ai/src/assembly/assembly.xml
@@ -29,6 +29,10 @@
<outputDirectory>${hop.plugin.libdir}</outputDirectory>
<filtered>true</filtered>
</file>
+ <file>
+
<source>${project.basedir}/src/main/resources/dependencies.xml</source>
+ <outputDirectory>${hop.plugin.libdir}</outputDirectory>
+ </file>
</files>
<fileSets>
<fileSet>
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/engine/AiChatFactory.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/engine/AiChatFactory.java
index 59415aa263..c4903ef21d 100644
--- a/plugins/tech/ai/src/main/java/org/apache/hop/ai/engine/AiChatFactory.java
+++ b/plugins/tech/ai/src/main/java/org/apache/hop/ai/engine/AiChatFactory.java
@@ -25,6 +25,7 @@ import dev.langchain4j.model.output.TokenUsage;
import java.util.ArrayList;
import java.util.List;
import java.util.concurrent.TimeUnit;
+import org.apache.hop.ai.metadata.AiModelRole;
import org.apache.hop.ai.metadata.AiProvider;
import org.apache.hop.ai.provider.IAiProvider;
import org.apache.hop.core.exception.HopException;
@@ -110,7 +111,7 @@ public final class AiChatFactory {
if (Utils.isEmpty(baseUrl) && useProviderDefaults) {
baseUrl = backend.getDefaultBaseUrl();
}
- String modelName = resolve(variables, provider.getModelName());
+ String modelName = resolve(variables,
provider.resolveModelName(AiModelRole.CHAT));
if (Utils.isEmpty(modelName) && useProviderDefaults) {
modelName = backend.getDefaultModelName();
}
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/engine/AiEmbeddingFactory.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/engine/AiEmbeddingFactory.java
new file mode 100644
index 0000000000..24d25f341d
--- /dev/null
+++
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/engine/AiEmbeddingFactory.java
@@ -0,0 +1,163 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.hop.ai.engine;
+
+import dev.langchain4j.model.embedding.EmbeddingModel;
+import dev.langchain4j.model.ollama.OllamaEmbeddingModel;
+import dev.langchain4j.model.openai.OpenAiEmbeddingModel;
+import java.time.Duration;
+import org.apache.hop.ai.metadata.AiModelRole;
+import org.apache.hop.ai.metadata.AiProvider;
+import org.apache.hop.ai.provider.IAiProvider;
+import org.apache.hop.core.exception.HopException;
+import org.apache.hop.core.util.Utils;
+import org.apache.hop.core.variables.IVariables;
+import org.apache.hop.metadata.api.IHopMetadataProvider;
+
+/**
+ * Builds an embedding model from an {@link AiProvider}, the counterpart of
{@link AiChatFactory}
+ * for the chat model.
+ *
+ * <p>The model comes from the provider's {@link AiModelRole#EMBEDDING} entry,
so one provider can
+ * serve a chat transform and an embedding transform at the same time. A
caller that needs a
+ * different model for one pipeline passes it explicitly and it wins.
+ */
+public final class AiEmbeddingFactory {
+
+ private AiEmbeddingFactory() {}
+
+ /**
+ * Loads the named provider once and builds its embedding model, returning
the model together with
+ * the model name that was actually used.
+ *
+ * <p>Both come from one load: the name is wanted for the optional output
field and for error
+ * messages, and reading the metadata twice to get it risks the two
disagreeing.
+ *
+ * @param providerName the {@code AiProvider} to use
+ * @param modelName an embedding model that overrides the provider's, or
empty to use its {@link
+ * AiModelRole#EMBEDDING} entry
+ * @param variables used to resolve the provider's fields
+ * @param metadataProvider where the provider is loaded from
+ * @throws HopException when the provider is missing, incomplete, or serves
no embedding model
+ */
+ public static ResolvedEmbeddingModel resolveEmbeddingModel(
+ String providerName,
+ String modelName,
+ IVariables variables,
+ IHopMetadataProvider metadataProvider)
+ throws HopException {
+ AiProvider provider;
+ try {
+ provider =
metadataProvider.getSerializer(AiProvider.class).load(providerName);
+ } catch (Exception e) {
+ throw new HopException("Error loading AI provider '" + providerName +
"'", e);
+ }
+ if (provider == null) {
+ throw new HopException("AI provider not found: " + providerName);
+ }
+ String resolvedName =
+ Utils.isEmpty(modelName)
+ ?
variables.resolve(provider.resolveModelName(AiModelRole.EMBEDDING))
+ : variables.resolve(modelName);
+ return new ResolvedEmbeddingModel(
+ createEmbeddingModel(provider, resolvedName, variables), resolvedName);
+ }
+
+ /** An embedding model and the name of the model it talks to. */
+ public record ResolvedEmbeddingModel(EmbeddingModel model, String modelName)
{}
+
+ public static EmbeddingModel createEmbeddingModel(
+ AiProvider provider, String modelName, IVariables variables) throws
HopException {
+ if (provider == null) {
+ throw new HopException("An AI provider is required to create an
embedding model");
+ }
+ IAiProvider backend = provider.getProvider();
+ if (backend == null) {
+ throw new HopException("AI provider type is not set on '" +
provider.getName() + "'");
+ }
+
+ // Resolution order: the transform's override, then the provider's
EMBEDDING row. The
+ // provider's plain modelName is the chat model and is deliberately not a
fallback here.
+ String model =
+ Utils.isEmpty(modelName)
+ ?
variables.resolve(provider.resolveModelName(AiModelRole.EMBEDDING))
+ : modelName;
+ if (Utils.isEmpty(model)) {
+ throw new HopException(
+ "No embedding model is configured. Set one on this transform, or add
an EMBEDDING model"
+ + " to AI provider '"
+ + provider.getName()
+ + "'.");
+ }
+
+ String baseUrl = variables.resolve(provider.getBaseUrl());
+ if (Utils.isEmpty(baseUrl)) {
+ baseUrl = backend.getDefaultBaseUrl();
+ }
+ String apiKey = variables.resolve(provider.getApiKey());
+ Duration timeout =
parseTimeout(variables.resolve(provider.getTimeoutSeconds()));
+
+ String hopType = backend.getHopModelType();
+ return switch (hopType == null ? "" : hopType) {
+ case "OLLAMA" -> ollamaModel(baseUrl, model, timeout);
+ case "OPEN_AI" -> openAiModel(baseUrl, apiKey, model, timeout);
+ default ->
+ throw new HopException(
+ "Provider type '"
+ + hopType
+ + "' does not serve embedding models yet. Use an Ollama or
OpenAI compatible"
+ + " provider.");
+ };
+ }
+
+ private static EmbeddingModel ollamaModel(String baseUrl, String model,
Duration timeout) {
+ OllamaEmbeddingModel.OllamaEmbeddingModelBuilder builder =
+ OllamaEmbeddingModel.builder().baseUrl(baseUrl).modelName(model);
+ if (timeout != null) {
+ builder.timeout(timeout);
+ }
+ return builder.build();
+ }
+
+ private static EmbeddingModel openAiModel(
+ String baseUrl, String apiKey, String model, Duration timeout) {
+ OpenAiEmbeddingModel.OpenAiEmbeddingModelBuilder builder =
+ OpenAiEmbeddingModel.builder().modelName(model);
+ if (!Utils.isEmpty(baseUrl)) {
+ builder.baseUrl(baseUrl);
+ }
+ if (!Utils.isEmpty(apiKey)) {
+ builder.apiKey(apiKey);
+ }
+ if (timeout != null) {
+ builder.timeout(timeout);
+ }
+ return builder.build();
+ }
+
+ private static Duration parseTimeout(String seconds) {
+ if (Utils.isEmpty(seconds)) {
+ return null;
+ }
+ try {
+ long value = Long.parseLong(seconds.trim());
+ return value > 0 ? Duration.ofSeconds(value) : null;
+ } catch (NumberFormatException e) {
+ return null;
+ }
+ }
+}
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiModelRole.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiModelRole.java
new file mode 100644
index 0000000000..17fa23feed
--- /dev/null
+++ b/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiModelRole.java
@@ -0,0 +1,61 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.hop.ai.metadata;
+
+/**
+ * What a model on an {@link AiProvider} is used for. A transform resolves the
role it needs rather
+ * than asking the user to pick one, so a single provider can serve a chat
transform, an embedding
+ * transform and a reranker at the same time.
+ *
+ * <p>The constants deliberately do not override {@code toString()}. Generated
dialogs fill an enum
+ * combo with {@code toString()} and read it back with {@code Enum.valueOf},
which only accepts the
+ * constant name.
+ */
+public enum AiModelRole {
+ /** Conversation and completion. The role {@code modelName} falls back to. */
+ CHAT,
+
+ /** Turning text into an embedding vector. */
+ EMBEDDING,
+
+ /** Scoring a passage against a query, used by rerankers. */
+ SCORING,
+
+ /** Image generation. */
+ IMAGE,
+
+ /** Content moderation. */
+ MODERATION;
+
+ /**
+ * Resolves a stored value to a role, tolerating case and unknown values.
+ *
+ * @param value the stored role name
+ * @return the matching role, or {@link #CHAT} when the value is empty or
unrecognised
+ */
+ public static AiModelRole fromString(String value) {
+ if (value == null || value.isEmpty()) {
+ return CHAT;
+ }
+ for (AiModelRole role : values()) {
+ if (role.name().equalsIgnoreCase(value.trim())) {
+ return role;
+ }
+ }
+ return CHAT;
+ }
+}
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProvider.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProvider.java
index 2374a9f85e..d974d6d5c4 100644
--- a/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProvider.java
+++ b/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProvider.java
@@ -23,6 +23,7 @@ import lombok.Getter;
import lombok.Setter;
import org.apache.hop.ai.engine.AiChatFactory;
import org.apache.hop.ai.provider.IAiProvider;
+import org.apache.hop.core.Const;
import org.apache.hop.core.exception.HopException;
import org.apache.hop.core.gui.plugin.GuiElementType;
import org.apache.hop.core.gui.plugin.GuiPlugin;
@@ -132,6 +133,17 @@ public class AiProvider extends HopMetadataBase implements
IHopMetadata {
toolTip = "i18n::AiProvider.Temperature.Tooltip")
private String temperature = "0.3";
+ /**
+ * Models this provider serves, one entry per {@link AiModelRole}. A
transform resolves the role
+ * it needs, so one provider can back a chat transform, an embedding
transform and a reranker at
+ * once.
+ *
+ * <p>Empty on a provider created before roles existed, in which case {@link
AiModelRole#CHAT}
+ * falls back to {@link #modelName} and nothing changes.
+ */
+ @HopMetadataProperty(key = "models", injectionGroupKey = "MODELS")
+ private List<AiProviderModel> models = new ArrayList<>();
+
public AiProvider() {}
public AiProvider(AiProvider other) {
@@ -144,6 +156,29 @@ public class AiProvider extends HopMetadataBase implements
IHopMetadata {
this.timeoutSeconds = other.timeoutSeconds;
this.modelName = other.modelName;
this.temperature = other.temperature;
+ for (AiProviderModel model : other.models) {
+ this.models.add(new AiProviderModel(model));
+ }
+ }
+
+ /**
+ * The model name to use for a role: the entry for that role when there is
one, otherwise {@link
+ * #modelName} for {@link AiModelRole#CHAT}.
+ *
+ * <p>{@code modelName} is the chat model, so it is deliberately not a
fallback for the other
+ * roles: embedding a text with a chat model fails at the provider with a
confusing error, and an
+ * empty result here produces a clear one instead.
+ *
+ * @param role the role the caller needs a model for
+ * @return the model name, or an empty string when none is configured for
that role
+ */
+ public String resolveModelName(AiModelRole role) {
+ for (AiProviderModel model : models) {
+ if (model != null && model.getRole() == role &&
!Utils.isEmpty(model.getModelName())) {
+ return model.getModelName();
+ }
+ }
+ return role == AiModelRole.CHAT ? Const.NVL(modelName, "") : "";
}
public String getPluginId() {
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProviderEditor.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProviderEditor.java
index c155bd5efc..96bbe58441 100644
---
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProviderEditor.java
+++
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProviderEditor.java
@@ -36,6 +36,8 @@ import org.apache.hop.ui.core.gui.GuiCompositeWidgets;
import org.apache.hop.ui.core.gui.GuiCompositeWidgetsAdapter;
import org.apache.hop.ui.core.metadata.MetadataEditor;
import org.apache.hop.ui.core.metadata.MetadataManager;
+import org.apache.hop.ui.core.widget.ColumnInfo;
+import org.apache.hop.ui.core.widget.TableView;
import org.apache.hop.ui.core.widget.TextVar;
import org.apache.hop.ui.hopgui.HopGui;
import org.eclipse.swt.SWT;
@@ -50,6 +52,7 @@ import org.eclipse.swt.widgets.Combo;
import org.eclipse.swt.widgets.Composite;
import org.eclipse.swt.widgets.Control;
import org.eclipse.swt.widgets.Label;
+import org.eclipse.swt.widgets.TableItem;
/** Metadata editor for {@link AiProvider}. */
public class AiProviderEditor extends MetadataEditor<AiProvider> {
@@ -63,6 +66,7 @@ public class AiProviderEditor extends
MetadataEditor<AiProvider> {
private ScrolledComposite wScrolled;
private Composite wContent;
private final AtomicBoolean busyChangingType = new AtomicBoolean(false);
+ private TableView wModels;
public AiProviderEditor(HopGui hopGui, MetadataManager<AiProvider> manager,
AiProvider metadata) {
super(hopGui, manager, metadata);
@@ -119,6 +123,8 @@ public class AiProviderEditor extends
MetadataEditor<AiProvider> {
widgets.createCompositeWidgets(
getMetadata(), null, wContent, AiProvider.GUI_WIDGETS_PARENT_ID, null);
+ addModelsTable();
+
wScrolled.addListener(SWT.Resize, e -> relayoutScrolledContent());
setWidgetsContent();
@@ -135,6 +141,65 @@ public class AiProviderEditor extends
MetadataEditor<AiProvider> {
});
}
+ /**
+ * The per-role model table. It sits below the generated widgets rather than
being one of them,
+ * because a list of rows is not something {@code @GuiWidgetElement} can
express.
+ */
+ private void addModelsTable() {
+ Control last = widgets.getWidgetsMap().get(AiProvider.WIDGET_TEMPERATURE);
+
+ Label wlModels = new Label(wContent, SWT.LEFT);
+ wlModels.setText(BaseMessages.getString(PKG,
"AiProviderEditor.Models.Label"));
+ wlModels.setToolTipText(BaseMessages.getString(PKG,
"AiProviderEditor.Models.Tooltip"));
+ PropsUi.setLook(wlModels);
+ FormData fdlModels = new FormData();
+ fdlModels.left = new FormAttachment(0, 0);
+ fdlModels.right = new FormAttachment(100, 0);
+ fdlModels.top =
+ last == null
+ ? new FormAttachment(0, PropsUi.getMargin())
+ : new FormAttachment(last, PropsUi.getMargin() * 3);
+ wlModels.setLayoutData(fdlModels);
+
+ ColumnInfo[] columns =
+ new ColumnInfo[] {
+ new ColumnInfo(
+ BaseMessages.getString(PKG,
"AiProviderEditor.Models.Column.Role"),
+ ColumnInfo.COLUMN_TYPE_CCOMBO,
+ roleNames(),
+ false),
+ new ColumnInfo(
+ BaseMessages.getString(PKG,
"AiProviderEditor.Models.Column.ModelName"),
+ ColumnInfo.COLUMN_TYPE_TEXT,
+ false)
+ };
+
+ wModels =
+ new TableView(
+ manager.getVariables(),
+ wContent,
+ SWT.BORDER | SWT.FULL_SELECTION | SWT.MULTI,
+ columns,
+ 0,
+ e -> setChanged(),
+ PropsUi.getInstance());
+ FormData fdModels = new FormData();
+ fdModels.left = new FormAttachment(0, 0);
+ fdModels.right = new FormAttachment(100, 0);
+ fdModels.top = new FormAttachment(wlModels, PropsUi.getMargin());
+ fdModels.height = (int) (PropsUi.getInstance().getZoomFactor() * 140);
+ wModels.setLayoutData(fdModels);
+ }
+
+ private static String[] roleNames() {
+ AiModelRole[] roles = AiModelRole.values();
+ String[] names = new String[roles.length];
+ for (int i = 0; i < roles.length; i++) {
+ names[i] = roles[i].name();
+ }
+ return names;
+ }
+
private void changeProviderType() {
if (busyChangingType.get()) {
return;
@@ -203,6 +268,17 @@ public class AiProviderEditor extends
MetadataEditor<AiProvider> {
wProviderType.setText(meta.getPluginName());
}
widgets.setWidgetsContents(meta, wContent,
AiProvider.GUI_WIDGETS_PARENT_ID);
+ if (wModels != null) {
+ wModels.clearAll();
+ for (AiProviderModel model : meta.getModels()) {
+ TableItem item = new TableItem(wModels.table, SWT.NONE);
+ item.setText(1, model.getRole() == null ? AiModelRole.CHAT.name() :
model.getRole().name());
+ item.setText(2, Const.NVL(model.getModelName(), ""));
+ }
+ wModels.removeEmptyRows();
+ wModels.setRowNums();
+ wModels.optWidth(true);
+ }
updateVisibility();
}
@@ -210,6 +286,7 @@ public class AiProviderEditor extends
MetadataEditor<AiProvider> {
public void getWidgetsContent(AiProvider meta) {
meta.setName(wName.getText());
widgets.getWidgetsContents(meta, AiProvider.GUI_WIDGETS_PARENT_ID);
+ meta.setModels(readModels());
String selected = wProviderType.getText();
if (selected != null && !selected.isEmpty()) {
try {
@@ -222,6 +299,20 @@ public class AiProviderEditor extends
MetadataEditor<AiProvider> {
}
}
+ private List<AiProviderModel> readModels() {
+ List<AiProviderModel> models = new ArrayList<>();
+ if (wModels == null || wModels.isDisposed()) {
+ return models;
+ }
+ for (TableItem item : wModels.getNonEmptyItems()) {
+ String modelName = item.getText(2);
+ if (!Utils.isEmpty(modelName)) {
+ models.add(new
AiProviderModel(AiModelRole.fromString(item.getText(1)), modelName));
+ }
+ }
+ return models;
+ }
+
@Override
public Button[] createButtonsForButtonBar(Composite composite) {
Button wbRefresh = new Button(composite, SWT.PUSH | SWT.CENTER);
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProviderModel.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProviderModel.java
new file mode 100644
index 0000000000..513fec1c75
--- /dev/null
+++
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/metadata/AiProviderModel.java
@@ -0,0 +1,74 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.hop.ai.metadata;
+
+import java.util.Objects;
+import org.apache.hop.metadata.api.HopMetadataProperty;
+
+/** One model served by an {@link AiProvider}, for one {@link AiModelRole}. */
+public class AiProviderModel {
+
+ @HopMetadataProperty(key = "role", injectionKey = "MODEL_ROLE")
+ private AiModelRole role = AiModelRole.CHAT;
+
+ @HopMetadataProperty(key = "model_name", injectionKey = "MODEL_NAME")
+ private String modelName = "";
+
+ public AiProviderModel() {}
+
+ public AiProviderModel(AiModelRole role, String modelName) {
+ this.role = role;
+ this.modelName = modelName;
+ }
+
+ public AiProviderModel(AiProviderModel other) {
+ this.role = other.role;
+ this.modelName = other.modelName;
+ }
+
+ public AiModelRole getRole() {
+ return role;
+ }
+
+ public void setRole(AiModelRole role) {
+ this.role = role;
+ }
+
+ public String getModelName() {
+ return modelName;
+ }
+
+ public void setModelName(String modelName) {
+ this.modelName = modelName;
+ }
+
+ @Override
+ public boolean equals(Object o) {
+ if (this == o) {
+ return true;
+ }
+ if (!(o instanceof AiProviderModel other)) {
+ return false;
+ }
+ return role == other.role && Objects.equals(modelName, other.modelName);
+ }
+
+ @Override
+ public int hashCode() {
+ return Objects.hash(role, modelName);
+ }
+}
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedText.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedText.java
new file mode 100644
index 0000000000..709116845c
--- /dev/null
+++
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedText.java
@@ -0,0 +1,238 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.hop.ai.transforms.embedtext;
+
+import dev.langchain4j.data.embedding.Embedding;
+import dev.langchain4j.data.segment.TextSegment;
+import java.util.ArrayList;
+import java.util.List;
+import org.apache.hop.ai.engine.AiEmbeddingFactory;
+import org.apache.hop.core.Const;
+import org.apache.hop.core.exception.HopException;
+import org.apache.hop.core.row.RowDataUtil;
+import org.apache.hop.core.util.Utils;
+import org.apache.hop.i18n.BaseMessages;
+import org.apache.hop.pipeline.Pipeline;
+import org.apache.hop.pipeline.PipelineMeta;
+import org.apache.hop.pipeline.transform.BaseTransform;
+import org.apache.hop.pipeline.transform.TransformMeta;
+
+/** Turns a text field into an embedding vector using the model of an AI
provider. */
+public class EmbedText extends BaseTransform<EmbedTextMeta, EmbedTextData> {
+
+ private static final Class<?> PKG = EmbedTextMeta.class;
+
+ public EmbedText(
+ TransformMeta transformMeta,
+ EmbedTextMeta meta,
+ EmbedTextData data,
+ int copyNr,
+ PipelineMeta pipelineMeta,
+ Pipeline pipeline) {
+ super(transformMeta, meta, data, copyNr, pipelineMeta, pipeline);
+ }
+
+ @Override
+ public boolean init() {
+ if (Utils.isEmpty(meta.getAiProvider())) {
+ logError(BaseMessages.getString(PKG,
"EmbedText.Validation.ProviderRequired"));
+ return false;
+ }
+ data.batchSize = Const.toInt(resolve(meta.getBatchSize()), 16);
+ if (data.batchSize < 1) {
+ logError(BaseMessages.getString(PKG,
"EmbedText.Validation.BatchSizePositive"));
+ return false;
+ }
+ data.emitVector = meta.getOutputFormat() == EmbedTextOutputFormat.VECTOR;
+ return super.init();
+ }
+
+ @Override
+ public boolean processRow() throws HopException {
+ Object[] row = getRow();
+ if (row == null) {
+ flushBatch();
+ setOutputDone();
+ return false;
+ }
+
+ if (first) {
+ first = false;
+ data.inputRowMeta = getInputRowMeta();
+ data.outputRowMeta = data.inputRowMeta.clone();
+ meta.getFields(
+ data.outputRowMeta, getTransformName(), null, null, this,
getMetadataProvider());
+ resolveInputField();
+ resolveOutputFieldNames();
+ openModel();
+ }
+
+ String text = data.inputRowMeta.getString(row, data.inputFieldIndex);
+ data.pendingRows.add(row);
+ // Null marks a row that needs no embedding. It still queues, so the
output keeps input order.
+ data.pendingTexts.add(Utils.isEmpty(text) ? null : text);
+ if (data.pendingRows.size() >= data.batchSize) {
+ flushBatch();
+ }
+ return true;
+ }
+
+ /**
+ * Embeds the pending rows that have text, in one call, and releases every
pending row in input
+ * order.
+ *
+ * <p>A provider failure diverts the whole batch when an error hop is
attached, because the
+ * provider answers per batch and cannot say which text it choked on.
+ */
+ private void flushBatch() throws HopException {
+ if (data.pendingRows.isEmpty()) {
+ return;
+ }
+ List<Object[]> rows = new ArrayList<>(data.pendingRows);
+ List<String> texts = new ArrayList<>(data.pendingTexts);
+ data.pendingRows.clear();
+ data.pendingTexts.clear();
+
+ List<TextSegment> segments = new ArrayList<>();
+ for (String text : texts) {
+ if (text != null) {
+ segments.add(TextSegment.from(text));
+ }
+ }
+
+ List<Embedding> embeddings = List.of();
+ if (!segments.isEmpty()) {
+ try {
+ embeddings = data.model.embedAll(segments).content();
+ } catch (Exception e) {
+ if (!getTransformMeta().isDoingErrorHandling()) {
+ throw new HopException(
+ BaseMessages.getString(PKG, "EmbedText.Error.Embedding",
data.modelName), e);
+ }
+ // Only the rows that were actually in the request failed. The others
carry no text, were
+ // never sent, and belong in the output stream.
+ for (int i = 0; i < rows.size(); i++) {
+ if (texts.get(i) == null) {
+ putRow(data.outputRowMeta, resize(rows.get(i)));
+ } else {
+ putError(
+ data.inputRowMeta,
+ rows.get(i),
+ 1,
+ e.getMessage(),
+ meta.getInputField(),
+ "EMBEDTEXT001");
+ }
+ }
+ return;
+ }
+ if (embeddings == null || embeddings.size() != segments.size()) {
+ throw new HopException(
+ BaseMessages.getString(
+ PKG,
+ "EmbedText.Error.BatchSizeMismatch",
+ String.valueOf(embeddings == null ? 0 : embeddings.size()),
+ String.valueOf(segments.size())));
+ }
+ }
+
+ int embedded = 0;
+ for (int i = 0; i < rows.size(); i++) {
+ Object[] output = resize(rows.get(i));
+ if (texts.get(i) != null) {
+ withEmbedding(output, embeddings.get(embedded++).vector());
+ }
+ putRow(data.outputRowMeta, output);
+ }
+ }
+
+ private void withEmbedding(Object[] output, float[] vector) {
+ int index = data.inputRowMeta.size();
+ output[index++] = data.emitVector ? vector : toJsonArray(vector);
+ // These mirror the columns getFields added, which it decided on the same
resolved names.
+ if (meta.isIncludeModelMetadata()) {
+ if (!Utils.isEmpty(data.modelFieldName)) {
+ output[index++] = data.modelName;
+ }
+ if (!Utils.isEmpty(data.dimensionsFieldName)) {
+ output[index] = (long) vector.length;
+ }
+ }
+ }
+
+ private Object[] resize(Object[] row) {
+ return RowDataUtil.createResizedCopy(row, data.outputRowMeta.size());
+ }
+
+ /** The canonical text form, which pgvector accepts as a literal and any
JSON reader can parse. */
+ static String toJsonArray(float[] vector) {
+ StringBuilder builder = new StringBuilder(vector.length * 8 + 2);
+ builder.append('[');
+ for (int i = 0; i < vector.length; i++) {
+ if (i > 0) {
+ builder.append(',');
+ }
+ builder.append(vector[i]);
+ }
+ return builder.append(']').toString();
+ }
+
+ /** Visible for testing: the first-row resolution step, without needing a
live model. */
+ void resolveOutputFieldNamesForTesting() throws HopException {
+ resolveOutputFieldNames();
+ }
+
+ private void resolveOutputFieldNames() throws HopException {
+ data.outputFieldName = resolve(meta.getOutputField());
+ data.modelFieldName = resolve(meta.getModelField());
+ data.dimensionsFieldName = resolve(meta.getDimensionsField());
+ // getFields adds no embedding column for an empty name, so writing one
anyway would run off
+ // the end of the row. A variable that resolves to nothing is a mistake
worth reporting, not
+ // a reason to drop the embedding quietly.
+ if (Utils.isEmpty(data.outputFieldName)) {
+ throw new HopException(
+ BaseMessages.getString(PKG,
"EmbedText.Validation.OutputFieldRequired"));
+ }
+ }
+
+ private void resolveInputField() throws HopException {
+ data.inputFieldIndex =
data.inputRowMeta.indexOfValue(meta.getInputField());
+ if (data.inputFieldIndex < 0) {
+ throw new HopException(
+ BaseMessages.getString(
+ PKG,
+ "EmbedText.Validation.InputFieldNotFound",
+ String.valueOf(meta.getInputField())));
+ }
+ }
+
+ private void openModel() throws HopException {
+ AiEmbeddingFactory.ResolvedEmbeddingModel resolved =
+ AiEmbeddingFactory.resolveEmbeddingModel(
+ resolve(meta.getAiProvider()), meta.getModelName(), this,
getMetadataProvider());
+ data.model = resolved.model();
+ data.modelName = resolved.modelName();
+ }
+
+ @Override
+ public void dispose() {
+ data.pendingRows.clear();
+ data.pendingTexts.clear();
+ data.model = null;
+ super.dispose();
+ }
+}
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextData.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextData.java
new file mode 100644
index 0000000000..bbb8da0534
--- /dev/null
+++
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextData.java
@@ -0,0 +1,70 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.hop.ai.transforms.embedtext;
+
+import dev.langchain4j.model.embedding.EmbeddingModel;
+import java.util.ArrayList;
+import java.util.List;
+import org.apache.hop.core.row.IRowMeta;
+import org.apache.hop.pipeline.transform.BaseTransformData;
+import org.apache.hop.pipeline.transform.ITransformData;
+
+public class EmbedTextData extends BaseTransformData implements ITransformData
{
+
+ public IRowMeta inputRowMeta;
+ public IRowMeta outputRowMeta;
+
+ public EmbeddingModel model;
+
+ /** Model name actually in use, for the optional output field and for error
messages. */
+ public String modelName;
+
+ /** Batch size resolved once in init, so a variable is not re-resolved per
row. */
+ public int batchSize;
+
+ /** True when the embedding is emitted as a Vector field rather than a JSON
string. */
+ public boolean emitVector;
+
+ public int inputFieldIndex = -1;
+
+ /**
+ * Output field names resolved once in the first row, so the layout {@code
getFields} produced and
+ * the slots written here are decided by the same values. Judging them
separately lets a variable
+ * that resolves to empty put a value in the wrong column.
+ */
+ public String outputFieldName;
+
+ public String modelFieldName;
+
+ public String dimensionsFieldName;
+
+ /**
+ * Rows waiting to be released, in input order. They are only passed
downstream once the provider
+ * has answered, so no downstream transform sees a row before its embedding
exists.
+ *
+ * <p>A row whose text is empty needs no embedding, but it still waits here
rather than being
+ * emitted straight away: letting it overtake the rows already buffered
would reorder the stream.
+ * Its entry in {@link #pendingTexts} is null.
+ */
+ public final List<Object[]> pendingRows = new ArrayList<>();
+
+ public final List<String> pendingTexts = new ArrayList<>();
+
+ public EmbedTextData() {
+ super();
+ }
+}
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextDialog.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextDialog.java
new file mode 100644
index 0000000000..0a6d56d4db
--- /dev/null
+++
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextDialog.java
@@ -0,0 +1,150 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.hop.ai.transforms.embedtext;
+
+import org.apache.hop.core.logging.LogChannel;
+import org.apache.hop.core.row.IRowMeta;
+import org.apache.hop.core.util.Utils;
+import org.apache.hop.core.variables.IVariables;
+import org.apache.hop.i18n.BaseMessages;
+import org.apache.hop.pipeline.PipelineMeta;
+import org.apache.hop.ui.core.dialog.BaseDialog;
+import org.apache.hop.ui.core.gui.GuiCompositeWidgets;
+import org.apache.hop.ui.core.gui.GuiCompositeWidgetsAdapter;
+import org.apache.hop.ui.core.widget.ComboVar;
+import org.apache.hop.ui.pipeline.transform.BaseTransformDialog;
+import org.eclipse.swt.widgets.Combo;
+import org.eclipse.swt.widgets.Control;
+import org.eclipse.swt.widgets.Shell;
+
+public class EmbedTextDialog extends BaseTransformDialog {
+
+ private static final Class<?> PKG = EmbedTextMeta.class;
+
+ private final EmbedTextMeta input;
+ private GuiCompositeWidgets widgets;
+ private boolean loading;
+
+ public EmbedTextDialog(
+ Shell parent, IVariables variables, EmbedTextMeta transformMeta,
PipelineMeta pipelineMeta) {
+ super(parent, variables, transformMeta, pipelineMeta);
+ input = transformMeta;
+ }
+
+ @Override
+ public String open() {
+ createShell(BaseMessages.getString(PKG, "EmbedTextDialog.Shell.Title"));
+ buildButtonBar().ok(e -> ok()).cancel(e -> cancel()).build();
+
+ changed = input.hasChanged();
+ loading = true;
+
+ widgets =
+ GuiCompositeWidgets.addScrolledComposite(
+ shell,
+ variables,
+ wTransformName,
+ wOk,
+ EmbedTextMeta.GUI_PLUGIN_ELEMENT_PARENT_ID,
+ input);
+ widgets.setWidgetsListener(
+ new GuiCompositeWidgetsAdapter() {
+ @Override
+ public void widgetModified(
+ GuiCompositeWidgets compositeWidgets, Control changedWidget,
String widgetId) {
+ if (!loading) {
+ input.setChanged();
+ }
+ }
+ });
+
+ setFieldComboValues();
+ loading = false;
+ input.setChanged(changed);
+
+ focusTransformName();
+ BaseDialog.defaultShellHandling(shell, c -> ok(), c -> cancel());
+ return transformName;
+ }
+
+ /**
+ * Stream field names cannot come from {@code comboValuesMethod}, which is
handed only a log
+ * channel and a metadata provider, so the input field combo is filled once
the widgets exist.
+ */
+ private void setFieldComboValues() {
+ try {
+ IRowMeta fields = pipelineMeta.getPrevTransformFields(variables,
transformName);
+ String[] names = fields == null ? new String[0] : fields.getFieldNames();
+ setComboItems(EmbedTextMeta.WIDGET_INPUT_FIELD, names);
+ } catch (Exception e) {
+ LogChannel.UI.logError("Error getting source fields", e);
+ }
+ }
+
+ /** Fills a combo without losing the selection the transform was saved with.
*/
+ private void setComboItems(String widgetId, String[] names) {
+ // Setting items clears the widget's text, so put the saved selection back
afterwards.
+ String selected = comboText(widgetId);
+ widgets.setComboValues(widgetId, names);
+ if (!Utils.isEmpty(selected)) {
+ setComboText(widgetId, selected);
+ }
+ }
+
+ /**
+ * {@code GuiCompositeWidgets} builds a plain SWT {@link Combo} when the
element has no variable
+ * support and a {@link ComboVar} when it does, so both have to be handled.
+ */
+ private String comboText(String widgetId) {
+ Control control = widgets.getWidgetsMap().get(widgetId);
+ if (control instanceof ComboVar comboVar && !comboVar.isDisposed()) {
+ return comboVar.getText();
+ }
+ if (control instanceof Combo combo && !combo.isDisposed()) {
+ return combo.getText();
+ }
+ return "";
+ }
+
+ private void setComboText(String widgetId, String text) {
+ Control control = widgets.getWidgetsMap().get(widgetId);
+ if (control == null || control.isDisposed()) {
+ return;
+ }
+ if (control instanceof ComboVar comboVar) {
+ comboVar.setText(text);
+ } else if (control instanceof Combo combo) {
+ combo.setText(text);
+ }
+ }
+
+ private void cancel() {
+ transformName = null;
+ input.setChanged(changed);
+ dispose();
+ }
+
+ private void ok() {
+ if (Utils.isEmpty(wTransformName.getText())) {
+ return;
+ }
+ widgets.getWidgetsContents(input,
EmbedTextMeta.GUI_PLUGIN_ELEMENT_PARENT_ID);
+ transformName = wTransformName.getText();
+ input.setChanged();
+ dispose();
+ }
+}
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextMeta.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextMeta.java
new file mode 100644
index 0000000000..ee9f05c28e
--- /dev/null
+++
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextMeta.java
@@ -0,0 +1,340 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.hop.ai.transforms.embedtext;
+
+import java.util.List;
+import java.util.function.IntPredicate;
+import lombok.Getter;
+import lombok.Setter;
+import org.apache.hop.ai.metadata.AiProvider;
+import org.apache.hop.core.CheckResult;
+import org.apache.hop.core.Const;
+import org.apache.hop.core.ICheckResult;
+import org.apache.hop.core.annotations.Transform;
+import org.apache.hop.core.exception.HopTransformException;
+import org.apache.hop.core.gui.plugin.GuiElementType;
+import org.apache.hop.core.gui.plugin.GuiPlugin;
+import org.apache.hop.core.gui.plugin.GuiWidgetElement;
+import org.apache.hop.core.gui.plugin.GuiWidgetGroupType;
+import org.apache.hop.core.row.IRowMeta;
+import org.apache.hop.core.row.IValueMeta;
+import org.apache.hop.core.row.value.ValueMetaFactory;
+import org.apache.hop.core.util.StringUtil;
+import org.apache.hop.core.util.Utils;
+import org.apache.hop.core.variables.IVariables;
+import org.apache.hop.i18n.BaseMessages;
+import org.apache.hop.metadata.api.HopMetadataProperty;
+import org.apache.hop.metadata.api.IHopMetadataProvider;
+import org.apache.hop.pipeline.PipelineMeta;
+import org.apache.hop.pipeline.transform.BaseTransformMeta;
+import org.apache.hop.pipeline.transform.TransformMeta;
+
+@Getter
+@Setter
+@Transform(
+ id = "EmbedText",
+ image = "embedtext.svg",
+ name = "i18n::EmbedText.Name",
+ description = "i18n::EmbedText.Description",
+ categoryDescription =
"i18n:org.apache.hop.pipeline.transform:BaseTransform.Category.AI",
+ keywords = "embedding,vector,ai,rag,retrieval,semantic",
+ documentationUrl = "/pipeline/transforms/embedtext.html",
+ // Shares a child-first classloader with hop-tech-ai, which supplies
AiEmbeddingFactory
+ // and langchain4j. Those jars are therefore excluded from this plugin's
own lib.
+ classLoaderGroup = "hop-ai")
+@GuiPlugin(classLoaderGroup = "hop-ai")
+public class EmbedTextMeta extends BaseTransformMeta<EmbedText, EmbedTextData>
{
+
+ public static final String GUI_PLUGIN_ELEMENT_PARENT_ID =
"EMBED_TEXT_DIALOG_OPTIONS";
+ public static final String WIDGET_INPUT_FIELD = "EMBED_TEXT_INPUT_FIELD";
+ public static final String WIDGET_MODEL_NAME = "EMBED_TEXT_MODEL_NAME";
+
+ private static final String TAB_MAIN = "i18n::EmbedText.Tab.Main";
+ private static final String TAB_MAIN_ORDER = "0100";
+
+ /** The dimension of an OpenAI text-embedding-3-small vector, and the Vector
value type's id. */
+ private static final int TYPE_VECTOR = 1536;
+
+ private static final Class<?> PKG = EmbedTextMeta.class;
+
+ @GuiWidgetElement(
+ order = "0100",
+ type = GuiElementType.METADATA,
+ metadata = AiProvider.class,
+ label = "i18n::EmbedText.aiProvider.Label",
+ toolTip = "i18n::EmbedText.aiProvider.Tooltip",
+ parentId = GUI_PLUGIN_ELEMENT_PARENT_ID,
+ groupType = GuiWidgetGroupType.TABS,
+ group = TAB_MAIN,
+ groupOrder = TAB_MAIN_ORDER)
+ @HopMetadataProperty(
+ key = "ai_provider",
+ injectionKey = "AI_PROVIDER",
+ injectionKeyDescription = "EmbedTextMeta.Injection.AI_PROVIDER")
+ private String aiProvider;
+
+ @GuiWidgetElement(
+ id = WIDGET_MODEL_NAME,
+ order = "0200",
+ type = GuiElementType.TEXT,
+ label = "i18n::EmbedText.modelName.Label",
+ toolTip = "i18n::EmbedText.modelName.Tooltip",
+ variables = true,
+ parentId = GUI_PLUGIN_ELEMENT_PARENT_ID,
+ groupType = GuiWidgetGroupType.TABS,
+ group = TAB_MAIN,
+ groupOrder = TAB_MAIN_ORDER)
+ @HopMetadataProperty(
+ key = "model_name",
+ injectionKey = "MODEL_NAME",
+ injectionKeyDescription = "EmbedTextMeta.Injection.MODEL_NAME")
+ private String modelName = "";
+
+ @GuiWidgetElement(
+ id = WIDGET_INPUT_FIELD,
+ order = "0300",
+ type = GuiElementType.COMBO,
+ label = "i18n::EmbedText.inputField.Label",
+ toolTip = "i18n::EmbedText.inputField.Tooltip",
+ parentId = GUI_PLUGIN_ELEMENT_PARENT_ID,
+ groupType = GuiWidgetGroupType.TABS,
+ group = TAB_MAIN,
+ groupOrder = TAB_MAIN_ORDER)
+ @HopMetadataProperty(
+ key = "input_field",
+ injectionKey = "INPUT_FIELD",
+ injectionKeyDescription = "EmbedTextMeta.Injection.INPUT_FIELD")
+ private String inputField = "chunk_text";
+
+ @GuiWidgetElement(
+ order = "0400",
+ type = GuiElementType.TEXT,
+ label = "i18n::EmbedText.outputField.Label",
+ toolTip = "i18n::EmbedText.outputField.Tooltip",
+ parentId = GUI_PLUGIN_ELEMENT_PARENT_ID,
+ groupType = GuiWidgetGroupType.TABS,
+ group = TAB_MAIN,
+ groupOrder = TAB_MAIN_ORDER)
+ @HopMetadataProperty(
+ key = "output_field",
+ injectionKey = "OUTPUT_FIELD",
+ injectionKeyDescription = "EmbedTextMeta.Injection.OUTPUT_FIELD")
+ private String outputField = "embedding";
+
+ @GuiWidgetElement(
+ order = "0500",
+ type = GuiElementType.COMBO,
+ label = "i18n::EmbedText.outputFormat.Label",
+ toolTip = "i18n::EmbedText.outputFormat.Tooltip",
+ parentId = GUI_PLUGIN_ELEMENT_PARENT_ID,
+ groupType = GuiWidgetGroupType.TABS,
+ group = TAB_MAIN,
+ groupOrder = TAB_MAIN_ORDER)
+ @HopMetadataProperty(
+ key = "output_format",
+ injectionKey = "OUTPUT_FORMAT",
+ injectionKeyDescription = "EmbedTextMeta.Injection.OUTPUT_FORMAT")
+ private EmbedTextOutputFormat outputFormat = EmbedTextOutputFormat.STRING;
+
+ @GuiWidgetElement(
+ order = "0600",
+ type = GuiElementType.TEXT,
+ label = "i18n::EmbedText.batchSize.Label",
+ toolTip = "i18n::EmbedText.batchSize.Tooltip",
+ variables = true,
+ parentId = GUI_PLUGIN_ELEMENT_PARENT_ID,
+ groupType = GuiWidgetGroupType.TABS,
+ group = TAB_MAIN,
+ groupOrder = TAB_MAIN_ORDER)
+ @HopMetadataProperty(
+ key = "batch_size",
+ injectionKey = "BATCH_SIZE",
+ injectionKeyDescription = "EmbedTextMeta.Injection.BATCH_SIZE")
+ private String batchSize = "16";
+
+ @GuiWidgetElement(
+ order = "0700",
+ type = GuiElementType.CHECKBOX,
+ label = "i18n::EmbedText.includeModelMetadata.Label",
+ toolTip = "i18n::EmbedText.includeModelMetadata.Tooltip",
+ parentId = GUI_PLUGIN_ELEMENT_PARENT_ID,
+ groupType = GuiWidgetGroupType.TABS,
+ group = TAB_MAIN,
+ groupOrder = TAB_MAIN_ORDER)
+ @HopMetadataProperty(
+ key = "include_model_metadata",
+ injectionKey = "INCLUDE_MODEL_METADATA",
+ injectionKeyDescription =
"EmbedTextMeta.Injection.INCLUDE_MODEL_METADATA")
+ private boolean includeModelMetadata = true;
+
+ @GuiWidgetElement(
+ order = "0800",
+ type = GuiElementType.TEXT,
+ label = "i18n::EmbedText.modelField.Label",
+ toolTip = "i18n::EmbedText.modelField.Tooltip",
+ parentId = GUI_PLUGIN_ELEMENT_PARENT_ID,
+ groupType = GuiWidgetGroupType.TABS,
+ group = TAB_MAIN,
+ groupOrder = TAB_MAIN_ORDER)
+ @HopMetadataProperty(
+ key = "model_field",
+ injectionKey = "MODEL_FIELD",
+ injectionKeyDescription = "EmbedTextMeta.Injection.MODEL_FIELD")
+ private String modelField = "embedding_model";
+
+ @GuiWidgetElement(
+ order = "0900",
+ type = GuiElementType.TEXT,
+ label = "i18n::EmbedText.dimensionsField.Label",
+ toolTip = "i18n::EmbedText.dimensionsField.Tooltip",
+ parentId = GUI_PLUGIN_ELEMENT_PARENT_ID,
+ groupType = GuiWidgetGroupType.TABS,
+ group = TAB_MAIN,
+ groupOrder = TAB_MAIN_ORDER)
+ @HopMetadataProperty(
+ key = "dimensions_field",
+ injectionKey = "DIMENSIONS_FIELD",
+ injectionKeyDescription = "EmbedTextMeta.Injection.DIMENSIONS_FIELD")
+ private String dimensionsField = "embedding_dimensions";
+
+ @Override
+ public void setDefault() {
+ aiProvider = "";
+ modelName = "";
+ inputField = "chunk_text";
+ outputField = "embedding";
+ outputFormat = EmbedTextOutputFormat.STRING;
+ batchSize = "16";
+ includeModelMetadata = true;
+ modelField = "embedding_model";
+ dimensionsField = "embedding_dimensions";
+ }
+
+ @Override
+ public void getFields(
+ IRowMeta row,
+ String origin,
+ IRowMeta[] info,
+ TransformMeta nextTransform,
+ IVariables variables,
+ IHopMetadataProvider metadataProvider)
+ throws HopTransformException {
+ // EmbedText resolves the same three names once per run and writes the
slots in this order.
+ // Both sides have to judge emptiness on the resolved value, or a variable
that resolves to
+ // nothing adds no column here while the transform still writes one.
+ addField(row, origin, variables.resolve(outputField), embeddingType());
+ if (includeModelMetadata) {
+ addField(row, origin, variables.resolve(modelField),
IValueMeta.TYPE_STRING);
+ addField(row, origin, variables.resolve(dimensionsField),
IValueMeta.TYPE_INTEGER);
+ }
+ }
+
+ /**
+ * The Vector value type is an optional plugin. When it is not installed the
embedding falls back
+ * to a String holding a JSON array, which every downstream transform can
still read.
+ */
+ private int embeddingType() {
+ if (outputFormat != EmbedTextOutputFormat.VECTOR) {
+ return IValueMeta.TYPE_STRING;
+ }
+ try {
+ ValueMetaFactory.createValueMeta("probe", TYPE_VECTOR);
+ return TYPE_VECTOR;
+ } catch (Exception e) {
+ return IValueMeta.TYPE_STRING;
+ }
+ }
+
+ private static void addField(IRowMeta row, String origin, String name, int
type)
+ throws HopTransformException {
+ if (Utils.isEmpty(name)) {
+ return;
+ }
+ try {
+ IValueMeta value = ValueMetaFactory.createValueMeta(name, type);
+ value.setOrigin(origin);
+ row.addValueMeta(value);
+ } catch (Exception e) {
+ throw new HopTransformException("Unable to add output field '" + name +
"'", e);
+ }
+ }
+
+ @Override
+ public boolean supportsErrorHandling() {
+ return true;
+ }
+
+ @Override
+ public void check(
+ List<ICheckResult> remarks,
+ PipelineMeta pipelineMeta,
+ TransformMeta transformMeta,
+ IRowMeta prev,
+ String[] input,
+ String[] output,
+ IRowMeta info,
+ IVariables variables,
+ IHopMetadataProvider metadataProvider) {
+
+ if (Utils.isEmpty(aiProvider)) {
+ error(remarks, transformMeta, "EmbedText.Validation.ProviderRequired");
+ }
+ if (Utils.isEmpty(inputField)) {
+ error(remarks, transformMeta, "EmbedText.Validation.InputFieldRequired");
+ } else if (prev != null && prev.indexOfValue(inputField) < 0) {
+ error(remarks, transformMeta, "EmbedText.Validation.InputFieldNotFound",
inputField);
+ }
+ if (Utils.isEmpty(outputField)) {
+ error(remarks, transformMeta,
"EmbedText.Validation.OutputFieldRequired");
+ }
+ // Batch size accepts variables, which a design-time check cannot resolve.
Only a value that is
+ // genuinely fixed can be judged here.
+ if (isResolvedNumberBad(variables, batchSize, size -> size < 1)) {
+ error(remarks, transformMeta, "EmbedText.Validation.BatchSizePositive");
+ }
+ if (includeModelMetadata && Utils.isEmpty(modelField) &&
Utils.isEmpty(dimensionsField)) {
+ warning(remarks, transformMeta, "EmbedText.Validation.NoMetadataFields");
+ }
+ }
+
+ private static boolean isResolvedNumberBad(
+ IVariables variables, String value, IntPredicate invalid) {
+ String resolved = variables.resolve(value);
+ if (StringUtil.containsVariableToken(resolved)) {
+ return false;
+ }
+ return invalid.test(Const.toInt(resolved, -1));
+ }
+
+ private static void error(
+ List<ICheckResult> remarks, TransformMeta transformMeta, String key,
String... parameters) {
+ remarks.add(
+ new CheckResult(
+ ICheckResult.TYPE_RESULT_ERROR,
+ BaseMessages.getString(PKG, key, parameters),
+ transformMeta));
+ }
+
+ private static void warning(
+ List<ICheckResult> remarks, TransformMeta transformMeta, String key,
String... parameters) {
+ remarks.add(
+ new CheckResult(
+ ICheckResult.TYPE_RESULT_WARNING,
+ BaseMessages.getString(PKG, key, parameters),
+ transformMeta));
+ }
+}
diff --git
a/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextOutputFormat.java
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextOutputFormat.java
new file mode 100644
index 0000000000..8318114a3e
--- /dev/null
+++
b/plugins/tech/ai/src/main/java/org/apache/hop/ai/transforms/embedtext/EmbedTextOutputFormat.java
@@ -0,0 +1,44 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.hop.ai.transforms.embedtext;
+
+/**
+ * How the embedding is put on the output row.
+ *
+ * <p>The constants deliberately do not override {@code toString()}. Generated
dialogs fill an enum
+ * combo with {@code toString()} and read it back with {@code Enum.valueOf},
which only accepts the
+ * constant name.
+ */
+public enum EmbedTextOutputFormat {
+ /** A String holding a JSON array, readable anywhere and accepted by
pgvector upsert. */
+ STRING,
+
+ /** A field of the Vector value type, which avoids rendering the numbers to
text and back. */
+ VECTOR;
+
+ public static EmbedTextOutputFormat fromString(String value) {
+ if (value == null || value.isEmpty()) {
+ return STRING;
+ }
+ for (EmbedTextOutputFormat format : values()) {
+ if (format.name().equalsIgnoreCase(value.trim())) {
+ return format;
+ }
+ }
+ return STRING;
+ }
+}
diff --git a/plugins/tech/ai/src/main/resources/dependencies.xml
b/plugins/tech/ai/src/main/resources/dependencies.xml
new file mode 100644
index 0000000000..07a6814096
--- /dev/null
+++ b/plugins/tech/ai/src/main/resources/dependencies.xml
@@ -0,0 +1,25 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<!--
+ ~ Licensed to the Apache Software Foundation (ASF) under one or more
+ ~ contributor license agreements. See the NOTICE file distributed with
+ ~ this work for additional information regarding copyright ownership.
+ ~ The ASF licenses this file to You under the Apache License, Version 2.0
+ ~ (the "License"); you may not use this file except in compliance with
+ ~ the License. You may obtain a copy of the License at
+ ~
+ ~ http://www.apache.org/licenses/LICENSE-2.0
+ ~
+ ~ Unless required by applicable law or agreed to in writing, software
+ ~ distributed under the License is distributed on an "AS IS" BASIS,
+ ~ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ ~ See the License for the specific language governing permissions and
+ ~ limitations under the License.
+ ~
+ -->
+<dependencies>
+ <!-- Language Model Chat ships the langchain4j libraries this plugin's
model factories are
+ built on. They were already reached through the shared hop-ai class
loader group, but only
+ once that plugin happened to be asked for. Naming the folder here
puts them on the loader
+ from the start, which is what a transform needs on its very first
row. -->
+ <folder>../../transforms/languagemodelchat/lib</folder>
+</dependencies>
diff --git a/plugins/tech/ai/src/main/resources/embedtext.svg
b/plugins/tech/ai/src/main/resources/embedtext.svg
new file mode 100644
index 0000000000..3f2017dc61
--- /dev/null
+++ b/plugins/tech/ai/src/main/resources/embedtext.svg
@@ -0,0 +1,10 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<svg width="24" height="24" viewBox="0 0 24 24" fill="none"
xmlns="http://www.w3.org/2000/svg">
+ <line x1="2" y1="6" x2="9" y2="6" stroke="#2A5D8F" stroke-width="1.6"
stroke-linecap="round"/>
+ <line x1="2" y1="10" x2="9" y2="10" stroke="#2A5D8F" stroke-width="1.6"
stroke-linecap="round"/>
+ <line x1="2" y1="14" x2="7" y2="14" stroke="#2A5D8F" stroke-width="1.6"
stroke-linecap="round"/>
+ <path d="M11 10 L15 10 M13.5 8 L15.5 10 L13.5 12" stroke="#2A5D8F"
stroke-width="1.6" stroke-linecap="round" stroke-linejoin="round"/>
+ <circle cx="19" cy="5" r="1.7" stroke="#2A5D8F" stroke-width="1.6"/>
+ <circle cx="19" cy="12" r="1.7" stroke="#2A5D8F" stroke-width="1.6"/>
+ <circle cx="19" cy="19" r="1.7" stroke="#2A5D8F" stroke-width="1.6"/>
+</svg>
diff --git
a/plugins/tech/ai/src/main/resources/org/apache/hop/ai/metadata/messages/messages_en_US.properties
b/plugins/tech/ai/src/main/resources/org/apache/hop/ai/metadata/messages/messages_en_US.properties
index 8f38836196..75471cb17f 100644
---
a/plugins/tech/ai/src/main/resources/org/apache/hop/ai/metadata/messages/messages_en_US.properties
+++
b/plugins/tech/ai/src/main/resources/org/apache/hop/ai/metadata/messages/messages_en_US.properties
@@ -35,6 +35,11 @@ AiProvider.ApiKey.Label=API key
AiProvider.ApiKey.Tooltip=API key or token. Do not paste a live key into this
metadata. Prefer '${AI_API_KEY}' from an environment configuration file, or a
keystore expression such as '#{vault:secret/data/ai:api-key}'. Resolved at send
time.
AiProvider.Timeout.Label=Timeout (seconds)
AiProvider.Timeout.Tooltip=Request timeout in seconds
+AiProviderEditor.Models.Label=Models per role
+AiProviderEditor.Models.Tooltip=Optional. A model for each role this provider
serves, so one provider can back a chat transform, an embedding transform and a
reranker. A transform picks the row for the role it needs. Leave empty to use
the model name above, which is the chat model.
+AiProviderEditor.Models.Column.Role=Role
+AiProviderEditor.Models.Column.ModelName=Model name
+
AiProvider.ModelName.Label=Model name
AiProvider.ModelName.Tooltip=Model identifier sent to the provider. Use
Refresh models to load names from the endpoint. You can still type a value such
as '${AI_MODEL}'.
AiProvider.Temperature.Label=Temperature
diff --git
a/plugins/tech/ai/src/main/resources/org/apache/hop/ai/transforms/embedtext/messages/messages_en_US.properties
b/plugins/tech/ai/src/main/resources/org/apache/hop/ai/transforms/embedtext/messages/messages_en_US.properties
new file mode 100644
index 0000000000..eb6d46a256
--- /dev/null
+++
b/plugins/tech/ai/src/main/resources/org/apache/hop/ai/transforms/embedtext/messages/messages_en_US.properties
@@ -0,0 +1,59 @@
+# Licensed to the Apache Software Foundation (ASF) under one or more
+# contributor license agreements. See the NOTICE file distributed with
+# this work for additional information regarding copyright ownership.
+# The ASF licenses this file to You under the Apache License, Version 2.0
+# (the "License"); you may not use this file except in compliance with
+# the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+EmbedText.Name=Embed text
+EmbedText.Description=Turn a text field into an embedding vector using an AI
provider
+
+EmbedTextDialog.Shell.Title=Embed text
+EmbedText.Tab.Main=Main
+
+EmbedText.aiProvider.Label=AI provider
+EmbedText.aiProvider.Tooltip=AI provider serving the embedding model. The
model comes from the provider's EMBEDDING entry.
+EmbedText.modelName.Label=Model
+EmbedText.modelName.Tooltip=Optional. Overrides the provider's EMBEDDING model
for this transform only.
+EmbedText.inputField.Label=Input field
+EmbedText.inputField.Tooltip=Field holding the text to embed. A row with empty
text is passed on with empty output fields.
+EmbedText.outputField.Label=Output field
+EmbedText.outputField.Tooltip=Field the embedding is written to
+EmbedText.outputFormat.Label=Output format
+EmbedText.outputFormat.Tooltip=STRING writes a JSON array, which pgvector
upsert accepts as a literal. VECTOR writes a Vector field, which avoids
rendering the numbers to text and back, and needs the Vector value type plugin.
+EmbedText.batchSize.Label=Batch size
+EmbedText.batchSize.Tooltip=Rows buffered before the provider is called. Rows
whose text is empty are buffered too, so they keep their place in the stream,
but they are not sent. Embedding calls are billed and rate limited per request,
so batching matters.
+EmbedText.includeModelMetadata.Label=Include model metadata
+EmbedText.includeModelMetadata.Tooltip=Add the model name and the vector width
to each row, so a later re-index can tell which model produced an embedding
+EmbedText.modelField.Label=Model field
+EmbedText.modelField.Tooltip=Output field for the embedding model name
+EmbedText.dimensionsField.Label=Dimensions field
+EmbedText.dimensionsField.Tooltip=Output field for the number of dimensions in
the vector
+
+EmbedText.Validation.ProviderRequired=An AI provider is required
+EmbedText.Validation.InputFieldRequired=An input field is required
+EmbedText.Validation.InputFieldNotFound=Input field ''{0}'' not found in the
input stream
+EmbedText.Validation.OutputFieldRequired=An output field is required
+EmbedText.Validation.BatchSizePositive=Batch size must be at least 1
+EmbedText.Validation.NoMetadataFields=Include model metadata is on, but
neither the model field nor the dimensions field has a name, so nothing is added
+
+EmbedText.Error.Embedding=Error calling embedding model {0}
+EmbedText.Error.BatchSizeMismatch=The provider returned {0} embeddings for the
{1} text(s) it was sent
+
+EmbedTextMeta.Injection.AI_PROVIDER=AI provider serving the embedding model
+EmbedTextMeta.Injection.MODEL_NAME=Embedding model name, overriding the
provider's
+EmbedTextMeta.Injection.INPUT_FIELD=Field holding the text to embed
+EmbedTextMeta.Injection.OUTPUT_FIELD=Field the embedding is written to
+EmbedTextMeta.Injection.OUTPUT_FORMAT=STRING or VECTOR
+EmbedTextMeta.Injection.BATCH_SIZE=Rows sent to the provider in one call
+EmbedTextMeta.Injection.INCLUDE_MODEL_METADATA=Add the model name and vector
width to each row
+EmbedTextMeta.Injection.MODEL_FIELD=Output field for the embedding model name
+EmbedTextMeta.Injection.DIMENSIONS_FIELD=Output field for the number of
dimensions
diff --git
a/plugins/tech/ai/src/test/java/org/apache/hop/ai/metadata/AiProviderTest.java
b/plugins/tech/ai/src/test/java/org/apache/hop/ai/metadata/AiProviderTest.java
index 642d191eee..1ee37f03e6 100644
---
a/plugins/tech/ai/src/test/java/org/apache/hop/ai/metadata/AiProviderTest.java
+++
b/plugins/tech/ai/src/test/java/org/apache/hop/ai/metadata/AiProviderTest.java
@@ -28,6 +28,73 @@ import org.junit.jupiter.api.Test;
class AiProviderTest {
+ @Test
+ void aProviderWithoutRolesBehavesExactlyAsBefore() {
+ // The compatibility guarantee: anything serialized before roles existed
has no models entry,
+ // so CHAT still comes from modelName and no other role resolves.
+ AiProvider provider = new AiProvider();
+ provider.setModelName("gpt-4o-mini");
+
+ assertTrue(provider.getModels().isEmpty());
+ assertEquals("gpt-4o-mini", provider.resolveModelName(AiModelRole.CHAT));
+ assertEquals("", provider.resolveModelName(AiModelRole.EMBEDDING));
+ }
+
+ @Test
+ void eachRoleResolvesItsOwnModel() {
+ AiProvider provider = new AiProvider();
+ provider.setModelName("gpt-4o-mini");
+ provider.setModels(
+ List.of(
+ new AiProviderModel(AiModelRole.CHAT, "llama3.2"),
+ new AiProviderModel(AiModelRole.EMBEDDING, "nomic-embed-text"),
+ new AiProviderModel(AiModelRole.SCORING, "bge-reranker-v2-m3")));
+
+ assertEquals("llama3.2", provider.resolveModelName(AiModelRole.CHAT));
+ assertEquals("nomic-embed-text",
provider.resolveModelName(AiModelRole.EMBEDDING));
+ assertEquals("bge-reranker-v2-m3",
provider.resolveModelName(AiModelRole.SCORING));
+ assertEquals("", provider.resolveModelName(AiModelRole.IMAGE));
+ }
+
+ @Test
+ void theChatModelIsNotAFallbackForTheOtherRoles() {
+ // Embedding a text with a chat model fails at the provider with a
confusing message. An empty
+ // result here lets the caller say what is actually wrong.
+ AiProvider provider = new AiProvider();
+ provider.setModelName("gpt-4o-mini");
+ provider.setModels(List.of(new AiProviderModel(AiModelRole.SCORING,
"bge-reranker-v2-m3")));
+
+ assertEquals("", provider.resolveModelName(AiModelRole.EMBEDDING));
+ }
+
+ @Test
+ void anEmptyModelNameOnARoleDoesNotCount() {
+ AiProvider provider = new AiProvider();
+ provider.setModelName("gpt-4o-mini");
+ provider.setModels(List.of(new AiProviderModel(AiModelRole.CHAT, "")));
+
+ assertEquals("gpt-4o-mini", provider.resolveModelName(AiModelRole.CHAT));
+ }
+
+ @Test
+ void copyConstructorDeepCopiesTheModels() {
+ AiProvider source = new AiProvider();
+ source.setModels(List.of(new AiProviderModel(AiModelRole.EMBEDDING,
"nomic-embed-text")));
+
+ AiProvider copy = new AiProvider(source);
+ copy.getModels().get(0).setModelName("changed");
+
+ assertEquals("nomic-embed-text",
source.resolveModelName(AiModelRole.EMBEDDING));
+ }
+
+ @Test
+ void theRoleEnumReadsBackFromItsDisplayedText() {
+ // Generated dialogs fill an enum combo with toString() and read it back
with Enum.valueOf.
+ for (AiModelRole role : AiModelRole.values()) {
+ assertEquals(role.name(), role.toString());
+ }
+ }
+
@Test
void copyConstructorClonesProvider() {
AiProvider source = new AiProvider();
diff --git
a/plugins/tech/ai/src/test/java/org/apache/hop/ai/transforms/embedtext/EmbedTextMetaTest.java
b/plugins/tech/ai/src/test/java/org/apache/hop/ai/transforms/embedtext/EmbedTextMetaTest.java
new file mode 100644
index 0000000000..06a3041d47
--- /dev/null
+++
b/plugins/tech/ai/src/test/java/org/apache/hop/ai/transforms/embedtext/EmbedTextMetaTest.java
@@ -0,0 +1,241 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.hop.ai.transforms.embedtext;
+
+import static org.junit.jupiter.api.Assertions.assertDoesNotThrow;
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+import java.lang.reflect.Field;
+import java.util.ArrayList;
+import java.util.List;
+import org.apache.hop.core.HopClientEnvironment;
+import org.apache.hop.core.ICheckResult;
+import org.apache.hop.core.annotations.Transform;
+import org.apache.hop.core.gui.plugin.GuiWidgetElement;
+import org.apache.hop.core.row.IRowMeta;
+import org.apache.hop.core.row.IValueMeta;
+import org.apache.hop.core.row.RowMeta;
+import org.apache.hop.core.row.value.ValueMetaString;
+import org.apache.hop.core.util.TranslateUtil;
+import org.apache.hop.core.variables.Variables;
+import org.apache.hop.core.xml.XmlHandler;
+import org.apache.hop.metadata.api.HopMetadataProperty;
+import org.apache.hop.metadata.serializer.xml.XmlMetadataUtil;
+import org.apache.hop.pipeline.transform.TransformMeta;
+import org.junit.jupiter.api.BeforeAll;
+import org.junit.jupiter.api.Test;
+import org.w3c.dom.Document;
+
+class EmbedTextMetaTest {
+
+ @BeforeAll
+ static void setUpClass() throws Exception {
+ HopClientEnvironment.init();
+ }
+
+ @Test
+ void survivesAnXmlRoundTrip() throws Exception {
+ EmbedTextMeta original = completeMeta();
+ original.setOutputFormat(EmbedTextOutputFormat.VECTOR);
+ original.setBatchSize("32");
+
+ EmbedTextMeta copy = roundTrip(original);
+
+ assertEquals("ollama-local", copy.getAiProvider());
+ assertEquals(EmbedTextOutputFormat.VECTOR, copy.getOutputFormat());
+ assertEquals("32", copy.getBatchSize());
+ assertEquals("chunk_text", copy.getInputField());
+ }
+
+ @Test
+ void addsTheEmbeddingAndItsMetadataToTheOutputRow() throws Exception {
+ EmbedTextMeta meta = completeMeta();
+ IRowMeta row = inputFields();
+
+ meta.getFields(row, "Embed", null, null, new Variables(), null);
+
+ assertEquals(4, row.size(), "one embedding field plus the two metadata
fields");
+ assertEquals("embedding", row.getValueMeta(1).getName());
+ assertEquals("embedding_model", row.getValueMeta(2).getName());
+ assertEquals(IValueMeta.TYPE_INTEGER, row.getValueMeta(3).getType());
+ }
+
+ @Test
+ void addsOnlyTheEmbeddingWhenMetadataIsOff() throws Exception {
+ EmbedTextMeta meta = completeMeta();
+ meta.setIncludeModelMetadata(false);
+ IRowMeta row = inputFields();
+
+ meta.getFields(row, "Embed", null, null, new Variables(), null);
+
+ assertEquals(2, row.size());
+ }
+
+ @Test
+ void fallsBackToAStringWhenTheVectorTypeIsNotInstalled() throws Exception {
+ // The Vector value type is an optional plugin, and this module does not
depend on it. Without
+ // it the embedding still has to reach the stream, as a JSON array.
+ EmbedTextMeta meta = completeMeta();
+ meta.setOutputFormat(EmbedTextOutputFormat.VECTOR);
+ IRowMeta row = inputFields();
+
+ meta.getFields(row, "Embed", null, null, new Variables(), null);
+
+ assertEquals(IValueMeta.TYPE_STRING, row.getValueMeta(1).getType());
+ }
+
+ @Test
+ void acceptsAVariableForBatchSize() {
+ EmbedTextMeta meta = completeMeta();
+ meta.setBatchSize("${EMBED_BATCH_SIZE}");
+
+ assertTrue(
+ errorText(check(meta)).isEmpty(), "a variable is resolved at run time,
not a design error");
+ }
+
+ @Test
+ void stillRejectsAFixedBatchSizeBelowOne() {
+ EmbedTextMeta meta = completeMeta();
+ meta.setBatchSize("0");
+
+ assertTrue(errorText(check(meta)).contains("Batch size"),
errorText(check(meta)));
+ }
+
+ @Test
+ void reportsAMissingProviderAndInputField() {
+ EmbedTextMeta meta = new EmbedTextMeta();
+ meta.setDefault();
+ meta.setInputField("not_in_the_stream");
+
+ String errors = errorText(check(meta));
+
+ assertTrue(errors.contains("AI provider is required"), errors);
+ assertTrue(errors.contains("not found"), errors);
+ }
+
+ @Test
+ void supportsAnErrorHop() {
+ // processRow calls putError, which Hop Gui only offers when this says so.
+ assertTrue(new EmbedTextMeta().supportsErrorHandling());
+ }
+
+ @Test
+ void everyPropertyIsOnTheDialog() {
+ for (Field field : EmbedTextMeta.class.getDeclaredFields()) {
+ if (field.getAnnotation(HopMetadataProperty.class) == null) {
+ continue;
+ }
+ assertTrue(
+ field.getAnnotation(GuiWidgetElement.class) != null,
+ "Property "
+ + field.getName()
+ + " is persisted but has no @GuiWidgetElement, so it is missing
from the dialog");
+ }
+ }
+
+ @Test
+ void everyWidgetLabelTooltipAndTabResolves() {
+ for (Field field : EmbedTextMeta.class.getDeclaredFields()) {
+ GuiWidgetElement widget = field.getAnnotation(GuiWidgetElement.class);
+ if (widget == null) {
+ continue;
+ }
+ assertResolves(field.getName(), "label", widget.label());
+ assertResolves(field.getName(), "toolTip", widget.toolTip());
+ assertResolves(field.getName(), "group", widget.group());
+ }
+ }
+
+ @Test
+ void everyEnumOnTheDialogReadsBackFromItsDisplayedText() {
+ // Generated dialogs fill an enum combo with toString() and read it back
with Enum.valueOf.
+ for (EmbedTextOutputFormat format : EmbedTextOutputFormat.values()) {
+ assertEquals(format.name(), format.toString());
+ }
+ }
+
+ @Test
+ void aVariableThatResolvesToNothingAddsNoColumn() {
+ // The transform writes the metadata slots only when the resolved name is
non-empty. If
+ // getFields judged the raw value instead, the two would disagree and a
String would land in
+ // the Integer dimensions column.
+ EmbedTextMeta meta = completeMeta();
+ meta.setModelField("${EMPTY_MODEL_FIELD}");
+ Variables variables = new Variables();
+ variables.setVariable("EMPTY_MODEL_FIELD", "");
+ IRowMeta row = inputFields();
+
+ assertDoesNotThrow(() -> meta.getFields(row, "Embed", null, null,
variables, null));
+
+ assertEquals(3, row.size(), "embedding and dimensions, but no model
column");
+ assertEquals("embedding_dimensions", row.getValueMeta(2).getName());
+ }
+
+ @Test
+ void theCategoryResolves() {
+ // A dangling categoryDescription puts the transform under a raw key in
the Hop Gui tree.
+ Transform annotation = EmbedTextMeta.class.getAnnotation(Transform.class);
+ String category =
+ TranslateUtil.translate(annotation.categoryDescription(),
EmbedTextMeta.class);
+ assertFalse(
+ category.startsWith("!") && category.endsWith("!"),
+ "the transform category does not resolve: " + category);
+ }
+
+ private static void assertResolves(String fieldName, String what, String
value) {
+ String translated = TranslateUtil.translate(value, EmbedTextMeta.class);
+ assertFalse(
+ translated.startsWith("!") && translated.endsWith("!"),
+ "The " + what + " of " + fieldName + " does not resolve: " +
translated);
+ }
+
+ private static EmbedTextMeta completeMeta() {
+ EmbedTextMeta meta = new EmbedTextMeta();
+ meta.setDefault();
+ meta.setAiProvider("ollama-local");
+ return meta;
+ }
+
+ private static IRowMeta inputFields() {
+ IRowMeta row = new RowMeta();
+ row.addValueMeta(new ValueMetaString("chunk_text"));
+ return row;
+ }
+
+ private static List<ICheckResult> check(EmbedTextMeta meta) {
+ List<ICheckResult> remarks = new ArrayList<>();
+ meta.check(
+ remarks, null, new TransformMeta(), inputFields(), null, null, null,
new Variables(), null);
+ return remarks;
+ }
+
+ private static String errorText(List<ICheckResult> remarks) {
+ return remarks.stream()
+ .filter(r -> r.getType() == ICheckResult.TYPE_RESULT_ERROR)
+ .map(ICheckResult::getText)
+ .collect(java.util.stream.Collectors.joining(" | "));
+ }
+
+ private static EmbedTextMeta roundTrip(EmbedTextMeta original) throws
Exception {
+ String xml = "<transform>" +
XmlMetadataUtil.serializeObjectToXml(original) + "</transform>";
+ Document document = XmlHandler.loadXmlString(xml);
+ return XmlMetadataUtil.deSerializeFromXml(
+ XmlHandler.getSubNode(document, "transform"), EmbedTextMeta.class,
null);
+ }
+}
diff --git
a/plugins/tech/ai/src/test/java/org/apache/hop/ai/transforms/embedtext/EmbedTextTest.java
b/plugins/tech/ai/src/test/java/org/apache/hop/ai/transforms/embedtext/EmbedTextTest.java
new file mode 100644
index 0000000000..ba2200e5c1
--- /dev/null
+++
b/plugins/tech/ai/src/test/java/org/apache/hop/ai/transforms/embedtext/EmbedTextTest.java
@@ -0,0 +1,206 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.hop.ai.transforms.embedtext;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertNull;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+import static org.mockito.ArgumentMatchers.any;
+import static org.mockito.ArgumentMatchers.anyLong;
+import static org.mockito.Mockito.doAnswer;
+import static org.mockito.Mockito.mock;
+import static org.mockito.Mockito.spy;
+import static org.mockito.Mockito.when;
+
+import dev.langchain4j.data.embedding.Embedding;
+import dev.langchain4j.data.segment.TextSegment;
+import dev.langchain4j.model.embedding.EmbeddingModel;
+import dev.langchain4j.model.output.Response;
+import java.util.ArrayList;
+import java.util.Arrays;
+import java.util.Iterator;
+import java.util.List;
+import org.apache.hop.core.HopClientEnvironment;
+import org.apache.hop.core.exception.HopException;
+import org.apache.hop.core.row.IRowMeta;
+import org.apache.hop.core.row.RowMeta;
+import org.apache.hop.core.row.value.ValueMetaString;
+import org.apache.hop.pipeline.transforms.mock.TransformMockHelper;
+import org.junit.jupiter.api.AfterEach;
+import org.junit.jupiter.api.BeforeAll;
+import org.junit.jupiter.api.BeforeEach;
+import org.junit.jupiter.api.Test;
+
+class EmbedTextTest {
+
+ private TransformMockHelper<EmbedTextMeta, EmbedTextData> helper;
+ private List<Object[]> passed;
+ private List<Object[]> diverted;
+ private boolean failTheProvider;
+
+ @BeforeAll
+ static void setUpClass() throws Exception {
+ HopClientEnvironment.init();
+ }
+
+ @BeforeEach
+ void setUp() {
+ helper = new TransformMockHelper<>("EmbedText", EmbedTextMeta.class,
EmbedTextData.class);
+ when(helper.logChannelFactory.create(any(),
any())).thenReturn(helper.iLogChannel);
+ when(helper.pipeline.isRunning()).thenReturn(true);
+ passed = new ArrayList<>();
+ diverted = new ArrayList<>();
+ failTheProvider = false;
+ }
+
+ @AfterEach
+ void tearDown() {
+ helper.cleanUp();
+ }
+
+ @Test
+ void keepsInputOrderWhenSomeRowsHaveNoTextToEmbed() throws Exception {
+ // A row with no text needs no provider call, but emitting it straight
away would let it
+ // overtake the rows already waiting in the batch.
+ run(Arrays.asList(new Object[] {"first"}, new Object[] {""}, new Object[]
{"third"}));
+
+ assertEquals(3, passed.size());
+ assertEquals("first", passed.get(0)[0]);
+ assertEquals("", passed.get(1)[0]);
+ assertEquals("third", passed.get(2)[0]);
+ }
+
+ @Test
+ void embedsOnlyTheRowsThatHaveText() throws Exception {
+ run(Arrays.asList(new Object[] {"first"}, new Object[] {""}, new Object[]
{"third"}));
+
+ // The embedding column is index 1; the empty row must not get one.
+ assertEquals("[1.0,2.0]", passed.get(0)[1]);
+ assertNull(passed.get(1)[1], "a row with no text gets no embedding");
+ assertEquals("[1.0,2.0]", passed.get(2)[1]);
+ }
+
+ @Test
+ void passesEveryRowOnEvenWhenNoneHaveText() throws Exception {
+ run(Arrays.asList(new Object[] {""}, new Object[] {""}));
+
+ assertEquals(2, passed.size(), "rows with nothing to embed are not
dropped");
+ }
+
+ @Test
+ void divertsOnlyTheRowsThatWereActuallySent() throws Exception {
+ // A row with no text is not in the request, so a provider failure is not
its failure.
+ when(helper.transformMeta.isDoingErrorHandling()).thenReturn(true);
+ failTheProvider = true;
+
+ run(Arrays.asList(new Object[] {"first"}, new Object[] {""}, new Object[]
{"third"}));
+
+ assertEquals(1, passed.size(), "the row with no text still reaches the
output");
+ assertEquals("", passed.get(0)[0]);
+ assertEquals(2, diverted.size(), "only the two rows that were sent are
diverted");
+ }
+
+ @Test
+ void failsWhenTheOutputFieldNameResolvesToNothing() {
+ // getFields adds no embedding column for an empty name, so writing one
would run off the row.
+ EmbedTextMeta meta = new EmbedTextMeta();
+ meta.setDefault();
+ meta.setAiProvider("ollama");
+ meta.setOutputField("${EMPTY_OUTPUT_FIELD}");
+
+ EmbedText transform =
+ new EmbedText(
+ helper.transformMeta,
+ meta,
+ new EmbedTextData(),
+ 0,
+ helper.pipelineMeta,
+ helper.pipeline);
+ transform.setVariable("EMPTY_OUTPUT_FIELD", "");
+
+ HopException e =
+ assertThrows(HopException.class, () ->
transform.resolveOutputFieldNamesForTesting());
+ assertTrue(e.getMessage().contains("output field"), e.getMessage());
+ }
+
+ private void run(List<Object[]> rows) throws Exception {
+ EmbedTextMeta meta = new EmbedTextMeta();
+ meta.setDefault();
+ meta.setAiProvider("ollama");
+ meta.setInputField("chunk_text");
+ meta.setIncludeModelMetadata(false);
+
+ IRowMeta inputRowMeta = new RowMeta();
+ inputRowMeta.addValueMeta(new ValueMetaString("chunk_text"));
+
+ EmbedTextData data = new EmbedTextData();
+ data.inputRowMeta = inputRowMeta;
+ data.outputRowMeta = inputRowMeta.clone();
+ data.outputRowMeta.addValueMeta(new ValueMetaString("embedding"));
+ data.inputFieldIndex = 0;
+ data.batchSize = 16;
+ data.modelName = "test-model";
+ data.outputFieldName = "embedding";
+
+ EmbeddingModel model = mock(EmbeddingModel.class);
+ when(model.embedAll(any()))
+ .thenAnswer(
+ invocation -> {
+ if (failTheProvider) {
+ throw new RuntimeException("provider is down");
+ }
+ List<TextSegment> segments = invocation.getArgument(0);
+ List<Embedding> embeddings = new ArrayList<>();
+ for (int i = 0; i < segments.size(); i++) {
+ embeddings.add(Embedding.from(new float[] {1.0f, 2.0f}));
+ }
+ return Response.from(embeddings);
+ });
+ data.model = model;
+
+ EmbedText transform =
+ spy(
+ new EmbedText(
+ helper.transformMeta, meta, data, 0, helper.pipelineMeta,
helper.pipeline));
+ transform.init();
+ transform.setInputRowMeta(inputRowMeta);
+ // The model and the field indexes are already set up, so skip the
first-row branch.
+ transform.first = false;
+
+ Iterator<Object[]> iterator = rows.iterator();
+ doAnswer(invocation -> iterator.hasNext() ? iterator.next() :
null).when(transform).getRow();
+ doAnswer(
+ invocation -> {
+ passed.add(invocation.getArgument(1));
+ return null;
+ })
+ .when(transform)
+ .putRow(any(IRowMeta.class), any(Object[].class));
+ doAnswer(
+ invocation -> {
+ diverted.add(invocation.getArgument(1));
+ return null;
+ })
+ .when(transform)
+ .putError(any(), any(), anyLong(), any(), any(), any());
+
+ while (transform.processRow()) {
+ // drain
+ }
+ }
+}