This is an automated email from the ASF dual-hosted git repository.
github-merge-queue[bot] pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/texera.git
The following commit(s) were added to refs/heads/main by this push:
new 7a79db153e fix(workflow-operator): raise ValueError for HF config
checks instead of assert (stripped under -O) (#7297)
7a79db153e is described below
commit 7a79db153e9aeaafd6120805838f56e05c702ae8
Author: Prateek Ganigi <[email protected]>
AuthorDate: Wed Aug 5 12:24:11 2026 -0700
fix(workflow-operator): raise ValueError for HF config checks instead of
assert (stripped under -O) (#7297)
### What changes were proposed in this PR?
The HuggingFace operator's pre-loop config validations (prompt column,
zero-shot candidate labels, question-answering context column,
sentence-similarity/text-ranking sentences column) used Python `assert`.
That has two problems: `assert` is **stripped entirely under `python
-O`**, silently disabling the checks, and it raises `AssertionError`
instead of the `ValueError` used elsewhere in the operator.
This converts the four `assert X, (msg)` checks to `if not X: raise
ValueError(msg)`. Behavior is otherwise identical, same messages, same
fail-fast on misconfiguration, but the checks now survive `-O` and raise
a consistent error type.
This is the `assert`->`ValueError` half of #7199. The other half is
`zero-shot-image-classification` raising a `ValueError` inside the
per-row loop (which crashes the operator rather than writing a per-row
error), is left as a follow-up, since it needs a structural change and a
separate design decision.
### Any related issues, documentation, discussions?
Addresses #7199 (the `assert`->`ValueError` part; the zero-shot
per-row-crash follow-up remains).
### How was this PR tested?
`sbt "WorkflowOperator/testOnly
org.apache.texera.amber.operator.huggingFace.*
org.apache.texera.amber.util.PythonCodeRawInvalidTextSpec"`: passes (126
tests). Added a test asserting the generated script contains no `assert
` and uses the explicit `raise ValueError` form;
`PythonCodeRawInvalidTextSpec` py-compiles the generated Python (guards
the multi-line indentation change). scalafmt clean.
### Was this PR authored or co-authored using generative AI tooling?
This PR was co-authored with Claude Opus 4.8 in compliance with ASF
policy.
---
.../codegen/HuggingFaceCodegenBase.scala | 36 ++++++++++++----------
.../HuggingFaceInferenceOpDescSpec.scala | 11 +++++++
2 files changed, 31 insertions(+), 16 deletions(-)
diff --git
a/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/huggingFace/codegen/HuggingFaceCodegenBase.scala
b/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/huggingFace/codegen/HuggingFaceCodegenBase.scala
index 7db809a07a..9b59ce6d92 100644
---
a/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/huggingFace/codegen/HuggingFaceCodegenBase.scala
+++
b/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/huggingFace/codegen/HuggingFaceCodegenBase.scala
@@ -504,28 +504,32 @@ object HuggingFaceCodegenBase {
|
| # --- validate prompt column exists (skipped for image tasks
and binary-only audio tasks) ---
| if task not in image_tasks and task not in audio_only_tasks:
- | assert prompt_col in table.columns, (
- | f"Prompt column '{prompt_col}' not found in input
table. "
- | f"Available columns: {list(table.columns)}"
- | )
+ | if prompt_col not in table.columns:
+ | raise ValueError(
+ | f"Prompt column '{prompt_col}' not found in input
table. "
+ | f"Available columns: {list(table.columns)}"
+ | )
| if task == "zero-shot-classification":
| labels = [l.strip() for l in
str(self.CANDIDATE_LABELS).split(",") if l.strip()]
- | assert labels, (
- | "Candidate Labels are required for
zero-shot-classification. "
- | "Provide a comma-separated list of labels."
- | )
+ | if not labels:
+ | raise ValueError(
+ | "Candidate Labels are required for
zero-shot-classification. "
+ | "Provide a comma-separated list of labels."
+ | )
| if task == "question-answering":
| ctx_col = self.CONTEXT_COLUMN
- | assert ctx_col and ctx_col in table.columns, (
- | f"Context column '{ctx_col}' not found in input table.
"
- | f"Available columns: {list(table.columns)}"
- | )
+ | if not (ctx_col and ctx_col in table.columns):
+ | raise ValueError(
+ | f"Context column '{ctx_col}' not found in input
table. "
+ | f"Available columns: {list(table.columns)}"
+ | )
| if task in ("sentence-similarity", "text-ranking"):
| sent_col = self.SENTENCES_COLUMN
- | assert sent_col and sent_col in table.columns, (
- | f"Sentences column '{sent_col}' not found in input
table. "
- | f"Available columns: {list(table.columns)}"
- | )
+ | if not (sent_col and sent_col in table.columns):
+ | raise ValueError(
+ | f"Sentences column '{sent_col}' not found in input
table. "
+ | f"Available columns: {list(table.columns)}"
+ | )
|
| # --- handle empty table ---
| if table.empty:
diff --git
a/common/workflow-operator/src/test/scala/org/apache/texera/amber/operator/huggingFace/HuggingFaceInferenceOpDescSpec.scala
b/common/workflow-operator/src/test/scala/org/apache/texera/amber/operator/huggingFace/HuggingFaceInferenceOpDescSpec.scala
index d34a6e8fea..b6b0355435 100644
---
a/common/workflow-operator/src/test/scala/org/apache/texera/amber/operator/huggingFace/HuggingFaceInferenceOpDescSpec.scala
+++
b/common/workflow-operator/src/test/scala/org/apache/texera/amber/operator/huggingFace/HuggingFaceInferenceOpDescSpec.scala
@@ -649,6 +649,17 @@ class HuggingFaceInferenceOpDescSpec extends AnyFlatSpec
with Matchers {
outSchema.getAttributeNames.contains("hf_response") shouldBe true
}
+ it should "validate config with raise ValueError rather than assert (which
python -O strips)" in {
+ val code = makeDesc().generatePythonCode()
+ // No `assert` in the generated script — asserts are removed under `python
-O`,
+ // silently disabling the checks, and raise AssertionError rather than
ValueError.
+ code should not include ("assert ")
+ // The pre-loop config checks now raise ValueError explicitly.
+ code should include("if prompt_col not in table.columns:")
+ code should include("if not (ctx_col and ctx_col in table.columns):")
+ code should include("if not (sent_col and sent_col in table.columns):")
+ }
+
it should "validate base64 in the binary-column fallback so plain text isn't
decoded to garbage" in {
val code = makeDesc().generatePythonCode()
// validate=True makes b64decode reject non-base64 input, so real text
falls