Akash3121 commented on code in PR #10163:
URL: https://github.com/apache/paimon/pull/10163#discussion_r4096214016
##########
paimon-python/pypaimon/cli/cli_table.py:
##########
@@ -314,6 +314,115 @@ def cmd_table_full_text_search(args):
print(df.to_string(index=False))
+def _parse_query_vector(raw):
+ """
+ Parse a query vector supplied on the command line.
+
+ Accepts either a JSON-style array (e.g. ``[0.1, 0.2, 0.3]``) or a bare
+ comma-separated list (e.g. ``0.1,0.2,0.3``) and returns a list of floats.
+
+ Args:
+ raw: The raw ``--query`` argument value.
+
+ Raises:
+ ValueError: If the value is empty or contains a non-numeric element.
+ """
+ text = raw.strip()
+ if text.startswith('[') and text.endswith(']'):
+ text = text[1:-1]
+ parts = [p.strip() for p in text.split(',') if p.strip() != '']
+ if not parts:
+ raise ValueError(
+ 'Query vector is empty; provide comma-separated floats, '
+ 'e.g. --query "0.1,0.2,0.3"')
+ try:
+ return [float(p) for p in parts]
+ except ValueError:
+ raise ValueError(
+ 'Query vector must contain only numbers; got: %r' % raw) from None
+
+
+def cmd_table_vector_search(args):
+ """
+ Execute the 'table vector-search' command.
+
+ Performs nearest-neighbor vector search on a Paimon table and displays the
+ matching rows.
+
+ Args:
+ args: Parsed command line arguments.
+ """
+ from pypaimon.cli.cli import load_catalog_config, create_catalog
+
+ config_path = args.config
+ config = load_catalog_config(config_path)
+ catalog = create_catalog(config)
+
+ table_identifier = args.table
+ parts = table_identifier.split('.')
+ if len(parts) != 2:
+ print(f"Error: Invalid table identifier '{table_identifier}'. "
+ f"Expected format: 'database.table'", file=sys.stderr)
+ sys.exit(1)
+
+ database_name, table_name = parts
+
+ try:
+ table = catalog.get_table(f"{database_name}.{table_name}")
+ except Exception as e:
+ print(f"Error: Failed to get table '{table_identifier}': {e}",
file=sys.stderr)
+ sys.exit(1)
+
+ try:
+ query_vector = _parse_query_vector(args.query)
+ except ValueError as e:
+ print(f"Error: {e}", file=sys.stderr)
+ sys.exit(1)
+
+ limit = args.limit
+
+ try:
+ builder = table.new_vector_search_builder()
+ builder.with_vector_column(args.column)
+ builder.with_query_vector(query_vector)
+ builder.with_limit(limit)
+ result = builder.execute_local()
+ except Exception as e:
+ print(f"Error: Vector search failed: {e}", file=sys.stderr)
+ sys.exit(1)
+
+ if result.is_empty():
+ print("No matching rows found.")
+ return
+
+ # Read matching rows using global index result
+ read_builder = table.new_read_builder()
+
+ select_columns = args.select
+ if select_columns:
+ projection = [col.strip() for col in select_columns.split(',')]
+ available_fields = set(field.name for field in
table.table_schema.fields)
+ invalid_columns = [col for col in projection if col not in
available_fields]
+ if invalid_columns:
+ print(f"Error: Column(s) {invalid_columns} do not exist in table
'{table_identifier}'.",
+ file=sys.stderr)
+ sys.exit(1)
+ read_builder = read_builder.with_projection(projection)
+
+ scan = read_builder.new_scan().with_global_index_result(result)
+ plan = scan.plan()
+ splits = plan.splits()
+ read = read_builder.new_read()
+ df = read.to_pandas(splits)
+
+ output_format = getattr(args, 'format', 'table')
+ if output_format == 'json':
Review Comment:
`--format json` is not usable with the default vector-search projection.
Arrow list/vector columns become `numpy.ndarray` values in pandas, and
`json.dumps(df.to_dict(...))` raises `TypeError: Object of type ndarray is
not JSON serializable` . Since the searched vector column is included by
default, this is the normal JSON path rather than an edge case. Please
serialize from Arrow/Python-native values (for example `to_pylist()` ) or
normalize NumPy arrays/scalars before calling `json.dumps`, and add a
successful JSON-output test that includes the embedding column.
Validation: Direct parser probes passed. I reproduced both the ranking loss
( `bitmap_order=[2,10]` , `score_order=[10,2]` ) and the vector JSON
`ndarray` failure. The catalog-backed CLI tests cannot initialize locally
because of the repository’s Windows URI issue ( `Unrecognized filesystem type
in URI: c` ). GitHub’s failing Python jobs are unrelated and all report the
existing `interval_partition_test.py` `SimpleNamespace.path_factory` error.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]