JingsongLi commented on code in PR #9754:
URL: https://github.com/apache/paimon/pull/9754#discussion_r3996196212


##########
paimon-python/pypaimon/table/source/vector_search_read.py:
##########
@@ -751,41 +781,31 @@ def _table_options_map(table):
     return table_options.to_map() if table_options is not None else {}
 
 
-def _raw_search_metric(table, vector_column, options, index_type=None):
-    candidates = []
+def _configured_vector_metric(options, vector_column, index_type=None):
     field_prefix = "fields.%s." % vector_column.name
     index_prefix = "%s." % index_type if index_type else None
-    for key in [
-        field_prefix + "distance.metric",
-        field_prefix + "metric",
-        *(([
-            index_prefix + "distance.metric",
-            index_prefix + "metric",
-        ]) if index_prefix is not None else []),
-        "test.vector.metric",
-        "lumina.distance.metric",
-        "distance.metric",
-        "metric",
-    ]:
+    keys = [field_prefix + "distance.metric", field_prefix + "metric"]
+    if index_prefix is not None:
+        keys.extend([index_prefix + "distance.metric", index_prefix + 
"metric"])
+    keys.extend(["test.vector.metric", "lumina.distance.metric", 
"distance.metric", "metric"])
+    for key in keys:
         if key in options:
-            candidates.append(options[key])
+            return _normalize_metric(options[key])
+    return None
+
+
+def _raw_search_metric(table, vector_column, options, index_type=None):
+    from pypaimon.globalindex.vindex.vindex_vector_global_index_reader import 
VINDEX_IDENTIFIERS
+
     table_map = _table_options_map(table)
-    for key in [
-        field_prefix + "distance.metric",
-        field_prefix + "metric",
-        *(([
-            index_prefix + "distance.metric",
-            index_prefix + "metric",
-        ]) if index_prefix is not None else []),
-        "test.vector.metric",
-        "lumina.distance.metric",
-        "distance.metric",
-        "metric",
-    ]:
-        if key in table_map:
-            candidates.append(table_map[key])
-    if candidates:
-        return _normalize_metric(candidates[0])
+    for source in (options, table_map):
+        configured = _configured_vector_metric(source, vector_column, 
index_type)
+        if configured is not None:
+            return configured
+
+    # Before an index exists, use its writer's default, not another column's 
metric.
+    if index_type in VINDEX_IDENTIFIERS:
+        return "inner_product"
 

Review Comment:
   [P1] Resolve the primary-key column metric before applying the vindex default
   
   `_configured_vector_metric` does not recognize 
`fields.<column>.pk-vector.distance.metric`. Previously the suffix fallback 
below found that documented option; this new early return instead chooses 
`inner_product`. In full-mode primary-key searches with no usable payloads, 
`PrimaryKeyVectorRead._raw_candidates` reaches this path with `_index_metric` 
unset even when `CoreOptions.primary_key_vector_distance_metric()` is `l2` or 
`cosine`, so it can return the wrong Top-K. With the Java golden data, 
canonical L2 configuration, query `[0.1, 0, 0, 0]` and limit 1, the base 
resolver selects id 2 (score 0.60606062), while this revision selects id 1 
(inner-product score 0.1). Resolve the canonical PK metric for the selected 
column before the default, and add a raw-only PK regression without the 
fixture's legacy `ivf-flat.metric` option.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to