This is an automated email from the ASF dual-hosted git repository.

airborne12 pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/doris.git


The following commit(s) were added to refs/heads/master by this push:
     new 59859162cda [style](index) Translate SNII comments and test 
diagnostics to English (#67808)
59859162cda is described below

commit 59859162cda45bc02f1179f7ab9da13dbee59ef3
Author: Jack <[email protected]>
AuthorDate: Fri Sep 11 09:41:30 2026 +0800

    [style](index) Translate SNII comments and test diagnostics to English 
(#67808)
    
    ### What problem does this PR solve?
    
    Related PR: #67537
    
    Problem Summary:
    
    The SNII code and tests touched by #67537 contain Chinese comments and
    golden-corpus diagnostics. Translate that explanatory text to English,
    including the related FE policy and protobuf comments. Preserve all
    Unicode test inputs, executable logic, query expectations, and the
    on-disk format.
    
    Also clarify two comments against the existing behavior: the
    control-byte sample is not the full internal marker, and missing golden
    expectations fail verification.
---
 be/src/storage/index/index_file_writer.h           |  8 +-
 .../storage/index/inverted/inverted_index_cache.h  |  2 +-
 .../inverted/similarity/collection_statistics.cpp  |  2 +-
 .../inverted/similarity/collection_statistics.h    |  5 +-
 be/src/storage/index/snii/compaction/eligibility.h |  6 +-
 .../index/snii/compaction/posting_run_merger.cpp   |  4 +-
 .../index/snii/compaction/posting_run_merger.h     |  5 +-
 .../snii/compaction/snii_index_compaction.cpp      |  4 +-
 be/src/storage/index/snii/format/core_metadata.cpp | 17 +++--
 .../storage/index/snii/format/format_constants.h   |  4 +-
 be/src/storage/index/snii/format/phrase_bigram.h   |  6 +-
 be/src/storage/index/snii/format/prx_frame.h       |  6 +-
 .../index/snii/query/docid_posting_reader.cpp      |  2 +-
 .../index/snii/query/internal/phrase_query_split.h |  4 +-
 .../index/snii/query/internal/plain_term_routing.h |  6 +-
 be/src/storage/index/snii/query/phrase_emit.cpp    |  2 +-
 .../index/snii/query/phrase_position_source.cpp    |  3 +-
 be/src/storage/index/snii/query/scoring_query.cpp  |  2 +-
 be/src/storage/index/snii/query/term_expansion.cpp |  4 +-
 .../index/snii/reader/logical_index_reader.h       |  2 +-
 be/src/storage/index/snii/snii_index_reader.cpp    |  7 +-
 be/src/storage/index/snii/snii_index_reader.h      |  2 +-
 be/src/storage/index/snii/snii_index_writer.cpp    | 10 +--
 be/src/storage/index/snii/snii_index_writer.h      |  2 +-
 .../index/snii/stats/snii_stats_provider.cpp       |  5 +-
 .../index/snii/writer/logical_index_writer.h       |  4 +-
 .../index/snii/writer/posting_window_emitter.cpp   |  4 +-
 .../index/snii/writer/posting_window_emitter.h     |  5 +-
 .../index/snii/writer/snii_compound_writer.cpp     |  2 +-
 .../index/snii/writer/snii_compound_writer.h       |  7 +-
 .../storage/index/snii/writer/spill_run_codec.cpp  |  4 +-
 ...inverted_index_reader_analysis_purpose_test.cpp |  6 +-
 .../similarity/collection_statistics_test.cpp      |  8 +-
 .../snii_compaction_eligibility_test.cpp           |  6 +-
 .../snii/compaction/snii_index_compaction_test.cpp | 23 +++---
 .../snii/compaction/snii_posting_cursor_test.cpp   |  9 ++-
 .../snii/compaction/snii_streamed_session_test.cpp |  9 ++-
 .../index/snii/format/core_metadata_test.cpp       | 23 +++---
 .../index/snii/query/phrase_prefix_query_test.cpp  | 14 ++--
 .../index/snii/query/query_operator_error_test.cpp |  2 +-
 .../index/snii/query/query_profile_test.cpp        |  2 +-
 .../storage/index/snii/snii_golden_corpus_test.cpp | 89 +++++++++++-----------
 .../index/snii/writer/phase_a_readback_test.cpp    |  2 +-
 .../snii/writer/snii_compound_writer_test.cpp      |  7 +-
 .../index/snii/writer/spill_run_codec_test.cpp     |  2 +-
 be/test/storage/index/snii_query_test.cpp          |  5 +-
 be/test/storage/index/snii_spimi_intern_test.cpp   |  2 +-
 be/test/storage/index/snii_writer_test.cpp         |  4 +-
 .../org/apache/doris/indexpolicy/IndexPolicy.java  |  4 +-
 .../apache/doris/indexpolicy/IndexPolicyMgr.java   |  7 +-
 gensrc/proto/snii.proto                            |  6 +-
 51 files changed, 197 insertions(+), 179 deletions(-)

diff --git a/be/src/storage/index/index_file_writer.h 
b/be/src/storage/index/index_file_writer.h
index f5a2e4c1d67..3a0346b95d4 100644
--- a/be/src/storage/index/index_file_writer.h
+++ b/be/src/storage/index/index_file_writer.h
@@ -110,8 +110,8 @@ public:
         // the prx region compresses at snii_prx_zstd_level_direct_load;
         // compaction / schema change / ADD INDEX keep snii_prx_zstd_level.
         bool is_direct_load = false;
-        // 每 doc 一字节的 BM25 norms;为空表示该索引不写 norms(keyword / 不带位置)。
-        // 非空时长度必须等于 doc_count,且 posting 会保留 freq 区供打分读取。
+        // One byte of BM25 norms per document; empty for keyword or 
positionless indexes.
+        // If nonempty, its size must equal doc_count, and postings retain 
frequencies for scoring.
         std::vector<uint8_t> encoded_norms;
     };
     Status add_snii_index(const TabletIndex* index_meta, uint32_t doc_count,
@@ -138,8 +138,8 @@ public:
             doris::snii::format::IndexConfig index_config,
             std::shared_ptr<doris::snii::writer::MemoryReporter> mem_reporter,
             doris::snii::writer::SniiStreamedIndexSession** session);
-    // write_norms=true 的会话在 finish 之前必须通过 set_encoded_norms 交付 norms
-    // (compaction 在合并 postings 的同一趟里重建它们)。
+    // Sessions with write_norms=true must supply norms through 
set_encoded_norms before finish.
+    // Compaction rebuilds them in the same pass that merges postings.
     Status add_snii_index_streamed(
             const TabletIndex* index_meta, uint32_t doc_count,
             doris::snii::writer::TrackedNullDocids null_docids, bool 
write_norms,
diff --git a/be/src/storage/index/inverted/inverted_index_cache.h 
b/be/src/storage/index/inverted/inverted_index_cache.h
index 0b53f3c0da8..e470d88ec6c 100644
--- a/be/src/storage/index/inverted/inverted_index_cache.h
+++ b/be/src/storage/index/inverted/inverted_index_cache.h
@@ -216,7 +216,7 @@ private:
 
 class InvertedIndexQueryCacheHandle;
 
-// v2:结果缓存键不再携带 CommonGrams 查询计划开关位(该功能已删除)。
+// v2: Result cache keys no longer include the removed CommonGrams query-plan 
flags.
 inline constexpr uint32_t INVERTED_INDEX_QUERY_CACHE_SEMANTICS_VERSION = 2;
 
 // Stable identity shared by result-cache and row-accurate single-flight. It 
intentionally contains
diff --git a/be/src/storage/index/inverted/similarity/collection_statistics.cpp 
b/be/src/storage/index/inverted/similarity/collection_statistics.cpp
index d8e8d4a2620..69bd4155389 100644
--- a/be/src/storage/index/inverted/similarity/collection_statistics.cpp
+++ b/be/src/storage/index/inverted/similarity/collection_statistics.cpp
@@ -251,7 +251,7 @@ Status CollectionStatistics::process_segment(const 
RowsetSharedPtr& rowset,
 
             ::doris::snii::reader::DictBlockCache dict_block_cache;
             for (const auto& logical_term_bytes : collect_info.unique_terms) {
-                // SNII 的 term 键就是分词后的原始字节,没有任何转义或版本。
+                // SNII term keys are the raw analyzed bytes, with no escaping 
or version prefix.
                 const std::string& physical_term = logical_term_bytes;
                 const auto logical_term =
                         
segment_v2::inverted_index::StringHelper::to_wstring(logical_term_bytes);
diff --git a/be/src/storage/index/inverted/similarity/collection_statistics.h 
b/be/src/storage/index/inverted/similarity/collection_statistics.h
index d65ee4a3512..7e934985083 100644
--- a/be/src/storage/index/inverted/similarity/collection_statistics.h
+++ b/be/src/storage/index/inverted/similarity/collection_statistics.h
@@ -119,8 +119,9 @@ struct SniiScoringSegmentStats {
     uint64_t token_count = 0;
 };
 
-// 一个 SNII 段能参与打分的条件:带位置(词频来自位置)且带 norms(新版 writer 对分词 + 带位置
-// 的索引一律写出)。老段没有 norms → NOT_SUPPORTED,重建索引或等 compaction 补齐。
+// SNII scoring requires positions (which provide term frequencies) and norms. 
The current writer
+// emits norms for every analyzed index with positions. Older segments without 
norms return
+// NOT_SUPPORTED until an index rebuild or compaction supplies them.
 Result<SniiScoringSegmentStats> resolve_snii_scoring_segment(uint64_t 
index_doc_count,
                                                              uint64_t 
sum_total_term_freq,
                                                              bool 
has_positions, bool has_norms);
diff --git a/be/src/storage/index/snii/compaction/eligibility.h 
b/be/src/storage/index/snii/compaction/eligibility.h
index 8c24ea56cb3..7257258ef3a 100644
--- a/be/src/storage/index/snii/compaction/eligibility.h
+++ b/be/src/storage/index/snii/compaction/eligibility.h
@@ -52,9 +52,9 @@ using AnalyzerProviderFactory = 
std::function<segment_v2::inverted_index::Analyz
         const InvertedIndexAnalyzerConfig&)>;
 
 struct SniiCompactionEligibility {
-    // A2:目标索引是"分词 + 带位置"时,合并产物必须带 norms。norms 在合并 postings 的同一趟里
-    // 从各源的词频重建(每 doc Σfreq,clamp 到 1..255),所以老段(没有 norms 的 T2)也能
-    // 不重分词地升级成带 norms 的段。
+    // A2: An analyzed destination index with positions must include norms. 
The posting merge
+    // rebuilds them from source term frequencies (sum per document, clamped 
to 1..255), so older
+    // T2 segments without norms can gain norms without reanalysis.
     bool destination_writes_norms = false;
 };
 
diff --git a/be/src/storage/index/snii/compaction/posting_run_merger.cpp 
b/be/src/storage/index/snii/compaction/posting_run_merger.cpp
index f4cfaa7ff12..53a9a34aa41 100644
--- a/be/src/storage/index/snii/compaction/posting_run_merger.cpp
+++ b/be/src/storage/index/snii/compaction/posting_run_merger.cpp
@@ -363,8 +363,8 @@ Status MergedPostingRuns::select_run(ActivePostingChunk* 
active, size_t max_docs
                               : std::span<const uint32_t> {};
 
     if (!destination_doc_lengths_.empty()) {
-        // 重建 norms:每篇文档的长度 = 该文档在所有 term 上的词频之和(与 writer 的
-        // 词元计数逐字节一致,见 SniiIndexColumnWriter),按 u8 饱和累加,255 封顶。
+        // Rebuild norms by summing each document's frequencies across all 
terms, matching the
+        // token counts in SniiIndexColumnWriter byte for byte. Saturate the 
u8 sum at 255.
         std::vector<uint8_t>& lengths = 
destination_doc_lengths_[*active_destination_];
         for (size_t i = 0; i < document_count; ++i) {
             const uint32_t docid = run->docids[i];
diff --git a/be/src/storage/index/snii/compaction/posting_run_merger.h 
b/be/src/storage/index/snii/compaction/posting_run_merger.h
index 95afa259377..1562e9d4de2 100644
--- a/be/src/storage/index/snii/compaction/posting_run_merger.h
+++ b/be/src/storage/index/snii/compaction/posting_run_merger.h
@@ -57,8 +57,9 @@ class MergedPostingRuns final : public 
writer::TermPostingSource {
     };
 
 public:
-    // destination_doc_lengths 非空时,每个 (目标段, 目标 docid) 的词频按 u8 饱和累加(0..255),
-    // 合并结束后由调用方 encode 成 norms;为空表示目标不写 norms。
+    // If destination_doc_lengths is nonempty, accumulate frequencies for each 
destination
+    // (segment, docid), saturating at 255. The caller encodes them as norms 
after the merge.
+    // An empty vector means the destination does not write norms.
     MergedPostingRuns(std::vector<std::unique_ptr<SniiPostingCursor>> cursors,
                       bool retain_positions, std::span<const uint32_t> 
destination_doc_counts,
                       std::span<std::vector<uint8_t>> destination_doc_lengths);
diff --git a/be/src/storage/index/snii/compaction/snii_index_compaction.cpp 
b/be/src/storage/index/snii/compaction/snii_index_compaction.cpp
index a7c4cdcfb38..efa021d2f1e 100644
--- a/be/src/storage/index/snii/compaction/snii_index_compaction.cpp
+++ b/be/src/storage/index/snii/compaction/snii_index_compaction.cpp
@@ -412,8 +412,8 @@ Status SniiPlainT2MergePlan::merge_terms(
     }
 
     if (eligibility_.destination_writes_norms) {
-        // 累加的是原始长度(0..255 饱和);encode_norm 把 0 映射成 1,与 writer 的
-        // encode_norm(len) = clamp(len, 1, 255) 一致。
+        // Accumulated raw lengths saturate at 255. encode_norm maps 0 to 1, 
matching the
+        // writer's encode_norm(len) = clamp(len, 1, 255).
         for (size_t destination_ordinal = 0; destination_ordinal < 
sessions.size();
              ++destination_ordinal) {
             for (uint8_t& value : 
destination_encoded_norms_[destination_ordinal]) {
diff --git a/be/src/storage/index/snii/format/core_metadata.cpp 
b/be/src/storage/index/snii/format/core_metadata.cpp
index 3bc0196965e..0f6ef8f5a59 100644
--- a/be/src/storage/index/snii/format/core_metadata.cpp
+++ b/be/src/storage/index/snii/format/core_metadata.cpp
@@ -72,9 +72,9 @@ Status decode_core_pb(const doris::snii::SniiCoreMetadataPB& 
input, CoreMetadata
         !stats.has_null_count()) {
         return corrupted("core metadata: missing statistics field");
     }
-    // sum_total_term_freq(字段 5)与 norms(字段 5)是后加的可选字段:已上线的生产
-    // 3.1 系 writer 不写它们。缺失 = 该段没有打分统计 / 没有 norms,
-    // 只影响 BM25 打分是否可用,不影响任何过滤查询。
+    // sum_total_term_freq (stats field 5) and norms (section_refs field 5) 
are optional additions
+    // absent from the deployed 3.1-series writer. Missing fields mean no 
scoring statistics or
+    // norms, affecting BM25 availability but not filtering queries.
     out->stats = {.doc_count = stats.doc_count(),
                   .indexed_doc_count = stats.indexed_doc_count(),
                   .term_count = stats.term_count(),
@@ -97,15 +97,16 @@ Status decode_core_pb(const 
doris::snii::SniiCoreMetadataPB& input, CoreMetadata
     RETURN_IF_ERROR(decode_region_ref(refs.null_bitmap(), 
&out->section_refs.null_bitmap));
     RETURN_IF_ERROR(decode_region_ref(refs.bsbf(), &out->section_refs.bsbf));
 
-    // 墓碑:CommonGrams 功能已删除。带过字段 4/5 的段只可能是用 CommonGrams analyzer 写出来的
-    // (含 gram 词项、键转义或混合 posting 策略),term 键与查询语义都已不可解释,必须重建索引。
-    // 生产 writer 从未写过这两个字段,所以升级路径不受影响。
+    // Tombstones for the removed CommonGrams feature. Fields 4/5 identify 
segments written with
+    // a CommonGrams analyzer (gram terms, escaped keys, or mixed posting 
policies). Their term
+    // keys and query semantics are no longer supported, so these indexes must 
be rebuilt.
+    // Production writers never emitted these fields, so upgrades are 
unaffected.
     if (input.has_legacy_common_grams() || 
input.has_legacy_common_grams_posting_policy()) {
         return unsupported(
                 "core metadata: segment was written with CommonGrams, which is 
no longer "
                 "supported; rebuild the index");
     }
-    // norms(每 doc 一字节的 BM25 文档长度)只对带位置的段有意义:打分的词频来自位置。
+    // Norms encode BM25 document lengths in one byte and require positions 
for term frequencies.
     if (out->section_refs.norms.length != 0 && 
!has_positions(out->index_config)) {
         return corrupted("core metadata: norms require positions");
     }
@@ -130,7 +131,7 @@ Status encode_core_metadata(const CoreMetadata& metadata, 
ByteSink* out) {
     auto* refs = core.mutable_section_refs();
     encode_region_ref(metadata.section_refs.dict_region, 
refs->mutable_dict_region());
     encode_region_ref(metadata.section_refs.posting_region, 
refs->mutable_posting_region());
-    // 没有 norms 的段不写字段 5:与生产 writer 的字节形态一致,老 reader 也无需感知。
+    // Omit field 5 when norms are absent, matching production bytes without 
affecting old readers.
     if (metadata.section_refs.norms.length != 0) {
         encode_region_ref(metadata.section_refs.norms, refs->mutable_norms());
     }
diff --git a/be/src/storage/index/snii/format/format_constants.h 
b/be/src/storage/index/snii/format/format_constants.h
index 683acfb16d3..3073a18930e 100644
--- a/be/src/storage/index/snii/format/format_constants.h
+++ b/be/src/storage/index/snii/format/format_constants.h
@@ -68,8 +68,8 @@ enum class SectionType : uint8_t {
 enum class IndexConfig : uint8_t {
     kDocsOnly = 0,      // docid only: term/match filtering
     kDocsPositions = 1, // docid+positions: MATCH_PHRASE; BM25 tf = position 
count
-    // 2 曾是 kDocsPositionsScoring(CommonGrams 时代的打分 tier),已删除:打分能力由
-    // norms region 是否存在决定(见 CoreMetadata::section_refs.norms),reader 拒绝值 2。
+    // Value 2 was kDocsPositionsScoring, the removed CommonGrams scoring 
tier. Scoring now
+    // depends on the norms region (CoreMetadata::section_refs.norms); readers 
reject value 2.
     kPositionsOffsets = 3, // reserved (highlight/RAG), not implemented in 
this release
 };
 
diff --git a/be/src/storage/index/snii/format/phrase_bigram.h 
b/be/src/storage/index/snii/format/phrase_bigram.h
index 743810cc506..0c255e2ba6b 100644
--- a/be/src/storage/index/snii/format/phrase_bigram.h
+++ b/be/src/storage/index/snii/format/phrase_bigram.h
@@ -31,9 +31,9 @@ inline bool is_phrase_bigram_term(std::string_view term) {
     return term.starts_with(kPhraseBigramTermMarker);
 }
 
-// SNII 的 term 键就是分词后的原始字节,没有任何转义。唯一的内部命名空间是上面这个以 \x1F
-// 开头的 phrase-bigram 标记:用户 term(或前缀展开的前缀)若与它重叠,查询必须绕过 SNII,
-// 否则用户词项会命中内部词项。
+// SNII term keys are raw analyzed bytes, without escaping. The only internal 
namespace is the
+// phrase-bigram marker above, starting with \x1F. Queries whose user terms or 
expansion prefixes
+// overlap this marker must bypass SNII to avoid matching internal terms.
 inline bool term_overlaps_internal_namespace(std::string_view term) {
     return term.starts_with(kPhraseBigramTermMarker);
 }
diff --git a/be/src/storage/index/snii/format/prx_frame.h 
b/be/src/storage/index/snii/format/prx_frame.h
index 6159330c7bb..84f68381f15 100644
--- a/be/src/storage/index/snii/format/prx_frame.h
+++ b/be/src/storage/index/snii/format/prx_frame.h
@@ -35,9 +35,9 @@ struct PrxFrameView {
     Slice payload;
 };
 
-// verify_crc=false 只解析帧头并切出 payload、跳过 crc 比对:给"估算位置工作量"这类
-// 只读元数据的路径用,避免在真正解码之前对整个 prx 窗口多算一遍 crc32c;坏帧仍会在
-// 解码时被 crc 拦下。
+// With verify_crc=false, parse only the header and extract the payload 
without checking its CRC.
+// Metadata-only paths, such as position-work estimation, avoid an extra 
crc32c pass over the PRX
+// window. Decoding still checks the CRC and rejects corrupt frames.
 Status read_prx_frame(ByteSource* source, PrxFrameView* frame, bool verify_crc 
= true);
 
 } // namespace format
diff --git a/be/src/storage/index/snii/query/docid_posting_reader.cpp 
b/be/src/storage/index/snii/query/docid_posting_reader.cpp
index a813334c578..eaf5560881f 100644
--- a/be/src/storage/index/snii/query/docid_posting_reader.cpp
+++ b/be/src/storage/index/snii/query/docid_posting_reader.cpp
@@ -110,7 +110,7 @@ Status plan_window_prefix(const LogicalIndexReader& idx, 
WindowPlan* plan,
     RETURN_IF_ERROR(validate_windowed_docs_prefix(posting.entry));
     uint64_t abs = 0;
     RETURN_IF_ERROR(prelude_abs(idx, posting.entry, posting.frq_base, &abs));
-    // 生产布局:整个 .frq 载荷就是 [prelude][dd-block],一次 range 读完。
+    // Production layout: the entire .frq payload is [prelude][dd-block], read 
in one range request.
     plan->prefix_handle = fetcher->add(abs, posting.entry.frq_len);
     return Status::OK();
 }
diff --git a/be/src/storage/index/snii/query/internal/phrase_query_split.h 
b/be/src/storage/index/snii/query/internal/phrase_query_split.h
index 822dc00166e..f552543ca6d 100644
--- a/be/src/storage/index/snii/query/internal/phrase_query_split.h
+++ b/be/src/storage/index/snii/query/internal/phrase_query_split.h
@@ -173,8 +173,8 @@ struct PhraseTermMapping {
 
 PhraseTermMapping build_phrase_term_mapping(const std::vector<std::string>& 
terms);
 
-// 把一个已解析的 term 以给定位置偏移追加到 phrase 计划(unique_terms 去重)。定义在
-// phrase_plan.cpp;以前只靠 unity build 的拼接顺序"碰巧"可见。
+// Append a parsed term at the given position offset, deduplicating it in 
unique_terms.
+// Defined in phrase_plan.cpp; visibility previously depended on the unity 
build's source order.
 void append_resolved_phrase_clause(ResolvedQueryTerm term, uint32_t 
position_offset,
                                    internal::ResolvedPhrasePlan* plan);
 
diff --git a/be/src/storage/index/snii/query/internal/plain_term_routing.h 
b/be/src/storage/index/snii/query/internal/plain_term_routing.h
index 110aa1950b6..349fd9754e8 100644
--- a/be/src/storage/index/snii/query/internal/plain_term_routing.h
+++ b/be/src/storage/index/snii/query/internal/plain_term_routing.h
@@ -25,9 +25,9 @@
 
 namespace doris::snii::query::internal {
 
-// SNII 的 term 键就是分词后的原始字节:没有转义、没有版本、没有"物理键 / 逻辑键"之分。
-// 唯一的例外是 \x1F 开头的内部命名空间(phrase-bigram 标记):用户 term 若与之重叠,
-// 整条查询以 INVERTED_INDEX_BYPASS 绕过 SNII,由上层走非索引路径。
+// SNII term keys are raw analyzed bytes, without escaping, versioning, or a 
physical/logical
+// key distinction. The sole exception is the internal phrase-bigram namespace 
starting with
+// \x1F: overlapping user terms return INVERTED_INDEX_BYPASS for execution 
without the index.
 inline Status check_term_outside_internal_namespace(std::string_view term) {
     if (format::term_overlaps_internal_namespace(term)) {
         return Status::Error<ErrorCode::INVERTED_INDEX_BYPASS>(
diff --git a/be/src/storage/index/snii/query/phrase_emit.cpp 
b/be/src/storage/index/snii/query/phrase_emit.cpp
index 0d0cf4e46a5..416a0ea9f7a 100644
--- a/be/src/storage/index/snii/query/phrase_emit.cpp
+++ b/be/src/storage/index/snii/query/phrase_emit.cpp
@@ -90,7 +90,7 @@ bool should_use_streaming_exact_phrase(const 
std::vector<TermPlan>& plans,
         }
         const TermPlan& plan = plans[plan_index];
         DORIS_CHECK_NE(plan.df, 0);
-        // dict entry 不带词频统计:每 doc 的平均位置工作量从已解码的 prx 帧统计得来。
+        // Dictionary entries lack term frequencies; estimate position work 
from decoded PRX frames.
         DORIS_CHECK_NE(sources[plan_index].logical_position_docs, 0);
         const uint64_t position_work = 
sources[plan_index].logical_position_work /
                                        
sources[plan_index].logical_position_docs;
diff --git a/be/src/storage/index/snii/query/phrase_position_source.cpp 
b/be/src/storage/index/snii/query/phrase_position_source.cpp
index f8527a975c6..c2cabd61914 100644
--- a/be/src/storage/index/snii/query/phrase_position_source.cpp
+++ b/be/src/storage/index/snii/query/phrase_position_source.cpp
@@ -78,7 +78,8 @@ PhraseTermMapping build_phrase_term_mapping(const 
std::vector<std::string>& term
 }
 
 namespace {
-// 只读帧头估算工作量,不校验 crc:解码时才校验,坏帧在那里被拦下并保留已解码帧的统计。
+// Estimate work from frame headers without checking CRCs. Decoding rejects 
corrupt frames while
+// retaining statistics from frames already decoded.
 Status accumulate_frame_position_work(Slice frames, uint64_t* work) {
     ByteSource source(frames);
     while (!source.eof()) {
diff --git a/be/src/storage/index/snii/query/scoring_query.cpp 
b/be/src/storage/index/snii/query/scoring_query.cpp
index e1147cb6d35..cd63062cc65 100644
--- a/be/src/storage/index/snii/query/scoring_query.cpp
+++ b/be/src/storage/index/snii/query/scoring_query.cpp
@@ -52,7 +52,7 @@ struct TermPosting {
     double score = 0.0;
 };
 
-// 磁盘上没有词频区:BM25 的 tf 就是该 term 在文档里的位置个数(与 Lucene 系打分定义一致)。
+// With no on-disk frequency region, BM25 tf is the term's position count, as 
in Lucene scoring.
 Status require_positions(const LogicalIndexReader& idx) {
     if (!idx.has_positions()) {
         return Status::Error<ErrorCode::INVALID_ARGUMENT, false>(
diff --git a/be/src/storage/index/snii/query/term_expansion.cpp 
b/be/src/storage/index/snii/query/term_expansion.cpp
index 4e78c1543e8..ab716ab2ece 100644
--- a/be/src/storage/index/snii/query/term_expansion.cpp
+++ b/be/src/storage/index/snii/query/term_expansion.cpp
@@ -45,8 +45,8 @@ Status legacy_raw_prefix_exists(const 
reader::LogicalIndexReader& idx, std::stri
             cache);
 }
 
-// 空前缀会枚举整个词典:段里若存在内部命名空间的词项(phrase-bigram 标记),枚举结果会混入
-// 内部词项,此时整条查询绕过 SNII。
+// An empty prefix enumerates the entire dictionary. If the segment contains 
internal
+// phrase-bigram terms, bypass SNII for the whole query to avoid including 
those terms.
 Status prove_no_internal_terms(const reader::LogicalIndexReader& idx,
                                reader::DictBlockCache* cache) {
     bool exists = false;
diff --git a/be/src/storage/index/snii/reader/logical_index_reader.h 
b/be/src/storage/index/snii/reader/logical_index_reader.h
index 1500ebcb117..ee1e90b2571 100644
--- a/be/src/storage/index/snii/reader/logical_index_reader.h
+++ b/be/src/storage/index/snii/reader/logical_index_reader.h
@@ -178,7 +178,7 @@ public:
     const format::StatsBlock& stats() const { return core_.stats; }
     format::IndexTier tier() const { return tier_; }
     bool has_positions() const { return has_positions_; }
-    // 是否带 BM25 norms(分词 + 带位置的索引由新版 writer 写出;老段没有)。
+    // Whether BM25 norms exist: current writers emit them for analyzed 
indexes with positions.
     bool has_norms() const { return core_.section_refs.norms.length != 0; }
     LogicalIndexOpenMode open_mode() const { return open_mode_; }
     io::FileReader* reader() const { return reader_; }
diff --git a/be/src/storage/index/snii/snii_index_reader.cpp 
b/be/src/storage/index/snii/snii_index_reader.cpp
index 82d42a7f82d..bbcb78f7f84 100644
--- a/be/src/storage/index/snii/snii_index_reader.cpp
+++ b/be/src/storage/index/snii/snii_index_reader.cpp
@@ -624,9 +624,10 @@ Status SniiIndexReader::_query(const IndexQueryContextPtr& 
context, const std::s
     if (query_type == InvertedIndexQueryType::MATCH_PHRASE_QUERY) {
         parse_phrase_slop(&plain_analysis_str, &query_info);
     }
-    // 结果缓存只以 (索引文件, 列, 查询类型, 原始查询字节) 为键:分词结果由索引属性与
-    // policy 唯一决定(policy 被引用后不可变),因此打开 segment 之前就能判定是否可共享;
-    // 只有打分查询(结果随集合统计变化)不进缓存,也不走 single-flight 合并。
+    // Result cache keys contain only (index file, column, query type, raw 
query bytes). Analysis
+    // is determined by index properties and policies, which are immutable 
once referenced, so
+    // sharing can be decided before opening the segment. Scoring queries 
depend on collection
+    // statistics and use neither the result cache nor single-flight 
coalescing.
     const bool allow_result_cache = !actual_similarity;
     const InvertedIndexRawQuerySemantic raw_semantic {.raw_query_bytes = 
search_str,
                                                       .query_type = query_type,
diff --git a/be/src/storage/index/snii/snii_index_reader.h 
b/be/src/storage/index/snii/snii_index_reader.h
index 9fb0e188d18..91996110887 100644
--- a/be/src/storage/index/snii/snii_index_reader.h
+++ b/be/src/storage/index/snii/snii_index_reader.h
@@ -37,7 +37,7 @@ struct PhraseMatch;
 
 namespace doris::segment_v2 {
 
-// 一次查询在打开 logical reader 之后交给 _compute_query_bitmap 的全部输入。
+// All query inputs passed to _compute_query_bitmap after opening the logical 
reader.
 struct SniiQueryBitmapRequest {
     InvertedIndexQueryType query_type;
     const InvertedIndexQueryInfo& query_info;
diff --git a/be/src/storage/index/snii/snii_index_writer.cpp 
b/be/src/storage/index/snii/snii_index_writer.cpp
index 1720d4deaad..94ec6610c0b 100644
--- a/be/src/storage/index/snii/snii_index_writer.cpp
+++ b/be/src/storage/index/snii/snii_index_writer.cpp
@@ -115,9 +115,9 @@ Status SniiIndexColumnWriter::init() {
         return Status::Error<ErrorCode::INVERTED_INDEX_ANALYZER_ERROR>(
                 "SNII create analyzer failed: {}", e.what());
     }
-    // A2:分词 + 带位置的索引一律写 norms(每 doc 的词元数,clamp 到 1..255),与 CLucene 的
-    // 打分能力对齐;keyword 或不带位置的索引不写。norms 是 core 元数据里的可选 region,
-    // 不认识它的老 reader 会原样忽略。
+    // A2: Analyzed indexes with positions always write norms (tokens per 
document, clamped to
+    // 1..255), matching CLucene's scoring capabilities. Keyword or 
positionless indexes omit
+    // them. Norms are an optional core-metadata region ignored by older 
readers.
     _writes_norms = _should_analyzer && _has_positions;
     return Status::OK();
 }
@@ -253,8 +253,8 @@ Status SniiIndexColumnWriter::add_array_values(size_t 
field_size, const void* va
             row_token_count += token_count;
         }
         if (_writes_norms) {
-            // 一行 ARRAY 的文档长度 = 所有元素的词元数之和(NULL 行由 add_array_nulls 声明,
-            // 但同样经过这里,长度为 0)。
+            // An ARRAY row's document length is the total token count across 
its elements.
+            // NULL rows also pass here with length 0 and are marked by 
add_array_nulls.
             
_encoded_norms.push_back(::doris::snii::query::encode_norm(row_token_count));
             _report_encoded_norms_capacity();
         }
diff --git a/be/src/storage/index/snii/snii_index_writer.h 
b/be/src/storage/index/snii/snii_index_writer.h
index 5196e722afe..b17624dbfbf 100644
--- a/be/src/storage/index/snii/snii_index_writer.h
+++ b/be/src/storage/index/snii/snii_index_writer.h
@@ -93,7 +93,7 @@ private:
     bool _should_analyzer = false;
     bool _has_positions = false;
     const bool _is_char;
-    // A2:分词 + 带位置的索引一律写 norms(与 CLucene 一致),供 BM25 使用。
+    // A2: Analyzed indexes with positions always write BM25 norms, matching 
CLucene.
     bool _writes_norms = false;
     // Latch: set_direct_load() ran. The first call wins; a repeat or late call
     // is ignored (and logged) so one index keeps one stable compression-tier
diff --git a/be/src/storage/index/snii/stats/snii_stats_provider.cpp 
b/be/src/storage/index/snii/stats/snii_stats_provider.cpp
index e13f96a26f6..557c9c30417 100644
--- a/be/src/storage/index/snii/stats/snii_stats_provider.cpp
+++ b/be/src/storage/index/snii/stats/snii_stats_provider.cpp
@@ -46,8 +46,9 @@ Status SniiStatsProvider::open(const 
reader::LogicalIndexReader* idx, SniiStatsP
     if (idx == nullptr || out == nullptr) {
         return Status::Error<ErrorCode::INVALID_ARGUMENT, 
false>("stats_provider: null argument");
     }
-    // 统计全部来自物理统计:doc_count / indexed_doc_count / sum_total_term_freq 由 writer 
对
-    // 每个 posting 累加,avgdl 按注释定义用 indexed_doc_count(不含 NULL 行)。
+    // All statistics come from physical data collected by the writer: 
doc_count,
+    // indexed_doc_count, and sum_total_term_freq. As documented, avgdl uses 
indexed_doc_count,
+    // which excludes NULL rows.
     out->idx_ = idx;
     const auto& sb = idx->stats();
     out->doc_count_ = sb.doc_count;
diff --git a/be/src/storage/index/snii/writer/logical_index_writer.h 
b/be/src/storage/index/snii/writer/logical_index_writer.h
index 271975807cd..693d020cc6e 100644
--- a/be/src/storage/index/snii/writer/logical_index_writer.h
+++ b/be/src/storage/index/snii/writer/logical_index_writer.h
@@ -106,8 +106,8 @@ struct SniiIndexInput {
     // Per-doc 1-byte encoded norm (length doc_count); only consumed when the
     // config has scoring. May be empty otherwise.
     std::vector<uint8_t> encoded_norms;
-    // 流式合并会话:norms 在 finish 之前才交付(与 postings 同一趟重建),先用它声明
-    // "这个索引会有 norms",让 writer 在 finalize 时校验长度。
+    // Streaming merge sessions declare norms up front but supply them only 
before finish,
+    // after rebuilding them alongside postings. The writer validates their 
size at finalize.
     bool write_norms = false;
     // G16-h: zstd levels for the dict-block whole-block compression and the
     // .prx window auto mode (both default 3 == the historical constants).
diff --git a/be/src/storage/index/snii/writer/posting_window_emitter.cpp 
b/be/src/storage/index/snii/writer/posting_window_emitter.cpp
index 130087edbf9..c2826a83f01 100644
--- a/be/src/storage/index/snii/writer/posting_window_emitter.cpp
+++ b/be/src/storage/index/snii/writer/posting_window_emitter.cpp
@@ -197,8 +197,8 @@ private:
                     "window emitter: document frequency overflow");
         }
         stats_.df += static_cast<uint32_t>(run.docids.size());
-        // 没有 freqs 的 docs-only 输入按每 doc 一次计入总词频;带 freqs 的输入在
-        // emit_window_impl 的逐 doc 循环里累加。
+        // Docs-only input without freqs contributes one occurrence per 
document. Input with
+        // freqs is accumulated in emit_window_impl's per-document loop.
         if (run.freqs.empty()) {
             return checked_add(run.docids.size(), &stats_.total_freq);
         }
diff --git a/be/src/storage/index/snii/writer/posting_window_emitter.h 
b/be/src/storage/index/snii/writer/posting_window_emitter.h
index b59e458d9c5..ed79a92f1bd 100644
--- a/be/src/storage/index/snii/writer/posting_window_emitter.h
+++ b/be/src/storage/index/snii/writer/posting_window_emitter.h
@@ -48,8 +48,9 @@ struct PostingRunView {
 
 struct TermAggregateStats {
     uint32_t df = 0;
-    // 该 term 的总词频(有位置的 term = 位置个数;docs-only = 文档数或输入 freqs 之和),只进
-    // core 元数据的 sum_total_term_freq(BM25 的 avgdl),不进 dict entry。
+    // Total frequency of this term: position count, or document count / sum 
of input freqs for
+    // docs-only terms. Used only in core metadata's sum_total_term_freq (BM25 
avgdl), not in
+    // dictionary entries.
     uint64_t total_freq = 0;
 };
 
diff --git a/be/src/storage/index/snii/writer/snii_compound_writer.cpp 
b/be/src/storage/index/snii/writer/snii_compound_writer.cpp
index e46a6112f12..89256f88dd7 100644
--- a/be/src/storage/index/snii/writer/snii_compound_writer.cpp
+++ b/be/src/storage/index/snii/writer/snii_compound_writer.cpp
@@ -374,7 +374,7 @@ Status 
SniiStreamedIndexSession::set_encoded_norms(TrackedEncodedNorms encoded_n
                 "compound: norms length {} differs from doc_count {}", 
encoded_norms.size(),
                 input_.doc_count);
     }
-    // writer_ 持有 input_.encoded_norms 的引用:就地移入即可,finalize 时按引用读取。
+    // writer_ references input_.encoded_norms; move into it here for finalize 
to read by reference.
     encoded_norms_reservation_ = std::move(encoded_norms.reservation_);
     input_.encoded_norms = std::move(encoded_norms.norms_);
     norms_set_ = true;
diff --git a/be/src/storage/index/snii/writer/snii_compound_writer.h 
b/be/src/storage/index/snii/writer/snii_compound_writer.h
index f17bcd1482d..cb3092d780d 100644
--- a/be/src/storage/index/snii/writer/snii_compound_writer.h
+++ b/be/src/storage/index/snii/writer/snii_compound_writer.h
@@ -131,8 +131,9 @@ public:
     // every rejection is terminal here because posting bytes may already have
     // entered the compound output; all later calls return the first error.
     Status push_term(StreamedTermPostings&& tp);
-    // 交付本目标段的 norms(compaction 在合并 postings 的同一趟里重建)。声明了
-    // write_norms 的会话在 finish 之前必须恰好调用一次;长度必须等于 doc_count。
+    // Supply this destination segment's norms, rebuilt alongside postings 
during compaction.
+    // Sessions declaring write_norms must call this exactly once before 
finish, with doc_count
+    // entries.
     Status set_encoded_norms(TrackedEncodedNorms encoded_norms);
     // Seals this index: flushes the trailing DICT block, streams the DICT 
region
     // right after the posting region and records the placements. A failed 
finish
@@ -222,7 +223,7 @@ public:
     // rejected while a session is unfinished, as is finish(). The returned
     // handle is owned by this writer and valid for its lifetime.
     Status begin_streamed_index(SniiIndexInput in, SniiStreamedIndexSession** 
session);
-    // in.write_norms=true 的会话在 finish 之前必须通过 set_encoded_norms 交付 norms。
+    // Sessions with in.write_norms=true must supply norms through 
set_encoded_norms before finish.
     Status begin_streamed_index(SniiIndexInput in, TrackedNullDocids 
null_docids,
                                 SniiStreamedIndexSession** session);
 
diff --git a/be/src/storage/index/snii/writer/spill_run_codec.cpp 
b/be/src/storage/index/snii/writer/spill_run_codec.cpp
index f70b2303012..1a669789361 100644
--- a/be/src/storage/index/snii/writer/spill_run_codec.cpp
+++ b/be/src/storage/index/snii/writer/spill_run_codec.cpp
@@ -49,8 +49,8 @@ constexpr size_t kWriteFlushBytes = 1u << 22; // 4 MiB
 // merge-phase peak RSS at low spill thresholds (each reader holds a window).
 constexpr size_t kReadChunkBytes = 1u << 16; // 64 KiB
 
-// 形状值 0 曾是 CommonGrams 的无频次 docs-only 记录,已删除;run 文件是构建期临时文件,
-// 读到 0 一律按损坏处理。
+// Shape 0 was the removed CommonGrams docs-only record without frequencies. 
Run files are
+// temporary build artifacts, so encountering shape 0 is always treated as 
corruption.
 enum class RunPostingShape : uint8_t {
     kDocsAndFreqs = 1,
     kPositioned = 2,
diff --git 
a/be/test/storage/index/inverted/inverted_index_reader_analysis_purpose_test.cpp
 
b/be/test/storage/index/inverted/inverted_index_reader_analysis_purpose_test.cpp
index 50f1ae3346f..e0e9620e964 100644
--- 
a/be/test/storage/index/inverted/inverted_index_reader_analysis_purpose_test.cpp
+++ 
b/be/test/storage/index/inverted/inverted_index_reader_analysis_purpose_test.cpp
@@ -303,7 +303,7 @@ protected:
         EXPECT_EQ(execution.stats.inverted_index_query_cache_miss, 0);
         EXPECT_EQ(execution.stats.inverted_index_query_cache_lookup, 1);
         EXPECT_EQ(execution.stats.inverted_index_query_cache_insert, 0);
-        // 命中发生在打开 logical reader 之前:searcher cache 完全没有被触碰。
+        // The hit occurs before opening the logical reader, without touching 
the searcher cache.
         EXPECT_EQ(execution.stats.inverted_index_searcher_cache_hit, 0);
         EXPECT_EQ(execution.stats.inverted_index_searcher_cache_miss, 0);
     }
@@ -365,8 +365,8 @@ TEST(InvertedIndexRawQuerySemanticTest, 
CacheEnvelopeSeparatesSlashAndNulBoundar
     EXPECT_NE(nul_left.encode(), nul_right.encode());
 }
 
-// 结果缓存的键只有 (索引文件, 列, 查询类型, 原始查询字节):命中发生在打开 segment 与
-// 任何分词之前,所以即使 analyzer provider 会失败,缓存命中也照常返回。
+// Result cache keys contain only (index file, column, query type, raw query 
bytes). Hits precede
+// opening the segment and analysis, so they succeed even if the analyzer 
provider would fail.
 TEST_F(InvertedIndexReaderAnalysisPurposeTest, 
SniiRawCacheHitHappensBeforeSegmentOpenAndAnalysis) {
     preload_legacy_searcher_cache_entries();
     expect_raw_cache_hit_after_segment_admission(
diff --git 
a/be/test/storage/index/inverted/similarity/collection_statistics_test.cpp 
b/be/test/storage/index/inverted/similarity/collection_statistics_test.cpp
index cb57ea12b03..21b62a02f18 100644
--- a/be/test/storage/index/inverted/similarity/collection_statistics_test.cpp
+++ b/be/test/storage/index/inverted/similarity/collection_statistics_test.cpp
@@ -453,7 +453,7 @@ protected:
         return file_writer.finish_close();
     }
 
-    // 分词 + 带位置 + norms 的普通 SNII 段:这就是新 writer 对可打分索引写出的形态。
+    // A normal analyzed SNII segment with positions and norms, as emitted for 
scoring indexes.
     Status write_snii_scoring_segment(const std::string& segment_path) {
         const std::string index_path_prefix {
                 
segment_v2::InvertedIndexDescriptor::get_index_file_path_prefix(segment_path)};
@@ -603,7 +603,7 @@ protected:
         return tablet_schema;
     }
 
-    // 落在 SNII 内部命名空间(\x1f 开头)里的词项,对 V3(CLucene)索引只是普通字节。
+    // Terms in SNII's internal namespace (starting with \x1f) are ordinary 
bytes in V3 (CLucene).
     VExprContextSPtrs create_reserved_exact_search_contexts() {
         return create_search_contexts("EXACT",
                                       
std::string(snii::format::kPhraseBigramTermMarker) + "user");
@@ -1090,7 +1090,7 @@ protected:
     std::unique_ptr<TestableCollectionStatistics> stats_;
 };
 
-// 一个 SNII 段能参与打分的条件只有两个物理事实:带位置、带 norms;统计量直接取 stats 块。
+// SNII scoring requires only positions and norms; statistics come directly 
from the stats block.
 TEST(CollectionStatisticsSniiScoringTest, 
ResolveUsesPhysicalDocAndTokenCounts) {
     auto result = resolve_snii_scoring_segment(3, 7, /*has_positions=*/true, 
/*has_norms=*/true);
 
@@ -1132,7 +1132,7 @@ TEST_F(CollectionStatisticsTest, 
CollectionStatisticsInstancesKeepAdmissionState
     EXPECT_FLOAT_EQ(second.get_or_calculate_avg_dl(L"1"), 5.0F);
 }
 
-// 老段(没有 norms)混进来就整体拒绝打分,已收集的统计量一并清空。
+// An older segment without norms disables scoring for the whole collection 
and clears its stats.
 TEST_F(CollectionStatisticsTest, SegmentWithoutNormsRejectsWholeCollection) {
     ASSERT_TRUE(admit_snii_segment_for_test(stats_.get(), L"1", 2, 6).ok());
 
diff --git 
a/be/test/storage/index/snii/compaction/snii_compaction_eligibility_test.cpp 
b/be/test/storage/index/snii/compaction/snii_compaction_eligibility_test.cpp
index 007e836d727..bbf5c0493d3 100644
--- a/be/test/storage/index/snii/compaction/snii_compaction_eligibility_test.cpp
+++ b/be/test/storage/index/snii/compaction/snii_compaction_eligibility_test.cpp
@@ -211,7 +211,7 @@ TEST(SniiCompactionEligibilityTest, 
RejectsPhysicalShapesOutsidePlainT2) {
     }
 }
 
-// 带 norms 的段(新 writer 对分词 + 带位置索引的产物)与不带 norms 的老段都是合法的 T2 源。
+// Both current segments with norms and older segments without norms are valid 
T2 sources.
 TEST(SniiCompactionEligibilityTest, AcceptsSourcesWithOrWithoutNorms) {
     auto with_norms = open_index(IndexShape {.has_norms = true});
     auto without_norms = open_index({});
@@ -223,8 +223,8 @@ TEST(SniiCompactionEligibilityTest, 
AcceptsSourcesWithOrWithoutNorms) {
     EXPECT_TRUE(compaction::validate_plain_t2_compaction_eligibility(sources, 
*destination).ok());
 }
 
-// A2:目标索引分词就写 norms(哪怕所有源都是没有 norms 的老段——合并时从 postings 重建);
-// 不分词的 keyword 索引不写。
+// A2: An analyzed destination writes norms, rebuilt from postings even if all 
sources lack them.
+// Keyword indexes without analysis do not write norms.
 TEST(SniiCompactionEligibilityTest, DestinationWritesNormsExactlyWhenAnalyzed) 
{
     auto legacy = open_index({});
 
diff --git 
a/be/test/storage/index/snii/compaction/snii_index_compaction_test.cpp 
b/be/test/storage/index/snii/compaction/snii_index_compaction_test.cpp
index 60f2a3cd4ef..5cee6dcf1fe 100644
--- a/be/test/storage/index/snii/compaction/snii_index_compaction_test.cpp
+++ b/be/test/storage/index/snii/compaction/snii_index_compaction_test.cpp
@@ -78,8 +78,8 @@ SniiIndexInput make_input(uint32_t doc_count, 
std::vector<uint32_t> null_docids,
     return input;
 }
 
-// 带 norms 的 T2 输入(A2:分词 + 带位置的索引一律写 norms)。norms 由调用方给出,必须与
-// postings 的每 doc 词频一致,这样 compaction 从 postings 重建出的 norms 才能与之逐字节相同。
+// T2 input with norms (A2: analyzed indexes with positions always write 
norms). Caller-supplied
+// norms must match per-document posting frequencies so compaction can rebuild 
identical bytes.
 SniiIndexInput make_norms_input(uint32_t doc_count, std::vector<uint32_t> 
null_docids,
                                 std::vector<uint8_t> norms, 
std::vector<TermPostings> terms) {
     SniiIndexInput input = make_input(doc_count, std::move(null_docids), 
std::move(terms));
@@ -509,19 +509,20 @@ TEST(SniiIndexCompactionTest, 
MergesTwentyFourRunSourcesByteIdenticallyToReferen
     }
 }
 
-// A2 验收:目标写 norms 时,compaction 不重分词,而是在合并 postings 的同一趟里按每 doc Σfreq
-// 重建 norms;结果必须与"用同样 postings 与一致 norms 重新构建"的段逐字节相同。
+// Verify A2: compaction rebuilds destination norms from per-document 
frequency sums in the
+// posting merge pass, without reanalysis. The result must be byte-identical 
to a segment rebuilt
+// from the same postings and matching norms.
 TEST(SniiIndexCompactionTest, NormsMergeMatchesRebuildAfterDeletesAndRemap) {
     OpenedIndex source_zero;
     OpenedIndex source_one;
-    // 源 0:doc0 = alpha×2,doc1 = alpha×1,doc2(null 但带 posting)= beta×1 → norms 
{2, 1, 1}
+    // Source 0: doc0 = alpha*2, doc1 = alpha*1, doc2 (null with a posting) = 
beta*1; norms {2, 1, 1}.
     build_index(make_norms_input(
                         /*doc_count=*/3, /*null_docids=*/ {2}, /*norms=*/ {2, 
1, 1},
                         {make_term("alpha", {{.docid = 0, .positions = {0, 2}},
                                              {.docid = 1, .positions = {0}}}),
                          make_term("beta", {{.docid = 2, .positions = {0}}})}),
                 &source_zero, reader::LogicalIndexOpenMode::kCompaction);
-    // 源 1:doc0 = alpha×1,doc1 = gamma×2 → norms {1, 2}
+    // Source 1: doc0 = alpha*1, doc1 = gamma*2; norms {1, 2}.
     build_index(make_norms_input(
                         /*doc_count=*/2, /*null_docids=*/ {}, /*norms=*/ {1, 
2},
                         {make_term("alpha", {{.docid = 0, .positions = {0}}}),
@@ -561,7 +562,8 @@ TEST(SniiIndexCompactionTest, 
NormsMergeMatchesRebuildAfterDeletesAndRemap) {
         assert_ok(compound->finish());
     }
 
-    // 目标 0 = [源0 doc0, 源1 doc0, 源1 doc1] → norms {2, 1, 2};目标 1 = [源0 doc2] → 
{1}
+    // Destination 0 = [source0 doc0, source1 doc0, source1 doc1], norms {2, 
1, 2}.
+    // Destination 1 = [source0 doc2], norms {1}.
     std::array<OpenedIndex, 2> rebuilt;
     build_index(make_norms_input(
                         /*doc_count=*/3, /*null_docids=*/ {}, /*norms=*/ {2, 
1, 2},
@@ -603,8 +605,9 @@ TEST(SniiIndexCompactionTest, 
NormsMergeMatchesRebuildAfterDeletesAndRemap) {
     EXPECT_EQ(merged_docs, rebuilt_docs);
 }
 
-// 老段(没有 norms 的 T2)参与 compaction 时同样能重建出 norms:这是生产升级后无需重建索引就
-// 获得打分能力的路径。空文档(无任何 token)编码为 1,超过 255 个 token 饱和到 255。
+// Compaction also rebuilds norms from older T2 segments without norms, 
enabling scoring after
+// a production upgrade without rebuilding indexes. Empty documents encode as 
1; token counts
+// above 255 saturate at 255.
 TEST(SniiIndexCompactionTest, 
NormsAreReconstructedFromLegacySourcesWithSaturation) {
     std::vector<uint32_t> long_positions(300);
     std::iota(long_positions.begin(), long_positions.end(), 0U);
@@ -678,7 +681,7 @@ TEST(SniiIndexCompactionTest, 
NormsMergeReclaimsResidentDictBeforeLargePlainTerm
     source_terms.push_back(make_term(
             "zeta", {{.docid = 0, .positions = make_positions(kPlainPositions, 
/*salt=*/99)}}));
 
-    // 单 doc 里的 token 远超 255,norm 饱和到 255。
+    // The document contains far more than 255 tokens, so its norm saturates 
at 255.
     SniiIndexInput source_input = make_norms_input(/*doc_count=*/1, 
/*null_docids=*/ {},
                                                    /*norms=*/ {255}, 
std::move(source_terms));
     source_input.target_dict_block_bytes = 1;
diff --git a/be/test/storage/index/snii/compaction/snii_posting_cursor_test.cpp 
b/be/test/storage/index/snii/compaction/snii_posting_cursor_test.cpp
index 9e05de65023..effcb6d90d0 100644
--- a/be/test/storage/index/snii/compaction/snii_posting_cursor_test.cpp
+++ b/be/test/storage/index/snii/compaction/snii_posting_cursor_test.cpp
@@ -57,7 +57,7 @@ static_assert(!std::is_constructible_v<
 
 constexpr uint64_t kIndexId = 9;
 constexpr std::string_view kIndexSuffix = "body";
-constexpr uint32_t kDocCount = 20000; // slim 项的 docid 间距不规则,最大约 1.6 万
+constexpr uint32_t kDocCount = 20000; // Slim docids have irregular gaps, 
reaching about 16,000.
 constexpr uint32_t kFreqDroppedDocCount = 65536;
 constexpr auto kDeleted = std::pair<uint32_t, uint32_t> 
{std::numeric_limits<uint32_t>::max(),
                                                          
std::numeric_limits<uint32_t>::max()};
@@ -95,8 +95,8 @@ Status build_source(std::vector<writer::TermPostings> terms, 
uint32_t doc_count,
     return fixture->segment.open_index(kIndexId, kIndexSuffix, 
&fixture->index);
 }
 
-// 500 个 docid,间距 1..62 不规则:dd 区(PFOR)超过 256B 的 inline 阈值,
-// 且 df < 512,仍是 slim pod_ref。docid 0 一定在列表里。
+// 500 docids with irregular gaps of 1..62: the PFOR dd region exceeds the 
256-byte inline
+// threshold, while df < 512 still selects a slim pod_ref. The list always 
includes docid 0.
 std::vector<uint32_t> slim_docids() {
     std::vector<uint32_t> docids;
     docids.reserve(500);
@@ -110,7 +110,8 @@ std::vector<uint32_t> slim_docids() {
     return docids;
 }
 
-// slim 项每个 doc 的位置只由 docid 决定(哈希取 1..97 个位置),测试据此重算期望。
+// Each slim document's positions depend only on its docid (hashed to 1..97 
positions), allowing
+// the test to recompute the expected values.
 std::vector<uint32_t> slim_positions(uint32_t docid) {
     uint32_t mixed = docid * 2654435761U;
     mixed ^= mixed >> 16;
diff --git 
a/be/test/storage/index/snii/compaction/snii_streamed_session_test.cpp 
b/be/test/storage/index/snii/compaction/snii_streamed_session_test.cpp
index ba424c5c6d3..eeb6d6e1f45 100644
--- a/be/test/storage/index/snii/compaction/snii_streamed_session_test.cpp
+++ b/be/test/storage/index/snii/compaction/snii_streamed_session_test.cpp
@@ -384,7 +384,7 @@ Status 
begin_scoring_session_from_local_input(SniiCompoundWriter* compound,
     input.config = format::IndexConfig::kDocsPositions;
     input.doc_count = 4;
     input.null_docids = {1, 3};
-    // 流式会话的 norms 只能晚绑定(compaction 在合并 postings 之后才知道每 doc 长度)。
+    // Streaming norms are bound late: compaction learns document lengths 
after merging postings.
     input.write_norms = true;
     return compound->begin_streamed_index(std::move(input), session);
 }
@@ -1125,14 +1125,15 @@ TEST(SniiStreamedWriterSessionTest, 
ActiveAndFinishedSessionLifecycleIsEnforced)
     EXPECT_FALSE(exists);
 }
 
-// A2:compaction 的目标会话先声明 write_norms,postings 合并完毕后才把重建出来的 norms 交给
-// 会话;必须恰好交付一次、长度等于 doc_count,finish 之前缺失则整个 compound 中毒。
+// A2: A compaction destination declares write_norms before merging postings, 
then receives the
+// rebuilt norms exactly once, with doc_count entries. Missing norms at finish 
poison the entire
+// compound writer.
 TEST(SniiStreamedWriterSessionTest, 
EncodedNormsAreLateBoundExactlyOnceBeforeFinish) {
     MemoryFile file;
     SniiCompoundWriter compound(&file);
     SniiStreamedIndexSession* session = nullptr;
     SniiIndexInput input = empty_input(105, "late_norms", /*doc_count=*/2);
-    // norms 需要词频(BM25 的 tf 来自位置计数,存储上要求 freq 区存在)。
+    // Norms require frequencies: BM25 tf comes from position counts and 
requires a freq region.
     input.write_norms = true;
     assert_ok(compound.begin_streamed_index(std::move(input), &session));
     ASSERT_NE(session, nullptr);
diff --git a/be/test/storage/index/snii/format/core_metadata_test.cpp 
b/be/test/storage/index/snii/format/core_metadata_test.cpp
index 809a9426da0..ce717fe9477 100644
--- a/be/test/storage/index/snii/format/core_metadata_test.cpp
+++ b/be/test/storage/index/snii/format/core_metadata_test.cpp
@@ -46,7 +46,7 @@ CoreMetadata sample_core(IndexConfig index_config = 
IndexConfig::kDocsOnly) {
                              .norms = {},
                              .null_bitmap = {.offset = 41, .length = 42},
                              .bsbf = {.offset = 51, .length = 52}};
-    // norms 只对带位置的段合法(BM25 的词频来自位置);docs-only 样本不带 norms。
+    // Norms require positions for BM25 frequencies; docs-only samples omit 
them.
     if (has_positions(index_config)) {
         metadata.section_refs.norms = {.offset = 31, .length = 32};
     }
@@ -109,7 +109,7 @@ void expect_core_eq(const CoreMetadata& expected, const 
CoreMetadata& actual) {
     EXPECT_EQ(expected.section_refs.bsbf.length, 
actual.section_refs.bsbf.length);
 }
 
-// docs-only 段不能带 norms(norms 需要位置):这是 A2 之后 core 元数据的一条硬约束。
+// Since A2, core metadata forbids norms on docs-only segments: norms require 
positions.
 TEST(SniiCoreMetadata, RejectsNormsOnDocsOnlyIndex) {
     auto metadata = sample_core();
     metadata.section_refs.norms = {.offset = 31, .length = 32};
@@ -132,11 +132,11 @@ TEST(SniiCoreMetadata, RoundTripsPositions) {
     expect_core_eq(expected, actual);
 }
 
-// CommonGrams 已删除:写过字段 4(CommonGrams 元数据)或字段 5(posting 策略)的段是墓碑,
-// 必须重建索引。生产 writer 从未写过这两个字段。
+// CommonGrams was removed: fields 4 (CommonGrams metadata) and 5 (posting 
policy) mark unsupported
+// segments that require an index rebuild. Production writers never emitted 
these fields.
 TEST(SniiCoreMetadata, RejectsLegacyCommonGramsMetadataFieldAsUnsupported) {
     auto payload = 
payload_of(encode(sample_core(IndexConfig::kDocsPositions)));
-    // 字段 4,length-delimited,任意内容。
+    // Field 4, length-delimited, with arbitrary contents.
     payload.push_back(static_cast<uint8_t>((4u << 3) | 2u));
     payload.push_back(3);
     payload.insert(payload.end(), {'c', 'g', '1'});
@@ -193,8 +193,9 @@ TEST(SniiCoreMetadata, RejectsUnsupportedIndexConfig) {
     EXPECT_TRUE(status.is<ErrorCode::INVERTED_INDEX_NOT_SUPPORTED>()) << 
status;
 }
 
-// master 开发期曾用 index_config=2 表示"带打分 tier";打分能力现在由 norms region 表达,
-// 该取值不再有意义,按不支持拒绝(从未进入生产,不存在兼容负担)。
+// During development on master, index_config=2 denoted a scoring tier. The 
norms region now
+// determines scoring capability, so reject this obsolete value. It was never 
deployed to
+// production and has no compatibility requirements.
 TEST(SniiCoreMetadata, RejectsLegacyScoringIndexConfigAsUnsupported) {
     auto payload = payload_of(encode(sample_core()));
     ByteSink field;
@@ -207,8 +208,8 @@ TEST(SniiCoreMetadata, 
RejectsLegacyScoringIndexConfigAsUnsupported) {
     EXPECT_TRUE(status.is<ErrorCode::INVERTED_INDEX_NOT_SUPPORTED>()) << 
status;
 }
 
-// 已上线的 3.1 系生产 writer 不写 stats.sum_total_term_freq 与
-// section_refs.norms:这种形态必须能打开,缺失的统计按 0、norms 按空 region 处理。
+// The deployed 3.1-series writer omits stats.sum_total_term_freq and 
section_refs.norms. Such
+// segments must remain readable, with missing statistics set to 0 and an 
empty norms region.
 TEST(SniiCoreMetadata, AcceptsProductionShapeWithoutNormsAndTotalTermFreq) {
     const auto metadata = sample_core(IndexConfig::kDocsPositions);
     const auto payload = mutate_core_payload(metadata, [](auto* core) {
@@ -226,7 +227,7 @@ TEST(SniiCoreMetadata, 
AcceptsProductionShapeWithoutNormsAndTotalTermFreq) {
     EXPECT_EQ(actual.section_refs.bsbf.length, 
metadata.section_refs.bsbf.length);
 }
 
-// 没有 norms 的段编码时不写 section_refs.norms(与生产 writer 的字节形态一致)。
+// Segments without norms omit section_refs.norms, matching the production 
writer's bytes.
 TEST(SniiCoreMetadata, OmitsEmptyNormsRefOnEncode) {
     auto metadata = sample_core(IndexConfig::kDocsPositions);
     metadata.section_refs.norms = {};
@@ -244,7 +245,7 @@ TEST(SniiCoreMetadata, OmitsEmptyNormsRefOnEncode) {
 
 TEST(SniiCoreMetadata, RejectsMissingEachStatsField) {
     const auto metadata = sample_core();
-    // sum_total_term_freq 是可选字段(生产 writer 不写),不在必填之列。
+    // sum_total_term_freq is optional and omitted by the production writer.
     for (const auto clear : std::array<void (doris::snii::SniiStatsPB::*)(), 
4> {
                  &doris::snii::SniiStatsPB::clear_doc_count,
                  &doris::snii::SniiStatsPB::clear_indexed_doc_count,
diff --git a/be/test/storage/index/snii/query/phrase_prefix_query_test.cpp 
b/be/test/storage/index/snii/query/phrase_prefix_query_test.cpp
index 0fd31b5d80f..2ce64fb114e 100644
--- a/be/test/storage/index/snii/query/phrase_prefix_query_test.cpp
+++ b/be/test/storage/index/snii/query/phrase_prefix_query_test.cpp
@@ -333,15 +333,15 @@ Corpus BuildCrossWindowTailCorpus() {
 // is byte-wise, matching the index's dict enumeration order.
 Corpus BuildCjkTailCorpus() {
     Corpus c;
-    const char* const tails[] = {"\xE7\xBB\x93\xE6\x9E\x9C\xE7\x94\xB2",  // 
结果甲
-                                 "\xE7\xBB\x93\xE6\x9E\x9C\xE4\xB9\x99",  // 
结果乙
-                                 "\xE7\xBB\x93\xE6\x9E\x9C\xE4\xB8\x99"}; // 
结果丙
-    const std::string lead = "\xE8\xBF\x9E\xE6\x8E\xA5";                  // 连接
+    const char* const tails[] = {"\xE7\xBB\x93\xE6\x9E\x9C\xE7\x94\xB2",  // 
Result A
+                                 "\xE7\xBB\x93\xE6\x9E\x9C\xE4\xB9\x99",  // 
Result B
+                                 "\xE7\xBB\x93\xE6\x9E\x9C\xE4\xB8\x99"}; // 
Result C
+    const std::string lead = "\xE8\xBF\x9E\xE6\x8E\xA5";                  // 
Join
     c.docs.resize(120);
     for (uint32_t d = 0; d < c.docs.size(); ++d) {
         if (d % 20 == 3) {
             c.docs[d] = {lead, "\xE9\x97\xB4\xE9\x9A\x94",
-                         tails[d % 3]}; // 间隔 filler, not adjacent
+                         tails[d % 3]}; // Separating filler, not adjacent
         } else {
             c.docs[d] = {lead, tails[d % 3]};
         }
@@ -966,8 +966,8 @@ TEST(SniiPhrasePrefixMerge, CjkUnicodeTailsMatchOracle) {
     SniiSegmentReader segment;
     LogicalIndexReader idx = OpenIndex(&file, &segment, path);
 
-    const std::vector<std::string> terms = {"\xE8\xBF\x9E\xE6\x8E\xA5",  // 连接
-                                            "\xE7\xBB\x93\xE6\x9E\x9C"}; // 结果
+    const std::vector<std::string> terms = {"\xE8\xBF\x9E\xE6\x8E\xA5",  // 
Join
+                                            "\xE7\xBB\x93\xE6\x9E\x9C"}; // 
Result
     std::vector<uint32_t> got;
     ASSERT_TRUE(query::phrase_prefix_query(idx, terms, &got).ok());
     EXPECT_TRUE(std::ranges::is_sorted(got));
diff --git a/be/test/storage/index/snii/query/query_operator_error_test.cpp 
b/be/test/storage/index/snii/query/query_operator_error_test.cpp
index 7773fab90af..4083c222f8b 100644
--- a/be/test/storage/index/snii/query/query_operator_error_test.cpp
+++ b/be/test/storage/index/snii/query/query_operator_error_test.cpp
@@ -128,7 +128,7 @@ void BuildIndexBytes(const Corpus& corpus, 
doris::snii::format::IndexConfig conf
     in.config = config;
     in.doc_count = static_cast<uint32_t>(corpus.docs.size());
     if (config == doris::snii::format::IndexConfig::kDocsPositions) {
-        // 带位置的索引一并写 norms(A2),让打分查询在这些边界用例里也可用。
+        // Indexes with positions also write norms (A2), enabling scoring in 
these edge cases.
         in.encoded_norms.assign(corpus.docs.size(), 1);
     }
     in.terms = buf.finalize_sorted();
diff --git a/be/test/storage/index/snii/query/query_profile_test.cpp 
b/be/test/storage/index/snii/query/query_profile_test.cpp
index 4a44493c43d..ec8d32db353 100644
--- a/be/test/storage/index/snii/query/query_profile_test.cpp
+++ b/be/test/storage/index/snii/query/query_profile_test.cpp
@@ -206,7 +206,7 @@ void WriteCorpus(const Corpus& c, const std::string& path, 
int prx_zstd_level =
     in.index_suffix = "body";
     in.config = doris::snii::format::IndexConfig::kDocsPositions;
     in.doc_count = static_cast<uint32_t>(c.docs.size());
-    // 分词 + 带位置的索引一律带 norms(A2),这样才能打分。
+    // Analyzed indexes with positions always include norms (A2) to enable 
scoring.
     in.encoded_norms.assign(c.docs.size(), 1);
     in.terms = buf.finalize_sorted();
     in.target_dict_block_bytes = 512;
diff --git a/be/test/storage/index/snii/snii_golden_corpus_test.cpp 
b/be/test/storage/index/snii/snii_golden_corpus_test.cpp
index dadead11106..aaabaeefe97 100644
--- a/be/test/storage/index/snii/snii_golden_corpus_test.cpp
+++ b/be/test/storage/index/snii/snii_golden_corpus_test.cpp
@@ -15,20 +15,20 @@
 // specific language governing permissions and limitations
 // under the License.
 
-// SNII 黄金语料回读测试(跨版本 / 跨改动的查询语义锁定)。
+// SNII golden-corpus readback test: preserve query semantics across versions 
and changes.
 //
-// 用途:把一组「普通」SNII 段(无 CommonGrams、无打分扩展)用生产写入栈写到磁盘,
-// 并把每条查询在写入时的结果(docid 集合、null bitmap、状态码)记成期望文件;
-// 之后任何改动(删 CommonGrams、格式对齐、norms 写入……)都必须能原样回读这些段,
-// 且每条查询结果逐条一致。段文件也可以由其它版本的 writer(例如生产分支)写出,
-// 只要目录里有同名的 .expect 即可校验。
+// Write ordinary SNII segments without CommonGrams or scoring extensions 
through the production
+// writer stack. Record each query's docids, null bitmap, and status codes as 
expectations.
+// Later changes, including CommonGrams removal, format alignment, or norms 
support, must preserve
+// readback and query results. Other writer versions, such as a production 
branch, may also supply
+// segment files with matching .expect files in the same directory.
 //
-//   SNII_GOLDEN_DIR=<dir> SNII_GOLDEN_MODE=write   写段 + 写期望
-//   SNII_GOLDEN_DIR=<dir>                          回读校验(默认)
-//   未设置 SNII_GOLDEN_DIR                          跳过
+//   SNII_GOLDEN_DIR=<dir> SNII_GOLDEN_MODE=write   Write segments and 
expectations.
+//   SNII_GOLDEN_DIR=<dir>                          Verify readback (default).
+//   SNII_GOLDEN_DIR unset                          Skip the test.
 //
-// 校验模式下每条查询跑三遍:结果缓存关闭(冷)、缓存开启(冷)、缓存开启(热),
-// 三遍都必须等于期望。
+// Verification runs each query with the result cache disabled, enabled but 
cold, and warm.
+// All three results must match the expectations.
 
 #include <gen_cpp/PaloInternalService_types.h>
 #include <gtest/gtest.h>
@@ -69,30 +69,30 @@ namespace {
 constexpr int64_t kIndexId = 7001;
 constexpr const char* kColumn = "c1";
 
-// ---------------------------------------------------------------- 语料
+// ---------------------------------------------------------------- Corpus
 
 using ScalarRow = std::optional<std::string>; // nullopt = NULL
 
 std::vector<ScalarRow> scalar_corpus() {
     std::vector<ScalarRow> rows;
-    rows.emplace_back("hello world hello doris");                              
       // 0
-    rows.emplace_back("");                                                     
       // 1 空串
-    rows.emplace_back("The QUICK brown-fox; jumped!! over_the lazy dog 42 
times");    // 2
+    rows.emplace_back("hello world hello doris");                              
    // 0
+    rows.emplace_back("");                                                     
    // 1 Empty string
+    rows.emplace_back("The QUICK brown-fox; jumped!! over_the lazy dog 42 
times"); // 2
     rows.emplace_back("重复 重复 重复 词元 Doris 数据库 全文检索 mixed 中英 tokens"); // 3
     rows.emplace_back(std::nullopt);                                           
       // 4 NULL
     rows.emplace_back(std::nullopt);                                           
       // 5 NULL
     rows.emplace_back(std::nullopt);                                           
       // 6 NULL
     rows.emplace_back(std::string(300, 'x'));                      // 7 > 
ignore_above
     rows.emplace_back("single");                                   // 8
-    rows.emplace_back("!!! ??? ,,,");                              // 9 无词元
+    rows.emplace_back("!!! ??? ,,,");                              // 9 No 
tokens
     rows.emplace_back("hello world again and again and again");    // 10
     rows.emplace_back(std::nullopt);                               // 11 NULL
-    rows.emplace_back(std::string("\x1f") + "hidden term inside"); // 12 
内部命名空间前缀
-    rows.emplace_back(std::string("\x1e") + "escaped start");      // 13 转义前缀
+    rows.emplace_back(std::string("\x1f") + "hidden term inside"); // 12 
Control-byte prefix
+    rows.emplace_back(std::string("\x1e") + "escaped start");      // 13 
Escape control byte
     rows.emplace_back("prefix prefixes prefixing prefab");         // 14
     rows.emplace_back("alpha beta gamma alpha beta alpha");        // 15
     {
-        std::string long_doc; // 16 超过 255 个词元(norm 饱和)
+        std::string long_doc; // 16 More than 255 tokens (norm saturation)
         for (int i = 0; i < 300; ++i) long_doc += (i ? " tok" : "tok");
         rows.emplace_back(std::move(long_doc));
     }
@@ -162,7 +162,7 @@ struct Sample {
     std::string name;
     std::map<std::string, std::string> properties;
     bool keyword_lane;
-    bool array = false; // ARRAY<STRING> 列:经 add_array_values / 
add_array_nulls 写入
+    bool array = false; // ARRAY<STRING>, written via add_array_values / 
add_array_nulls.
 };
 
 std::vector<Sample> samples() {
@@ -185,27 +185,27 @@ std::vector<Sample> samples() {
     };
 }
 
-// ARRAY 语料:每行是若干元素;nullopt 行 = 整行 NULL;元素级 NULL 用 std::nullopt 元素表示。
+// ARRAY corpus: an absent row is SQL NULL; absent elements represent 
element-level NULLs.
 using ArrayRow = std::optional<std::vector<std::optional<std::string>>>;
 
 std::vector<ArrayRow> array_corpus() {
     std::vector<ArrayRow> rows;
     rows.emplace_back(std::vector<std::optional<std::string>> {"hello world", 
"hello doris"}); // 0
-    rows.emplace_back(std::vector<std::optional<std::string>> {});         // 
1 空数组
-    rows.emplace_back(std::nullopt);                                       // 
2 NULL 行
+    rows.emplace_back(std::vector<std::optional<std::string>> {});         // 
1 Empty array
+    rows.emplace_back(std::nullopt);                                       // 
2 NULL row
     rows.emplace_back(std::vector<std::optional<std::string>> {"single"}); // 3
-    rows.emplace_back(std::vector<std::optional<std::string>> {"alpha beta", 
std::nullopt,
-                                                               "gamma 
alpha"}); // 4 含元素 NULL
-    rows.emplace_back(
-            std::vector<std::optional<std::string>> {"world", "hello"}); // 5 
跨元素不成短语
+    rows.emplace_back(std::vector<std::optional<std::string>> {
+            "alpha beta", std::nullopt, "gamma alpha"}); // 4 Contains a NULL 
element
+    rows.emplace_back(std::vector<std::optional<std::string>> {
+            "world", "hello"}); // 5 Phrases do not span elements
     rows.emplace_back(std::vector<std::optional<std::string>> {"重复 词元", "Doris 
数据库"}); // 6
-    rows.emplace_back(std::nullopt); // 7 NULL 行
+    rows.emplace_back(std::nullopt); // 7 NULL row
     rows.emplace_back(std::vector<std::optional<std::string>> {"prefix 
prefixes", "",
                                                                "hello world 
hello doris"}); // 8
     return rows;
 }
 
-// ---------------------------------------------------------------- 工具
+// ---------------------------------------------------------------- Helpers
 
 TabletIndex make_meta(const Sample& sample) {
     TabletIndexPB pb;
@@ -281,8 +281,8 @@ struct Observation {
     InvertedIndexQueryType type;
     std::string text;
     int status_code = 0;
-    std::string docids;      // 逗号分隔
-    std::string null_docids; // 逗号分隔
+    std::string docids;      // Comma-separated.
+    std::string null_docids; // Comma-separated.
     int null_status_code = 0;
 
     std::string line() const {
@@ -319,7 +319,7 @@ std::optional<Observation> parse_line(const std::string& 
sample, const std::stri
     return o;
 }
 
-// 查询上下文:仿生产的 IndexQueryContext 装配(enable_query_cache 可选)。
+// Assemble IndexQueryContext as in production, with optional query-cache 
support.
 struct QueryEnv {
     explicit QueryEnv(bool enable_query_cache) {
         TQueryOptions options;
@@ -337,7 +337,7 @@ struct QueryEnv {
     IndexQueryContextPtr context = std::make_shared<IndexQueryContext>();
 };
 
-// ---------------------------------------------------------------- 写入
+// ---------------------------------------------------------------- Writing
 
 Status write_sample(const std::string& dir, const Sample& sample, const 
TabletIndex& meta) {
     const std::string prefix = dir + "/" + sample.name;
@@ -354,8 +354,8 @@ Status write_sample(const std::string& dir, const Sample& 
sample, const TabletIn
     SniiIndexColumnWriter writer(&index_file_writer, &meta, 
FieldType::OLAP_FIELD_TYPE_VARCHAR);
     RETURN_IF_ERROR(writer.init());
     if (sample.array) {
-        // 仿 ArrayColumnWriter::append_nullable:所有行(含 NULL 行)都喂给 
add_array_values,
-        // NULL 行是空数组;行级 NULL 再通过 add_array_nulls 声明。
+        // Follow ArrayColumnWriter::append_nullable: pass all rows to 
add_array_values,
+        // representing NULL rows as empty arrays, then mark them with 
add_array_nulls.
         const auto rows = array_corpus();
         std::vector<std::string> storage;
         std::vector<uint8_t> element_nulls;
@@ -409,7 +409,7 @@ Status write_sample(const std::string& dir, const Sample& 
sample, const TabletIn
     return Status::OK();
 }
 
-// ---------------------------------------------------------------- 读取 + 查询
+// ---------------------------------------------------------------- Reading 
and querying
 
 struct OpenedSample {
     std::shared_ptr<IndexFileReader> file_reader;
@@ -459,7 +459,7 @@ protected:
     void SetUp() override {
         const char* dir = std::getenv("SNII_GOLDEN_DIR");
         if (dir == nullptr || *dir == '\0') {
-            GTEST_SKIP() << "SNII_GOLDEN_DIR 未设置";
+            GTEST_SKIP() << "SNII_GOLDEN_DIR is not set";
         }
         _dir = dir;
         const char* mode = std::getenv("SNII_GOLDEN_MODE");
@@ -506,16 +506,16 @@ TEST_F(SniiGoldenCorpus, WriteOrVerify) {
             continue;
         }
 
-        // 校验模式:期望文件缺失 = 该样本不存在(可能由其它版本 writer 未写出),跳过并提示。
+        // Report missing expectations as a failure, then continue checking 
the remaining samples.
         std::ifstream in(expect_path, std::ios::binary);
         if (!in.good()) {
-            ADD_FAILURE() << "缺少期望文件: " << expect_path;
+            ADD_FAILURE() << "Missing expectation file: " << expect_path;
             continue;
         }
         OpenedSample opened;
         const Status os = open_sample(_dir, sample, meta, &opened);
         if (!os.ok()) {
-            ADD_FAILURE() << sample.name << " 打开失败: " << os.to_string();
+            ADD_FAILURE() << sample.name << " failed to open: " << 
os.to_string();
             ++mismatches;
             continue;
         }
@@ -524,7 +524,8 @@ TEST_F(SniiGoldenCorpus, WriteOrVerify) {
         while (std::getline(in, line)) {
             if (line.empty() || line[0] == '#') continue;
             auto parsed = parse_line(sample.name, line);
-            ASSERT_TRUE(parsed.has_value()) << "期望文件格式错误: " << expect_path << 
": " << line;
+            ASSERT_TRUE(parsed.has_value())
+                    << "Invalid expectation file format: " << expect_path << 
": " << line;
             expected.push_back(*parsed);
         }
         for (const Observation& want : expected) {
@@ -540,14 +541,14 @@ TEST_F(SniiGoldenCorpus, WriteOrVerify) {
                                   got->null_docids == want.null_docids;
                 if (!same) {
                     ++mismatches;
-                    ADD_FAILURE() << want.describe() << "\n  期望: " << 
want.line()
-                                  << "\n  实际: " << got->line();
+                    ADD_FAILURE() << want.describe() << "\n  Expected: " << 
want.line()
+                                  << "\n  Actual: " << got->line();
                 }
             }
         }
     }
     if (!_write_mode) {
-        EXPECT_EQ(mismatches, 0) << "共校验 " << checked << " 条";
+        EXPECT_EQ(mismatches, 0) << "Checked " << checked << " observations";
         std::cout << "[golden] checked=" << checked << " mismatches=" << 
mismatches << std::endl;
     }
 }
diff --git a/be/test/storage/index/snii/writer/phase_a_readback_test.cpp 
b/be/test/storage/index/snii/writer/phase_a_readback_test.cpp
index 5e841157cce..471d0d04d09 100644
--- a/be/test/storage/index/snii/writer/phase_a_readback_test.cpp
+++ b/be/test/storage/index/snii/writer/phase_a_readback_test.cpp
@@ -114,7 +114,7 @@ SniiIndexInput MakeIndex(const Corpus& c) {
     SniiIndexInput in;
     in.index_id = 1;
     in.index_suffix = "body";
-    in.config = IndexConfig::kDocsPositions; // norms 随 encoded_norms 一起写出
+    in.config = IndexConfig::kDocsPositions; // Norms are written from 
encoded_norms.
     in.doc_count = c.doc_count;
     in.target_dict_block_bytes = 1; // one block per term
     in.encoded_norms.resize(c.doc_count);
diff --git a/be/test/storage/index/snii/writer/snii_compound_writer_test.cpp 
b/be/test/storage/index/snii/writer/snii_compound_writer_test.cpp
index 6c9dd56d1f7..1fd58df5042 100644
--- a/be/test/storage/index/snii/writer/snii_compound_writer_test.cpp
+++ b/be/test/storage/index/snii/writer/snii_compound_writer_test.cpp
@@ -1537,9 +1537,10 @@ void ExpectFixtureReadable(const std::vector<uint8_t>& 
file) {
 
 } // namespace
 
-// 夹具尺寸随格式演进漂移(core 元数据字段的增减就会改动几十字节)。从"跨 2*kMinPaddingLeverage
-// 个块"的导出值往下找第一个"确实要补齐、且补齐便宜(2*pad < block)"的块大小:块越小跨的块越多,
-// 阈值门只会更宽松,所以门不会成为这些用例的决定因素。
+// Fixture sizes vary with the format; core-metadata fields alone can shift 
them by dozens of
+// bytes. Starting from a size spanning 2*kMinPaddingLeverage blocks, search 
downward for a block
+// size that needs padding and satisfies 2*pad < block. Smaller blocks only 
relax the block-count
+// threshold, keeping that threshold from determining these test results.
 int64_t CheapPaddingBlockSize(size_t unpadded) {
     auto block = static_cast<int64_t>(unpadded / (2 * kMinPaddingLeverage));
     while (block >= 2) {
diff --git a/be/test/storage/index/snii/writer/spill_run_codec_test.cpp 
b/be/test/storage/index/snii/writer/spill_run_codec_test.cpp
index 5f1be4a0670..dbefdc7f77b 100644
--- a/be/test/storage/index/snii/writer/spill_run_codec_test.cpp
+++ b/be/test/storage/index/snii/writer/spill_run_codec_test.cpp
@@ -358,7 +358,7 @@ TEST(SniiSpillRunCodec, 
RunWriterStreamsWideTermWithinAccountedBound) {
     TempRun run;
     std::vector<uint32_t> docids(kDocs);
     std::iota(docids.begin(), docids.end(), 0U);
-    // 生产 SPIMI 落盘的 postings 一定带 freqs(to_postings 总会填充),run 记录没有无频次形状。
+    // SPIMI spill postings always contain freqs populated by to_postings; run 
records require them.
     TermPostings postings = MakeTerm(std::move(docids), 
std::vector<uint32_t>(kDocs, 1));
 
     int64_t observed = 0;
diff --git a/be/test/storage/index/snii_query_test.cpp 
b/be/test/storage/index/snii_query_test.cpp
index e16b7c577a4..4ccf4628462 100644
--- a/be/test/storage/index/snii_query_test.cpp
+++ b/be/test/storage/index/snii_query_test.cpp
@@ -901,8 +901,9 @@ struct OpaqueMatcherPlanCase {
     std::vector<bool> common;
 };
 
-// 内部命名空间(\x1f 开头)里的不透明词项:这些用例直接驱动 phrase 计划执行器,
-// 验证"覆盖两个位置的不透明词项"与普通位置匹配的等价性,与任何具体分析器无关。
+// Opaque terms in the internal namespace (starting with \x1f). These tests 
drive the phrase-plan
+// executor directly to verify that an opaque term spanning two positions 
matches ordinary
+// positions equivalently, independently of any particular analyzer.
 constexpr std::string_view kOpaqueInternalTermPrefix =
         "\x1f"
         "SNII_TEST_OPAQUE"
diff --git a/be/test/storage/index/snii_spimi_intern_test.cpp 
b/be/test/storage/index/snii_spimi_intern_test.cpp
index 54381a1afb4..756f8c9f2a4 100644
--- a/be/test/storage/index/snii_spimi_intern_test.cpp
+++ b/be/test/storage/index/snii_spimi_intern_test.cpp
@@ -110,7 +110,7 @@ TEST(SniiSpimiTermBufferTest, 
InternHashUsesFastStringViewHash) {
 }
 
 TEST(SniiSpimiTermBufferTest, OrdinaryDocsOnlyMarkerTermRetainsFrequency) {
-    // 落在内部命名空间(\x1f 开头)里的普通词项也只是普通词项:照常记录频次。
+    // Ordinary terms starting with the internal \x1f namespace still record 
frequencies normally.
     const std::string literal_marker_term =
             std::string("\x1f") + "SNII_TEST_MARKER\x1f" + "literal";
     SpimiTermBuffer ordinary(/*has_positions=*/false);
diff --git a/be/test/storage/index/snii_writer_test.cpp 
b/be/test/storage/index/snii_writer_test.cpp
index 77f08840065..2451c8aa578 100644
--- a/be/test/storage/index/snii_writer_test.cpp
+++ b/be/test/storage/index/snii_writer_test.cpp
@@ -96,8 +96,8 @@ void init_failure_index_meta(doris::TabletIndex* index_meta, 
int64_t index_id) {
     index_meta->init_from_pb(index_pb);
 }
 
-// 分词过程中抛 INVERTED_INDEX_ANALYZER_ERROR 的分析器:模拟任何 token filter 的运行期失败
-// (以前由某个词元过滤器的 UTF-8 校验扮演这个角色)。
+// An analyzer that throws INVERTED_INDEX_ANALYZER_ERROR to simulate a token 
filter's runtime
+// failure. Previously, a token filter's UTF-8 validation provided this 
failure path.
 class ThrowingTokenStream final : public lucene::analysis::TokenStream {
 public:
     lucene::analysis::Token* next(lucene::analysis::Token*) override {
diff --git 
a/fe/fe-core/src/main/java/org/apache/doris/indexpolicy/IndexPolicy.java 
b/fe/fe-core/src/main/java/org/apache/doris/indexpolicy/IndexPolicy.java
index 1630442c63e..31d51a06c99 100644
--- a/fe/fe-core/src/main/java/org/apache/doris/indexpolicy/IndexPolicy.java
+++ b/fe/fe-core/src/main/java/org/apache/doris/indexpolicy/IndexPolicy.java
@@ -123,8 +123,8 @@ public class IndexPolicy implements Writable, 
GsonPostProcessable {
                 GsonUtils.GSON.toJson(this.properties));
     }
 
-    // 已从 BE 删除、但老版本镜像/edit log 里可能仍持久化着的 token filter 类型。
-    // 这类策略可以被加载(否则 FE 起不来),但不可再被 analyzer 引用。
+    // Token filter types removed from BE but possibly retained in older 
images or edit logs.
+    // Load these policies so FE can start, but reject analyzers that 
reference them.
     public static final Set<String> LEGACY_UNSUPPORTED_TOKEN_FILTER_TYPES =
             ImmutableSet.of("common_grams");
 
diff --git 
a/fe/fe-core/src/main/java/org/apache/doris/indexpolicy/IndexPolicyMgr.java 
b/fe/fe-core/src/main/java/org/apache/doris/indexpolicy/IndexPolicyMgr.java
index 9f71a140c13..b547c482d81 100644
--- a/fe/fe-core/src/main/java/org/apache/doris/indexpolicy/IndexPolicyMgr.java
+++ b/fe/fe-core/src/main/java/org/apache/doris/indexpolicy/IndexPolicyMgr.java
@@ -117,9 +117,10 @@ public class IndexPolicyMgr implements Writable, 
GsonPostProcessable {
     }
 
     /**
-     * 安全网:老版本可能持久化了 BE 已不再支持的 token filter 类型(例如已删除的 common_grams)。
-     * 这类策略仍会被加载(不能让 FE 因为镜像里的一条策略起不来),但任何引用它的 analyzer
-     * 都必须在使用时被明确拒绝,而不是等到 BE 建索引/查询时才报"未知 token filter"。
+     * Older metadata may retain token filter types no longer supported by BE, 
such as common_grams.
+     * Load these policies so a single obsolete policy cannot prevent FE 
startup, but reject any
+     * analyzer that references them at use time, before BE reports an unknown 
token filter during
+     * index construction or querying.
      */
     private void validateReferencedTokenFiltersUsableLocked(String 
analyzerName, IndexPolicy analyzer)
             throws DdlException {
diff --git a/gensrc/proto/snii.proto b/gensrc/proto/snii.proto
index 188ddc9e5ea..1dc6793b1e2 100644
--- a/gensrc/proto/snii.proto
+++ b/gensrc/proto/snii.proto
@@ -77,9 +77,9 @@ message SniiCoreMetadataPB {
     optional uint32 index_config = 1;
     optional SniiStatsPB stats = 2;
     optional SniiSectionRefsPB section_refs = 3;
-    // 字段 4/5 曾是 CommonGrams 元数据(功能已删除)。字段号保留、类型降级成不透明
-    // 字节 / 整数:reader 只用它们识别"这段是用 CommonGrams 写出来的"并拒绝读取(墓碑),
-    // 不再解析内容。旧 reader(生产 4.1.7 系)本来就忽略这两个字段。
+    // Fields 4/5 belonged to the removed CommonGrams feature. Keep their 
numbers as opaque
+    // bytes/integers so readers can identify and reject CommonGrams segments 
without parsing
+    // the contents. Older production 4.1.7-series readers already ignore 
these fields.
     optional bytes legacy_common_grams = 4;
     optional uint32 legacy_common_grams_posting_policy = 5;
 }


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to