github-actions[bot] commented on code in PR #68646:
URL: https://github.com/apache/doris/pull/68646#discussion_r4219437320
##########
fe/fe-core/src/main/java/org/apache/doris/mtmv/MTMVRelationManager.java:
##########
@@ -349,34 +360,221 @@ public void dropTable(Table table) {
// because a dropped table is the one change whose query is gone
beyond doubt. What the two record
// is the same state either way. Unlike a rename it stays an
invalidation: the table is gone for
// good, so the state is not something a later alter can make obsolete.
- processBaseTableChange(new BaseTableInfo(table), "The base table has
been deleted:", false);
+ processBaseTableChange(new BaseTableInfo(table), "The base table has
been deleted:", null);
}
/**
* update mtmv status to `SCHEMA_CHANGE`.
*
* @param isReplace
+ * @param queryJudgedColumns the names the alter gives the table or takes
away from it, which leave the
+ * judgement about each MV's state to that MV's
own query, or null when the
+ * alter is not one a query decides. The names
are carried rather than judged
+ * before the call because the judgement is
about them; see
+ * {@code AlterOp#queryJudgedColumnNames} for
which operations name one, and
+ * {@link #invalidateMvUnlessQueryHolds} for
what is asked about it. A rename
+ * of the base table names no column: it is left
to the record below, which
+ * says what the MV that keeps spelling the old
name needs to hear
*/
@Override
- public void alterTable(BaseTableInfo oldTableInfo, Optional<BaseTableInfo>
newTableInfo, boolean isReplace) {
+ public void alterTable(BaseTableInfo oldTableInfo, Optional<BaseTableInfo>
newTableInfo, boolean isReplace,
+ QueryJudgedChange queryJudgedChange) {
// when replace, need deal two table
if (isReplace) {
// REPLACE TABLE already invalidates the IVM baseline explicitly,
see Alter#processReplaceTable
- processBaseTableChange(newTableInfo.get(), "The base table has
been updated:", false);
+ processBaseTableChange(newTableInfo.get(), "The base table has
been updated:", null);
}
- boolean renamed = !isReplace && newTableInfo.isPresent()
- && !Objects.equals(oldTableInfo.getTableName(),
newTableInfo.get().getTableName());
- // A rename is the one change whose query check is skipped: the MV
query keeps spelling the old
- // name, so it is unanalyzable by construction, and the reason it
would be invalidated with --
- // "the query is no longer analyzable" -- says less than the message
this call records anyway.
- boolean checkQueryUsable = !renamed;
- processBaseTableChange(oldTableInfo, "The base table has been
updated:", checkQueryUsable);
+ processBaseTableChange(oldTableInfo, "The base table has been
updated:", queryJudgedChange);
}
/**
- * An MV's query is only as good as the base table schema it was analyzed
against. Re-analyzing the
- * MV query here (right after the alter was applied) is what detects a
changed column identity:
+ * Whether the query, as it is analysed now, reads a column of any of
these names, and reads it where
+ * the change can reach it.
+ *
+ * <p>There are two places a name is the change's to answer for. One is a
column of the table the change
+ * is about: that is the column this view's rows were computed from, and
the names are matched
+ * case-insensitively because a name is what moves. The other is a column
the query reaches across a
+ * scope boundary -- the plan records those on the Apply that stands for
the subquery, whose correlation
+ * slots are the outer columns its right side reads -- because such a name
is the scopes' to answer for
+ * rather than the query's: the nearest column to the reference answers
for it, so a column the change
+ * takes away from a scope inside leaves the name to one outside, and a
column it gives to a scope inside
+ * takes the name over. A name reached with the qualifier of another table
inside the query's own scope
+ * is neither: no later change can move it, so one to a column it does not
name is one this view's rows
+ * do not depend on.
+ */
+ private static boolean reachesAnyColumnOf(Plan plan, BaseTableInfo
baseTableInfo, Set<String> columnNames) {
+ if (plan == null) {
+ // A query whose plan was not kept is one this cannot be answered
about, and "it does" is the
+ // answer that keeps the view safe.
+ return true;
+ }
+ Set<String> names = Sets.newTreeSet(String.CASE_INSENSITIVE_ORDER);
+ names.addAll(columnNames);
+ LineageInfo lineage = LineageInfoExtractor.extractLineageInfo(plan);
+ for (SetMultimap<?, Expression> byType :
lineage.getDirectLineageMap().values()) {
+ if (reachesAnyColumn(byType.values(), names, baseTableInfo)) {
+ return true;
+ }
+ }
+ // The dataset predicates once, not once per output column: the
per-output copy of them the lineage
+ // also offers holds the same expressions for every column the query
produces, and scanning it would
+ // visit each of them once per column.
+ if (reachesAnyColumn(lineage.getDatasetIndirectLineageMap().values(),
names, baseTableInfo)) {
+ return true;
+ }
+ if (reachesAnyColumnOfASubquery(plan, names, baseTableInfo)) {
+ return true;
+ }
+ return reachesAnyColumnAcrossScopes(plan, lineage, names,
baseTableInfo);
+ }
+
+ /** Whether this slot is a column of this table, through whatever views
stand between the two. */
+ private static boolean isColumnOf(Slot slot, BaseTableInfo baseTableInfo) {
+ if (!(slot instanceof SlotReference)) {
+ return false;
+ }
+ return ((SlotReference) slot).getOriginalTable()
+ .map(table -> new BaseTableInfo(table).equals(baseTableInfo))
+ .orElse(false);
+ }
+
+ /**
+ * Whether a name the change is about is answered for inside a subquery,
out of that subquery's own
+ * scope.
+ *
+ * <p>This is the one place a name can move without any column the view
produces depending on it: the
+ * scope of a subquery is internal, so which column answers for a name
there changes what the query
+ * returns -- a row, or none -- while every column of the view stays the
one it was. The lineage of the
+ * view's columns does not reach it, so the scope the subquery became is
read here, expression by
+ * expression, the way the lineage is read for the view's own.
+ *
+ * <p>Two things are read. One is a value the subquery itself names -- an
expression of its own under one
+ * of these names, rather than a column of a table -- because that is what
a name the change takes away
+ * falls back to, and it decides the rows whether the subquery is a
predicate or a value. It is read only
+ * where the table the change is about is one that subquery reads: what a
name falls back to is what the
+ * scope that answered for it holds, so a scope that does not read the
table holds nothing for the name
+ * and one of its own is one the change never reached. The other is a
column of the table the change is
+ * about, which decides the rows only when the subquery's output is one
the query reads: an EXISTS tests
+ * the rows of its subquery and not what it projects, so a name it
projects and never compares is one
+ * this view's rows do not depend on.
+ *
+ * <p>Each scope is read on its own. A subquery inside one of these is a
scope of its own, and it is
+ * judged where it is read and not as a part of its enclosing one: its
projection is held only where
+ * that scope's own output is read, so an EXISTS inside an IN is still not
compared by the IN.
+ */
+ private static boolean reachesAnyColumnOfASubquery(Plan plan, Set<String>
names,
+ BaseTableInfo baseTableInfo) {
+ for (LogicalApply<?, ?> apply :
plan.<LogicalApply>collectToList(LogicalApply.class::isInstance)) {
+ List<Plan> itsOwnScope = ownScopeOf(apply);
+ boolean outputDecidesRows = !apply.isExist();
+ boolean isOneOfItsTables = readsAnyTableOf(itsOwnScope,
baseTableInfo);
+ for (Plan node : itsOwnScope) {
+ for (Expression expression : node.getExpressions()) {
+ if ((isOneOfItsTables &&
readsAnyNameTheSubqueryAnswersFor(expression, names))
+ || (outputDecidesRows &&
reachesAnyColumn(expression, names, baseTableInfo))) {
+ return true;
+ }
+ }
+ }
+ }
+ return false;
+ }
+
+ /**
+ * The nodes of a subquery that belong to the scope this Apply stands for:
its right side, with the
+ * right side of every subquery nested in it left out.
+ *
+ * <p>A nested Apply is a scope of its own and is read as one, so its
right side belongs to that read
+ * rather than to this one; the relation it is asked about is the
enclosing scope's own and stays.
+ */
+ private static List<Plan> ownScopeOf(LogicalApply<?, ?> apply) {
+ List<Plan> itsOwnScope = Lists.newArrayList();
+ collectItsOwnScope((Plan) apply.right(), itsOwnScope);
+ return itsOwnScope;
+ }
+
+ private static void collectItsOwnScope(Plan node, List<Plan> itsOwnScope) {
+ itsOwnScope.add(node);
+ if (node instanceof LogicalApply) {
+ collectItsOwnScope(((LogicalApply<?, ?>) node).left(),
itsOwnScope);
+ return;
+ }
+ for (Plan child : node.children()) {
+ collectItsOwnScope(child, itsOwnScope);
+ }
+ }
+
+ /** Whether one of these nodes reads this table, through whatever views
stand between the two. */
+ private static boolean readsAnyTableOf(List<Plan> itsOwnScope,
BaseTableInfo baseTableInfo) {
+ return itsOwnScope.stream().anyMatch(node -> node instanceof
LogicalCatalogRelation
+ && new BaseTableInfo(((LogicalCatalogRelation)
node).getTable()).equals(baseTableInfo));
+ }
+
+ /**
+ * Whether this expression reads a value the subquery answers for itself,
under one of these names: a
+ * slot of the subquery's own -- an alias or a value it computed -- rather
than a column of a table.
+ *
+ * <p>Read rather than merely named, because an expression of the subquery
carrying one of these names
+ * says nothing on its own: a subquery that names a `flag` of its own
while no reference in it resolves
+ * to that name is one whose rows the change cannot reach, and one that
reads the name it names is where
+ * a reference that answered for the changed column falls back to.
+ */
+ private static boolean readsAnyNameTheSubqueryAnswersFor(Expression
expression, Set<String> names) {
+ for (Slot slot : expression.getInputSlots()) {
+ if (names.contains(slot.getName()) && !isColumnOfATable(slot)) {
+ return true;
+ }
+ }
+ return false;
+ }
+
+ /** Whether this slot is a column of some table, or a value produced
inside the query. */
+ private static boolean isColumnOfATable(Slot slot) {
+ return slot instanceof SlotReference && ((SlotReference)
slot).getOriginalTable().isPresent();
+ }
+
+ /** Whether this expression reads a column of one of these names from this
table. */
+ private static boolean reachesAnyColumn(Expression expression, Set<String>
names,
+ BaseTableInfo baseTableInfo) {
+ return reachesAnyColumn(ImmutableList.of(expression), names,
baseTableInfo);
+ }
+
+ /** Whether any of these expressions reads a column of one of these names
from this table. */
+ private static boolean reachesAnyColumn(Collection<? extends Expression>
expressions, Set<String> names,
+ BaseTableInfo baseTableInfo) {
+ for (Expression expression : expressions) {
+ for (Slot slot : expression.getInputSlots()) {
Review Comment:
[P2] Compare the source column behind an IN-subquery alias. For a refreshed
MV on `SELECT o.id FROM outer_t o WHERE o.k IN (SELECT d.flag FROM (SELECT c.x
AS flag FROM changed_t c) d)`, adding `changed_t.flag` leaves `d.flag` bound to
`c.x` and all rows unchanged. `Alias.toSlot()` preserves
`originalTable=changed_t` and `originalColumn=x` while naming the slot `flag`;
this raw Apply-right scan compares only that visible name and table, so it
invalidates the MV and drops its rewrite snapshot. Match a source-backed slot's
original column to the changed column, or shuttle the Apply expression through
its alias before matching.
##########
fe/fe-core/src/main/java/org/apache/doris/mtmv/MTMVRelationManager.java:
##########
@@ -349,34 +360,221 @@ public void dropTable(Table table) {
// because a dropped table is the one change whose query is gone
beyond doubt. What the two record
// is the same state either way. Unlike a rename it stays an
invalidation: the table is gone for
// good, so the state is not something a later alter can make obsolete.
- processBaseTableChange(new BaseTableInfo(table), "The base table has
been deleted:", false);
+ processBaseTableChange(new BaseTableInfo(table), "The base table has
been deleted:", null);
}
/**
* update mtmv status to `SCHEMA_CHANGE`.
*
* @param isReplace
+ * @param queryJudgedColumns the names the alter gives the table or takes
away from it, which leave the
+ * judgement about each MV's state to that MV's
own query, or null when the
+ * alter is not one a query decides. The names
are carried rather than judged
+ * before the call because the judgement is
about them; see
+ * {@code AlterOp#queryJudgedColumnNames} for
which operations name one, and
+ * {@link #invalidateMvUnlessQueryHolds} for
what is asked about it. A rename
+ * of the base table names no column: it is left
to the record below, which
+ * says what the MV that keeps spelling the old
name needs to hear
*/
@Override
- public void alterTable(BaseTableInfo oldTableInfo, Optional<BaseTableInfo>
newTableInfo, boolean isReplace) {
+ public void alterTable(BaseTableInfo oldTableInfo, Optional<BaseTableInfo>
newTableInfo, boolean isReplace,
+ QueryJudgedChange queryJudgedChange) {
// when replace, need deal two table
if (isReplace) {
// REPLACE TABLE already invalidates the IVM baseline explicitly,
see Alter#processReplaceTable
- processBaseTableChange(newTableInfo.get(), "The base table has
been updated:", false);
+ processBaseTableChange(newTableInfo.get(), "The base table has
been updated:", null);
}
- boolean renamed = !isReplace && newTableInfo.isPresent()
- && !Objects.equals(oldTableInfo.getTableName(),
newTableInfo.get().getTableName());
- // A rename is the one change whose query check is skipped: the MV
query keeps spelling the old
- // name, so it is unanalyzable by construction, and the reason it
would be invalidated with --
- // "the query is no longer analyzable" -- says less than the message
this call records anyway.
- boolean checkQueryUsable = !renamed;
- processBaseTableChange(oldTableInfo, "The base table has been
updated:", checkQueryUsable);
+ processBaseTableChange(oldTableInfo, "The base table has been
updated:", queryJudgedChange);
}
/**
- * An MV's query is only as good as the base table schema it was analyzed
against. Re-analyzing the
- * MV query here (right after the alter was applied) is what detects a
changed column identity:
+ * Whether the query, as it is analysed now, reads a column of any of
these names, and reads it where
+ * the change can reach it.
+ *
+ * <p>There are two places a name is the change's to answer for. One is a
column of the table the change
+ * is about: that is the column this view's rows were computed from, and
the names are matched
+ * case-insensitively because a name is what moves. The other is a column
the query reaches across a
+ * scope boundary -- the plan records those on the Apply that stands for
the subquery, whose correlation
+ * slots are the outer columns its right side reads -- because such a name
is the scopes' to answer for
+ * rather than the query's: the nearest column to the reference answers
for it, so a column the change
+ * takes away from a scope inside leaves the name to one outside, and a
column it gives to a scope inside
+ * takes the name over. A name reached with the qualifier of another table
inside the query's own scope
+ * is neither: no later change can move it, so one to a column it does not
name is one this view's rows
+ * do not depend on.
+ */
+ private static boolean reachesAnyColumnOf(Plan plan, BaseTableInfo
baseTableInfo, Set<String> columnNames) {
+ if (plan == null) {
+ // A query whose plan was not kept is one this cannot be answered
about, and "it does" is the
+ // answer that keeps the view safe.
+ return true;
+ }
+ Set<String> names = Sets.newTreeSet(String.CASE_INSENSITIVE_ORDER);
+ names.addAll(columnNames);
+ LineageInfo lineage = LineageInfoExtractor.extractLineageInfo(plan);
+ for (SetMultimap<?, Expression> byType :
lineage.getDirectLineageMap().values()) {
+ if (reachesAnyColumn(byType.values(), names, baseTableInfo)) {
+ return true;
+ }
+ }
+ // The dataset predicates once, not once per output column: the
per-output copy of them the lineage
+ // also offers holds the same expressions for every column the query
produces, and scanning it would
+ // visit each of them once per column.
+ if (reachesAnyColumn(lineage.getDatasetIndirectLineageMap().values(),
names, baseTableInfo)) {
+ return true;
+ }
+ if (reachesAnyColumnOfASubquery(plan, names, baseTableInfo)) {
+ return true;
+ }
+ return reachesAnyColumnAcrossScopes(plan, lineage, names,
baseTableInfo);
+ }
+
+ /** Whether this slot is a column of this table, through whatever views
stand between the two. */
+ private static boolean isColumnOf(Slot slot, BaseTableInfo baseTableInfo) {
+ if (!(slot instanceof SlotReference)) {
+ return false;
+ }
+ return ((SlotReference) slot).getOriginalTable()
+ .map(table -> new BaseTableInfo(table).equals(baseTableInfo))
+ .orElse(false);
+ }
+
+ /**
+ * Whether a name the change is about is answered for inside a subquery,
out of that subquery's own
+ * scope.
+ *
+ * <p>This is the one place a name can move without any column the view
produces depending on it: the
+ * scope of a subquery is internal, so which column answers for a name
there changes what the query
+ * returns -- a row, or none -- while every column of the view stays the
one it was. The lineage of the
+ * view's columns does not reach it, so the scope the subquery became is
read here, expression by
+ * expression, the way the lineage is read for the view's own.
+ *
+ * <p>Two things are read. One is a value the subquery itself names -- an
expression of its own under one
+ * of these names, rather than a column of a table -- because that is what
a name the change takes away
+ * falls back to, and it decides the rows whether the subquery is a
predicate or a value. It is read only
+ * where the table the change is about is one that subquery reads: what a
name falls back to is what the
+ * scope that answered for it holds, so a scope that does not read the
table holds nothing for the name
+ * and one of its own is one the change never reached. The other is a
column of the table the change is
+ * about, which decides the rows only when the subquery's output is one
the query reads: an EXISTS tests
+ * the rows of its subquery and not what it projects, so a name it
projects and never compares is one
+ * this view's rows do not depend on.
+ *
+ * <p>Each scope is read on its own. A subquery inside one of these is a
scope of its own, and it is
+ * judged where it is read and not as a part of its enclosing one: its
projection is held only where
+ * that scope's own output is read, so an EXISTS inside an IN is still not
compared by the IN.
+ */
+ private static boolean reachesAnyColumnOfASubquery(Plan plan, Set<String>
names,
+ BaseTableInfo baseTableInfo) {
+ for (LogicalApply<?, ?> apply :
plan.<LogicalApply>collectToList(LogicalApply.class::isInstance)) {
+ List<Plan> itsOwnScope = ownScopeOf(apply);
+ boolean outputDecidesRows = !apply.isExist();
+ boolean isOneOfItsTables = readsAnyTableOf(itsOwnScope,
baseTableInfo);
+ for (Plan node : itsOwnScope) {
+ for (Expression expression : node.getExpressions()) {
+ if ((isOneOfItsTables &&
readsAnyNameTheSubqueryAnswersFor(expression, names))
+ || (outputDecidesRows &&
reachesAnyColumn(expression, names, baseTableInfo))) {
+ return true;
+ }
+ }
+ }
+ }
+ return false;
+ }
+
+ /**
+ * The nodes of a subquery that belong to the scope this Apply stands for:
its right side, with the
+ * right side of every subquery nested in it left out.
+ *
+ * <p>A nested Apply is a scope of its own and is read as one, so its
right side belongs to that read
+ * rather than to this one; the relation it is asked about is the
enclosing scope's own and stays.
+ */
+ private static List<Plan> ownScopeOf(LogicalApply<?, ?> apply) {
+ List<Plan> itsOwnScope = Lists.newArrayList();
+ collectItsOwnScope((Plan) apply.right(), itsOwnScope);
+ return itsOwnScope;
+ }
+
+ private static void collectItsOwnScope(Plan node, List<Plan> itsOwnScope) {
+ itsOwnScope.add(node);
+ if (node instanceof LogicalApply) {
+ collectItsOwnScope(((LogicalApply<?, ?>) node).left(),
itsOwnScope);
+ return;
+ }
+ for (Plan child : node.children()) {
+ collectItsOwnScope(child, itsOwnScope);
+ }
+ }
+
+ /** Whether one of these nodes reads this table, through whatever views
stand between the two. */
+ private static boolean readsAnyTableOf(List<Plan> itsOwnScope,
BaseTableInfo baseTableInfo) {
+ return itsOwnScope.stream().anyMatch(node -> node instanceof
LogicalCatalogRelation
+ && new BaseTableInfo(((LogicalCatalogRelation)
node).getTable()).equals(baseTableInfo));
+ }
+
+ /**
+ * Whether this expression reads a value the subquery answers for itself,
under one of these names: a
+ * slot of the subquery's own -- an alias or a value it computed -- rather
than a column of a table.
+ *
+ * <p>Read rather than merely named, because an expression of the subquery
carrying one of these names
+ * says nothing on its own: a subquery that names a `flag` of its own
while no reference in it resolves
+ * to that name is one whose rows the change cannot reach, and one that
reads the name it names is where
+ * a reference that answered for the changed column falls back to.
+ */
+ private static boolean readsAnyNameTheSubqueryAnswersFor(Expression
expression, Set<String> names) {
+ for (Slot slot : expression.getInputSlots()) {
+ if (names.contains(slot.getName()) && !isColumnOfATable(slot)) {
+ return true;
+ }
+ }
+ return false;
+ }
+
+ /** Whether this slot is a column of some table, or a value produced
inside the query. */
+ private static boolean isColumnOfATable(Slot slot) {
+ return slot instanceof SlotReference && ((SlotReference)
slot).getOriginalTable().isPresent();
+ }
+
+ /** Whether this expression reads a column of one of these names from this
table. */
+ private static boolean reachesAnyColumn(Expression expression, Set<String>
names,
+ BaseTableInfo baseTableInfo) {
+ return reachesAnyColumn(ImmutableList.of(expression), names,
baseTableInfo);
+ }
+
+ /** Whether any of these expressions reads a column of one of these names
from this table. */
+ private static boolean reachesAnyColumn(Collection<? extends Expression>
expressions, Set<String> names,
+ BaseTableInfo baseTableInfo) {
+ for (Expression expression : expressions) {
+ for (Slot slot : expression.getInputSlots()) {
+ if (names.contains(slot.getName()) && isColumnOf(slot,
baseTableInfo)) {
+ return true;
+ }
+ }
+ }
+ return false;
+ }
+
+ /**
+ * Whether the query resolves a column of one of these names across a
scope boundary, which is a name
+ * the change can move whatever the query writes it against.
+ */
+ private static boolean reachesAnyColumnAcrossScopes(Plan plan, LineageInfo
lineage, Set<String> names,
+ BaseTableInfo baseTableInfo) {
+ // A name is only one the change can move if the table it is about is
one the query reads at all.
+ boolean isOneOfItsTables = lineage.getTableLineageSet().stream()
+ .anyMatch(table -> new
BaseTableInfo(table).equals(baseTableInfo));
+ if (!isOneOfItsTables) {
+ return false;
+ }
Review Comment:
[P2] Exclude ignored EXISTS projections from the global correlation check. A
refreshed MV of `SELECT o.id FROM outer_t o WHERE EXISTS (SELECT spare FROM
inner_t i WHERE i.id=o.id)` can have `spare` on both tables. After light `DROP
inner_t.spare`, the unqualified Project name binds `o.spare`, but EXISTS still
tests only matching `i.id` rows, so the MV is unchanged. The analyzed Apply
records that projection-only `o.spare` as a correlation. The local Apply scan
correctly skips its output for EXISTS, but this global scan sees `inner_t` and
the `spare` correlation and invalidates the MV, discarding its rewrite
snapshot. Count only correlations that affect EXISTS rows; the earlier ADD
projection thread was a separate raw-output branch.
##########
fe/fe-core/src/main/java/org/apache/doris/mtmv/MTMVRelationManager.java:
##########
@@ -349,34 +360,221 @@ public void dropTable(Table table) {
// because a dropped table is the one change whose query is gone
beyond doubt. What the two record
// is the same state either way. Unlike a rename it stays an
invalidation: the table is gone for
// good, so the state is not something a later alter can make obsolete.
- processBaseTableChange(new BaseTableInfo(table), "The base table has
been deleted:", false);
+ processBaseTableChange(new BaseTableInfo(table), "The base table has
been deleted:", null);
}
/**
* update mtmv status to `SCHEMA_CHANGE`.
*
* @param isReplace
+ * @param queryJudgedColumns the names the alter gives the table or takes
away from it, which leave the
+ * judgement about each MV's state to that MV's
own query, or null when the
+ * alter is not one a query decides. The names
are carried rather than judged
+ * before the call because the judgement is
about them; see
+ * {@code AlterOp#queryJudgedColumnNames} for
which operations name one, and
+ * {@link #invalidateMvUnlessQueryHolds} for
what is asked about it. A rename
+ * of the base table names no column: it is left
to the record below, which
+ * says what the MV that keeps spelling the old
name needs to hear
*/
@Override
- public void alterTable(BaseTableInfo oldTableInfo, Optional<BaseTableInfo>
newTableInfo, boolean isReplace) {
+ public void alterTable(BaseTableInfo oldTableInfo, Optional<BaseTableInfo>
newTableInfo, boolean isReplace,
+ QueryJudgedChange queryJudgedChange) {
// when replace, need deal two table
if (isReplace) {
// REPLACE TABLE already invalidates the IVM baseline explicitly,
see Alter#processReplaceTable
- processBaseTableChange(newTableInfo.get(), "The base table has
been updated:", false);
+ processBaseTableChange(newTableInfo.get(), "The base table has
been updated:", null);
}
- boolean renamed = !isReplace && newTableInfo.isPresent()
- && !Objects.equals(oldTableInfo.getTableName(),
newTableInfo.get().getTableName());
- // A rename is the one change whose query check is skipped: the MV
query keeps spelling the old
- // name, so it is unanalyzable by construction, and the reason it
would be invalidated with --
- // "the query is no longer analyzable" -- says less than the message
this call records anyway.
- boolean checkQueryUsable = !renamed;
- processBaseTableChange(oldTableInfo, "The base table has been
updated:", checkQueryUsable);
+ processBaseTableChange(oldTableInfo, "The base table has been
updated:", queryJudgedChange);
}
/**
- * An MV's query is only as good as the base table schema it was analyzed
against. Re-analyzing the
- * MV query here (right after the alter was applied) is what detects a
changed column identity:
+ * Whether the query, as it is analysed now, reads a column of any of
these names, and reads it where
+ * the change can reach it.
+ *
+ * <p>There are two places a name is the change's to answer for. One is a
column of the table the change
+ * is about: that is the column this view's rows were computed from, and
the names are matched
+ * case-insensitively because a name is what moves. The other is a column
the query reaches across a
+ * scope boundary -- the plan records those on the Apply that stands for
the subquery, whose correlation
+ * slots are the outer columns its right side reads -- because such a name
is the scopes' to answer for
+ * rather than the query's: the nearest column to the reference answers
for it, so a column the change
+ * takes away from a scope inside leaves the name to one outside, and a
column it gives to a scope inside
+ * takes the name over. A name reached with the qualifier of another table
inside the query's own scope
+ * is neither: no later change can move it, so one to a column it does not
name is one this view's rows
+ * do not depend on.
+ */
+ private static boolean reachesAnyColumnOf(Plan plan, BaseTableInfo
baseTableInfo, Set<String> columnNames) {
+ if (plan == null) {
+ // A query whose plan was not kept is one this cannot be answered
about, and "it does" is the
+ // answer that keeps the view safe.
+ return true;
+ }
+ Set<String> names = Sets.newTreeSet(String.CASE_INSENSITIVE_ORDER);
+ names.addAll(columnNames);
+ LineageInfo lineage = LineageInfoExtractor.extractLineageInfo(plan);
+ for (SetMultimap<?, Expression> byType :
lineage.getDirectLineageMap().values()) {
+ if (reachesAnyColumn(byType.values(), names, baseTableInfo)) {
+ return true;
+ }
+ }
+ // The dataset predicates once, not once per output column: the
per-output copy of them the lineage
+ // also offers holds the same expressions for every column the query
produces, and scanning it would
+ // visit each of them once per column.
+ if (reachesAnyColumn(lineage.getDatasetIndirectLineageMap().values(),
names, baseTableInfo)) {
+ return true;
+ }
+ if (reachesAnyColumnOfASubquery(plan, names, baseTableInfo)) {
+ return true;
+ }
+ return reachesAnyColumnAcrossScopes(plan, lineage, names,
baseTableInfo);
+ }
+
+ /** Whether this slot is a column of this table, through whatever views
stand between the two. */
+ private static boolean isColumnOf(Slot slot, BaseTableInfo baseTableInfo) {
+ if (!(slot instanceof SlotReference)) {
+ return false;
+ }
+ return ((SlotReference) slot).getOriginalTable()
+ .map(table -> new BaseTableInfo(table).equals(baseTableInfo))
+ .orElse(false);
+ }
+
+ /**
+ * Whether a name the change is about is answered for inside a subquery,
out of that subquery's own
+ * scope.
+ *
+ * <p>This is the one place a name can move without any column the view
produces depending on it: the
+ * scope of a subquery is internal, so which column answers for a name
there changes what the query
+ * returns -- a row, or none -- while every column of the view stays the
one it was. The lineage of the
+ * view's columns does not reach it, so the scope the subquery became is
read here, expression by
+ * expression, the way the lineage is read for the view's own.
+ *
+ * <p>Two things are read. One is a value the subquery itself names -- an
expression of its own under one
+ * of these names, rather than a column of a table -- because that is what
a name the change takes away
+ * falls back to, and it decides the rows whether the subquery is a
predicate or a value. It is read only
+ * where the table the change is about is one that subquery reads: what a
name falls back to is what the
+ * scope that answered for it holds, so a scope that does not read the
table holds nothing for the name
+ * and one of its own is one the change never reached. The other is a
column of the table the change is
+ * about, which decides the rows only when the subquery's output is one
the query reads: an EXISTS tests
+ * the rows of its subquery and not what it projects, so a name it
projects and never compares is one
+ * this view's rows do not depend on.
+ *
+ * <p>Each scope is read on its own. A subquery inside one of these is a
scope of its own, and it is
+ * judged where it is read and not as a part of its enclosing one: its
projection is held only where
+ * that scope's own output is read, so an EXISTS inside an IN is still not
compared by the IN.
+ */
+ private static boolean reachesAnyColumnOfASubquery(Plan plan, Set<String>
names,
+ BaseTableInfo baseTableInfo) {
+ for (LogicalApply<?, ?> apply :
plan.<LogicalApply>collectToList(LogicalApply.class::isInstance)) {
+ List<Plan> itsOwnScope = ownScopeOf(apply);
+ boolean outputDecidesRows = !apply.isExist();
+ boolean isOneOfItsTables = readsAnyTableOf(itsOwnScope,
baseTableInfo);
+ for (Plan node : itsOwnScope) {
+ for (Expression expression : node.getExpressions()) {
+ if ((isOneOfItsTables &&
readsAnyNameTheSubqueryAnswersFor(expression, names))
+ || (outputDecidesRows &&
reachesAnyColumn(expression, names, baseTableInfo))) {
+ return true;
+ }
+ }
+ }
+ }
+ return false;
+ }
+
+ /**
+ * The nodes of a subquery that belong to the scope this Apply stands for:
its right side, with the
+ * right side of every subquery nested in it left out.
+ *
+ * <p>A nested Apply is a scope of its own and is read as one, so its
right side belongs to that read
+ * rather than to this one; the relation it is asked about is the
enclosing scope's own and stays.
+ */
+ private static List<Plan> ownScopeOf(LogicalApply<?, ?> apply) {
+ List<Plan> itsOwnScope = Lists.newArrayList();
+ collectItsOwnScope((Plan) apply.right(), itsOwnScope);
+ return itsOwnScope;
+ }
+
+ private static void collectItsOwnScope(Plan node, List<Plan> itsOwnScope) {
+ itsOwnScope.add(node);
+ if (node instanceof LogicalApply) {
+ collectItsOwnScope(((LogicalApply<?, ?>) node).left(),
itsOwnScope);
+ return;
+ }
+ for (Plan child : node.children()) {
+ collectItsOwnScope(child, itsOwnScope);
+ }
+ }
+
+ /** Whether one of these nodes reads this table, through whatever views
stand between the two. */
+ private static boolean readsAnyTableOf(List<Plan> itsOwnScope,
BaseTableInfo baseTableInfo) {
+ return itsOwnScope.stream().anyMatch(node -> node instanceof
LogicalCatalogRelation
+ && new BaseTableInfo(((LogicalCatalogRelation)
node).getTable()).equals(baseTableInfo));
+ }
+
+ /**
+ * Whether this expression reads a value the subquery answers for itself,
under one of these names: a
+ * slot of the subquery's own -- an alias or a value it computed -- rather
than a column of a table.
+ *
+ * <p>Read rather than merely named, because an expression of the subquery
carrying one of these names
+ * says nothing on its own: a subquery that names a `flag` of its own
while no reference in it resolves
+ * to that name is one whose rows the change cannot reach, and one that
reads the name it names is where
+ * a reference that answered for the changed column falls back to.
+ */
+ private static boolean readsAnyNameTheSubqueryAnswersFor(Expression
expression, Set<String> names) {
+ for (Slot slot : expression.getInputSlots()) {
Review Comment:
[P1] Keep a forwarding alias as a possible post-DROP binding. A refreshed MV
of `SELECT o.id FROM outer_t o WHERE EXISTS (SELECT b.x AS flag, COUNT(*) n
FROM changed_t c JOIN other_t b ON b.id=c.id WHERE c.id=o.id GROUP BY b.x, flag
HAVING flag=1)` can be empty with `c.flag=0` and `b.x=1`. After a light `DROP
changed_t.flag`, the stored unqualified GROUP BY/HAVING `flag` resolves to the
SELECT alias `b.x`, so the query returns `o.id` with the same MV schema.
`Alias.toSlot()` carries `originalTable=other_t` even for that alias, and this
`!isColumnOfATable` test skips it; no other post-DROP check sees
`changed_t.flag`, leaving stale MV rows NORMAL. Preserve the old binding or
recognize source-backed aliases here. The earlier constant-alias thread is
fixed; this forwarding alias takes the opposite branch.
##########
fe/fe-core/src/main/java/org/apache/doris/mtmv/MTMVRelationManager.java:
##########
@@ -349,34 +360,221 @@ public void dropTable(Table table) {
// because a dropped table is the one change whose query is gone
beyond doubt. What the two record
// is the same state either way. Unlike a rename it stays an
invalidation: the table is gone for
// good, so the state is not something a later alter can make obsolete.
- processBaseTableChange(new BaseTableInfo(table), "The base table has
been deleted:", false);
+ processBaseTableChange(new BaseTableInfo(table), "The base table has
been deleted:", null);
}
/**
* update mtmv status to `SCHEMA_CHANGE`.
*
* @param isReplace
+ * @param queryJudgedColumns the names the alter gives the table or takes
away from it, which leave the
+ * judgement about each MV's state to that MV's
own query, or null when the
+ * alter is not one a query decides. The names
are carried rather than judged
+ * before the call because the judgement is
about them; see
+ * {@code AlterOp#queryJudgedColumnNames} for
which operations name one, and
+ * {@link #invalidateMvUnlessQueryHolds} for
what is asked about it. A rename
+ * of the base table names no column: it is left
to the record below, which
+ * says what the MV that keeps spelling the old
name needs to hear
*/
@Override
- public void alterTable(BaseTableInfo oldTableInfo, Optional<BaseTableInfo>
newTableInfo, boolean isReplace) {
+ public void alterTable(BaseTableInfo oldTableInfo, Optional<BaseTableInfo>
newTableInfo, boolean isReplace,
+ QueryJudgedChange queryJudgedChange) {
// when replace, need deal two table
if (isReplace) {
// REPLACE TABLE already invalidates the IVM baseline explicitly,
see Alter#processReplaceTable
- processBaseTableChange(newTableInfo.get(), "The base table has
been updated:", false);
+ processBaseTableChange(newTableInfo.get(), "The base table has
been updated:", null);
}
- boolean renamed = !isReplace && newTableInfo.isPresent()
- && !Objects.equals(oldTableInfo.getTableName(),
newTableInfo.get().getTableName());
- // A rename is the one change whose query check is skipped: the MV
query keeps spelling the old
- // name, so it is unanalyzable by construction, and the reason it
would be invalidated with --
- // "the query is no longer analyzable" -- says less than the message
this call records anyway.
- boolean checkQueryUsable = !renamed;
- processBaseTableChange(oldTableInfo, "The base table has been
updated:", checkQueryUsable);
+ processBaseTableChange(oldTableInfo, "The base table has been
updated:", queryJudgedChange);
}
/**
- * An MV's query is only as good as the base table schema it was analyzed
against. Re-analyzing the
- * MV query here (right after the alter was applied) is what detects a
changed column identity:
+ * Whether the query, as it is analysed now, reads a column of any of
these names, and reads it where
+ * the change can reach it.
+ *
+ * <p>There are two places a name is the change's to answer for. One is a
column of the table the change
+ * is about: that is the column this view's rows were computed from, and
the names are matched
+ * case-insensitively because a name is what moves. The other is a column
the query reaches across a
+ * scope boundary -- the plan records those on the Apply that stands for
the subquery, whose correlation
+ * slots are the outer columns its right side reads -- because such a name
is the scopes' to answer for
+ * rather than the query's: the nearest column to the reference answers
for it, so a column the change
+ * takes away from a scope inside leaves the name to one outside, and a
column it gives to a scope inside
+ * takes the name over. A name reached with the qualifier of another table
inside the query's own scope
+ * is neither: no later change can move it, so one to a column it does not
name is one this view's rows
+ * do not depend on.
+ */
+ private static boolean reachesAnyColumnOf(Plan plan, BaseTableInfo
baseTableInfo, Set<String> columnNames) {
+ if (plan == null) {
+ // A query whose plan was not kept is one this cannot be answered
about, and "it does" is the
+ // answer that keeps the view safe.
+ return true;
+ }
+ Set<String> names = Sets.newTreeSet(String.CASE_INSENSITIVE_ORDER);
+ names.addAll(columnNames);
+ LineageInfo lineage = LineageInfoExtractor.extractLineageInfo(plan);
+ for (SetMultimap<?, Expression> byType :
lineage.getDirectLineageMap().values()) {
+ if (reachesAnyColumn(byType.values(), names, baseTableInfo)) {
+ return true;
+ }
+ }
+ // The dataset predicates once, not once per output column: the
per-output copy of them the lineage
+ // also offers holds the same expressions for every column the query
produces, and scanning it would
+ // visit each of them once per column.
+ if (reachesAnyColumn(lineage.getDatasetIndirectLineageMap().values(),
names, baseTableInfo)) {
+ return true;
+ }
+ if (reachesAnyColumnOfASubquery(plan, names, baseTableInfo)) {
+ return true;
+ }
+ return reachesAnyColumnAcrossScopes(plan, lineage, names,
baseTableInfo);
+ }
+
+ /** Whether this slot is a column of this table, through whatever views
stand between the two. */
+ private static boolean isColumnOf(Slot slot, BaseTableInfo baseTableInfo) {
+ if (!(slot instanceof SlotReference)) {
+ return false;
+ }
+ return ((SlotReference) slot).getOriginalTable()
+ .map(table -> new BaseTableInfo(table).equals(baseTableInfo))
+ .orElse(false);
+ }
+
+ /**
+ * Whether a name the change is about is answered for inside a subquery,
out of that subquery's own
+ * scope.
+ *
+ * <p>This is the one place a name can move without any column the view
produces depending on it: the
+ * scope of a subquery is internal, so which column answers for a name
there changes what the query
+ * returns -- a row, or none -- while every column of the view stays the
one it was. The lineage of the
+ * view's columns does not reach it, so the scope the subquery became is
read here, expression by
+ * expression, the way the lineage is read for the view's own.
+ *
+ * <p>Two things are read. One is a value the subquery itself names -- an
expression of its own under one
+ * of these names, rather than a column of a table -- because that is what
a name the change takes away
+ * falls back to, and it decides the rows whether the subquery is a
predicate or a value. It is read only
+ * where the table the change is about is one that subquery reads: what a
name falls back to is what the
+ * scope that answered for it holds, so a scope that does not read the
table holds nothing for the name
+ * and one of its own is one the change never reached. The other is a
column of the table the change is
+ * about, which decides the rows only when the subquery's output is one
the query reads: an EXISTS tests
+ * the rows of its subquery and not what it projects, so a name it
projects and never compares is one
+ * this view's rows do not depend on.
+ *
+ * <p>Each scope is read on its own. A subquery inside one of these is a
scope of its own, and it is
+ * judged where it is read and not as a part of its enclosing one: its
projection is held only where
+ * that scope's own output is read, so an EXISTS inside an IN is still not
compared by the IN.
+ */
+ private static boolean reachesAnyColumnOfASubquery(Plan plan, Set<String>
names,
+ BaseTableInfo baseTableInfo) {
+ for (LogicalApply<?, ?> apply :
plan.<LogicalApply>collectToList(LogicalApply.class::isInstance)) {
+ List<Plan> itsOwnScope = ownScopeOf(apply);
+ boolean outputDecidesRows = !apply.isExist();
+ boolean isOneOfItsTables = readsAnyTableOf(itsOwnScope,
baseTableInfo);
+ for (Plan node : itsOwnScope) {
+ for (Expression expression : node.getExpressions()) {
+ if ((isOneOfItsTables &&
readsAnyNameTheSubqueryAnswersFor(expression, names))
+ || (outputDecidesRows &&
reachesAnyColumn(expression, names, baseTableInfo))) {
+ return true;
+ }
+ }
+ }
+ }
+ return false;
+ }
+
+ /**
+ * The nodes of a subquery that belong to the scope this Apply stands for:
its right side, with the
+ * right side of every subquery nested in it left out.
+ *
+ * <p>A nested Apply is a scope of its own and is read as one, so its
right side belongs to that read
+ * rather than to this one; the relation it is asked about is the
enclosing scope's own and stays.
+ */
+ private static List<Plan> ownScopeOf(LogicalApply<?, ?> apply) {
+ List<Plan> itsOwnScope = Lists.newArrayList();
+ collectItsOwnScope((Plan) apply.right(), itsOwnScope);
+ return itsOwnScope;
+ }
+
+ private static void collectItsOwnScope(Plan node, List<Plan> itsOwnScope) {
+ itsOwnScope.add(node);
+ if (node instanceof LogicalApply) {
+ collectItsOwnScope(((LogicalApply<?, ?>) node).left(),
itsOwnScope);
+ return;
+ }
+ for (Plan child : node.children()) {
+ collectItsOwnScope(child, itsOwnScope);
+ }
+ }
+
+ /** Whether one of these nodes reads this table, through whatever views
stand between the two. */
+ private static boolean readsAnyTableOf(List<Plan> itsOwnScope,
BaseTableInfo baseTableInfo) {
+ return itsOwnScope.stream().anyMatch(node -> node instanceof
LogicalCatalogRelation
+ && new BaseTableInfo(((LogicalCatalogRelation)
node).getTable()).equals(baseTableInfo));
+ }
+
+ /**
+ * Whether this expression reads a value the subquery answers for itself,
under one of these names: a
+ * slot of the subquery's own -- an alias or a value it computed -- rather
than a column of a table.
+ *
+ * <p>Read rather than merely named, because an expression of the subquery
carrying one of these names
+ * says nothing on its own: a subquery that names a `flag` of its own
while no reference in it resolves
+ * to that name is one whose rows the change cannot reach, and one that
reads the name it names is where
+ * a reference that answered for the changed column falls back to.
+ */
+ private static boolean readsAnyNameTheSubqueryAnswersFor(Expression
expression, Set<String> names) {
+ for (Slot slot : expression.getInputSlots()) {
+ if (names.contains(slot.getName()) && !isColumnOfATable(slot)) {
+ return true;
+ }
+ }
+ return false;
+ }
+
+ /** Whether this slot is a column of some table, or a value produced
inside the query. */
+ private static boolean isColumnOfATable(Slot slot) {
+ return slot instanceof SlotReference && ((SlotReference)
slot).getOriginalTable().isPresent();
+ }
+
+ /** Whether this expression reads a column of one of these names from this
table. */
+ private static boolean reachesAnyColumn(Expression expression, Set<String>
names,
+ BaseTableInfo baseTableInfo) {
+ return reachesAnyColumn(ImmutableList.of(expression), names,
baseTableInfo);
+ }
+
+ /** Whether any of these expressions reads a column of one of these names
from this table. */
+ private static boolean reachesAnyColumn(Collection<? extends Expression>
expressions, Set<String> names,
+ BaseTableInfo baseTableInfo) {
+ for (Expression expression : expressions) {
+ for (Slot slot : expression.getInputSlots()) {
+ if (names.contains(slot.getName()) && isColumnOf(slot,
baseTableInfo)) {
+ return true;
+ }
+ }
+ }
+ return false;
+ }
+
+ /**
+ * Whether the query resolves a column of one of these names across a
scope boundary, which is a name
+ * the change can move whatever the query writes it against.
+ */
+ private static boolean reachesAnyColumnAcrossScopes(Plan plan, LineageInfo
lineage, Set<String> names,
+ BaseTableInfo baseTableInfo) {
+ // A name is only one the change can move if the table it is about is
one the query reads at all.
Review Comment:
[P2] Scope correlated names to the subquery the changed table can reach. For
a refreshed MV of `SELECT o.id FROM outer_t o WHERE EXISTS (SELECT 1 FROM
changed_t c WHERE c.id=o.id) AND EXISTS (SELECT 1 FROM other_t u WHERE
u.id=o.id AND flag=1)`, only `outer_t` has `flag`, so the second subquery's
unqualified `flag` stays bound to `o.flag` after a light `ADD changed_t.flag`.
The global table-lineage check sees `changed_t` in the first Apply, while
`plan.anyMatch` sees `flag` in the second Apply's correlation slots, and
invalidates the unchanged MV and discards its rewrite snapshot. Match each
correlation to a table that can enter that Apply's scope. This is separate from
the fixed sibling-alias case: it is an outer-column correlation in this global
branch.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]