Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion datafusion/common/src/rounding.rs
Original file line number Diff line number Diff line change
Expand Up @@ -254,7 +254,7 @@ where
}
}
_ => {}
};
}
Ok(result)
}

Expand Down
122 changes: 122 additions & 0 deletions datafusion/core/tests/custom_sources_cases/statistics.rs
Original file line number Diff line number Diff line change
Expand Up @@ -285,6 +285,128 @@ async fn sql_filter() -> Result<()> {
Ok(())
}

fn string_filter_ctx(
data_type: DataType,
distinct_count: Precision<usize>,
) -> Result<SessionContext> {
init_ctx(
Statistics {
num_rows: Precision::Exact(1000),
total_byte_size: Precision::Absent,
column_statistics: vec![ColumnStatistics {
null_count: Precision::Exact(200),
distinct_count,
..ColumnStatistics::new_unknown()
}],
},
Schema::new(vec![Field::new("c1", data_type, true)]),
)
}

async fn string_filter_rows(
ctx: &SessionContext,
predicate: &str,
) -> Result<Precision<usize>> {
let plan = ctx
.sql(&format!("SELECT * FROM stats_table WHERE {predicate}"))
.await?
.create_physical_plan()
.await?;
Ok(StatisticsContext::new()
.compute(plan.as_ref(), &StatisticsArgs::new())?
.num_rows)
}

#[tokio::test]
async fn sql_string_filter_selectivity() -> Result<()> {
// There are 800 non-null rows and 20 distinct strings. Equality estimates
// 40 matches, and negated predicates exclude nulls as well as matches.
// LIKE estimates distinguish unanchored patterns from patterns anchored
// at one or both ends; escaped wildcards are literal characters.
let cases = [
("c1 = 'foo'", 40),
("'foo' = c1", 40),
("c1 != 'foo'", 760),
("'foo' != c1", 760),
("c1 LIKE 'foo'", 40),
("c1 NOT LIKE 'foo'", 760),
("c1 LIKE '%foo%'", 160),
("c1 NOT LIKE '%foo%'", 640),
("c1 LIKE 'foo%'", 80),
("c1 NOT LIKE 'foo%'", 720),
("c1 LIKE '%foo'", 80),
("c1 NOT LIKE '%foo'", 720),
("c1 LIKE 'f_o'", 40),
("c1 NOT LIKE 'f_o'", 760),
(r"c1 LIKE 'foo\%'", 40),
(r"c1 NOT LIKE 'foo\%'", 760),
(r"c1 LIKE 'foo\%%'", 80),
(r"c1 NOT LIKE 'foo\%%'", 720),
(r"c1 LIKE 'foo\'", 40),
(r"c1 NOT LIKE 'foo\'", 760),
(r"c1 LIKE '%foo\'", 80),
(r"c1 NOT LIKE '%foo\'", 720),
// Case-sensitive NDV cannot estimate a case-insensitive equality.
("c1 ILIKE 'foo'", 160),
("c1 NOT ILIKE 'foo'", 640),
];
for data_type in [DataType::Utf8, DataType::LargeUtf8, DataType::Utf8View] {
let ctx = string_filter_ctx(data_type.clone(), Precision::Exact(20))?;
for (predicate, expected_rows) in cases {
assert_eq!(
string_filter_rows(&ctx, predicate).await?,
Precision::Inexact(expected_rows),
"{data_type:?}: {predicate}"
);
}
}
Ok(())
}

#[tokio::test]
async fn sql_string_filter_without_distinct_count() -> Result<()> {
let ctx = string_filter_ctx(DataType::Utf8View, Precision::Absent)?;
// Without NDV, use the configured default (20%) over non-null rows.
for (predicate, expected_rows) in [
("c1 = 'foo'", 160),
("c1 != 'foo'", 640),
("c1 LIKE 'foo'", 160),
("c1 NOT LIKE 'foo'", 640),
] {
assert_eq!(
string_filter_rows(&ctx, predicate).await?,
Precision::Inexact(expected_rows),
"{predicate}"
);
}
Ok(())
}

#[tokio::test]
async fn sql_string_filter_custom_selectivity() -> Result<()> {
let ctx = string_filter_ctx(DataType::Utf8View, Precision::Absent)?;
ctx.sql("SET datafusion.optimizer.default_filter_selectivity = 40")
.await?
.collect()
.await?;

for (predicate, expected_rows) in [
("c1 = 'foo'", 320),
("c1 != 'foo'", 480),
("c1 LIKE '%foo%'", 320),
("c1 NOT LIKE '%foo%'", 480),
("c1 LIKE 'foo%'", 160),
("c1 NOT LIKE 'foo%'", 640),
] {
assert_eq!(
string_filter_rows(&ctx, predicate).await?,
Precision::Inexact(expected_rows),
"{predicate}"
);
}
Ok(())
}

#[tokio::test]
async fn sql_limit() -> Result<()> {
let (stats, schema) = fully_defined();
Expand Down
Loading
Loading