Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 21 additions & 0 deletions bindings/python/python/pypaimon_rust/datafusion.pyi
Original file line number Diff line number Diff line change
Expand Up @@ -222,6 +222,7 @@ class Table:
def new_read_builder(self, options: Optional[Dict[str, str]] = None) -> ReadBuilder: ...
def new_stream_scan(self) -> StreamTableScan: ...
def new_full_text_search_builder(self) -> FullTextSearchBuilder: ...
def new_hybrid_search_builder(self) -> HybridSearchBuilder: ...
def new_vector_search_builder(self) -> "VectorSearchBuilder": ...
def new_batch_vector_search_builder(self) -> "BatchVectorSearchBuilder": ...
def new_global_index_build_builder(self) -> "GlobalIndexBuildBuilder": ...
Expand Down Expand Up @@ -615,6 +616,26 @@ class SearchResultReadBuilder:
"""Read user columns and scores in relevance order from the search snapshot."""
...

class HybridSearchBuilder:
def with_limit(self, limit: int) -> HybridSearchBuilder: ...
def with_snapshot(self, snapshot_json: Optional[str]) -> HybridSearchBuilder:
"""Reuse resolved Java Snapshot JSON; None pins an empty read view."""
...
def with_ranker(self, ranker: Optional[str]) -> HybridSearchBuilder: ...
def with_rrf_ranker(self) -> HybridSearchBuilder: ...
def with_weighted_score_ranker(self) -> HybridSearchBuilder: ...
def with_filter(self, predicate: Dict[str, Any]) -> HybridSearchBuilder: ...
def with_partition_filter(self, predicate: Dict[str, Any]) -> HybridSearchBuilder: ...
def add_vector_route(self, field_name: str, vector: List[float], limit: int,
weight: float = 1.0, options: Optional[Dict[str, str]] = None) -> HybridSearchBuilder: ...
def add_full_text_route(self, field_name: str, query: str, limit: int,
weight: float = 1.0, options: Optional[Dict[str, str]] = None) -> HybridSearchBuilder: ...
def execute_local(self) -> HybridSearchResult: ...

class HybridSearchResult:
def __len__(self) -> int: ...
def row_ids(self) -> Dict[int, float]: ...

class FullTextSearchBuilder:
def with_query(self, field_name: str, query: str) -> FullTextSearchBuilder: ...
def with_limit(self, limit: int) -> FullTextSearchBuilder: ...
Expand Down
2 changes: 2 additions & 0 deletions bindings/python/src/context.rs
Original file line number Diff line number Diff line change
Expand Up @@ -732,6 +732,8 @@ pub fn register_module(py: Python<'_>, m: &Bound<'_, PyModule>) -> PyResult<()>
this.add_class::<crate::full_text_search::PyFullTextScanPlan>()?;
this.add_class::<crate::full_text_search::PyFullTextRead>()?;
this.add_class::<crate::full_text_search::PyFullTextSearchResult>()?;
this.add_class::<crate::hybrid_search::PyHybridSearchBuilder>()?;
this.add_class::<crate::hybrid_search::PyHybridSearchResult>()?;
this.add_class::<crate::vector_search::PyVectorSearchBuilder>()?;
this.add_class::<crate::vector_search::PyBatchVectorSearchBuilder>()?;
this.add_class::<crate::vector_search::PyVectorScan>()?;
Expand Down
206 changes: 206 additions & 0 deletions bindings/python/src/hybrid_search.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,206 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.

//! Hybrid configuration forwarding; snapshot, routes and fusion belong to core.

use crate::error::to_py_err;
use crate::predicate::dict_to_table_predicate;
use paimon::spec::{Predicate, Snapshot};
use paimon::table::{HybridSearchRanker, HybridSearchRoute, Table};
use paimon::vector_search::ScoredRowIds;
use paimon_datafusion::runtime::runtime;
use pyo3::prelude::*;
use pyo3::types::PyDict;
use std::collections::HashMap;
use std::sync::Arc;

#[pyclass(name = "HybridSearchBuilder", module = "pypaimon_rust.datafusion")]
pub struct PyHybridSearchBuilder {
table: Arc<Table>,
routes: Vec<HybridSearchRoute>,
filters: Vec<Predicate>,
limit: Option<usize>,
ranker: HybridSearchRanker,
pinned_snapshot: Option<Option<Snapshot>>,
}

impl PyHybridSearchBuilder {
pub(crate) fn new(table: Arc<Table>) -> Self {
Self {
table,
routes: Vec::new(),
filters: Vec::new(),
limit: None,
ranker: HybridSearchRanker::Rrf,
pinned_snapshot: None,
}
}

fn builder(&self) -> paimon::table::HybridSearchBuilder<'_> {
let mut builder = self.table.new_hybrid_search_builder();
for route in &self.routes {
builder.add_route(route.clone());
}
if let Some(limit) = self.limit {
builder.with_limit(limit);
}
// The stored ranker has already been validated by the core parser.
builder
.with_ranker(self.ranker.as_str())
.expect("validated ranker");
for filter in &self.filters {
builder.with_filter(filter.clone());
}
if let Some(snapshot) = &self.pinned_snapshot {
builder.with_snapshot(snapshot.as_ref());
}
builder
}
}

#[pymethods]
impl PyHybridSearchBuilder {
#[pyo3(signature = (field_name, vector, limit, weight=1.0, options=None))]
fn add_vector_route(
mut slf: PyRefMut<'_, Self>,
field_name: String,
vector: Vec<f32>,
limit: usize,
weight: f32,
options: Option<HashMap<String, String>>,
) -> PyResult<PyRefMut<'_, Self>> {
let route = HybridSearchRoute::vector(
field_name,
vector,
limit,
weight,
options.unwrap_or_default(),
)
.map_err(to_py_err)?;
slf.routes.push(route);
Ok(slf)
}

#[pyo3(signature = (field_name, query, limit, weight=1.0, options=None))]
fn add_full_text_route(
mut slf: PyRefMut<'_, Self>,
field_name: String,
query: String,
limit: usize,
weight: f32,
options: Option<HashMap<String, String>>,
) -> PyResult<PyRefMut<'_, Self>> {
let route = HybridSearchRoute::full_text(
field_name,
query,
limit,
weight,
options.unwrap_or_default(),
)
.map_err(to_py_err)?;
slf.routes.push(route);
Ok(slf)
}

fn with_limit(mut slf: PyRefMut<'_, Self>, limit: usize) -> PyRefMut<'_, Self> {
slf.limit = Some(limit);
slf
}

#[pyo3(signature = (snapshot_json))]
fn with_snapshot<'py>(
mut slf: PyRefMut<'py, Self>,
snapshot_json: Option<&str>,
) -> PyResult<PyRefMut<'py, Self>> {
let snapshot = snapshot_json
.map(serde_json::from_str)
.transpose()
.map_err(|error| {
pyo3::exceptions::PyValueError::new_err(format!("Invalid snapshot JSON: {error}"))
})?;
slf.pinned_snapshot = Some(snapshot);
Ok(slf)
}

#[pyo3(signature = (ranker))]
fn with_ranker<'py>(
mut slf: PyRefMut<'py, Self>,
ranker: Option<&str>,
) -> PyResult<PyRefMut<'py, Self>> {
slf.ranker = HybridSearchRanker::parse(ranker.unwrap_or_default()).map_err(to_py_err)?;
Ok(slf)
}

fn with_rrf_ranker(mut slf: PyRefMut<'_, Self>) -> PyRefMut<'_, Self> {
slf.ranker = HybridSearchRanker::Rrf;
slf
}

fn with_weighted_score_ranker(mut slf: PyRefMut<'_, Self>) -> PyRefMut<'_, Self> {
slf.ranker = HybridSearchRanker::WeightedScore;
slf
}

fn with_filter<'py>(
mut slf: PyRefMut<'py, Self>,
predicate: &Bound<'_, PyDict>,
) -> PyResult<PyRefMut<'py, Self>> {
let filter = dict_to_table_predicate(predicate, slf.table.schema(), true)?;
slf.filters.push(filter);
Ok(slf)
}

fn with_partition_filter<'py>(
mut slf: PyRefMut<'py, Self>,
predicate: &Bound<'_, PyDict>,
) -> PyResult<PyRefMut<'py, Self>> {
let filter = dict_to_table_predicate(predicate, slf.table.schema(), true)?;
slf.builder()
.with_partition_filter(filter.clone())
.map_err(to_py_err)?;
slf.filters.push(filter);
Ok(slf)
}

fn execute_local(&self, py: Python<'_>) -> PyResult<PyHybridSearchResult> {
let builder = self.builder();
py.detach(|| runtime().block_on(builder.execute_scored()))
.map(|inner| PyHybridSearchResult { inner })
.map_err(to_py_err)
}
}

#[pyclass(name = "HybridSearchResult", module = "pypaimon_rust.datafusion")]
pub struct PyHybridSearchResult {
inner: ScoredRowIds,
}

#[pymethods]
impl PyHybridSearchResult {
fn __len__(&self) -> usize {
self.inner.len()
}

fn row_ids(&self) -> HashMap<u64, f32> {
self.inner
.row_ids
.iter()
.copied()
.zip(self.inner.scores.iter().copied())
.collect()
}
}
1 change: 1 addition & 0 deletions bindings/python/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@ mod blob_uri_reader;
mod context;
mod error;
mod full_text_search;
mod hybrid_search;
mod index_build;
mod merge_into;
mod oss_cpp;
Expand Down
4 changes: 4 additions & 0 deletions bindings/python/src/table.rs
Original file line number Diff line number Diff line change
Expand Up @@ -202,6 +202,10 @@ impl PyTable {
crate::full_text_search::PyFullTextSearchBuilder::new(Arc::clone(&self.inner))
}

fn new_hybrid_search_builder(&self) -> crate::hybrid_search::PyHybridSearchBuilder {
crate::hybrid_search::PyHybridSearchBuilder::new(Arc::clone(&self.inner))
}

fn new_vector_search_builder(&self) -> crate::vector_search::PyVectorSearchBuilder {
crate::vector_search::PyVectorSearchBuilder::new(Arc::clone(&self.inner))
}
Expand Down
9 changes: 5 additions & 4 deletions crates/integrations/datafusion/src/hybrid_search.rs
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@
//! SELECT * FROM hybrid_search(
//! 'table_name',
//! array(named_struct('field', 'embedding', 'query_vector', array(1.0, 0.0))),
//! array(named_struct('column', 'content', 'query', 'paimon')),
//! array(named_struct('column', 'content', 'query', '{"match":{"query":"paimon"}}')),
//! 10,
//! 'rrf')
//! ```
Expand Down Expand Up @@ -53,7 +53,9 @@ use paimon::table::{HybridSearchRanker, HybridSearchRoute, Table};
use crate::error::to_datafusion_error;
use crate::physical_plan::{SearchScoreExec, SearchScoreOutputColumn};
use crate::runtime::{await_with_runtime, block_on_with_runtime};
use crate::table::{datafusion_read_fields, PaimonScanBuilder, PaimonTableProvider};
use crate::table::{
datafusion_arrow_schema, datafusion_read_fields, PaimonScanBuilder, PaimonTableProvider,
};
use crate::table_function_args::{
extract_int_literal, extract_string_literal, parse_table_identifier,
};
Expand Down Expand Up @@ -248,8 +250,7 @@ impl TableProvider for HybridSearchTableProvider {
let inner_schema = self.inner.schema();
let score_index = inner_schema.fields().len();
let input_read_fields = search_read_fields(table)?;
let input_schema = paimon::arrow::build_target_arrow_schema(&input_read_fields)
.map_err(to_datafusion_error)?;
let input_schema = datafusion_arrow_schema(&input_read_fields, true)?;
let row_id_table_index = input_read_fields
.iter()
.position(|field| field.name() == ROW_ID_FIELD_NAME)
Expand Down
Loading
Loading