blob: b998402ce34ba07c737d14ba7aecbd31ab2e2b8e [file]
/*
* Licensed to the Apache Software Foundation (ASF) under one
* or more contributor license agreements. See the NOTICE file
* distributed with this work for additional information
* regarding copyright ownership. The ASF licenses this file
* to you under the Apache License, Version 2.0 (the
* "License"); you may not use this file except in compliance
* with the License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing,
* software distributed under the License is distributed on an
* "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
* KIND, either express or implied. See the License for the
* specific language governing permissions and limitations
* under the License.
*/
#include <algorithm>
#include <cassert>
#include <memory>
#include <string>
#include <vector>
#include "arrow/array.h"
#include "arrow/c/bridge.h"
#include "arrow/ipc/api.h"
#include "arrow/type.h"
#include "gtest/gtest.h"
#include "paimon/common/utils/path_util.h"
#include "paimon/core/global_index/global_index_file_manager.h"
#include "paimon/core/index/index_path_factory.h"
#include "paimon/fs/local/local_file_system.h"
#include "paimon/global_index/bitmap_global_index_result.h"
#include "paimon/global_index/bitmap_scored_global_index_result.h"
#include "paimon/global_index/tantivy/tantivy_archive_layout.h"
#include "paimon/global_index/tantivy/tantivy_defs.h"
#include "paimon/global_index/tantivy/tantivy_global_index.h"
#include "paimon/global_index/tantivy/tantivy_global_index_reader.h"
#include "paimon/global_index/tantivy/tantivy_global_index_writer.h"
#include "paimon/predicate/full_text_search.h"
#include "paimon/testing/utils/testharness.h"
#ifndef JIEBA_TEST_DICT_DIR
#error "JIEBA_TEST_DICT_DIR must be set at compile time"
#endif
#ifndef PAIMON_TANTIVY_JAVA_FIXTURE_DIR
#error "PAIMON_TANTIVY_JAVA_FIXTURE_DIR must be set at compile time"
#endif
namespace paimon::tantivy::test {
namespace {
class FixturePathFactory : public IndexPathFactory {
public:
explicit FixturePathFactory(const std::string& root) : root_(root) {}
std::string NewPath() const override {
assert(false);
return "";
}
std::string ToPath(const std::shared_ptr<IndexFileMeta>&) const override {
assert(false);
return "";
}
std::string ToPath(const std::string& file_name) const override {
return PathUtil::JoinPath(root_, file_name);
}
bool IsExternalPath() const override {
return false;
}
private:
std::string root_;
};
class JavaCompatTest : public ::testing::Test {
public:
/// Build a TantivyGlobalIndexReader on top of the Java-produced fixture.
/// `fixture_name` is relative to `PAIMON_TANTIVY_JAVA_FIXTURE_DIR`.
std::shared_ptr<GlobalIndexReader> OpenFixture(const std::string& fixture_name) {
std::string fixture_dir = PAIMON_TANTIVY_JAVA_FIXTURE_DIR;
std::string archive_path = PathUtil::JoinPath(fixture_dir, fixture_name);
EXPECT_OK_AND_ASSIGN(auto file_status, fs_->GetFileStatus(archive_path));
int64_t file_size = file_status.GetLen();
EXPECT_GT(file_size, 4) << "fixture archive must exist and be > 4 bytes";
// Empty metadata (options not needed for cross-read — we use defaults)
std::string metadata_json = "{}";
auto meta_bytes = std::make_shared<Bytes>(metadata_json, pool_.get());
GlobalIndexIOMeta io_meta(archive_path, file_size, meta_bytes);
std::map<std::string, std::string> options;
auto global_index = std::make_shared<TantivyGlobalIndex>(options);
auto path_factory = std::make_shared<FixturePathFactory>(fixture_dir);
auto file_reader = std::make_shared<GlobalIndexFileManager>(fs_, path_factory);
auto data_type = arrow::struct_({arrow::field("f0", arrow::utf8())});
auto c_schema = std::make_unique<::ArrowSchema>();
EXPECT_TRUE(arrow::ExportType(*data_type, c_schema.get()).ok());
EXPECT_OK_AND_ASSIGN(auto reader_res, global_index->CreateReader(
c_schema.get(), file_reader, {io_meta}, pool_));
return reader_res;
}
std::shared_ptr<FullTextSearch> BuildFts(FullTextSearch::SearchType type,
const std::string& query) {
return std::make_shared<FullTextSearch>(
/*_field_name=*/"f0",
/*_limit=*/std::optional<int32_t>{},
/*_query=*/query,
/*_search_type=*/type,
/*_pre_filter=*/std::optional<RoaringBitmap64>{});
}
/// Run the search and return the sorted row_ids from the result bitmap.
std::vector<int64_t> RunSearchRowIds(const std::shared_ptr<GlobalIndexReader>& reader,
FullTextSearch::SearchType type,
const std::string& query) {
auto fts = BuildFts(type, query);
// This helper returns a value, so gtest ASSERT_* (which `return;`) cannot
// be used here; use the EXPECT_OK family.
EXPECT_OK_AND_ASSIGN(std::shared_ptr<GlobalIndexResult> r,
reader->VisitFullTextSearch(fts));
const RoaringBitmap64* bitmap = nullptr;
if (auto plain = std::dynamic_pointer_cast<BitmapGlobalIndexResult>(r)) {
EXPECT_OK_AND_ASSIGN(bitmap, plain->GetBitmap());
} else if (auto scored = std::dynamic_pointer_cast<BitmapScoredGlobalIndexResult>(r)) {
EXPECT_OK_AND_ASSIGN(bitmap, scored->GetBitmap());
}
EXPECT_TRUE(bitmap != nullptr);
if (bitmap == nullptr) {
return {};
}
std::vector<int64_t> out;
for (auto it = bitmap->Begin(); it != bitmap->End(); ++it) {
out.push_back(static_cast<int64_t>(*it));
}
std::sort(out.begin(), out.end());
return out;
}
protected:
std::shared_ptr<MemoryPool> pool_ = GetDefaultPool();
std::shared_ptr<FileSystem> fs_ = std::make_shared<LocalFileSystem>();
};
} // namespace
// ============================================================================
// 1. Archive basics: opening the Java-produced fixture succeeds
// ============================================================================
TEST_F(JavaCompatTest, OpenJavaArchiveSucceeds) {
auto reader = OpenFixture("english_simple.archive");
ASSERT_TRUE(reader != nullptr);
}
// ============================================================================
// 2. MATCH_ALL — single and multi-term
// ============================================================================
TEST_F(JavaCompatTest, MatchAllApple) {
auto reader = OpenFixture("english_simple.archive");
auto ids = RunSearchRowIds(reader, FullTextSearch::SearchType::MATCH_ALL, "apple");
// Docs containing "apple": 0 ("apple banana cherry"), 1 ("apple durian"),
// 4 ("apple cherry fig"), 7 ("apple")
ASSERT_EQ(ids, (std::vector<int64_t>{0, 1, 4, 7}));
}
TEST_F(JavaCompatTest, MatchAllAppleBananaIntersection) {
auto reader = OpenFixture("english_simple.archive");
auto ids = RunSearchRowIds(reader, FullTextSearch::SearchType::MATCH_ALL, "apple banana");
// Only doc 0 contains both "apple" and "banana"
ASSERT_EQ(ids, (std::vector<int64_t>{0}));
}
// ============================================================================
// 3. MATCH_ANY — union
// ============================================================================
TEST_F(JavaCompatTest, MatchAnyDurianElderberryUnion) {
auto reader = OpenFixture("english_simple.archive");
auto ids = RunSearchRowIds(reader, FullTextSearch::SearchType::MATCH_ANY, "durian elderberry");
// durian: 1, 6 elderberry: 5, 8 union: {1, 5, 6, 8}
ASSERT_EQ(ids, (std::vector<int64_t>{1, 5, 6, 8}));
}
// ============================================================================
// 4. PHRASE — consecutive term order matters
// ============================================================================
TEST_F(JavaCompatTest, PhraseAppleBanana) {
auto reader = OpenFixture("english_simple.archive");
auto ids = RunSearchRowIds(reader, FullTextSearch::SearchType::PHRASE, "apple banana");
// Only doc 0 has "apple banana" as consecutive phrase
ASSERT_EQ(ids, (std::vector<int64_t>{0}));
}
TEST_F(JavaCompatTest, PhraseBananaCherry) {
auto reader = OpenFixture("english_simple.archive");
auto ids = RunSearchRowIds(reader, FullTextSearch::SearchType::PHRASE, "banana cherry");
// "banana cherry" consecutive in doc 0 ("apple banana cherry") and doc 2 ("banana cherry")
ASSERT_EQ(ids, (std::vector<int64_t>{0, 2}));
}
// ============================================================================
// 5. PREFIX — byte-level (not tokenized) via RegexQuery
// ============================================================================
TEST_F(JavaCompatTest, PrefixAp) {
auto reader = OpenFixture("english_simple.archive");
auto ids = RunSearchRowIds(reader, FullTextSearch::SearchType::PREFIX, "ap");
// Tokens starting with "ap": "apple" → docs 0, 1, 4, 7
ASSERT_EQ(ids, (std::vector<int64_t>{0, 1, 4, 7}));
}
// ============================================================================
// 6. WILDCARD — glob-style via regex
// ============================================================================
TEST_F(JavaCompatTest, WildcardErr) {
auto reader = OpenFixture("english_simple.archive");
auto ids = RunSearchRowIds(reader, FullTextSearch::SearchType::WILDCARD, "*err*");
// Tokens matching *err*: "cherry" (0,2,4,6,9), "elderberry" (5,8)
ASSERT_EQ(ids, (std::vector<int64_t>{0, 2, 4, 5, 6, 8, 9}));
}
// ============================================================================
// 7. row_id invariant — must return the *caller-supplied* row_ids (not doc_ids)
// ============================================================================
TEST_F(JavaCompatTest, AllDocsReachableByRowId) {
auto reader = OpenFixture("english_simple.archive");
// Union of all terms matches all 10 docs.
auto ids = RunSearchRowIds(reader, FullTextSearch::SearchType::MATCH_ANY,
"apple banana cherry durian fig grape elderberry");
ASSERT_EQ(ids, (std::vector<int64_t>{0, 1, 2, 3, 4, 5, 6, 7, 8, 9}));
// This confirms Java wrote row_ids 0..9 via `addDocument(rowId, text)` and
// paimon-cpp reader extracted them via fast_fields().u64("row_id") —
// the schema invariant survives round-trip across implementations.
}
// ============================================================================
// 8. Probe: real paimon-java production archive (handed over by Java team).
// Data was claimed to be (id INT, content STRING) with 5 rows but ids
// rewritten multiple times; dump layout + per-term hits so caller can
// reverse-engineer what's actually inside.
// ============================================================================
TEST_F(JavaCompatTest, ProductionSampleProbe) {
const std::string fixture_name = "production_sample.archive";
// Open a reader over the Java-written production sample archive.
auto reader = OpenFixture(fixture_name);
ASSERT_TRUE(reader != nullptr);
// Keywords expected from the production text samples; tokenizer is "default"
// (lowercased, word-granular).
const std::vector<std::string> probes = {
"apache", "paimon", "is", "a", "lake", "format", "supports",
"full", "text", "search", "in", "vector", "similarity", "using",
"lumina", "streaming", "and", "batch", "processing", "engine",
};
// The archive must be readable — at least one probe term hits.
bool any_hit = false;
for (const auto& term : probes) {
if (!RunSearchRowIds(reader, FullTextSearch::SearchType::MATCH_ALL, term).empty()) {
any_hit = true;
break;
}
}
ASSERT_TRUE(any_hit) << "no probe term hit; archive may be empty or schema mismatched";
}
// ============================================================================
// 9. Reverse direction: paimon-cpp writes with tokenizer="default" → fixture
// consumed by paimon-java test. This test emits the archive into
// test/test_data/cpp_tantivy_fixtures/english_default.archive and
// round-trips it through the cpp reader first (schema-driven tokenizer
// dispatch picks "default" automatically).
// ============================================================================
namespace {
/// GlobalIndexFileWriter that emits to a single fixed filename under `root`.
/// Mirrors paimon-java's `FixedNameLocalFileWriter` from
/// `TantivyIndexFixtureGen.java`: `newFileName(prefix)` ignores the prefix and
/// always returns the caller-chosen name. Used to produce a stable fixture
/// path consumed by the paimon-java cross-read test.
class FixedNameGlobalIndexFileWriter : public GlobalIndexFileWriter {
public:
FixedNameGlobalIndexFileWriter(std::shared_ptr<FileSystem> fs, std::string root,
std::string fixed_name)
: fs_(std::move(fs)), root_(std::move(root)), fixed_name_(std::move(fixed_name)) {}
Result<std::string> NewFileName(const std::string& /*prefix*/) const override {
return fixed_name_;
}
std::string ToPath(const std::string& file_name) const override {
return PathUtil::JoinPath(root_, file_name);
}
Result<std::unique_ptr<OutputStream>> NewOutputStream(
const std::string& file_name) const override {
return fs_->Create(ToPath(file_name), /*overwrite=*/true);
}
Result<int64_t> GetFileSize(const std::string& file_name) const override {
PAIMON_ASSIGN_OR_RAISE(FileStatus file_status, fs_->GetFileStatus(ToPath(file_name)));
return file_status.GetLen();
}
private:
std::shared_ptr<FileSystem> fs_;
std::string root_;
std::string fixed_name_;
};
/// Same 10-doc English corpus paimon-java uses in TantivyIndexFixtureGen
/// (pure ASCII, no punctuation inside words). SimpleTokenizer (tantivy's
/// "default") tokenizes identically on both sides for this subset, so the
/// golden row_ids match byte-for-byte between cpp-write and java-read.
constexpr const char* kEnglishDocs[] = {
"apple banana cherry", // 0
"apple durian", // 1
"banana cherry", // 2
"fig grape", // 3
"apple cherry fig", // 4
"banana elderberry", // 5
"cherry durian", // 6
"apple", // 7
"grape fig elderberry", // 8
"cherry fig", // 9
};
} // namespace
TEST_F(JavaCompatTest, CppWriteDefaultTokenizerForJavaCrossRead) {
// 1) Produce an archive into test/test_data/cpp_tantivy_fixtures/ via the
// production TantivyGlobalIndexWriter, configured with tantivy's
// built-in "default" tokenizer (same as paimon-java's TEXT field).
const std::string out_dir = PAIMON_TANTIVY_CPP_FIXTURE_DIR;
const std::string fixture_name = "english_default.archive";
// Ensure dir exists (CMake does NOT create it automatically).
ASSERT_OK(fs_->Mkdirs(out_dir));
// Clean any prior fixture so each test run writes fresh bytes.
{
const std::string archive_path_cleanup = PathUtil::JoinPath(out_dir, fixture_name);
auto existing = fs_->GetFileStatus(archive_path_cleanup);
if (existing.ok()) {
ASSERT_TRUE(fs_->Delete(archive_path_cleanup, false).ok());
}
}
auto file_writer = std::make_shared<FixedNameGlobalIndexFileWriter>(fs_, out_dir, fixture_name);
auto data_type = arrow::struct_({arrow::field("f0", arrow::utf8())});
std::map<std::string, std::string> options{
{kTantivyWriteTokenizer, "default"},
};
ASSERT_OK_AND_ASSIGN(auto writer_res, TantivyGlobalIndexWriter::Create(
"f0", data_type, file_writer, options, pool_));
auto writer = writer_res;
// Build an arrow batch from kEnglishDocs.
std::string json = "[";
for (std::size_t i = 0; i < sizeof(kEnglishDocs) / sizeof(kEnglishDocs[0]); ++i) {
if (i > 0) {
json += ",";
}
json += "[\"";
json += kEnglishDocs[i];
json += "\"]";
}
json += "]";
auto array = arrow::ipc::internal::json::ArrayFromJSON(data_type, json).ValueOrDie();
::ArrowArray c_array;
ASSERT_TRUE(arrow::ExportArray(*array, &c_array).ok());
std::vector<int64_t> relative_row_ids(array->length());
for (int64_t i = 0; i < array->length(); ++i) {
relative_row_ids[i] = i;
}
ASSERT_TRUE(writer->AddBatch(&c_array, std::move(relative_row_ids)).ok());
ASSERT_OK_AND_ASSIGN(auto metas_res, writer->Finish());
ASSERT_EQ(metas_res.size(), 1u);
const auto& meta = metas_res.front();
const std::string archive_path = meta.file_path;
// 2) Archive header sanity: 16+ files, meta.json present, tokenizer in schema.
ASSERT_OK_AND_ASSIGN(std::shared_ptr<InputStream> stream, fs_->Open(archive_path));
ASSERT_OK_AND_ASSIGN(auto layout_res, ArchiveLayout::Parse(stream.get()));
const auto& layout = layout_res;
bool has_meta_json = false;
for (std::size_t i = 0; i < layout.count; ++i) {
if (layout.names[i] == "meta.json") {
has_meta_json = true;
}
}
ASSERT_TRUE(has_meta_json);
// 3) Round-trip through the cpp reader first — the reader must auto-register
// "default" from the schema so the search path works without passing
// any reader-side tokenizer config.
// Build a reader directly off the archive path (mirrors OpenFixture
// but rooted at the cpp fixtures dir).
ASSERT_OK_AND_ASSIGN(auto file_status, fs_->GetFileStatus(archive_path));
int64_t file_size = file_status.GetLen();
auto meta_bytes = std::make_shared<Bytes>(std::string("{}"), pool_.get());
GlobalIndexIOMeta io_meta(archive_path, file_size, meta_bytes);
auto reader_factory =
std::make_shared<TantivyGlobalIndex>(std::map<std::string, std::string>{});
auto reader_path_factory = std::make_shared<FixturePathFactory>(out_dir);
auto reader_file_mgr = std::make_shared<GlobalIndexFileManager>(fs_, reader_path_factory);
auto c_schema = std::make_unique<::ArrowSchema>();
ASSERT_TRUE(arrow::ExportType(*data_type, c_schema.get()).ok());
ASSERT_OK_AND_ASSIGN(auto reader_res, reader_factory->CreateReader(
c_schema.get(), reader_file_mgr, {io_meta}, pool_));
auto reader = reader_res;
// Golden expectations (identical to paimon-java's english_simple.golden.json)
ASSERT_EQ(RunSearchRowIds(reader, FullTextSearch::SearchType::MATCH_ALL, "apple"),
(std::vector<int64_t>{0, 1, 4, 7}));
ASSERT_EQ(RunSearchRowIds(reader, FullTextSearch::SearchType::MATCH_ALL, "apple banana"),
(std::vector<int64_t>{0}));
ASSERT_EQ(RunSearchRowIds(reader, FullTextSearch::SearchType::MATCH_ANY, "durian elderberry"),
(std::vector<int64_t>{1, 5, 6, 8}));
ASSERT_EQ(RunSearchRowIds(reader, FullTextSearch::SearchType::PHRASE, "apple banana"),
(std::vector<int64_t>{0}));
ASSERT_EQ(RunSearchRowIds(reader, FullTextSearch::SearchType::PHRASE, "banana cherry"),
(std::vector<int64_t>{0, 2}));
ASSERT_EQ(RunSearchRowIds(reader, FullTextSearch::SearchType::PREFIX, "ap"),
(std::vector<int64_t>{0, 1, 4, 7}));
ASSERT_EQ(RunSearchRowIds(reader, FullTextSearch::SearchType::WILDCARD, "*err*"),
(std::vector<int64_t>{0, 2, 4, 5, 6, 8, 9}));
ASSERT_EQ(RunSearchRowIds(reader, FullTextSearch::SearchType::MATCH_ANY,
"apple banana cherry durian fig grape elderberry"),
(std::vector<int64_t>{0, 1, 2, 3, 4, 5, 6, 7, 8, 9}));
}
} // namespace paimon::tantivy::test