tree: 5c65b7c899259cfd44ca5ae7cbda305aeaebb05a
  1. python/
  2. src/
  3. tests/
  4. .gitignore
  5. Cargo.toml
  6. DEPENDENCIES.rust.tsv
  7. LICENSE
  8. Makefile
  9. NOTICE
  10. project-description.md
  11. pyproject.toml
  12. README.md
bindings/python/README.md

PyPaimon Rust

This project builds the Rust-powered core for PyPaimon while also providing DataFusion integration for querying Paimon tables.

Usage

import pyarrow as pa
from pypaimon_rust.datafusion import SQLContext

# Create a SQL context and register a Paimon catalog
ctx = SQLContext()
ctx.register_catalog("paimon", {"warehouse": "/tmp/paimon-warehouse"})

# Create a table and insert data
ctx.sql("CREATE SCHEMA paimon.my_db")
ctx.sql("CREATE TABLE paimon.my_db.users (id INT, name STRING, PRIMARY KEY (id))")
ctx.sql("INSERT INTO paimon.my_db.users VALUES (1, 'alice'), (2, 'bob')")

# Query data
batches = ctx.sql("SELECT id, name FROM paimon.my_db.users ORDER BY id")

# Inspect BLOB media or build thumbnails when installed with pypaimon-rust[video]
batches = ctx.sql(
    "SELECT id, media_info(content), media_thumbnail(content, 160, 90) "
    "FROM paimon.my_db.assets"
)

# Register a temporary table from a PyArrow RecordBatch
batch = pa.record_batch([[1, 2], ["alice", "bob"]], names=["id", "name"])
ctx.register_batch("paimon.default.my_temp", batch)
batches = ctx.sql("SELECT * FROM paimon.default.my_temp")

# Drop it via SQL when no longer needed
ctx.sql("DROP TEMPORARY TABLE paimon.default.my_temp")

For the full SQL reference, see the SQL Integration docs.

Native Read / Write

Beyond SQL, you can use the lower-level read and write APIs directly from Python. Time travel is supported via the options dict on new_read_builder.

import pyarrow as pa
from pypaimon_rust.datafusion import SQLContext, PaimonCatalog

WAREHOUSE = "/tmp/paimon-warehouse"

# --- DDL/DML via DataFusion SQLContext ---
ctx = SQLContext()
ctx.register_catalog("paimon", {"warehouse": WAREHOUSE})
ctx.sql("CREATE SCHEMA paimon.my_db")
ctx.sql("CREATE TABLE paimon.my_db.users (id INT, name STRING, PRIMARY KEY (id))")
ctx.sql("INSERT INTO paimon.my_db.users VALUES (1, 'alice'), (2, 'bob')")
catalog = PaimonCatalog({"warehouse": WAREHOUSE})
table = catalog.get_table("my_db.users")

# --- Read data ---
read_builder = table.new_read_builder().with_projection(["id", "name"]).with_limit(100)
scan = read_builder.new_scan()
plan = scan.plan()
batches = read_builder.new_read().read(plan.splits())

print(f"\nRead: {batches[0].num_rows} rows")
print(batches[0])

# --- Write data, from a PyArrow RecordBatch ---
batch = pa.record_batch(
    [[3, 4], ["charlie", "diana"]],
    schema=pa.schema([("id", pa.int32()), ("name", pa.utf8())]),
)
write_builder = table.new_write_builder()
writer = write_builder.new_write()
writer.write_arrow(batch)
commit_messages = writer.prepare_commit()
write_builder.new_commit().commit(commit_messages)

# --- Time travel: read a past version ---
# Supported options: scan.version, scan.timestamp-millis, scan.snapshot-id, or scan.tag-name
read_builder_tt = table.new_read_builder({"scan.snapshot-id": "1"})
scan_tt = read_builder_tt.new_scan()
plan_tt = scan_tt.plan()
batches_tt = read_builder_tt.new_read().read(plan_tt.splits())

print(f"\nRead: {batches_tt[0].num_rows} rows")
print(batches_tt[0])

Setup

Install uv:

pip install uv

Set up the development environment:

make install

Build

make build

Test

Python integration tests expect the shared Paimon test warehouse to be prepared first from the repository root:

make docker-up
cd bindings/python
make test