ADBC

CedarDB is compatible with the ADBC PostgreSQL driver for Arrow-native database access.

Installing Driver

Install the PostgreSQL ADBC driver with dbc:

dbc install postgresql

Connecting and Querying

The examples below load the 2015 NYC Street Tree Census into a trees table with ADBC bulk ingestion and then query the three cedars with the thickest trunks. Download the dataset (~220 MB) into your project directory:

curl -L -o trees.csv "https://data.cityofnewyork.us/api/views/uvpi-gqnh/rows.csv?accessType=DOWNLOAD"

Then pick your preferred language to install the ADBC client library and run the example:

Installing the C++ Client

Install the Arrow C++ and ADBC libraries with your system package manager.

On Debian or Ubuntu:

sudo apt install libarrow-dev libadbc-driver-manager-dev

Connecting and Querying with C++

#include <cstdlib>
#include <iostream>

#include <arrow-adbc/adbc.h>
#include <arrow-adbc/adbc_driver_manager.h>
#include <arrow/api.h>
#include <arrow/c/bridge.h>
#include <arrow/csv/api.h>
#include <arrow/io/api.h>

void Check(AdbcStatusCode status, AdbcError* error) {
  if (status != ADBC_STATUS_OK) {
    std::cerr << error->message << std::endl;
    std::exit(EXIT_FAILURE);
  }
}

int main() {
  // Read trees.csv into Arrow
  auto input = arrow::io::ReadableFile::Open("trees.csv").ValueOrDie();
  auto trees = arrow::csv::TableReader::Make(arrow::io::default_io_context(), input,
                                             arrow::csv::ReadOptions::Defaults(),
                                             arrow::csv::ParseOptions::Defaults(),
                                             arrow::csv::ConvertOptions::Defaults())
                   .ValueOrDie()
                   ->Read()
                   .ValueOrDie();

  // Connect to CedarDB
  AdbcError error = {};

  AdbcDatabase database = {};
  Check(AdbcDatabaseNew(&database, &error), &error);
  Check(AdbcDatabaseSetOption(&database, "driver", "postgresql", &error), &error);
  Check(AdbcDatabaseSetOption(&database, "uri",
                              "postgresql://<username>:<password>@localhost:5432/<dbname>", &error),
        &error);
  Check(AdbcDriverManagerDatabaseSetLoadFlags(&database, ADBC_LOAD_FLAG_DEFAULT, &error), &error);
  Check(AdbcDatabaseInit(&database, &error), &error);

  AdbcConnection connection = {};
  Check(AdbcConnectionNew(&connection, &error), &error);
  Check(AdbcConnectionInit(&connection, &database, &error), &error);

  AdbcStatement statement = {};
  Check(AdbcStatementNew(&connection, &statement, &error), &error);

  // Create the trees table and bulk load the data
  struct ArrowArrayStream stream = {};
  arrow::ExportRecordBatchReader(std::make_shared<arrow::TableBatchReader>(trees), &stream).ok();
  Check(AdbcStatementSetOption(&statement, ADBC_INGEST_OPTION_TARGET_TABLE, "trees", &error),
        &error);
  Check(AdbcStatementSetOption(&statement, ADBC_INGEST_OPTION_MODE,
                               ADBC_INGEST_OPTION_MODE_REPLACE, &error),
        &error);
  Check(AdbcStatementBindStream(&statement, &stream, &error), &error);
  Check(AdbcStatementExecuteQuery(&statement, nullptr, nullptr, &error), &error);

  // Run the query
  Check(AdbcStatementSetSqlQuery(&statement,
                                 "SELECT tree_id, spc_common, tree_dbh, address, borough FROM trees "
                                 "WHERE spc_common ILIKE '%cedar%' ORDER BY tree_dbh DESC LIMIT 3",
                                 &error),
        &error);
  Check(AdbcStatementExecuteQuery(&statement, &stream, nullptr, &error), &error);

  // Fetch and print the results
  auto reader = arrow::ImportRecordBatchReader(&stream).ValueOrDie();
  std::cout << reader->ToTable().ValueOrDie()->ToString() << std::endl;

  AdbcStatementRelease(&statement, &error);
  AdbcConnectionRelease(&connection, &error);
  AdbcDatabaseRelease(&database, &error);
  return EXIT_SUCCESS;
}

Query results are returned in Apache Arrow format.