Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 26 additions & 4 deletions BerlinMOD/benchmarks/batch/bench/bench_mduck.sh
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,7 @@
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
BERLINMOD_DIR="$(cd "${SCRIPT_DIR}/.." && pwd)"
QUERIES_SQL="${SCRIPT_DIR}/queries.sql" # shared single-source query set

# ── defaults ──────────────────────────────────────────────────────────────────
DUCKDB="${DUCKDB:-duckdb}"
Expand Down Expand Up @@ -114,7 +115,7 @@ _duck_q() { "$DUCKDB" "$DBFILE" -noheader -list -c "$1" 2>/dev/null; }
if $LOAD; then
echo "=== Loading data into: $DBFILE ==="
rm -f "$DBFILE"
LOAD_BODY="$(sed '/^SET VARIABLE DATADIR/d' "${BERLINMOD_DIR}/load_mduck.sql")"
LOAD_BODY="$(sed '/^SET VARIABLE DATADIR/d' "${SCRIPT_DIR}/load_mduck.sql")"
LOAD_SQL="${MOBILITY_LOAD} SET VARIABLE DATADIR='${DATADIR}/'; ${LOAD_BODY}"
"$DUCKDB" "$DBFILE" -c "$LOAD_SQL"
echo " done."
Expand Down Expand Up @@ -165,10 +166,31 @@ echo ""
TIMEFILE=$(mktemp)
trap 'rm -f "$TIMEFILE"' EXIT

# ── extract each selected query from the shared queries.sql ──────────────────
# queries.sql is the single canonical source; every query is delimited by a
# `-- BerlinMOD Q<n>:` (or `-- BerlinMOD QRT:`) marker and runs to the next one.
marker_for() { # q07 -> Q7 ; q13 -> Q13 ; qrt -> QRT
local q="$1"
if [[ "$q" == "qrt" ]]; then echo "QRT"; else echo "Q$((10#${q#q}))"; fi
}
extract_query() { # emit the SQL body for a marker
awk -v start="-- BerlinMOD ${1}:" '
index($0, start)==1 { grab=1; print; next }
grab && /^-- BerlinMOD Q/ { exit }
grab { print }
' "$QUERIES_SQL" | grep -v '^\s*--' | tr '\n' ' '
}

for Q in "${QUERIES[@]}"; do
QFILE="${BERLINMOD_DIR}/${Q}.sql"
[[ -f "$QFILE" ]] || { echo " [skip] ${Q} — SQL file not found"; continue; }
QSQL=$(grep -v '^\s*--' "$QFILE" | tr '\n' ' ')
QSQL=$(extract_query "$(marker_for "$Q")")
[[ -n "${QSQL// /}" ]] || { echo " [skip] ${Q} — not found in queries.sql"; continue; }
# Validate once: a query the extension cannot run (missing function, etc.) is
# reported and EXCLUDED, never recorded as a phantom sub-second timing.
ERR=$("$DUCKDB" "$DBFILE" -c "${MOBILITY_LOAD} SET search_path='portable,main'; ${QSQL}" 2>&1 >/dev/null) || true
if echo "$ERR" | grep -qiE "error"; then
echo " [FAIL] ${Q} — $(echo "$ERR" | grep -iE 'error' | head -1)"
continue
fi
printf " timing %-6s: " "$Q"
for RUN in $(seq 1 "$RUNS"); do
T0=$(date +%s%3N)
Expand Down
75 changes: 75 additions & 0 deletions BerlinMOD/benchmarks/batch/bench/load_mduck.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,75 @@
--------------------------------------------------------------------------------
-- BerlinMOD Batch Load — MobilityDuck / DuckDB
--------------------------------------------------------------------------------
--
-- This file is part of MobilityDB.
-- Copyright(c) 2020-2026, Université libre de Bruxelles and MobilityDB
-- contributors
--
--------------------------------------------------------------------------------
--
-- Loads the cross-platform BerlinMOD portability export (produced by
-- berlinmod_portability_export() in berlinmod_export.sql) into the schema the
-- shared batch queries (bench/queries.sql) expect — the SAME CSV set the
-- MobilityDB and MobilitySpark runners consume, so all platforms benchmark the
-- identical corpus.
--
-- Input CSVs (DATADIR is injected by bench_mduck.sh):
-- vehicles.csv : vehId, licence, type, model
-- trips.csv : tripId, vehId, trip -- tgeompoint as hex-EWKB
-- query_licences.csv : licenceId, licence
-- query_instants.csv : instantId, instant
-- query_points.csv : pointId, geom -- geometry as EWKT
-- query_periods.csv : periodId, period -- tstzspan as text
-- query_regions.csv : regionId, geom -- geometry as EWKT
--
-- SRID is carried by the serialization and resolved by MEOS parsers — never
-- parsed or reprojected by hand: the trip comes as hex-EWKB
-- (tgeompointFromHexEWKB) and the static geometries as SRID-tagged EWKT
-- (ST_GeomFromText accepts the SRID= prefix). The th3index prefilter column is
-- built in-engine from the lat/lon trajectory, exactly as a real ingest would.
--------------------------------------------------------------------------------

-- Default for standalone use; bench_mduck.sh strips this line and injects the
-- real data directory.
SET VARIABLE DATADIR='./data/';

-- The runner selects `SET search_path='portable,main'`; create the schema so that
-- selection resolves (the tables live in main and are found via the fallback).
CREATE SCHEMA IF NOT EXISTS portable;

-- ── Vehicles ──────────────────────────────────────────────────────────────────
CREATE OR REPLACE TABLE Vehicles AS
SELECT vehId, licence, type, model
FROM read_csv(getvariable('DATADIR') || 'vehicles.csv', header = true);

-- ── Trips ─────────────────────────────────────────────────────────────────────
-- trip arrives as hex-EWKB (SRID embedded); trip_h3 is computed here.
CREATE OR REPLACE TABLE Trips AS
SELECT tripId, vehId, tgeompointFromHexEWKB(trip) AS trip
FROM read_csv(getvariable('DATADIR') || 'trips.csv', header = true, all_varchar = true);
ALTER TABLE Trips ADD COLUMN trip_h3 TH3INDEX;
UPDATE Trips SET trip_h3 = th3index(transform(trip, 4326), 7);

-- ── Query parameter tables (100-row selectors used by queries.sql) ────────────
CREATE OR REPLACE TABLE QueryLicences AS
SELECT licenceId, licence
FROM read_csv(getvariable('DATADIR') || 'query_licences.csv', header = true);

CREATE OR REPLACE TABLE QueryInstants AS
SELECT instantId, instant::TIMESTAMPTZ AS instant
FROM read_csv(getvariable('DATADIR') || 'query_instants.csv', header = true, all_varchar = true);

CREATE OR REPLACE TABLE QueryPeriods AS
SELECT periodId, period::TSTZSPAN AS period
FROM read_csv(getvariable('DATADIR') || 'query_periods.csv', header = true, all_varchar = true);

-- Points carry both the parsed geometry and the SRID-tagged EWKT text
-- (queries.sql surfaces p.geomWKT); geomWKT is the export text verbatim.
CREATE OR REPLACE TABLE QueryPoints AS
SELECT pointId, ST_GeomFromText(geom) AS geom, geom AS geomWKT
FROM read_csv(getvariable('DATADIR') || 'query_points.csv', header = true, all_varchar = true);

CREATE OR REPLACE TABLE QueryRegions AS
SELECT regionId, ST_GeomFromText(geom) AS geom
FROM read_csv(getvariable('DATADIR') || 'query_regions.csv', header = true, all_varchar = true);
Loading