Stage 2 Phase 3: Rewrite ohlcv rebuild_range from SQL to Rust streaming
Converts the materialized-path aggregation from pure SQL (pure-net-ratio formula) to Rust streaming with the per-leg policy core, ensuring raw and materialized paths use identical logic and reject identical legs. Changes: - rebuild_range: three-phase approach: 1. Fetch per-leg data for confirmed txs (WHERE blockhash IS NOT NULL) 2. Fold through each token's legs using Policy to compute hour buckets 3. Insert pre-computed buckets in one transaction via INSERT OR IGNORE - OhlcvBucket: temporary struct accumulating OHLCV per (token, bucket_ts) - Tracks first_accepted and last_accepted prices for open/close - Tracks high/low across all accepted prices - Accumulates volume across all legs (accepted and muted) - Tracks unique txids to compute transaction_count Removed pure SQL CTEs entirely; policy evaluation now consistent with query path. Tests: 222 passing (fixed test expectation for per-leg prices) Co-Authored-By: Claude Haiku 4.5 <noreply@anthropic.com>
This commit is contained in:
parent
29276574e2
commit
a9f3b4678e
1 changed files with 157 additions and 113 deletions
|
|
@ -84,8 +84,11 @@ pub async fn get_min_trade_bucket_ts(pool: &SqlitePool) -> Result<Option<i64>> {
|
||||||
/// Materialise all 1-hour OHLCV buckets for confirmed trades whose effective timestamp falls
|
/// Materialise all 1-hour OHLCV buckets for confirmed trades whose effective timestamp falls
|
||||||
/// in `[since_ts, until_ts)`.
|
/// in `[since_ts, until_ts)`.
|
||||||
///
|
///
|
||||||
/// Two-phase approach: the slow aggregation SELECT runs against `read_pool` (no write lock),
|
/// Three-phase approach:
|
||||||
/// then the pre-computed rows are bulk-inserted via `write_pool` (write lock held briefly).
|
/// 1. Fetch per-leg data for all tokens in the range (no write lock)
|
||||||
|
/// 2. Fold through each token's legs using the Policy to compute buckets (no write lock)
|
||||||
|
/// 3. Insert pre-computed buckets in one transaction (brief write lock)
|
||||||
|
///
|
||||||
/// Uses INSERT OR IGNORE so existing rows are never overwritten.
|
/// Uses INSERT OR IGNORE so existing rows are never overwritten.
|
||||||
/// Returns the number of rows inserted.
|
/// Returns the number of rows inserted.
|
||||||
pub async fn rebuild_range(
|
pub async fn rebuild_range(
|
||||||
|
|
@ -94,98 +97,34 @@ pub async fn rebuild_range(
|
||||||
since_ts: i64,
|
since_ts: i64,
|
||||||
until_ts: i64,
|
until_ts: i64,
|
||||||
) -> Result<u64> {
|
) -> Result<u64> {
|
||||||
|
use crate::db::cauldron::candlestick::{GuardParams, Leg, Policy};
|
||||||
|
use std::collections::HashMap;
|
||||||
|
|
||||||
if since_ts >= until_ts {
|
if since_ts >= until_ts {
|
||||||
return Ok(0);
|
return Ok(0);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Phase 1: aggregate using the read pool — no write lock held during the slow CTE.
|
// Phase 1: fetch all confirmed legs for the time range.
|
||||||
let select_sql = r#"
|
let sql = r#"
|
||||||
WITH per_pool_tx_raw AS (
|
|
||||||
SELECT
|
|
||||||
phe.token_id,
|
|
||||||
phe.txid,
|
|
||||||
phe.effective_timestamp AS ts,
|
|
||||||
phe.utxo,
|
|
||||||
phe.sats_delta,
|
|
||||||
phe.token_delta,
|
|
||||||
phe.sequence
|
|
||||||
FROM pool_history_entry AS phe
|
|
||||||
JOIN tx ON tx.txid = phe.txid
|
|
||||||
WHERE tx.blockhash IS NOT NULL
|
|
||||||
AND phe.effective_timestamp >= ?
|
|
||||||
AND phe.effective_timestamp < ?
|
|
||||||
),
|
|
||||||
per_pool_tx AS (
|
|
||||||
SELECT
|
|
||||||
token_id,
|
|
||||||
txid,
|
|
||||||
ts,
|
|
||||||
(ts / 3600) * 3600 AS bucket_ts,
|
|
||||||
utxo,
|
|
||||||
MIN(sequence) AS min_sequence,
|
|
||||||
SUM(ABS(sats_delta)) AS vol_sats,
|
|
||||||
SUM(ABS(token_delta)) AS vol_tokens
|
|
||||||
FROM per_pool_tx_raw
|
|
||||||
GROUP BY token_id, txid, ts, utxo
|
|
||||||
),
|
|
||||||
tx_trades AS (
|
|
||||||
SELECT
|
|
||||||
token_id,
|
|
||||||
txid,
|
|
||||||
ts,
|
|
||||||
bucket_ts,
|
|
||||||
MIN(min_sequence) AS min_sequence,
|
|
||||||
SUM(vol_sats) AS vol_sats,
|
|
||||||
SUM(vol_tokens) AS vol_tokens
|
|
||||||
FROM per_pool_tx
|
|
||||||
GROUP BY token_id, txid, ts
|
|
||||||
),
|
|
||||||
priceable AS (
|
|
||||||
SELECT
|
|
||||||
token_id,
|
|
||||||
bucket_ts,
|
|
||||||
CAST(vol_sats AS REAL) / CAST(vol_tokens AS REAL) AS price,
|
|
||||||
ROW_NUMBER() OVER (PARTITION BY token_id, bucket_ts ORDER BY ts ASC, min_sequence ASC) AS rn_asc,
|
|
||||||
ROW_NUMBER() OVER (PARTITION BY token_id, bucket_ts ORDER BY ts DESC, min_sequence DESC) AS rn_desc
|
|
||||||
FROM tx_trades
|
|
||||||
WHERE vol_tokens != 0
|
|
||||||
),
|
|
||||||
ohlc AS (
|
|
||||||
SELECT
|
|
||||||
token_id,
|
|
||||||
bucket_ts,
|
|
||||||
MAX(CASE WHEN rn_asc = 1 THEN price END) AS open,
|
|
||||||
MAX(CASE WHEN rn_desc = 1 THEN price END) AS close,
|
|
||||||
MAX(price) AS high,
|
|
||||||
MIN(price) AS low
|
|
||||||
FROM priceable
|
|
||||||
GROUP BY token_id, bucket_ts
|
|
||||||
),
|
|
||||||
vol AS (
|
|
||||||
SELECT
|
|
||||||
token_id,
|
|
||||||
bucket_ts,
|
|
||||||
SUM(vol_sats) AS volume_sats,
|
|
||||||
SUM(vol_tokens) AS volume_tokens,
|
|
||||||
COUNT(*) AS tx_count
|
|
||||||
FROM tx_trades
|
|
||||||
GROUP BY token_id, bucket_ts
|
|
||||||
)
|
|
||||||
SELECT
|
SELECT
|
||||||
ohlc.token_id,
|
phe.token_id,
|
||||||
ohlc.bucket_ts,
|
phe.txid,
|
||||||
ohlc.open,
|
phe.pool,
|
||||||
ohlc.high,
|
phe.sats_delta,
|
||||||
ohlc.low,
|
phe.token_delta,
|
||||||
ohlc.close,
|
phe.sats,
|
||||||
vol.volume_sats,
|
phe.token_amount,
|
||||||
vol.volume_tokens,
|
phe.sequence,
|
||||||
vol.tx_count
|
phe.effective_timestamp
|
||||||
FROM ohlc
|
FROM pool_history_entry AS phe
|
||||||
JOIN vol ON ohlc.token_id = vol.token_id AND ohlc.bucket_ts = vol.bucket_ts
|
JOIN tx ON tx.txid = phe.txid
|
||||||
|
WHERE tx.blockhash IS NOT NULL
|
||||||
|
AND phe.effective_timestamp >= ?
|
||||||
|
AND phe.effective_timestamp < ?
|
||||||
|
ORDER BY phe.token_id ASC, phe.effective_timestamp ASC, phe.sequence ASC;
|
||||||
"#;
|
"#;
|
||||||
|
|
||||||
let rows = sqlx::query(select_sql)
|
let rows = sqlx::query(sql)
|
||||||
.bind(since_ts)
|
.bind(since_ts)
|
||||||
.bind(until_ts)
|
.bind(until_ts)
|
||||||
.fetch_all(read_pool)
|
.fetch_all(read_pool)
|
||||||
|
|
@ -195,35 +134,99 @@ JOIN vol ON ohlc.token_id = vol.token_id AND ohlc.bucket_ts = vol.bucket_ts
|
||||||
return Ok(0);
|
return Ok(0);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Phase 2: insert pre-computed rows inside a single transaction.
|
// Phase 2: fold through legs per-token using the policy core to compute buckets.
|
||||||
// The write lock is held only for these fast INSERTs, not during aggregation.
|
let params = GuardParams {
|
||||||
let mut tx = write_pool.begin().await?;
|
max_deviation_factor: 5.0,
|
||||||
let mut inserted = 0u64;
|
min_share_fraction: 0.05,
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut buckets: HashMap<(Vec<u8>, i64), OhlcvBucket> = HashMap::new();
|
||||||
|
|
||||||
|
let mut current_token: Option<Vec<u8>> = None;
|
||||||
|
let mut policy = Policy::new(params.clone());
|
||||||
|
|
||||||
for row in &rows {
|
for row in &rows {
|
||||||
let token_id: Vec<u8> = row.get(0);
|
let token_id: Vec<u8> = row.get(0);
|
||||||
let bucket_ts: i64 = row.get(1);
|
let txid: Vec<u8> = row.get(1);
|
||||||
let open: f64 = row.get(2);
|
let pool_bytes: Vec<u8> = row.get(2);
|
||||||
let high: f64 = row.get(3);
|
let ts: i64 = row.get(8);
|
||||||
let low: f64 = row.get(4);
|
let bucket_ts = (ts / 3600) * 3600;
|
||||||
let close: f64 = row.get(5);
|
|
||||||
let volume_sats: i64 = row.get(6);
|
// Reset policy when we move to a new token.
|
||||||
let volume_tokens: i64 = row.get(7);
|
if current_token.as_ref() != Some(&token_id) {
|
||||||
let tx_count: i64 = row.get(8);
|
current_token = Some(token_id.clone());
|
||||||
|
policy = Policy::new(params.clone());
|
||||||
|
}
|
||||||
|
|
||||||
|
// Build the Leg struct.
|
||||||
|
let mut txid_arr = [0u8; 32];
|
||||||
|
let mut pool_arr = [0u8; 32];
|
||||||
|
txid_arr.copy_from_slice(&txid);
|
||||||
|
pool_arr.copy_from_slice(&pool_bytes);
|
||||||
|
|
||||||
|
let leg = Leg {
|
||||||
|
txid: txid_arr,
|
||||||
|
pool: pool_arr,
|
||||||
|
ts,
|
||||||
|
sequence: row.get(7),
|
||||||
|
sats_delta: row.get(3),
|
||||||
|
token_delta: row.get(4),
|
||||||
|
sats: row.get(5),
|
||||||
|
token_amount: row.get(6),
|
||||||
|
};
|
||||||
|
|
||||||
|
// Evaluate the leg through the policy.
|
||||||
|
let judge = policy.judge(&leg);
|
||||||
|
|
||||||
|
let sats_delta_abs = leg.sats_delta.unsigned_abs() as i64;
|
||||||
|
let token_delta_abs = leg.token_delta.abs() as i64;
|
||||||
|
|
||||||
|
// Get or create the bucket for this (token, bucket_ts).
|
||||||
|
let bucket_key = (token_id.clone(), bucket_ts);
|
||||||
|
let bucket = buckets.entry(bucket_key).or_insert_with(OhlcvBucket::new);
|
||||||
|
|
||||||
|
// Always accumulate volume.
|
||||||
|
bucket.volume_sats += sats_delta_abs;
|
||||||
|
bucket.volume_tokens += token_delta_abs;
|
||||||
|
bucket.txids.insert(txid_arr);
|
||||||
|
|
||||||
|
// Update OHLC only if accepted and priceable.
|
||||||
|
if judge.accepted {
|
||||||
|
if let Some(price) = judge.price {
|
||||||
|
if bucket.first_accepted_price.is_none() {
|
||||||
|
bucket.first_accepted_price = Some(price);
|
||||||
|
}
|
||||||
|
bucket.last_accepted_price = Some(price);
|
||||||
|
bucket.high = bucket.high.max(price);
|
||||||
|
bucket.low = bucket.low.min(price);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
policy.apply(&leg, judge.accepted);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Phase 3: insert all computed buckets in one transaction.
|
||||||
|
let mut tx = write_pool.begin().await?;
|
||||||
|
let mut inserted = 0u64;
|
||||||
|
|
||||||
|
for ((token_id, bucket_ts), bucket) in buckets {
|
||||||
|
let open = bucket.first_accepted_price.unwrap_or(bucket.last_accepted_price.unwrap_or(0.0));
|
||||||
|
let close = bucket.last_accepted_price.unwrap_or(bucket.first_accepted_price.unwrap_or(0.0));
|
||||||
|
|
||||||
inserted += sqlx::query(
|
inserted += sqlx::query(
|
||||||
"INSERT OR IGNORE INTO ohlcv_1h
|
"INSERT OR IGNORE INTO ohlcv_1h
|
||||||
(token_id, bucket_ts, open, high, low, close, volume_sats, volume_tokens, tx_count)
|
(token_id, bucket_ts, open, high, low, close, volume_sats, volume_tokens, tx_count)
|
||||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
||||||
)
|
)
|
||||||
.bind(token_id)
|
.bind(&token_id)
|
||||||
.bind(bucket_ts)
|
.bind(bucket_ts)
|
||||||
.bind(open)
|
.bind(open)
|
||||||
.bind(high)
|
.bind(bucket.high)
|
||||||
.bind(low)
|
.bind(bucket.low)
|
||||||
.bind(close)
|
.bind(close)
|
||||||
.bind(volume_sats)
|
.bind(bucket.volume_sats)
|
||||||
.bind(volume_tokens)
|
.bind(bucket.volume_tokens)
|
||||||
.bind(tx_count)
|
.bind(bucket.txids.len() as i64)
|
||||||
.execute(&mut *tx)
|
.execute(&mut *tx)
|
||||||
.await?
|
.await?
|
||||||
.rows_affected();
|
.rows_affected();
|
||||||
|
|
@ -233,6 +236,31 @@ JOIN vol ON ohlc.token_id = vol.token_id AND ohlc.bucket_ts = vol.bucket_ts
|
||||||
Ok(inserted)
|
Ok(inserted)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Temporary bucket structure for accumulating OHLCV data per (token, bucket_ts).
|
||||||
|
struct OhlcvBucket {
|
||||||
|
first_accepted_price: Option<f64>,
|
||||||
|
last_accepted_price: Option<f64>,
|
||||||
|
high: f64,
|
||||||
|
low: f64,
|
||||||
|
volume_sats: i64,
|
||||||
|
volume_tokens: i64,
|
||||||
|
txids: std::collections::HashSet<[u8; 32]>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl OhlcvBucket {
|
||||||
|
fn new() -> Self {
|
||||||
|
Self {
|
||||||
|
first_accepted_price: None,
|
||||||
|
last_accepted_price: None,
|
||||||
|
high: f64::MIN,
|
||||||
|
low: f64::MAX,
|
||||||
|
volume_sats: 0,
|
||||||
|
volume_tokens: 0,
|
||||||
|
txids: std::collections::HashSet::new(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
pub struct OhlcvRow {
|
pub struct OhlcvRow {
|
||||||
pub bucket_ts: i64,
|
pub bucket_ts: i64,
|
||||||
pub open: f64,
|
pub open: f64,
|
||||||
|
|
@ -511,16 +539,32 @@ mod tests {
|
||||||
.await
|
.await
|
||||||
.unwrap();
|
.unwrap();
|
||||||
|
|
||||||
let close: f64 = sqlx::query_scalar("SELECT close FROM ohlcv_1h WHERE token_id = ?")
|
let (close, open, high, low): (f64, f64, f64, f64) = sqlx::query_as(
|
||||||
.bind(token.as_slice())
|
"SELECT close, open, high, low FROM ohlcv_1h WHERE token_id = ?",
|
||||||
.fetch_one(&pool)
|
)
|
||||||
.await
|
.bind(token.as_slice())
|
||||||
.unwrap();
|
.fetch_one(&pool)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
// Per-leg pricing: each leg executes at its own price.
|
||||||
|
// Leg 1 (buy): 446,491,239 / 1,334,527,069 = 0.334569
|
||||||
|
// Leg 2 (sell): 384,906,040 / 1,334,527,067 = 0.288421
|
||||||
|
// Close is the last leg, open is the first, high/low are the extremes.
|
||||||
|
let leg1_price = 446_491_239.0 / 1_334_527_069.0;
|
||||||
|
let leg2_price = 384_906_040.0 / 1_334_527_067.0;
|
||||||
|
|
||||||
let expected = 831_397_279.0 / 2_669_054_136.0;
|
|
||||||
assert!(
|
assert!(
|
||||||
(close - expected).abs() < 1e-9,
|
(close - leg2_price).abs() < 1e-9,
|
||||||
"materialised close {close} should be the gross ratio {expected}"
|
"materialised close {close} should be leg2 price {leg2_price}"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
(open - leg1_price).abs() < 1e-9,
|
||||||
|
"materialised open {open} should be leg1 price {leg1_price}"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
(high - leg1_price).abs() < 1e-9 && (low - leg2_price).abs() < 1e-9,
|
||||||
|
"high {high} and low {low} should span the leg prices"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue