push repo
This commit is contained in:
commit
99a1db2fb2
11 changed files with 3379 additions and 0 deletions
0
README.md
Normal file
0
README.md
Normal file
392
dags/anomaly_detection_dag.py
Normal file
392
dags/anomaly_detection_dag.py
Normal file
|
|
@ -0,0 +1,392 @@
|
||||||
|
from airflow import DAG
|
||||||
|
from airflow.decorators import task
|
||||||
|
from airflow.providers.amazon.aws.hooks.s3 import S3Hook
|
||||||
|
from airflow.providers.postgres.hooks.postgres import PostgresHook
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
import pandas as pd
|
||||||
|
import requests
|
||||||
|
import os
|
||||||
|
import json
|
||||||
|
import uuid
|
||||||
|
import logging
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
# --- Configuration & Constants ---
|
||||||
|
MINIO_BUCKET = "fraud-features"
|
||||||
|
AWS_CONN_ID = "minio_conn"
|
||||||
|
POSTGRES_CONN_ID = "postgres_conn"
|
||||||
|
BATCH_INTERVAL_HOURS = int(os.environ.get('BATCH_INTERVAL_HOURS', 6))
|
||||||
|
default_args = {
|
||||||
|
'owner': 'data_engineering',
|
||||||
|
'depends_on_past': False,
|
||||||
|
'retries': 1,
|
||||||
|
'retry_delay': timedelta(minutes=5),
|
||||||
|
}
|
||||||
|
|
||||||
|
with DAG(
|
||||||
|
dag_id='session_anomaly_detection_pipeline',
|
||||||
|
default_args=default_args,
|
||||||
|
description='Batch fraud detection pipeline using PostHog, MinIO, MLFlow, and Postgres',
|
||||||
|
schedule_interval=timedelta(hours=BATCH_INTERVAL_HOURS),
|
||||||
|
start_date=datetime(2026, 6, 20, tzinfo=timezone.utc),
|
||||||
|
catchup=False,
|
||||||
|
tags=['fraud', 'posthog', 'mlflow'],
|
||||||
|
) as dag:
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
@task
|
||||||
|
def extract_features_to_minio(**kwargs) -> dict:
|
||||||
|
logger.info("=== Starting feature extraction task ===")
|
||||||
|
|
||||||
|
try:
|
||||||
|
run_id = kwargs["run_id"]
|
||||||
|
logger.info("Run ID: %s", run_id)
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# 1. Determine extraction window
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
logger.info("Connecting to PostgreSQL...")
|
||||||
|
pg_hook = PostgresHook(postgres_conn_id=POSTGRES_CONN_ID)
|
||||||
|
|
||||||
|
logger.info("Fetching previous successful batch...")
|
||||||
|
last_run = pg_hook.get_first(
|
||||||
|
"SELECT MAX(window_end) FROM batch_runs WHERE status = 'SUCCESS'"
|
||||||
|
)
|
||||||
|
|
||||||
|
last_window_end = last_run[0] if last_run and last_run[0] else None
|
||||||
|
logger.info("Last successful window_end: %s", last_window_end)
|
||||||
|
|
||||||
|
interval_hours = BATCH_INTERVAL_HOURS
|
||||||
|
logger.info("Batch interval: %s hours", interval_hours)
|
||||||
|
|
||||||
|
if last_window_end:
|
||||||
|
start_date = last_window_end
|
||||||
|
end_date = start_date + timedelta(hours=interval_hours)
|
||||||
|
else:
|
||||||
|
end_date = datetime.now(timezone.utc)
|
||||||
|
start_date = end_date - timedelta(hours=interval_hours)
|
||||||
|
|
||||||
|
start_str = start_date.strftime("%Y-%m-%d %H:%M:%S")
|
||||||
|
end_str = end_date.strftime("%Y-%m-%d %H:%M:%S")
|
||||||
|
|
||||||
|
logger.info("Extraction window: %s -> %s", start_str, end_str)
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# 2. Build HogQL query
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
dag_dir = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
sql_file_path = os.path.join(dag_dir, "sql", "session_features.hql")
|
||||||
|
|
||||||
|
logger.info("Reading HogQL template from %s", sql_file_path)
|
||||||
|
|
||||||
|
with open(sql_file_path, "r") as f:
|
||||||
|
template = f.read()
|
||||||
|
|
||||||
|
query = template.format(
|
||||||
|
start_date=start_str,
|
||||||
|
end_date=end_str,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info("Generated HogQL query:")
|
||||||
|
logger.info("\n%s", query)
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# 3. Query PostHog
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
POSTHOG_HOST = os.environ.get("POSTHOG_HOST")
|
||||||
|
POSTHOG_PROJECT_ID = os.environ.get("POSTHOG_PROJECT_ID", "1")
|
||||||
|
POSTHOG_API_KEY = os.environ.get("POSTHOG_API_KEY")
|
||||||
|
logger.info("POSTHOG_API_KEY: %s", POSTHOG_API_KEY)
|
||||||
|
logger.info("Sending request to PostHog...")
|
||||||
|
logger.info("Host: %s", POSTHOG_HOST)
|
||||||
|
logger.info("Project ID: %s", POSTHOG_PROJECT_ID)
|
||||||
|
|
||||||
|
response = requests.post(
|
||||||
|
f"{POSTHOG_HOST}/api/projects/{POSTHOG_PROJECT_ID}/query/",
|
||||||
|
headers={
|
||||||
|
"Authorization": f"Bearer {POSTHOG_API_KEY}",
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
},
|
||||||
|
json={
|
||||||
|
"query": {
|
||||||
|
"kind": "HogQLQuery",
|
||||||
|
"query": query,
|
||||||
|
}
|
||||||
|
},
|
||||||
|
timeout=300,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info("PostHog status code: %s", response.status_code)
|
||||||
|
|
||||||
|
if not response.ok:
|
||||||
|
logger.error("PostHog response:\n%s", response.text)
|
||||||
|
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
data = response.json()
|
||||||
|
logger.info("Successfully parsed JSON response.")
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# 4. Convert to DataFrame
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
columns = data.get("columns", [])
|
||||||
|
results = data.get("results", [])
|
||||||
|
|
||||||
|
logger.info("Columns: %s", columns)
|
||||||
|
logger.info("Rows returned: %d", len(results))
|
||||||
|
|
||||||
|
df = pd.DataFrame(results, columns=columns)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"DataFrame shape: %s x %s",
|
||||||
|
df.shape[0],
|
||||||
|
df.shape[1],
|
||||||
|
)
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# 5. Save parquet
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
local_path = f"/tmp/features_{run_id}.parquet"
|
||||||
|
|
||||||
|
logger.info("Writing parquet to %s", local_path)
|
||||||
|
df.to_parquet(local_path, index=False)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"Parquet size: %.2f MB",
|
||||||
|
os.path.getsize(local_path) / (1024 * 1024),
|
||||||
|
)
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# 6. Upload to MinIO
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
logger.info("Connecting to MinIO...")
|
||||||
|
|
||||||
|
s3_hook = S3Hook(aws_conn_id=AWS_CONN_ID)
|
||||||
|
s3_key = f"batches/{run_id}/session_features.parquet"
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
"Uploading to bucket=%s key=%s",
|
||||||
|
MINIO_BUCKET,
|
||||||
|
s3_key,
|
||||||
|
)
|
||||||
|
|
||||||
|
s3_hook.load_file(
|
||||||
|
filename=local_path,
|
||||||
|
key=s3_key,
|
||||||
|
bucket_name=MINIO_BUCKET,
|
||||||
|
replace=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.info("Upload complete.")
|
||||||
|
|
||||||
|
os.remove(local_path)
|
||||||
|
logger.info("Temporary parquet deleted.")
|
||||||
|
|
||||||
|
logger.info("=== Feature extraction completed successfully ===")
|
||||||
|
|
||||||
|
return {
|
||||||
|
"s3_uri": f"s3://{MINIO_BUCKET}/{s3_key}",
|
||||||
|
"window_start": start_date.isoformat(),
|
||||||
|
"window_end": end_date.isoformat(),
|
||||||
|
"record_count": len(df),
|
||||||
|
}
|
||||||
|
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Feature extraction task failed!")
|
||||||
|
raise
|
||||||
|
|
||||||
|
@task
|
||||||
|
def call_mlflow_inference(extraction_result: dict, **kwargs) -> str:
|
||||||
|
"""
|
||||||
|
Task 2: Reads features from MinIO, sends payload to local MLflow endpoint,
|
||||||
|
saves predictions (including SHAP values) back to MinIO.
|
||||||
|
"""
|
||||||
|
logger.info("=== Starting MLflow inference task ===")
|
||||||
|
run_id = kwargs['run_id']
|
||||||
|
|
||||||
|
features_s3_uri = extraction_result.get("s3_uri") if isinstance(extraction_result, dict) else extraction_result
|
||||||
|
|
||||||
|
s3_hook = S3Hook(aws_conn_id=AWS_CONN_ID)
|
||||||
|
|
||||||
|
bucket = features_s3_uri.split("/")[2]
|
||||||
|
key = "/".join(features_s3_uri.split("/")[3:])
|
||||||
|
|
||||||
|
logger.info("Downloading features from %s", features_s3_uri)
|
||||||
|
local_features_path = s3_hook.download_file(key=key, bucket_name=bucket, local_path="/tmp")
|
||||||
|
df_features = pd.read_parquet(local_features_path)
|
||||||
|
|
||||||
|
if df_features.empty:
|
||||||
|
logger.info("DataFrame is empty. Skipping inference.")
|
||||||
|
os.remove(local_features_path)
|
||||||
|
return features_s3_uri
|
||||||
|
|
||||||
|
logger.info("Loaded %d rows for inference.", len(df_features))
|
||||||
|
|
||||||
|
payload = {"dataframe_split": df_features.to_dict(orient="split")}
|
||||||
|
|
||||||
|
mlflow_url = os.environ.get("MLFLOW_API_URL", "http://host.docker.internal:5001/invocations")
|
||||||
|
|
||||||
|
response = requests.post(
|
||||||
|
mlflow_url,
|
||||||
|
data=json.dumps(payload),
|
||||||
|
headers={"Content-Type": "application/json"},
|
||||||
|
timeout=120
|
||||||
|
)
|
||||||
|
|
||||||
|
if not response.ok:
|
||||||
|
logger.error("MLflow API error: %s - %s", response.status_code, response.text)
|
||||||
|
response.raise_for_status()
|
||||||
|
|
||||||
|
predictions_data = response.json().get("predictions", [])
|
||||||
|
|
||||||
|
# Extract model outputs into the dataframe
|
||||||
|
df_features['prediction'] = [p.get('prediction') for p in predictions_data]
|
||||||
|
df_features['anomaly_score'] = [p.get('anomaly_score') for p in predictions_data]
|
||||||
|
df_features['decision_score'] = [p.get('decision_score') for p in predictions_data]
|
||||||
|
df_features['is_anomaly'] = df_features['prediction'] == -1
|
||||||
|
|
||||||
|
# Extract SHAP values (returned as stringified JSON by your model)
|
||||||
|
df_features['shap_values'] = [p.get('shap_values', '{}') for p in predictions_data]
|
||||||
|
|
||||||
|
local_preds_path = f"/tmp/predictions_{run_id}.parquet"
|
||||||
|
df_features.to_parquet(local_preds_path, index=False)
|
||||||
|
|
||||||
|
preds_s3_key = f"batches/{run_id}/predictions.parquet"
|
||||||
|
|
||||||
|
s3_hook.load_file(
|
||||||
|
filename=local_preds_path,
|
||||||
|
key=preds_s3_key,
|
||||||
|
bucket_name=MINIO_BUCKET,
|
||||||
|
replace=True
|
||||||
|
)
|
||||||
|
|
||||||
|
os.remove(local_features_path)
|
||||||
|
os.remove(local_preds_path)
|
||||||
|
|
||||||
|
return f"s3://{MINIO_BUCKET}/{preds_s3_key}"
|
||||||
|
|
||||||
|
@task
|
||||||
|
def load_predictions_to_postgres(predictions_s3_uri: str, extraction_result: dict, **kwargs):
|
||||||
|
logger.info("=== Loading predictions into PostgreSQL ===")
|
||||||
|
run_id = kwargs["run_id"]
|
||||||
|
|
||||||
|
s3_hook = S3Hook(aws_conn_id=AWS_CONN_ID)
|
||||||
|
pg_hook = PostgresHook(postgres_conn_id=POSTGRES_CONN_ID)
|
||||||
|
|
||||||
|
bucket = predictions_s3_uri.split("/")[2]
|
||||||
|
key = "/".join(predictions_s3_uri.split("/")[3:])
|
||||||
|
|
||||||
|
local_path = s3_hook.download_file(
|
||||||
|
key=key,
|
||||||
|
bucket_name=bucket,
|
||||||
|
local_path="/tmp",
|
||||||
|
)
|
||||||
|
|
||||||
|
df = pd.read_parquet(local_path)
|
||||||
|
logger.info("Loaded %d prediction rows", len(df))
|
||||||
|
|
||||||
|
window_start = extraction_result["window_start"]
|
||||||
|
window_end = extraction_result["window_end"]
|
||||||
|
|
||||||
|
total_sessions = len(df)
|
||||||
|
total_anomalies = int(df["is_anomaly"].sum()) if total_sessions else 0
|
||||||
|
contamination_rate = total_anomalies / total_sessions if total_sessions else 0.0
|
||||||
|
mean_score = float(df["anomaly_score"].mean()) if total_sessions else None
|
||||||
|
max_score = float(df["anomaly_score"].max()) if total_sessions else None
|
||||||
|
|
||||||
|
# ---------------------------
|
||||||
|
# ISOLATE FEATURE COLUMNS
|
||||||
|
# ---------------------------
|
||||||
|
# Define which columns are NOT part of the JSONB feature payload
|
||||||
|
metadata_cols = {
|
||||||
|
'session_id', 'user_id', 'prediction',
|
||||||
|
'anomaly_score', 'decision_score', 'is_anomaly', 'shap_values'
|
||||||
|
}
|
||||||
|
|
||||||
|
# Everything else is a feature
|
||||||
|
feature_cols = [col for col in df.columns if col not in metadata_cols]
|
||||||
|
|
||||||
|
from psycopg2.extras import Json, execute_values
|
||||||
|
|
||||||
|
records = []
|
||||||
|
for _, row in df.iterrows():
|
||||||
|
# 1. Build a dict of features for this specific row (dropping nulls safely)
|
||||||
|
row_features = {
|
||||||
|
col: row[col]
|
||||||
|
for col in feature_cols
|
||||||
|
if pd.notna(row[col])
|
||||||
|
}
|
||||||
|
|
||||||
|
# 2. Parse the SHAP values back into a dict (since MLflow returned stringified JSON)
|
||||||
|
shap_raw = row.get("shap_values")
|
||||||
|
shap_dict = json.loads(shap_raw) if isinstance(shap_raw, str) else (shap_raw or {})
|
||||||
|
|
||||||
|
records.append((
|
||||||
|
None, # batch_id placeholder
|
||||||
|
row["session_id"],
|
||||||
|
row.get("user_id"),
|
||||||
|
window_start,
|
||||||
|
row["anomaly_score"],
|
||||||
|
bool(row["is_anomaly"]),
|
||||||
|
Json(row_features), # Automatically adapts dict to JSONB
|
||||||
|
Json(shap_dict) # Automatically adapts dict to JSONB
|
||||||
|
))
|
||||||
|
|
||||||
|
insert_batch_sql = """
|
||||||
|
INSERT INTO batch_runs (
|
||||||
|
dag_run_id, window_start, window_end, mlflow_model_version, status
|
||||||
|
) VALUES (%s,%s,%s,%s,'SUCCESS')
|
||||||
|
RETURNING batch_id;
|
||||||
|
"""
|
||||||
|
|
||||||
|
prediction_sql = """
|
||||||
|
INSERT INTO session_predictions (
|
||||||
|
batch_id, session_id, user_id, session_start_time,
|
||||||
|
anomaly_score, is_anomaly, session_features, shap_values
|
||||||
|
) VALUES %s;
|
||||||
|
"""
|
||||||
|
|
||||||
|
update_batch_sql = """
|
||||||
|
UPDATE batch_runs
|
||||||
|
SET total_sessions_processed=%s, total_anomalies_detected=%s,
|
||||||
|
contamination_rate=%s, mean_anomaly_score=%s, max_anomaly_score=%s
|
||||||
|
WHERE batch_id=%s;
|
||||||
|
"""
|
||||||
|
|
||||||
|
conn = pg_hook.get_conn()
|
||||||
|
try:
|
||||||
|
with conn:
|
||||||
|
with conn.cursor() as cur:
|
||||||
|
cur.execute(
|
||||||
|
insert_batch_sql,
|
||||||
|
(run_id, window_start, window_end, "v1.0.0"),
|
||||||
|
)
|
||||||
|
batch_id = cur.fetchone()[0]
|
||||||
|
|
||||||
|
# Replace the 'None' placeholder with the actual batch_id
|
||||||
|
records = [(batch_id, *r[1:]) for r in records]
|
||||||
|
|
||||||
|
execute_values(cur, prediction_sql, records)
|
||||||
|
|
||||||
|
cur.execute(
|
||||||
|
update_batch_sql,
|
||||||
|
(total_sessions, total_anomalies, contamination_rate,
|
||||||
|
mean_score, max_score, batch_id),
|
||||||
|
)
|
||||||
|
logger.info("Inserted %d predictions for batch %s", total_sessions, batch_id)
|
||||||
|
|
||||||
|
except Exception:
|
||||||
|
conn.rollback()
|
||||||
|
with conn:
|
||||||
|
with conn.cursor() as cur:
|
||||||
|
cur.execute("UPDATE batch_runs SET status='FAILED' WHERE dag_run_id=%s;", (run_id,))
|
||||||
|
raise
|
||||||
|
finally:
|
||||||
|
conn.close()
|
||||||
|
os.remove(local_path)
|
||||||
|
|
||||||
|
# --- Pipeline Orchestration ---
|
||||||
|
features_uri = extract_features_to_minio()
|
||||||
|
predictions_uri = call_mlflow_inference(features_uri)
|
||||||
|
load_predictions_to_postgres(predictions_uri, features_uri)
|
||||||
83
dags/sql/init.sql
Normal file
83
dags/sql/init.sql
Normal file
|
|
@ -0,0 +1,83 @@
|
||||||
|
-- Ensure clean setup if restarting from scratch
|
||||||
|
DROP TABLE IF EXISTS session_predictions CASCADE;
|
||||||
|
DROP TABLE IF EXISTS batch_runs CASCADE;
|
||||||
|
|
||||||
|
-- ============================================================
|
||||||
|
-- BATCH RUN METADATA + AGGREGATED STATISTICS
|
||||||
|
-- ============================================================
|
||||||
|
CREATE TABLE batch_runs (
|
||||||
|
batch_id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
|
||||||
|
-- Airflow metadata
|
||||||
|
dag_run_id VARCHAR(255) NOT NULL UNIQUE,
|
||||||
|
|
||||||
|
-- Time window processed by this batch
|
||||||
|
window_start TIMESTAMP WITH TIME ZONE NOT NULL,
|
||||||
|
window_end TIMESTAMP WITH TIME ZONE NOT NULL,
|
||||||
|
|
||||||
|
-- Model lineage
|
||||||
|
mlflow_model_version VARCHAR(50) NOT NULL,
|
||||||
|
|
||||||
|
-- Execution status
|
||||||
|
status VARCHAR(20) NOT NULL
|
||||||
|
CHECK (status IN ('RUNNING', 'SUCCESS', 'FAILED')),
|
||||||
|
|
||||||
|
-- Batch statistics
|
||||||
|
total_sessions_processed INTEGER DEFAULT 0,
|
||||||
|
total_anomalies_detected INTEGER DEFAULT 0,
|
||||||
|
contamination_rate DOUBLE PRECISION,
|
||||||
|
mean_anomaly_score DOUBLE PRECISION,
|
||||||
|
max_anomaly_score DOUBLE PRECISION,
|
||||||
|
|
||||||
|
-- Dataset drift monitoring
|
||||||
|
feature_means_summary JSONB,
|
||||||
|
|
||||||
|
created_at TIMESTAMP WITH TIME ZONE DEFAULT CURRENT_TIMESTAMP,
|
||||||
|
|
||||||
|
CONSTRAINT unique_window UNIQUE (window_start, window_end)
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE INDEX idx_batch_window
|
||||||
|
ON batch_runs(window_start, window_end);
|
||||||
|
|
||||||
|
CREATE INDEX idx_batch_created_at
|
||||||
|
ON batch_runs(created_at);
|
||||||
|
|
||||||
|
-- ============================================================
|
||||||
|
-- SESSION-LEVEL PREDICTIONS
|
||||||
|
-- ============================================================
|
||||||
|
CREATE TABLE session_predictions (
|
||||||
|
prediction_id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
|
||||||
|
batch_id UUID NOT NULL
|
||||||
|
REFERENCES batch_runs(batch_id)
|
||||||
|
ON DELETE CASCADE,
|
||||||
|
|
||||||
|
session_id VARCHAR(255) NOT NULL,
|
||||||
|
user_id VARCHAR(255),
|
||||||
|
|
||||||
|
session_start_time TIMESTAMP WITH TIME ZONE NOT NULL,
|
||||||
|
|
||||||
|
-- Model outputs
|
||||||
|
anomaly_score DOUBLE PRECISION NOT NULL,
|
||||||
|
is_anomaly BOOLEAN NOT NULL,
|
||||||
|
|
||||||
|
-- Dynamic payloads
|
||||||
|
session_features JSONB NOT NULL,
|
||||||
|
shap_values JSONB NOT NULL,
|
||||||
|
|
||||||
|
created_at TIMESTAMP WITH TIME ZONE DEFAULT CURRENT_TIMESTAMP,
|
||||||
|
|
||||||
|
CONSTRAINT unique_session_per_batch
|
||||||
|
UNIQUE(batch_id, session_id)
|
||||||
|
);
|
||||||
|
|
||||||
|
-- Frequently queried anomaly rows
|
||||||
|
CREATE INDEX idx_session_predictions_anomalies
|
||||||
|
ON session_predictions(batch_id)
|
||||||
|
WHERE is_anomaly = TRUE;
|
||||||
|
|
||||||
|
-- Optional searches inside feature JSON
|
||||||
|
CREATE INDEX idx_session_features_gin
|
||||||
|
ON session_predictions
|
||||||
|
USING GIN(session_features);
|
||||||
234
dags/sql/session_features.hql
Normal file
234
dags/sql/session_features.hql
Normal file
|
|
@ -0,0 +1,234 @@
|
||||||
|
WITH base AS (
|
||||||
|
SELECT
|
||||||
|
person_id,
|
||||||
|
timestamp,
|
||||||
|
event,
|
||||||
|
properties,
|
||||||
|
row_number() OVER (
|
||||||
|
PARTITION BY person_id
|
||||||
|
ORDER BY timestamp, event
|
||||||
|
) AS global_order
|
||||||
|
FROM events
|
||||||
|
WHERE timestamp >= toDateTime('{start_date}')
|
||||||
|
AND timestamp < toDateTime('{end_date}')
|
||||||
|
AND event NOT IN ('flutter_error', 'platform_error')
|
||||||
|
),
|
||||||
|
|
||||||
|
ordered_events AS (
|
||||||
|
SELECT
|
||||||
|
*,
|
||||||
|
lagInFrame(timestamp) OVER (
|
||||||
|
PARTITION BY person_id
|
||||||
|
ORDER BY global_order
|
||||||
|
) AS prev_ts,
|
||||||
|
|
||||||
|
lagInFrame(event) OVER (
|
||||||
|
PARTITION BY person_id
|
||||||
|
ORDER BY global_order
|
||||||
|
) AS prev_event
|
||||||
|
FROM base
|
||||||
|
),
|
||||||
|
|
||||||
|
session_flags AS (
|
||||||
|
SELECT
|
||||||
|
*,
|
||||||
|
CASE
|
||||||
|
WHEN prev_ts IS NULL THEN 1
|
||||||
|
WHEN event = 'Application Opened' THEN 1
|
||||||
|
WHEN prev_event IN ('Application Backgrounded', 'Application Closed') THEN 1
|
||||||
|
WHEN dateDiff('minute', prev_ts, timestamp) > 10 THEN 1
|
||||||
|
ELSE 0
|
||||||
|
END AS is_new_session
|
||||||
|
FROM ordered_events
|
||||||
|
),
|
||||||
|
|
||||||
|
sessionized AS (
|
||||||
|
SELECT
|
||||||
|
*,
|
||||||
|
sum(is_new_session) OVER (
|
||||||
|
PARTITION BY person_id
|
||||||
|
ORDER BY global_order
|
||||||
|
) AS session_number
|
||||||
|
FROM session_flags
|
||||||
|
),
|
||||||
|
|
||||||
|
session_duration_calc AS (
|
||||||
|
SELECT
|
||||||
|
*,
|
||||||
|
dateDiff(
|
||||||
|
'second',
|
||||||
|
MIN(timestamp) OVER (
|
||||||
|
PARTITION BY person_id, session_number
|
||||||
|
),
|
||||||
|
timestamp
|
||||||
|
) AS seconds_since_session_start
|
||||||
|
FROM sessionized
|
||||||
|
),
|
||||||
|
|
||||||
|
split_sessions AS (
|
||||||
|
SELECT
|
||||||
|
*,
|
||||||
|
floor(seconds_since_session_start / 1800) AS session_sub_id
|
||||||
|
FROM session_duration_calc
|
||||||
|
),
|
||||||
|
|
||||||
|
session_events AS (
|
||||||
|
SELECT
|
||||||
|
*,
|
||||||
|
row_number() OVER (
|
||||||
|
PARTITION BY person_id, session_number, session_sub_id
|
||||||
|
ORDER BY timestamp, event
|
||||||
|
) AS session_index
|
||||||
|
FROM split_sessions
|
||||||
|
),
|
||||||
|
|
||||||
|
inter_event_calc AS (
|
||||||
|
SELECT
|
||||||
|
person_id,
|
||||||
|
session_number,
|
||||||
|
session_sub_id,
|
||||||
|
timestamp,
|
||||||
|
event,
|
||||||
|
properties,
|
||||||
|
session_index,
|
||||||
|
|
||||||
|
lagInFrame(timestamp) OVER (
|
||||||
|
PARTITION BY person_id, session_number, session_sub_id
|
||||||
|
ORDER BY session_index
|
||||||
|
) AS prev_ts_in_session,
|
||||||
|
|
||||||
|
CASE
|
||||||
|
WHEN session_index = 1 THEN NULL
|
||||||
|
ELSE greatest(
|
||||||
|
0,
|
||||||
|
dateDiff(
|
||||||
|
'second',
|
||||||
|
lagInFrame(timestamp) OVER (
|
||||||
|
PARTITION BY person_id, session_number, session_sub_id
|
||||||
|
ORDER BY session_index
|
||||||
|
),
|
||||||
|
timestamp
|
||||||
|
)
|
||||||
|
)
|
||||||
|
END AS inter_event_time_seconds
|
||||||
|
FROM session_events
|
||||||
|
),
|
||||||
|
|
||||||
|
session_stats AS (
|
||||||
|
SELECT
|
||||||
|
person_id,
|
||||||
|
session_number,
|
||||||
|
session_sub_id,
|
||||||
|
|
||||||
|
COUNT(*) AS event_count,
|
||||||
|
|
||||||
|
uniqExact(event) AS unique_event_types,
|
||||||
|
|
||||||
|
uniqExact(
|
||||||
|
if(
|
||||||
|
event = '$screen',
|
||||||
|
replaceRegexpOne(
|
||||||
|
JSONExtractString(properties, '$screen_name'),
|
||||||
|
'\\?.*$',
|
||||||
|
''
|
||||||
|
),
|
||||||
|
NULL
|
||||||
|
)
|
||||||
|
) AS screens_visited,
|
||||||
|
|
||||||
|
MIN(timestamp) AS session_start,
|
||||||
|
MAX(timestamp) AS session_end,
|
||||||
|
|
||||||
|
dateDiff(
|
||||||
|
'second',
|
||||||
|
MIN(timestamp),
|
||||||
|
MAX(timestamp)
|
||||||
|
) AS duration_seconds,
|
||||||
|
|
||||||
|
COUNT(*) * 60.0 /
|
||||||
|
greatest(
|
||||||
|
dateDiff(
|
||||||
|
'second',
|
||||||
|
MIN(timestamp),
|
||||||
|
MAX(timestamp)
|
||||||
|
),
|
||||||
|
1
|
||||||
|
) AS events_per_minute,
|
||||||
|
|
||||||
|
AVG(inter_event_time_seconds) AS inter_event_time_mean_seconds,
|
||||||
|
|
||||||
|
median(inter_event_time_seconds) AS inter_event_time_median_seconds,
|
||||||
|
|
||||||
|
max(inter_event_time_seconds) AS max_inter_event_gap_seconds,
|
||||||
|
|
||||||
|
sqrt(
|
||||||
|
varSamp(inter_event_time_seconds)
|
||||||
|
) AS inter_event_time_std_seconds,
|
||||||
|
|
||||||
|
toHour(MIN(timestamp)) AS session_start_hour,
|
||||||
|
|
||||||
|
toDayOfWeek(MIN(timestamp)) AS session_day_of_week,
|
||||||
|
|
||||||
|
uniqExact(
|
||||||
|
toStartOfHour(timestamp)
|
||||||
|
) AS distinct_hours_active
|
||||||
|
|
||||||
|
FROM inter_event_calc
|
||||||
|
GROUP BY
|
||||||
|
person_id,
|
||||||
|
session_number,
|
||||||
|
session_sub_id
|
||||||
|
|
||||||
|
HAVING
|
||||||
|
COUNT(*) > 2
|
||||||
|
AND dateDiff(
|
||||||
|
'second',
|
||||||
|
MIN(timestamp),
|
||||||
|
MAX(timestamp)
|
||||||
|
) >= 4
|
||||||
|
)
|
||||||
|
|
||||||
|
SELECT
|
||||||
|
person_id,
|
||||||
|
|
||||||
|
concat(
|
||||||
|
toString(person_id),
|
||||||
|
'_',
|
||||||
|
toString(session_number),
|
||||||
|
'_',
|
||||||
|
toString(session_sub_id)
|
||||||
|
) AS session_id,
|
||||||
|
|
||||||
|
session_start,
|
||||||
|
-- 24-Hour Cyclical Encoding for Session Start
|
||||||
|
sin(2 * pi() * (toUnixTimestamp(session_start) - toUnixTimestamp(toStartOfDay(session_start))) / 86400) AS session_start_sin,
|
||||||
|
cos(2 * pi() * (toUnixTimestamp(session_start) - toUnixTimestamp(toStartOfDay(session_start))) / 86400) AS session_start_cos,
|
||||||
|
|
||||||
|
session_end,
|
||||||
|
-- 24-Hour Cyclical Encoding for Session End
|
||||||
|
sin(2 * pi() * (toUnixTimestamp(session_end) - toUnixTimestamp(toStartOfDay(session_end))) / 86400) AS session_end_sin,
|
||||||
|
cos(2 * pi() * (toUnixTimestamp(session_end) - toUnixTimestamp(toStartOfDay(session_end))) / 86400) AS session_end_cos,
|
||||||
|
|
||||||
|
event_count,
|
||||||
|
unique_event_types,
|
||||||
|
screens_visited,
|
||||||
|
|
||||||
|
duration_seconds,
|
||||||
|
events_per_minute,
|
||||||
|
|
||||||
|
inter_event_time_mean_seconds,
|
||||||
|
inter_event_time_median_seconds,
|
||||||
|
inter_event_time_std_seconds,
|
||||||
|
max_inter_event_gap_seconds,
|
||||||
|
|
||||||
|
session_start_hour,
|
||||||
|
|
||||||
|
session_day_of_week,
|
||||||
|
-- 7-Day Cyclical Encoding for Day of the Week
|
||||||
|
sin(2 * pi() * session_day_of_week / 7) AS session_day_of_week_sin,
|
||||||
|
cos(2 * pi() * session_day_of_week / 7) AS session_day_of_week_cos,
|
||||||
|
|
||||||
|
distinct_hours_active
|
||||||
|
|
||||||
|
FROM session_stats
|
||||||
|
ORDER BY session_start DESC
|
||||||
108
docker-compose.yaml
Normal file
108
docker-compose.yaml
Normal file
|
|
@ -0,0 +1,108 @@
|
||||||
|
x-airflow-common: &airflow-common
|
||||||
|
image: apache/airflow:2.7.2
|
||||||
|
extra_hosts:
|
||||||
|
- "host.docker.internal:host-gateway"
|
||||||
|
environment:
|
||||||
|
&airflow-common-env
|
||||||
|
AIRFLOW__CORE__EXECUTOR: LocalExecutor
|
||||||
|
AIRFLOW__DATABASE__SQL_ALCHEMY_CONN: postgresql+psycopg2://airflow:airflow_password@postgres:5432/fraud_db
|
||||||
|
AIRFLOW__CORE__FERNET_KEY: ''
|
||||||
|
AIRFLOW__CORE__LOAD_EXAMPLES: 'False'
|
||||||
|
_PIP_ADDITIONAL_REQUIREMENTS: 'apache-airflow-providers-amazon apache-airflow-providers-postgres pandas pyarrow requests'
|
||||||
|
AIRFLOW__WEBSERVER__SECRET_KEY: 'this_is_a_very_secure_secret_key'
|
||||||
|
AIRFLOW_CONN_MINIO_CONN: 'aws://minio_admin:minio_password@/?endpoint_url=http%3A%2F%2Fminio%3A9000'
|
||||||
|
env_file:
|
||||||
|
- .env
|
||||||
|
volumes:
|
||||||
|
- ./dags:/opt/airflow/dags
|
||||||
|
- airflow_logs:/opt/airflow/logs
|
||||||
|
user: "50000:0"
|
||||||
|
services:
|
||||||
|
postgres:
|
||||||
|
image: postgres:15
|
||||||
|
container_name: postgres_db
|
||||||
|
environment:
|
||||||
|
POSTGRES_USER: airflow
|
||||||
|
POSTGRES_PASSWORD: airflow_password
|
||||||
|
POSTGRES_DB: fraud_db
|
||||||
|
ports:
|
||||||
|
- "5432:5432"
|
||||||
|
volumes:
|
||||||
|
- ./dags/sql/init.sql:/docker-entrypoint-initdb.d/init.sql
|
||||||
|
- postgres_data:/var/lib/postgresql/data
|
||||||
|
healthcheck:
|
||||||
|
test: ["CMD-SHELL", "pg_isready -U airflow -d fraud_db"]
|
||||||
|
interval: 5s
|
||||||
|
timeout: 5s
|
||||||
|
retries: 5
|
||||||
|
|
||||||
|
minio:
|
||||||
|
image: quay.io/minio/minio
|
||||||
|
container_name: minio_storage
|
||||||
|
command: server /data --console-address ":9001"
|
||||||
|
environment:
|
||||||
|
MINIO_ROOT_USER: minio_admin
|
||||||
|
MINIO_ROOT_PASSWORD: minio_password
|
||||||
|
ports:
|
||||||
|
- "9000:9000"
|
||||||
|
- "9001:9001"
|
||||||
|
volumes:
|
||||||
|
- minio_data:/data
|
||||||
|
|
||||||
|
minio-setup:
|
||||||
|
image: quay.io/minio/mc
|
||||||
|
container_name: minio_setup
|
||||||
|
depends_on:
|
||||||
|
- minio
|
||||||
|
entrypoint: >
|
||||||
|
/bin/sh -c "
|
||||||
|
sleep 5;
|
||||||
|
mc alias set myminio http://minio:9000 minio_admin minio_password;
|
||||||
|
mc mb myminio/fraud-features --ignore-existing;
|
||||||
|
mc policy set public myminio/fraud-features;
|
||||||
|
exit 0;
|
||||||
|
"
|
||||||
|
|
||||||
|
airflow-init:
|
||||||
|
<<: *airflow-common
|
||||||
|
container_name: airflow_init
|
||||||
|
command: version
|
||||||
|
environment:
|
||||||
|
<<: *airflow-common-env
|
||||||
|
_AIRFLOW_DB_MIGRATE: 'true'
|
||||||
|
_AIRFLOW_WWW_USER_CREATE: 'true'
|
||||||
|
_AIRFLOW_WWW_USER_USERNAME: airflow
|
||||||
|
_AIRFLOW_WWW_USER_PASSWORD: airflow_password
|
||||||
|
|
||||||
|
airflow-webserver:
|
||||||
|
<<: *airflow-common
|
||||||
|
container_name: airflow_webserver
|
||||||
|
command: webserver
|
||||||
|
ports:
|
||||||
|
- "8080:8080"
|
||||||
|
healthcheck:
|
||||||
|
test: ["CMD", "curl", "--fail", "http://localhost:8080/health"]
|
||||||
|
interval: 10s
|
||||||
|
timeout: 10s
|
||||||
|
retries: 5
|
||||||
|
restart: always
|
||||||
|
depends_on:
|
||||||
|
airflow-init:
|
||||||
|
condition: service_completed_successfully
|
||||||
|
|
||||||
|
airflow-scheduler:
|
||||||
|
<<: *airflow-common
|
||||||
|
container_name: airflow_scheduler
|
||||||
|
command: scheduler
|
||||||
|
restart: always
|
||||||
|
depends_on:
|
||||||
|
airflow-init:
|
||||||
|
condition: service_completed_successfully
|
||||||
|
environment:
|
||||||
|
<<: *airflow-common-env # <-- Add this line to merge the common variables
|
||||||
|
AIRFLOW_CONN_POSTGRES_CONN: "postgresql://airflow:airflow_password@postgres_db:5432/fraud_db"
|
||||||
|
|
||||||
|
volumes:
|
||||||
|
postgres_data:
|
||||||
|
minio_data:
|
||||||
|
airflow_logs: # <-- Add this line
|
||||||
6
main.py
Normal file
6
main.py
Normal file
|
|
@ -0,0 +1,6 @@
|
||||||
|
def main():
|
||||||
|
print("Hello from fraud-workflow!")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
BIN
predictions.parquet
Normal file
BIN
predictions.parquet
Normal file
Binary file not shown.
12
pyproject.toml
Normal file
12
pyproject.toml
Normal file
|
|
@ -0,0 +1,12 @@
|
||||||
|
[project]
|
||||||
|
name = "fraud-workflow"
|
||||||
|
version = "0.1.0"
|
||||||
|
description = "Add your description here"
|
||||||
|
readme = "README.md"
|
||||||
|
requires-python = ">=3.13"
|
||||||
|
dependencies = [
|
||||||
|
"apache-airflow>=3.2.2",
|
||||||
|
"fastparquet>=2026.5.0",
|
||||||
|
"pandas>=3.0.3",
|
||||||
|
"psycopg[binary]>=3.3.4",
|
||||||
|
]
|
||||||
BIN
session_features.parquet
Normal file
BIN
session_features.parquet
Normal file
Binary file not shown.
3
test.py
Normal file
3
test.py
Normal file
|
|
@ -0,0 +1,3 @@
|
||||||
|
import pandas
|
||||||
|
df = pandas.read_parquet("./session_features.parquet")
|
||||||
|
print(df.head())
|
||||||
Loading…
Add table
Reference in a new issue