push repo
This commit is contained in:
commit
99a1db2fb2
11 changed files with 3379 additions and 0 deletions
0
README.md
Normal file
0
README.md
Normal file
392
dags/anomaly_detection_dag.py
Normal file
392
dags/anomaly_detection_dag.py
Normal file
|
|
@ -0,0 +1,392 @@
|
|||
from airflow import DAG
|
||||
from airflow.decorators import task
|
||||
from airflow.providers.amazon.aws.hooks.s3 import S3Hook
|
||||
from airflow.providers.postgres.hooks.postgres import PostgresHook
|
||||
from datetime import datetime, timedelta, timezone
|
||||
import pandas as pd
|
||||
import requests
|
||||
import os
|
||||
import json
|
||||
import uuid
|
||||
import logging
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
# --- Configuration & Constants ---
|
||||
MINIO_BUCKET = "fraud-features"
|
||||
AWS_CONN_ID = "minio_conn"
|
||||
POSTGRES_CONN_ID = "postgres_conn"
|
||||
BATCH_INTERVAL_HOURS = int(os.environ.get('BATCH_INTERVAL_HOURS', 6))
|
||||
default_args = {
|
||||
'owner': 'data_engineering',
|
||||
'depends_on_past': False,
|
||||
'retries': 1,
|
||||
'retry_delay': timedelta(minutes=5),
|
||||
}
|
||||
|
||||
with DAG(
|
||||
dag_id='session_anomaly_detection_pipeline',
|
||||
default_args=default_args,
|
||||
description='Batch fraud detection pipeline using PostHog, MinIO, MLFlow, and Postgres',
|
||||
schedule_interval=timedelta(hours=BATCH_INTERVAL_HOURS),
|
||||
start_date=datetime(2026, 6, 20, tzinfo=timezone.utc),
|
||||
catchup=False,
|
||||
tags=['fraud', 'posthog', 'mlflow'],
|
||||
) as dag:
|
||||
|
||||
|
||||
|
||||
@task
|
||||
def extract_features_to_minio(**kwargs) -> dict:
|
||||
logger.info("=== Starting feature extraction task ===")
|
||||
|
||||
try:
|
||||
run_id = kwargs["run_id"]
|
||||
logger.info("Run ID: %s", run_id)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# 1. Determine extraction window
|
||||
# ------------------------------------------------------------------
|
||||
logger.info("Connecting to PostgreSQL...")
|
||||
pg_hook = PostgresHook(postgres_conn_id=POSTGRES_CONN_ID)
|
||||
|
||||
logger.info("Fetching previous successful batch...")
|
||||
last_run = pg_hook.get_first(
|
||||
"SELECT MAX(window_end) FROM batch_runs WHERE status = 'SUCCESS'"
|
||||
)
|
||||
|
||||
last_window_end = last_run[0] if last_run and last_run[0] else None
|
||||
logger.info("Last successful window_end: %s", last_window_end)
|
||||
|
||||
interval_hours = BATCH_INTERVAL_HOURS
|
||||
logger.info("Batch interval: %s hours", interval_hours)
|
||||
|
||||
if last_window_end:
|
||||
start_date = last_window_end
|
||||
end_date = start_date + timedelta(hours=interval_hours)
|
||||
else:
|
||||
end_date = datetime.now(timezone.utc)
|
||||
start_date = end_date - timedelta(hours=interval_hours)
|
||||
|
||||
start_str = start_date.strftime("%Y-%m-%d %H:%M:%S")
|
||||
end_str = end_date.strftime("%Y-%m-%d %H:%M:%S")
|
||||
|
||||
logger.info("Extraction window: %s -> %s", start_str, end_str)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# 2. Build HogQL query
|
||||
# ------------------------------------------------------------------
|
||||
dag_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
sql_file_path = os.path.join(dag_dir, "sql", "session_features.hql")
|
||||
|
||||
logger.info("Reading HogQL template from %s", sql_file_path)
|
||||
|
||||
with open(sql_file_path, "r") as f:
|
||||
template = f.read()
|
||||
|
||||
query = template.format(
|
||||
start_date=start_str,
|
||||
end_date=end_str,
|
||||
)
|
||||
|
||||
logger.info("Generated HogQL query:")
|
||||
logger.info("\n%s", query)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# 3. Query PostHog
|
||||
# ------------------------------------------------------------------
|
||||
POSTHOG_HOST = os.environ.get("POSTHOG_HOST")
|
||||
POSTHOG_PROJECT_ID = os.environ.get("POSTHOG_PROJECT_ID", "1")
|
||||
POSTHOG_API_KEY = os.environ.get("POSTHOG_API_KEY")
|
||||
logger.info("POSTHOG_API_KEY: %s", POSTHOG_API_KEY)
|
||||
logger.info("Sending request to PostHog...")
|
||||
logger.info("Host: %s", POSTHOG_HOST)
|
||||
logger.info("Project ID: %s", POSTHOG_PROJECT_ID)
|
||||
|
||||
response = requests.post(
|
||||
f"{POSTHOG_HOST}/api/projects/{POSTHOG_PROJECT_ID}/query/",
|
||||
headers={
|
||||
"Authorization": f"Bearer {POSTHOG_API_KEY}",
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
json={
|
||||
"query": {
|
||||
"kind": "HogQLQuery",
|
||||
"query": query,
|
||||
}
|
||||
},
|
||||
timeout=300,
|
||||
)
|
||||
|
||||
logger.info("PostHog status code: %s", response.status_code)
|
||||
|
||||
if not response.ok:
|
||||
logger.error("PostHog response:\n%s", response.text)
|
||||
|
||||
response.raise_for_status()
|
||||
|
||||
data = response.json()
|
||||
logger.info("Successfully parsed JSON response.")
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# 4. Convert to DataFrame
|
||||
# ------------------------------------------------------------------
|
||||
columns = data.get("columns", [])
|
||||
results = data.get("results", [])
|
||||
|
||||
logger.info("Columns: %s", columns)
|
||||
logger.info("Rows returned: %d", len(results))
|
||||
|
||||
df = pd.DataFrame(results, columns=columns)
|
||||
|
||||
logger.info(
|
||||
"DataFrame shape: %s x %s",
|
||||
df.shape[0],
|
||||
df.shape[1],
|
||||
)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# 5. Save parquet
|
||||
# ------------------------------------------------------------------
|
||||
local_path = f"/tmp/features_{run_id}.parquet"
|
||||
|
||||
logger.info("Writing parquet to %s", local_path)
|
||||
df.to_parquet(local_path, index=False)
|
||||
|
||||
logger.info(
|
||||
"Parquet size: %.2f MB",
|
||||
os.path.getsize(local_path) / (1024 * 1024),
|
||||
)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# 6. Upload to MinIO
|
||||
# ------------------------------------------------------------------
|
||||
logger.info("Connecting to MinIO...")
|
||||
|
||||
s3_hook = S3Hook(aws_conn_id=AWS_CONN_ID)
|
||||
s3_key = f"batches/{run_id}/session_features.parquet"
|
||||
|
||||
logger.info(
|
||||
"Uploading to bucket=%s key=%s",
|
||||
MINIO_BUCKET,
|
||||
s3_key,
|
||||
)
|
||||
|
||||
s3_hook.load_file(
|
||||
filename=local_path,
|
||||
key=s3_key,
|
||||
bucket_name=MINIO_BUCKET,
|
||||
replace=True,
|
||||
)
|
||||
|
||||
logger.info("Upload complete.")
|
||||
|
||||
os.remove(local_path)
|
||||
logger.info("Temporary parquet deleted.")
|
||||
|
||||
logger.info("=== Feature extraction completed successfully ===")
|
||||
|
||||
return {
|
||||
"s3_uri": f"s3://{MINIO_BUCKET}/{s3_key}",
|
||||
"window_start": start_date.isoformat(),
|
||||
"window_end": end_date.isoformat(),
|
||||
"record_count": len(df),
|
||||
}
|
||||
|
||||
except Exception:
|
||||
logger.exception("Feature extraction task failed!")
|
||||
raise
|
||||
|
||||
@task
|
||||
def call_mlflow_inference(extraction_result: dict, **kwargs) -> str:
|
||||
"""
|
||||
Task 2: Reads features from MinIO, sends payload to local MLflow endpoint,
|
||||
saves predictions (including SHAP values) back to MinIO.
|
||||
"""
|
||||
logger.info("=== Starting MLflow inference task ===")
|
||||
run_id = kwargs['run_id']
|
||||
|
||||
features_s3_uri = extraction_result.get("s3_uri") if isinstance(extraction_result, dict) else extraction_result
|
||||
|
||||
s3_hook = S3Hook(aws_conn_id=AWS_CONN_ID)
|
||||
|
||||
bucket = features_s3_uri.split("/")[2]
|
||||
key = "/".join(features_s3_uri.split("/")[3:])
|
||||
|
||||
logger.info("Downloading features from %s", features_s3_uri)
|
||||
local_features_path = s3_hook.download_file(key=key, bucket_name=bucket, local_path="/tmp")
|
||||
df_features = pd.read_parquet(local_features_path)
|
||||
|
||||
if df_features.empty:
|
||||
logger.info("DataFrame is empty. Skipping inference.")
|
||||
os.remove(local_features_path)
|
||||
return features_s3_uri
|
||||
|
||||
logger.info("Loaded %d rows for inference.", len(df_features))
|
||||
|
||||
payload = {"dataframe_split": df_features.to_dict(orient="split")}
|
||||
|
||||
mlflow_url = os.environ.get("MLFLOW_API_URL", "http://host.docker.internal:5001/invocations")
|
||||
|
||||
response = requests.post(
|
||||
mlflow_url,
|
||||
data=json.dumps(payload),
|
||||
headers={"Content-Type": "application/json"},
|
||||
timeout=120
|
||||
)
|
||||
|
||||
if not response.ok:
|
||||
logger.error("MLflow API error: %s - %s", response.status_code, response.text)
|
||||
response.raise_for_status()
|
||||
|
||||
predictions_data = response.json().get("predictions", [])
|
||||
|
||||
# Extract model outputs into the dataframe
|
||||
df_features['prediction'] = [p.get('prediction') for p in predictions_data]
|
||||
df_features['anomaly_score'] = [p.get('anomaly_score') for p in predictions_data]
|
||||
df_features['decision_score'] = [p.get('decision_score') for p in predictions_data]
|
||||
df_features['is_anomaly'] = df_features['prediction'] == -1
|
||||
|
||||
# Extract SHAP values (returned as stringified JSON by your model)
|
||||
df_features['shap_values'] = [p.get('shap_values', '{}') for p in predictions_data]
|
||||
|
||||
local_preds_path = f"/tmp/predictions_{run_id}.parquet"
|
||||
df_features.to_parquet(local_preds_path, index=False)
|
||||
|
||||
preds_s3_key = f"batches/{run_id}/predictions.parquet"
|
||||
|
||||
s3_hook.load_file(
|
||||
filename=local_preds_path,
|
||||
key=preds_s3_key,
|
||||
bucket_name=MINIO_BUCKET,
|
||||
replace=True
|
||||
)
|
||||
|
||||
os.remove(local_features_path)
|
||||
os.remove(local_preds_path)
|
||||
|
||||
return f"s3://{MINIO_BUCKET}/{preds_s3_key}"
|
||||
|
||||
@task
|
||||
def load_predictions_to_postgres(predictions_s3_uri: str, extraction_result: dict, **kwargs):
|
||||
logger.info("=== Loading predictions into PostgreSQL ===")
|
||||
run_id = kwargs["run_id"]
|
||||
|
||||
s3_hook = S3Hook(aws_conn_id=AWS_CONN_ID)
|
||||
pg_hook = PostgresHook(postgres_conn_id=POSTGRES_CONN_ID)
|
||||
|
||||
bucket = predictions_s3_uri.split("/")[2]
|
||||
key = "/".join(predictions_s3_uri.split("/")[3:])
|
||||
|
||||
local_path = s3_hook.download_file(
|
||||
key=key,
|
||||
bucket_name=bucket,
|
||||
local_path="/tmp",
|
||||
)
|
||||
|
||||
df = pd.read_parquet(local_path)
|
||||
logger.info("Loaded %d prediction rows", len(df))
|
||||
|
||||
window_start = extraction_result["window_start"]
|
||||
window_end = extraction_result["window_end"]
|
||||
|
||||
total_sessions = len(df)
|
||||
total_anomalies = int(df["is_anomaly"].sum()) if total_sessions else 0
|
||||
contamination_rate = total_anomalies / total_sessions if total_sessions else 0.0
|
||||
mean_score = float(df["anomaly_score"].mean()) if total_sessions else None
|
||||
max_score = float(df["anomaly_score"].max()) if total_sessions else None
|
||||
|
||||
# ---------------------------
|
||||
# ISOLATE FEATURE COLUMNS
|
||||
# ---------------------------
|
||||
# Define which columns are NOT part of the JSONB feature payload
|
||||
metadata_cols = {
|
||||
'session_id', 'user_id', 'prediction',
|
||||
'anomaly_score', 'decision_score', 'is_anomaly', 'shap_values'
|
||||
}
|
||||
|
||||
# Everything else is a feature
|
||||
feature_cols = [col for col in df.columns if col not in metadata_cols]
|
||||
|
||||
from psycopg2.extras import Json, execute_values
|
||||
|
||||
records = []
|
||||
for _, row in df.iterrows():
|
||||
# 1. Build a dict of features for this specific row (dropping nulls safely)
|
||||
row_features = {
|
||||
col: row[col]
|
||||
for col in feature_cols
|
||||
if pd.notna(row[col])
|
||||
}
|
||||
|
||||
# 2. Parse the SHAP values back into a dict (since MLflow returned stringified JSON)
|
||||
shap_raw = row.get("shap_values")
|
||||
shap_dict = json.loads(shap_raw) if isinstance(shap_raw, str) else (shap_raw or {})
|
||||
|
||||
records.append((
|
||||
None, # batch_id placeholder
|
||||
row["session_id"],
|
||||
row.get("user_id"),
|
||||
window_start,
|
||||
row["anomaly_score"],
|
||||
bool(row["is_anomaly"]),
|
||||
Json(row_features), # Automatically adapts dict to JSONB
|
||||
Json(shap_dict) # Automatically adapts dict to JSONB
|
||||
))
|
||||
|
||||
insert_batch_sql = """
|
||||
INSERT INTO batch_runs (
|
||||
dag_run_id, window_start, window_end, mlflow_model_version, status
|
||||
) VALUES (%s,%s,%s,%s,'SUCCESS')
|
||||
RETURNING batch_id;
|
||||
"""
|
||||
|
||||
prediction_sql = """
|
||||
INSERT INTO session_predictions (
|
||||
batch_id, session_id, user_id, session_start_time,
|
||||
anomaly_score, is_anomaly, session_features, shap_values
|
||||
) VALUES %s;
|
||||
"""
|
||||
|
||||
update_batch_sql = """
|
||||
UPDATE batch_runs
|
||||
SET total_sessions_processed=%s, total_anomalies_detected=%s,
|
||||
contamination_rate=%s, mean_anomaly_score=%s, max_anomaly_score=%s
|
||||
WHERE batch_id=%s;
|
||||
"""
|
||||
|
||||
conn = pg_hook.get_conn()
|
||||
try:
|
||||
with conn:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(
|
||||
insert_batch_sql,
|
||||
(run_id, window_start, window_end, "v1.0.0"),
|
||||
)
|
||||
batch_id = cur.fetchone()[0]
|
||||
|
||||
# Replace the 'None' placeholder with the actual batch_id
|
||||
records = [(batch_id, *r[1:]) for r in records]
|
||||
|
||||
execute_values(cur, prediction_sql, records)
|
||||
|
||||
cur.execute(
|
||||
update_batch_sql,
|
||||
(total_sessions, total_anomalies, contamination_rate,
|
||||
mean_score, max_score, batch_id),
|
||||
)
|
||||
logger.info("Inserted %d predictions for batch %s", total_sessions, batch_id)
|
||||
|
||||
except Exception:
|
||||
conn.rollback()
|
||||
with conn:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute("UPDATE batch_runs SET status='FAILED' WHERE dag_run_id=%s;", (run_id,))
|
||||
raise
|
||||
finally:
|
||||
conn.close()
|
||||
os.remove(local_path)
|
||||
|
||||
# --- Pipeline Orchestration ---
|
||||
features_uri = extract_features_to_minio()
|
||||
predictions_uri = call_mlflow_inference(features_uri)
|
||||
load_predictions_to_postgres(predictions_uri, features_uri)
|
||||
83
dags/sql/init.sql
Normal file
83
dags/sql/init.sql
Normal file
|
|
@ -0,0 +1,83 @@
|
|||
-- Ensure clean setup if restarting from scratch
|
||||
DROP TABLE IF EXISTS session_predictions CASCADE;
|
||||
DROP TABLE IF EXISTS batch_runs CASCADE;
|
||||
|
||||
-- ============================================================
|
||||
-- BATCH RUN METADATA + AGGREGATED STATISTICS
|
||||
-- ============================================================
|
||||
CREATE TABLE batch_runs (
|
||||
batch_id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
|
||||
-- Airflow metadata
|
||||
dag_run_id VARCHAR(255) NOT NULL UNIQUE,
|
||||
|
||||
-- Time window processed by this batch
|
||||
window_start TIMESTAMP WITH TIME ZONE NOT NULL,
|
||||
window_end TIMESTAMP WITH TIME ZONE NOT NULL,
|
||||
|
||||
-- Model lineage
|
||||
mlflow_model_version VARCHAR(50) NOT NULL,
|
||||
|
||||
-- Execution status
|
||||
status VARCHAR(20) NOT NULL
|
||||
CHECK (status IN ('RUNNING', 'SUCCESS', 'FAILED')),
|
||||
|
||||
-- Batch statistics
|
||||
total_sessions_processed INTEGER DEFAULT 0,
|
||||
total_anomalies_detected INTEGER DEFAULT 0,
|
||||
contamination_rate DOUBLE PRECISION,
|
||||
mean_anomaly_score DOUBLE PRECISION,
|
||||
max_anomaly_score DOUBLE PRECISION,
|
||||
|
||||
-- Dataset drift monitoring
|
||||
feature_means_summary JSONB,
|
||||
|
||||
created_at TIMESTAMP WITH TIME ZONE DEFAULT CURRENT_TIMESTAMP,
|
||||
|
||||
CONSTRAINT unique_window UNIQUE (window_start, window_end)
|
||||
);
|
||||
|
||||
CREATE INDEX idx_batch_window
|
||||
ON batch_runs(window_start, window_end);
|
||||
|
||||
CREATE INDEX idx_batch_created_at
|
||||
ON batch_runs(created_at);
|
||||
|
||||
-- ============================================================
|
||||
-- SESSION-LEVEL PREDICTIONS
|
||||
-- ============================================================
|
||||
CREATE TABLE session_predictions (
|
||||
prediction_id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
|
||||
batch_id UUID NOT NULL
|
||||
REFERENCES batch_runs(batch_id)
|
||||
ON DELETE CASCADE,
|
||||
|
||||
session_id VARCHAR(255) NOT NULL,
|
||||
user_id VARCHAR(255),
|
||||
|
||||
session_start_time TIMESTAMP WITH TIME ZONE NOT NULL,
|
||||
|
||||
-- Model outputs
|
||||
anomaly_score DOUBLE PRECISION NOT NULL,
|
||||
is_anomaly BOOLEAN NOT NULL,
|
||||
|
||||
-- Dynamic payloads
|
||||
session_features JSONB NOT NULL,
|
||||
shap_values JSONB NOT NULL,
|
||||
|
||||
created_at TIMESTAMP WITH TIME ZONE DEFAULT CURRENT_TIMESTAMP,
|
||||
|
||||
CONSTRAINT unique_session_per_batch
|
||||
UNIQUE(batch_id, session_id)
|
||||
);
|
||||
|
||||
-- Frequently queried anomaly rows
|
||||
CREATE INDEX idx_session_predictions_anomalies
|
||||
ON session_predictions(batch_id)
|
||||
WHERE is_anomaly = TRUE;
|
||||
|
||||
-- Optional searches inside feature JSON
|
||||
CREATE INDEX idx_session_features_gin
|
||||
ON session_predictions
|
||||
USING GIN(session_features);
|
||||
234
dags/sql/session_features.hql
Normal file
234
dags/sql/session_features.hql
Normal file
|
|
@ -0,0 +1,234 @@
|
|||
WITH base AS (
|
||||
SELECT
|
||||
person_id,
|
||||
timestamp,
|
||||
event,
|
||||
properties,
|
||||
row_number() OVER (
|
||||
PARTITION BY person_id
|
||||
ORDER BY timestamp, event
|
||||
) AS global_order
|
||||
FROM events
|
||||
WHERE timestamp >= toDateTime('{start_date}')
|
||||
AND timestamp < toDateTime('{end_date}')
|
||||
AND event NOT IN ('flutter_error', 'platform_error')
|
||||
),
|
||||
|
||||
ordered_events AS (
|
||||
SELECT
|
||||
*,
|
||||
lagInFrame(timestamp) OVER (
|
||||
PARTITION BY person_id
|
||||
ORDER BY global_order
|
||||
) AS prev_ts,
|
||||
|
||||
lagInFrame(event) OVER (
|
||||
PARTITION BY person_id
|
||||
ORDER BY global_order
|
||||
) AS prev_event
|
||||
FROM base
|
||||
),
|
||||
|
||||
session_flags AS (
|
||||
SELECT
|
||||
*,
|
||||
CASE
|
||||
WHEN prev_ts IS NULL THEN 1
|
||||
WHEN event = 'Application Opened' THEN 1
|
||||
WHEN prev_event IN ('Application Backgrounded', 'Application Closed') THEN 1
|
||||
WHEN dateDiff('minute', prev_ts, timestamp) > 10 THEN 1
|
||||
ELSE 0
|
||||
END AS is_new_session
|
||||
FROM ordered_events
|
||||
),
|
||||
|
||||
sessionized AS (
|
||||
SELECT
|
||||
*,
|
||||
sum(is_new_session) OVER (
|
||||
PARTITION BY person_id
|
||||
ORDER BY global_order
|
||||
) AS session_number
|
||||
FROM session_flags
|
||||
),
|
||||
|
||||
session_duration_calc AS (
|
||||
SELECT
|
||||
*,
|
||||
dateDiff(
|
||||
'second',
|
||||
MIN(timestamp) OVER (
|
||||
PARTITION BY person_id, session_number
|
||||
),
|
||||
timestamp
|
||||
) AS seconds_since_session_start
|
||||
FROM sessionized
|
||||
),
|
||||
|
||||
split_sessions AS (
|
||||
SELECT
|
||||
*,
|
||||
floor(seconds_since_session_start / 1800) AS session_sub_id
|
||||
FROM session_duration_calc
|
||||
),
|
||||
|
||||
session_events AS (
|
||||
SELECT
|
||||
*,
|
||||
row_number() OVER (
|
||||
PARTITION BY person_id, session_number, session_sub_id
|
||||
ORDER BY timestamp, event
|
||||
) AS session_index
|
||||
FROM split_sessions
|
||||
),
|
||||
|
||||
inter_event_calc AS (
|
||||
SELECT
|
||||
person_id,
|
||||
session_number,
|
||||
session_sub_id,
|
||||
timestamp,
|
||||
event,
|
||||
properties,
|
||||
session_index,
|
||||
|
||||
lagInFrame(timestamp) OVER (
|
||||
PARTITION BY person_id, session_number, session_sub_id
|
||||
ORDER BY session_index
|
||||
) AS prev_ts_in_session,
|
||||
|
||||
CASE
|
||||
WHEN session_index = 1 THEN NULL
|
||||
ELSE greatest(
|
||||
0,
|
||||
dateDiff(
|
||||
'second',
|
||||
lagInFrame(timestamp) OVER (
|
||||
PARTITION BY person_id, session_number, session_sub_id
|
||||
ORDER BY session_index
|
||||
),
|
||||
timestamp
|
||||
)
|
||||
)
|
||||
END AS inter_event_time_seconds
|
||||
FROM session_events
|
||||
),
|
||||
|
||||
session_stats AS (
|
||||
SELECT
|
||||
person_id,
|
||||
session_number,
|
||||
session_sub_id,
|
||||
|
||||
COUNT(*) AS event_count,
|
||||
|
||||
uniqExact(event) AS unique_event_types,
|
||||
|
||||
uniqExact(
|
||||
if(
|
||||
event = '$screen',
|
||||
replaceRegexpOne(
|
||||
JSONExtractString(properties, '$screen_name'),
|
||||
'\\?.*$',
|
||||
''
|
||||
),
|
||||
NULL
|
||||
)
|
||||
) AS screens_visited,
|
||||
|
||||
MIN(timestamp) AS session_start,
|
||||
MAX(timestamp) AS session_end,
|
||||
|
||||
dateDiff(
|
||||
'second',
|
||||
MIN(timestamp),
|
||||
MAX(timestamp)
|
||||
) AS duration_seconds,
|
||||
|
||||
COUNT(*) * 60.0 /
|
||||
greatest(
|
||||
dateDiff(
|
||||
'second',
|
||||
MIN(timestamp),
|
||||
MAX(timestamp)
|
||||
),
|
||||
1
|
||||
) AS events_per_minute,
|
||||
|
||||
AVG(inter_event_time_seconds) AS inter_event_time_mean_seconds,
|
||||
|
||||
median(inter_event_time_seconds) AS inter_event_time_median_seconds,
|
||||
|
||||
max(inter_event_time_seconds) AS max_inter_event_gap_seconds,
|
||||
|
||||
sqrt(
|
||||
varSamp(inter_event_time_seconds)
|
||||
) AS inter_event_time_std_seconds,
|
||||
|
||||
toHour(MIN(timestamp)) AS session_start_hour,
|
||||
|
||||
toDayOfWeek(MIN(timestamp)) AS session_day_of_week,
|
||||
|
||||
uniqExact(
|
||||
toStartOfHour(timestamp)
|
||||
) AS distinct_hours_active
|
||||
|
||||
FROM inter_event_calc
|
||||
GROUP BY
|
||||
person_id,
|
||||
session_number,
|
||||
session_sub_id
|
||||
|
||||
HAVING
|
||||
COUNT(*) > 2
|
||||
AND dateDiff(
|
||||
'second',
|
||||
MIN(timestamp),
|
||||
MAX(timestamp)
|
||||
) >= 4
|
||||
)
|
||||
|
||||
SELECT
|
||||
person_id,
|
||||
|
||||
concat(
|
||||
toString(person_id),
|
||||
'_',
|
||||
toString(session_number),
|
||||
'_',
|
||||
toString(session_sub_id)
|
||||
) AS session_id,
|
||||
|
||||
session_start,
|
||||
-- 24-Hour Cyclical Encoding for Session Start
|
||||
sin(2 * pi() * (toUnixTimestamp(session_start) - toUnixTimestamp(toStartOfDay(session_start))) / 86400) AS session_start_sin,
|
||||
cos(2 * pi() * (toUnixTimestamp(session_start) - toUnixTimestamp(toStartOfDay(session_start))) / 86400) AS session_start_cos,
|
||||
|
||||
session_end,
|
||||
-- 24-Hour Cyclical Encoding for Session End
|
||||
sin(2 * pi() * (toUnixTimestamp(session_end) - toUnixTimestamp(toStartOfDay(session_end))) / 86400) AS session_end_sin,
|
||||
cos(2 * pi() * (toUnixTimestamp(session_end) - toUnixTimestamp(toStartOfDay(session_end))) / 86400) AS session_end_cos,
|
||||
|
||||
event_count,
|
||||
unique_event_types,
|
||||
screens_visited,
|
||||
|
||||
duration_seconds,
|
||||
events_per_minute,
|
||||
|
||||
inter_event_time_mean_seconds,
|
||||
inter_event_time_median_seconds,
|
||||
inter_event_time_std_seconds,
|
||||
max_inter_event_gap_seconds,
|
||||
|
||||
session_start_hour,
|
||||
|
||||
session_day_of_week,
|
||||
-- 7-Day Cyclical Encoding for Day of the Week
|
||||
sin(2 * pi() * session_day_of_week / 7) AS session_day_of_week_sin,
|
||||
cos(2 * pi() * session_day_of_week / 7) AS session_day_of_week_cos,
|
||||
|
||||
distinct_hours_active
|
||||
|
||||
FROM session_stats
|
||||
ORDER BY session_start DESC
|
||||
108
docker-compose.yaml
Normal file
108
docker-compose.yaml
Normal file
|
|
@ -0,0 +1,108 @@
|
|||
x-airflow-common: &airflow-common
|
||||
image: apache/airflow:2.7.2
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
environment:
|
||||
&airflow-common-env
|
||||
AIRFLOW__CORE__EXECUTOR: LocalExecutor
|
||||
AIRFLOW__DATABASE__SQL_ALCHEMY_CONN: postgresql+psycopg2://airflow:airflow_password@postgres:5432/fraud_db
|
||||
AIRFLOW__CORE__FERNET_KEY: ''
|
||||
AIRFLOW__CORE__LOAD_EXAMPLES: 'False'
|
||||
_PIP_ADDITIONAL_REQUIREMENTS: 'apache-airflow-providers-amazon apache-airflow-providers-postgres pandas pyarrow requests'
|
||||
AIRFLOW__WEBSERVER__SECRET_KEY: 'this_is_a_very_secure_secret_key'
|
||||
AIRFLOW_CONN_MINIO_CONN: 'aws://minio_admin:minio_password@/?endpoint_url=http%3A%2F%2Fminio%3A9000'
|
||||
env_file:
|
||||
- .env
|
||||
volumes:
|
||||
- ./dags:/opt/airflow/dags
|
||||
- airflow_logs:/opt/airflow/logs
|
||||
user: "50000:0"
|
||||
services:
|
||||
postgres:
|
||||
image: postgres:15
|
||||
container_name: postgres_db
|
||||
environment:
|
||||
POSTGRES_USER: airflow
|
||||
POSTGRES_PASSWORD: airflow_password
|
||||
POSTGRES_DB: fraud_db
|
||||
ports:
|
||||
- "5432:5432"
|
||||
volumes:
|
||||
- ./dags/sql/init.sql:/docker-entrypoint-initdb.d/init.sql
|
||||
- postgres_data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U airflow -d fraud_db"]
|
||||
interval: 5s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
|
||||
minio:
|
||||
image: quay.io/minio/minio
|
||||
container_name: minio_storage
|
||||
command: server /data --console-address ":9001"
|
||||
environment:
|
||||
MINIO_ROOT_USER: minio_admin
|
||||
MINIO_ROOT_PASSWORD: minio_password
|
||||
ports:
|
||||
- "9000:9000"
|
||||
- "9001:9001"
|
||||
volumes:
|
||||
- minio_data:/data
|
||||
|
||||
minio-setup:
|
||||
image: quay.io/minio/mc
|
||||
container_name: minio_setup
|
||||
depends_on:
|
||||
- minio
|
||||
entrypoint: >
|
||||
/bin/sh -c "
|
||||
sleep 5;
|
||||
mc alias set myminio http://minio:9000 minio_admin minio_password;
|
||||
mc mb myminio/fraud-features --ignore-existing;
|
||||
mc policy set public myminio/fraud-features;
|
||||
exit 0;
|
||||
"
|
||||
|
||||
airflow-init:
|
||||
<<: *airflow-common
|
||||
container_name: airflow_init
|
||||
command: version
|
||||
environment:
|
||||
<<: *airflow-common-env
|
||||
_AIRFLOW_DB_MIGRATE: 'true'
|
||||
_AIRFLOW_WWW_USER_CREATE: 'true'
|
||||
_AIRFLOW_WWW_USER_USERNAME: airflow
|
||||
_AIRFLOW_WWW_USER_PASSWORD: airflow_password
|
||||
|
||||
airflow-webserver:
|
||||
<<: *airflow-common
|
||||
container_name: airflow_webserver
|
||||
command: webserver
|
||||
ports:
|
||||
- "8080:8080"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "--fail", "http://localhost:8080/health"]
|
||||
interval: 10s
|
||||
timeout: 10s
|
||||
retries: 5
|
||||
restart: always
|
||||
depends_on:
|
||||
airflow-init:
|
||||
condition: service_completed_successfully
|
||||
|
||||
airflow-scheduler:
|
||||
<<: *airflow-common
|
||||
container_name: airflow_scheduler
|
||||
command: scheduler
|
||||
restart: always
|
||||
depends_on:
|
||||
airflow-init:
|
||||
condition: service_completed_successfully
|
||||
environment:
|
||||
<<: *airflow-common-env # <-- Add this line to merge the common variables
|
||||
AIRFLOW_CONN_POSTGRES_CONN: "postgresql://airflow:airflow_password@postgres_db:5432/fraud_db"
|
||||
|
||||
volumes:
|
||||
postgres_data:
|
||||
minio_data:
|
||||
airflow_logs: # <-- Add this line
|
||||
6
main.py
Normal file
6
main.py
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
def main():
|
||||
print("Hello from fraud-workflow!")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
BIN
predictions.parquet
Normal file
BIN
predictions.parquet
Normal file
Binary file not shown.
12
pyproject.toml
Normal file
12
pyproject.toml
Normal file
|
|
@ -0,0 +1,12 @@
|
|||
[project]
|
||||
name = "fraud-workflow"
|
||||
version = "0.1.0"
|
||||
description = "Add your description here"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.13"
|
||||
dependencies = [
|
||||
"apache-airflow>=3.2.2",
|
||||
"fastparquet>=2026.5.0",
|
||||
"pandas>=3.0.3",
|
||||
"psycopg[binary]>=3.3.4",
|
||||
]
|
||||
BIN
session_features.parquet
Normal file
BIN
session_features.parquet
Normal file
Binary file not shown.
3
test.py
Normal file
3
test.py
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
import pandas
|
||||
df = pandas.read_parquet("./session_features.parquet")
|
||||
print(df.head())
|
||||
Loading…
Add table
Reference in a new issue