173 lines
No EOL
6.3 KiB
Python
173 lines
No EOL
6.3 KiB
Python
import os
|
|
import requests
|
|
import pandas as pd
|
|
import mlflow
|
|
import mlflow.data
|
|
import mlflow.pyfunc
|
|
from mlflow.models.signature import infer_signature
|
|
from sklearn.svm import OneClassSVM
|
|
from sklearn.preprocessing import StandardScaler
|
|
from sklearn.pipeline import Pipeline
|
|
from datetime import datetime
|
|
|
|
# Configuration
|
|
PERSONAL_API_KEY = os.getenv("PERSONAL_API_KEY")
|
|
POSTHOG_PROJECT_ID = os.getenv("POSTHOG_PROJECT_ID")
|
|
POSTHOG_HOST = os.getenv("POSTHOG_HOST")
|
|
|
|
def load_query_template(filepath: str, start_date: str, end_date: str) -> str:
|
|
"""Reads SQL file and replaces {start_date}/{end_date} with formatted strings."""
|
|
with open(filepath, 'r') as f:
|
|
template = f.read()
|
|
return template.format(start_date=start_date, end_date=end_date)
|
|
|
|
def fetch_posthog_data(query_string: str) -> pd.DataFrame:
|
|
"""Executes HogQL/ClickHouse query on the PostHog API."""
|
|
response = requests.post(
|
|
f"{POSTHOG_HOST}/api/projects/{POSTHOG_PROJECT_ID}/query/",
|
|
headers={
|
|
"Authorization": f"Bearer {PERSONAL_API_KEY}",
|
|
"Content-Type": "application/json",
|
|
},
|
|
json={
|
|
"query": {
|
|
"kind": "HogQLQuery",
|
|
"query": query_string,
|
|
}
|
|
},
|
|
)
|
|
response.raise_for_status()
|
|
data = response.json()
|
|
|
|
# PostHog query API responses return arrays for rows ('results') and column names ('columns')
|
|
df = pd.DataFrame(data=data['results'], columns=data['columns'])
|
|
return df
|
|
|
|
def run_training_pipeline():
|
|
mlflow.set_tracking_uri("http://127.0.0.1:8080")
|
|
mlflow.set_experiment("Session_Fraud_Detection")
|
|
|
|
# 1. Parse your template file
|
|
sql_filepath = "./fetch_clean_sessions.sql"
|
|
start_date = "2026-03-21 00:00:00"
|
|
end_date = "2026-04-21 00:00:00"
|
|
populated_query = load_query_template(sql_filepath, start_date=start_date, end_date=end_date)
|
|
|
|
# 2. Ingest your baseline from PostHog
|
|
print("Fetching training baseline from PostHog...")
|
|
df_raw = fetch_posthog_data(populated_query)
|
|
|
|
# Isolate feature space (dropping IDs and raw timestamps)
|
|
feature_cols = [col for col in df_raw.columns if col not in ['person_id', 'session_id', 'session_start', 'session_end']]
|
|
df_features = df_raw[feature_cols].astype(float)
|
|
|
|
# ------------------------------------------------------------------
|
|
# Custom wrapper so inference handles scaling and returns scores
|
|
# ------------------------------------------------------------------
|
|
class OneClassSVMWithScores(mlflow.pyfunc.PythonModel):
|
|
|
|
def load_context(self, context):
|
|
import pickle
|
|
with open(context.artifacts["pipeline"], "rb") as f:
|
|
# This loads the combined StandardScaler + OneClassSVM pipeline
|
|
self.pipeline = pickle.load(f)
|
|
|
|
def predict(self, context, model_input):
|
|
# Pipeline automatically scales inputs before passing them to the SVM
|
|
predictions = self.pipeline.predict(model_input)
|
|
raw_scores = self.pipeline.score_samples(model_input)
|
|
decision_scores = self.pipeline.decision_function(model_input)
|
|
|
|
return pd.DataFrame(
|
|
{
|
|
"prediction": predictions,
|
|
"anomaly_score": raw_scores,
|
|
"decision_score": decision_scores,
|
|
}
|
|
)
|
|
|
|
with mlflow.start_run(run_name="PostHog_NuSVM_Baseline"):
|
|
|
|
# ------------------------------------------------------------------
|
|
# Dataset lineage tracking
|
|
# ------------------------------------------------------------------
|
|
mlflow_dataset = mlflow.data.from_pandas(
|
|
df_features,
|
|
source=populated_query,
|
|
name="posthog_clean_sessions",
|
|
targets=None,
|
|
)
|
|
|
|
mlflow.log_input(
|
|
mlflow_dataset,
|
|
context="training",
|
|
tags={
|
|
"sql_template_file": sql_filepath,
|
|
"start_date": start_date,
|
|
"end_date": end_date,
|
|
},
|
|
)
|
|
|
|
# ------------------------------------------------------------------
|
|
# Train Pipeline (Scaling + Nu-SVM / OneClassSVM)
|
|
# ------------------------------------------------------------------
|
|
# For OneClassSVM, 'nu' acts as an upper bound on training errors
|
|
# and behaves similarly to the contamination rate.
|
|
nu_param = 0.01
|
|
|
|
mlflow.log_params({
|
|
"nu": nu_param,
|
|
"kernel": "rbf",
|
|
"scaler": "StandardScaler"
|
|
})
|
|
|
|
print("Training Nu-SVM (One-Class SVM) Pipeline with feature scaling...")
|
|
|
|
# Bundling preprocessing and model execution
|
|
svm_pipeline = Pipeline([
|
|
('scaler', StandardScaler()),
|
|
('svm', OneClassSVM(nu=nu_param, kernel='rbf'))
|
|
])
|
|
|
|
svm_pipeline.fit(df_features)
|
|
|
|
# ------------------------------------------------------------------
|
|
# Build example output signature
|
|
# ------------------------------------------------------------------
|
|
example_output = pd.DataFrame(
|
|
{
|
|
"prediction": svm_pipeline.predict(df_features.head(5)),
|
|
"anomaly_score": svm_pipeline.score_samples(df_features.head(5)),
|
|
"decision_score": svm_pipeline.decision_function(df_features.head(5)),
|
|
}
|
|
)
|
|
|
|
signature = infer_signature(
|
|
df_features.head(5),
|
|
example_output,
|
|
)
|
|
|
|
# ------------------------------------------------------------------
|
|
# Save pipeline artifact
|
|
# ------------------------------------------------------------------
|
|
import pickle
|
|
import tempfile
|
|
|
|
with tempfile.TemporaryDirectory() as tmpdir:
|
|
pipeline_path = os.path.join(tmpdir, "svm_pipeline.pkl")
|
|
|
|
with open(pipeline_path, "wb") as f:
|
|
pickle.dump(svm_pipeline, f)
|
|
|
|
mlflow.pyfunc.log_model(
|
|
artifact_path="svm_anomaly_model",
|
|
python_model=OneClassSVMWithScores(),
|
|
artifacts={"pipeline": pipeline_path},
|
|
signature=signature,
|
|
registered_model_name="NuSVM_Anomaly_Detector",
|
|
)
|
|
|
|
print("Model tracking and registry entry complete.")
|
|
|
|
if __name__ == "__main__":
|
|
run_training_pipeline() |