mlflow-repo/training/train_and_register_nusvm.py

173 lines
No EOL
6.3 KiB
Python

import os
import requests
import pandas as pd
import mlflow
import mlflow.data
import mlflow.pyfunc
from mlflow.models.signature import infer_signature
from sklearn.svm import OneClassSVM
from sklearn.preprocessing import StandardScaler
from sklearn.pipeline import Pipeline
from datetime import datetime
# Configuration
PERSONAL_API_KEY = os.getenv("PERSONAL_API_KEY")
POSTHOG_PROJECT_ID = os.getenv("POSTHOG_PROJECT_ID")
POSTHOG_HOST = os.getenv("POSTHOG_HOST")
def load_query_template(filepath: str, start_date: str, end_date: str) -> str:
"""Reads SQL file and replaces {start_date}/{end_date} with formatted strings."""
with open(filepath, 'r') as f:
template = f.read()
return template.format(start_date=start_date, end_date=end_date)
def fetch_posthog_data(query_string: str) -> pd.DataFrame:
"""Executes HogQL/ClickHouse query on the PostHog API."""
response = requests.post(
f"{POSTHOG_HOST}/api/projects/{POSTHOG_PROJECT_ID}/query/",
headers={
"Authorization": f"Bearer {PERSONAL_API_KEY}",
"Content-Type": "application/json",
},
json={
"query": {
"kind": "HogQLQuery",
"query": query_string,
}
},
)
response.raise_for_status()
data = response.json()
# PostHog query API responses return arrays for rows ('results') and column names ('columns')
df = pd.DataFrame(data=data['results'], columns=data['columns'])
return df
def run_training_pipeline():
mlflow.set_tracking_uri("http://127.0.0.1:8080")
mlflow.set_experiment("Session_Fraud_Detection")
# 1. Parse your template file
sql_filepath = "./fetch_clean_sessions.sql"
start_date = "2026-03-21 00:00:00"
end_date = "2026-04-21 00:00:00"
populated_query = load_query_template(sql_filepath, start_date=start_date, end_date=end_date)
# 2. Ingest your baseline from PostHog
print("Fetching training baseline from PostHog...")
df_raw = fetch_posthog_data(populated_query)
# Isolate feature space (dropping IDs and raw timestamps)
feature_cols = [col for col in df_raw.columns if col not in ['person_id', 'session_id', 'session_start', 'session_end']]
df_features = df_raw[feature_cols].astype(float)
# ------------------------------------------------------------------
# Custom wrapper so inference handles scaling and returns scores
# ------------------------------------------------------------------
class OneClassSVMWithScores(mlflow.pyfunc.PythonModel):
def load_context(self, context):
import pickle
with open(context.artifacts["pipeline"], "rb") as f:
# This loads the combined StandardScaler + OneClassSVM pipeline
self.pipeline = pickle.load(f)
def predict(self, context, model_input):
# Pipeline automatically scales inputs before passing them to the SVM
predictions = self.pipeline.predict(model_input)
raw_scores = self.pipeline.score_samples(model_input)
decision_scores = self.pipeline.decision_function(model_input)
return pd.DataFrame(
{
"prediction": predictions,
"anomaly_score": raw_scores,
"decision_score": decision_scores,
}
)
with mlflow.start_run(run_name="PostHog_NuSVM_Baseline"):
# ------------------------------------------------------------------
# Dataset lineage tracking
# ------------------------------------------------------------------
mlflow_dataset = mlflow.data.from_pandas(
df_features,
source=populated_query,
name="posthog_clean_sessions",
targets=None,
)
mlflow.log_input(
mlflow_dataset,
context="training",
tags={
"sql_template_file": sql_filepath,
"start_date": start_date,
"end_date": end_date,
},
)
# ------------------------------------------------------------------
# Train Pipeline (Scaling + Nu-SVM / OneClassSVM)
# ------------------------------------------------------------------
# For OneClassSVM, 'nu' acts as an upper bound on training errors
# and behaves similarly to the contamination rate.
nu_param = 0.01
mlflow.log_params({
"nu": nu_param,
"kernel": "rbf",
"scaler": "StandardScaler"
})
print("Training Nu-SVM (One-Class SVM) Pipeline with feature scaling...")
# Bundling preprocessing and model execution
svm_pipeline = Pipeline([
('scaler', StandardScaler()),
('svm', OneClassSVM(nu=nu_param, kernel='rbf'))
])
svm_pipeline.fit(df_features)
# ------------------------------------------------------------------
# Build example output signature
# ------------------------------------------------------------------
example_output = pd.DataFrame(
{
"prediction": svm_pipeline.predict(df_features.head(5)),
"anomaly_score": svm_pipeline.score_samples(df_features.head(5)),
"decision_score": svm_pipeline.decision_function(df_features.head(5)),
}
)
signature = infer_signature(
df_features.head(5),
example_output,
)
# ------------------------------------------------------------------
# Save pipeline artifact
# ------------------------------------------------------------------
import pickle
import tempfile
with tempfile.TemporaryDirectory() as tmpdir:
pipeline_path = os.path.join(tmpdir, "svm_pipeline.pkl")
with open(pipeline_path, "wb") as f:
pickle.dump(svm_pipeline, f)
mlflow.pyfunc.log_model(
artifact_path="svm_anomaly_model",
python_model=OneClassSVMWithScores(),
artifacts={"pipeline": pipeline_path},
signature=signature,
registered_model_name="NuSVM_Anomaly_Detector",
)
print("Model tracking and registry entry complete.")
if __name__ == "__main__":
run_training_pipeline()