import os import requests import pandas as pd import mlflow import mlflow.data import mlflow.pyfunc from mlflow.models.signature import infer_signature from sklearn.svm import OneClassSVM from sklearn.preprocessing import StandardScaler from sklearn.pipeline import Pipeline from datetime import datetime # Configuration PERSONAL_API_KEY = os.getenv("PERSONAL_API_KEY") POSTHOG_PROJECT_ID = os.getenv("POSTHOG_PROJECT_ID") POSTHOG_HOST = os.getenv("POSTHOG_HOST") def load_query_template(filepath: str, start_date: str, end_date: str) -> str: """Reads SQL file and replaces {start_date}/{end_date} with formatted strings.""" with open(filepath, 'r') as f: template = f.read() return template.format(start_date=start_date, end_date=end_date) def fetch_posthog_data(query_string: str) -> pd.DataFrame: """Executes HogQL/ClickHouse query on the PostHog API.""" response = requests.post( f"{POSTHOG_HOST}/api/projects/{POSTHOG_PROJECT_ID}/query/", headers={ "Authorization": f"Bearer {PERSONAL_API_KEY}", "Content-Type": "application/json", }, json={ "query": { "kind": "HogQLQuery", "query": query_string, } }, ) response.raise_for_status() data = response.json() # PostHog query API responses return arrays for rows ('results') and column names ('columns') df = pd.DataFrame(data=data['results'], columns=data['columns']) return df def run_training_pipeline(): mlflow.set_tracking_uri("http://127.0.0.1:8080") mlflow.set_experiment("Session_Fraud_Detection") # 1. Parse your template file sql_filepath = "./fetch_clean_sessions.sql" start_date = "2026-03-21 00:00:00" end_date = "2026-04-21 00:00:00" populated_query = load_query_template(sql_filepath, start_date=start_date, end_date=end_date) # 2. Ingest your baseline from PostHog print("Fetching training baseline from PostHog...") df_raw = fetch_posthog_data(populated_query) # Isolate feature space (dropping IDs and raw timestamps) feature_cols = [col for col in df_raw.columns if col not in ['person_id', 'session_id', 'session_start', 'session_end']] df_features = df_raw[feature_cols].astype(float) # ------------------------------------------------------------------ # Custom wrapper so inference handles scaling and returns scores # ------------------------------------------------------------------ class OneClassSVMWithScores(mlflow.pyfunc.PythonModel): def load_context(self, context): import pickle with open(context.artifacts["pipeline"], "rb") as f: # This loads the combined StandardScaler + OneClassSVM pipeline self.pipeline = pickle.load(f) def predict(self, context, model_input): # Pipeline automatically scales inputs before passing them to the SVM predictions = self.pipeline.predict(model_input) raw_scores = self.pipeline.score_samples(model_input) decision_scores = self.pipeline.decision_function(model_input) return pd.DataFrame( { "prediction": predictions, "anomaly_score": raw_scores, "decision_score": decision_scores, } ) with mlflow.start_run(run_name="PostHog_NuSVM_Baseline"): # ------------------------------------------------------------------ # Dataset lineage tracking # ------------------------------------------------------------------ mlflow_dataset = mlflow.data.from_pandas( df_features, source=populated_query, name="posthog_clean_sessions", targets=None, ) mlflow.log_input( mlflow_dataset, context="training", tags={ "sql_template_file": sql_filepath, "start_date": start_date, "end_date": end_date, }, ) # ------------------------------------------------------------------ # Train Pipeline (Scaling + Nu-SVM / OneClassSVM) # ------------------------------------------------------------------ # For OneClassSVM, 'nu' acts as an upper bound on training errors # and behaves similarly to the contamination rate. nu_param = 0.01 mlflow.log_params({ "nu": nu_param, "kernel": "rbf", "scaler": "StandardScaler" }) print("Training Nu-SVM (One-Class SVM) Pipeline with feature scaling...") # Bundling preprocessing and model execution svm_pipeline = Pipeline([ ('scaler', StandardScaler()), ('svm', OneClassSVM(nu=nu_param, kernel='rbf')) ]) svm_pipeline.fit(df_features) # ------------------------------------------------------------------ # Build example output signature # ------------------------------------------------------------------ example_output = pd.DataFrame( { "prediction": svm_pipeline.predict(df_features.head(5)), "anomaly_score": svm_pipeline.score_samples(df_features.head(5)), "decision_score": svm_pipeline.decision_function(df_features.head(5)), } ) signature = infer_signature( df_features.head(5), example_output, ) # ------------------------------------------------------------------ # Save pipeline artifact # ------------------------------------------------------------------ import pickle import tempfile with tempfile.TemporaryDirectory() as tmpdir: pipeline_path = os.path.join(tmpdir, "svm_pipeline.pkl") with open(pipeline_path, "wb") as f: pickle.dump(svm_pipeline, f) mlflow.pyfunc.log_model( artifact_path="svm_anomaly_model", python_model=OneClassSVMWithScores(), artifacts={"pipeline": pipeline_path}, signature=signature, registered_model_name="NuSVM_Anomaly_Detector", ) print("Model tracking and registry entry complete.") if __name__ == "__main__": run_training_pipeline()