-- ================================================================================================ -- Migration 021: ML Model Versioning Schema -- Comprehensive model registry with metadata tracking, version history, and production tags -- ================================================================================================ -- Enable required extensions CREATE EXTENSION IF NOT EXISTS "btree_gin"; -- ================================================================================================ -- ML MODEL VERSIONS TABLE -- Comprehensive model registry with training metrics, hyperparameters, and S3 storage -- ================================================================================================ CREATE TABLE IF NOT EXISTS ml_model_versions ( -- Primary identifiers id SERIAL PRIMARY KEY, model_id VARCHAR(255) NOT NULL UNIQUE, -- Model classification model_type VARCHAR(50) NOT NULL, version VARCHAR(50) NOT NULL, -- Training metadata training_date TIMESTAMPTZ NOT NULL, hyperparameters JSONB NOT NULL DEFAULT '{}'::jsonb, metrics JSONB NOT NULL DEFAULT '{}'::jsonb, data_source VARCHAR(255) NOT NULL, -- Storage and integrity s3_location TEXT NOT NULL, checksum VARCHAR(255) NOT NULL, -- Production status flags is_production BOOLEAN NOT NULL DEFAULT false, is_experimental BOOLEAN NOT NULL DEFAULT true, is_archived BOOLEAN NOT NULL DEFAULT false, -- Additional metadata metadata JSONB NOT NULL DEFAULT '{}'::jsonb, -- Audit timestamps created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(), updated_at TIMESTAMPTZ NOT NULL DEFAULT NOW(), -- Constraints CONSTRAINT unique_model_version UNIQUE (model_type, version), CONSTRAINT chk_version_flags CHECK ( -- At most one of production/experimental can be true (is_production::int + is_experimental::int) <= 1 OR is_archived = true ) ); -- Add table comment COMMENT ON TABLE ml_model_versions IS 'ML model version registry with metadata, hyperparameters, and training metrics. Tracks production, experimental, and archived model versions with S3 storage locations.'; -- Add column comments COMMENT ON COLUMN ml_model_versions.model_id IS 'Unique model identifier (e.g., dqn-v1.0.0)'; COMMENT ON COLUMN ml_model_versions.model_type IS 'Type of ML model (DQN, MAMBA, TFT, etc.)'; COMMENT ON COLUMN ml_model_versions.version IS 'Semantic version (e.g., 1.0.0)'; COMMENT ON COLUMN ml_model_versions.training_date IS 'Date and time when model was trained'; COMMENT ON COLUMN ml_model_versions.hyperparameters IS 'Training hyperparameters (epochs, batch_size, learning_rate, etc.)'; COMMENT ON COLUMN ml_model_versions.metrics IS 'Training and validation metrics (loss, accuracy, Sharpe ratio, etc.)'; COMMENT ON COLUMN ml_model_versions.data_source IS 'Data source identifier (e.g., databento_2024_Q4)'; COMMENT ON COLUMN ml_model_versions.s3_location IS 'S3 path to model artifacts'; COMMENT ON COLUMN ml_model_versions.checksum IS 'SHA-256 checksum of model artifacts for integrity verification'; COMMENT ON COLUMN ml_model_versions.is_production IS 'True if model is deployed in production'; COMMENT ON COLUMN ml_model_versions.is_experimental IS 'True if model is experimental (not production-ready)'; COMMENT ON COLUMN ml_model_versions.is_archived IS 'True if model is archived (no longer in use)'; COMMENT ON COLUMN ml_model_versions.metadata IS 'Additional model-specific metadata'; -- ================================================================================================ -- HIGH-PERFORMANCE INDEXES -- Optimized for model registry query patterns -- ================================================================================================ -- Index for model type queries CREATE INDEX IF NOT EXISTS idx_ml_model_versions_model_type ON ml_model_versions(model_type); -- Index for version queries CREATE INDEX IF NOT EXISTS idx_ml_model_versions_version ON ml_model_versions(version); -- Index for time-series queries (most recent models first) CREATE INDEX IF NOT EXISTS idx_ml_model_versions_training_date ON ml_model_versions(training_date DESC); -- Partial index for production models (most common query) CREATE INDEX IF NOT EXISTS idx_ml_model_versions_is_production ON ml_model_versions(is_production) WHERE is_production = true; -- Partial index for experimental models CREATE INDEX IF NOT EXISTS idx_ml_model_versions_is_experimental ON ml_model_versions(is_experimental) WHERE is_experimental = true; -- Partial index for non-archived models (most common filter) CREATE INDEX IF NOT EXISTS idx_ml_model_versions_is_archived ON ml_model_versions(is_archived) WHERE is_archived = false; -- Composite index for production non-archived models CREATE INDEX IF NOT EXISTS idx_ml_model_versions_production_active ON ml_model_versions(model_type, training_date DESC) WHERE is_production = true AND is_archived = false; -- GIN indexes for JSONB queries CREATE INDEX IF NOT EXISTS idx_ml_model_versions_metadata_gin ON ml_model_versions USING GIN (metadata); CREATE INDEX IF NOT EXISTS idx_ml_model_versions_hyperparameters_gin ON ml_model_versions USING GIN (hyperparameters); CREATE INDEX IF NOT EXISTS idx_ml_model_versions_metrics_gin ON ml_model_versions USING GIN (metrics); -- ================================================================================================ -- TRIGGER FUNCTIONS FOR DATA INTEGRITY -- ================================================================================================ -- Function to update updated_at timestamp CREATE OR REPLACE FUNCTION update_ml_model_versions_timestamp() RETURNS TRIGGER AS $$ BEGIN NEW.updated_at := NOW(); RETURN NEW; END; $$ LANGUAGE plpgsql; -- Trigger to auto-update updated_at on modification CREATE TRIGGER tg_ml_model_versions_update_timestamp BEFORE UPDATE ON ml_model_versions FOR EACH ROW EXECUTE FUNCTION update_ml_model_versions_timestamp(); -- Function to ensure only one production model per type CREATE OR REPLACE FUNCTION ensure_single_production_model() RETURNS TRIGGER AS $$ BEGIN -- If marking as production, demote other production models of same type IF NEW.is_production = true AND OLD.is_production = false THEN UPDATE ml_model_versions SET is_production = false, is_experimental = true, updated_at = NOW() WHERE model_type = NEW.model_type AND is_production = true AND id != NEW.id; END IF; RETURN NEW; END; $$ LANGUAGE plpgsql; -- Trigger to ensure single production model per type CREATE TRIGGER tg_ml_model_versions_single_production BEFORE UPDATE ON ml_model_versions FOR EACH ROW WHEN (NEW.is_production = true) EXECUTE FUNCTION ensure_single_production_model(); -- ================================================================================================ -- ANALYTICAL VIEWS FOR REPORTING -- ================================================================================================ -- View for active models (non-archived) CREATE OR REPLACE VIEW v_active_ml_models AS SELECT model_id, model_type, version, training_date, CASE WHEN is_production THEN 'production' WHEN is_experimental THEN 'experimental' ELSE 'unknown' END as status, data_source, s3_location, checksum, created_at, updated_at FROM ml_model_versions WHERE is_archived = false ORDER BY training_date DESC; COMMENT ON VIEW v_active_ml_models IS 'Active (non-archived) ML models with status classification'; -- View for production models CREATE OR REPLACE VIEW v_production_ml_models AS SELECT model_id, model_type, version, training_date, hyperparameters, metrics, data_source, s3_location, checksum, created_at, updated_at FROM ml_model_versions WHERE is_production = true AND is_archived = false ORDER BY model_type, training_date DESC; COMMENT ON VIEW v_production_ml_models IS 'Production-ready ML models currently deployed'; -- View for model version history CREATE OR REPLACE VIEW v_ml_model_version_history AS SELECT model_type, COUNT(*) as total_versions, COUNT(*) FILTER (WHERE is_production) as production_versions, COUNT(*) FILTER (WHERE is_experimental) as experimental_versions, COUNT(*) FILTER (WHERE is_archived) as archived_versions, MAX(training_date) as latest_training_date, MIN(training_date) as earliest_training_date FROM ml_model_versions GROUP BY model_type ORDER BY total_versions DESC; COMMENT ON VIEW v_ml_model_version_history IS 'Version history statistics by model type'; -- ================================================================================================ -- QUERY FUNCTIONS FOR MODEL REGISTRY -- ================================================================================================ -- Function to get production model by type CREATE OR REPLACE FUNCTION get_production_model_by_type( p_model_type VARCHAR(50) ) RETURNS TABLE ( model_id VARCHAR(255), version VARCHAR(50), training_date TIMESTAMPTZ, s3_location TEXT, checksum VARCHAR(255), hyperparameters JSONB, metrics JSONB ) AS $$ BEGIN RETURN QUERY SELECT m.model_id, m.version, m.training_date, m.s3_location, m.checksum, m.hyperparameters, m.metrics FROM ml_model_versions m WHERE m.model_type = p_model_type AND m.is_production = true AND m.is_archived = false ORDER BY m.training_date DESC LIMIT 1; END; $$ LANGUAGE plpgsql; COMMENT ON FUNCTION get_production_model_by_type IS 'Get the current production model for a specific model type'; -- Function to get model performance comparison CREATE OR REPLACE FUNCTION compare_model_performance( p_model_type VARCHAR(50), p_metric_key VARCHAR(100) ) RETURNS TABLE ( model_id VARCHAR(255), version VARCHAR(50), training_date TIMESTAMPTZ, metric_value NUMERIC, is_production BOOLEAN, rank INTEGER ) AS $$ BEGIN RETURN QUERY SELECT m.model_id, m.version, m.training_date, (m.metrics->p_metric_key)::text::numeric as metric_value, m.is_production, ROW_NUMBER() OVER (ORDER BY (m.metrics->p_metric_key)::text::numeric DESC)::INTEGER as rank FROM ml_model_versions m WHERE m.model_type = p_model_type AND m.is_archived = false AND m.metrics ? p_metric_key ORDER BY metric_value DESC; END; $$ LANGUAGE plpgsql; COMMENT ON FUNCTION compare_model_performance IS 'Compare model performance by specific metric (e.g., accuracy, Sharpe ratio)'; -- ================================================================================================ -- SAMPLE DATA FOR TESTING (commented out for production) -- ================================================================================================ -- Example: Insert sample DQN model -- INSERT INTO ml_model_versions ( -- model_id, model_type, version, training_date, -- hyperparameters, metrics, data_source, s3_location, checksum, -- is_production, is_experimental -- ) VALUES ( -- 'dqn-v1.0.0', -- 'DQN', -- '1.0.0', -- NOW(), -- '{"epochs": 500, "batch_size": 128, "learning_rate": 0.0001}', -- '{"final_loss": 0.001, "best_epoch": 487, "training_time_seconds": 168}', -- 'databento_2024_Q4', -- 's3://foxhunt-ml-models/dqn/1.0.0/', -- 'sha256:abc123def456', -- true, -- false -- ); -- ================================================================================================ -- GRANTS AND PERMISSIONS -- ================================================================================================ -- Note: Uncomment and modify these grants based on your specific user roles -- ML training service permissions -- GRANT SELECT, INSERT, UPDATE ON ml_model_versions TO ml_training_user; -- GRANT USAGE, SELECT ON SEQUENCE ml_model_versions_id_seq TO ml_training_user; -- ML inference service permissions (read-only) -- GRANT SELECT ON ml_model_versions TO ml_inference_user; -- Analytics user (read-only for all views and tables) -- GRANT SELECT ON ALL TABLES IN SCHEMA public TO analytics_user; -- GRANT SELECT ON ALL SEQUENCES IN SCHEMA public TO analytics_user; -- ================================================================================================ -- FINAL VALIDATION -- ================================================================================================ -- Verify table was created successfully DO $$ BEGIN IF NOT EXISTS ( SELECT 1 FROM information_schema.tables WHERE table_name = 'ml_model_versions' ) THEN RAISE EXCEPTION 'Migration 021 failed: ml_model_versions table not created'; END IF; RAISE NOTICE 'Migration 021 completed successfully: ML model versioning schema created'; END $$;