docs: add talk-to-data case study
Add anonymized architecture, governance examples, diagrams, and interview materials.\n\nCo-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
This commit is contained in:
156
cortex_governance_config.yaml
Normal file
156
cortex_governance_config.yaml
Normal file
@@ -0,0 +1,156 @@
|
||||
# Cortex Analytics Guardrails & Governance Configuration
|
||||
# Semantic Layer: Quotation Analysis
|
||||
# Purpose: Control data access, freshness, and output quality for LLM consumption
|
||||
|
||||
cortex_policies:
|
||||
version: "1.0"
|
||||
semantic_model: "sv_quotation"
|
||||
|
||||
# Data freshness contract: How fresh must data be for different use cases
|
||||
freshness:
|
||||
default_max_age_hours: 24
|
||||
rules:
|
||||
- use_case: "chatbot_quote_inquiry"
|
||||
max_age_hours: 24
|
||||
refresh_trigger: "daily_batch_08_00_utc"
|
||||
description: "Daily batch refresh sufficient for quote research; cutoff 8am UTC (5am EST)"
|
||||
|
||||
- use_case: "real_time_dashboard"
|
||||
max_age_hours: 2
|
||||
refresh_trigger: "hourly"
|
||||
description: "Sales team dashboard requires 2-hour freshness maximum"
|
||||
|
||||
# Access control: Who can query this semantic model
|
||||
access_control:
|
||||
default_role: "[REDACTED: SNOWFLAKE_ROLE]"
|
||||
authorized_teams:
|
||||
- name: "sales_team"
|
||||
snowflake_role: "[REDACTED: SALES_ROLE]"
|
||||
tables: ["sv_quotation", "sv_chatbot_quotation"]
|
||||
max_rows_returned: 100000
|
||||
|
||||
- name: "operations_team"
|
||||
snowflake_role: "[REDACTED: OPS_ROLE]"
|
||||
tables: ["sv_quotation", "sv_chatbot_quotation"]
|
||||
max_rows_returned: 500000
|
||||
|
||||
- name: "cortex_agent"
|
||||
snowflake_role: "[REDACTED: CORTEX_AGENT_ROLE]"
|
||||
tables: ["sv_chatbot_quotation"]
|
||||
max_rows_returned: 1000
|
||||
allowed_functions: ["semantic_search", "similarity_score"]
|
||||
|
||||
# LLM output guardrails: Control what the AI model can do with the data
|
||||
llm_guardrails:
|
||||
|
||||
- rule: "discount_policy_redaction"
|
||||
description: "Hide raw discount values from end users; show only 'within policy' or 'requires approval'"
|
||||
pattern: "discount_.*_pct"
|
||||
action: "redact_numeric_values"
|
||||
replacement_logic: |
|
||||
if discount_total_pct <= discount_ceiling_pct then
|
||||
"Within policy"
|
||||
else
|
||||
"Requires manager approval"
|
||||
end if
|
||||
impact: "User sees governance status, not raw discount rules"
|
||||
|
||||
- rule: "customer_name_masking"
|
||||
description: "Mask actual customer names in chatbot responses"
|
||||
pattern: "customer_name|cliente_nome"
|
||||
action: "mask_value"
|
||||
replacement: "[REDACTED: Customer Information]"
|
||||
exception: "Sales team in authorized_teams can see unmasked values"
|
||||
|
||||
- rule: "pii_scrubbing"
|
||||
description: "Remove personally identifiable information (contact names, emails, phone)"
|
||||
pattern: "sales_person|email|phone|contact_name"
|
||||
action: "redact"
|
||||
exception: "Internal sales team dashboard only"
|
||||
|
||||
- rule: "discount_exception_escalation"
|
||||
description: "Flag when chatbot encounters discount above policy ceiling"
|
||||
pattern: "requires_exception_approval = 1"
|
||||
action: "escalate_to_human"
|
||||
notification: "Send to operations team for manual review"
|
||||
max_escalations_per_hour: 10
|
||||
description: "Prevent bot from auto-approving exceptions"
|
||||
|
||||
# Quality gates: Validation rules before response generation
|
||||
quality_gates:
|
||||
|
||||
- gate: "data_completeness"
|
||||
check: "All required dimensions present (quote_id, material_id, customer_id)"
|
||||
failure_action: "return_error_to_user"
|
||||
error_message: "Quote data incomplete. Please provide quote number."
|
||||
|
||||
- gate: "data_staleness"
|
||||
check: "Data age < freshness.max_age_hours"
|
||||
failure_action: "warn_user"
|
||||
warning_message: "Quote data may be up to 24 hours old."
|
||||
|
||||
- gate: "output_relevance"
|
||||
check: "Cortex confidence score > 0.7 on semantic match"
|
||||
failure_action: "escalate_to_human"
|
||||
threshold_score: 0.7
|
||||
description: "Only respond if model is confident about context"
|
||||
|
||||
# Cost controls: Prevent runaway LLM usage
|
||||
cost_controls:
|
||||
monthly_budget_usd: "[REDACTED: BUDGET]"
|
||||
alert_threshold_pct: 80
|
||||
per_query_max_tokens: 2000
|
||||
max_concurrent_queries: 10
|
||||
|
||||
# Audit & compliance: Track all access and LLM decisions
|
||||
audit:
|
||||
log_level: "full"
|
||||
events_logged:
|
||||
- "user_query"
|
||||
- "data_accessed"
|
||||
- "llm_response_generated"
|
||||
- "redaction_applied"
|
||||
- "exception_escalated"
|
||||
retention_days: 90
|
||||
compliance_flags:
|
||||
- "discount_policy_violation"
|
||||
- "unauthorized_access_attempt"
|
||||
- "data_freshness_breach"
|
||||
|
||||
semantic_model_lineage:
|
||||
description: "How quotation data flows through transformations"
|
||||
stages:
|
||||
1_source:
|
||||
system: "[REDACTED: SAP_SYSTEM]"
|
||||
frequency: "Daily batch 05:00 UTC"
|
||||
tables: ["[REDACTED: SOURCE_QUOTE_TABLE]", "[REDACTED: SOURCE_DISCOUNT_TABLE]"]
|
||||
|
||||
2_staging:
|
||||
schema: "STAGING"
|
||||
models: ["stg_quotes.sql", "stg_discount_conditions.sql"]
|
||||
transformations: "Column renaming, type casting, null handling"
|
||||
|
||||
3_transform:
|
||||
schema: "TRANSFORM"
|
||||
models: ["trf_quotation.sql", "trf_chatbot_quotation.sql"]
|
||||
transformations: "Discount calculation, portfolio mapping, governance flags"
|
||||
|
||||
4_semantic_layer:
|
||||
schema: "DISTRIBUTE_DDL"
|
||||
models: ["sv_quotation", "sv_chatbot_quotation"]
|
||||
purpose: "Semantic views expose business logic, hide implementation details"
|
||||
|
||||
5_cortex_consumption:
|
||||
interface_1: "Snowflake Native App (BI)"
|
||||
interface_2: "[REDACTED: SIEMENS_WORKSPACE]"
|
||||
llm_model: "[REDACTED: LLM_VERSION]"
|
||||
prompt_template: |
|
||||
You are a sales support assistant. Answer quote questions using only
|
||||
the data provided. If discount exceeds policy, flag for human review.
|
||||
Do not disclose raw discount rules.
|
||||
|
||||
notes:
|
||||
- "This configuration enforces 'semantic layer first' — governance lives in data, not prompt engineering."
|
||||
- "Redaction rules + access control + LLM guardrails work together: one fails, user sees safe fallback."
|
||||
- "Cortex policies are version-controlled; audit trail shows all policy changes and who approved them."
|
||||
- "Refresh schedule validates data freshness; chatbot rejects queries on stale data."
|
||||
Reference in New Issue
Block a user