Question 1mediummultiple choice
Study the full Python automation breakdown →Databricks-DE-Assoc Data Transformation and Modeling • Complete Question Bank
Complete Databricks-DE-Assoc Data Transformation and Modeling question bank — all 0 questions with answers and detailed explanations.
{
"table": "orders",
"schema": "order_id LONG, amount DOUBLE, event_time TIMESTAMP",
"error": "AnalysisException: Cannot resolve 'event_time' given input columns: ['order_id', 'amount']",
"pipeline_stage": "Silver_Transformation"
}SparkSession
.readStream
.format("cloudFiles")
.option("cloudFiles.format", "csv")
.option("cloudFiles.inferSchema", "true")
.load("/mnt/source/csvs")
.writeStream
.format("delta")
.option("checkpointLocation", "/mnt/checkpoints/csv_stream")
.table("silver_csv_data"){
"operation": "MERGE INTO target USING source ON target.id = source.id",
"when_matched": "UPDATE SET target.val = source.val",
"when_not_matched": "INSERT *",
"error": "java.lang.IllegalArgumentException: Multiple source rows matched one target row"
}{
"source": "Kafka",
"processing": "Structured Streaming",
"issue": "State store is growing indefinitely.",
"current_code": "df.groupBy('user_id').count().writeStream..."
}{
"operation": "INSERT",
"source_format": "JSON",
"target_table": "raw_data",
"constraint": "NOT NULL",
"status": "FAILED",
"error": "java.lang.NullPointerException: Value at column 'user_id' is null"
}spark.sql.streaming.checkpointLocation: /mnt/delta/checkpoints/sales_data spark.databricks.delta.optimizeWrite.enabled: true spark.databricks.delta.autoCompact.enabled: true
{
"table": "orders",
"zorder_columns": ["customer_id", "order_date"],
"frequency": "daily",
"status": "PENDING"
}{
"table": "orders",
"partition_columns": ["order_date"],
"clustering": "none",
"zorder": "none",
"file_size": "small"
}spark.conf.set("spark.databricks.delta.optimizeWrite.enabled", "true")
spark.conf.set("spark.databricks.delta.autoCompact.enabled", "true")--- Config Snippet --- spark.databricks.delta.retentionDurationCheck.enabled = true spark.databricks.delta.logRetentionDuration = interval 30 days