diff --git a/data-track/embeds/README.md b/data-track/embeds/README.md new file mode 100644 index 0000000..f62ce9d --- /dev/null +++ b/data-track/embeds/README.md @@ -0,0 +1,7 @@ +# Data Track embeds + +Static HTML (and assets) hosted for **Notion iframes** in the HYF Data Track curriculum. + +GitHub Pages: after Pages is enabled on this repo, decks are served from: + +`https://hackyourfuture.github.io/Learning-Resources/data-track/embeds/...` diff --git a/data-track/embeds/week-13-slides/ch1_delta_vs_parquet.png b/data-track/embeds/week-13-slides/ch1_delta_vs_parquet.png new file mode 100644 index 0000000..66724fe Binary files /dev/null and b/data-track/embeds/week-13-slides/ch1_delta_vs_parquet.png differ diff --git a/data-track/embeds/week-13-slides/ch1_lake_vs_warehouse.png b/data-track/embeds/week-13-slides/ch1_lake_vs_warehouse.png new file mode 100644 index 0000000..6df01bb Binary files /dev/null and b/data-track/embeds/week-13-slides/ch1_lake_vs_warehouse.png differ diff --git a/data-track/embeds/week-13-slides/ch1_lakehouse_collapse.png b/data-track/embeds/week-13-slides/ch1_lakehouse_collapse.png new file mode 100644 index 0000000..8ff96fe Binary files /dev/null and b/data-track/embeds/week-13-slides/ch1_lakehouse_collapse.png differ diff --git a/data-track/embeds/week-13-slides/ch1_one_vs_many.png b/data-track/embeds/week-13-slides/ch1_one_vs_many.png new file mode 100644 index 0000000..8a56603 Binary files /dev/null and b/data-track/embeds/week-13-slides/ch1_one_vs_many.png differ diff --git a/data-track/embeds/week-13-slides/ch3_lazy_plan.png b/data-track/embeds/week-13-slides/ch3_lazy_plan.png new file mode 100644 index 0000000..f3c2d82 Binary files /dev/null and b/data-track/embeds/week-13-slides/ch3_lazy_plan.png differ diff --git a/data-track/embeds/week-13-slides/ch3_read_actions.png b/data-track/embeds/week-13-slides/ch3_read_actions.png new file mode 100644 index 0000000..0b39d1a Binary files /dev/null and b/data-track/embeds/week-13-slides/ch3_read_actions.png differ diff --git a/data-track/embeds/week-13-slides/ch5_dbt_job_architecture.png b/data-track/embeds/week-13-slides/ch5_dbt_job_architecture.png new file mode 100644 index 0000000..5a6ba47 Binary files /dev/null and b/data-track/embeds/week-13-slides/ch5_dbt_job_architecture.png differ diff --git a/data-track/embeds/week-13-slides/dbx_catalog_explorer.png b/data-track/embeds/week-13-slides/dbx_catalog_explorer.png new file mode 100644 index 0000000..a6ffefe Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_catalog_explorer.png differ diff --git a/data-track/embeds/week-13-slides/dbx_compute.png b/data-track/embeds/week-13-slides/dbx_compute.png new file mode 100644 index 0000000..c683796 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_compute.png differ diff --git a/data-track/embeds/week-13-slides/dbx_create_query.png b/data-track/embeds/week-13-slides/dbx_create_query.png new file mode 100644 index 0000000..9df89de Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_create_query.png differ diff --git a/data-track/embeds/week-13-slides/dbx_dbt_build_timing.png b/data-track/embeds/week-13-slides/dbx_dbt_build_timing.png new file mode 100644 index 0000000..7a638d1 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_dbt_build_timing.png differ diff --git a/data-track/embeds/week-13-slides/dbx_incremental_fact_history.png b/data-track/embeds/week-13-slides/dbx_incremental_fact_history.png new file mode 100644 index 0000000..26576e7 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_incremental_fact_history.png differ diff --git a/data-track/embeds/week-13-slides/dbx_job_create.png b/data-track/embeds/week-13-slides/dbx_job_create.png new file mode 100644 index 0000000..20e9463 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_job_create.png differ diff --git a/data-track/embeds/week-13-slides/dbx_job_runs.png b/data-track/embeds/week-13-slides/dbx_job_runs.png new file mode 100644 index 0000000..95a290b Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_job_runs.png differ diff --git a/data-track/embeds/week-13-slides/dbx_job_schedule.png b/data-track/embeds/week-13-slides/dbx_job_schedule.png new file mode 100644 index 0000000..c92592d Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_job_schedule.png differ diff --git a/data-track/embeds/week-13-slides/dbx_job_schedule_paused_panel.png b/data-track/embeds/week-13-slides/dbx_job_schedule_paused_panel.png new file mode 100644 index 0000000..4d661b8 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_job_schedule_paused_panel.png differ diff --git a/data-track/embeds/week-13-slides/dbx_jobs_list.png b/data-track/embeds/week-13-slides/dbx_jobs_list.png new file mode 100644 index 0000000..4f10a24 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_jobs_list.png differ diff --git a/data-track/embeds/week-13-slides/dbx_nb_attach_sql_warehouse.png b/data-track/embeds/week-13-slides/dbx_nb_attach_sql_warehouse.png new file mode 100644 index 0000000..de64229 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_nb_attach_sql_warehouse.png differ diff --git a/data-track/embeds/week-13-slides/dbx_nb_groupby_show.png b/data-track/embeds/week-13-slides/dbx_nb_groupby_show.png new file mode 100644 index 0000000..6d45b12 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_nb_groupby_show.png differ diff --git a/data-track/embeds/week-13-slides/dbx_nb_hello_cluster.png b/data-track/embeds/week-13-slides/dbx_nb_hello_cluster.png new file mode 100644 index 0000000..f388e81 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_nb_hello_cluster.png differ diff --git a/data-track/embeds/week-13-slides/dbx_nb_new_empty.png b/data-track/embeds/week-13-slides/dbx_nb_new_empty.png new file mode 100644 index 0000000..2ad1808 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_nb_new_empty.png differ diff --git a/data-track/embeds/week-13-slides/dbx_query_attach_warehouse.png b/data-track/embeds/week-13-slides/dbx_query_attach_warehouse.png new file mode 100644 index 0000000..749e773 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_query_attach_warehouse.png differ diff --git a/data-track/embeds/week-13-slides/dbx_sidebar.png b/data-track/embeds/week-13-slides/dbx_sidebar.png new file mode 100644 index 0000000..77f21a7 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_sidebar.png differ diff --git a/data-track/embeds/week-13-slides/dbx_warehouse_connection.png b/data-track/embeds/week-13-slides/dbx_warehouse_connection.png new file mode 100644 index 0000000..0a4b035 Binary files /dev/null and b/data-track/embeds/week-13-slides/dbx_warehouse_connection.png differ diff --git a/data-track/embeds/week-13-slides/week_13__fct_trips_incremental_code_explainer.html b/data-track/embeds/week-13-slides/week_13__fct_trips_incremental_code_explainer.html new file mode 100644 index 0000000..939ede8 --- /dev/null +++ b/data-track/embeds/week-13-slides/week_13__fct_trips_incremental_code_explainer.html @@ -0,0 +1,357 @@ + + + + + +Code Explainer: Incremental fct_trips — compile fork + + + +
+
+ + + + Compile fork · 1st build ↔ 2nd+ +
+ + +
+
+ +
+ is_incremental() = false + scans ~128M rows + creates fct_trips +
+ +
+
+
Model (Jinja — same file both runs)
+
+
1{{
+
2 config(
+
3 materialized='incremental',
+
4 incremental_strategy='merge',
+
5 unique_key='trip_id'
+
6 )
+
7}}
+
8
+
9select t.trip_id, t.pickup_datetime, …
+
10from {{ ref('stg_trips') }} t
+
11left join {{ ref('stg_zones') }} pz …
+
12
+
On this run the guard is false — Jinja deletes the block before SQL reaches the warehouse.
+
On this run the guard is true — the where survives into compiled SQL.
+
13{% if is_incremental() %}
+
14 -- only trips newer than what we already have
+
15 where t.pickup_datetime > (
+
16 select max(pickup_datetime) from {{ this }}
+
17 )
+
18{% endif %}
+
+
+ +
+
Compiled SQL (what the warehouse runs)
+
+
1-- is_incremental() was false → no WHERE
+
2select t.trip_id, t.pickup_datetime, …
+
3from hyf.dev_yourname.stg_trips t
+
4left join hyf.dev_yourname.stg_zones pz …
+
5
+
6-- (filter block compiled away)
+
+ +
+
+ +
+
Say this out loud
+
+ Same model file. First build: table missing → is_incremental() = false → filter deleted → full scan. +
+
+
+ + + + diff --git a/data-track/embeds/week-13-slides/week_13__fct_trips_merge_minisim.html b/data-track/embeds/week-13-slides/week_13__fct_trips_merge_minisim.html new file mode 100644 index 0000000..818404e --- /dev/null +++ b/data-track/embeds/week-13-slides/week_13__fct_trips_merge_minisim.html @@ -0,0 +1,265 @@ + + + + + +MERGE mini-sim: trip_id match vs insert + + + +
+
+ + Toy MERGE mini-sim · unique_key = trip_id +
+ +
+ + + + Step 1 / 5 +
+ +
+
+

Incoming batch (new run)

+ + + +
trip_idpickupfare
+
+ +
+

fct_trips ({{ this }})

+ + + +
trip_idpickupfare
+
+
+ +
+ MATCH → update + NO MATCH → insert + ≥ trap → duplicate risk +
+ +
+
Say this out loud
+
+
+
+ + + + diff --git a/data-track/embeds/week-13-slides/week_13__presentation.html b/data-track/embeds/week-13-slides/week_13__presentation.html new file mode 100644 index 0000000..ff43e2f --- /dev/null +++ b/data-track/embeds/week-13-slides/week_13__presentation.html @@ -0,0 +1,2250 @@ +Mid-week: Big Data on Databricks
+

Mid-week: Big Data on Databricks

+

Week 13 lock-in session

+
+
+

👋 Where are you mid-week?

+

(Quick sense of the room: who has finished Ch1–Ch3? Who has dbt debug green on Databricks? Who has already created a Job?)

+

You are already on the shared workspace. Keep it open in a tab; if a login expired, refresh now. Stuck on access? Raise a hand.

+
+

This session locks the through-line, demos the hard bits live, and clears blockers before the PR.

+
+
+
+

🧭 New platform, familiar craft

+

Weeks 6-12 ran on one machine. This week: Databricks + the lakehouse at 128M rows. Week 10 craft (fct_trips, ref(), tests) mostly travels; adapter + profile switch.

+
Week 10: dbt models + tests          (56K rows, Postgres)
+Week 11: dashboards on the marts
+Week 12: orchestrate + secrets
+Week 13: Databricks + dbt at scale   (128M rows; port, incremental, Jobs)
+
+

Centre of gravity: Ch4 (incremental at 128M). Ch5 closes with a Git-backed Job. Today we re-anchor Ch1–Ch3 so Tasks 2–3 land.

+
+
+

📅 Today's agenda

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
TimeActivityDuration
0:00Progress check8 min
0:08Agenda + Ex A–E frame (Tasks 1–3)7 min
0:15§1 Ch1 Lakehouse recap + live check10 min
0:25§2 Ch2 lock-in + Ex A (128M proof)15 min
0:40§3 Ch3 + Ex B (Task 1 mini)20 min
1:00☕ Break10 min
1:10§4 Ch4 + Ex C + Ex D37 min
1:47§5 Ch5 + Ex E18 min
2:05§6 Live Quiz (Ch1-Ch5)20 min
2:25§7 Optional demos (only if Jobs landed)10 min
2:35Buffer / assignment blockers25 min
+
+
+

⌨️ Live exercises today (workspace, not git switch)

+

Every content chapter has a live beat. Ex B–E = Tasks 1–3 (Ex A is Ch2 lock-in only). Task 4 optional bonuses come later if Jobs land.

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Ex / liveFeedsStartSuccess check
Ch1 liveCh1 lock-inOne past pipeline from Weeks 6–12Neighbor hears "one machine" or "need distributed" + why
Ex ACh2 lock-inSQL editor + hyf-dbt-warehousecount(*) about 128,202,548
Ex BTask 1Python notebook + shared clusterTwo show() tables; no raw collect()
Ex CTask 2dbt project + warehouse connection detailsdbt debugAll checks passed!
Ex DTask 2Incremental fct_tripsTwo timings; second clearly faster
Ex ETask 3Workflows → your Job (or demo Job)Green run + paused schedule + Run URL
+

Already finished an Ex? Help a neighbor or jump to the matching assignment task. Submit as a fork + pull request to data-assignment-week-13.

+
+

⚠️ Never commit your Databricks token. It is an automatic fail.

+
+
+
+

§1 Ch1: The lakehouse idea

+

📖 Recap: mental model

+

10 minutes

+
+
+

📖 Ch1: One machine vs many

+

A single machine has fixed memory and CPUs. pandas / one Postgres server hit that wall. Distributed computing spreads work across machines - with coordination cost and billable cluster time.

+

+

At 56K rows, the cluster is often slower. At 128M rows, the parallel path earns its keep.

+
+

💡 Reach for distributed computing when the data genuinely does not fit, not because it sounds impressive.

+
+
+
+

📖 Ch1: Lake vs warehouse

+

+

Lake: path to files. Warehouse: named tables. The pain was storing (and syncing) both.

+
+
+

📖 Ch1: The lakehouse collapses the split

+

+

Lakehouse: one copy in object storage + a table layer on top.

+

Warehouse guarantees. Lake economics. No brittle copy pipeline between them.

+
+
+

📖 Ch1: Why a folder of Parquet is not a table

+

+

Two nightly jobs write the same folder; a dashboard reads mid-write and sees half of January 15. The transaction log is what makes that impossible (ACID).

+

Same log also gives schema checks, time travel, and atomic MERGE: MERGE is what makes Ex D's second build fast.

+
+
+

⌨️ Ch1 live: is your past work big data?

+

Start: pick one pipeline you built in Weeks 6–12 (taxi dbt, weather job, Airflow DAG, …).

+

Do: rough size (rows is enough). Would one Postgres VM like the shared Azure database handle it comfortably?

+

Success check: tell a neighbor in one sentence: still one machine, or would need distributed, and why.

+
+

For almost everything in this course so far, the answer is one machine. That is the point of Week 13. Cost habit later: when you finish a hands-on, let the cluster auto-terminate.

+
+
+
+

§2 Ch2: Workspace & Unity Catalog

+

📖 Lock-in + Ex A

+

15 minutes

+
+
+

📖 Ch2: The five places

+

+
    +
  • Workspace · Catalog · Compute · SQL · Workflows (Jobs; §5)
  • +
+

Scheduling lives here too. Do not create a Job yet.

+
+
+

📖 Ch2: Cluster vs serverless warehouse

+

+
+
+

📖 Ch2: Cluster vs warehouse (when to use which)

+ + + + + + + + + + + + + + + + + +
Reach for…When…
ClusterPySpark notebooks (driver + executors)
Serverless SQL warehouseSQL editor and dbt
+

Rule of thumb this week: PySpark/Python → cluster; SQL work (dbt, SQL Editor, SQL notebooks) → warehouse.

+
+

Clusters bill per minute and auto-terminate. On this subscription a cold start can take ten minutes. The shared cluster should already be running for today's demos.

+
+
+
+

📖 Ch2: catalog.schema.table

+

Every table has a three-part name. The shared taxi data:

+
hyf . nyc_yellow . raw_trips
+ │         │            │
+catalog  schema       table
+
+

Unity Catalog is the shared namespace across every workspace, cluster, and warehouse. Same name from a notebook and from dbt.

+
+
+

📖 Ch2: Find the table in Catalog Explorer

+

+

Expand hyfnyc_yellowraw_trips.

+
+
+

📖 Ch2: Create → Query

+

From the raw_trips table page: Create → Query.

+

+
+
+

📖 Ch2: Attach hyf-dbt-warehouse

+

Before you run anything, attach the shared warehouse: SQL Warehousehyf-dbt-warehouse.

+

+
+
+

⌨️ Ex A: the 128M-row proof

+

Start: SQL editor attached to hyf-dbt-warehouse.

+

Do:

+
select count(*) from hyf.nyc_yellow.raw_trips
+
+

Success check: you see about 128,202,548. Say the three-part name out loud: catalog · schema · table.

+
+

Already did Ex A earlier this week? Confirm the number once, then help a neighbor.

+
+
+
+

§3 Ch3: PySpark in Databricks

+

📖 Setup + Ex B (Task 1 mini)

+

20 minutes

+
+
+

📖 Ch3: What is a notebook?

+

A notebook is cells: type code, run that cell, see the result under it. Variables stay in memory. On Databricks a cell can be Python (PySpark) or SQL. Unlike a .py script, the session stays up - load once, try filters many times.

+
+

⚠️ Cell order is the order you ran, not the order on screen. When unsure, run top to bottom.

+
+

Follow along if you already have one:

+
    +
  1. Open (or create) a Python notebook under Workspace.
  2. +
  3. Attach hyf-dbt-warehouse, run a SQL count(*) cell.
  4. +
  5. Attach the shared (pre-started) cluster.
  6. +
  7. Run: print("hello from the cluster").
  8. +
+
+

Forgetting to attach is still the most common reason a cell does nothing.

+
+
+
+

📖 Ch3: Create notebook → attach warehouse

+

New Python notebook (empty cell, compute dropdown in the toolbar):

+

+
+
+

📖 Ch3: Attach hyf-dbt-warehouse

+

Compute type → SQL Warehousehyf-dbt-warehouse → Attach. Run a SQL count(*) cell (same ballpark as Ex A).

+

+
+
+

📖 Ch3: Then attach the cluster + hello

+

Switch the notebook to the shared cluster (warehouse for SQL/dbt; cluster for PySpark - same notebook, different compute):

+

+
+
+

📖 Ch3: Read the shared tables

+
trips = spark.read.table("hyf.nyc_yellow.raw_trips")
+zones = spark.read.table("hyf.nyc_yellow.raw_zones")
+
+

+
+
+

📖 Ch3: Lazy plan, then an action

+

+

Transformations return instantly (a plan). An action (show / count / write) is what runs it across the cluster.

+
+
+

📖 Ch3: …then show() runs it

+

+
+

⚠️ collect() on a raw 128M-row table crashes the driver. Aggregate first; use show().

+
+
+
+

⌨️ Ex B: Task 1 mini (live)

+

Start: Python notebook on the shared cluster.

+

Do: two lazy queries + one show() each (next slide). Success: two small tables; no raw collect().

+
+

Already finished Task 1? Help a neighbor, or write the PySpark-vs-dbt note.

+
+
+
+

⌨️ Ex B: Task 1 queries (1/2)

+

Top pickup borough (join zones):

+
from pyspark.sql import functions as F
+
+borough_trips = (
+    trips.join(zones, trips.pickup_location_id == zones.location_id)
+    .groupBy("borough")
+    .agg(F.count("*").alias("trip_count"))
+    .orderBy(F.desc("trip_count"))
+)
+borough_trips.show(1)
+
+
+
+

⌨️ Ex B: Task 1 queries (2/2)

+

Average total_amount by payment_type:

+
payment_avg = (
+    trips
+    .groupBy("payment_type")
+    .agg(F.avg("total_amount").alias("avg_total_amount"))
+)
+payment_avg.show()
+
+
+
+

📖 Ch3: PySpark or dbt?

+ + + + + + + + + + + + + + + + + + + + + +
Reach for dbt SQL when…Reach for PySpark when…
The transform is expressible in SQLYou need Python (a library, a per-row API)
You want tests, docs, lineage, ref()The logic is genuinely procedural
The work is a scheduled, reviewable modelYou are exploring interactively
+

For analytics engineering, dbt SQL is the default. Chapter 4 is dbt.

+
+
+

☕ Break

+

10 minutes

+

§4 runs on the serverless warehouse, not the cluster. No restart needed if the cluster auto-terminated.

+
+
+

§4 Ch4: dbt on Databricks

+

📖 Ex C (debug) + Ex D (build twice)

+

37 minutes

+
+
+

📖 Ch4: Port dbt (adapter + profile)

+
pip install dbt-databricks
+
+
nyc_taxi:
+  target: databricks
+  outputs:
+    databricks:
+      type: databricks
+      catalog: hyf
+      schema: "{{ env_var('DBT_SCHEMA') }}"
+      host: "{{ env_var('DATABRICKS_HOST') }}"
+      http_path: "{{ env_var('DATABRICKS_HTTP_PATH') }}"
+      token: "{{ env_var('DATABRICKS_TOKEN') }}"
+      threads: 4
+
+

Models, ref(), tests, YAML: unchanged. Incremental config is the Week 13 twist.

+
+
+

📖 Ch4: Connection details → env vars

+

SQL → SQL Warehouses → hyf-dbt-warehouseConnection details:

+
    +
  • Server hostname → DATABRICKS_HOST
  • +
  • HTTP path → DATABRICKS_HTTP_PATH
  • +
  • Token → DATABRICKS_TOKEN (env var only, never in git)
  • +
  • DBT_SCHEMA=dev_yourname
  • +
+
+
+

📖 Ch4: Connection details in the UI

+

+
+
+

⌨️ Ex C: dbt debug green

+

Start: your Week 10 port or assignment task-2/ with env vars set.

+

Do: dbt debug (or uv run dbt debug).

+

Success check: All checks passed! against catalog hyf and schema dev_<name>.

+
+

Stuck on the port? Diff against nyc-taxi-dbt-reference branch week-13-ch-4-dbt-solution (catch-up only, not the class spine).

+
+
+
+

📖 Ch4: Reality check on 128M rows

+

Pointing Week 10 tests at real data surfaces what the clean 57K sample hid:

+
    +
  • payment_type gains a code 0
  • +
  • A few thousand rows have pickup after dropoff
  • +
+

That is not a porting bug. Widening a test or lowering severity is normal AE work.

+
+
+

📖 Ch4: Reality check as schema tests

+
# Week 10 (clean 57K) — passed
+- accepted_values:
+    values: [1, 2, 3, 4, 5, 6]
+
+# At 128M — fails: yellow adds code 0 (Flex Fare / unknown)
+- accepted_values:
+    values: [0, 1, 2, 3, 4, 5, 6]   # widen the rule = normal AE work
+
+
-- ~6K rows: pickup after dropoff (real dirty data, not a porting bug)
+{{ config(severity='warn') }}  -- or keep error while you investigate
+select * from {{ ref('stg_trips') }}
+where pickup_datetime > dropoff_datetime
+
+
+
+

📖 Ch4: Make fct_trips incremental

+

Week 10:

+
{{ config(materialized='table') }}
+
+

On Databricks:

+
{{
+  config(
+    materialized='incremental',
+    incremental_strategy='merge',
+    unique_key='trip_id'
+  )
+}}
+
+

Add a surrogate trip_id in stg_trips (dbt_utils.generate_surrogate_key). Guard new rows with is_incremental() and > (not >=).

+
+
+

📖 Ch4: The incremental filter

+
{% if is_incremental() %}
+    where t.pickup_datetime > (select max(pickup_datetime) from {{ this }})
+{% endif %}
+
+
    +
  • merge uses Delta's atomic MERGE (Ch1).
  • +
  • unique_key='trip_id' tells dbt how to match rows.
  • +
  • > avoids re-reading the boundary and creating duplicates.
  • +
+
+
+

📖 Ch4: fct_trips (full model, commented)

+
{{
+  config(
+    materialized='incremental',    -- update table; don't drop/rebuild
+    incremental_strategy='merge',  -- Delta MERGE INTO
+    unique_key='trip_id'           -- match key for MERGE
+  )
+}}
+
+select
+    t.trip_id,  -- surrogate key from stg_trips (dbt_utils)
+    t.pickup_datetime, t.dropoff_datetime,
+    t.fare_amount, t.tip_amount, t.trip_distance,
+    t.trip_duration_minutes, t.tip_pct, t.fare_per_mile,
+    t.payment_type_label,
+    pz.borough as pickup_borough, pz.zone as pickup_zone,
+    dz.borough as dropoff_borough, dz.zone as dropoff_zone
+from {{ ref('stg_trips') }} t              -- DAG → your schema
+left join {{ ref('stg_zones') }} pz
+    on t.pickup_location_id = pz.location_id
+left join {{ ref('stg_zones') }} dz
+    on t.dropoff_location_id = dz.location_id
+
+{% if is_incremental() %}  -- true on 2nd+ runs only
+    -- only trips newer than what we already have
+    where t.pickup_datetime > (
+        select max(pickup_datetime) from {{ this }}  -- this = fct_trips
+    )
+{% endif %}
+
+
+
+
+
+
+
+

⌨️ Ex D: build it twice (live)

+

Start: Ex C green; fct_trips incremental config in place.

+

Do:

+
dbt build --select fct_trips   # full history over 128M (often ~1 min; illustrative)
+dbt build --select fct_trips   # incremental (often much faster)
+
+

Success check: two wall-clock times; second clearly faster. Say is_incremental() and {{ this }} out loud.

+

Do not promise exact seconds: warehouse load varies.

+
+
+

⌨️ Ex D: what the timings look like

+

+
+
+

📖 Ch4: Prove it in Delta history

+

+
+
+

📖 Ch4: What to paste in WRITEUP.md

+

In Catalog Explorer or SQL: DESCRIBE HISTORY hyf.dev_yourname.fct_trips.

+

Expect a full create/replace, then MERGE versions. Paste into WRITEUP.md for Task 2.

+
+
+

§5 Ch5: Scheduling dbt Jobs

+

📖 Ex E: Run now, then pause

+

18 minutes

+
+
+

📖 Ch5: Why schedule here?

+

Local dbt build is for development. Production needs the same build without an open laptop.

+ + + + + + + + + + + + + + + + + + + + + +
OptionBest used for
Laptop CLIDeveloping and debugging
Databricks JobDatabricks-native scheduled dbt
Airflow (Week 12)Pipelines that span many systems
+

Workflows / Jobs is a first-class part of Databricks. Task 3 is often the mid-week gap: leave with a green Run URL.

+
+
+

📖 Ch5: How a dbt Job runs

+

Schedule or Run now → Job pulls Git → dbt CLI → warehouse → Unity Catalog.

+

+
+
+

📖 Ch5: Git provider, not Workspace upload

+

+
+
+

📖 Ch5: Demo Job (known-good)

+
    +
  • Repo: lassebenni/nyc-taxi-dbt-reference · branch week-13-ch-4-dbt-solution
  • +
  • Warehouse: hyf-dbt-warehouse · commands: dbt deps then dbt build --select fct_trips
  • +
  • Your Task 3 Job: fork of data-assignment-week-13, branch main, path task-2, name dev_yourname_fct_trips
  • +
+
+
+

📖 Ch5: Demo Job in the UI

+

+
+
+

⌨️ Ex E: Run now, then pause

+

Start: Workflows → your Job (after the demo Job).

+

Do:

+
    +
  1. Run now once → wait for green.
  2. +
  3. Copy the Job Run URL into SCHEDULING.md.
  4. +
  5. Add a schedule for UI proof, then pause the trigger.
  6. +
+

Success check: green run + paused schedule + Run URL saved. Shared bill: no nightly runs left overnight.

+
+
+

⌨️ Ex E: green run

+

+
+
+

⌨️ Ex E: schedule, then pause

+

+
+
+

⌨️ Ex E: paused schedule

+

+
+
+

📖 Ch5: Jobs vs Airflow (30 seconds)

+
    +
  • Databricks Jobs when the work is already on Databricks (dbt, notebooks, SQL).
  • +
  • Airflow when you orchestrate many systems (blob, Postgres, Databricks, …).
  • +
+

Write 2-3 sentences in SCHEDULING.md.

+
+
+

🧠 Live Quiz · one round, 12 questions (Ch1-Ch5)

+

20 minutes.

+

📱 Open our Live Q&A site on your phone (URL + QR code on screen).

+

🔢 Game code: (your teacher will project it)

+

A retrieval check on this week's material: lakehouse, Unity Catalog, PySpark, dbt incremental, and Git-backed Jobs. No notes. Use it to find gaps before you open the PR.

+
+

💡 Still catching up on a chapter? Observe and listen on those questions. Don't guess-answer just to participate.

+
+
+
+

⌨️ Optional demos (only if Jobs landed)

+

10 minutes.

+

Only if §5 landed with time to spare:

+
    +
  • Easy - Workflows alerting (~2 min): email on Job failure/completion (Task 4 Easy).
  • +
  • Medium - Governance: SHOW GRANTS, lineage graph, written SET TAGS statement.
  • +
  • Harder - Streaming: rate-source demo; stop the query afterward.
  • +
+
+
+

🎯 Week 13 Assignment

+

Task 1 ← Ex B: PySpark + show() (not raw collect()); short PySpark vs dbt note.

+

Task 2 ← Ex C + Ex D: port + incremental fct_trips; two timings + DESCRIBE HISTORY in WRITEUP.md.

+

Task 3 ← Ex E: Git-backed Job on your fork; green Run now; schedule paused; SCHEDULING.md.

+

Task 4 (optional): alerting / governance / streaming under task-4/.

+

Submit: fork data-assignment-week-13task-1/ notebook, task-2/ dbt + WRITEUP, task-3/ scheduling evidence → PR. Never commit a token.

+
+

Blocker rule: stuck more than 10 minutes? Ask in the buffer or Slack now.

+
+
+
+

Next steps

+

By now you can: explain lakehouse + Delta; navigate Databricks; run lazy PySpark safely; port dbt incremental; schedule a Git-backed Job and pause it.

+
    +
  • Close Tasks 1–3 and open the PR before the deadline.
  • +
  • Pause every Job schedule; let clusters auto-terminate.
  • +
  • Week 14: infrastructure as code for the workspace, warehouse, and policies.
  • +
+

Well done: Week 10's models still travel at real scale. Finish the Job + PR.

+
+
+

Thank you

+

Questions? Open a thread in Slack or drop them on the Live Q&A board.

+
+

slide #1 · http://127.0.0.1:8765/week_13__presentation.html#1

slide #2 · http://127.0.0.1:8765/week_13__presentation.html#2

slide #3 · http://127.0.0.1:8765/week_13__presentation.html#3

slide #4 · http://127.0.0.1:8765/week_13__presentation.html#4

slide #5 · http://127.0.0.1:8765/week_13__presentation.html#5

slide #6 · http://127.0.0.1:8765/week_13__presentation.html#6

slide #7 · http://127.0.0.1:8765/week_13__presentation.html#7

slide #8 · http://127.0.0.1:8765/week_13__presentation.html#8

slide #9 · http://127.0.0.1:8765/week_13__presentation.html#9

slide #10 · http://127.0.0.1:8765/week_13__presentation.html#10

slide #11 · http://127.0.0.1:8765/week_13__presentation.html#11

slide #12 · http://127.0.0.1:8765/week_13__presentation.html#12

slide #13 · http://127.0.0.1:8765/week_13__presentation.html#13

slide #14 · http://127.0.0.1:8765/week_13__presentation.html#14

slide #15 · http://127.0.0.1:8765/week_13__presentation.html#15

slide #16 · http://127.0.0.1:8765/week_13__presentation.html#16

slide #17 · http://127.0.0.1:8765/week_13__presentation.html#17

slide #18 · http://127.0.0.1:8765/week_13__presentation.html#18

slide #19 · http://127.0.0.1:8765/week_13__presentation.html#19

slide #20 · http://127.0.0.1:8765/week_13__presentation.html#20

slide #21 · http://127.0.0.1:8765/week_13__presentation.html#21

slide #22 · http://127.0.0.1:8765/week_13__presentation.html#22

slide #23 · http://127.0.0.1:8765/week_13__presentation.html#23

slide #24 · http://127.0.0.1:8765/week_13__presentation.html#24

slide #25 · http://127.0.0.1:8765/week_13__presentation.html#25

slide #26 · http://127.0.0.1:8765/week_13__presentation.html#26

slide #27 · http://127.0.0.1:8765/week_13__presentation.html#27

slide #28 · http://127.0.0.1:8765/week_13__presentation.html#28

slide #29 · http://127.0.0.1:8765/week_13__presentation.html#29

slide #30 · http://127.0.0.1:8765/week_13__presentation.html#30

slide #31 · http://127.0.0.1:8765/week_13__presentation.html#31

slide #32 · http://127.0.0.1:8765/week_13__presentation.html#32

slide #33 · http://127.0.0.1:8765/week_13__presentation.html#33

slide #34 · http://127.0.0.1:8765/week_13__presentation.html#34

slide #35 · http://127.0.0.1:8765/week_13__presentation.html#35

slide #36 · http://127.0.0.1:8765/week_13__presentation.html#36

slide #37 · http://127.0.0.1:8765/week_13__presentation.html#37

slide #38 · http://127.0.0.1:8765/week_13__presentation.html#38

slide #39 · http://127.0.0.1:8765/week_13__presentation.html#39

slide #40 · http://127.0.0.1:8765/week_13__presentation.html#40

slide #41 · http://127.0.0.1:8765/week_13__presentation.html#41

slide #42 · http://127.0.0.1:8765/week_13__presentation.html#42

slide #43 · http://127.0.0.1:8765/week_13__presentation.html#43

slide #44 · http://127.0.0.1:8765/week_13__presentation.html#44

slide #45 · http://127.0.0.1:8765/week_13__presentation.html#45

slide #46 · http://127.0.0.1:8765/week_13__presentation.html#46

slide #47 · http://127.0.0.1:8765/week_13__presentation.html#47

slide #48 · http://127.0.0.1:8765/week_13__presentation.html#48

slide #49 · http://127.0.0.1:8765/week_13__presentation.html#49

slide #50 · http://127.0.0.1:8765/week_13__presentation.html#50

slide #51 · http://127.0.0.1:8765/week_13__presentation.html#51

slide #52 · http://127.0.0.1:8765/week_13__presentation.html#52

slide #53 · http://127.0.0.1:8765/week_13__presentation.html#53

slide #54 · http://127.0.0.1:8765/week_13__presentation.html#54

slide #55 · http://127.0.0.1:8765/week_13__presentation.html#55

slide #56 · http://127.0.0.1:8765/week_13__presentation.html#56

slide #57 · http://127.0.0.1:8765/week_13__presentation.html#57

slide #58 · http://127.0.0.1:8765/week_13__presentation.html#58

slide #59 · http://127.0.0.1:8765/week_13__presentation.html#59

slide #60 · http://127.0.0.1:8765/week_13__presentation.html#60

slide #61 · http://127.0.0.1:8765/week_13__presentation.html#61

slide #62 · http://127.0.0.1:8765/week_13__presentation.html#62

slide #63 · http://127.0.0.1:8765/week_13__presentation.html#63

slide #64 · http://127.0.0.1:8765/week_13__presentation.html#64

slide #65 · http://127.0.0.1:8765/week_13__presentation.html#65

\ No newline at end of file