From a80f20990abb15e06d27c6dac293ec8834e56b99 Mon Sep 17 00:00:00 2001 From: Troy Scott <846218+troyscott@users.noreply.github.com> Date: Tue, 1 Sep 2026 23:59:36 -0700 Subject: [PATCH 1/6] Publish complete DP-700 study content --- content/published/dp700/book.yaml | 183 +++++++++++++++++- .../published/dp700/chapters/batch-data.md | 141 +++++++++++++- .../dp700/chapters/lifecycle-management.md | 90 ++++++++- .../dp700/chapters/loading-patterns.md | 96 ++++++++- .../published/dp700/chapters/monitor-items.md | 89 ++++++++- .../dp700/chapters/optimize-performance.md | 118 ++++++++++- .../published/dp700/chapters/orchestration.md | 84 +++++++- .../dp700/chapters/resolve-errors.md | 95 ++++++++- .../dp700/chapters/security-governance.md | 143 +++++++++++++- .../dp700/chapters/streaming-data.md | 137 ++++++++++++- tests/content/test_dp700_outline.py | 30 ++- 11 files changed, 1119 insertions(+), 87 deletions(-) diff --git a/content/published/dp700/book.yaml b/content/published/dp700/book.yaml index 5e4afda..f34516c 100644 --- a/content/published/dp700/book.yaml +++ b/content/published/dp700/book.yaml @@ -58,6 +58,162 @@ sources: retrieved_at: 2026-09-02T05:18:00Z content_sha256: 058c6d6feb30c2984ff481f9719866b4df8ae245a6497dcb232cc9a22df243fb chapter_ids: [workspace-settings] + - id: fabric-cicd-overview + title: Introduction to CI/CD in Microsoft Fabric + url: https://learn.microsoft.com/en-us/fabric/cicd/cicd-overview + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: c6d1ff511b78b958eefedd97f7fda84f1a90c28f038f5a9a8608fc9203d1ae12 + chapter_ids: [lifecycle-management] + - id: fabric-permission-model + title: Microsoft Fabric permission model + url: https://learn.microsoft.com/en-us/fabric/security/permission-model + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 9f56c7b377bb93cf4faf351e0d376db1dc11ac85cb8e588cb2b5472147d793cc + chapter_ids: [security-governance] + - id: onelake-data-access-control + title: OneLake data access control model + url: https://learn.microsoft.com/en-us/fabric/onelake/security/data-access-control-model + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 354acecf19058ef3b9b46cc66e79cac85fae72d09dae6b16ad0ca1a5c70d2a37 + chapter_ids: [security-governance] + - id: warehouse-dynamic-masking + title: Dynamic data masking in Fabric Data Warehouse + url: https://learn.microsoft.com/en-us/fabric/data-warehouse/dynamic-data-masking + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 934514244c3d1d7524befcceaf5a19abb4a9ef628fe4bb38c83586b2cf65da55 + chapter_ids: [security-governance] + - id: fabric-information-protection + title: Information protection in Fabric + url: https://learn.microsoft.com/en-us/fabric/governance/information-protection + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 444ee731afe960b7bede916f9a8c8e0e9fd3eb636e019231e7a587b251d7f84a + chapter_ids: [security-governance] + - id: fabric-endorsement + title: Endorsement overview + url: https://learn.microsoft.com/en-us/fabric/governance/endorsement-overview + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 9394961cf55a6e0efa296695d5f45c94dd84bd30e37b1f4e7a1b5680bc25f59d + chapter_ids: [security-governance] + - id: fabric-audit-activities + title: Track user activities in Microsoft Fabric + url: https://learn.microsoft.com/en-us/fabric/admin/track-user-activities + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 5883ebfb1e0078495253d2d10e479e811cb40aee5cdbd6e90cc19cc4aa3a6ae2 + chapter_ids: [security-governance] + - id: pipeline-overview + title: Pipeline overview + url: https://learn.microsoft.com/en-us/fabric/data-factory/pipeline-overview + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: dba592f9593cf0d012435d0b69378878aad33e7db1aadf43859a67c8fcaefeaa + chapter_ids: [orchestration] + - id: pipeline-parameters + title: Parameters for Data Factory in Microsoft Fabric + url: https://learn.microsoft.com/en-us/fabric/data-factory/parameters + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: afd48104e7cabaa3379d941584061a6ecd8a9591537b1a1b2a20c80700c91171 + chapter_ids: [orchestration] + - id: data-movement-decision-guide + title: Choose a data movement strategy + url: https://learn.microsoft.com/en-us/fabric/data-factory/decision-guide-data-movement + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 9c28e69274c076d7db280c6c5b334b74b00c4756e45fa19e984dd7d422a264de + chapter_ids: [loading-patterns, batch-data] + - id: dimensional-model-loading + title: Load tables in a dimensional model + url: https://learn.microsoft.com/en-us/fabric/data-warehouse/dimensional-modeling-load-tables + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: e6148d2232549d19b1788f733b54b66c7fe7bb486d93a8525e7cab358bbb62f3 + chapter_ids: [loading-patterns] + - id: dataflows-gen2-overview + title: Dataflows Gen2 overview + url: https://learn.microsoft.com/en-us/fabric/data-factory/dataflows-gen2-overview + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 207bccd75fa09ec51d15af4b6d02f0ddba2eaa2ce3984ab10edc85626aa66615 + chapter_ids: [batch-data] + - id: onelake-shortcuts + title: OneLake shortcuts + url: https://learn.microsoft.com/en-us/fabric/onelake/onelake-shortcuts + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 5744075b37e0f09ffe6a8ad53ee395c2bf39d424aaefed579028f0fbacb14ed6 + chapter_ids: [batch-data, resolve-errors] + - id: fabric-mirroring + title: What is mirroring in Fabric? + url: https://learn.microsoft.com/en-us/fabric/mirroring/overview + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 6053b3f02e2500a7343f4a0f48bb81b0292aef79d9d78fb886176abee5d020e7 + chapter_ids: [batch-data] + - id: real-time-intelligence-overview + title: What is Real-Time Intelligence? + url: https://learn.microsoft.com/en-us/fabric/real-time-intelligence/overview + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: db73725804e0643422b247e2648b9b7a6b34c7f96914bd45d52d9b2349ae690d + chapter_ids: [streaming-data] + - id: query-acceleration-overview + title: Query acceleration for OneLake shortcuts + url: https://learn.microsoft.com/en-us/fabric/real-time-intelligence/query-acceleration-overview + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: dbeb43e271e47b8d481b4947ed65460fd096d80ae367636b4af27083d18b8282 + chapter_ids: [streaming-data] + - id: structured-streaming-state + title: Stateful processing with Structured Streaming + url: https://learn.microsoft.com/en-us/fabric/data-engineering/structured-streaming-stateful-processing + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 533e2f97674a8f549d70ff090fc1c5389a0bd8ce5fa7cb01455b8b17f92f6772 + chapter_ids: [loading-patterns, streaming-data] + - id: fabric-monitoring-hub + title: Use the Monitoring hub + url: https://learn.microsoft.com/en-us/fabric/admin/monitoring-hub + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: b4accbd5cb3ace8d9461c5c4e2d8bc1e284a1eaaceb90bb0155d71b688a9e760 + chapter_ids: [monitor-items] + - id: dataflow-monitoring + title: Monitor Dataflow Gen2 refreshes + url: https://learn.microsoft.com/en-us/fabric/data-factory/dataflows-gen2-monitor + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 3db701d42efb9edb28e31eeeab2a9ab722f17636e54a107e85477f6564a30389 + chapter_ids: [monitor-items, resolve-errors] + - id: semantic-refresh-summaries + title: Refresh summaries + url: https://learn.microsoft.com/en-us/power-bi/connect-data/refresh-summaries + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 2b0e7caf1d23cbc7becb9acbdeb52982b280757f4f0336856a4236d9581f000b + chapter_ids: [monitor-items] + - id: fabric-activator + title: What is Fabric Activator? + url: https://learn.microsoft.com/en-us/fabric/real-time-intelligence/data-activator/activator-introduction + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: df31c30d7d1997890bb07e029f7896701342be422644e73bbb5c71379addb369 + chapter_ids: [monitor-items] + - id: pipeline-troubleshooting + title: Pipeline troubleshooting guide + url: https://learn.microsoft.com/en-us/fabric/data-factory/pipeline-troubleshoot-guide + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: b73934e3aac0206c60ce1a05e15233be0af212201019d2bfc9da9ae57368cd26 + chapter_ids: [resolve-errors] + - id: lakehouse-delta-tables + title: Lakehouse and Delta tables + url: https://learn.microsoft.com/en-us/fabric/data-engineering/lakehouse-and-delta-tables + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 1f95d4f63b5776cb61fc7b7a6ed76c181ac1f212b4da99264a9d252bb9839820 + chapter_ids: [optimize-performance] + - id: delta-v-order + title: Delta optimization and V-Order + url: https://learn.microsoft.com/en-us/fabric/data-engineering/delta-optimization-and-v-order + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 0608a02fe58ff6011a005af236fa849ff97bdedb81a059171a8814000b3c2b45 + chapter_ids: [optimize-performance] + - id: warehouse-performance + title: Guidelines for Fabric Data Warehouse performance + url: https://learn.microsoft.com/en-us/fabric/data-warehouse/guidelines-warehouse-performance + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 70fea084b235f11015ee4e55670d258bcedceb6b6ee66d86a5191664d53fa025 + chapter_ids: [optimize-performance] + - id: kql-query-best-practices + title: KQL query best practices + url: https://learn.microsoft.com/en-us/kusto/query/best-practices + retrieved_at: 2026-09-02T06:52:59Z + content_sha256: 9e4ddae5cb3f3a33152876d2583ee99c8df9924f8bf03125c9d750a33d456607 + chapter_ids: [optimize-performance] domains: - id: implement-manage title: Implement and manage an analytics solution @@ -78,7 +234,8 @@ domains: slug: lifecycle-management title: Implement lifecycle management in Fabric content_path: lifecycle-management.md - source_ids: [dp700-study-guide] + status: published + source_ids: [dp700-study-guide, fabric-cicd-overview] objectives: - {id: implement.lifecycle.version-control, title: Configure version control} - {id: implement.lifecycle.database-projects, title: Implement database projects} @@ -87,7 +244,8 @@ domains: slug: security-governance title: Configure security and governance content_path: security-governance.md - source_ids: [dp700-study-guide] + status: published + source_ids: [dp700-study-guide, fabric-permission-model, onelake-data-access-control, warehouse-dynamic-masking, fabric-information-protection, fabric-endorsement, fabric-audit-activities] objectives: - {id: implement.security.workspace-access, title: Implement workspace-level access controls} - {id: implement.security.item-access, title: Implement item-level access controls} @@ -101,7 +259,8 @@ domains: slug: orchestration title: Orchestrate processes content_path: orchestration.md - source_ids: [dp700-study-guide] + status: published + source_ids: [dp700-study-guide, pipeline-overview, pipeline-parameters] objectives: - {id: implement.orchestration.choose-tool, title: "Choose between Dataflow Gen2, a pipeline, and a notebook"} - {id: implement.orchestration.triggers, title: Design and implement schedules and event-based triggers} @@ -114,7 +273,8 @@ domains: slug: loading-patterns title: Design and implement loading patterns content_path: loading-patterns.md - source_ids: [dp700-study-guide] + status: published + source_ids: [dp700-study-guide, data-movement-decision-guide, dimensional-model-loading, structured-streaming-state] objectives: - {id: ingest.loading.full-incremental, title: Design and implement full and incremental data loads} - {id: ingest.loading.dimensional, title: Prepare data for loading into a dimensional model} @@ -123,7 +283,8 @@ domains: slug: batch-data title: Ingest and transform batch data content_path: batch-data.md - source_ids: [dp700-study-guide] + status: published + source_ids: [dp700-study-guide, data-movement-decision-guide, dataflows-gen2-overview, onelake-shortcuts, fabric-mirroring] objectives: - {id: ingest.batch.store, title: Choose an appropriate data store} - {id: ingest.batch.transform-tool, title: "Choose between Dataflows Gen2, notebooks, KQL, and T-SQL for data transformation"} @@ -138,7 +299,8 @@ domains: slug: streaming-data title: Ingest and transform streaming data content_path: streaming-data.md - source_ids: [dp700-study-guide] + status: published + source_ids: [dp700-study-guide, real-time-intelligence-overview, query-acceleration-overview, structured-streaming-state] objectives: - {id: ingest.streaming.engine, title: Choose an appropriate streaming engine} - {id: ingest.streaming.native-shortcut, title: Choose between native tables and OneLake shortcuts in Real-Time Intelligence} @@ -155,7 +317,8 @@ domains: slug: monitor-items title: Monitor Fabric items content_path: monitor-items.md - source_ids: [dp700-study-guide] + status: published + source_ids: [dp700-study-guide, fabric-monitoring-hub, dataflow-monitoring, semantic-refresh-summaries, fabric-activator] objectives: - {id: monitor.items.ingestion, title: Monitor data ingestion} - {id: monitor.items.transformation, title: Monitor data transformation} @@ -165,7 +328,8 @@ domains: slug: resolve-errors title: Identify and resolve errors content_path: resolve-errors.md - source_ids: [dp700-study-guide] + status: published + source_ids: [dp700-study-guide, pipeline-troubleshooting, dataflow-monitoring, onelake-shortcuts] objectives: - {id: monitor.errors.pipeline, title: Identify and resolve pipeline errors} - {id: monitor.errors.dataflow, title: Identify and resolve Dataflow Gen2 errors} @@ -178,7 +342,8 @@ domains: slug: optimize-performance title: Optimize performance content_path: optimize-performance.md - source_ids: [dp700-study-guide] + status: published + source_ids: [dp700-study-guide, lakehouse-delta-tables, delta-v-order, warehouse-performance, kql-query-best-practices] objectives: - {id: monitor.optimize.lakehouse, title: Optimize a Lakehouse table} - {id: monitor.optimize.pipeline, title: Optimize a pipeline} diff --git a/content/published/dp700/chapters/batch-data.md b/content/published/dp700/chapters/batch-data.md index 030e2f5..6443b56 100644 --- a/content/published/dp700/chapters/batch-data.md +++ b/content/published/dp700/chapters/batch-data.md @@ -1,14 +1,139 @@ # Ingest and transform batch data - -This chapter will connect store selection, ingestion mechanisms, transformation languages, and data-quality handling into an end-to-end batch design. + +Batch design begins with workload shape: data volume, latency, source ownership, transformation complexity, serving engine, and operational skill. The correct answer is rarely “use the newest feature”; it is the mechanism whose semantics fit the requirement. -## Planned study work +## Objective coverage -- Choose among shortcuts, mirroring, and pipeline copies. -- Compare Dataflows Gen2, notebooks, KQL, and T-SQL. -- Handle duplicate, missing, and late-arriving records explicitly. +| Measured objective | What to master | +| --- | --- | +| Choose an appropriate data store | Lakehouse, warehouse, Eventhouse, and operational-store fit | +| Choose between Dataflows Gen2, notebooks, KQL, and T-SQL for data transformation | User, engine, language, scale, and destination fit | +| Create and manage OneLake shortcuts | References, target/source behavior, credentials, schema, caching, and deletion | +| Implement mirroring | Database, metadata, and open mirroring; replication versus in-place access | +| Ingest data by using pipelines | Copy, parameters, staging, retries, and monitoring | +| Transform data by using PySpark, SQL, and KQL | Equivalent relational operations and engine-specific strengths | +| Denormalize data | Join normalized entities into a consumption-oriented shape | +| Group and aggregate data | Grain, grouping keys, measures, and null behavior | +| Handle duplicate, missing, and late-arriving data | Explicit quality rules, quarantine, correction, and observability | -## Source +## Choose an appropriate data store -See the [official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700). +| Store | Best fit | Key clue | +| --- | --- | --- | +| Lakehouse | Open files/Delta, Spark engineering, mixed structured data | Multiple engines need the same OneLake data | +| Warehouse | Governed relational analytics and T-SQL | Dimensional SQL serving and transactions dominate | +| Eventhouse | High-ingest, time-series/log analytics with KQL | Low-latency event exploration and time windows dominate | +| Operational database | Application transactions | Point writes and application consistency, not analytical scans | + +Do not choose from language preference alone. Consider concurrency, update pattern, governance, latency, file/table format, and consumers. + +## Choose between Dataflows Gen2, notebooks, KQL, and T-SQL for data transformation + +**Dataflow Gen2** is Power Query-based, visual, and approachable for reusable tabular shaping. **Notebooks** provide PySpark/Python/SQL flexibility and distributed control. **T-SQL** fits set-based relational transformations close to a warehouse. **KQL** fits telemetry, logs, semi-structured events, and time-series analysis in Eventhouse. + +### Dataflow Gen2 in depth + +A useful Dataflow Gen2 review follows the data: + +1. Connect and authenticate to a source. +2. Assign correct data types early; type drives comparison, arithmetic, and folding behavior. +3. Select/rename columns and filter rows to reduce unnecessary data. +4. Clean values: replace errors, trim/standardize text, handle nulls, and remove duplicates using an explicit key. +5. Combine data: **merge** performs a join; **append** stacks compatible rows. +6. Reshape: split/merge columns, pivot/unpivot, group, aggregate, and add derived columns. +7. Define destination and write behavior, then monitor refresh. + +**Query folding** pushes supported transformations back to the source. Keep foldable filters and projections early, inspect folding indicators/query plans when available, and avoid assuming every connector or step folds. A nonfolding custom operation can force much more data into the mashup engine. + +Schema strategy matters. Fixed destination schema gives predictable downstream contracts; automatic or dynamic handling eases evolution but can surprise consumers. Treat added columns, removed columns, type changes, and renamed columns as different events with explicit policy. Dataflow destinations and update methods vary, so verify whether append, replace, or another supported behavior matches the target. + +Use Dataflow Gen2 for analyst-friendly transformations and managed destinations; use a notebook when logic needs libraries, tests, complex algorithms, or distributed tuning. Use SQL/KQL when the data already lives in the corresponding engine and set-based work can remain close to storage. + +## Create and manage OneLake shortcuts + +A shortcut is an independent OneLake object pointing to a target path. It avoids an extra copy and exposes supported external or internal data through the OneLake namespace. Deleting the shortcut does not delete the target; moving or deleting the target can break the shortcut. + +Create table shortcuts in the lakehouse `Tables` area for supported table formats and file shortcuts in `Files`. Plan the cloud connection and credentials, source permissions, shortcut security, cache behavior, and schema evolution. The shortcut is read-through access, so source availability and network latency remain relevant. + +## Implement mirroring + +Mirroring adds a source database or catalog to Fabric. **Database mirroring** continuously replicates supported operational data into OneLake Delta tables. **Metadata mirroring** synchronizes catalog metadata and uses shortcuts to open-format data in place. **Open mirroring** accepts change files written to a Fabric landing zone according to the published specification. + +Choose a shortcut for selected open-format tables/folders, mirroring for a database/catalog as a unit, and pipeline/copy when custom cadence, complex shaping, or a non-OneLake destination is needed. Mirrored tables are not general-purpose writable replicas; change the source and let synchronization propagate. + +## Ingest data by using pipelines + +Pipeline Copy Activity supports controlled movement across connectors and destinations. Parameterize dataset/path/table names, choose mappings deliberately, configure parallelism from measurements, and preserve source metadata. Use staging only when required by the connector or performance design. + +Capture rows read/written/skipped, duration, throughput, run ID, watermark, and rejected records. Retry transient throttling or network faults with backoff; do not repeatedly retry invalid credentials or deterministic schema errors. + +## Transform data by using PySpark, SQL, and KQL + +These examples all aggregate valid sales by region, but they execute in different engines: + +```python +from pyspark.sql import functions as F + +result = ( + sales.filter(F.col("Amount").isNotNull()) + .groupBy("Region") + .agg(F.sum("Amount").alias("Revenue")) +) +``` + +```sql +SELECT Region, SUM(Amount) AS Revenue +FROM dbo.Sales +WHERE Amount IS NOT NULL +GROUP BY Region; +``` + +```kusto +Sales +| where isnotnull(Amount) +| summarize Revenue = sum(Amount) by Region +``` + +PySpark favors distributed file/table processing, T-SQL relational warehouse workloads, and KQL event/time-series exploration. Equivalent syntax does not imply identical null, type, optimizer, or consistency behavior. + +## Denormalize data + +Denormalization joins entities into a consumption-oriented table to simplify queries and reduce repeated joins. First declare output grain, then select the correct one-to-one or many-to-one joins. A one-to-many join can multiply rows and measures; validate counts and key uniqueness before and after. + +## Group and aggregate data + +Grouping changes grain. Every nonaggregated output column must be a grouping key. Decide whether measures are additive, semi-additive, or nonadditive. `SUM(Amount)` can be additive; inventory snapshots should not be summed across time; ratios should normally be recomputed from numerator and denominator. + +## Handle duplicate, missing, and late-arriving data + +- **Duplicates:** define the business/event key, rank by trusted version/time, retain one winner, and record discarded rows. +- **Missing values:** distinguish unknown, not applicable, and invalid. Impute only with a defensible rule; otherwise quarantine or preserve null with a quality flag. +- **Late data:** use overlap windows and event/business time; reopen affected partitions or dimensions and restate downstream aggregates when policy requires it. + +Quality handling must be observable. Track rule, count, sample, source, run, and disposition. Silently dropping bad rows makes a pipeline look successful while corrupting completeness. + +## Exam distinctions + +- Merge joins columns; append stacks rows. +- Shortcut references selected data; mirroring brings in a database/catalog and may replicate or reference depending on source. +- Query folding is source pushdown, not merely a successful Dataflow refresh. +- Grouping changes grain; denormalization can multiply rows if join cardinality is wrong. +- Tool choice follows workload and engine, not just which languages the engineer knows. + +## Active recall + +1. When does a shortcut beat a copy, and when does mirroring beat both? +2. What breaks query folding, and why does that matter? +3. Explain merge versus append in Power Query. +4. Why can a dimension join double a fact-table total? +5. Which records would you quarantine rather than impute? +6. Translate a SQL `GROUP BY` into KQL and PySpark. + +## Authoritative sources + +- [Dataflows Gen2 overview](https://learn.microsoft.com/en-us/fabric/data-factory/dataflows-gen2-overview) +- [Choose a data movement strategy](https://learn.microsoft.com/en-us/fabric/data-factory/decision-guide-data-movement) +- [OneLake shortcuts](https://learn.microsoft.com/en-us/fabric/onelake/onelake-shortcuts) +- [Mirroring overview](https://learn.microsoft.com/en-us/fabric/mirroring/overview) +- [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/content/published/dp700/chapters/lifecycle-management.md b/content/published/dp700/chapters/lifecycle-management.md index 5c68a5a..95678bb 100644 --- a/content/published/dp700/chapters/lifecycle-management.md +++ b/content/published/dp700/chapters/lifecycle-management.md @@ -1,14 +1,88 @@ # Implement lifecycle management in Fabric - -This chapter will connect version control, database projects, and deployment pipelines into one reviewable DEV-to-production lifecycle. + +Lifecycle management answers three different questions: **How do we review changes? How do we represent database code? How do we promote tested content?** Git integration, database projects, and deployment pipelines cooperate, but none replaces the others. -## Planned study work +## Objective coverage -- Compare Git integration with deployment pipelines. -- Trace how database project changes are built and reviewed. -- Practice selecting the correct promotion mechanism for a scenario. +| Measured objective | What to master | +| --- | --- | +| Configure version control | Workspace-to-branch connection, sync direction, branches, commits, conflicts, and supported items | +| Implement database projects | Declarative schema, build validation, publish artifacts, and state-based deployment | +| Create and configure deployment pipelines | Dev/test/prod stages, workspace assignment, comparison, deployment rules, and promotion | -## Source +## Configure version control -See the [official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700). +Fabric Git integration connects a workspace to a branch and folder in Azure DevOps or GitHub. The workspace is a live collaboration surface; the repository is the durable, reviewable representation. A **commit to Git** exports supported workspace changes to the branch. An **update from Git** applies branch changes to the workspace. Neither direction is automatic simply because the connection exists. + +Use a development workspace for authoring. Protect the integration branch with pull requests and automated validation. Before syncing, inspect the change list: an update can change or delete live workspace items, while a commit can record unintended portal edits. + +| Situation | Best action | +| --- | --- | +| Developer needs isolation | Work on a feature branch and, where supported, a dedicated workspace | +| Workspace and branch both changed | Compare, resolve the conflict deliberately, then synchronize | +| Secret or environment-specific value | Keep it out of source; use supported parameters, rules, or variable libraries | +| Item is not supported by Git integration | Use an API or deployment tool and document that separate release path | + +Git records definitions, not all external dependencies or data. Connections, credentials, gateway bindings, permissions, and some item properties can require post-deployment configuration. A successful Git sync therefore proves definition synchronization—not production readiness. + +## Implement database projects + +A SQL database project stores the intended database schema as code: tables, views, procedures, functions, roles, and other supported objects. Its build checks whether those declarations form a coherent model and produces a deployable artifact such as a DACPAC. Publishing compares the model with a target database and generates a change plan. + +```sql +CREATE TABLE dbo.Customer ( + CustomerKey bigint NOT NULL, + CustomerName varchar(200) NOT NULL, + RegionCode char(2) NULL, + CONSTRAINT PK_Customer PRIMARY KEY (CustomerKey) +); +``` + +The important distinction is **state based versus migration based**. A database project declares the desired end state; the deployment engine calculates changes. A migration script declares ordered steps. Review generated plans carefully, especially when a rename could be interpreted as drop-and-create or when a type change could lose data. + +A robust flow builds the project in CI, checks naming and static-analysis rules, creates the deployment artifact, tests it against a disposable database, and requires human approval before production. Database projects cover database objects; they do not version Fabric workspaces or orchestrate the release of every Fabric item. + +## Create and configure deployment pipelines + +Fabric deployment pipelines promote supported items through ordered stages commonly named Development, Test, and Production. Assign a workspace to each stage, compare adjacent stages, select content, configure supported deployment rules, and deploy forward. Pairing recognizes corresponding items across stages; it does not mean every dependency is automatically rebound. + +| Stage | Purpose | Evidence before promotion | +| --- | --- | --- | +| Development | Integrate approved source changes | Build and content validation pass | +| Test | Exercise realistic configuration and data cases | Functional, permission, refresh, and performance checks pass | +| Production | Serve governed consumers | Approval, rollback plan, and post-deployment checks exist | + +Deployment rules can vary supported data-source or parameter values by stage. They are not a secret store and do not grant target permissions. After deployment, verify connections, credentials, schedules, gateway mappings, item bindings, and access separately. + +### Put the three mechanisms together + +1. Author in a development workspace connected to a feature branch. +2. Commit definitions, open a pull request, and run validation. +3. Merge to the integration branch and update the development workspace. +4. Promote approved items through test and production deployment stages. +5. Run post-deployment configuration and smoke tests, then retain the evidence. + +This separation creates traceability: Git answers *what changed and who reviewed it*; a database project validates a SQL schema model; a deployment pipeline answers *what was promoted between Fabric environments*. + +## Exam distinctions + +- Git integration synchronizes a workspace and branch; it is not a Dev/Test/Prod promotion engine. +- Deployment pipelines promote supported Fabric items; they are not general-purpose Git repositories. +- Database projects model database schema; they do not capture data or every server-level setting. +- A deployment that reports success can still have broken credentials, bindings, permissions, or schedules. +- Never assume all Fabric item types support identical Git and deployment behavior; check the current supported-item matrix. + +## Active recall + +1. Which direction does **Update from Git** move definitions? +2. Why should a generated database deployment plan be reviewed before publishing? +3. What is the difference between a deployment rule and a secret store? +4. A production item deploys but cannot reach its source. Which lifecycle boundary was missed? +5. When would you choose an isolated feature branch and workspace? + +## Authoritative sources + +- [Introduction to CI/CD in Microsoft Fabric](https://learn.microsoft.com/en-us/fabric/cicd/cicd-overview) +- [CI/CD workflow options in Fabric](https://learn.microsoft.com/en-us/fabric/cicd/manage-deployment) +- [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/content/published/dp700/chapters/loading-patterns.md b/content/published/dp700/chapters/loading-patterns.md index 619c67a..5469284 100644 --- a/content/published/dp700/chapters/loading-patterns.md +++ b/content/published/dp700/chapters/loading-patterns.md @@ -1,14 +1,94 @@ # Design and implement loading patterns - -This chapter will model full, incremental, dimensional, and streaming loads as choices driven by source behavior, latency, correctness, and recovery needs. + +A loading pattern is a correctness contract, not merely a transfer method. Define how records are selected, keyed, validated, applied, retried, and reconciled before choosing the Fabric activity. -## Planned study work +## Objective coverage -- Compare watermarks, change tracking, and full reloads. -- Prepare facts and dimensions for reliable loading. -- Design a recoverable streaming ingestion path. +| Measured objective | What to master | +| --- | --- | +| Design and implement full and incremental data loads | Snapshots, watermarks, CDC, upserts, deletes, idempotency, and reconciliation | +| Prepare data for loading into a dimensional model | Grain, surrogate keys, dimensions, facts, SCD behavior, and late members | +| Design and implement a loading pattern for streaming data | Event time, checkpoints, deduplication, windows, late data, and serving layers | -## Source +## Design and implement full and incremental data loads -See the [official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700). +A **full load** replaces or rebuilds the target from the complete selected source. It is simple and self-reconciling but can be expensive and disruptive. An **incremental load** selects only changes since a trusted position, reducing work while increasing state-management complexity. + +| Change signal | Strength | Main risk | +| --- | --- | --- | +| Monotonic ID | Simple | Updates and deletes may be invisible | +| Modified timestamp | Widely available | Clock precision, ties, and late commits | +| Source CDC/version | Captures inserts, updates, and often deletes | Retention and ordering must be managed | +| File/path partition | Efficient for append-oriented sources | Rewritten or late partitions need reconciliation | + +Persist a **high-water mark** only after the target transaction or durable write succeeds. Read an overlap window when timestamps can collide or arrive late, then deduplicate by business key and source version. A safe retry either overwrites a deterministic partition or performs an idempotent upsert. + +```sql +MERGE dbo.Customer AS target +USING #CustomerStage AS source +ON target.CustomerBusinessKey = source.CustomerBusinessKey +WHEN MATCHED AND source.ModifiedAt > target.ModifiedAt THEN + UPDATE SET CustomerName = source.CustomerName, + ModifiedAt = source.ModifiedAt +WHEN NOT MATCHED THEN + INSERT (CustomerBusinessKey, CustomerName, ModifiedAt) + VALUES (source.CustomerBusinessKey, source.CustomerName, source.ModifiedAt); +``` + +`MERGE` is not automatically correct: the staged source must contain at most one winning row per target key, delete semantics must be explicit, and concurrency must be tested. Periodic full or bounded reconciliation detects changes missed by the incremental signal. + +## Prepare data for loading into a dimensional model + +Declare a fact table's **grain** in one sentence before selecting columns—for example, “one row per order line at posting time.” Every fact and dimension key must agree with that grain. + +Dimensions describe business entities and normally use warehouse surrogate keys. Facts store measurements and foreign keys to dimensions. Load dimensions before facts so natural/business keys can be resolved to surrogate keys. + +| Dimension change | Behavior | Use when | +| --- | --- | --- | +| Type 1 | Overwrite current value | History is unneeded or correction is intended | +| Type 2 | Expire old row and insert a new version | Reports must reproduce historical attributes | + +A Type 2 row commonly has `ValidFrom`, `ValidTo`, and `IsCurrent`. Detect a change in tracked attributes, expire the current record, and insert a new surrogate-keyed version. Facts resolve the version valid at the event's business time. + +Late-arriving dimensions require an **unknown/inferred member** rather than rejecting every fact. Load the fact against that stable surrogate key, then update or restate according to the chosen policy when the dimension arrives. Degenerate dimensions such as order number can remain on the fact when they have no useful descriptive dimension row. + +Validate uniqueness of dimension business keys, referential integrity, fact counts, additive behavior, and totals against a trusted control. A star schema is not merely denormalization; it is a deliberate grain-and-relationship model for analytics. + +## Design and implement a loading pattern for streaming data + +Streaming replaces a single batch boundary with continuously advancing state. Distinguish **event time** (when the business event occurred) from **processing time** (when the engine handled it). Windowing and late-data policy normally use event time. + +A reliable pattern is: + +1. Ingest immutable events and retain source metadata. +2. Validate schema and quarantine malformed events. +3. Deduplicate using an event ID and bounded state. +4. Apply event-time windows and a watermark defining tolerated lateness. +5. Write to an idempotent sink with a durable checkpoint. +6. Serve curated tables and reconcile them from retained raw events. + +“Exactly once” should be treated as an end-to-end property. A streaming engine checkpoint cannot prevent duplication if the source reuses IDs or the sink performs non-idempotent side effects. Choose the watermark from measured lateness: too short drops legitimate events; too long retains more state and delays final results. + +## Exam distinctions + +- Full versus incremental describes selection and application, not a specific Fabric tool. +- A timestamp watermark is not the same as Spark's event-time watermark, though both bound progress. +- SCD Type 1 overwrites; Type 2 preserves history through new dimension versions. +- A surrogate key is warehouse-managed; a business key comes from the business/source domain. +- Checkpointing supports recovery; idempotent target logic prevents duplicate effects. + +## Active recall + +1. When should the persisted high-water mark advance? +2. Why is a timestamp overlap window paired with deduplication? +3. State the grain of an order-line fact table. +4. How does a fact arriving before its dimension remain loadable? +5. What is the cost of setting a streaming watermark too aggressively? + +## Authoritative sources + +- [Load tables in a dimensional model](https://learn.microsoft.com/en-us/fabric/data-warehouse/dimensional-modeling-load-tables) +- [Choose a data movement strategy](https://learn.microsoft.com/en-us/fabric/data-factory/decision-guide-data-movement) +- [Stateful processing with Structured Streaming](https://learn.microsoft.com/en-us/fabric/data-engineering/structured-streaming-stateful-processing) +- [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/content/published/dp700/chapters/monitor-items.md b/content/published/dp700/chapters/monitor-items.md index 2cb56f9..9a974f3 100644 --- a/content/published/dp700/chapters/monitor-items.md +++ b/content/published/dp700/chapters/monitor-items.md @@ -1,14 +1,87 @@ # Monitor Fabric items - -This chapter will develop an evidence path across ingestion, transformation, semantic model refresh, and actionable alerts. + +Monitoring turns execution into evidence. Start with a service objective, then collect signals that distinguish freshness, correctness, reliability, and efficiency. A green “Succeeded” state proves execution completion—not data quality. -## Planned study work +## Objective coverage -- Identify the correct monitoring surface for each item. -- Correlate upstream ingestion with downstream refresh behavior. -- Configure alerts around operationally meaningful conditions. +| Measured objective | Evidence to inspect | +| --- | --- | +| Monitor data ingestion | Run state, rows/bytes, throughput, latency, watermark, lag, and rejected records | +| Monitor data transformation | Activity/query/job status, duration, resources, input/output counts, and quality results | +| Monitor semantic model refresh | Refresh status, type, duration, partition/table detail, and failure message | +| Configure alerts | Signal, threshold, evaluation cadence, recipient/action, suppression, and recovery | -## Source +## Monitor data ingestion -See the [official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700). +Use the Fabric Monitoring hub for a cross-workspace view of supported job runs, then drill into the owning item for activity-level detail. For pipelines, inspect input/output, activity duration, copy throughput, integration runtime/gateway, and error details. For Eventstreams/Eventhouse, inspect incoming-event rate, processing lag, rejected events, destination status, and ingestion failures. + +Define freshness as an observable equation: + +```text +freshness lag = observation time - newest successfully published business event time +``` + +This is more useful than “last run succeeded.” Also compare rows read, rows written, rows rejected, and source control totals. Track the last committed watermark independently from the current attempt. + +## Monitor data transformation + +The right signals depend on the engine: + +| Item | Operational evidence | +| --- | --- | +| Dataflow Gen2 | Refresh status, duration, query/destination failures, and destination results | +| Notebook/Spark job | Application/session state, stage/task failures, executor behavior, skew/spill, and output checks | +| Pipeline | Activity graph, dependency path, retries, parameters, copy metrics, and child-run IDs | +| Warehouse/Eventhouse query | Duration, resource use, scanned data, queuing, and query text/plan context | + +Attach data assertions to the run: uniqueness, null rate, referential integrity, allowed ranges, schema contract, and reconciliation totals. Store a small quality result containing rule, severity, observed value, threshold, run ID, and disposition. + +## Monitor semantic model refresh + +Semantic model refresh is downstream of ingestion and transformation. Use refresh history or the administrative refresh summary to see status, start/end time, refresh type, and error detail. For large models, inspect table/partition behavior and whether incremental refresh covers the intended date range. + +Classify failures before acting: expired credentials, gateway unavailable, capacity pressure, source timeout, schema change, memory pressure, or invalid query. A retry can help transient capacity/network faults; it does not repair a renamed source column. + +End-to-end freshness is governed by the slowest required stage: + +```text +source event -> ingestion -> transformation -> semantic refresh -> consumer-visible data +``` + +## Configure alerts + +An actionable alert specifies the signal, threshold, evaluation window, recipient, action, deduplication/suppression, and recovery condition. Fabric Activator can evaluate events or periodic observations and trigger actions such as notifications or Power Automate flows. + +Prefer sustained conditions over noisy single samples: “freshness lag exceeds 30 minutes for two evaluations” is usually better than “one run lasted 31 minutes.” Route alerts to an owner who can act, include item/run identifiers and a runbook link, and test both firing and recovery. + +| Alert | Why it matters | First diagnostic | +| --- | --- | --- | +| No successful ingestion by cutoff | Consumer SLA at risk | Trigger/run history and source availability | +| Rejected-row rate above threshold | Correctness degraded | Quality rule and source sample | +| Semantic refresh failed | Published model is stale | Refresh detail and upstream completion | +| Streaming lag rising continuously | Backpressure or destination issue | Input/output rate and resource saturation | + +## Exam distinctions + +- Monitoring hub gives centralized visibility; item detail provides engine-specific evidence. +- Execution success is not data-quality success. +- Audit logs record user activities; monitoring logs record workload execution; OneLake diagnostics records access events. +- Alerts notify or act on conditions; they do not diagnose root cause automatically. +- Refresh history describes the semantic-model operation, not whether upstream data is complete. + +## Active recall + +1. Which metric proves consumer-visible freshness better than last run status? +2. Why compare rows read, written, and rejected? +3. Where do you begin for a cross-workspace job view? +4. Which semantic refresh failures should not be retried unchanged? +5. What makes an alert actionable instead of noisy? + +## Authoritative sources + +- [Use the Monitoring hub](https://learn.microsoft.com/en-us/fabric/admin/monitoring-hub) +- [Monitor Dataflow Gen2 refreshes](https://learn.microsoft.com/en-us/fabric/data-factory/dataflows-gen2-monitor) +- [Refresh summaries](https://learn.microsoft.com/en-us/power-bi/connect-data/refresh-summaries) +- [What is Fabric Activator?](https://learn.microsoft.com/en-us/fabric/real-time-intelligence/data-activator/activator-introduction) +- [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/content/published/dp700/chapters/optimize-performance.md b/content/published/dp700/chapters/optimize-performance.md index 901e144..2726f09 100644 --- a/content/published/dp700/chapters/optimize-performance.md +++ b/content/published/dp700/chapters/optimize-performance.md @@ -1,14 +1,116 @@ # Optimize performance - -This chapter will separate storage layout, orchestration, compute, and query optimization so that each performance change is tied to measured evidence. + +Optimization is a measured loop: define the workload and target, capture a baseline, locate the bottleneck, change one relevant mechanism, and compare correctness, latency, throughput, and capacity cost. More compute is a hypothesis—not a diagnosis. -## Planned study work +## Objective coverage -- Diagnose before selecting an optimization. -- Compare Lakehouse, warehouse, Spark, and real-time tuning levers. -- Validate improvements against a repeatable workload. +| Measured objective | Primary levers | +| --- | --- | +| Optimize a Lakehouse table | Delta file size, compaction, V-Order, partitioning, statistics, and maintenance | +| Optimize a pipeline | Copy parallelism, incremental selection, concurrency, staging, and orchestration overhead | +| Optimize a data warehouse | Distribution/scan reduction, statistics, query plans, concurrency, and table design | +| Optimize Eventstreams and Eventhouses | Early filtering, routing, ingestion batching, retention/cache, KQL shape, and materialization | +| Optimize Spark performance | Partitioning, pruning, shuffle, skew, caching, join strategy, and compute sizing | +| Optimize query performance | Filter/project early, minimize scanned data, inspect plans, and precompute repeated work | -## Source +## Optimize a Lakehouse table -See the [official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700). +Delta tables can accumulate many small Parquet files, increasing listing, metadata, and task overhead. `OPTIMIZE` compacts files; V-Order reorganizes Parquet encoding for Fabric read engines. They address file layout, not incorrect partitioning or poor queries. + +```sql +OPTIMIZE lakehouse.sales_order +VORDER; +``` + +Partition only on columns that support frequent selective pruning and do not create extreme cardinality or tiny partitions. Date is often useful; customer ID often creates too many directories. Compact after meaningful write accumulation, not after every tiny batch. Retain files according to recovery/time-travel and governance requirements before vacuuming obsolete data. + +## Optimize a pipeline + +Measure source read, transfer, sink write, activity queue, and orchestration time separately. Select only changed rows/files and needed columns. Tune Copy Activity parallelism against source, network, capacity, and destination limits; maximum parallelism can trigger throttling and reduce total throughput. + +Avoid thousands of tiny activities when one parameterized or batched operation can do the work. Use bounded ForEach concurrency. Stage only when it enables a required connector path or proven bulk-load benefit. Keep retries exponential and limited so a failing dependency does not multiply load. + +## Optimize a data warehouse + +Start with actual workload evidence: duration, scans, data movement, queuing, blocking, and execution plan. Use appropriate data types and a star schema, keep statistics current where applicable, filter early, avoid `SELECT *`, and precompute stable expensive logic when justified. + +| Symptom | Investigate before scaling | +| --- | --- | +| Large scans | Predicate selectivity, column projection, table organization, statistics | +| Join explosion | Grain, duplicate keys, join cardinality, preaggregation | +| High concurrency latency | Queuing, workload shape, repeated scans, capacity saturation | +| One query regressed | Plan/data-distribution change, statistics, parameter sensitivity | + +V-Order improves file organization for reads; it does not replace sound SQL, statistics, or relational design. + +## Optimize Eventstreams and Eventhouses + +In Eventstreams, filter and project early, avoid needless transformations on branches, route only required events, and monitor input/output/lag at each destination. Batch and schema choices affect destination ingestion efficiency. + +In Eventhouse, set retention for business need and caching for the hot query horizon. In KQL, put selective `where` predicates early, constrain time, select needed columns, prefer term-aware string operators, and reduce both sides before large joins. Use materialized views or update policies for repeated transformations only when their ingestion cost and maintenance semantics are justified. + +For OneLake shortcuts, compare standard access, query acceleration, and native ingestion. Acceleration is attractive for a recent hot window; native tables are better when native policies and consistently low latency are required. + +## Optimize Spark performance + +Spark performance is dominated by data scanned, partition count/size, shuffle, skew, serialization, and compute utilization. + +1. Use the Spark UI to locate the slow stage and inspect task distribution, input, shuffle, spill, and failures. +2. Filter and project before wide transformations. +3. Avoid Python UDFs when built-in Spark functions express the operation. +4. Broadcast only genuinely small join inputs; otherwise choose partitioning that reduces movement. +5. Repartition deliberately before a large join/write; coalesce mainly to reduce partitions without a full shuffle. +6. Cache only reused, expensive intermediate data and unpersist it afterward. +7. Address skew through better keys, preaggregation, adaptive execution, or selective salting before adding executors. + +```python +from pyspark.sql import functions as F + +recent = orders.select("CustomerKey", "OrderDate", "Amount").filter( + F.col("OrderDate") >= F.lit("2026-01-01") +) + +result = recent.join(F.broadcast(customer_region), "CustomerKey") +``` + +The broadcast is appropriate only if `customer_region` is small enough for each executor. Confirm with plan and runtime evidence. + +## Optimize query performance + +Across SQL, KQL, and Spark SQL, the durable principles are similar: + +- reduce rows and columns early; +- use predicates that enable pruning and indexes/statistics where supported; +- avoid repeated parsing, conversion, and computation; +- join at compatible grain and reduce before joining; +- preaggregate/materialize repeated expensive results when freshness permits; +- inspect the engine's actual plan and runtime metrics. + +Optimization must preserve correctness. Compare row counts, totals, null behavior, and representative results after a rewrite. Report both latency and capacity/resource change so a faster query that costs far more is visible as a trade-off. + +## Exam distinctions + +- `OPTIMIZE` compacts Delta files; V-Order changes Parquet layout; partitioning organizes data for pruning. +- Pipeline concurrency and Copy Activity parallelism are related but different controls. +- Cache hot Eventhouse data based on query horizon; retention controls how long data remains. +- Spark `repartition` shuffles to reshape partitions; `coalesce` normally reduces them with less movement. +- A broadcast join copies the small side to executors; it is harmful when the “small” side is not small. +- Scaling capacity can relieve saturation but does not repair unnecessary scans, skew, or bad grain. + +## Active recall + +1. Which symptoms indicate a small-file problem? +2. Why can more Copy Activity parallelism make a pipeline slower? +3. What evidence would justify a materialized view in Eventhouse? +4. Compare repartition and coalesce. +5. Why must optimization results include correctness and capacity cost? +6. A query sped up after scaling. What has—and has not—been proven? + +## Authoritative sources + +- [Lakehouse and Delta tables](https://learn.microsoft.com/en-us/fabric/data-engineering/lakehouse-and-delta-tables) +- [Delta optimization and V-Order](https://learn.microsoft.com/en-us/fabric/data-engineering/delta-optimization-and-v-order) +- [Warehouse performance guidelines](https://learn.microsoft.com/en-us/fabric/data-warehouse/guidelines-warehouse-performance) +- [KQL query best practices](https://learn.microsoft.com/en-us/kusto/query/best-practices) +- [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/content/published/dp700/chapters/orchestration.md b/content/published/dp700/chapters/orchestration.md index bfd9710..e4d8b4c 100644 --- a/content/published/dp700/chapters/orchestration.md +++ b/content/published/dp700/chapters/orchestration.md @@ -1,14 +1,82 @@ # Orchestrate processes - -This chapter will compare Dataflow Gen2, pipelines, and notebooks, then develop schedules, event triggers, parameters, and dynamic orchestration patterns. + +Orchestration coordinates work; transformation changes data. Choose the smallest tool that expresses the work clearly, then use a pipeline when multiple activities need dependencies, retries, parameters, observability, or a trigger. -## Planned study work +## Objective coverage -- Select an orchestration tool from workload constraints. -- Trace parameters through a multi-step pipeline. -- Compare scheduled and event-driven execution. +| Measured objective | What to master | +| --- | --- | +| Choose between Dataflow Gen2, a pipeline, and a notebook | Low-code shaping, workflow coordination, and code-first distributed processing | +| Design and implement schedules and event-based triggers | Time-driven versus event-driven execution, concurrency, idempotency, and late events | +| Implement orchestration patterns with notebooks and pipelines, including parameters and dynamic expressions | Parent-child workflows, metadata-driven loops, dependencies, parameters, variables, and expressions | -## Source +## Choose between Dataflow Gen2, a pipeline, and a notebook -See the [official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700). +| Tool | Primary strength | Choose it when | Do not confuse it with | +| --- | --- | --- | --- | +| Dataflow Gen2 | Visual Power Query transformations | Analysts need repeatable tabular shaping and managed destinations | A general workflow scheduler | +| Pipeline | Coordination and data movement | Activities require ordering, branching, copying, retry, or triggers | The compute engine that performs every transformation | +| Notebook | Code-first Spark/Python/SQL work | Logic is complex, iterative, library-driven, or distributed | A no-code control plane | + +A pipeline can invoke a notebook or Dataflow Gen2 activity. That nesting is often the correct answer: the pipeline owns control flow while the invoked item owns transformation logic. Avoid putting complex business rules into dynamic expressions merely because the pipeline can evaluate them. + +## Design and implement schedules and event-based triggers + +A schedule starts work from a clock: hourly, daily, or another recurrence. It fits predictable availability and periodic reconciliation. An event-based trigger starts from an occurrence such as a file arrival. It reduces polling delay but requires careful event filtering and duplicate handling. + +| Requirement | Prefer | Design concern | +| --- | --- | --- | +| Source closes its books nightly | Schedule | Time zone, daylight saving, and source completion | +| Process each arriving object quickly | Event trigger | Duplicate/out-of-order events and partially written files | +| Guarantee eventual completeness | Scheduled reconciliation, possibly alongside events | Reprocess a bounded watermark safely | + +Triggers do not guarantee business correctness. Design the target operation to be **idempotent**: rerunning the same logical input should not duplicate facts or corrupt state. Record a run ID, source object/version, high-water mark, and outcome. Set concurrency according to whether activities and targets can safely overlap. + +## Implement orchestration patterns with notebooks and pipelines, including parameters and dynamic expressions + +Parameters are immutable inputs for a run. Variables are mutable values used during pipeline execution. System variables expose context such as the pipeline run ID and trigger time. Dynamic expressions combine these values to construct paths, queries, and activity inputs. + +```text +@concat(pipeline().parameters.basePath, '/', + formatDateTime(pipeline().TriggerTime, 'yyyy/MM/dd'), '/') +``` + +Keep configuration typed and explicit. Validate required parameters at the boundary. Do not pass secrets as ordinary parameters or print them into logs. + +### Reusable patterns + +**Parent-child pipeline:** a parent validates shared inputs and invokes specialized child pipelines. This reduces duplication and provides one operational entry point. + +**Metadata-driven ingestion:** a lookup reads enabled datasets; a ForEach invokes a parameterized copy or notebook for each row. Bound concurrency to source and capacity limits, and capture per-dataset results. + +**Notebook chain:** a pipeline passes primitive parameters to a notebook, waits for its outcome, and branches on success or failure. Make the notebook return a small status contract rather than parsing arbitrary log text. + +**Dependency graph:** success, failure, completion, and skip conditions determine which activity runs next. Add a deliberate failure path that records diagnostic context and alerts an owner. A retry is appropriate for transient faults; it is harmful for deterministic schema or permission failures. + +### Scenario walkthrough + +A daily sales load receives `businessDate` and `fullReload` parameters. The pipeline checks that the landing file exists, invokes a notebook to validate schema, runs a parameterized copy for valid data, invokes a SQL procedure to merge the target, and records the watermark only after the merge succeeds. An event trigger provides low latency, while a nightly scheduled run reconciles missing events. Because the merge key is stable, either path can retry safely. + +## Exam distinctions + +- A schedule answers *when*; an activity dependency answers *after what*. +- An event trigger reduces polling but does not eliminate duplicate-event or partial-file risks. +- Parameters do not change during a run; variables can. +- Pipeline expressions configure and route work; notebooks and dataflows normally hold substantial transformation logic. +- Retries address transient failures. Fix authentication, invalid schemas, and bad expressions instead of retrying them repeatedly. + +## Active recall + +1. Which tool owns ordering when a Dataflow Gen2 must run before a notebook? +2. Why pair an event trigger with scheduled reconciliation? +3. When is a pipeline variable more appropriate than a parameter? +4. What makes a metadata-driven ForEach safe to rerun? +5. Which evidence distinguishes a transient failure from a deterministic one? + +## Authoritative sources + +- [Pipeline overview](https://learn.microsoft.com/en-us/fabric/data-factory/pipeline-overview) +- [Parameters for Data Factory in Fabric](https://learn.microsoft.com/en-us/fabric/data-factory/parameters) +- [Data Factory overview](https://learn.microsoft.com/en-us/fabric/data-factory/data-factory-overview) +- [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/content/published/dp700/chapters/resolve-errors.md b/content/published/dp700/chapters/resolve-errors.md index dbcc6fe..dd2d79f 100644 --- a/content/published/dp700/chapters/resolve-errors.md +++ b/content/published/dp700/chapters/resolve-errors.md @@ -1,14 +1,93 @@ # Identify and resolve errors - -This chapter will use a consistent diagnose-isolate-correct-verify loop across pipelines, Dataflows Gen2, notebooks, Eventhouse, Eventstreams, T-SQL, and shortcuts. + +Troubleshooting is an evidence funnel: define the symptom and scope, identify the failing component, capture the most specific error and correlation ID, classify the fault, change one cause, and rerun the smallest representative case. -## Planned study work +## Objective coverage -- Map common failures to their best diagnostic evidence. -- Separate configuration, identity, data, and runtime causes. -- Turn recurring failures into regression checks. +| Measured objective | First evidence | +| --- | --- | +| Identify and resolve pipeline errors | Failed activity, inputs/outputs, parameters, error code, and child run | +| Identify and resolve Dataflow Gen2 errors | Refresh detail, failing query/step, connector, schema, and destination | +| Identify and resolve notebook errors | Spark application, failed cell/stage/task, driver/executor log, and environment | +| Identify and resolve Eventhouse errors | Ingestion failure, KQL error, table policy, extent, and capacity evidence | +| Identify and resolve Eventstream errors | Source/destination status, event schema, transformation, and throughput | +| Identify and resolve T-SQL errors | Error number/message, query text, permissions, schema, and execution plan | +| Identify and resolve OneLake shortcut errors | Target path, connection, source permission, format/schema, and cache status | -## Source +## A repeatable diagnostic method -See the [official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700). +1. Record time range, workspace/item, run/query ID, identity, and visible symptom. +2. Decide whether one record, one item, one workspace, or the capacity is affected. +3. Find the earliest failing component; downstream errors are often consequences. +4. Classify: authentication/authorization, connectivity, schema/data, expression/code, resource/capacity, concurrency, or transient service. +5. Reproduce with the smallest safe input and preserve the original evidence. +6. Apply a targeted correction, rerun, and add a regression check. + +## Identify and resolve pipeline errors + +Open the run and failed activity. Inspect resolved inputs and outputs rather than only the activity definition. Validate parameter types, dynamic expression results, connection identity, source/sink paths, mappings, and child pipeline/notebook run IDs. + +Use retries for throttling, temporary connectivity, or service unavailability. Fix invalid expressions, missing files, denied permissions, incompatible mappings, and constraint violations. If a ForEach partially succeeds, rerun only when the target is idempotent or the successful subset is excluded. + +## Identify and resolve Dataflow Gen2 errors + +Locate the failed refresh and query. Step through Power Query transformations to find the first error, checking source credentials/privacy, gateway, type conversions, renamed/missing columns, query folding, and destination schema/write settings. + +A preview can succeed on sampled data while refresh fails on an unseen value or full volume. Profile the entire failing column or capture representative bad rows. Explicitly handle conversion errors and schema drift; do not replace all errors with null unless that loss is an approved quality rule. + +## Identify and resolve notebook errors + +Separate driver failures from executor/task failures. A Python stack trace in one cell suggests code or input; repeated task loss can suggest skew, bad records, memory, or infrastructure. Confirm runtime, environment publication, libraries, attached lakehouse, credentials, and Spark configuration. + +For out-of-memory errors, identify whether the driver collected too much data or a partition/task is oversized. Avoid `collect()` on large results, reduce shuffle width, correct skew, prune early, and scale only after measuring. Reproduce with the same runtime and a small failing partition. + +## Identify and resolve Eventhouse errors + +Distinguish ingestion from query. For ingestion, inspect failures by table/source and validate mapping, format, encoding, required fields, identity, retention/caching policies, and capacity. For KQL, keep the request ID, reduce the query, verify names/types, and inspect resource/scanned-data behavior. + +An empty query result is not automatically an ingestion failure: check time filters, event-time parsing, database context, and whether data landed in another table. + +## Identify and resolve Eventstream errors + +Follow the graph from source through transformations to each destination. Check connector state and credentials, incoming/output event rates, schema and type assumptions, malformed events, window time field, and destination health. One failing destination should be distinguished from a source that stopped producing. + +Protect against poison events with validation and a quarantine route. If input exceeds output and lag grows, find the saturated transformation or destination before increasing capacity. + +## Identify and resolve T-SQL errors + +Capture the exact error number/message and statement. Verify database/schema context, object names, permissions, data types, nullability, key constraints, transaction state, and concurrency. For slow rather than failed queries, inspect the actual plan and runtime evidence—do not guess from SQL text alone. + +Common distinctions: “invalid object” is usually context/name/deployment; “permission denied” is identity/grant; truncation/conversion is data/schema; deadlock is concurrency and transaction ordering; timeout can be client, resource, blocking, or plan related. + +## Identify and resolve OneLake shortcut errors + +Verify that the shortcut target still exists, the stored connection is valid, the executing identity has required OneLake/source permissions, the target format is supported in its location, and schema synchronization/caching has not encountered an incompatible change. Deleting a shortcut leaves the target; deleting or moving the target breaks the reference. + +For KQL shortcuts, validate with `external_table('name')`. For lakehouse table shortcuts, confirm the target is a supported Delta table and appears at the correct Tables path. Acceleration has separate status, cache-window, and schema constraints. + +## Exam distinctions + +- Retry transient faults; correct deterministic faults. +- Find the earliest failure, not merely the last red activity. +- Preview success does not prove full Dataflow refresh success. +- Driver memory and executor/task memory are different Spark diagnoses. +- Eventhouse ingestion failure and KQL query failure have different evidence. +- A broken shortcut does not mean the target data was deleted. + +## Active recall + +1. Which pipeline view shows resolved runtime values? +2. Why can Dataflow preview pass while refresh fails? +3. What evidence separates Spark driver failure from executor failure? +4. Input rate is stable but output rate falls—where do you look in an Eventstream? +5. Which shortcut changes can break a formerly valid table? +6. What turns a production incident into a regression test? + +## Authoritative sources + +- [Pipeline troubleshooting guide](https://learn.microsoft.com/en-us/fabric/data-factory/pipeline-troubleshoot-guide) +- [Monitor Dataflow Gen2 refreshes](https://learn.microsoft.com/en-us/fabric/data-factory/dataflows-gen2-monitor) +- [Manage and monitor a KQL database](https://learn.microsoft.com/en-us/fabric/real-time-intelligence/manage-monitor-database) +- [OneLake shortcuts](https://learn.microsoft.com/en-us/fabric/onelake/onelake-shortcuts) +- [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/content/published/dp700/chapters/security-governance.md b/content/published/dp700/chapters/security-governance.md index 5cba37f..ab52efe 100644 --- a/content/published/dp700/chapters/security-governance.md +++ b/content/published/dp700/chapters/security-governance.md @@ -1,14 +1,141 @@ # Configure security and governance - -This chapter will distinguish workspace, item, data, and OneLake security boundaries, then connect them to labels, endorsements, masking, and audit evidence. + +Fabric security is layered. Start with identity, then determine the required scope—workspace, item, or data—and grant the least privilege at that scope. Governance features such as sensitivity labels and endorsement describe and manage data; they do not silently replace authorization. -## Planned study work +## Objective coverage -- Build a role-and-scope decision table. -- Compare row, column, object, and file-level controls. -- Work through governance scenarios without conflating discovery and authorization. +| Measured objective | Core idea | +| --- | --- | +| Implement workspace-level access controls | Roles grant broad capabilities across a collaboration boundary | +| Implement item-level access controls | Sharing and item permissions narrow access without workspace membership | +| Implement row-level, column-level, object-level, and folder/file-level access controls | Data-plane policies constrain what an identity can see or do | +| Implement dynamic data masking | Query results are obscured for principals without unmask privilege | +| Apply sensitivity labels to items | Purview classifications travel through supported inheritance and export paths | +| Endorse items | Promotion, certification, and master-data badges communicate trust | +| Implement and use Microsoft Fabric audit logs | Microsoft Purview audit records activities for investigation and compliance | +| Configure and implement OneLake security | Roles secure OneLake folders and tables independently of broad workspace roles | -## Source +## Implement workspace-level access controls -See the [official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700). +Workspace roles are Admin, Member, Contributor, and Viewer. They apply broadly to items in the workspace and include authoring or administration capabilities according to role. Use security groups rather than many individual assignments, and reserve Admin for principals that truly manage the workspace. + +| Need | Starting point | +| --- | --- | +| Manage access and workspace configuration | Admin | +| Collaborate and manage content without full workspace administration | Member | +| Create and modify content | Contributor | +| Consume workspace content | Viewer, then verify data-plane behavior | + +A role assignment is not the same as a data permission. Fabric evaluates control-plane and data-plane paths, and item types can enforce additional SQL, semantic-model, or OneLake rules. Test the actual access route the user will employ. + +## Implement item-level access controls + +Item sharing grants access to a specific item without placing the recipient in the entire workspace. Use it when a consumer needs a lakehouse, warehouse, report, or other supported item but should not discover or modify unrelated content. Item owners can manage permissions through the item, while workspace roles may already confer broader rights. + +The least-privilege order is useful: first ask whether data can be served through a curated report or model; next consider item sharing; use a workspace role only when collaboration across the workspace is intended. Revoke stale direct shares and prefer groups for durable ownership. + +## Implement row-level, column-level, object-level, and folder/file-level access controls + +These controls protect different dimensions: + +| Control | Restricts | Typical implementation | +| --- | --- | --- | +| Row-level security (RLS) | Which records a principal sees | Predicate/filter based on identity or role | +| Column-level security (CLS) | Which columns can be selected | Grant or deny column permissions | +| Object-level security (OLS) | Whether an object is visible or accessible | Permissions on tables, views, models, or schemas | +| Folder/file-level access | Which OneLake paths can be read or written | OneLake security roles with path/table permissions | + +RLS and CLS are enforcement, while a view is an interface. A view that omits a sensitive column is helpful, but protect the base object so an alternate query path cannot bypass the intended boundary. Evaluate SQL endpoints, Spark, Direct Lake, shortcuts, and OneLake APIs because enforcement support can differ by route. + +Example RLS pattern in a warehouse: + +```sql +CREATE FUNCTION Security.fn_region(@RegionCode char(2)) +RETURNS TABLE WITH SCHEMABINDING AS +RETURN SELECT 1 AS allowed +WHERE @RegionCode = CAST(SESSION_CONTEXT(N'region') AS char(2)); + +CREATE SECURITY POLICY Security.RegionPolicy +ADD FILTER PREDICATE Security.fn_region(RegionCode) ON Sales.OrderFact +WITH (STATE = ON); +``` + +The exact identity mapping must be designed and tested; the predicate alone does not establish a trustworthy user-to-region assignment. + +## Implement dynamic data masking + +Dynamic data masking (DDM) changes query results for users without the `UNMASK` privilege; it does not change stored values. Fabric Warehouse supports default, email, random, and custom-string masks as documented for supported data types. + +```sql +ALTER TABLE dbo.Customer +ALTER COLUMN Email varchar(320) MASKED WITH (FUNCTION = 'email()'); +``` + +DDM reduces accidental exposure in ordinary queries. It is not encryption and is not a strong boundary against a user who can infer values through repeated predicates or joins. Combine it with RLS, CLS, object permissions, and least privilege. + +## Apply sensitivity labels to items + +Sensitivity labels come from Microsoft Purview Information Protection. A tenant administrator enables their use, and labels must be published to the relevant users. Depending on supported item and path, Fabric can apply defaults, require labels, inherit labels downstream, and carry them into supported exports. + +Do not equate a label with universal access control. Labels classify content and can invoke protection in documented cases, but access control is unsupported in other routes and cross-tenant scenarios. Verify the current item and export-path matrix. + +## Endorse items + +Endorsement helps users identify trustworthy assets: + +| Badge | Meaning | Who can apply it | +| --- | --- | --- | +| Promoted | Owner believes the item is ready for reuse | A user with write permission | +| Certified | Authorized reviewer confirms organizational quality standards | Administratively designated certifiers | +| Master data | Data item is an authoritative organizational source | Administratively designated reviewers | + +Endorsement improves discovery and trust signaling. It does not grant permission, encrypt data, or prove that every downstream use is correct. + +## Implement and use Microsoft Fabric audit logs + +Fabric user activities are available through Microsoft Purview Audit. Search by time, user, operation, workload, or item context; export results for a documented investigation when needed. Audit records answer *who performed which recorded action and when*. Diagnostic and workload logs answer different operational questions. + +Establish retention, reviewer roles, alerting criteria, and an evidence-handling procedure before an incident. A missing result can reflect retention, licensing, ingestion delay, filter choice, or an activity not represented by the searched operation—not necessarily proof that nothing happened. + +## Configure and implement OneLake security + +OneLake security roles provide data access for selected tables or folders. A role can include members and define read or read/write permissions at paths, allowing data access without broad workspace authoring rights. Plan roles around job functions and stable groups. + +Evaluate access as a path: + +1. Does the principal have access to the Fabric item or relevant endpoint? +2. Which workspace or item permission applies? +3. Which OneLake security role and path applies? +4. Is a downstream engine enforcing another policy? +5. Does a shortcut introduce source credentials or source-side security? + +Test with representative users, including denied cases. Administrators and item owners can have elevated rights that make their tests misleading. + +## Exam distinctions + +- Workspace roles are broad collaboration grants; item permissions are narrower. +- RLS filters rows; CLS restricts columns; OLS restricts objects; OneLake roles restrict tables or paths. +- DDM changes displayed results, not stored data, and is not encryption. +- Sensitivity labels classify and can protect supported flows; endorsement signals trust. +- Audit logs, OneLake diagnostics, and workload execution logs have different purposes. +- A OneLake shortcut can expose a reference while source-side security and credentials still matter. + +## Active recall + +1. Why is Viewer not a universal statement about every data access path? +2. When is item sharing preferable to workspace membership? +3. Which control hides a column entirely, and which only masks its returned value? +4. What is the difference between Certified and a sensitivity label? +5. Which identities should be used for negative access tests? +6. Where would you investigate a recorded item-deletion activity? + +## Authoritative sources + +- [Fabric permission model](https://learn.microsoft.com/en-us/fabric/security/permission-model) +- [OneLake data access control model](https://learn.microsoft.com/en-us/fabric/onelake/security/data-access-control-model) +- [Dynamic data masking in Fabric Data Warehouse](https://learn.microsoft.com/en-us/fabric/data-warehouse/dynamic-data-masking) +- [Information protection in Fabric](https://learn.microsoft.com/en-us/fabric/governance/information-protection) +- [Endorsement overview](https://learn.microsoft.com/en-us/fabric/governance/endorsement-overview) +- [Track user activities in Microsoft Fabric](https://learn.microsoft.com/en-us/fabric/admin/track-user-activities) +- [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/content/published/dp700/chapters/streaming-data.md b/content/published/dp700/chapters/streaming-data.md index 4fccc60..4c40ee1 100644 --- a/content/published/dp700/chapters/streaming-data.md +++ b/content/published/dp700/chapters/streaming-data.md @@ -1,14 +1,135 @@ # Ingest and transform streaming data - -This chapter will compare streaming engines, Real-Time Intelligence storage choices, Eventstreams, Spark structured streaming, KQL, and window semantics. + +Streaming systems trade bounded batch simplicity for continuous state. Choose the engine from latency, event volume, transformation complexity, operational ownership, and serving/query needs; then make time and failure semantics explicit. -## Planned study work +## Objective coverage -- Select an engine from latency and processing requirements. -- Compare native tables with standard and accelerated shortcuts. -- Reason about tumbling, hopping, and sliding windows. +| Measured objective | What to master | +| --- | --- | +| Choose an appropriate streaming engine | Eventstreams, Eventhouse/KQL, Spark Structured Streaming, and pipeline boundaries | +| Choose between native tables and OneLake shortcuts in Real-Time Intelligence | Ingest/index versus query data in place | +| Choose between Query acceleration for OneLake shortcuts and standard OneLake shortcuts in Real-Time Intelligence | Cache window, performance, freshness, cost, and limitations | +| Process data by using Eventstreams | Sources, transformations, derived streams, routing, and destinations | +| Process data by using Spark structured streaming | Sources/sinks, checkpoints, output modes, watermarks, and state | +| Process data by using KQL | Filtering, parsing, summarizing, joining, and time-series operations | +| Create windowing functions | Tumbling, hopping/sliding, and session windows with event time | -## Source +## Choose an appropriate streaming engine -See the [official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700). +| Engine | Choose it for | Main trade-off | +| --- | --- | --- | +| Eventstreams | No-code/low-code ingestion, filtering, transformation, and routing | Less arbitrary code than Spark | +| Eventhouse and KQL | High-rate event storage and low-latency time-series/log queries | Optimized for event analytics rather than general ETL | +| Spark Structured Streaming | Code-first, complex/stateful processing with Delta integration | Greater engineering and state-management responsibility | +| Pipeline | Orchestrating bounded jobs around a stream | Not the continuous event-processing engine itself | + +Use Eventstreams to connect and route; Eventhouse to retain and query time-oriented data; Spark when custom algorithms, libraries, or complex state are required. They can form one design rather than mutually exclusive choices. + +## Choose between native tables and OneLake shortcuts in Real-Time Intelligence + +A native Eventhouse table ingests data into Eventhouse-managed, indexed storage. It gives predictable KQL performance and supports table policies, at the cost of ingestion and another managed representation. A OneLake shortcut exposes Delta data as an external table through `external_table()` without moving it. + +Choose native ingestion for consistently low-latency, high-concurrency event queries and Eventhouse policy features. Choose a standard shortcut when data already lives in OneLake/open storage, freshness through the source is acceptable, and avoiding movement matters more than native query speed. + +## Choose between Query acceleration for OneLake shortcuts and standard OneLake shortcuts in Real-Time Intelligence + +Query acceleration caches a configured recent period of shortcut Delta data and builds structures that approach native-table query performance. It is useful for hot recent data or joins between historical OneLake data and live Eventhouse events. + +| Choice | Performance | Storage/cost | Constraints | +| --- | --- | --- | --- | +| Standard shortcut | Reads external Delta at query time | Avoids accelerated cache | Network/file layout and lack of indexes affect latency | +| Accelerated shortcut | Caches a time window for faster queries | Premium cache and indexing consume resources | External-table limitations remain; schema/policy constraints apply | + +Set the cache period from the actual query horizon. Acceleration is not a full ingestion conversion: accelerated external tables still do not support every native-table feature, including documented materialized-view and update-policy scenarios. + +## Process data by using Eventstreams + +An Eventstream connects event sources, applies no-code operations, and sends results to one or more destinations. Typical operations include filtering, field management, aggregation, group-by, union, expansion, and content-based routing. A **derived stream** makes a transformed branch reusable. + +Design steps: + +1. Define event schema, ID, event-time field, and expected rate. +2. Select and secure the source connector. +3. Filter early and normalize types/names. +4. Apply windowed aggregation or routing only after time semantics are clear. +5. Route raw and curated branches to their respective destinations. +6. Monitor input, output, invalid events, lag, and destination errors. + +## Process data by using Spark structured streaming + +Spark treats a stream as an incrementally updated table. A durable checkpoint stores progress and state so a query can recover. Each production query needs its own stable checkpoint path; deleting or reusing it changes recovery semantics. + +```python +from pyspark.sql import functions as F + +events = spark.readStream.format("delta").table("bronze_events") + +windowed = ( + events.withWatermark("event_time", "10 minutes") + .dropDuplicatesWithinWatermark(["event_id"]) + .groupBy(F.window("event_time", "5 minutes"), "device_id") + .agg(F.avg("temperature").alias("avg_temperature")) +) + +( + windowed.writeStream.format("delta") + .outputMode("append") + .option("checkpointLocation", "Files/checkpoints/device_5m") + .toTable("silver_device_windows") +) +``` + +Output modes describe emitted results: append emits final new rows where supported, update emits changed rows, and complete emits the full result table. The sink must support the chosen mode. `foreachBatch` enables batch logic per micro-batch, but that logic must use the batch ID or stable keys to remain idempotent. + +## Process data by using KQL + +KQL is a pipeline language: each operator receives a tabular result and passes another result forward. + +```kusto +DeviceEvents +| where EventTime > ago(1h) +| where isnotnull(Temperature) +| summarize AvgTemperature=avg(Temperature), Events=count() + by DeviceId, bin(EventTime, 5m) +| where AvgTemperature > 80 +| order by EventTime desc +``` + +Filter early, select only needed columns, parse dynamic fields deliberately, and use time-bounded joins. `summarize ... by bin()` forms fixed time buckets; time-series functions and materialized views serve repeated analytical patterns where supported. + +## Create windowing functions + +| Window | Behavior | Example use | +| --- | --- | --- | +| Tumbling | Fixed, nonoverlapping intervals | Five-minute totals | +| Hopping/sliding | Fixed width, starts more frequently than width | Ten-minute moving measure every minute | +| Session | Variable window closes after inactivity gap | User/device activity sessions | + +Windows should use event time when results describe when events happened. A watermark states how late the engine expects events and bounds retained state; it does not reorder the entire infinite stream or guarantee that later records will be accepted. + +## Exam distinctions + +- Native Eventhouse tables ingest and index; shortcuts query supported data in place. +- Query acceleration caches a shortcut window but does not turn it into a fully native table. +- Event time describes occurrence; processing time describes observation by the engine. +- A watermark bounds late-data state; a checkpoint supports restart/recovery. +- Tumbling windows do not overlap; hopping windows can; session windows depend on inactivity. +- Eventstreams route/shape events; Activator evaluates conditions and takes actions. + +## Active recall + +1. Why might an accelerated shortcut still be the wrong choice for an update policy? +2. Which engine would you choose for custom stateful correlation, and why? +3. What happens when a checkpoint directory is discarded? +4. Compare a ten-minute tumbling window with a ten-minute window sliding every minute. +5. Why should a streaming join be time bounded? +6. Which metrics reveal backpressure or destination failure? + +## Authoritative sources + +- [Real-Time Intelligence overview](https://learn.microsoft.com/en-us/fabric/real-time-intelligence/overview) +- [OneLake shortcuts in a KQL database](https://learn.microsoft.com/en-us/fabric/real-time-intelligence/onelake-shortcuts) +- [Query acceleration overview](https://learn.microsoft.com/en-us/fabric/real-time-intelligence/query-acceleration-overview) +- [Stateful processing with Structured Streaming](https://learn.microsoft.com/en-us/fabric/data-engineering/structured-streaming-stateful-processing) +- [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/tests/content/test_dp700_outline.py b/tests/content/test_dp700_outline.py index 8983eb1..1ceeef4 100644 --- a/tests/content/test_dp700_outline.py +++ b/tests/content/test_dp700_outline.py @@ -41,27 +41,45 @@ def test_every_dp700_chapter_maps_objectives_sources_and_content() -> None: def test_source_registry_records_current_learn_provenance() -> None: book = BookCatalog(CONTENT_ROOT).load_book("dp700") - assert len(book.sources) == 9 + assert len(book.sources) >= 30 assert {source.url.host for source in book.sources} == {"learn.microsoft.com"} assert all(source.retrieved_at.tzinfo is not None for source in book.sources) assert all(len(source.content_sha256) == 64 for source in book.sources) assert set(book.sources[0].chapter_ids) == {chapter.id for chapter in book.chapters} -def test_representative_chapter_is_published_and_directly_sourced() -> None: +def test_every_chapter_is_published_substantive_and_directly_sourced() -> None: + catalog = BookCatalog(CONTENT_ROOT) + book = catalog.load_book("dp700") + + for chapter in book.chapters: + markdown = catalog.load_chapter_markdown(book, chapter) + + assert chapter.status == "published" + assert len(chapter.source_ids) >= 2 + assert len(markdown.split()) >= 700 + assert all(objective.title in markdown for objective in chapter.objectives) + assert "## Objective coverage" in markdown + assert "## Exam distinctions" in markdown + assert "## Active recall" in markdown + assert ( + "## Authoritative sources" in markdown + or "## Sources and provenance" in markdown + ) + assert "Planned study work" not in markdown + assert "placeholder" not in markdown.lower() + + +def test_workspace_settings_retains_representative_technical_depth() -> None: catalog = BookCatalog(CONTENT_ROOT) book = catalog.load_book("dp700") chapter = book.chapter_by_slug("workspace-settings") markdown = catalog.load_chapter_markdown(book, chapter) - assert chapter.status == "published" assert len(chapter.source_ids) == 9 - assert all(objective.title in markdown for objective in chapter.objectives) assert "## Configure Spark workspace settings" in markdown assert "## Configure domain workspace settings" in markdown assert "## What the OneLake settings control" in markdown assert "## Configure Apache Airflow workspace settings" in markdown assert "## Responsibility and prerequisite boundaries" in markdown - assert "## Exam distinctions" in markdown - assert "## Active recall" in markdown assert "OneLake.Read.All" in markdown From 6ad4ebe5b2231b7b1b801589145ec4bfa368487b Mon Sep 17 00:00:00 2001 From: Troy Scott <846218+troyscott@users.noreply.github.com> Date: Wed, 2 Sep 2026 04:02:32 -0700 Subject: [PATCH 2/6] Add objective coverage audit contract --- content/published/dp700/book.yaml | 4 +- content/published/dp700/coverage.yaml | 327 ++++++++++++++++++++++++++ docs/content-model.md | 34 ++- src/study_reader/content/catalog.py | 13 + src/study_reader/content/coverage.py | 116 +++++++++ tests/content/test_dp700_outline.py | 15 ++ tests/unit/test_models.py | 53 +++++ 7 files changed, 559 insertions(+), 3 deletions(-) create mode 100644 content/published/dp700/coverage.yaml create mode 100644 src/study_reader/content/coverage.py diff --git a/content/published/dp700/book.yaml b/content/published/dp700/book.yaml index f34516c..62e64d4 100644 --- a/content/published/dp700/book.yaml +++ b/content/published/dp700/book.yaml @@ -123,7 +123,7 @@ sources: url: https://learn.microsoft.com/en-us/fabric/data-warehouse/dimensional-modeling-load-tables retrieved_at: 2026-09-02T06:52:59Z content_sha256: e6148d2232549d19b1788f733b54b66c7fe7bb486d93a8525e7cab358bbb62f3 - chapter_ids: [loading-patterns] + chapter_ids: [loading-patterns, batch-data] - id: dataflows-gen2-overview title: Dataflows Gen2 overview url: https://learn.microsoft.com/en-us/fabric/data-factory/dataflows-gen2-overview @@ -284,7 +284,7 @@ domains: title: Ingest and transform batch data content_path: batch-data.md status: published - source_ids: [dp700-study-guide, data-movement-decision-guide, dataflows-gen2-overview, onelake-shortcuts, fabric-mirroring] + source_ids: [dp700-study-guide, data-movement-decision-guide, dimensional-model-loading, dataflows-gen2-overview, onelake-shortcuts, fabric-mirroring] objectives: - {id: ingest.batch.store, title: Choose an appropriate data store} - {id: ingest.batch.transform-tool, title: "Choose between Dataflows Gen2, notebooks, KQL, and T-SQL for data transformation"} diff --git a/content/published/dp700/coverage.yaml b/content/published/dp700/coverage.yaml new file mode 100644 index 0000000..9d60bed --- /dev/null +++ b/content/published/dp700/coverage.yaml @@ -0,0 +1,327 @@ +book_id: dp700 +blueprint_effective_date: 2026-07-21 +objectives: + - objective_id: implement.workspace.spark + chapter_id: workspace-settings + status: draft + subtopics: [Workspace defaults, Environment overrides, Runtimes, Pools, Driver and executor sizing, Session configuration] + source_ids: [dp700-study-guide, spark-compute-settings] + gaps: [Add a hands-on configuration lab and objective-specific evidence blocks] + - objective_id: implement.workspace.domain + chapter_id: workspace-settings + status: draft + subtopics: [Domain hierarchy, Workspace assignment, Domain roles, Default domains, Delegated governance, Catalog discovery] + source_ids: [dp700-study-guide, fabric-domains] + gaps: [Add a domain-assignment lab and negative authorization test] + - objective_id: implement.workspace.onelake + chapter_id: workspace-settings + status: draft + subtopics: [Diagnostics, Immutability, Storage tiers, Lifecycle rules, REST API, Permissions and placement] + source_ids: [dp700-study-guide, onelake-overview, onelake-diagnostics, onelake-storage-tiers, onelake-lifecycle, onelake-settings-api] + gaps: [Add an end-to-end diagnostics and lifecycle scenario] + - objective_id: implement.workspace.airflow + chapter_id: workspace-settings + status: draft + subtopics: [Workspace runtime, Starter pools, Custom pools, Sizing, Autoscaling, Concurrency and limitations] + source_ids: [dp700-study-guide, airflow-workspace-settings] + gaps: [Add configuration procedure and pool-sizing lab] + - objective_id: implement.lifecycle.version-control + chapter_id: lifecycle-management + status: draft + subtopics: [Git providers, Workspace connection, Branches, Commit and update direction, Conflicts, Supported items, Secrets] + source_ids: [dp700-study-guide, fabric-cicd-overview] + gaps: [Add portal workflow, conflict exercise, and supported-item limitations] + - objective_id: implement.lifecycle.database-projects + chapter_id: lifecycle-management + status: draft + subtopics: [Declarative schema, Project structure, Build validation, DACPAC, Publish plan, State versus migration, Data-loss safeguards] + source_ids: [dp700-study-guide, fabric-cicd-overview] + gaps: [Add direct database-project sources and a build-deploy walkthrough] + - objective_id: implement.lifecycle.deployment-pipelines + chapter_id: lifecycle-management + status: draft + subtopics: [Stages, Workspace assignment, Item pairing, Comparison, Deployment rules, Dependency binding, Post-deployment checks] + source_ids: [dp700-study-guide, fabric-cicd-overview] + gaps: [Add current pipeline configuration procedure and failed-binding scenario] + - objective_id: implement.security.workspace-access + chapter_id: security-governance + status: draft + subtopics: [Admin, Member, Contributor, Viewer, Group assignment, Least privilege, Workspace scope] + source_ids: [dp700-study-guide, fabric-permission-model] + gaps: [Add complete capability matrix and representative access tests] + - objective_id: implement.security.item-access + chapter_id: security-governance + status: draft + subtopics: [Item sharing, Direct permissions, Read and reshare, Ownership, Group-based grants, Revocation] + source_ids: [dp700-study-guide, fabric-permission-model] + gaps: [Add item-type examples and portal/API procedure] + - objective_id: implement.security.data-access + chapter_id: security-governance + status: draft + subtopics: [Row-level security, Column-level security, Object-level security, Folder and file access, Engine enforcement paths, Bypass testing] + source_ids: [dp700-study-guide, fabric-permission-model, onelake-data-access-control] + gaps: [Add implementation examples for every control and cross-engine enforcement matrix] + - objective_id: implement.security.masking + chapter_id: security-governance + status: draft + subtopics: [Mask functions, UNMASK permission, Stored versus returned values, Inference risk, Combining DDM with RLS and CLS] + source_ids: [dp700-study-guide, warehouse-dynamic-masking] + gaps: [Add complete mask examples, grants, and negative tests] + - objective_id: implement.security.sensitivity + chapter_id: security-governance + status: draft + subtopics: [Purview labels, Tenant enablement, Label publishing, Defaults, Mandatory labels, Inheritance, Export behavior] + source_ids: [dp700-study-guide, fabric-information-protection] + gaps: [Add application procedure, supported-item matrix, and inheritance scenario] + - objective_id: implement.security.endorsement + chapter_id: security-governance + status: draft + subtopics: [Promoted, Certified, Master data, Authorized reviewers, Discovery, Trust versus authorization] + source_ids: [dp700-study-guide, fabric-endorsement] + gaps: [Add endorsement workflow and governance scenario] + - objective_id: implement.security.audit + chapter_id: security-governance + status: draft + subtopics: [Purview Audit, Search filters, Operations, Retention, Export, Investigation procedure, Audit versus diagnostic logs] + source_ids: [dp700-study-guide, fabric-audit-activities] + gaps: [Add investigation walkthrough and evidence-handling lab] + - objective_id: implement.security.onelake + chapter_id: security-governance + status: draft + subtopics: [OneLake roles, Tables and folders, Read and write grants, Role membership, Shortcut security, Cross-engine access paths] + source_ids: [dp700-study-guide, onelake-data-access-control] + gaps: [Add role-creation procedure, API example, and denied-access tests] + - objective_id: implement.orchestration.choose-tool + chapter_id: orchestration + status: draft + subtopics: [Dataflow Gen2, Pipeline, Notebook, Transformation versus orchestration, Personas, Scale and maintainability] + source_ids: [dp700-study-guide, pipeline-overview] + gaps: [Add representative tool-selection cases and cost/operability trade-offs] + - objective_id: implement.orchestration.triggers + chapter_id: orchestration + status: draft + subtopics: [Schedules, Time zones, Event triggers, File events, Filtering, Concurrency, Idempotency, Reconciliation] + source_ids: [dp700-study-guide, pipeline-overview] + gaps: [Add trigger configuration procedures and duplicate-event lab] + - objective_id: implement.orchestration.patterns + chapter_id: orchestration + status: draft + subtopics: [Parameters, Variables, Dynamic expressions, Parent-child pipelines, Metadata-driven loops, Dependencies, Retry and failure paths] + source_ids: [dp700-study-guide, pipeline-overview, pipeline-parameters] + gaps: [Add executable expression catalog and end-to-end pipeline/notebook example] + - objective_id: ingest.loading.full-incremental + chapter_id: loading-patterns + status: draft + subtopics: [Full refresh, Watermarks, CDC, Overlap windows, Upserts, Deletes, Idempotency, Reconciliation] + source_ids: [dp700-study-guide, data-movement-decision-guide] + gaps: [Add Fabric pipeline and notebook implementations with failure recovery] + - objective_id: ingest.loading.dimensional + chapter_id: loading-patterns + status: draft + subtopics: [Grain, Surrogate keys, Dimension-first loading, Facts, SCD Type 1, SCD Type 2, Inferred members, Referential checks] + source_ids: [dp700-study-guide, dimensional-model-loading] + gaps: [Add complete dimension/fact load example and late-member lab] + - objective_id: ingest.loading.streaming + chapter_id: loading-patterns + status: draft + subtopics: [Raw retention, Schema validation, Deduplication, Event time, Watermarks, Checkpoints, Idempotent sinks, Reconciliation] + source_ids: [dp700-study-guide, structured-streaming-state] + gaps: [Add Fabric architecture, checkpoint recovery, and replay scenario] + - objective_id: ingest.batch.store + chapter_id: batch-data + status: draft + subtopics: [Lakehouse, Warehouse, Eventhouse, Operational database, Open formats, Transactions, Concurrency, Serving patterns] + source_ids: [dp700-study-guide, data-movement-decision-guide] + gaps: [Add workload decision matrix and migration scenarios] + - objective_id: ingest.batch.transform-tool + chapter_id: batch-data + status: draft + subtopics: [Dataflow Gen2, Notebooks, T-SQL, KQL, Personas, Engine locality, Scale, Testing and maintainability] + source_ids: [dp700-study-guide, dataflows-gen2-overview] + gaps: [Add detailed tool-comparison cases and equivalent transformations] + - objective_id: ingest.batch.shortcuts + chapter_id: batch-data + status: draft + subtopics: [Internal shortcuts, External shortcuts, Tables versus Files, Connections, Caching, Schema synchronization, Security, Deletion behavior] + source_ids: [dp700-study-guide, onelake-shortcuts] + gaps: [Add creation and management procedures plus broken-target lab] + - objective_id: ingest.batch.mirroring + chapter_id: batch-data + status: draft + subtopics: [Database mirroring, Metadata mirroring, Open mirroring, Replication, Source support, Read-only targets, Monitoring] + source_ids: [dp700-study-guide, fabric-mirroring] + gaps: [Add implementation workflow, selection matrix, and failure recovery] + - objective_id: ingest.batch.pipelines + chapter_id: batch-data + status: draft + subtopics: [Copy Activity, Connectors, Mapping, Parameters, Parallelism, Staging, Retry, Metrics and rejects] + source_ids: [dp700-study-guide, data-movement-decision-guide] + gaps: [Add parameterized copy walkthrough and performance exercise] + - objective_id: ingest.batch.languages + chapter_id: batch-data + status: draft + subtopics: [PySpark DataFrames, Spark SQL, T-SQL, KQL pipelines, Types, Null behavior, Engine-specific optimization] + source_ids: [dp700-study-guide, dataflows-gen2-overview] + gaps: [Add multi-step equivalent transformations and validation outputs] + - objective_id: ingest.batch.denormalize + chapter_id: batch-data + status: draft + subtopics: [Output grain, Join cardinality, Flattening, Repeated attributes, Performance trade-offs, Validation] + source_ids: [dp700-study-guide, dimensional-model-loading] + gaps: [Map dimensional source to chapter and add worked denormalization example] + - objective_id: ingest.batch.aggregate + chapter_id: batch-data + status: draft + subtopics: [Grouping keys, Grain changes, Additive measures, Semi-additive measures, Ratios, Nulls, Incremental aggregates] + source_ids: [dp700-study-guide, dataflows-gen2-overview] + gaps: [Add Power Query, SQL, PySpark, and KQL aggregation examples] + - objective_id: ingest.batch.data-quality + chapter_id: batch-data + status: draft + subtopics: [Duplicate keys, Missing values, Invalid values, Late records, Quarantine, Imputation, Restatement, Quality metrics] + source_ids: [dp700-study-guide, dataflows-gen2-overview] + gaps: [Add executable quality rules and late-arrival correction lab] + - objective_id: ingest.streaming.engine + chapter_id: streaming-data + status: draft + subtopics: [Eventstreams, Eventhouse, Spark Structured Streaming, Latency, Volume, State, Serving and operations] + source_ids: [dp700-study-guide, real-time-intelligence-overview, structured-streaming-state] + gaps: [Add decision cases and end-to-end architecture comparison] + - objective_id: ingest.streaming.native-shortcut + chapter_id: streaming-data + status: draft + subtopics: [Native ingestion, Indexed storage, External Delta, Feature support, Latency, Cost, Source availability] + source_ids: [dp700-study-guide, real-time-intelligence-overview] + gaps: [Add direct shortcut source mapping and measured selection scenario] + - objective_id: ingest.streaming.acceleration + chapter_id: streaming-data + status: draft + subtopics: [Standard shortcut, Accelerated cache, Cache period, Performance, Billing, Schema changes, External-table limitations] + source_ids: [dp700-study-guide, query-acceleration-overview] + gaps: [Add configuration procedure, monitoring commands, and cost/performance lab] + - objective_id: ingest.streaming.eventstreams + chapter_id: streaming-data + status: draft + subtopics: [Sources, Transformations, Filtering, Aggregation, Derived streams, Routing, Destinations, Monitoring] + source_ids: [dp700-study-guide, real-time-intelligence-overview] + gaps: [Add full Eventstream construction walkthrough and poison-event route] + - objective_id: ingest.streaming.spark + chapter_id: streaming-data + status: draft + subtopics: [Sources, Sinks, Checkpoints, Output modes, foreachBatch, Stateful operations, Recovery, Observability] + source_ids: [dp700-study-guide, structured-streaming-state] + gaps: [Add runnable Fabric notebook pattern and checkpoint recovery lab] + - objective_id: ingest.streaming.kql + chapter_id: streaming-data + status: draft + subtopics: [Filtering, Projection, Dynamic parsing, Summarize, Joins, Time series, Materialization, Query optimization] + source_ids: [dp700-study-guide, real-time-intelligence-overview] + gaps: [Add complete KQL transformation sequence and validation scenario] + - objective_id: ingest.streaming.windows + chapter_id: streaming-data + status: draft + subtopics: [Event time, Tumbling windows, Hopping windows, Sliding windows, Session windows, Watermarks, Late events] + source_ids: [dp700-study-guide, structured-streaming-state] + gaps: [Add Eventstream, Spark, and KQL window examples with expected results] + - objective_id: monitor.items.ingestion + chapter_id: monitor-items + status: draft + subtopics: [Monitoring hub, Pipeline runs, Copy metrics, Streaming rate, Lag, Watermark, Rejected rows, Freshness] + source_ids: [dp700-study-guide, fabric-monitoring-hub] + gaps: [Add item-specific screens, thresholds, and incident scenario] + - objective_id: monitor.items.transformation + chapter_id: monitor-items + status: draft + subtopics: [Dataflow refresh, Notebook and Spark jobs, Pipeline activities, SQL and KQL queries, Quality assertions, Capacity evidence] + source_ids: [dp700-study-guide, fabric-monitoring-hub, dataflow-monitoring] + gaps: [Add per-engine diagnostic workflow and data-quality evidence model] + - objective_id: monitor.items.semantic-refresh + chapter_id: monitor-items + status: draft + subtopics: [Refresh history, Administrative summary, Types, Tables and partitions, Incremental refresh, Credentials, Gateway, Capacity] + source_ids: [dp700-study-guide, semantic-refresh-summaries] + gaps: [Add refresh investigation walkthrough and upstream freshness chain] + - objective_id: monitor.items.alerts + chapter_id: monitor-items + status: draft + subtopics: [Activator, Signals, Thresholds, Windows, Recipients, Actions, Suppression, Recovery, Runbooks] + source_ids: [dp700-study-guide, fabric-activator] + gaps: [Add alert configuration procedure and firing/recovery test] + - objective_id: monitor.errors.pipeline + chapter_id: resolve-errors + status: draft + subtopics: [Failed activity, Resolved inputs, Error codes, Connections, Expressions, Child runs, Retry classification, Idempotent reruns] + source_ids: [dp700-study-guide, pipeline-troubleshooting] + gaps: [Add representative failure cases and resolution lab] + - objective_id: monitor.errors.dataflow + chapter_id: resolve-errors + status: draft + subtopics: [Refresh detail, Power Query step, Credentials, Gateway, Type errors, Schema drift, Folding, Destination errors] + source_ids: [dp700-study-guide, dataflow-monitoring] + gaps: [Add bad-row diagnosis, schema failure, and destination repair examples] + - objective_id: monitor.errors.notebook + chapter_id: resolve-errors + status: draft + subtopics: [Driver errors, Executor failures, Spark UI, Runtime, Libraries, Lakehouse attachment, Memory, Skew and bad records] + source_ids: [dp700-study-guide] + gaps: [Add direct notebook troubleshooting sources and Spark failure labs] + - objective_id: monitor.errors.eventhouse + chapter_id: resolve-errors + status: draft + subtopics: [Ingestion failures, Mapping and format, Permissions, Table policies, KQL request IDs, Capacity, Empty results] + source_ids: [dp700-study-guide] + gaps: [Add direct Eventhouse troubleshooting sources and commands] + - objective_id: monitor.errors.eventstream + chapter_id: resolve-errors + status: draft + subtopics: [Connector state, Source rate, Transform schema, Event time, Destination health, Poison events, Backpressure] + source_ids: [dp700-study-guide] + gaps: [Add direct Eventstream troubleshooting sources and failure-routing lab] + - objective_id: monitor.errors.tsql + chapter_id: resolve-errors + status: draft + subtopics: [Error numbers, Context and names, Permissions, Conversions, Constraints, Transactions, Deadlocks, Timeouts and plans] + source_ids: [dp700-study-guide] + gaps: [Add Fabric-specific T-SQL troubleshooting sources and worked cases] + - objective_id: monitor.errors.shortcut + chapter_id: resolve-errors + status: draft + subtopics: [Target existence, Connections, Source permissions, Format, Tables path, Schema sync, Cache and acceleration] + source_ids: [dp700-study-guide, onelake-shortcuts] + gaps: [Add systematic shortcut diagnostic procedure and repair lab] + - objective_id: monitor.optimize.lakehouse + chapter_id: optimize-performance + status: draft + subtopics: [Small files, OPTIMIZE, V-Order, Partitioning, Statistics, VACUUM, Retention, Maintenance cadence] + source_ids: [dp700-study-guide, lakehouse-delta-tables, delta-v-order] + gaps: [Add measurement-based before/after lab and safety checks] + - objective_id: monitor.optimize.pipeline + chapter_id: optimize-performance + status: draft + subtopics: [Source selection, Copy parallelism, ForEach concurrency, Staging, Activity overhead, Retry backoff, Throughput metrics] + source_ids: [dp700-study-guide] + gaps: [Add direct performance sources and measured tuning scenario] + - objective_id: monitor.optimize.warehouse + chapter_id: optimize-performance + status: draft + subtopics: [Scans, Statistics, Data types, Star schema, Joins, Plans, Concurrency, V-Order and materialization] + source_ids: [dp700-study-guide, warehouse-performance] + gaps: [Add execution-plan walkthrough and before/after query lab] + - objective_id: monitor.optimize.realtime + chapter_id: optimize-performance + status: draft + subtopics: [Eventstream filtering, Routing, Ingestion batching, Retention, Caching, KQL shape, Materialized views, Shortcut acceleration] + source_ids: [dp700-study-guide, kql-query-best-practices] + gaps: [Add Eventhouse policy sources and measured streaming scenario] + - objective_id: monitor.optimize.spark + chapter_id: optimize-performance + status: draft + subtopics: [Spark UI, Partition pruning, Shuffle, Skew, Broadcast joins, Repartition, Coalesce, Cache, Compute sizing] + source_ids: [dp700-study-guide, lakehouse-delta-tables] + gaps: [Add Fabric Spark tuning sources and evidence-driven lab] + - objective_id: monitor.optimize.query + chapter_id: optimize-performance + status: draft + subtopics: [Projection, Predicate pushdown, Pruning, Join grain, Plans, Repeated computation, Preaggregation, Correctness and cost] + source_ids: [dp700-study-guide, warehouse-performance, kql-query-best-practices] + gaps: [Add cross-engine query examples with plans and measured results] diff --git a/docs/content-model.md b/docs/content-model.md index 0b2d73b..78c8409 100644 --- a/docs/content-model.md +++ b/docs/content-model.md @@ -75,11 +75,43 @@ grouping, shaping, schema handling, and query folding—along with destinations and tool-selection trade-offs. Merely listing “Dataflow Gen2” does not satisfy the objective. +### Comprehensive coverage audit + +Each published book MUST include a `coverage.yaml` file with exactly one record +for every measured objective in `book.yaml`. The record identifies the owning +chapter, the objective's required subtopics, its authoritative sources, known +gaps, and durable Markdown block IDs that provide evidence for the approved +teaching rubric. + +An objective MAY move through `planned`, `draft`, and `complete` states. A +non-complete objective MUST state its remaining gaps. An objective MUST NOT be +marked `complete` unless it has no remaining gaps and identifies durable blocks +for all of these evidence categories: + +- conceptual model and terminology; +- prerequisites, responsibilities, and security boundaries; +- procedure or operational workflow; +- decision guidance and trade-offs; +- a worked example; +- limitations and failure modes; +- troubleshooting or monitoring guidance; +- exam distinctions; +- at least two active-recall questions; and +- a scenario or mini-lab. + +One strong block MAY support more than one category or objective when its +content genuinely supplies that evidence. The evidence mapping exists to make +editorial review inspectable; it MUST NOT be satisfied with empty headings, +duplicated boilerplate, or identifiers that are absent from the chapter. + ## Validation Pydantic rejects duplicate identifiers, duplicate slugs, non-Learn source hosts, malformed hashes, one-way source mappings, broken source references, invalid paths, and unknown manifest fields. Content tests confirm that every -planned chapter file exists and maps at least one objective and one source. +planned chapter file exists and maps at least one objective and one source. The +coverage audit additionally rejects missing or extra objectives, incorrect +chapter ownership, invalid source mappings, unsupported completion claims, and +incomplete objectives that conceal their remaining gaps. Rendering tests verify sanitized HTML, stable anchors, external-link behavior, and a reviewed golden fixture. diff --git a/src/study_reader/content/catalog.py b/src/study_reader/content/catalog.py index ac2c42c..d093749 100644 --- a/src/study_reader/content/catalog.py +++ b/src/study_reader/content/catalog.py @@ -5,6 +5,7 @@ import yaml +from study_reader.content.coverage import CoverageAudit from study_reader.content.models import Book, Chapter SAFE_ID = re.compile(r"^[a-z0-9][a-z0-9-]*$") @@ -33,6 +34,18 @@ def chapter_path(self, book: Book, chapter: Chapter) -> Path: return self.root / book.id / "chapters" / chapter.content_path + def load_coverage_audit(self, book: Book) -> CoverageAudit: + """Load and validate the objective-level content evidence contract.""" + + audit_path = self.root / book.id / "coverage.yaml" + try: + raw_audit = yaml.safe_load(audit_path.read_text(encoding="utf-8")) + except (OSError, yaml.YAMLError) as error: + raise FileNotFoundError(audit_path) from error + audit = CoverageAudit.model_validate(raw_audit) + audit.validate_against(book) + return audit + def load_chapter_markdown(self, book: Book, chapter: Chapter) -> str: """Read one authored chapter from the published snapshot.""" diff --git a/src/study_reader/content/coverage.py b/src/study_reader/content/coverage.py new file mode 100644 index 0000000..6f32ccf --- /dev/null +++ b/src/study_reader/content/coverage.py @@ -0,0 +1,116 @@ +"""Objective-level evidence contract for comprehensive study content.""" + +from datetime import date +from typing import Literal, Self + +from pydantic import BaseModel, ConfigDict, Field, model_validator + +from study_reader.content.models import Book + +BLOCK_ID = r"^[a-z0-9][a-z0-9-]*$" +OBJECTIVE_ID = r"^[a-z0-9][a-z0-9.-]*$" + + +class CoverageModel(BaseModel): + """Strict immutable base for coverage records.""" + + model_config = ConfigDict(extra="forbid", frozen=True) + + +class CoverageEvidence(CoverageModel): + """Durable Markdown blocks that prove an objective's teaching depth.""" + + conceptual_model: str | None = Field(default=None, pattern=BLOCK_ID) + prerequisites: str | None = Field(default=None, pattern=BLOCK_ID) + procedure: str | None = Field(default=None, pattern=BLOCK_ID) + decision_guidance: str | None = Field(default=None, pattern=BLOCK_ID) + worked_example: str | None = Field(default=None, pattern=BLOCK_ID) + limitations: str | None = Field(default=None, pattern=BLOCK_ID) + troubleshooting: str | None = Field(default=None, pattern=BLOCK_ID) + exam_distinctions: str | None = Field(default=None, pattern=BLOCK_ID) + recall_questions: str | None = Field(default=None, pattern=BLOCK_ID) + scenario_lab: str | None = Field(default=None, pattern=BLOCK_ID) + + @property + def block_ids(self) -> tuple[str, ...]: + """Return all recorded evidence block identifiers.""" + + return tuple( + value + for name in type(self).model_fields + if (value := getattr(self, name)) is not None + ) + + @property + def is_complete(self) -> bool: + """Return whether every rubric category has durable evidence.""" + + return len(self.block_ids) == len(type(self).model_fields) + + +class ObjectiveCoverage(CoverageModel): + """Coverage plan and evidence for one measured objective.""" + + objective_id: str = Field(pattern=OBJECTIVE_ID) + chapter_id: str = Field(pattern=BLOCK_ID) + status: Literal["planned", "draft", "complete"] + subtopics: tuple[str, ...] = Field(min_length=2) + source_ids: tuple[str, ...] = Field(min_length=1) + gaps: tuple[str, ...] = () + evidence: CoverageEvidence = Field(default_factory=CoverageEvidence) + + @model_validator(mode="after") + def validate_status(self) -> Self: + """Make completion a claim backed by every rubric category.""" + + if self.status == "complete" and not self.evidence.is_complete: + raise ValueError("complete objectives require every evidence category") + if self.status == "complete" and self.gaps: + raise ValueError("complete objectives cannot retain coverage gaps") + if self.status != "complete" and not self.gaps: + raise ValueError("incomplete objectives must record coverage gaps") + return self + + +class CoverageAudit(CoverageModel): + """Book-wide, objective-level comprehensive-content contract.""" + + book_id: str = Field(pattern=BLOCK_ID) + blueprint_effective_date: date + objectives: tuple[ObjectiveCoverage, ...] = Field(min_length=1) + + @model_validator(mode="after") + def validate_unique_objectives(self) -> Self: + """Reject ambiguous coverage claims.""" + + objective_ids = [objective.objective_id for objective in self.objectives] + if len(objective_ids) != len(set(objective_ids)): + raise ValueError("coverage objective identifiers must be unique") + return self + + def validate_against(self, book: Book) -> None: + """Verify exact objective, chapter, source, and blueprint mappings.""" + + if self.book_id != book.id: + raise ValueError("coverage book identifier does not match manifest") + if self.blueprint_effective_date != book.blueprint_effective_date: + raise ValueError("coverage blueprint date does not match manifest") + + expected = { + objective.id: chapter + for chapter in book.chapters + for objective in chapter.objectives + } + actual = {coverage.objective_id: coverage for coverage in self.objectives} + if actual.keys() != expected.keys(): + raise ValueError("coverage objectives must exactly match the book manifest") + + known_sources = {source.id for source in book.sources} + for objective_id, coverage in actual.items(): + chapter = expected[objective_id] + if coverage.chapter_id != chapter.id: + raise ValueError(f"coverage chapter mismatch for {objective_id}") + if not set(coverage.source_ids) <= known_sources: + raise ValueError(f"unknown coverage source for {objective_id}") + if not set(coverage.source_ids) <= set(chapter.source_ids): + raise ValueError(f"coverage source not mapped to {chapter.id}") diff --git a/tests/content/test_dp700_outline.py b/tests/content/test_dp700_outline.py index 1ceeef4..79fc754 100644 --- a/tests/content/test_dp700_outline.py +++ b/tests/content/test_dp700_outline.py @@ -38,6 +38,21 @@ def test_every_dp700_chapter_maps_objectives_sources_and_content() -> None: assert sum(len(chapter.objectives) for chapter in book.chapters) == 54 +def test_dp700_coverage_audit_records_every_objective_and_known_gap() -> None: + catalog = BookCatalog(CONTENT_ROOT) + book = catalog.load_book("dp700") + audit = catalog.load_coverage_audit(book) + + assert len(audit.objectives) == 54 + assert {coverage.objective_id for coverage in audit.objectives} == { + objective.id for chapter in book.chapters for objective in chapter.objectives + } + assert all(len(coverage.subtopics) >= 2 for coverage in audit.objectives) + assert all(coverage.source_ids for coverage in audit.objectives) + assert all(coverage.gaps for coverage in audit.objectives) + assert all(coverage.status == "draft" for coverage in audit.objectives) + + def test_source_registry_records_current_learn_provenance() -> None: book = BookCatalog(CONTENT_ROOT).load_book("dp700") diff --git a/tests/unit/test_models.py b/tests/unit/test_models.py index 59e851e..e5e1c04 100644 --- a/tests/unit/test_models.py +++ b/tests/unit/test_models.py @@ -5,6 +5,11 @@ import pytest from pydantic import HttpUrl, ValidationError +from study_reader.content.coverage import ( + CoverageAudit, + CoverageEvidence, + ObjectiveCoverage, +) from study_reader.content.models import Book, Chapter, Domain, Objective, Source @@ -92,3 +97,51 @@ def test_book_rejects_one_way_source_mapping() -> None: with pytest.raises(ValidationError, match="mappings must be bidirectional"): Book.model_validate(book_data) + + +def test_coverage_audit_matches_an_exam_agnostic_book() -> None: + book = minimal_book() + audit = CoverageAudit( + book_id=book.id, + blueprint_effective_date=book.blueprint_effective_date, + objectives=[ + ObjectiveCoverage( + objective_id="design.choose", + chapter_id="design-basics", + status="draft", + subtopics=["Constraints", "Trade-offs"], + source_ids=["official-guide"], + gaps=["Add a worked example"], + ) + ], + ) + + audit.validate_against(book) + + +def test_complete_coverage_requires_every_evidence_category() -> None: + with pytest.raises(ValidationError, match="every evidence category"): + ObjectiveCoverage( + objective_id="design.choose", + chapter_id="design-basics", + status="complete", + subtopics=["Constraints", "Trade-offs"], + source_ids=["official-guide"], + ) + + +def test_complete_coverage_rejects_remaining_gaps() -> None: + evidence = CoverageEvidence( + **{name: name.replace("_", "-") for name in CoverageEvidence.model_fields} + ) + + with pytest.raises(ValidationError, match="cannot retain coverage gaps"): + ObjectiveCoverage( + objective_id="design.choose", + chapter_id="design-basics", + status="complete", + subtopics=["Constraints", "Trade-offs"], + source_ids=["official-guide"], + gaps=["Still incomplete"], + evidence=evidence, + ) From 2d714834563b0805d0ef6efebffc569faf6a6489 Mon Sep 17 00:00:00 2001 From: Troy Scott <846218+troyscott@users.noreply.github.com> Date: Wed, 2 Sep 2026 04:13:59 -0700 Subject: [PATCH 3/6] Complete implement and manage study domain --- .../dp700/chapters/lifecycle-management.md | 171 +++++++++++ .../published/dp700/chapters/orchestration.md | 166 +++++++++++ .../dp700/chapters/security-governance.md | 206 +++++++++++++ .../dp700/chapters/workspace-settings.md | 44 +++ content/published/dp700/coverage.yaml | 270 +++++++++++++++--- src/study_reader/content/catalog.py | 6 + src/study_reader/content/coverage.py | 23 ++ tests/content/test_dp700_outline.py | 12 +- tests/unit/test_models.py | 28 ++ 9 files changed, 887 insertions(+), 39 deletions(-) diff --git a/content/published/dp700/chapters/lifecycle-management.md b/content/published/dp700/chapters/lifecycle-management.md index 95678bb..ca82cd1 100644 --- a/content/published/dp700/chapters/lifecycle-management.md +++ b/content/published/dp700/chapters/lifecycle-management.md @@ -65,6 +65,177 @@ Deployment rules can vary supported data-source or parameter values by stage. Th This separation creates traceability: Git answers *what changed and who reviewed it*; a database project validates a SQL schema model; a deployment pipeline answers *what was promoted between Fabric environments*. +## Version control field guide + + +Think of Git integration as a **two-sided synchronization contract**. One side is +the item definition stored in a particular workspace; the other is a folder in +a particular repository branch. Fabric calculates a status for supported items +by comparing those sides. `Commit to Git` writes workspace definitions to the +connected branch. `Update from Git` reads branch definitions into the +workspace. A branch change made through a pull request does not alter a +workspace until an update occurs, and a portal edit is not durable source +history until it is committed. + + +Before connecting, confirm that the tenant and workspace allow Git integration, +the user has the required workspace and repository permissions, the provider is +supported, and the target branch and directory are intentional. Connect a +development workspace—not production—to the chosen branch and folder. After +connection, inspect the source-control status, commit only the intended items, +and use a pull request to validate and review the resulting definitions. After +merge, update the integration workspace from the shared branch and run item +smoke tests. Secrets, credentials, data, and every runtime setting are not +automatically captured just because an item definition is versioned. + + +Use one workspace per developer or feature when simultaneous portal authoring +would otherwise collide. A shared development workspace is cheaper and simpler, +but it increases coordination and makes unrelated changes appear in the same +sync operation. Prefer trunk-based development when small changes can merge +frequently; use longer-lived stage branches only when the release mechanism +truly depends on them. In either model, protect the integration branch and keep +the branch-to-workspace mapping explicit. A workspace is a deployment target, +not a substitute for a branch. + + +**Worked scenario.** Maya changes a notebook in a feature workspace connected +to `feature/customer-quality`. The source-control pane also shows an unrelated +pipeline edit made by another user. Maya commits only the notebook, opens a pull +request, and CI validates the definition. After approval and merge into `main`, +the team updates the development workspace from `main`. They then run the +notebook against test data. The correct evidence chain is workspace change → +feature commit → pull request and CI → merge → workspace update → runtime test. +The successful merge alone does not prove the notebook runs in Fabric. + + +When synchronization fails, first classify the state: workspace-only change, +Git-only change, conflict, unsupported item, missing dependency, or permission +failure. Preserve both versions before resolving a conflict. Check the connected +branch and folder, repository authorization, workspace role, item support, and +whether another operation is still running. After an update, diagnose a broken +item as a runtime or dependency problem rather than repeatedly syncing it. A +useful mini-lab is to change a harmless notebook comment in a feature branch, +observe each status transition, resolve a deliberately created conflict, and +record which actions affected Git versus the workspace. + +## Database projects field guide + + +A database project is a **declarative model of desired schema state**. Source +files describe objects and their relationships; a build resolves references and +packages the model; publish compares that model with a target and produces a +deployment plan. This differs from a migration sequence that says “run step 17 +after step 16.” The project says what the database should look like, while the +deployment engine infers how to reach that state. + + +Start by importing or authoring supported tables, views, procedures, functions, +and security objects as individual SQL files. Add project references where one +model depends on another. Build on every pull request, treat unresolved object +references and incompatible syntax as failures, and retain the build artifact. +Before publish, generate and review the deployment script or report against the +actual target. Deploy first to a disposable or test database, execute schema and +data-preservation checks, and require approval for production. The deployment +identity needs only the target permissions required by the planned changes. + + +Choose a database project when the desired schema can be represented +declaratively and drift detection matters. Choose explicit migrations when the +order of data movement is itself the essential design—for example, populating a +replacement column before making it non-null. Many mature releases combine +them: the project owns ordinary schema state and reviewed pre/post-deployment +scripts handle exceptional transitions. Do not hide destructive changes in an +automatic publish profile. Renames, narrowing data types, changing distribution +choices, and dropping objects require deliberate review and often a backup or +copy strategy. + + +**Worked scenario.** A new `RegionCode` column must become required. Publishing +`char(2) NOT NULL` directly fails because old rows contain no value. The safe +release adds the nullable column, backfills and validates every row, then makes +the column non-null in a later reviewed change. In a practice database, build a +project with a view that references a misspelled column and confirm the build +catches it; then compare the generated plan before and after renaming a table. +The lesson is that build validation proves model consistency, not safe data +transition semantics. + + +For build errors, inspect the first unresolved reference, target platform, SQL +syntax, and project dependencies. For publish errors, separate authentication +and authorization from model incompatibility, target drift, locks, or data that +violates a new constraint. Never retry a partially applied destructive plan +blindly. Capture the generated script, target schema version, error, and objects +already changed; decide whether to roll forward or restore. A DACPAC contains a +schema model, not a copy of production data and not a complete Fabric workspace. + +## Deployment pipeline field guide + + +A Fabric deployment pipeline is an **environment-promotion mechanism**. Stages +represent lifecycle environments; assigned workspaces hold the live items; +pairing relates corresponding items; comparison shows differences; deployment +moves supported definitions forward. Deployment rules and variable values can +change supported configuration by stage. None of these constructs is a general +secret manager, test runner, or source-control review system. + + +Create the pipeline, define and order its stages, and assign the correct +workspace to each stage. Confirm item support and pairing before the first +promotion. Configure stage-specific supported rules or variables, then deploy a +small coherent set from development to test. Inspect the comparison, run data, +permission, refresh, and dependency checks in test, and obtain release approval. +Promote the same reviewed definitions to production, complete any documented +post-deployment bindings, and run smoke tests under representative identities. +Record the source revision, deployment operation, approver, and verification +result. + + +Deploy a dependency set together when a consumer cannot operate safely without +its producer; deploy independently when the contract is backward compatible. +Use deployment rules for supported environment differences such as connection +or parameter values, and a governed secret store or connection mechanism for +credentials. Selective deployment reduces blast radius but can produce an +inconsistent dependency graph. Full-stage deployment improves consistency but +can promote unrelated work. The correct choice follows tested dependency and +release boundaries, not convenience. + + +**Worked scenario.** A pipeline invokes a notebook that writes to a lakehouse. +All three exist in development, but only the pipeline is selected for test. The +deployment succeeds and the first run fails because the paired notebook or +lakehouse binding is absent. The repair is to compare the dependency set, +promote the compatible items, apply the test connection configuration, and +rerun. As a mini-lab, document a Dev/Test/Prod matrix containing workspace, +capacity, connection, schedule, and owner; promote a harmless parameter change +and verify the resolved value at each stage. + + +If an item is missing from comparison, verify support, workspace assignment, +pairing, and whether it resides in the expected stage. If deployment fails, +check permissions, capacity state, item dependencies, and concurrent +operations. If deployment succeeds but execution fails, move to runtime checks: +credentials, gateways, item IDs, shortcut targets, schedules, and data-plane +permissions. Rollback might mean redeploying the previous known-good definition +or restoring external state; a pipeline has no universal undo for data mutated +by a job. + + +The exam often describes all three lifecycle mechanisms in one scenario. Name +the boundary before choosing: review and history point to Git; declarative SQL +schema points to a database project; staged Fabric promotion points to a +deployment pipeline. “Deployment succeeded” is control-plane evidence only. +The strongest answer also verifies dependencies, configuration, permissions, +and behavior in the target environment. + + +**Recall and mini-lab.** Explain why `Update from Git` can change a workspace +without creating a new Git commit. Describe one schema change a build can +validate but cannot prove safe for existing data. List three post-deployment +checks for a notebook-to-lakehouse pipeline. Then sketch a release with one +feature branch, one pull request, one test promotion, and one intentional +failure; label the evidence produced at each gate. + ## Exam distinctions - Git integration synchronizes a workspace and branch; it is not a Dev/Test/Prod promotion engine. diff --git a/content/published/dp700/chapters/orchestration.md b/content/published/dp700/chapters/orchestration.md index e4d8b4c..85c6517 100644 --- a/content/published/dp700/chapters/orchestration.md +++ b/content/published/dp700/chapters/orchestration.md @@ -58,6 +58,172 @@ Keep configuration typed and explicit. Validate required parameters at the bound A daily sales load receives `businessDate` and `fullReload` parameters. The pipeline checks that the landing file exists, invokes a notebook to validate schema, runs a parameterized copy for valid data, invokes a SQL procedure to merge the target, and records the watermark only after the merge succeeds. An event trigger provides low latency, while a nightly scheduled run reconciles missing events. Because the merge key is stable, either path can retry safely. +## Tool selection as a separation of concerns + + +The three tools sit at different layers. **Dataflow Gen2** is a visual, +Power Query-based transformation experience with managed destinations. +**Notebook** is a code-first execution surface for Spark, Python, and SQL logic. +**Pipeline** is a control plane that moves data and coordinates activities. +The fact that a pipeline can copy data or evaluate an expression does not make +it the best home for complex transformation logic; the fact that a notebook can +call APIs does not make it a maintainable enterprise scheduler. + + +Begin with the unit of work. If a business analyst can express repeatable +tabular shaping in Power Query and use a supported destination, prototype a +Dataflow Gen2. If the work requires distributed computation, custom libraries, +complex testing, or code reuse, create a notebook and parameterize its inputs. +When two or more activities require dependencies, data movement, retries, +branching, parameters, or schedules, put the control flow in a pipeline and +invoke the transformation item. Give the pipeline identity access to each +invoked item and data endpoint; do not embed secrets in expressions. + + +Select by maintainability as well as capability. Dataflow Gen2 improves visual +accessibility but complicated M expressions can become difficult to test. +Notebooks provide flexibility but require software discipline and can incur +Spark startup cost. Pipelines provide operational visibility but deeply nested +activities and business rules in expression strings become brittle. A common +design is pipeline → parameterized notebook or Dataflow Gen2 → governed +destination. Use a single tool when it genuinely owns the whole job; adding an +orchestrator to one simple transformation can create needless failure points. + + +**Worked decision.** A CSV requires column renaming, type conversion, a lookup +join, and loading to a warehouse table. A Dataflow Gen2 is suitable when the +volume and transformations fit its connectors and the owning team works in +Power Query. If the lookup is a very large Delta table and the logic uses a +tested Python library, choose a notebook. If the file must first be copied from +an on-premises source, the transform run after validation, and an alert sent on +failure, use a pipeline to coordinate the chosen transformer. The tool decision +is about responsibilities, not a contest for one universal winner. + + +When a solution becomes hard to operate, look for logic at the wrong layer: +hundreds of pipeline expressions, a notebook reimplementing scheduling and +retry, or a Dataflow whose steps hide an opaque procedural algorithm. Inspect +run history at the orchestration layer and the invoked item's detailed logs at +the compute layer. As a mini-lab, implement the same three-column cleanup once +in a Dataflow and once in a notebook, then write a pipeline that invokes one; +compare authoring, testability, startup, lineage, and error evidence. + +## Trigger engineering + + +A schedule asserts that **time is the readiness signal**. An event trigger +asserts that **an observed event is the readiness signal**. Neither proves that +all business inputs are complete. Scheduled runs must reason about time zone, +daylight-saving transitions, source close times, and overlap. Event-driven runs +must reason about event filtering, duplication, ordering, partial writes, and +bursts. A reconciliation schedule often complements events because delivery +systems can be at-least-once or temporarily unavailable. + + +For a schedule, define recurrence, start and end boundaries, time zone, missed +run policy, expected duration, and safe concurrency. For an event trigger, +select the event source, filter to the intended objects, pass stable event +metadata into pipeline parameters, and validate that the object is ready before +processing. In both cases, generate or receive an idempotency key, persist the +source version and watermark, and make retries safe. Configure monitoring for +failed, unusually long, and unexpectedly absent runs. + + +Prefer schedules for periodic snapshots, closed accounting periods, and +reconciliation. Prefer events for low-latency response to discrete arrivals. +Use both when fast processing and eventual completeness matter. Prevent overlap +when the target uses destructive replace semantics; allow controlled +concurrency when inputs and target partitions are independent. A “file created” +event can fire before an upstream multi-file delivery is complete, so use a +manifest, completion marker, stable-size check, or upstream contract rather +than an arbitrary delay. + + +**Worked trigger.** An event for +`landing/region=CA/business_date=2026-09-01/orders.parquet` passes the URL, +event ID, and modification timestamp to a pipeline. The first activity rejects +unexpected paths and checks a control table keyed by URL plus version. A valid +new object is processed and the key recorded atomically. A duplicate event +finds the completed key and exits successfully without inserting rows again. A +02:00 scheduled reconciliation compares the manifest with processed keys and +submits only missing versions. + + +For a run that never started, inspect whether the trigger is enabled, its time +zone or event subscription, filter, source event, and permissions. For duplicate +runs, compare event IDs and business idempotency keys; do not simply increase a +delay. For overlapping schedules, compare trigger time, actual start time, +duration, queueing, and concurrency settings. A useful mini-lab is to submit the +same event twice and prove the target has one logical result, then intentionally +withhold an event and prove reconciliation finds it. + +## Parameterized orchestration patterns + + +Parameters are run inputs and should be treated as immutable. Variables hold +mutable run state. System variables expose orchestration context. Dynamic +expressions resolve values from parameters, activity outputs, variables, and +system context at runtime. A parent-child pattern centralizes common control +flow; a metadata-driven pattern turns configuration rows into repeated work; +fan-out/fan-in runs independent units in parallel and then joins their results. + + +Define parameter names, types, defaults, allowed values, and ownership before +building expressions. Validate required inputs in the first activity. Pass only +the values a child requires and return a small, documented result. In a +metadata-driven pipeline, look up enabled configuration rows, iterate with a +bounded concurrency, and parameterize datasets, paths, or notebook arguments. +Use activity dependencies for success, failure, completion, and skip paths. +Route secrets through managed connections or secret integration, mask sensitive +outputs, and include the pipeline run ID in operational records. + + +Use a parent-child pipeline when the child is a coherent reusable workflow, not +merely to reduce the number of boxes on screen. Use metadata-driven iteration +when many entities share one algorithm and differ in configuration. Use +fan-out/fan-in only when target isolation and capacity support parallelism. +Prefer explicit expressions over clever nested expressions, and calculate +complex business logic in a tested transform. Parameters configure behavior; +copying whole environment-specific JSON documents into them can create an +unreviewed second configuration system. + + +**Worked pattern.** A control table contains `entity`, `source_path`, +`target_table`, `watermark_column`, and `enabled`. The parent looks up enabled +rows and invokes child `load_entity` with those five typed values. The child +reads the previous watermark, copies the bounded range to staging, validates +counts, merges into the target, advances the watermark only after success, and +returns rows read and written. The parent aggregates results and fails if any +required entity failed. Rerunning one entity uses the same algorithm and a +deliberate watermark override. + + +Expression failures often come from the wrong evaluation context, null activity +output, incorrect JSON path, unintended string conversion, or escaping. Inspect +the resolved activity input in run details rather than only the expression +source. For a failed child, retain both parent and child run IDs. For loops, +record the current entity and concurrency. Reproduce with one known metadata row +before scaling out. A mini-lab should process two entities, force one child to +fail, confirm the parent captures both results, and rerun only the failed unit +without duplicating the successful target. + + +On the exam, “transform with a visual Power Query experience” points to +Dataflow Gen2; “distributed custom code” points to a notebook; “coordinate, +copy, branch, retry, or trigger” points to a pipeline. Parameters are immutable +run inputs, variables are mutable run state, and dynamic expressions compute +runtime values. An event trigger reduces latency but does not remove the need +for idempotency and reconciliation. + + +**Recall and mini-lab.** Why might a pipeline invoke a notebook rather than +place all logic in pipeline expressions? Give two ways to prove a multi-file +delivery is complete. What state must advance only after a successful +incremental load? Contrast an event ID with a business idempotency key. Finally, +draw a parent pipeline with lookup, bounded fan-out, child invocation, failure +collection, and a reconciliation trigger; label parameters, variables, and +system values. + ## Exam distinctions - A schedule answers *when*; an activity dependency answers *after what*. diff --git a/content/published/dp700/chapters/security-governance.md b/content/published/dp700/chapters/security-governance.md index ab52efe..ce0f133 100644 --- a/content/published/dp700/chapters/security-governance.md +++ b/content/published/dp700/chapters/security-governance.md @@ -112,6 +112,212 @@ Evaluate access as a path: Test with representative users, including denied cases. Administrators and item owners can have elevated rights that make their tests misleading. +## Build an access decision from the outside in + + +Workspace roles define a broad collaboration boundary. **Admin** manages the +workspace and access; **Member** collaborates broadly and can manage many +workspace capabilities; **Contributor** creates and changes content; **Viewer** +primarily consumes. Admin, Member, and Contributor are elevated identities for +OneLake because their workspace rights include write access. A narrow OneLake +read role therefore cannot be used to reduce the data rights already granted by +one of those broader workspace roles. + + +Inventory people, service principals, and groups; assign each to a job function; +and grant the lowest workspace role that permits the required collaboration. +Prefer Microsoft Entra security groups so joiner, mover, and leaver changes occur +in one identity system. Separate administration from content development, keep +at least two governed owners, and review assignments periodically. Validate with +a user who has no overlapping group membership. For services, document the +identity, credential owner, rotation method, allowed workspace, and exact jobs. + + +Choose a workspace role only when the identity needs capabilities across the +workspace. Do not grant Contributor merely to let a consumer query one table; +use item and data permissions. Conversely, repeated direct shares across most +items are a signal that the person may genuinely belong in the workspace. The +failure mode to avoid is **permission accumulation**: a user receives narrow +access directly but also inherits broad access through a workspace group, so a +test of the narrow policy appears ineffective. + + +**Scenario.** Data engineers belong to a Contributor group, the platform team +uses a small Admin group, and report consumers are not workspace members. A +contractor who needs one lakehouse is shared that item and assigned a narrow +OneLake role. Test the contractor, a Contributor, and an unassigned account. +The contractor should not discover unrelated workspace items; the Contributor +will still have broad OneLake rights. This three-identity test is more +informative than testing only as the workspace owner. + + +When access is unexpectedly allowed, enumerate every workspace role and group, +direct item permission, OneLake role, SQL grant, semantic-model role, and cached +session. When access is denied, check the same layers in order plus tenant +settings and endpoint-specific prerequisites. Remove ambiguity by using a fresh +test identity and a private browser session. Record both positive and negative +tests; “the administrator could open it” is not evidence of least privilege. + +## Item permissions and the data plane + + +Item permissions answer “may this identity interact with this item?” but not +always “may it read every underlying row through every engine?” `Read` commonly +exposes item metadata. Additional permissions such as `ReadData` or `ReadAll` +govern documented compute or OneLake routes for supported items. A report, +semantic model, SQL endpoint, and lakehouse can therefore participate in one +experience while enforcing different permissions. + + +Start at the consumer experience and trace backward: report → semantic model → +SQL endpoint or OneLake table → source or shortcut target. Grant the report or +item permission first, then only the downstream data permission the intended +route requires. Use Manage permissions to inspect direct grants and reshare +rights. Avoid granting build, write, or reshare unless the user must create new +content or delegate access. Re-test after removing a grant because tokens and +sessions can temporarily obscure the result. + + +Use a curated report when consumers need answers, item sharing when they need a +specific reusable item, and workspace membership when they participate in the +workspace lifecycle. Granting `ReadAll` for OneLake access is materially +different from granting basic `Read` metadata access. Direct Lake also requires +careful reasoning about which identity and fallback path performs the query. +Never infer the data plane solely from whether the item appears in navigation. + + +**Worked access trace.** Lee can open a report but receives an error when +connecting directly to the lakehouse SQL endpoint. That can be correct: the +report path may be authorized through its semantic model while direct SQL data +access was never granted. If the requirement is report consumption only, do not +“fix” the result with Contributor. If direct analysis is required, grant the +documented endpoint permission and test a permitted and forbidden table. + + +Classify the failing action precisely: discover item, open item, query SQL, +read OneLake, build a semantic model, write data, or reshare. Each action points +to a different permission. Examine group expansion and inherited workspace +roles before adding another direct grant. A mini-lab should create a matrix of +three users by four actions and predict each outcome before testing it. + +## Data-level controls + + +Object or folder security chooses the tables, schemas, or paths a role can +reach. Column security removes selected attributes. Row security applies a +predicate to records. These controls can be combined, but OneLake roles use a +**grant model**: a restrictive role does not deny access obtained from another +role or permission path. Engine-native SQL or semantic-model security may have +different authoring and enforcement surfaces, so identify the query route in +every design. + + +Define access from a business policy, not from the current folder layout. Create +roles for stable functions, grant only required tables or folders, then apply +row and column rules where supported. Use immutable business keys and a governed +identity-to-scope mapping for dynamic row filtering. Validate Spark, SQL, +Direct Lake, and OneLake API paths that are actually in scope. Test nulls, +multiple group memberships, new rows, schema changes, and an identity that +should see nothing. + + +Prefer object or folder grants when whole datasets differ by audience. Use CLS +when a column must not be visible, RLS when records differ by user context, and +a curated view when a stable consumer contract or derived logic is needed. +Masking is not a substitute for removing a sensitive column. Avoid elaborate +per-user roles; group-driven roles and mapping tables scale better and are +auditable. Check current feature support before assuming the same rule is +enforced by every Fabric engine. + + +**Worked policy.** Regional analysts may read `Sales.OrderFact` only for their +region and must not see `Customer.Email`; finance may read all regions and the +email column. Create separate group-based roles, apply the region predicate to +the analyst role and exclude or restrict Email, then test an analyst in two +regions, finance, and an unassigned user. Adding an analyst to Contributor would +invalidate the intended OneLake restriction and should be caught by the test. + + +If a filter appears bypassed, look for broad workspace write, another granting +role, ownership, a different engine, or a cached result. If a query fails, +distinguish inability to reach the item from denial at table, column, or row +evaluation. Use a minimal query against one known object and progressively add +columns and predicates. Preserve the identity, endpoint, query, expected rows, +actual rows, and effective memberships as security-test evidence. + +## Masking, classification, trust, and evidence + + +Dynamic data masking is a presentation control evaluated at query time for +principals without `UNMASK`. Configure a supported mask on the target column, +grant ordinary query access to a test role, and compare results with and without +`UNMASK`. Choose a mask that reduces accidental exposure without misleading +users about the data type. Because values remain unchanged and inference can be +possible, use encryption, RLS, CLS, or denied object access for true +confidentiality boundaries. Troubleshoot by checking the querying principal, +`UNMASK` grants, object permissions, endpoint, and supported type. + + +Sensitivity labels classify an item according to organizational information +protection policy. Prerequisites include tenant configuration, labels published +to the author, and a supported item or propagation path. Choose a label based on +the data's policy classification, not the item's popularity. Apply it, verify +the displayed label and any documented downstream inheritance or export +behavior, and test the relevant route. A label can persist as governance +metadata or invoke supported protection, but it is not evidence that workspace, +item, and data permissions are correct. If a label is unavailable, inspect +Purview publication, licensing, tenant settings, user scope, and item support. + + +Endorsement describes organizational trust. Owners can promote suitable items; +authorized reviewers certify items or designate master data according to the +organization's governance process. Define criteria—owner, documentation, +quality checks, freshness, security review, and support contact—before applying +a badge. Review it when those facts change. Certification does not grant access +or guarantee that a consumer's interpretation is valid. If endorsement options +are missing, verify write permission, tenant configuration, and designated +certifier status. + + +Fabric audit evidence is searched through Microsoft Purview Audit. Establish +the investigation question and time zone, then query an appropriate time range, +users, operations, and workload context. Preserve exported results with the +query criteria and investigation record. Audit is for recorded user and admin +activity; Monitoring hub and workload logs explain operational executions. +Account for retention, licensing, ingestion delay, operation naming, and clock +boundaries before concluding an event is absent. As a mini-lab, perform a +benign item-share change, wait for ingestion, locate its audit record, and +compare what the audit event tells you with what the item's current permission +page tells you. + + +OneLake security roles grant supported table and folder access within a data +item. The author needs the documented Fabric Write or Reshare capability, and +role members should be stable groups where possible. Create the role, choose +Read or ReadWrite as required, select data objects, apply optional row or column +filters, assign members, and test through the intended engine. Default roles can +map documented item or workspace permissions to data access. Because grants are +additive, always inspect other roles and broad workspace rights when a user sees +too much. For a shortcut, evaluate both the OneLake reference and the remote +source identity or credential path. + + +For exam scenarios, separate four verbs: **authorize** with workspace, item, and +data permissions; **obscure returned values** with masking; **classify** with a +sensitivity label; **signal trust** with endorsement. Then distinguish audit +evidence from operational monitoring. A request for “one table only” points +away from Contributor and toward item plus OneLake data permissions. A request +to hide rows by region points to RLS, not a label or mask. + + +**Recall and mini-lab.** Why can a Contributor bypass the intent of a narrow +OneLake read role? What additional capability does a user need when `Read` +allows item discovery but not direct data access? Contrast CLS with DDM in one +sentence. Explain why Certified and Confidential answer different questions. +Finally, design a five-row access matrix for an administrator, engineer, +analyst, report-only consumer, and unassigned user; include at least one denied +test for workspace, item, SQL, and OneLake access. + ## Exam distinctions - Workspace roles are broad collaboration grants; item permissions are narrower. diff --git a/content/published/dp700/chapters/workspace-settings.md b/content/published/dp700/chapters/workspace-settings.md index 1d1c4d4..861cd5d 100644 --- a/content/published/dp700/chapters/workspace-settings.md +++ b/content/published/dp700/chapters/workspace-settings.md @@ -213,6 +213,50 @@ A useful decision sequence is: Current Microsoft Learn documentation states that Fabric Apache Airflow jobs do not support private networks or virtual networks. Treat that as a time-sensitive product limitation and recheck the authoritative source when designing a secured deployment. +## Troubleshooting and practice + + +**Spark practice.** In a test workspace, record the default pool and whether item +customization is enabled. Attach an environment to a notebook, change its +runtime or pool, save without publishing, and predict which configuration the +next session will use. Publish and repeat. If publication fails, inspect runtime +and library compatibility; if a session starts with unexpected resources, +compare workspace default, environment publication state, attached environment, +and session-level configuration. Distinguish slow startup from slow execution: +starter-pool availability affects the former, while partitioning, shuffle, +skew, executor sizing, and capacity pressure affect the latter. + + +**Domain practice.** Draw a tenant with Sales and Finance domains, one subdomain, +two default-domain groups, and an already assigned shared workspace. Predict the +workspace assignment after each administrator creates a new workspace. Then +verify who can assign each workspace and whether a consumer gains access. If an +assignment control is unavailable, check Fabric/domain role, workspace Admin, +the domain's allowed contributors, and tenant delegation. If catalog placement +is correct but access is denied, stop troubleshooting domains and inspect +workspace, item, and data permissions. + + +**OneLake practice.** Design a diagnostics destination and a lifecycle rule for +`Files/DiagnosticExports/`. Record capacity placement, configuring identity, +immutability period, path scope, age condition, tier action, and cleanup owner. +After enabling, allow for documented activation and asynchronous evaluation +instead of repeatedly toggling settings. For missing diagnostic events, check +destination prerequisites, permissions, activation time, and the requested +access route. For unexpected tiering, inspect default tier, explicit file tier, +rule scope, time basis, access-time tracking, minimum-retention cost, and the +policy's asynchronous run—not only the file's current modified timestamp. + + +**Airflow practice.** Given four concurrent DAGs, separate scheduler delay, +pool-resume delay, worker saturation, and slow task code. Compare starter versus +custom pool, node size, extra nodes, autoscale, and uptime ownership. Create a +decision record for an intermittent development workload and a production +workload with a start-time objective. If an Airflow environment uses an +unexpected pool, check the workspace default and whether item customization is +allowed. If tasks queue after the environment is running, examine worker +concurrency and task demand before increasing compute. + ## Exam distinctions diff --git a/content/published/dp700/coverage.yaml b/content/published/dp700/coverage.yaml index 9d60bed..25c3249 100644 --- a/content/published/dp700/coverage.yaml +++ b/content/published/dp700/coverage.yaml @@ -3,112 +3,310 @@ blueprint_effective_date: 2026-07-21 objectives: - objective_id: implement.workspace.spark chapter_id: workspace-settings - status: draft + status: complete subtopics: [Workspace defaults, Environment overrides, Runtimes, Pools, Driver and executor sizing, Session configuration] source_ids: [dp700-study-guide, spark-compute-settings] - gaps: [Add a hands-on configuration lab and objective-specific evidence blocks] + gaps: [] + evidence: + conceptual_model: spark-settings + prerequisites: spark-layers + procedure: spark-settings + decision_guidance: spark-decisions + worked_example: spark-troubleshooting-lab + limitations: spark-layers + troubleshooting: spark-troubleshooting-lab + exam_distinctions: exam-distinctions + recall_questions: active-recall + scenario_lab: spark-troubleshooting-lab - objective_id: implement.workspace.domain chapter_id: workspace-settings - status: draft + status: complete subtopics: [Domain hierarchy, Workspace assignment, Domain roles, Default domains, Delegated governance, Catalog discovery] source_ids: [dp700-study-guide, fabric-domains] - gaps: [Add a domain-assignment lab and negative authorization test] + gaps: [] + evidence: + conceptual_model: domain-settings + prerequisites: domain-roles + procedure: domain-roles + decision_guidance: domain-defaults + worked_example: domain-troubleshooting-lab + limitations: domain-defaults + troubleshooting: domain-troubleshooting-lab + exam_distinctions: exam-distinctions + recall_questions: active-recall + scenario_lab: domain-troubleshooting-lab - objective_id: implement.workspace.onelake chapter_id: workspace-settings - status: draft + status: complete subtopics: [Diagnostics, Immutability, Storage tiers, Lifecycle rules, REST API, Permissions and placement] source_ids: [dp700-study-guide, onelake-overview, onelake-diagnostics, onelake-storage-tiers, onelake-lifecycle, onelake-settings-api] - gaps: [Add an end-to-end diagnostics and lifecycle scenario] + gaps: [] + evidence: + conceptual_model: terminology + prerequisites: responsibility-boundaries + procedure: lifecycle-rules + decision_guidance: tier-decision + worked_example: lifecycle-rules + limitations: settings-map + troubleshooting: onelake-troubleshooting-lab + exam_distinctions: exam-distinctions + recall_questions: active-recall + scenario_lab: onelake-troubleshooting-lab - objective_id: implement.workspace.airflow chapter_id: workspace-settings - status: draft + status: complete subtopics: [Workspace runtime, Starter pools, Custom pools, Sizing, Autoscaling, Concurrency and limitations] source_ids: [dp700-study-guide, airflow-workspace-settings] - gaps: [Add configuration procedure and pool-sizing lab] + gaps: [] + evidence: + conceptual_model: airflow-settings + prerequisites: airflow-settings + procedure: airflow-settings + decision_guidance: airflow-decisions + worked_example: airflow-troubleshooting-lab + limitations: airflow-decisions + troubleshooting: airflow-troubleshooting-lab + exam_distinctions: exam-distinctions + recall_questions: active-recall + scenario_lab: airflow-troubleshooting-lab - objective_id: implement.lifecycle.version-control chapter_id: lifecycle-management - status: draft + status: complete subtopics: [Git providers, Workspace connection, Branches, Commit and update direction, Conflicts, Supported items, Secrets] source_ids: [dp700-study-guide, fabric-cicd-overview] - gaps: [Add portal workflow, conflict exercise, and supported-item limitations] + gaps: [] + evidence: + conceptual_model: version-control-model + prerequisites: version-control-operations + procedure: version-control-operations + decision_guidance: version-control-decisions + worked_example: version-control-example + limitations: version-control-decisions + troubleshooting: version-control-diagnostics + exam_distinctions: lifecycle-exam-distinctions + recall_questions: lifecycle-recall-lab + scenario_lab: version-control-diagnostics - objective_id: implement.lifecycle.database-projects chapter_id: lifecycle-management - status: draft + status: complete subtopics: [Declarative schema, Project structure, Build validation, DACPAC, Publish plan, State versus migration, Data-loss safeguards] source_ids: [dp700-study-guide, fabric-cicd-overview] - gaps: [Add direct database-project sources and a build-deploy walkthrough] + gaps: [] + evidence: + conceptual_model: database-projects-model + prerequisites: database-projects-operations + procedure: database-projects-operations + decision_guidance: database-projects-decisions + worked_example: database-projects-example + limitations: database-projects-decisions + troubleshooting: database-projects-diagnostics + exam_distinctions: lifecycle-exam-distinctions + recall_questions: lifecycle-recall-lab + scenario_lab: database-projects-example - objective_id: implement.lifecycle.deployment-pipelines chapter_id: lifecycle-management - status: draft + status: complete subtopics: [Stages, Workspace assignment, Item pairing, Comparison, Deployment rules, Dependency binding, Post-deployment checks] source_ids: [dp700-study-guide, fabric-cicd-overview] - gaps: [Add current pipeline configuration procedure and failed-binding scenario] + gaps: [] + evidence: + conceptual_model: deployment-pipelines-model + prerequisites: deployment-pipelines-operations + procedure: deployment-pipelines-operations + decision_guidance: deployment-pipelines-decisions + worked_example: deployment-pipelines-example + limitations: deployment-pipelines-decisions + troubleshooting: deployment-pipelines-diagnostics + exam_distinctions: lifecycle-exam-distinctions + recall_questions: lifecycle-recall-lab + scenario_lab: deployment-pipelines-example - objective_id: implement.security.workspace-access chapter_id: security-governance - status: draft + status: complete subtopics: [Admin, Member, Contributor, Viewer, Group assignment, Least privilege, Workspace scope] source_ids: [dp700-study-guide, fabric-permission-model] - gaps: [Add complete capability matrix and representative access tests] + gaps: [] + evidence: + conceptual_model: workspace-access-model + prerequisites: workspace-access-operations + procedure: workspace-access-operations + decision_guidance: workspace-access-decisions + worked_example: workspace-access-example + limitations: workspace-access-decisions + troubleshooting: workspace-access-diagnostics + exam_distinctions: security-exam-distinctions + recall_questions: security-recall-lab + scenario_lab: workspace-access-example - objective_id: implement.security.item-access chapter_id: security-governance - status: draft + status: complete subtopics: [Item sharing, Direct permissions, Read and reshare, Ownership, Group-based grants, Revocation] source_ids: [dp700-study-guide, fabric-permission-model] - gaps: [Add item-type examples and portal/API procedure] + gaps: [] + evidence: + conceptual_model: item-access-model + prerequisites: item-access-operations + procedure: item-access-operations + decision_guidance: item-access-decisions + worked_example: item-access-example + limitations: item-access-decisions + troubleshooting: item-access-diagnostics + exam_distinctions: security-exam-distinctions + recall_questions: security-recall-lab + scenario_lab: item-access-diagnostics - objective_id: implement.security.data-access chapter_id: security-governance - status: draft + status: complete subtopics: [Row-level security, Column-level security, Object-level security, Folder and file access, Engine enforcement paths, Bypass testing] source_ids: [dp700-study-guide, fabric-permission-model, onelake-data-access-control] - gaps: [Add implementation examples for every control and cross-engine enforcement matrix] + gaps: [] + evidence: + conceptual_model: data-access-model + prerequisites: data-access-operations + procedure: data-access-operations + decision_guidance: data-access-decisions + worked_example: data-access-example + limitations: data-access-decisions + troubleshooting: data-access-diagnostics + exam_distinctions: security-exam-distinctions + recall_questions: security-recall-lab + scenario_lab: data-access-example - objective_id: implement.security.masking chapter_id: security-governance - status: draft + status: complete subtopics: [Mask functions, UNMASK permission, Stored versus returned values, Inference risk, Combining DDM with RLS and CLS] source_ids: [dp700-study-guide, warehouse-dynamic-masking] - gaps: [Add complete mask examples, grants, and negative tests] + gaps: [] + evidence: + conceptual_model: masking-comprehensive + prerequisites: masking-comprehensive + procedure: masking-comprehensive + decision_guidance: masking-comprehensive + worked_example: masking-comprehensive + limitations: masking-comprehensive + troubleshooting: masking-comprehensive + exam_distinctions: security-exam-distinctions + recall_questions: security-recall-lab + scenario_lab: security-recall-lab - objective_id: implement.security.sensitivity chapter_id: security-governance - status: draft + status: complete subtopics: [Purview labels, Tenant enablement, Label publishing, Defaults, Mandatory labels, Inheritance, Export behavior] source_ids: [dp700-study-guide, fabric-information-protection] - gaps: [Add application procedure, supported-item matrix, and inheritance scenario] + gaps: [] + evidence: + conceptual_model: sensitivity-comprehensive + prerequisites: sensitivity-comprehensive + procedure: sensitivity-comprehensive + decision_guidance: sensitivity-comprehensive + worked_example: sensitivity-comprehensive + limitations: sensitivity-comprehensive + troubleshooting: sensitivity-comprehensive + exam_distinctions: security-exam-distinctions + recall_questions: security-recall-lab + scenario_lab: security-recall-lab - objective_id: implement.security.endorsement chapter_id: security-governance - status: draft + status: complete subtopics: [Promoted, Certified, Master data, Authorized reviewers, Discovery, Trust versus authorization] source_ids: [dp700-study-guide, fabric-endorsement] - gaps: [Add endorsement workflow and governance scenario] + gaps: [] + evidence: + conceptual_model: endorsement-comprehensive + prerequisites: endorsement-comprehensive + procedure: endorsement-comprehensive + decision_guidance: endorsement-comprehensive + worked_example: endorsement-comprehensive + limitations: endorsement-comprehensive + troubleshooting: endorsement-comprehensive + exam_distinctions: security-exam-distinctions + recall_questions: security-recall-lab + scenario_lab: security-recall-lab - objective_id: implement.security.audit chapter_id: security-governance - status: draft + status: complete subtopics: [Purview Audit, Search filters, Operations, Retention, Export, Investigation procedure, Audit versus diagnostic logs] source_ids: [dp700-study-guide, fabric-audit-activities] - gaps: [Add investigation walkthrough and evidence-handling lab] + gaps: [] + evidence: + conceptual_model: audit-comprehensive + prerequisites: audit-comprehensive + procedure: audit-comprehensive + decision_guidance: audit-comprehensive + worked_example: audit-comprehensive + limitations: audit-comprehensive + troubleshooting: audit-comprehensive + exam_distinctions: security-exam-distinctions + recall_questions: security-recall-lab + scenario_lab: audit-comprehensive - objective_id: implement.security.onelake chapter_id: security-governance - status: draft + status: complete subtopics: [OneLake roles, Tables and folders, Read and write grants, Role membership, Shortcut security, Cross-engine access paths] source_ids: [dp700-study-guide, onelake-data-access-control] - gaps: [Add role-creation procedure, API example, and denied-access tests] + gaps: [] + evidence: + conceptual_model: onelake-security-comprehensive + prerequisites: onelake-security-comprehensive + procedure: onelake-security-comprehensive + decision_guidance: onelake-security-comprehensive + worked_example: data-access-example + limitations: onelake-security-comprehensive + troubleshooting: data-access-diagnostics + exam_distinctions: security-exam-distinctions + recall_questions: security-recall-lab + scenario_lab: data-access-example - objective_id: implement.orchestration.choose-tool chapter_id: orchestration - status: draft + status: complete subtopics: [Dataflow Gen2, Pipeline, Notebook, Transformation versus orchestration, Personas, Scale and maintainability] source_ids: [dp700-study-guide, pipeline-overview] - gaps: [Add representative tool-selection cases and cost/operability trade-offs] + gaps: [] + evidence: + conceptual_model: choose-tool-model + prerequisites: choose-tool-operations + procedure: choose-tool-operations + decision_guidance: choose-tool-decisions + worked_example: choose-tool-example + limitations: choose-tool-decisions + troubleshooting: choose-tool-diagnostics + exam_distinctions: orchestration-exam-distinctions + recall_questions: orchestration-recall-lab + scenario_lab: choose-tool-diagnostics - objective_id: implement.orchestration.triggers chapter_id: orchestration - status: draft + status: complete subtopics: [Schedules, Time zones, Event triggers, File events, Filtering, Concurrency, Idempotency, Reconciliation] source_ids: [dp700-study-guide, pipeline-overview] - gaps: [Add trigger configuration procedures and duplicate-event lab] + gaps: [] + evidence: + conceptual_model: triggers-model + prerequisites: triggers-operations + procedure: triggers-operations + decision_guidance: triggers-decisions + worked_example: triggers-example + limitations: triggers-decisions + troubleshooting: triggers-diagnostics + exam_distinctions: orchestration-exam-distinctions + recall_questions: orchestration-recall-lab + scenario_lab: triggers-diagnostics - objective_id: implement.orchestration.patterns chapter_id: orchestration - status: draft + status: complete subtopics: [Parameters, Variables, Dynamic expressions, Parent-child pipelines, Metadata-driven loops, Dependencies, Retry and failure paths] source_ids: [dp700-study-guide, pipeline-overview, pipeline-parameters] - gaps: [Add executable expression catalog and end-to-end pipeline/notebook example] + gaps: [] + evidence: + conceptual_model: patterns-model + prerequisites: patterns-operations + procedure: patterns-operations + decision_guidance: patterns-decisions + worked_example: patterns-example + limitations: patterns-decisions + troubleshooting: patterns-diagnostics + exam_distinctions: orchestration-exam-distinctions + recall_questions: orchestration-recall-lab + scenario_lab: patterns-diagnostics - objective_id: ingest.loading.full-incremental chapter_id: loading-patterns status: draft diff --git a/src/study_reader/content/catalog.py b/src/study_reader/content/catalog.py index d093749..b0faee1 100644 --- a/src/study_reader/content/catalog.py +++ b/src/study_reader/content/catalog.py @@ -44,6 +44,12 @@ def load_coverage_audit(self, book: Book) -> CoverageAudit: raise FileNotFoundError(audit_path) from error audit = CoverageAudit.model_validate(raw_audit) audit.validate_against(book) + audit.validate_evidence_blocks( + { + chapter.id: self.load_chapter_markdown(book, chapter) + for chapter in book.chapters + } + ) return audit def load_chapter_markdown(self, book: Book, chapter: Chapter) -> str: diff --git a/src/study_reader/content/coverage.py b/src/study_reader/content/coverage.py index 6f32ccf..261b49a 100644 --- a/src/study_reader/content/coverage.py +++ b/src/study_reader/content/coverage.py @@ -5,6 +5,7 @@ from pydantic import BaseModel, ConfigDict, Field, model_validator +from study_reader.content.markdown import BLOCK_MARKER from study_reader.content.models import Book BLOCK_ID = r"^[a-z0-9][a-z0-9-]*$" @@ -114,3 +115,25 @@ def validate_against(self, book: Book) -> None: raise ValueError(f"unknown coverage source for {objective_id}") if not set(coverage.source_ids) <= set(chapter.source_ids): raise ValueError(f"coverage source not mapped to {chapter.id}") + + def validate_evidence_blocks(self, chapter_markdown: dict[str, str]) -> None: + """Verify that completion evidence names durable blocks in its chapter.""" + + for coverage in self.objectives: + if coverage.status != "complete": + continue + markdown = chapter_markdown.get(coverage.chapter_id) + if markdown is None: + raise ValueError(f"missing chapter content for {coverage.chapter_id}") + block_ids = { + marker.group(1) + for line in markdown.splitlines() + if (marker := BLOCK_MARKER.fullmatch(line)) is not None + } + missing = set(coverage.evidence.block_ids) - block_ids + if missing: + missing_list = ", ".join(sorted(missing)) + raise ValueError( + f"missing evidence blocks for {coverage.objective_id}: " + f"{missing_list}" + ) diff --git a/tests/content/test_dp700_outline.py b/tests/content/test_dp700_outline.py index 79fc754..2243f4b 100644 --- a/tests/content/test_dp700_outline.py +++ b/tests/content/test_dp700_outline.py @@ -38,7 +38,7 @@ def test_every_dp700_chapter_maps_objectives_sources_and_content() -> None: assert sum(len(chapter.objectives) for chapter in book.chapters) == 54 -def test_dp700_coverage_audit_records_every_objective_and_known_gap() -> None: +def test_dp700_coverage_audit_records_every_objective_and_evidence() -> None: catalog = BookCatalog(CONTENT_ROOT) book = catalog.load_book("dp700") audit = catalog.load_coverage_audit(book) @@ -49,8 +49,14 @@ def test_dp700_coverage_audit_records_every_objective_and_known_gap() -> None: } assert all(len(coverage.subtopics) >= 2 for coverage in audit.objectives) assert all(coverage.source_ids for coverage in audit.objectives) - assert all(coverage.gaps for coverage in audit.objectives) - assert all(coverage.status == "draft" for coverage in audit.objectives) + assert sum(coverage.status == "complete" for coverage in audit.objectives) == 18 + assert sum(coverage.status == "draft" for coverage in audit.objectives) == 36 + assert all( + coverage.evidence.is_complete and not coverage.gaps + if coverage.status == "complete" + else bool(coverage.gaps) + for coverage in audit.objectives + ) def test_source_registry_records_current_learn_provenance() -> None: diff --git a/tests/unit/test_models.py b/tests/unit/test_models.py index e5e1c04..6b1d87c 100644 --- a/tests/unit/test_models.py +++ b/tests/unit/test_models.py @@ -145,3 +145,31 @@ def test_complete_coverage_rejects_remaining_gaps() -> None: gaps=["Still incomplete"], evidence=evidence, ) + + +def test_complete_coverage_requires_evidence_blocks_in_its_chapter() -> None: + book = minimal_book() + evidence = CoverageEvidence( + **{name: "design-evidence" for name in CoverageEvidence.model_fields} + ) + audit = CoverageAudit( + book_id=book.id, + blueprint_effective_date=book.blueprint_effective_date, + objectives=[ + ObjectiveCoverage( + objective_id="design.choose", + chapter_id="design-basics", + status="complete", + subtopics=["Constraints", "Trade-offs"], + source_ids=["official-guide"], + evidence=evidence, + ) + ], + ) + + with pytest.raises(ValueError, match="missing evidence blocks"): + audit.validate_evidence_blocks({"design-basics": "# Design basics"}) + + audit.validate_evidence_blocks( + {"design-basics": "\nEvidence."} + ) From 973f17147666ce151b3bd456a70aebae4e0810a2 Mon Sep 17 00:00:00 2001 From: Troy Scott <846218+troyscott@users.noreply.github.com> Date: Wed, 2 Sep 2026 04:22:24 -0700 Subject: [PATCH 4/6] Complete ingest and transform study domain --- .../published/dp700/chapters/batch-data.md | 263 ++++++++++++++++ .../dp700/chapters/loading-patterns.md | 156 ++++++++++ .../dp700/chapters/streaming-data.md | 191 ++++++++++++ content/published/dp700/coverage.yaml | 285 +++++++++++++++--- tests/content/test_dp700_outline.py | 4 +- 5 files changed, 859 insertions(+), 40 deletions(-) diff --git a/content/published/dp700/chapters/batch-data.md b/content/published/dp700/chapters/batch-data.md index 6443b56..70f8a8f 100644 --- a/content/published/dp700/chapters/batch-data.md +++ b/content/published/dp700/chapters/batch-data.md @@ -113,6 +113,269 @@ Grouping changes grain. Every nonaggregated output column must be a grouping key Quality handling must be observable. Track rule, count, sample, source, run, and disposition. Silently dropping bad rows makes a pipeline look successful while corrupting completeness. +## Data-store decision workshop + + +Choose the store from the dominant access and mutation pattern. A lakehouse keeps +Delta and files in OneLake for open, multi-engine engineering and analytics. A +warehouse provides a relational, governed T-SQL serving experience for +dimensional models and concurrent BI. An Eventhouse is designed for high-rate, +time-oriented ingestion and KQL exploration. An operational database owns +application transactions and point reads/writes; copying or mirroring its data +to an analytical store protects the application from scan-heavy workloads. + +Prerequisites include the right Fabric capacity, workspace/item permissions, +source connectivity, and a retention/security design. Procedure: quantify data +volume and velocity, latency objective, update/delete frequency, transaction +needs, query language, concurrency, file-format interoperability, and consumers; +score each store; prototype the highest-risk query and load. A lakehouse is not +automatically fastest for every SQL dashboard, and a warehouse is not the right +landing zone for arbitrary binary files. + +**Scenario.** A team receives Parquet telemetry plus curated sales dimensions. +Land and engineer open telemetry in a lakehouse, serve governed star-schema BI +from a warehouse when T-SQL concurrency dominates, and use Eventhouse when +subsecond KQL exploration over incoming events is required. Do not force one +store merely to avoid architecture decisions. Troubleshoot a poor fit by +measuring ingestion delay, scan volume, concurrency waits, update complexity, +and duplicated copies. Mini-lab: rank all four stores for (1) images plus JSON, +(2) a finance star schema, (3) live device logs, and (4) order entry, explaining +which requirement eliminated each alternative. + +## Dataflow Gen2 transformation clinic + + +Tool choice combines authoring model and execution locality. Dataflow Gen2 uses +the Power Query engine and a visual sequence of M transformations. A notebook +uses code and Spark for distributed, library-driven processing. T-SQL executes +set-based relational logic near warehouse data. KQL executes pipeline-shaped +event and time-series logic near Eventhouse data. Moving data to a favorite +language can cost more than using the engine already holding it. + + +A disciplined Dataflow Gen2 build follows this order: connect with a governed +connection; profile source columns; set locale-aware types; remove unused +columns and filter rows early; normalize text and null representations; define +keys; merge or append; reshape; calculate derived values; validate row counts +and errors; select a supported destination and update method; publish and +monitor refresh. Name queries and steps for business meaning. Separate staging +queries from destination queries and disable load for helpers when appropriate. + + +Types are semantic, not cosmetic. Text `"01/02/2026"` can mean different dates +by locale; decimal currency and floating point have different guarantees; +joining numeric `42` to text `"42"` can fail or coerce unexpectedly. Assign type +with an explicit locale, inspect conversion errors, and decide whether invalid +values are corrected, replaced, or quarantined. Use trim/clean and case rules +before key comparison. Replace null only when zero or a default has defensible +business meaning; “unknown” and “none” are not generally interchangeable. + + +**Merge** joins columns using key equality and a join kind: left outer preserves +all left rows; inner keeps matches; anti joins isolate missing/unexpected keys. +**Append** stacks rows and aligns columns by name, producing null for absent +fields. Before merging, prove uniqueness on the expected one-side. Afterward, +compare row counts and unmatched keys. Before appending monthly extracts, +standardize names and types and add source-period metadata. A fuzzy merge may +help entity matching but requires thresholds and human-reviewed error cases; it +is not a substitute for a governed key. + + +Group By changes grain and should output only grouping keys plus aggregates. +Pivot turns values into columns and needs an aggregation when a cell has +multiple rows. Unpivot turns repeated columns such as `Jan`, `Feb`, `Mar` into +attribute/value rows and is often safer for evolving periods. Split, extract, +and conditional columns should preserve the original value until validation is +complete. Compute ratios from aggregated numerator and denominator rather than +averaging row-level percentages unless the measure definition explicitly calls +for it. + + +Query folding lets the connector translate supported Power Query steps into a +source query. Place selective filters, projections, and supported joins early; +inspect folding indicators or native query where available. A custom function, +unsupported type conversion, privacy boundary, or nonfoldable source can stop +folding at that step and everything after it. The consequence is often a large +source read and mashup-engine processing—not refresh failure. Use staged/native +queries cautiously, preserve parameterization, and compare source rows scanned +plus refresh duration before and after a change. + + +Choose Dataflow Gen2 when maintainers value visual Power Query, transformations +are tabular, and managed destinations fit. Choose notebooks for complex +algorithms, custom packages, reusable tests, distributed tuning, or mixed +file/table work. Choose T-SQL for relational transformations within a warehouse, +especially joins, window functions, and dimensional loads. Choose KQL for logs, +dynamic payloads, time bins, sequence, and high-rate event analysis. Limitations +include Dataflow folding/connector variability, notebook startup and engineering +overhead, T-SQL's relational boundary, and KQL's different update/serving model. + + +**Worked Dataflow.** Import `Orders` and `Customers`; explicitly type +`OrderDate`, `CustomerId`, and `Amount`; filter to the requested period; left +anti-join Orders to Customers to create an unmatched-key quality query; left +join valid orders to Customers; unpivot monthly target columns; group by Region +and Month; calculate Revenue and OrderCount; load the curated output and quality +output separately. Verify folding through the source filter and projection, +then compare input rows, valid rows, unmatched rows, and output totals. + + +For Dataflow failures, open refresh details and find the first failing query and +step. Separate connection/gateway errors, source schema drift, type conversion, +privacy/firewall constraints, folding/performance, and destination write errors. +Reproduce with a filtered representative sample but retest at scale. For wrong +results, check join cardinality, nulls, locale/types, grouping grain, and +implicit conversions. Mini-lab: build the worked flow, deliberately duplicate a +Customer key and add an invalid date; prove the quality outputs expose both +without silently changing sales totals. + +## Shortcuts and mirroring + + +A OneLake shortcut is metadata that exposes a target path under another OneLake +location. Internal shortcuts reference Fabric/OneLake data; external shortcuts +use a connection to supported storage. Create the connection with least +privilege, choose a unique shortcut name and correct target, and place supported +Delta tables under `Tables` or general content under `Files`. Validate Spark, +SQL, or other intended access and define who owns credential rotation, source +schema changes, cache behavior, and target availability. Deleting the shortcut +deletes the reference, not target data; deleting or moving the target breaks the +reference. Choose a shortcut to avoid copying selected existing data, not when +you need isolation from source outages or a transformed historical snapshot. + +**Failure lab.** Create a test shortcut, query a known row count, revoke source +read, and observe the error without changing the target. Restore permission, +then rename or move a test target and diagnose the reference. Check shortcut +path, connection identity, source ACL, supported format, Delta log, schema +refresh/synchronization, and any cache or acceleration layer. A user may have +access to the shortcut item but lack source data authorization through the +chosen credential path. + + +Mirroring represents a broader, continuously synchronized source. Database +mirroring captures supported operational changes into read-only analytical +Delta representation. Metadata mirroring synchronizes catalog metadata and +uses shortcuts to open data in place. Open mirroring accepts correctly formed +change data written to a Fabric landing zone. Verify source/version support, +network and authentication, required source privileges, key/change semantics, +capacity, and unsupported data types before creation. Select source objects, +start replication, monitor initial snapshot and ongoing lag, and validate counts +and changes—including deletes. + +Choose mirroring for low-friction replication or catalog-wide access when the +source fits; choose shortcut for selected in-place open data; choose pipeline +copy for scheduled/custom mappings and transformations. Do not write directly +to a mirrored target as if it were the operational source. For stale data, +inspect source change capture/retention, replication status, rejected tables or +types, permissions, capacity, and lag. Mini-lab: insert, update, and delete one +source row, record when each appears, pause or break connectivity, and prove the +monitoring evidence identifies the last synchronized point. + +## Pipeline ingestion + + +Copy Activity separates source connection, source query/path, column mapping, +sink behavior, and performance settings. Build a parameterized pipeline with +typed source/destination identifiers, validate them against an allow-list, and +use dynamic expressions only to assemble controlled values. Configure explicit +mapping when schema stability matters. Capture a run ID and source version, +write rejects separately, and validate rows read, copied, skipped, and written. +Grant the pipeline identity only the necessary read and target write rights; +store secrets in managed connections, not parameters or logs. + +Tune from measurements: source query selectivity, partitioned reads, parallel +copies, ForEach concurrency, staging, sink batch/commit behavior, network, and +capacity. More parallelism can throttle a source or create small files. Retry +transient network or throttling faults with backoff; repair authentication, +mapping, constraint, and type errors before rerun. Mini-lab: copy two date +partitions through one parameterized child pipeline, force one deterministic +mapping error, verify only that partition fails, correct it, and rerun without +duplicating the successful partition. + +## Language translation clinic + + +PySpark, T-SQL, and KQL share relational ideas but differ in type systems, +null semantics, physical execution, and durable write behavior. A sound +procedure is filter/select early, make types explicit, validate keys, transform, +write to the engine-native target, and reconcile counts/totals. PySpark can +distribute file/Delta work and expose Spark plans; T-SQL uses relational plans, +constraints, and transactions in a warehouse; KQL excels at time-bounded event +pipelines and dynamic values. Avoid collecting large Spark data to the driver, +unbounded KQL scans, or row-by-row SQL operations. + +**Equivalent join and aggregate:** the PySpark shape is: + +```python +orders.join(customers, "CustomerId", "left").groupBy("Region").agg(sum("Amount")) +``` + +In T-SQL use a `LEFT JOIN` plus `GROUP BY`; in KQL use `join kind=leftouter` +then `summarize`. +Before declaring equivalence, decide how unmatched keys, duplicate customer +rows, null regions, decimal precision, and event-time bounds behave. Mini-lab: +run a six-row fixture containing each edge case in all three engines and compare +the ordered normalized result, not just that each query completed. + +## Denormalization, aggregation, and quality + + +Denormalization creates a consumer-oriented shape by joining descriptive data +at a declared output grain. Confirm every joined lookup has at most one winning +row for each fact at the relevant time. Select only needed attributes and decide +whether history is current-state or as-of-event. Compare fact row count, distinct +fact key, unmatched count, and additive totals before and after. A flattened +table simplifies consumption and can improve scans, but repeats attributes, +increases storage/update cost, and can become inconsistent if refresh timing is +not governed. Mini-lab: introduce two current dimension rows for one key and +show how a $10 fact becomes $20 after the join; repair the winning-row rule. + + +Aggregation replaces detail grain with grouping grain. Define keys and measure +algebra first: additive measures sum across all intended dimensions; +semi-additive measures such as balance may sum across accounts but not time; +nonadditive ratios and distinct counts need special handling. Track numerator +and denominator for recomputable ratios, define null behavior, and avoid +double-counting after joins. Incremental aggregate maintenance must restate a +bucket when a late change affects it. Mini-lab: calculate daily revenue, closing +inventory, average price, and distinct customers; explain why four identical +`SUM` operations would be wrong. + + +Data quality is an explicit disposition system. Define rules for key uniqueness, +required values, valid domains/ranges, referential integrity, and timeliness. +For duplicates, keep a winner only from a trusted ordering and retain evidence +of losers. For missing data, distinguish unknown, not applicable, delayed, and +invalid before defaulting. For late data, use overlap/replay, reopen affected +partitions, resolve late dimensions, and restate downstream aggregates. Each +run records rule ID, evaluated count, failed count/rate, sample/reference, +source version, disposition, and owner. + +Quarantine preserves bad records for correction without contaminating curated +outputs; it is not a permanent trash folder. Define replay and expiry. If a rule +fails suddenly, check source schema/version and code deployment before blaming +the data. Mini-lab: process a fixture with one exact duplicate, one newer update, +one null amount, one unknown customer, and one yesterday event arriving today; +predict the winner, quarantine, unknown-member handling, and which daily +aggregate must be restated. + + +Shortcuts reference selected data in place; mirroring continuously represents a +supported database or catalog; Copy Activity moves a bounded selection. Merge +joins columns; append stacks rows. Query folding is source execution, not merely +successful refresh. Denormalization changes shape and may preserve grain; +aggregation intentionally changes grain. A null-replacement step is not a data +quality strategy unless the business meaning and audit trail are defined. + + +**Recall.** Which store fits concurrent star-schema T-SQL? What evidence proves +a Dataflow filter folded? Why can a left join increase row count? How does a +shortcut's credential path differ from copied data? Which mirroring mode uses a +published change-file contract? Why should aggregate ratios retain components? +For practice, take one customer/order dataset through Dataflow, PySpark, SQL, +and KQL; document types, joins, quality dispositions, output grain, and expected +totals before comparing results. + ## Exam distinctions - Merge joins columns; append stacks rows. diff --git a/content/published/dp700/chapters/loading-patterns.md b/content/published/dp700/chapters/loading-patterns.md index 5469284..1186eca 100644 --- a/content/published/dp700/chapters/loading-patterns.md +++ b/content/published/dp700/chapters/loading-patterns.md @@ -70,6 +70,162 @@ A reliable pattern is: “Exactly once” should be treated as an end-to-end property. A streaming engine checkpoint cannot prevent duplication if the source reuses IDs or the sink performs non-idempotent side effects. Choose the watermark from measured lateness: too short drops legitimate events; too long retains more state and delays final results. +## Full and incremental load field guide + + +Separate four decisions: **selection** chooses source changes, **transport** moves +them, **application** changes the target, and **control state** records progress. +A full load selects the complete scope and commonly replaces, truncates/reloads, +or swaps a rebuilt target. An incremental load selects a bounded change set and +applies inserts, updates, and possibly deletes. A pipeline, notebook, or SQL +procedure can participate in either pattern; the tool does not define the +semantics. + + +For timestamp incrementals, read the last committed watermark `W0`, choose an +upper bound `W1` at run start, and extract `ModifiedAt > W0 AND ModifiedAt <= +W1`, usually with a justified overlap before `W0`. Land the extract with source +version and run ID, validate it, select one winning version per business key, +and apply it idempotently. Reconcile counts and control totals. Only after the +target commit succeeds, store `W1` as the new committed watermark. Preserve the +staged range until the run is recoverable. + + +Use a full load for small datasets, first loads, recovery, or sources without a +trustworthy change signal. Use CDC when inserts, updates, deletes, and ordering +must be captured. Use a timestamp or version watermark when it is stable and +indexed. Use partition discovery for append-oriented files, but include a late +partition policy. Incremental loading saves work at the cost of state, delete +handling, and reconciliation. Periodic bounded or full comparisons are the +safety net against silent gaps. + + +**Worked recovery.** The committed watermark is 10:00. A run chooses 10:15, +loads staging, merges the target, and crashes before saving control state. On +retry it reads the same interval, ranks duplicate staged rows by source version, +and performs the same keyed upsert; the result remains one current row per key. +If the watermark had advanced before the merge, the retry would begin after +10:15 and permanently miss the interval. This is why progress follows durable +target success. + + +For missing rows, inspect source change retention, extraction bounds, time-zone +conversion, precision ties, overlap, rejected rows, merge predicates, and when +the watermark advanced. For duplicates, inspect source-key uniqueness, winning +version logic, target constraints, and concurrent writers. For slow loads, +measure source scan, transfer, staging, target application, and validation +separately. Mini-lab: run the same increment twice, fail once before target +commit and once after target commit but before control update, and prove the +final target and watermark are correct in both cases. + +## Dimensional loading field guide + + +Dimensional loading preserves a declared analytical grain while translating +source business keys into warehouse surrogate keys. Dimensions supply durable +descriptive context; facts record events or measurements at the chosen grain. +A Type 1 change corrects or replaces current attributes. A Type 2 change creates +a new version so a historical fact can resolve the attributes valid at its +business time. The order is normally stage → validate → load dimensions → +resolve keys → load facts → reconcile. + + +Profile each dimension business key for nulls and duplicates. Insert a stable +unknown member. For Type 1, update changed tracked columns. For Type 2, compare +a normalized attribute hash or explicit columns; expire the current record and +insert a new surrogate-keyed row with nonoverlapping validity. Then join staged +facts to the dimension on business key and event time where history applies. +Reject or route facts that violate grain, and use the unknown/inferred member +for permitted late dimensions. Validate foreign keys and totals before publish. + + +Use Type 1 for corrections or attributes whose history has no analytical value; +use Type 2 when reports must reproduce the past. Do not make every attribute +Type 2: it increases row count and fact-key lookup complexity. Choose an +accumulating snapshot for a process whose milestones update one logical row, a +periodic snapshot for measurements at regular periods, and a transaction fact +for individual events. A degenerate identifier belongs on the fact when it has +no useful dimension attributes. + + +**Worked late member.** Order 501 arrives for customer `C42`, but the customer +dimension has no `C42`. The fact load uses surrogate key 0, the governed unknown +member, and records the unresolved business key. When `C42` arrives, the +dimension receives a real surrogate key. Policy determines whether the order is +restated to that key or remains unknown for the historical load. Silently +dropping the order would preserve referential cleanliness by destroying +completeness—an unacceptable trade. + + +If fact totals multiply, verify the fact grain and dimension join is one winning +row per business key and event time. If a Type 2 dimension has two current rows, +inspect concurrent loads and enforce a serialized/keyed transition. If many +facts use the unknown key, alert on rate and age, then inspect dimension timing, +normalization, and source keys. Mini-lab: load one customer, change a Type 1 +attribute and a Type 2 attribute, then load facts before and after the change; +write the expected surrogate keys and historical report output before running. + +## Streaming load field guide + + +A durable streaming load usually has bronze, validated, and serving boundaries. +The raw/bronze layer preserves replayable events and source metadata. Stateful +processing uses event time, event identity, checkpoints, watermarks, and bounded +windows. The serving layer uses an idempotent Delta or Eventhouse write pattern. +Recovery correctness spans source replay plus processing state plus sink +semantics; no single “exactly once” checkbox proves the whole chain. + + +Define schema, event ID, event-time column, expected lateness, source retention, +and replay ownership. Land raw events before lossy filtering when possible. +Validate and quarantine malformed records. Apply a watermark chosen from an +observed lateness distribution, deduplicate within a bounded horizon, and keep a +unique checkpoint for each query. Write stable keys or deterministic partitions +to the sink and record batch/offset information. Monitor input rate, processing +rate, lag, state size, late drops, bad events, and sink failures. + + +Use a short watermark only when the business accepts more late-event loss in +exchange for smaller state and earlier final output. Use `foreachBatch` when +micro-batches need existing batch logic, but key writes by batch ID or event +identity. Retain raw data long enough to replay beyond checkpoint corruption or +logic defects. A dead-letter stream preserves diagnosable bad input; it should +include reason, original payload reference, source position, and processing +version without leaking sensitive data into broad logs. + + +**Worked recovery.** A device event reaches bronze twice with one `event_id`. +The silver query uses a 30-minute watermark and deduplicates that ID, then +writes by deterministic key. After a checkpointed restart, the source replays a +range, but already committed progress and the idempotent sink prevent a second +logical output. An event arriving 45 minutes late is routed or counted according +to policy; the system does not pretend it never existed. + + +When lag grows, compare input and processing rates, micro-batch duration, +shuffle/state size, sink latency, and capacity. When counts drift, inspect event +IDs, watermark delay, late-drop metrics, checkpoint history, replay range, and +sink idempotency. Never delete a checkpoint merely to clear an error without a +replay plan. Mini-lab: process an on-time event, a duplicate, a late-within- +watermark event, a too-late event, and a malformed event; predict bronze, +silver, quarantine, and aggregate results before execution. + + +Exam questions often mix similarly named state. A batch high-water mark bounds +source changes; an event-time watermark bounds streaming lateness and state; a +checkpoint stores streaming recovery progress. Surrogate keys are warehouse +identities; business keys come from source domains. Checkpoint recovery does not +repair non-idempotent side effects, and `MERGE` does not repair duplicate source +keys without a winning-row rule. + + +**Recall.** Why select the upper extraction bound at run start? When can a full +load be safer than incremental? Why must dimensions load before facts? Contrast +Type 1 and Type 2 for an address correction. What data must survive to replay a +stream after logic changes? For practice, write control records for one batch +increment and one streaming micro-batch, including state before, durable effects, +validation evidence, and state after. + ## Exam distinctions - Full versus incremental describes selection and application, not a specific Fabric tool. diff --git a/content/published/dp700/chapters/streaming-data.md b/content/published/dp700/chapters/streaming-data.md index 4c40ee1..0a8f179 100644 --- a/content/published/dp700/chapters/streaming-data.md +++ b/content/published/dp700/chapters/streaming-data.md @@ -108,6 +108,197 @@ Filter early, select only needed columns, parse dynamic fields deliberately, and Windows should use event time when results describe when events happened. A watermark states how late the engine expects events and bounds retained state; it does not reorder the entire infinite stream or guarantee that later records will be accepted. +## Streaming architecture decisions + + +Select an engine by answering where events enter, where durable state lives, +what transformations require, and how results are served. Eventstreams provides +connector-driven ingestion, visual transformations, routing, and derived +streams. Eventhouse stores/indexes high-rate events and serves KQL. Spark +Structured Streaming provides code-first stateful processing over supported +sources and Delta sinks. Pipelines can deploy, schedule bounded reconciliation, +or coordinate companion jobs; they do not continuously evaluate each event. + +Prerequisites include source authorization, destination write access, capacity, +network reachability, schema and event-time contract, retention, and an +operational owner. Prefer Eventstreams for accessible routing and straightforward +stream transforms; Eventhouse/KQL for low-latency log and time-series serving; +Spark for custom libraries, complex state, and Delta-centric engineering. A +combined design can route raw events through Eventstreams to Eventhouse for live +operations and OneLake/Delta for durable replay and Spark enrichment. + +**Scenario.** Correlating device events across 30 minutes with a custom model +points to Spark; filtering and routing by device type without code points to +Eventstreams; ad hoc “last 15 minutes by firmware” queries point to Eventhouse. +Diagnose the architecture by measuring end-to-end latency, input/processing +rate, state growth, query concurrency, retention, and operator skill—not by +asking which product is newest. Mini-lab: draw two valid architectures for the +same stream, one low-code and one code-first, and identify recovery state, +serving store, and reconciliation in each. + +## Native tables, shortcuts, and acceleration + + +A native Eventhouse table ingests records into Eventhouse-controlled indexed +storage. It supports predictable KQL query performance and native policies at +the cost of ingestion, retention, and another representation. A OneLake +shortcut exposes supported Delta data as an external table without copying it; +query performance and availability depend on file layout, source, and external- +table support. Verify Delta format, schema, source permission/connection, and +the exact KQL access syntax before selecting a shortcut. + +Choose native when continuous high-rate ingestion, low-latency concurrent KQL, +and native table features dominate. Choose a standard shortcut for historical +or shared open Delta data when avoiding movement matters and direct-read +performance is acceptable. If external queries are empty or fail, check target +path, `_delta_log`, schema compatibility, connection identity, source ACL, +partition/file health, and whether the query uses the external table correctly. +Mini-lab: query the same small Delta table through a shortcut and a native copy; +compare freshness boundary, features, rows scanned, latency, and ownership. + + +Query acceleration adds an optimized cache for a configured recent period of a +OneLake shortcut. Choose the period from observed query predicates—for example, +seven days when most dashboards read seven days—not from total source retention. +Enable it on a supported shortcut, allow cache population, and inspect the +documented status/metrics before measuring warm and cold queries. Account for +premium cache/storage use, refresh lag, schema evolution, and external-table +feature limits. + +Acceleration is preferable when a hot recent slice is repeatedly queried or +joined with native live data and the measured benefit justifies cost. Standard +shortcut remains preferable for infrequent, broad, or cost-sensitive access. +Native ingestion remains preferable when update policies, materialized views, +or other unsupported external-table features are required. For poor performance, +verify predicate time range overlaps the cache period, cache readiness, schema, +file layout, capacity, and query shape. Mini-lab: measure a 24-hour and 90-day +query before and after a seven-day cache; explain why only one should materially +benefit. + +## Eventstream implementation + + +Define source schema, stable event ID, event-time field, units, expected rate, +and bad-event disposition. Create and authenticate the source, preview events, +normalize fields and types, filter unnecessary events early, then add derived +streams for reusable branches. Use group/aggregate and window operations only +after choosing event time and lateness behavior. Route raw, curated, and poison +branches to independently governed destinations. Validate one known event along +every intended route and monitor input, output, dropped/invalid events, latency, +and destination status. + +**Worked route.** An IoT source branches raw events to a durable lakehouse, +valid temperature readings to Eventhouse, and invalid schema or out-of-range +values to a restricted quarantine destination with reason metadata. A derived +stream calculates five-minute device averages. Do not discard raw data merely +because the visual transform works; retained source events support replay after +logic changes. If output stops, walk source connection → incoming rate → each +operator's schema/output → route condition → destination connection/capacity. +Mini-lab: inject a valid event, wrong type, duplicate ID, late timestamp, and +out-of-range value and predict every branch before observing it. + +## Structured Streaming implementation + + +A Structured Streaming query has a logical plan, trigger/micro-batch execution, +state store when needed, sink, and checkpoint. Give each deployed query a +stable unique checkpoint path and protect it like operational state. Use +explicit schema for production inputs. Add event-time watermark before bounded +deduplication or stateful aggregation, choose an output mode supported by the +operation/sink, and write to a transactional Delta target. Monitor query progress +JSON, batch duration, input/processed rows per second, state rows/bytes, and sink +commit failures. + +`foreachBatch` receives a DataFrame plus `batch_id` and is useful for `MERGE` or +multi-target batch logic. Make the body idempotent using `batch_id` control or +stable business/event keys; a failed micro-batch can be retried. Avoid calling +unbounded actions repeatedly, reusing a checkpoint for a changed incompatible +query, or placing checkpoints in temporary locations. If recovery fails, retain +the old checkpoint and source offsets, determine whether the query/state schema +changed, and create a controlled replay to a validated target rather than +deleting state and hoping. + +**Mini-lab.** Run two micro-batches containing a duplicate ID and a late event, +stop gracefully, restart from the same checkpoint, and prove no logical output +duplicates. Then point a test copy at a fresh checkpoint and observe replay. +Compare append, update, and complete output semantics for one windowed count. +If processing falls behind, use progress metrics and Spark UI to distinguish +source rate, shuffle/state, skew, sink latency, and capacity. + +## KQL transformation implementation + + +KQL reads as an operator pipeline. Start from the smallest time/data scope; +`where` early, `project` required columns, parse dynamic JSON only where needed, +`extend` derived fields, and `summarize` at the declared grain. Use `join` with +the smaller side and a time/key bound, and use `materialize()` only when one +bounded intermediate is reused and measurement supports caching it. Retain a +request ID and inspect diagnostics for failed or slow queries. + +```kusto +DeviceEvents +| where EventTime between (ago(30m) .. now()) +| where isnotempty(DeviceId) +| extend Payload = todynamic(RawPayload) +| extend Temperature = todouble(Payload.temperature) +| where isnotnull(Temperature) +| summarize Events=count(), AvgTemp=avg(Temperature), + P95=percentile(Temperature, 95) + by DeviceId, bin(EventTime, 5m) +``` + +Validate input count, parse failures, filtered count, distinct devices, and +bucket totals. KQL null/empty and dynamic conversion rules deserve explicit +tests. An empty result may be correct because of time range or ingestion delay; +check table, database, time field, time zone, ingestion status, and filters +before changing syntax. Mini-lab: add malformed JSON, null temperature, and one +event outside the range; predict which metric accounts for each exclusion. + +## Windowing implementation + + +A tumbling window of width five minutes assigns each event to one nonoverlapping +bucket. A hopping window of width ten minutes and hop two minutes assigns one +event to up to five overlapping windows. A session window groups events for one +key until an inactivity gap closes the session. Sliding is sometimes used +conceptually for continuously moving boundaries; verify the engine's exact +operator terminology. Window choice follows the question: accounting totals +often tumble, moving signals hop, and user/device visits use sessions. + +Suppose events for device A occur at 10:01, 10:04, 10:06, and 10:20. Five-minute +tumbling windows place the first two together and 10:06 separately. A session +gap of five minutes groups 10:01/10:04/10:06 and starts a new session at 10:20. +A 10-minute hopping window every five minutes can count an event in two outputs. +This overlap is expected, so summing hopping-window outputs again usually +double-counts. + +Use event time for business windows, define time zone, establish watermark from +lateness evidence, and specify correction policy after finalization. A longer +watermark retains state and delays append-final output; a short watermark drops +or excludes more late data. Diagnose missing window counts through event-time +parse, watermark, window boundaries, time zone, late metrics, checkpoint state, +and output mode. Mini-lab: hand-calculate the four events above for tumbling, +hopping, and session windows, then add one event arriving 12 minutes late under +a 10-minute watermark. + + +Native tables ingest/index; standard shortcuts read Delta in place; accelerated +shortcuts cache a recent external slice but retain external-table limitations. +Eventstreams shape/route, Spark executes custom stateful code, KQL queries and +analyzes event stores, and Activator takes actions from conditions. Event time +drives business windows; processing time measures observation. Watermark bounds +lateness/state; checkpoint supports recovery; neither alone makes a sink +idempotent. + + +**Recall.** When is a native Eventhouse table worth another representation? Why +can a seven-day acceleration cache fail to help a 90-day query? What must every +Eventstream poison route retain? Why is each Spark query's checkpoint unique? +How can a KQL query distinguish parse failures from filtered values? How many +10-minute windows with a two-minute hop can contain one event? Build a one-page +operational contract listing event ID, event time, allowed lateness, checkpoint, +raw retention, sink key, replay method, metrics, and owner. + ## Exam distinctions - Native Eventhouse tables ingest and index; shortcuts query supported data in place. diff --git a/content/published/dp700/coverage.yaml b/content/published/dp700/coverage.yaml index 25c3249..a488ba7 100644 --- a/content/published/dp700/coverage.yaml +++ b/content/published/dp700/coverage.yaml @@ -309,118 +309,327 @@ objectives: scenario_lab: patterns-diagnostics - objective_id: ingest.loading.full-incremental chapter_id: loading-patterns - status: draft + status: complete subtopics: [Full refresh, Watermarks, CDC, Overlap windows, Upserts, Deletes, Idempotency, Reconciliation] source_ids: [dp700-study-guide, data-movement-decision-guide] - gaps: [Add Fabric pipeline and notebook implementations with failure recovery] + gaps: [] + evidence: + conceptual_model: full-incremental-model + prerequisites: full-incremental-operations + procedure: full-incremental-operations + decision_guidance: full-incremental-decisions + worked_example: full-incremental-example + limitations: full-incremental-decisions + troubleshooting: full-incremental-diagnostics + exam_distinctions: loading-exam-distinctions + recall_questions: loading-recall-lab + scenario_lab: full-incremental-diagnostics - objective_id: ingest.loading.dimensional chapter_id: loading-patterns - status: draft + status: complete subtopics: [Grain, Surrogate keys, Dimension-first loading, Facts, SCD Type 1, SCD Type 2, Inferred members, Referential checks] source_ids: [dp700-study-guide, dimensional-model-loading] - gaps: [Add complete dimension/fact load example and late-member lab] + gaps: [] + evidence: + conceptual_model: dimensional-model + prerequisites: dimensional-operations + procedure: dimensional-operations + decision_guidance: dimensional-decisions + worked_example: dimensional-example + limitations: dimensional-decisions + troubleshooting: dimensional-diagnostics + exam_distinctions: loading-exam-distinctions + recall_questions: loading-recall-lab + scenario_lab: dimensional-diagnostics - objective_id: ingest.loading.streaming chapter_id: loading-patterns - status: draft + status: complete subtopics: [Raw retention, Schema validation, Deduplication, Event time, Watermarks, Checkpoints, Idempotent sinks, Reconciliation] source_ids: [dp700-study-guide, structured-streaming-state] - gaps: [Add Fabric architecture, checkpoint recovery, and replay scenario] + gaps: [] + evidence: + conceptual_model: streaming-load-model + prerequisites: streaming-load-operations + procedure: streaming-load-operations + decision_guidance: streaming-load-decisions + worked_example: streaming-load-example + limitations: streaming-load-decisions + troubleshooting: streaming-load-diagnostics + exam_distinctions: loading-exam-distinctions + recall_questions: loading-recall-lab + scenario_lab: streaming-load-diagnostics - objective_id: ingest.batch.store chapter_id: batch-data - status: draft + status: complete subtopics: [Lakehouse, Warehouse, Eventhouse, Operational database, Open formats, Transactions, Concurrency, Serving patterns] source_ids: [dp700-study-guide, data-movement-decision-guide] - gaps: [Add workload decision matrix and migration scenarios] + gaps: [] + evidence: + conceptual_model: batch-store-comprehensive + prerequisites: batch-store-comprehensive + procedure: batch-store-comprehensive + decision_guidance: batch-store-comprehensive + worked_example: batch-store-comprehensive + limitations: batch-store-comprehensive + troubleshooting: batch-store-comprehensive + exam_distinctions: batch-exam-distinctions + recall_questions: batch-recall-lab + scenario_lab: batch-store-comprehensive - objective_id: ingest.batch.transform-tool chapter_id: batch-data - status: draft + status: complete subtopics: [Dataflow Gen2, Notebooks, T-SQL, KQL, Personas, Engine locality, Scale, Testing and maintainability] source_ids: [dp700-study-guide, dataflows-gen2-overview] - gaps: [Add detailed tool-comparison cases and equivalent transformations] + gaps: [] + evidence: + conceptual_model: transform-tool-model + prerequisites: transform-tool-operations + procedure: transform-tool-operations + decision_guidance: transform-tool-decisions + worked_example: transform-tool-example + limitations: transform-tool-decisions + troubleshooting: transform-tool-diagnostics + exam_distinctions: batch-exam-distinctions + recall_questions: batch-recall-lab + scenario_lab: transform-tool-diagnostics - objective_id: ingest.batch.shortcuts chapter_id: batch-data - status: draft + status: complete subtopics: [Internal shortcuts, External shortcuts, Tables versus Files, Connections, Caching, Schema synchronization, Security, Deletion behavior] source_ids: [dp700-study-guide, onelake-shortcuts] - gaps: [Add creation and management procedures plus broken-target lab] + gaps: [] + evidence: + conceptual_model: shortcuts-comprehensive + prerequisites: shortcuts-comprehensive + procedure: shortcuts-comprehensive + decision_guidance: shortcuts-comprehensive + worked_example: shortcuts-comprehensive + limitations: shortcuts-comprehensive + troubleshooting: shortcuts-comprehensive + exam_distinctions: batch-exam-distinctions + recall_questions: batch-recall-lab + scenario_lab: shortcuts-comprehensive - objective_id: ingest.batch.mirroring chapter_id: batch-data - status: draft + status: complete subtopics: [Database mirroring, Metadata mirroring, Open mirroring, Replication, Source support, Read-only targets, Monitoring] source_ids: [dp700-study-guide, fabric-mirroring] - gaps: [Add implementation workflow, selection matrix, and failure recovery] + gaps: [] + evidence: + conceptual_model: mirroring-comprehensive + prerequisites: mirroring-comprehensive + procedure: mirroring-comprehensive + decision_guidance: mirroring-comprehensive + worked_example: mirroring-comprehensive + limitations: mirroring-comprehensive + troubleshooting: mirroring-comprehensive + exam_distinctions: batch-exam-distinctions + recall_questions: batch-recall-lab + scenario_lab: mirroring-comprehensive - objective_id: ingest.batch.pipelines chapter_id: batch-data - status: draft + status: complete subtopics: [Copy Activity, Connectors, Mapping, Parameters, Parallelism, Staging, Retry, Metrics and rejects] source_ids: [dp700-study-guide, data-movement-decision-guide] - gaps: [Add parameterized copy walkthrough and performance exercise] + gaps: [] + evidence: + conceptual_model: pipelines-comprehensive + prerequisites: pipelines-comprehensive + procedure: pipelines-comprehensive + decision_guidance: pipelines-comprehensive + worked_example: pipelines-comprehensive + limitations: pipelines-comprehensive + troubleshooting: pipelines-comprehensive + exam_distinctions: batch-exam-distinctions + recall_questions: batch-recall-lab + scenario_lab: pipelines-comprehensive - objective_id: ingest.batch.languages chapter_id: batch-data - status: draft + status: complete subtopics: [PySpark DataFrames, Spark SQL, T-SQL, KQL pipelines, Types, Null behavior, Engine-specific optimization] source_ids: [dp700-study-guide, dataflows-gen2-overview] - gaps: [Add multi-step equivalent transformations and validation outputs] + gaps: [] + evidence: + conceptual_model: languages-comprehensive + prerequisites: languages-comprehensive + procedure: languages-comprehensive + decision_guidance: languages-comprehensive + worked_example: languages-comprehensive + limitations: languages-comprehensive + troubleshooting: languages-comprehensive + exam_distinctions: batch-exam-distinctions + recall_questions: batch-recall-lab + scenario_lab: languages-comprehensive - objective_id: ingest.batch.denormalize chapter_id: batch-data - status: draft + status: complete subtopics: [Output grain, Join cardinality, Flattening, Repeated attributes, Performance trade-offs, Validation] source_ids: [dp700-study-guide, dimensional-model-loading] - gaps: [Map dimensional source to chapter and add worked denormalization example] + gaps: [] + evidence: + conceptual_model: denormalize-comprehensive + prerequisites: denormalize-comprehensive + procedure: denormalize-comprehensive + decision_guidance: denormalize-comprehensive + worked_example: denormalize-comprehensive + limitations: denormalize-comprehensive + troubleshooting: denormalize-comprehensive + exam_distinctions: batch-exam-distinctions + recall_questions: batch-recall-lab + scenario_lab: denormalize-comprehensive - objective_id: ingest.batch.aggregate chapter_id: batch-data - status: draft + status: complete subtopics: [Grouping keys, Grain changes, Additive measures, Semi-additive measures, Ratios, Nulls, Incremental aggregates] source_ids: [dp700-study-guide, dataflows-gen2-overview] - gaps: [Add Power Query, SQL, PySpark, and KQL aggregation examples] + gaps: [] + evidence: + conceptual_model: aggregate-comprehensive + prerequisites: aggregate-comprehensive + procedure: aggregate-comprehensive + decision_guidance: aggregate-comprehensive + worked_example: aggregate-comprehensive + limitations: aggregate-comprehensive + troubleshooting: aggregate-comprehensive + exam_distinctions: batch-exam-distinctions + recall_questions: batch-recall-lab + scenario_lab: aggregate-comprehensive - objective_id: ingest.batch.data-quality chapter_id: batch-data - status: draft + status: complete subtopics: [Duplicate keys, Missing values, Invalid values, Late records, Quarantine, Imputation, Restatement, Quality metrics] source_ids: [dp700-study-guide, dataflows-gen2-overview] - gaps: [Add executable quality rules and late-arrival correction lab] + gaps: [] + evidence: + conceptual_model: data-quality-comprehensive + prerequisites: data-quality-comprehensive + procedure: data-quality-comprehensive + decision_guidance: data-quality-comprehensive + worked_example: data-quality-comprehensive + limitations: data-quality-comprehensive + troubleshooting: data-quality-comprehensive + exam_distinctions: batch-exam-distinctions + recall_questions: batch-recall-lab + scenario_lab: data-quality-comprehensive - objective_id: ingest.streaming.engine chapter_id: streaming-data - status: draft + status: complete subtopics: [Eventstreams, Eventhouse, Spark Structured Streaming, Latency, Volume, State, Serving and operations] source_ids: [dp700-study-guide, real-time-intelligence-overview, structured-streaming-state] - gaps: [Add decision cases and end-to-end architecture comparison] + gaps: [] + evidence: + conceptual_model: streaming-engine-comprehensive + prerequisites: streaming-engine-comprehensive + procedure: streaming-engine-comprehensive + decision_guidance: streaming-engine-comprehensive + worked_example: streaming-engine-comprehensive + limitations: streaming-engine-comprehensive + troubleshooting: streaming-engine-comprehensive + exam_distinctions: streaming-exam-distinctions + recall_questions: streaming-recall-lab + scenario_lab: streaming-engine-comprehensive - objective_id: ingest.streaming.native-shortcut chapter_id: streaming-data - status: draft + status: complete subtopics: [Native ingestion, Indexed storage, External Delta, Feature support, Latency, Cost, Source availability] source_ids: [dp700-study-guide, real-time-intelligence-overview] - gaps: [Add direct shortcut source mapping and measured selection scenario] + gaps: [] + evidence: + conceptual_model: native-shortcut-comprehensive + prerequisites: native-shortcut-comprehensive + procedure: native-shortcut-comprehensive + decision_guidance: native-shortcut-comprehensive + worked_example: native-shortcut-comprehensive + limitations: native-shortcut-comprehensive + troubleshooting: native-shortcut-comprehensive + exam_distinctions: streaming-exam-distinctions + recall_questions: streaming-recall-lab + scenario_lab: native-shortcut-comprehensive - objective_id: ingest.streaming.acceleration chapter_id: streaming-data - status: draft + status: complete subtopics: [Standard shortcut, Accelerated cache, Cache period, Performance, Billing, Schema changes, External-table limitations] source_ids: [dp700-study-guide, query-acceleration-overview] - gaps: [Add configuration procedure, monitoring commands, and cost/performance lab] + gaps: [] + evidence: + conceptual_model: acceleration-comprehensive + prerequisites: acceleration-comprehensive + procedure: acceleration-comprehensive + decision_guidance: acceleration-comprehensive + worked_example: acceleration-comprehensive + limitations: acceleration-comprehensive + troubleshooting: acceleration-comprehensive + exam_distinctions: streaming-exam-distinctions + recall_questions: streaming-recall-lab + scenario_lab: acceleration-comprehensive - objective_id: ingest.streaming.eventstreams chapter_id: streaming-data - status: draft + status: complete subtopics: [Sources, Transformations, Filtering, Aggregation, Derived streams, Routing, Destinations, Monitoring] source_ids: [dp700-study-guide, real-time-intelligence-overview] - gaps: [Add full Eventstream construction walkthrough and poison-event route] + gaps: [] + evidence: + conceptual_model: eventstreams-comprehensive + prerequisites: eventstreams-comprehensive + procedure: eventstreams-comprehensive + decision_guidance: eventstreams-comprehensive + worked_example: eventstreams-comprehensive + limitations: eventstreams-comprehensive + troubleshooting: eventstreams-comprehensive + exam_distinctions: streaming-exam-distinctions + recall_questions: streaming-recall-lab + scenario_lab: eventstreams-comprehensive - objective_id: ingest.streaming.spark chapter_id: streaming-data - status: draft + status: complete subtopics: [Sources, Sinks, Checkpoints, Output modes, foreachBatch, Stateful operations, Recovery, Observability] source_ids: [dp700-study-guide, structured-streaming-state] - gaps: [Add runnable Fabric notebook pattern and checkpoint recovery lab] + gaps: [] + evidence: + conceptual_model: spark-streaming-comprehensive + prerequisites: spark-streaming-comprehensive + procedure: spark-streaming-comprehensive + decision_guidance: spark-streaming-comprehensive + worked_example: spark-streaming-comprehensive + limitations: spark-streaming-comprehensive + troubleshooting: spark-streaming-comprehensive + exam_distinctions: streaming-exam-distinctions + recall_questions: streaming-recall-lab + scenario_lab: spark-streaming-comprehensive - objective_id: ingest.streaming.kql chapter_id: streaming-data - status: draft + status: complete subtopics: [Filtering, Projection, Dynamic parsing, Summarize, Joins, Time series, Materialization, Query optimization] source_ids: [dp700-study-guide, real-time-intelligence-overview] - gaps: [Add complete KQL transformation sequence and validation scenario] + gaps: [] + evidence: + conceptual_model: kql-comprehensive + prerequisites: kql-comprehensive + procedure: kql-comprehensive + decision_guidance: kql-comprehensive + worked_example: kql-comprehensive + limitations: kql-comprehensive + troubleshooting: kql-comprehensive + exam_distinctions: streaming-exam-distinctions + recall_questions: streaming-recall-lab + scenario_lab: kql-comprehensive - objective_id: ingest.streaming.windows chapter_id: streaming-data - status: draft + status: complete subtopics: [Event time, Tumbling windows, Hopping windows, Sliding windows, Session windows, Watermarks, Late events] source_ids: [dp700-study-guide, structured-streaming-state] - gaps: [Add Eventstream, Spark, and KQL window examples with expected results] + gaps: [] + evidence: + conceptual_model: windows-comprehensive + prerequisites: windows-comprehensive + procedure: windows-comprehensive + decision_guidance: windows-comprehensive + worked_example: windows-comprehensive + limitations: windows-comprehensive + troubleshooting: windows-comprehensive + exam_distinctions: streaming-exam-distinctions + recall_questions: streaming-recall-lab + scenario_lab: windows-comprehensive - objective_id: monitor.items.ingestion chapter_id: monitor-items status: draft diff --git a/tests/content/test_dp700_outline.py b/tests/content/test_dp700_outline.py index 2243f4b..d7ae373 100644 --- a/tests/content/test_dp700_outline.py +++ b/tests/content/test_dp700_outline.py @@ -49,8 +49,8 @@ def test_dp700_coverage_audit_records_every_objective_and_evidence() -> None: } assert all(len(coverage.subtopics) >= 2 for coverage in audit.objectives) assert all(coverage.source_ids for coverage in audit.objectives) - assert sum(coverage.status == "complete" for coverage in audit.objectives) == 18 - assert sum(coverage.status == "draft" for coverage in audit.objectives) == 36 + assert sum(coverage.status == "complete" for coverage in audit.objectives) == 37 + assert sum(coverage.status == "draft" for coverage in audit.objectives) == 17 assert all( coverage.evidence.is_complete and not coverage.gaps if coverage.status == "complete" From 572c6e4d3188137bee5dd800cdbe5b988d04c993 Mon Sep 17 00:00:00 2001 From: Troy Scott <846218+troyscott@users.noreply.github.com> Date: Wed, 2 Sep 2026 04:29:42 -0700 Subject: [PATCH 5/6] Complete monitor and optimize study domain --- .../published/dp700/chapters/monitor-items.md | 160 +++++++++++ .../dp700/chapters/optimize-performance.md | 221 +++++++++++++++ .../dp700/chapters/resolve-errors.md | 192 +++++++++++++ content/published/dp700/coverage.yaml | 255 +++++++++++++++--- tests/content/test_dp700_outline.py | 4 +- 5 files changed, 796 insertions(+), 36 deletions(-) diff --git a/content/published/dp700/chapters/monitor-items.md b/content/published/dp700/chapters/monitor-items.md index 9a974f3..3abbf25 100644 --- a/content/published/dp700/chapters/monitor-items.md +++ b/content/published/dp700/chapters/monitor-items.md @@ -62,6 +62,166 @@ Prefer sustained conditions over noisy single samples: “freshness lag exceeds | Semantic refresh failed | Published model is stale | Refresh detail and upstream completion | | Streaming lag rising continuously | Backpressure or destination issue | Input/output rate and resource saturation | +## Monitoring model: outcome, execution, resource + + +Every production data product needs three layers of evidence. **Outcome** asks +whether the published data is fresh, complete, valid, and available. +**Execution** asks whether the pipeline, refresh, query, or stream ran as +designed. **Resource** asks whether capacity, compute, gateway, network, source, +or destination limited it. Monitoring hub is an entry point for supported jobs; +item run detail and workspace monitoring provide deeper execution evidence; +the Capacity Metrics app provides capacity evidence. None alone proves the +business outcome. + +Define a service-level indicator with an owner and unit, then a target and +measurement window. Examples include “99% of hourly partitions published within +20 minutes,” “fewer than 0.1% rejected records per daily load,” or “semantic +model ready by 07:00 on business days.” Record UTC timestamps plus business +time-zone meaning, stable item/run IDs, source version/watermark, code version, +and target version so evidence can be correlated. + +## Ingestion monitoring field guide + + +Monitor batch ingestion as a control equation: selected source rows/files = +accepted + quarantined + explicitly ignored, and accepted should reconcile to +the target application semantics. Record bytes/rows read and written, files, +duration, throughput, watermark range, retries, rejects, and publish time. For +streaming, add input and processed rates, end-to-end/event-time lag, backlog, +late or dropped events, state size, checkpoint progress, destination failures, +and newest published event time. + +Procedure: open Monitoring hub, filter to item/status/time, capture the run ID, +then drill into pipeline Copy details, Eventstream graph, Eventhouse ingestion, +or Spark progress. Compare with source controls and target counts rather than +reading “Succeeded” in isolation. A run identity needs access to monitoring and +the item; sensitive payloads should not be copied into broadly accessible logs. +Retain enough history to compare the same hour/day and detect gradual drift. + +**Worked incident.** A pipeline reports success and copied one million rows, but +the source control total is 1.05 million. The watermark advanced. Evidence shows +50,000 conversion rejects that the job treated as tolerated. Operational success +is green; completeness is red. Freeze or correct the watermark according to the +replay design, repair the mapping, replay the bounded interval, reconcile, and +add a reject-rate alert. Do not rerun the entire source blindly. + +Troubleshoot freshness by finding the oldest boundary that is late: source +availability, trigger/start queue, extraction, transfer, transform, target +commit, or consumer publish. Mini-lab: create a run record with read/written/ +rejected counts and watermark; inject one missing file and one slow destination; +show which metrics distinguish completeness failure from throughput failure. + +## Transformation monitoring field guide + + +A transformation is healthy only when code ran, output meets its contract, and +resource cost remains acceptable. For Dataflow Gen2, examine refresh start, +status, duration, submitter/capacity, failing query/step, rows or bytes where +reported, detailed logs, and destination results. For notebooks/Spark jobs, +examine application, stage and task distribution, shuffle/spill/skew, executor +loss, input/output, and assertions. For SQL/KQL, retain query/request ID, +duration, scanned data, queue/resource evidence, plan/query context, and result +validation. + +Attach a quality result table to every curated output: + +```text +run_id | dataset | rule_id | severity | evaluated | failed | threshold | disposition +``` + +Rules include schema, uniqueness, required values, range/domain, referential +integrity, timeliness, and reconciliation. Use a warning only when the documented +disposition allows publish; otherwise fail before replacing the last known-good +output. Protect quality samples because they may contain sensitive values. + +**Scenario.** A Dataflow duration doubles but output counts remain correct. Drill +into source scan/bytes and folding evidence, destination write, gateway, and +capacity rather than treating it as data corruption. A notebook whose median +task is 20 seconds but one task is 12 minutes suggests skew; adding executors +may not fix one oversized partition. Mini-lab: set an allowed null-rate threshold, +run one pass below it and one above it, and verify publish, evidence, and alert +behavior match the declared severity. + +## Semantic-model refresh field guide + + +Refresh monitoring begins with refresh history for the model and refresh +summaries for administrative scope where authorized. Capture model/workspace, +refresh ID or time, trigger type, status, start/end/duration, table/partition +detail, error, gateway/connection context, and capacity evidence. Distinguish +full, incremental/partition, and metadata-only behaviors as applicable. The +refreshing identity needs model and source/connection rights; broad workspace +Admin is not the default repair. + +Trace consumer freshness backward. A model can refresh successfully from a +stale warehouse, so compare the upstream published watermark with the model's +source boundary and a consumer-visible maximum business timestamp. Incremental +refresh must include the intended partitions and handle late changes according +to policy. A renamed source column is deterministic and should be repaired; a +temporary gateway or capacity fault may justify bounded retry. + +**Worked incident.** The 06:30 model refresh succeeds, but the dashboard still +shows yesterday. Refresh detail confirms every partition completed. The +warehouse publish watermark is also yesterday because upstream ingestion missed +its cutoff. The correct alert is end-to-end freshness, not another semantic +refresh. Mini-lab: document a source → lakehouse → warehouse → semantic model +chain, deliberately stale one boundary, and prove which timestamp identifies +the first stale stage. + +For long refreshes, compare historical duration, table/partition time, source +query, gateway, capacity throttling, model size, and concurrent operations. For +failure, preserve the exact message and refresh context before retry. Validate +both failure and recovery; an alert that never resolves creates permanent noise. + +## Alert engineering field guide + + +An alert contract contains signal and unit, scope, threshold, evaluation window, +minimum duration/consecutive evaluations, schedule, severity, owner, delivery +channel or action, deduplication key, suppression/cooldown, recovery condition, +and runbook. Use Activator where its event/condition/action model fits; use +workspace monitoring queries plus supported alert/action mechanisms where +log-derived conditions are required. Recipients need access to enough context to +act, but alerts should not disclose credentials or raw sensitive records. + +Choose symptoms close to user impact: missed freshness cutoff, sustained lag, +failed required refresh, or quality threshold. Pair them with diagnostic signals +but avoid paging on every retry. Static thresholds fit contractual cutoffs; +rate-of-change or historical baselines fit gradual deviations. Route warning and +critical differently. Require acknowledgment/escalation for critical alerts and +define maintenance suppression. + +**Worked alert.** Evaluate every five minutes: if consumer freshness lag exceeds +30 minutes for two consecutive evaluations, create one incident keyed by +dataset, notify the data-operations group, attach newest source and publish +timestamps plus last run ID, and link the freshness runbook. Recover only after +lag is below 15 minutes for two evaluations. This hysteresis prevents flapping. +Test with synthetic late data, verify exactly one firing, inspect the action, +restore freshness, and verify recovery. + +If an alert does not fire, check signal production, time field/window, condition, +enabled state, permissions, action connection, and suppression. If it fires too +often, compare raw values and evaluation history before raising the threshold. +Mini-lab: design one failed-run alert and one freshness alert for the same +pipeline; explain why they can legitimately disagree. + + +Monitoring hub centralizes supported job status; item views expose engine detail; +workspace monitoring stores supported log-level evidence; Capacity Metrics +explains CU use and throttling; Purview Audit records user/admin activity; +OneLake diagnostics records data access. “Succeeded” is execution evidence, +freshness is outcome evidence, and retry is a response only to a classified +transient fault. + + +**Recall.** What equation detects silently rejected ingestion rows? Why can a +semantic refresh succeed while a dashboard is stale? Which metrics distinguish +Spark skew from general capacity pressure? What fields make a quality assertion +auditable? Why use hysteresis for recovery? For practice, write one outcome, +execution, and resource signal for pipeline, Dataflow, notebook, Eventstream, +and semantic model; identify the authoritative screen/log for each. + ## Exam distinctions - Monitoring hub gives centralized visibility; item detail provides engine-specific evidence. diff --git a/content/published/dp700/chapters/optimize-performance.md b/content/published/dp700/chapters/optimize-performance.md index 2726f09..7823aaf 100644 --- a/content/published/dp700/chapters/optimize-performance.md +++ b/content/published/dp700/chapters/optimize-performance.md @@ -89,6 +89,227 @@ Across SQL, KQL, and Spark SQL, the durable principles are similar: Optimization must preserve correctness. Compare row counts, totals, null behavior, and representative results after a rewrite. Report both latency and capacity/resource change so a faster query that costs far more is visible as a trade-off. +## The optimization experiment contract + + +Write the experiment before changing the system: workload/query/data version, +concurrency and cache state, environment/capacity, baseline runs, metric target, +correctness assertions, proposed bottleneck, one change, repeated result, and +cost/resource comparison. Use median and a tail percentile when variability +matters; one warm run after a cold baseline is not evidence. Preserve query/run +IDs and actual plans or engine metrics. + +Optimize the dominant bottleneck. If time is source I/O, adding pipeline +activities can hurt. If one Spark partition is skewed, adding executors leaves +one long task. If capacity throttles many workloads, a local SQL rewrite may not +explain all latency. Scaling is valid when measured demand genuinely exceeds +available resources after reasonable efficiency work, but it proves only that +more resources helped that tested workload. + +## Lakehouse optimization field guide + + +Inspect table size, file count and size distribution, partition columns/count, +write cadence, query predicates, bytes/files scanned, and Delta history before +maintenance. Many tiny files create metadata and task overhead. `OPTIMIZE` +compacts eligible files; V-Order reorganizes Parquet for Fabric read efficiency. +Partitioning creates directory-level pruning boundaries. These mechanisms solve +different problems and can be combined deliberately. + +Choose low-cardinality, frequently filtered columns with balanced volume for +partitioning—often date at an appropriate grain. High-cardinality customer or +transaction IDs produce tiny partitions. Use optimized write/maintenance based +on accumulated data and query SLA, not after every micro-batch. Maintain +statistics where the engine uses them. Before `VACUUM`, confirm time-travel, +concurrent-reader, replay, legal retention, and recovery requirements; removed +files cannot support old versions. + +**Lab.** Write the same representative table as thousands of tiny files and as +compacted files. Run identical selective and broad queries three times under +documented cache state; record planning/tasks, files/bytes, duration, and CU or +compute evidence. Apply `OPTIMIZE`/V-Order as supported and compare. Then test a +date partition versus an intentionally bad high-cardinality partition. Validate +row count, distinct key, and totals after every layout change. + +If optimization appears ineffective, verify the query actually reads the +optimized table/version, predicate can prune, file accumulation warranted +compaction, and competing capacity/cache conditions are comparable. V-Order +does not repair skewed partitioning, nonselective queries, or incorrect data +modeling. + +## Pipeline optimization field guide + + +Break elapsed time into trigger/queue, orchestration overhead, source query, +transfer, transform, staging, sink write/commit, and retry. Capture rows/bytes, +throughput, parallel copies, ForEach concurrency, source and sink utilization, +throttling, file counts, and capacity. First reduce work with incremental +selection, partition pruning, column projection, and compression. Then tune +parallelism within source, gateway/network, Fabric capacity, and sink limits. + +Copy parallelism controls work within a copy; ForEach concurrency controls +simultaneous activities. Raising both multiplies pressure. Many tiny activities +increase scheduling and connection overhead, while one giant serial copy may +underutilize resources. Partition/batch at recoverable boundaries. Use staging +only when a connector path or measured bulk-loading improvement warrants its +extra I/O. Backoff protects a throttled dependency; instant high retry amplifies +the outage. + +**Lab.** Copy 20 equally sized partitions at concurrency 1, 4, and 12 while +recording total throughput, source/sink throttling, activity duration, and +capacity. Keep mapping and data fixed. The best point is the lowest reliable +resource cost meeting SLA, not automatically 12. Repeat with 2,000 tiny files +versus consolidated inputs to expose orchestration/file overhead. Validate +source-to-target counts and repeat-safe rerun. + +If throughput falls as concurrency rises, inspect source connection limits, +gateway/network, sink commits/locks, capacity, and generated small files. If +queue time dominates, inspect capacity and activity count. If one partition +dominates, rebalance boundaries rather than globally increasing parallelism. + +## Warehouse optimization field guide + + +Begin with Query insights/monitor, actual plan and runtime metrics, query text +and parameters, data volume/distribution, statistics, concurrency, and Capacity +Metrics. Classify scan, join/data movement, sort/aggregate, blocking, queueing, +spill, or compilation/plan change. Select only required columns, use types that +represent values without needless width, filter with sargable/selective +predicates, preaggregate before many-to-many joins, and preserve a clean star +grain. + +Statistics help estimates; stale or absent statistics can lead to a poor join +order or movement. Materialized views or persisted curated tables can trade +refresh/storage cost for repeated query savings. Avoid wrapping a filter column +in a conversion when the same boundary can be expressed on the original type. +Avoid `SELECT *`, accidental Cartesian joins, and row-by-row procedural logic. +Capacity scale can help concurrency saturation but not a query that scans every +wide row unnecessarily. + +**Lab.** Start with a query joining order facts to a dimension whose business +key is duplicated. Capture the wrong row count and plan. Repair dimension +uniqueness, project only required columns, add a selective date predicate, and +compare scan, duration, and correct total. Next preaggregate facts before a +consumer-level join and compare. Run at one and several concurrent users so a +single-query gain is not mistaken for workload capacity. + +For regressions, compare code/data/statistics/plan/concurrency/capacity changes. +If all queries slow, suspect shared resource or blocking before rewriting each. +If only one parameter shape slows, investigate selectivity and plan behavior. +Do not “optimize” by removing correctness filters or changing decimal semantics. + +## Real-Time Intelligence optimization field guide + + +For Eventstreams, measure source input, operator/branch output, processing lag, +invalid/drop rate, and each destination. Filter and project early, avoid +duplicating expensive transforms on every branch, and route only required +events. Choose batching and destinations with end-to-end latency in mind; one +slow destination must be identifiable rather than silently backing up every +route. + +For Eventhouse, compare ingestion rate/batch metrics, hot-cache horizon, +retention, storage, query logs/insights, scanned extents/data, concurrency, and +capacity. Put time and selective `where` early, `project` needed columns, use +term-aware operators, reduce both sides before joins, and avoid repeatedly +parsing broad dynamic payloads. Set caching to the frequently queried hot +horizon and retention to the business/governance horizon; they answer different +questions. + +Use materialized views for frequently repeated aggregates when ingestion-time +maintenance and freshness semantics are acceptable. Use update policies for +supported deterministic derived ingestion when duplicated storage and failure +behavior are understood. For external Delta, compare standard shortcut, +accelerated recent window, and native ingestion; acceleration does not unlock +all native-table features. + +**Lab.** Query 90 days when most users need 24 hours. Add an explicit time +predicate and projection, measure scanned data and latency, then configure a hot +cache/accelerated period aligned to the observed horizon where appropriate. +Create a repeated five-minute aggregation and compare direct query with a +materialized design including ingestion cost. Validate bucket totals and late- +event behavior. If lag increases, localize source, operator, Eventhouse +ingestion, or query/capacity pressure before scaling. + +## Spark optimization field guide + + +Use Spark UI to identify the critical stage and its task distribution. Record +input/output rows and bytes, partitions, median/max task duration, shuffle read +and write, spill, skew, executor CPU/memory/GC, failures, and plan. Reduce scan +through partition pruning, column projection, and predicate pushdown. Prefer +built-in functions and vectorized execution to Python UDFs. Avoid unnecessary +wide transformations and repeated recomputation. + +Broadcast a dimension only when its serialized size safely fits every executor +and the plan confirms broadcast. Otherwise align/repartition around large joins +carefully. `repartition` performs a shuffle to reshape/increase/decrease +partitions; `coalesce` typically reduces without full balancing. Address skew +with better keys, preaggregation, adaptive query execution, or selective +salting. Cache only an expensive reused intermediate that fits memory and +unpersist it; caching one-use data adds cost. + +Size compute after the plan is efficient. More executors increase parallel +capacity but cannot split a single indivisible/skewed task automatically. A +larger driver helps driver duties but can hide an unsafe collect. Too many tiny +partitions add scheduling overhead; too few large partitions underuse compute +or spill. Match output partitioning/file sizes to downstream reads as well as +current job speed. + +**Lab.** Create a skewed join and capture one long task. Compare baseline, +preaggregation, and selective salting/adaptive execution; record task spread, +shuffle, spill, duration, and totals. Separately compare a built-in expression +with an equivalent Python UDF and inspect the plan. Cache a reused intermediate +for two actions, then unpersist; prove a one-action workload does not benefit. + +## Cross-engine query optimization field guide + + +The common model is scan → filter/project → join → aggregate/sort → return or +write. Reduce data as early as semantics permit, use engine-prunable predicates, +make join keys compatible, prevent many-to-many explosion, and calculate +expensive repeated results once. But verify the actual engine plan: SQL +statistics and relational operations, KQL extent/time pruning and operator +pipeline, and Spark partition/file pruning plus shuffle have different evidence. + +Compare both cold and representative warm-cache conditions, several runs, +concurrency, result size, and capacity cost. A query can be faster because the +result changed, cache warmed, data volume shrank, or capacity was quieter. Use a +fixed correctness fixture plus production-scale measurement. Validate row count, +key uniqueness, null behavior, decimal/time-zone semantics, and totals before +accepting a rewrite. + +**Worked comparison.** SQL `WHERE OrderDate >= @start`, KQL `where EventTime >= +start`, and Spark `.filter(col("OrderDate") >= start)` can all reduce scans only +when types, storage organization, and optimizer/source pushdown support it. +Wrapping the stored date in string conversion may prevent pruning. Preaggregate +a large fact by join key before joining to a small classification when only +group totals are needed—but never if detail-level matching changes meaning. + +If a rewrite shows no gain, check plan equivalence, predicate selectivity, +storage/partitioning, statistics, cache, result transfer, and shared resource +noise. If it is faster but CU/resource cost rises sharply, state the trade-off +and decide against the SLA/cost objective. Mini-lab: build a result checksum and +metrics table, then optimize one SQL, KQL, and Spark query using the same +filter/project/join principles and engine-specific evidence. + + +`OPTIMIZE` compacts Delta files, V-Order changes file layout, partitioning enables +pruning, and `VACUUM` removes obsolete files under retention rules. Copy +parallelism differs from loop concurrency. Eventhouse retention differs from hot +cache. Spark repartition differs from coalesce, and broadcast is a join strategy, +not “send output everywhere.” A faster run is not proven optimization until +correctness, comparable conditions, and resource cost are included. + + +**Recall.** Which metrics prove a small-file problem? Why can 12 concurrent +copies be slower than four? What plan evidence distinguishes warehouse scan from +queueing? When does a materialized view move cost rather than remove it? Why does +one long Spark task suggest skew? How can conversion prevent pruning? For +practice, write an A/B optimization record with hypothesis, fixed conditions, +three baseline and three changed runs, correctness checksum, p50/p95, bytes +scanned, and CU/compute result. + ## Exam distinctions - `OPTIMIZE` compacts Delta files; V-Order changes Parquet layout; partitioning organizes data for pruning. diff --git a/content/published/dp700/chapters/resolve-errors.md b/content/published/dp700/chapters/resolve-errors.md index dd2d79f..4ca8796 100644 --- a/content/published/dp700/chapters/resolve-errors.md +++ b/content/published/dp700/chapters/resolve-errors.md @@ -66,6 +66,198 @@ Verify that the shortcut target still exists, the stored connection is valid, th For KQL shortcuts, validate with `external_table('name')`. For lakehouse table shortcuts, confirm the target is a supported Delta table and appears at the correct Tables path. Acceleration has separate status, cache-window, and schema constraints. +## Pipeline failure clinic + + +A pipeline run is a graph of resolved activities, not just a canvas definition. +Start with run ID, trigger, parameters, identity, and the earliest failed +activity. Open its resolved input/output and exact error code. Trace a child +pipeline or notebook using its child run ID. Validate dynamic-expression type +and JSON path, source/sink connection, path/table, mapping, target constraints, +timeout, retry, and concurrency. Do not expose secret values while capturing +evidence. + +Classify before repair: permission/authentication requires identity or connection +correction; missing/renamed input requires source contract or routing; mapping +and constraint failures require schema/data handling; throttling and temporary +network faults may use bounded backoff. For partial ForEach completion, enumerate +successful units and prove target idempotency before replay. A retry cannot make +an invalid column mapping valid. + +**Lab.** A metadata-driven child receives `sourcePath=null`, so `concat()` builds +an unintended path and Copy fails “not found.” Capture the lookup output and +resolved child input, add boundary validation, correct the metadata, and rerun +only that entity. Then simulate throttling and verify backoff succeeds without +duplicate rows. The regression checks both null rejection and repeat-safe +application. If the activity remains queued, add capacity, integration runtime, +and concurrency evidence rather than changing the path again. + +## Dataflow Gen2 failure clinic + + +Separate **author/save validation**, **refresh execution**, and **destination +write**. In refresh history identify the run, failing query, first failing step, +connection/gateway, and detailed log. Reproduce on the offending query and data, +not only the small preview. Test credentials and privacy/firewall boundaries, +then schema names/types, locale conversions, merge cardinality, custom functions, +folding/source timeout, and destination schema/update behavior. + +An added source column may be harmless while a removed/renamed or changed-type +column is breaking. Replacing every error with null can turn a visible failure +into silent corruption. Route conversion failures with source value, reason, +run, and rule; correct or approve disposition. If refresh is slow rather than +failed, inspect folding, source rows/bytes, gateway routing, and destination +write. Some connectors report rows and others bytes, so do not compare unlike +statistics. + +**Lab.** Preview contains 1,000 valid dates, while the full source contains +`31/13/2026`. Refresh fails at `Changed Type`. Create an errors query, apply an +explicit locale, quarantine the invalid row, and reconcile accepted plus rejected +to source. Next rename a source column and prove schema validation fails before +publish. The correct regression includes the previously unseen value, not just +another preview. + +## Notebook and Spark failure clinic + + +Identify whether failure occurs before session start, in the driver/cell, or in +distributed stages/tasks. Pre-session failures point to pool/capacity, +environment runtime, library publication, permissions, or lakehouse attachment. +A driver stack trace points to Python/SQL/control logic or collecting too much +data. Repeated executor/task failure points to one bad record/partition, skew, +shuffle, memory, serialization, or transient worker loss. Use Spark UI and logs +for the first failing stage and compare task duration, input, shuffle, spill, and +failure reasons. + +Confirm code version, parameters, runtime, published environment, libraries, +Spark configuration, attached item, identity, input partition, and checkpoint +for streaming. Avoid `collect()`/`toPandas()` on unbounded data; filter/project +early; replace Python UDFs with built-ins where possible; isolate corrupt input; +fix skew before scaling. A library that imports interactively but is absent from +the published environment will fail scheduled execution. + +**Lab.** One key holds 70% of rows. The Spark stage shows one task far longer +than its peers and spilling, while executor utilization elsewhere is low. +Preaggregate or redesign the key; if legitimate, use a measured skew strategy +such as adaptive execution or selective salting. Compare task distribution and +correct totals before/after. Separately force a driver OOM with a test `collect` +and repair it with distributed aggregation; explain why adding executors would +not increase driver memory. + +## Eventhouse failure clinic + + +Split ingestion, command/policy, query, and capacity. Eventhouse monitoring can +expose metrics plus command, data-operation, ingestion-result, and query logs. +For ingestion, keep source/connection, database/table, operation/request ID, +format and ingestion mapping, schema/type, identity, batching, and result. For a +query, keep request ID, database context, text, time range, scanned/result rows, +duration, and error. Check that event time—not ingestion time—is used as intended. + +Mapping name, delimiter/encoding, dynamic JSON path, unsupported type, missing +table permission, source connectivity, and retention/caching policy can fail or +misroute ingestion. An empty query can result from the wrong database/table, +too-narrow time filter, time-zone/parse error, or ingestion delay. Reduce a +failing KQL pipeline operator by operator; start with `take` and a known broad +time range, then reapply filters and parsing. + +**Lab.** Five events are sent; four ingest and one rejects because +`Temperature="hot"` conflicts with mapping. Locate the ingestion result, retain +the source position, correct or route the event, and reconcile five outcomes. +Then query a future event-time range and prove empty results are a query-boundary +issue, not ingestion failure. If all databases slow simultaneously, examine +Eventhouse system/capacity evidence before rewriting one mapping. + +## Eventstream failure clinic + + +Walk the graph and compare rate at every boundary: source connected/input, +operator input/output/error, route match, destination accepted/failed, and +end-to-end lag. Capture item, time, source partition/offset or event ID, +event-time/schema version, and destination status. Validate connector credentials +and network, then transformation fields/types, window time/watermark, route +conditions, and destination permissions/capacity. + +A flat zero input points upstream; normal source input and zero after an operator +points to filter/schema logic; normal branch output plus destination failures +points downstream. Rising lag with input greater than output indicates +backpressure or a slow sink, not necessarily lost events. Route poison messages +to a restricted quarantine with reason and replay reference; do not create an +infinite retry loop around deterministic bad payloads. + +**Lab.** Rename `eventTime` to `event_time` at the source without updating the +window transform. Raw route continues, curated output stops. Boundary rates +localize the first zero-output operator. Add schema validation/version handling, +replay retained raw events, and verify window totals. Then deny only one +destination and show the healthy branch continues; the incident scope is one +sink, not the entire stream. + +## T-SQL failure clinic + + +Capture statement/query ID, database and schema context, executing identity, +parameters, exact number/message, transaction state, and time. Compilation/name +errors differ from permission, conversion, constraint, blocking/deadlock, +resource, and timeout failures. Query Monitor, query insights, DMVs, or plans +provide runtime evidence where supported. Use the least-privilege grant rather +than changing ownership or granting a broad workspace role. + +For truncation/conversion, find the column and offending value, compare source +and target types/length/precision, and choose validated cleansing or schema +change. For key/null constraints, test staged uniqueness and required values +before the target. For deadlocks, keep the graph/context, make transaction order +consistent and transactions short, and use safe retry only for the chosen +victim. For timeout, separate client timeout, blocking, capacity queueing, scan, +and plan regression. + +**Lab.** Two staged rows share a target business key and `MERGE` fails. Rank to +one trusted source version, quarantine ambiguous ties, enforce the invariant, +and rerun idempotently. Then create a safe blocking scenario in a test database, +identify blocker versus victim, and resolve transaction scope rather than adding +an index at random. Regression fixtures retain the duplicate key and verify one +deterministic outcome. + +## Shortcut failure clinic + + +A shortcut failure spans reference metadata, connection identity, source target, +format, consuming engine, and optional cache/acceleration. Record shortcut/item, +target URI/path, internal versus external type, connection, querying identity, +engine, schema, and exact error. Confirm target existence first, then connection +validity and source ACL, supported location/format, Delta `_delta_log` for table +use, Tables versus Files placement, and schema/cache status. + +Different users can see different results because credential delegation and +permissions differ. Deleting a shortcut is not target recovery; moving the +target requires updating/recreating the reference. For KQL validate +`external_table()` and time/schema assumptions; for accelerated shortcuts check +cache status/window and external-table limitations separately. A standard +shortcut cannot hide source outage or latency. + +**Lab.** Query a known Delta shortcut, revoke the connection identity's source +read, and confirm an authorization failure while target files remain. Restore +read, then remove `_delta_log` in a disposable copy and show file access may not +qualify as a table. Repair the test target and validate row count and schema. +Record a denied identity as well as the allowed one so success is not proved only +with owner privileges. + + +Resolved pipeline inputs show runtime truth; Dataflow preview is sampled authoring +evidence; Spark UI separates driver, stage, and task behavior; Eventhouse +ingestion logs differ from query logs; Eventstream boundary rates localize graph +failures; T-SQL error/plan context distinguishes correctness from slowness; a +shortcut is a reference whose target can remain intact. Retry only a classified +transient fault and only with repeat-safe effects. + + +**Recall.** What is the earliest-failure rule? Why can `collect()` cause driver +rather than executor failure? Which Eventhouse evidence proves a rejected event? +How do Eventstream boundary rates distinguish source from destination? What +makes a T-SQL timeout ambiguous? Which three identities/permissions can affect a +shortcut path? For practice, write one incident record with time, scope, +correlation ID, classification, evidence, correction, rerun boundary, outcome, +and regression test. + ## Exam distinctions - Retry transient faults; correct deterministic faults. diff --git a/content/published/dp700/coverage.yaml b/content/published/dp700/coverage.yaml index a488ba7..a43957b 100644 --- a/content/published/dp700/coverage.yaml +++ b/content/published/dp700/coverage.yaml @@ -632,103 +632,290 @@ objectives: scenario_lab: windows-comprehensive - objective_id: monitor.items.ingestion chapter_id: monitor-items - status: draft + status: complete subtopics: [Monitoring hub, Pipeline runs, Copy metrics, Streaming rate, Lag, Watermark, Rejected rows, Freshness] source_ids: [dp700-study-guide, fabric-monitoring-hub] - gaps: [Add item-specific screens, thresholds, and incident scenario] + gaps: [] + evidence: + conceptual_model: monitoring-foundation + prerequisites: monitor-ingestion-comprehensive + procedure: monitor-ingestion-comprehensive + decision_guidance: monitor-ingestion-comprehensive + worked_example: monitor-ingestion-comprehensive + limitations: monitor-ingestion-comprehensive + troubleshooting: monitor-ingestion-comprehensive + exam_distinctions: monitoring-exam-distinctions + recall_questions: monitoring-recall-lab + scenario_lab: monitor-ingestion-comprehensive - objective_id: monitor.items.transformation chapter_id: monitor-items - status: draft + status: complete subtopics: [Dataflow refresh, Notebook and Spark jobs, Pipeline activities, SQL and KQL queries, Quality assertions, Capacity evidence] source_ids: [dp700-study-guide, fabric-monitoring-hub, dataflow-monitoring] - gaps: [Add per-engine diagnostic workflow and data-quality evidence model] + gaps: [] + evidence: + conceptual_model: monitoring-foundation + prerequisites: monitor-transformation-comprehensive + procedure: monitor-transformation-comprehensive + decision_guidance: monitor-transformation-comprehensive + worked_example: monitor-transformation-comprehensive + limitations: monitor-transformation-comprehensive + troubleshooting: monitor-transformation-comprehensive + exam_distinctions: monitoring-exam-distinctions + recall_questions: monitoring-recall-lab + scenario_lab: monitor-transformation-comprehensive - objective_id: monitor.items.semantic-refresh chapter_id: monitor-items - status: draft + status: complete subtopics: [Refresh history, Administrative summary, Types, Tables and partitions, Incremental refresh, Credentials, Gateway, Capacity] source_ids: [dp700-study-guide, semantic-refresh-summaries] - gaps: [Add refresh investigation walkthrough and upstream freshness chain] + gaps: [] + evidence: + conceptual_model: monitor-semantic-comprehensive + prerequisites: monitor-semantic-comprehensive + procedure: monitor-semantic-comprehensive + decision_guidance: monitor-semantic-comprehensive + worked_example: monitor-semantic-comprehensive + limitations: monitor-semantic-comprehensive + troubleshooting: monitor-semantic-comprehensive + exam_distinctions: monitoring-exam-distinctions + recall_questions: monitoring-recall-lab + scenario_lab: monitor-semantic-comprehensive - objective_id: monitor.items.alerts chapter_id: monitor-items - status: draft + status: complete subtopics: [Activator, Signals, Thresholds, Windows, Recipients, Actions, Suppression, Recovery, Runbooks] source_ids: [dp700-study-guide, fabric-activator] - gaps: [Add alert configuration procedure and firing/recovery test] + gaps: [] + evidence: + conceptual_model: alerts-comprehensive + prerequisites: alerts-comprehensive + procedure: alerts-comprehensive + decision_guidance: alerts-comprehensive + worked_example: alerts-comprehensive + limitations: alerts-comprehensive + troubleshooting: alerts-comprehensive + exam_distinctions: monitoring-exam-distinctions + recall_questions: monitoring-recall-lab + scenario_lab: alerts-comprehensive - objective_id: monitor.errors.pipeline chapter_id: resolve-errors - status: draft + status: complete subtopics: [Failed activity, Resolved inputs, Error codes, Connections, Expressions, Child runs, Retry classification, Idempotent reruns] source_ids: [dp700-study-guide, pipeline-troubleshooting] - gaps: [Add representative failure cases and resolution lab] + gaps: [] + evidence: + conceptual_model: resolve-pipeline-comprehensive + prerequisites: resolve-pipeline-comprehensive + procedure: resolve-pipeline-comprehensive + decision_guidance: resolve-pipeline-comprehensive + worked_example: resolve-pipeline-comprehensive + limitations: resolve-pipeline-comprehensive + troubleshooting: resolve-pipeline-comprehensive + exam_distinctions: errors-exam-distinctions + recall_questions: errors-recall-lab + scenario_lab: resolve-pipeline-comprehensive - objective_id: monitor.errors.dataflow chapter_id: resolve-errors - status: draft + status: complete subtopics: [Refresh detail, Power Query step, Credentials, Gateway, Type errors, Schema drift, Folding, Destination errors] source_ids: [dp700-study-guide, dataflow-monitoring] - gaps: [Add bad-row diagnosis, schema failure, and destination repair examples] + gaps: [] + evidence: + conceptual_model: resolve-dataflow-comprehensive + prerequisites: resolve-dataflow-comprehensive + procedure: resolve-dataflow-comprehensive + decision_guidance: resolve-dataflow-comprehensive + worked_example: resolve-dataflow-comprehensive + limitations: resolve-dataflow-comprehensive + troubleshooting: resolve-dataflow-comprehensive + exam_distinctions: errors-exam-distinctions + recall_questions: errors-recall-lab + scenario_lab: resolve-dataflow-comprehensive - objective_id: monitor.errors.notebook chapter_id: resolve-errors - status: draft + status: complete subtopics: [Driver errors, Executor failures, Spark UI, Runtime, Libraries, Lakehouse attachment, Memory, Skew and bad records] source_ids: [dp700-study-guide] - gaps: [Add direct notebook troubleshooting sources and Spark failure labs] + gaps: [] + evidence: + conceptual_model: resolve-notebook-comprehensive + prerequisites: resolve-notebook-comprehensive + procedure: resolve-notebook-comprehensive + decision_guidance: resolve-notebook-comprehensive + worked_example: resolve-notebook-comprehensive + limitations: resolve-notebook-comprehensive + troubleshooting: resolve-notebook-comprehensive + exam_distinctions: errors-exam-distinctions + recall_questions: errors-recall-lab + scenario_lab: resolve-notebook-comprehensive - objective_id: monitor.errors.eventhouse chapter_id: resolve-errors - status: draft + status: complete subtopics: [Ingestion failures, Mapping and format, Permissions, Table policies, KQL request IDs, Capacity, Empty results] source_ids: [dp700-study-guide] - gaps: [Add direct Eventhouse troubleshooting sources and commands] + gaps: [] + evidence: + conceptual_model: resolve-eventhouse-comprehensive + prerequisites: resolve-eventhouse-comprehensive + procedure: resolve-eventhouse-comprehensive + decision_guidance: resolve-eventhouse-comprehensive + worked_example: resolve-eventhouse-comprehensive + limitations: resolve-eventhouse-comprehensive + troubleshooting: resolve-eventhouse-comprehensive + exam_distinctions: errors-exam-distinctions + recall_questions: errors-recall-lab + scenario_lab: resolve-eventhouse-comprehensive - objective_id: monitor.errors.eventstream chapter_id: resolve-errors - status: draft + status: complete subtopics: [Connector state, Source rate, Transform schema, Event time, Destination health, Poison events, Backpressure] source_ids: [dp700-study-guide] - gaps: [Add direct Eventstream troubleshooting sources and failure-routing lab] + gaps: [] + evidence: + conceptual_model: resolve-eventstream-comprehensive + prerequisites: resolve-eventstream-comprehensive + procedure: resolve-eventstream-comprehensive + decision_guidance: resolve-eventstream-comprehensive + worked_example: resolve-eventstream-comprehensive + limitations: resolve-eventstream-comprehensive + troubleshooting: resolve-eventstream-comprehensive + exam_distinctions: errors-exam-distinctions + recall_questions: errors-recall-lab + scenario_lab: resolve-eventstream-comprehensive - objective_id: monitor.errors.tsql chapter_id: resolve-errors - status: draft + status: complete subtopics: [Error numbers, Context and names, Permissions, Conversions, Constraints, Transactions, Deadlocks, Timeouts and plans] source_ids: [dp700-study-guide] - gaps: [Add Fabric-specific T-SQL troubleshooting sources and worked cases] + gaps: [] + evidence: + conceptual_model: resolve-tsql-comprehensive + prerequisites: resolve-tsql-comprehensive + procedure: resolve-tsql-comprehensive + decision_guidance: resolve-tsql-comprehensive + worked_example: resolve-tsql-comprehensive + limitations: resolve-tsql-comprehensive + troubleshooting: resolve-tsql-comprehensive + exam_distinctions: errors-exam-distinctions + recall_questions: errors-recall-lab + scenario_lab: resolve-tsql-comprehensive - objective_id: monitor.errors.shortcut chapter_id: resolve-errors - status: draft + status: complete subtopics: [Target existence, Connections, Source permissions, Format, Tables path, Schema sync, Cache and acceleration] source_ids: [dp700-study-guide, onelake-shortcuts] - gaps: [Add systematic shortcut diagnostic procedure and repair lab] + gaps: [] + evidence: + conceptual_model: resolve-shortcut-comprehensive + prerequisites: resolve-shortcut-comprehensive + procedure: resolve-shortcut-comprehensive + decision_guidance: resolve-shortcut-comprehensive + worked_example: resolve-shortcut-comprehensive + limitations: resolve-shortcut-comprehensive + troubleshooting: resolve-shortcut-comprehensive + exam_distinctions: errors-exam-distinctions + recall_questions: errors-recall-lab + scenario_lab: resolve-shortcut-comprehensive - objective_id: monitor.optimize.lakehouse chapter_id: optimize-performance - status: draft + status: complete subtopics: [Small files, OPTIMIZE, V-Order, Partitioning, Statistics, VACUUM, Retention, Maintenance cadence] source_ids: [dp700-study-guide, lakehouse-delta-tables, delta-v-order] - gaps: [Add measurement-based before/after lab and safety checks] + gaps: [] + evidence: + conceptual_model: optimize-lakehouse-comprehensive + prerequisites: optimize-lakehouse-comprehensive + procedure: optimize-lakehouse-comprehensive + decision_guidance: optimize-lakehouse-comprehensive + worked_example: optimize-lakehouse-comprehensive + limitations: optimize-lakehouse-comprehensive + troubleshooting: optimize-lakehouse-comprehensive + exam_distinctions: optimize-exam-distinctions + recall_questions: optimize-recall-lab + scenario_lab: optimize-lakehouse-comprehensive - objective_id: monitor.optimize.pipeline chapter_id: optimize-performance - status: draft + status: complete subtopics: [Source selection, Copy parallelism, ForEach concurrency, Staging, Activity overhead, Retry backoff, Throughput metrics] source_ids: [dp700-study-guide] - gaps: [Add direct performance sources and measured tuning scenario] + gaps: [] + evidence: + conceptual_model: optimize-pipeline-comprehensive + prerequisites: optimize-pipeline-comprehensive + procedure: optimize-pipeline-comprehensive + decision_guidance: optimize-pipeline-comprehensive + worked_example: optimize-pipeline-comprehensive + limitations: optimize-pipeline-comprehensive + troubleshooting: optimize-pipeline-comprehensive + exam_distinctions: optimize-exam-distinctions + recall_questions: optimize-recall-lab + scenario_lab: optimize-pipeline-comprehensive - objective_id: monitor.optimize.warehouse chapter_id: optimize-performance - status: draft + status: complete subtopics: [Scans, Statistics, Data types, Star schema, Joins, Plans, Concurrency, V-Order and materialization] source_ids: [dp700-study-guide, warehouse-performance] - gaps: [Add execution-plan walkthrough and before/after query lab] + gaps: [] + evidence: + conceptual_model: optimize-warehouse-comprehensive + prerequisites: optimize-warehouse-comprehensive + procedure: optimize-warehouse-comprehensive + decision_guidance: optimize-warehouse-comprehensive + worked_example: optimize-warehouse-comprehensive + limitations: optimize-warehouse-comprehensive + troubleshooting: optimize-warehouse-comprehensive + exam_distinctions: optimize-exam-distinctions + recall_questions: optimize-recall-lab + scenario_lab: optimize-warehouse-comprehensive - objective_id: monitor.optimize.realtime chapter_id: optimize-performance - status: draft + status: complete subtopics: [Eventstream filtering, Routing, Ingestion batching, Retention, Caching, KQL shape, Materialized views, Shortcut acceleration] source_ids: [dp700-study-guide, kql-query-best-practices] - gaps: [Add Eventhouse policy sources and measured streaming scenario] + gaps: [] + evidence: + conceptual_model: optimize-realtime-comprehensive + prerequisites: optimize-realtime-comprehensive + procedure: optimize-realtime-comprehensive + decision_guidance: optimize-realtime-comprehensive + worked_example: optimize-realtime-comprehensive + limitations: optimize-realtime-comprehensive + troubleshooting: optimize-realtime-comprehensive + exam_distinctions: optimize-exam-distinctions + recall_questions: optimize-recall-lab + scenario_lab: optimize-realtime-comprehensive - objective_id: monitor.optimize.spark chapter_id: optimize-performance - status: draft + status: complete subtopics: [Spark UI, Partition pruning, Shuffle, Skew, Broadcast joins, Repartition, Coalesce, Cache, Compute sizing] source_ids: [dp700-study-guide, lakehouse-delta-tables] - gaps: [Add Fabric Spark tuning sources and evidence-driven lab] + gaps: [] + evidence: + conceptual_model: optimize-spark-comprehensive + prerequisites: optimize-spark-comprehensive + procedure: optimize-spark-comprehensive + decision_guidance: optimize-spark-comprehensive + worked_example: optimize-spark-comprehensive + limitations: optimize-spark-comprehensive + troubleshooting: optimize-spark-comprehensive + exam_distinctions: optimize-exam-distinctions + recall_questions: optimize-recall-lab + scenario_lab: optimize-spark-comprehensive - objective_id: monitor.optimize.query chapter_id: optimize-performance - status: draft + status: complete subtopics: [Projection, Predicate pushdown, Pruning, Join grain, Plans, Repeated computation, Preaggregation, Correctness and cost] source_ids: [dp700-study-guide, warehouse-performance, kql-query-best-practices] - gaps: [Add cross-engine query examples with plans and measured results] + gaps: [] + evidence: + conceptual_model: optimize-query-comprehensive + prerequisites: optimize-query-comprehensive + procedure: optimize-query-comprehensive + decision_guidance: optimize-query-comprehensive + worked_example: optimize-query-comprehensive + limitations: optimize-query-comprehensive + troubleshooting: optimize-query-comprehensive + exam_distinctions: optimize-exam-distinctions + recall_questions: optimize-recall-lab + scenario_lab: optimize-query-comprehensive diff --git a/tests/content/test_dp700_outline.py b/tests/content/test_dp700_outline.py index d7ae373..bd9bdc0 100644 --- a/tests/content/test_dp700_outline.py +++ b/tests/content/test_dp700_outline.py @@ -49,8 +49,8 @@ def test_dp700_coverage_audit_records_every_objective_and_evidence() -> None: } assert all(len(coverage.subtopics) >= 2 for coverage in audit.objectives) assert all(coverage.source_ids for coverage in audit.objectives) - assert sum(coverage.status == "complete" for coverage in audit.objectives) == 37 - assert sum(coverage.status == "draft" for coverage in audit.objectives) == 17 + assert sum(coverage.status == "complete" for coverage in audit.objectives) == 54 + assert all(coverage.status == "complete" for coverage in audit.objectives) assert all( coverage.evidence.is_complete and not coverage.gaps if coverage.status == "complete" From 7702c349879ccdc312472e72e080f74e040f92f5 Mon Sep 17 00:00:00 2001 From: Troy Scott <846218+troyscott@users.noreply.github.com> Date: Wed, 2 Sep 2026 04:45:36 -0700 Subject: [PATCH 6/6] Add comprehensive DP-700 capstones and provenance --- content/published/dp700/book.yaml | 48 +++++- .../published/dp700/chapters/batch-data.md | 153 ++++++++++++++++++ .../dp700/chapters/lifecycle-management.md | 61 +++++++ .../dp700/chapters/loading-patterns.md | 68 ++++++++ .../published/dp700/chapters/monitor-items.md | 112 +++++++++++++ .../dp700/chapters/optimize-performance.md | 150 +++++++++++++++++ .../published/dp700/chapters/orchestration.md | 145 ++++++++++++++++- .../dp700/chapters/resolve-errors.md | 144 +++++++++++++++++ .../dp700/chapters/security-governance.md | 136 ++++++++++++++++ .../dp700/chapters/streaming-data.md | 134 +++++++++++++++ .../dp700/chapters/workspace-settings.md | 66 ++++++++ content/published/dp700/coverage.yaml | 20 +-- docs/dp700-completeness-report.md | 99 ++++++++++++ tests/content/test_dp700_outline.py | 19 +++ 14 files changed, 1336 insertions(+), 19 deletions(-) create mode 100644 docs/dp700-completeness-report.md diff --git a/content/published/dp700/book.yaml b/content/published/dp700/book.yaml index 62e64d4..9870428 100644 --- a/content/published/dp700/book.yaml +++ b/content/published/dp700/book.yaml @@ -147,7 +147,7 @@ sources: url: https://learn.microsoft.com/en-us/fabric/real-time-intelligence/overview retrieved_at: 2026-09-02T06:52:59Z content_sha256: db73725804e0643422b247e2648b9b7a6b34c7f96914bd45d52d9b2349ae690d - chapter_ids: [streaming-data] + chapter_ids: [streaming-data, resolve-errors] - id: query-acceleration-overview title: Query acceleration for OneLake shortcuts url: https://learn.microsoft.com/en-us/fabric/real-time-intelligence/query-acceleration-overview @@ -214,6 +214,42 @@ sources: retrieved_at: 2026-09-02T06:52:59Z content_sha256: 9e4ddae5cb3f3a33152876d2583ee99c8df9924f8bf03125c9d750a33d456607 chapter_ids: [optimize-performance] + - id: fabric-cicd-workflows + title: CI/CD workflow options in Fabric + url: https://learn.microsoft.com/en-us/fabric/cicd/manage-deployment + retrieved_at: 2026-09-02T11:39:23Z + content_sha256: 4b8410e415c7f9a0d78f3ee230a3213fd71dea79fe3966da572b68d8777a7ca2 + chapter_ids: [lifecycle-management] + - id: data-factory-overview + title: What is Data Factory in Microsoft Fabric? + url: https://learn.microsoft.com/en-us/fabric/data-factory/data-factory-overview + retrieved_at: 2026-09-02T11:39:23Z + content_sha256: 8c3ce1e027cd11fbbc64215a09104de4ae9084d7807f55e4fee76a515b48745e + chapter_ids: [orchestration] + - id: eventhouse-manage-monitor + title: Manage and monitor a KQL database + url: https://learn.microsoft.com/en-us/fabric/real-time-intelligence/manage-monitor-database + retrieved_at: 2026-09-02T11:39:23Z + content_sha256: b8ba7c360762e7ad040ab229ffed1a14c6af921ddb115cc06c3d4e08f9cfd709 + chapter_ids: [resolve-errors] + - id: rti-onelake-shortcuts + title: OneLake shortcuts in a KQL database + url: https://learn.microsoft.com/en-us/fabric/real-time-intelligence/onelake-shortcuts + retrieved_at: 2026-09-02T11:39:23Z + content_sha256: 2f276aec9680373d3b4cdd59b76f12431c9d6faa58d903203906412fa6b4373c + chapter_ids: [streaming-data] + - id: spark-troubleshooting + title: Spark errors overview in Microsoft Fabric + url: https://learn.microsoft.com/en-us/fabric/data-engineering/troubleshoot-spark + retrieved_at: 2026-09-02T11:39:23Z + content_sha256: 190a05719f9c53d3e680d4476ada36b6ab314be9cb86ac7419eac3346651dbe1 + chapter_ids: [resolve-errors, optimize-performance] + - id: warehouse-troubleshooting + title: Troubleshoot Fabric Data Warehouse + url: https://learn.microsoft.com/en-us/fabric/data-warehouse/troubleshoot-fabric-data-warehouse + retrieved_at: 2026-09-02T11:39:23Z + content_sha256: 52049dad7fd1f07e12f3273269585f1e6bf237edd50597292e97cc8bc4ccc945 + chapter_ids: [resolve-errors] domains: - id: implement-manage title: Implement and manage an analytics solution @@ -235,7 +271,7 @@ domains: title: Implement lifecycle management in Fabric content_path: lifecycle-management.md status: published - source_ids: [dp700-study-guide, fabric-cicd-overview] + source_ids: [dp700-study-guide, fabric-cicd-overview, fabric-cicd-workflows] objectives: - {id: implement.lifecycle.version-control, title: Configure version control} - {id: implement.lifecycle.database-projects, title: Implement database projects} @@ -260,7 +296,7 @@ domains: title: Orchestrate processes content_path: orchestration.md status: published - source_ids: [dp700-study-guide, pipeline-overview, pipeline-parameters] + source_ids: [dp700-study-guide, pipeline-overview, pipeline-parameters, data-factory-overview] objectives: - {id: implement.orchestration.choose-tool, title: "Choose between Dataflow Gen2, a pipeline, and a notebook"} - {id: implement.orchestration.triggers, title: Design and implement schedules and event-based triggers} @@ -300,7 +336,7 @@ domains: title: Ingest and transform streaming data content_path: streaming-data.md status: published - source_ids: [dp700-study-guide, real-time-intelligence-overview, query-acceleration-overview, structured-streaming-state] + source_ids: [dp700-study-guide, real-time-intelligence-overview, query-acceleration-overview, structured-streaming-state, rti-onelake-shortcuts] objectives: - {id: ingest.streaming.engine, title: Choose an appropriate streaming engine} - {id: ingest.streaming.native-shortcut, title: Choose between native tables and OneLake shortcuts in Real-Time Intelligence} @@ -329,7 +365,7 @@ domains: title: Identify and resolve errors content_path: resolve-errors.md status: published - source_ids: [dp700-study-guide, pipeline-troubleshooting, dataflow-monitoring, onelake-shortcuts] + source_ids: [dp700-study-guide, pipeline-troubleshooting, dataflow-monitoring, onelake-shortcuts, eventhouse-manage-monitor, real-time-intelligence-overview, spark-troubleshooting, warehouse-troubleshooting] objectives: - {id: monitor.errors.pipeline, title: Identify and resolve pipeline errors} - {id: monitor.errors.dataflow, title: Identify and resolve Dataflow Gen2 errors} @@ -343,7 +379,7 @@ domains: title: Optimize performance content_path: optimize-performance.md status: published - source_ids: [dp700-study-guide, lakehouse-delta-tables, delta-v-order, warehouse-performance, kql-query-best-practices] + source_ids: [dp700-study-guide, lakehouse-delta-tables, delta-v-order, warehouse-performance, kql-query-best-practices, spark-troubleshooting] objectives: - {id: monitor.optimize.lakehouse, title: Optimize a Lakehouse table} - {id: monitor.optimize.pipeline, title: Optimize a pipeline} diff --git a/content/published/dp700/chapters/batch-data.md b/content/published/dp700/chapters/batch-data.md index 70f8a8f..df44056 100644 --- a/content/published/dp700/chapters/batch-data.md +++ b/content/published/dp700/chapters/batch-data.md @@ -376,6 +376,159 @@ For practice, take one customer/order dataset through Dataflow, PySpark, SQL, and KQL; document types, joins, quality dispositions, output grain, and expected totals before comparing results. +## Batch scenario drills + +### Dataflow transformation review + +A Dataflow reads 40 million warehouse rows, then filters to the last day after a +custom text function. Refresh is slow. Inspect folding. Move supported date +filter and column projection before the nonfolding custom step, while preserving +semantics, and compare source rows/bytes plus duration. If the custom function +can be rewritten with foldable native operations, validate it on edge cases. + +The same flow merges Customers to Orders. Before accepting it, test Customer key +uniqueness, unmatched orders, row count before/after, and total Amount. A single +duplicate customer can multiply facts even though refresh succeeds. Keep an +anti-join quality output for missing customers and route invalid types rather +than turning all errors into null. + +### Shortcut, mirroring, or copy + +Three source cases look similar. Historical Delta tables already in supported +cloud storage need shared in-place access: use a shortcut if source availability, +security, and direct-read performance are acceptable. A supported operational +database needs continuously synchronized analytics with minimal custom logic: +evaluate database mirroring. A legacy source needs nightly extracts, column +mapping, and a warehouse destination: use pipeline Copy. + +For each, state ownership when the source moves or credentials rotate, how +deletes/schema changes propagate, whether data is writable in Fabric, and how +recovery works. “Avoid a copy” is not sufficient if consumers require isolation +from source outages or native Eventhouse features. + +### Denormalization at the wrong grain + +Orders has one row per order, OrderLines has many rows per order, and Payments +can also have many rows per order. Joining all three then summing line amount +multiplies each line by payment count. Declare the desired grain. Aggregate +payments to one row per order before joining to line grain, or keep separate fact +tables related through dimensions. Validate distinct line key, row count, and +known total. + +Denormalization is valuable when it simplifies a stable consumption pattern, +but it does not waive dimensional grain. Repeated order attributes are expected; +repeated line measures caused by a many-to-many join are not. + +### Semi-additive inventory + +An inventory snapshot records ending quantity by product and warehouse each +day. Summing products and warehouses for one day is meaningful; summing seven +daily ending balances is not weekly inventory. Select the last valid snapshot +for the period, or calculate a defined average, instead of `SUM` across date. +Revenue, by contrast, may be additive across those days. + +Create measures with their allowed aggregation dimensions documented. For a +late correction to Tuesday, restate aggregates whose chosen Tuesday value or +derived period metric changes. Retain numerator/denominator for ratios rather +than averaging percentages. + +### Quality threshold and quarantine replay + +A batch contains 0.02% invalid dates under a documented warning threshold of +0.05%. Publish accepted records, quarantine failures with rule and source +reference, record the rate, and alert at warning severity. The next day reaches +2%; stop publish according to policy because this likely indicates a source +contract change. Last known-good curated data remains available but freshness +is now at risk. + +After correcting parsing or source data, replay quarantined/bounded source rows, +reconcile counts, update affected aggregates, and close the incident. Quarantine +needs access control, retention, owner, and replay state; otherwise it merely +hides data loss. + +## Capstone: build a trustworthy sales mart + +The source landscape contains a supported operational sales database, customer +master Delta tables in external storage, monthly CSV targets, and application +logs. The product needs a governed dimensional sales mart, exploratory data +science access, and recent operational monitoring. + +### Place and move the data + +Evaluate database mirroring for supported sales tables that need low-latency +analytical replication without custom transforms. Use an external OneLake +shortcut for the customer master Delta table if source availability, credential +path, security, and schema ownership are acceptable. Use pipeline Copy for CSV +targets because they need scheduled acquisition, explicit mapping, file/run +metadata, and quarantine. Land application logs in the route best suited to +Eventhouse/KQL when time-series exploration dominates. + +The lakehouse holds open bronze and engineering tables used by Spark. The +warehouse serves the conformed star schema to T-SQL/BI consumers. This is not +unnecessary duplication if each representation has a documented workload and +refresh contract. Avoid copying the customer master again merely by habit, but +create a governed snapshot if consumers require isolation from source change or +outage. + +### Transform and validate + +Build Customer shaping in Dataflow Gen2 for maintainers who own Power Query: +explicit locale-aware types, trimmed/cased business key, anti-join for missing +reference values, duplicate detection, address cleanup, and a curated +destination. Preserve query folding for source filtering/projection as far as +supported and inspect it. Use notebooks for high-volume/order logic requiring +distributed processing and reusable tests. Use warehouse T-SQL for dimension and +fact application near the final model. Use KQL for log/time-window analysis +rather than moving logs into SQL only for language familiarity. + +Declare `SalesFact` grain: one row per posted order line. Load Customer and +Product dimensions before facts. Use surrogate keys, a stable unknown member, +and explicit Type 1/Type 2 attributes. Validate one current Type 2 row per +business key, nonoverlapping validity, fact foreign keys, distinct order-line +key, row count, quantity and amount controls. + +### Quality fixture + +Create a ten-row test fixture containing: + +- two versions of one order line, with the newer source sequence winning; +- one exact duplicate delivery; +- one missing customer that uses the governed unknown/inferred path; +- one customer business key duplicated in the dimension staging data; +- one invalid localized date; +- one null amount whose business rule forbids publish; +- one late order affecting yesterday's regional total; and +- one valid order at a Type 2 validity boundary. + +Write expected accepted, quarantined, superseded, and late-restatement counts +before execution. The duplicate dimension key must block or quarantine the +ambiguous dimension—not multiply fact totals. The late order must restate the +affected daily aggregate. The invalid date and forbidden null remain diagnosable +with rule and source reference. + +### Equivalent transformations + +Implement one filter, join, and regional aggregate in PySpark, T-SQL, and KQL on +the same normalized fixture. Compare key/type/null semantics and ordered output. +Do not claim equivalence from similar syntax. Power Query's Merge and Group By +version should produce the same declared output grain. Retain numerator and +denominator for average order value, then compute the ratio after aggregation. + +### Operations and change + +Monitor mirroring lag/change application, shortcut connectivity, Copy rows and +rejects, Dataflow refresh/folding/destination, notebook stages, warehouse load +controls, and consumer freshness. Rotate the shortcut connection in Test and +prove denied/allowed behavior. Rename a CSV column and verify schema handling +fails visibly. Move a disposable shortcut target and diagnose the reference +without deleting source data. + +The capstone is complete when every source has an accountable access/update +path; every tool owns a justified layer; the quality fixture matches expected +disposition; star-schema totals reconcile; late data restates the right period; +reruns are idempotent; and the published mart includes source/run/watermark and +quality evidence. + ## Exam distinctions - Merge joins columns; append stacks rows. diff --git a/content/published/dp700/chapters/lifecycle-management.md b/content/published/dp700/chapters/lifecycle-management.md index ca82cd1..ae06bd0 100644 --- a/content/published/dp700/chapters/lifecycle-management.md +++ b/content/published/dp700/chapters/lifecycle-management.md @@ -236,6 +236,67 @@ checks for a notebook-to-lakehouse pipeline. Then sketch a release with one feature branch, one pull request, one test promotion, and one intentional failure; label the evidence produced at each gate. +## Release scenario drills + +### Conflicting workspace and branch changes + +An engineer edits a notebook in the Fabric portal while another pull request +changes the same notebook definition in Git. The workspace now reports a +conflict. First preserve and compare both intentions; do not click update or +commit merely to clear the status. Decide the winning or combined definition in +an isolated branch/workspace, validate it, and use the normal pull-request gate. +Then update the integration workspace from the reviewed branch and run the +notebook. The conflict resolution is a content decision; synchronization is only +the mechanism that applies it. + +If the item runs locally but fails after update, inspect runtime, environment, +connections, data permissions, and unsupported properties. Git proves the +definition and review history, not all external state. Record both Git commit +and Fabric run ID so later investigators can connect source to execution. + +### Safe schema evolution + +A warehouse project changes `CustomerName varchar(200)` to `varchar(80)` and +renames `RegionCode` to `SalesRegionCode`. The project builds. That proves the +declarations resolve; it does not prove 200-character existing names fit or that +publish recognizes the rename instead of drop/create. Generate the deployment +plan against a realistic copy, profile maximum length/nulls, and explicitly +represent or stage the rename according to supported project tooling. Reject an +unreviewed destructive plan. + +For a zero-downtime transition, add the new representation, backfill and validate, +deploy compatible readers/writers, then remove the old object in a later release. +Keep a rollback or restore path and retain the exact artifact and plan. The exam +distinction is state-model validation versus safe data migration. + +### Successful deployment, failed application + +A deployment pipeline reports success when promoting a pipeline and notebook to +Test. The first execution cannot open the lakehouse. Compare paired items and +dependency bindings, then inspect stage-specific connection/variable values, +credentials, target permissions, and the execution identity. Deployment rules +can adapt supported values but do not supply secrets or grant access. + +Repair the narrow missing configuration, rerun a test input, validate target +rows and logs, and then record approval. Do not promote to Production because +the control-plane deployment succeeded. The gate should require definition +deployment plus target-environment smoke tests, data checks, negative permission +tests where relevant, and a rollback plan. + +### Choosing a release mechanism + +A team asks for “version control, database validation, and Dev/Test/Prod.” The +answer is not one feature. Connect the development workspace to Git for durable +definitions, branches, CI, and pull-request review. Build database projects to +validate declarative SQL schema and produce reviewed artifacts. Use deployment +pipelines or another selected release route to promote supported Fabric items +and environment configuration. + +Document unsupported items and post-deployment APIs rather than hiding them. +The complete release record contains issue/intent, commit and PR, CI results, +database artifact and plan where applicable, Fabric deployment operation, +configuration evidence, runtime smoke tests, approver, and source revision. + ## Exam distinctions - Git integration synchronizes a workspace and branch; it is not a Dev/Test/Prod promotion engine. diff --git a/content/published/dp700/chapters/loading-patterns.md b/content/published/dp700/chapters/loading-patterns.md index 1186eca..fc4da28 100644 --- a/content/published/dp700/chapters/loading-patterns.md +++ b/content/published/dp700/chapters/loading-patterns.md @@ -226,6 +226,74 @@ stream after logic changes? For practice, write control records for one batch increment and one streaming micro-batch, including state before, durable effects, validation evidence, and state after. +## Loading scenario drills + +Use each drill twice. First answer without looking at the explanation: name the +selection boundary, durable state, target application rule, validation equation, +failure recovery, and reconciliation. Then change one assumption—introduce +deletes, concurrent writers, late facts, or insufficient raw retention—and +redesign the pattern. A strong exam answer identifies the missing correctness +contract before choosing a Fabric tool. A strong production answer also records +who owns that contract, how it is observed, and which bounded input can be +replayed. If the design cannot explain what happens after a crash between target +write and state update, it is not yet complete. + +### Timestamp ties and late commits + +The source exposes millisecond `ModifiedAt`, but several rows share a timestamp +and one transaction commits late. Reading only `ModifiedAt > last_watermark` +can miss a tied row or late commit. At run start choose a fixed upper bound, +read a justified overlap from before the committed watermark, retain business +key plus source version/modified time, and select one trusted winner per key. +Apply idempotently and advance the watermark only after durable target success. + +If the source offers a stable composite cursor or CDC sequence, prefer it over +ambiguous wall-clock time. Reconcile a bounded source interval and alert on +unexpected gaps. Test a row with the exact previous timestamp, two versions of +one key, and a commit visible only on the next run. + +### Delete handling + +An incremental source sends inserts and updates but no delete flag. The target +will accumulate records removed upstream. Options include source CDC/tombstones, +a periodic key snapshot and anti-join, a bounded full comparison, or a business +rule that never physically deletes and instead closes validity. Choose from +source semantics and analytical requirements; a modified timestamp cannot infer +a row that no longer exists. + +Apply deletes only after validating scope and controls. An unexpectedly empty +snapshot could otherwise delete everything. Record deleted/expired counts, +protect required history, and make replay deterministic. Full loads naturally +reconcile absence when replace semantics are safe, which can make them preferable +for small tables. + +### Type 2 correction versus real change + +A customer address was entered incorrectly yesterday and corrected today. If +the organization wants historical reports to show the corrected address for all +time, treat it as Type 1 correction. If the customer genuinely moved and reports +must preserve the old address for prior facts, use Type 2: expire the current +row and add a new version. The source must distinguish correction from business +change or the warehouse needs an explicit policy. + +Test nonoverlapping validity, one current row, surrogate-key lookup at boundary +times, and facts arriving after the dimension but carrying earlier event time. +Do not use load time as business validity without acknowledging the consequence. + +### Streaming logic change and replay + +A bug undercounted events for three days. The current checkpoint faithfully +records progress through incorrect logic. Replaying requires retained raw events, +the affected source range, corrected code version, compatible/new checkpoint, +and a target plan that replaces or idempotently restates affected results. Simply +deleting the production checkpoint can replay more history than intended and +duplicate side effects. + +Build corrected output in isolation, reconcile to raw controls, atomically +publish or replace the affected partitions, and retain old/new version evidence. +The recovery design proves why raw retention and deterministic sink keys are +part of the loading contract, not optional observability. + ## Exam distinctions - Full versus incremental describes selection and application, not a specific Fabric tool. diff --git a/content/published/dp700/chapters/monitor-items.md b/content/published/dp700/chapters/monitor-items.md index 3abbf25..31192ec 100644 --- a/content/published/dp700/chapters/monitor-items.md +++ b/content/published/dp700/chapters/monitor-items.md @@ -222,6 +222,118 @@ auditable? Why use hysteresis for recovery? For practice, write one outcome, execution, and resource signal for pipeline, Dataflow, notebook, Eventstream, and semantic model; identify the authoritative screen/log for each. +## Monitoring scenario drills + +### Green run, stale consumer + +The latest pipeline and semantic-model refresh both succeeded, but the dashboard +is four hours stale. Compare newest business timestamp at source, committed +ingestion watermark, curated publish timestamp, refresh source boundary, and +consumer-visible maximum. The first boundary that stops advancing owns the +initial investigation. A green downstream job may have consumed stale upstream +data correctly. + +Alert on end-to-end freshness by dataset and cutoff, while retaining failed-run +alerts for execution diagnosis. Include last run IDs and stage timestamps. This +pair distinguishes “job failed” from “no fresh data arrived.” + +### Rising stream lag with no failures + +Input increases from 5,000 to 12,000 events/sec; output remains 7,000 and lag +grows. Examine boundary rates, micro-batch duration, state/shuffle, destination +latency, Eventhouse ingestion, and capacity. The absence of a red status does not +mean the service objective is healthy. Estimate time to exhaust backlog under +current rates and trigger before consumer freshness breaches. + +After tuning or scaling, verify output eventually exceeds input long enough to +drain backlog, no events were dropped, and cost remains acceptable. Recovery is +not merely the moment input falls. + +### Dataflow regression after a source release + +Refresh duration triples immediately after a source deployment. Output remains +correct. Compare folding indicators/native query, source rows/bytes, gateway +route, query steps, destination metrics, and capacity. A new source view or type +can prevent folding without generating an error. + +Record baseline and changed source/code versions. Repair foldability or move +selective operations earlier, then compare equivalent loads. Add a duration or +scan-volume anomaly alert only after understanding expected variation. + +### Alert fires but nobody can act + +An email says “pipeline failed” with no workspace, item, run ID, owner, or +runbook. The signal is technically correct and operationally weak. Redesign it +with dataset/item, environment, severity, failure/freshness context, correlation +ID, first diagnostic link, owner group, escalation, suppression key, and recovery +condition. Keep sensitive parameters and payloads out. + +Test delivery during staffed and after-hours paths, duplicate failure retries, +maintenance suppression, and recovery. An alert is complete only when the +intended responder can locate and classify the incident. + +## Capstone: the 06:00 operations review + +You are the on-call analytics engineer. Executives expect the sales dashboard by +06:00. The chain is source close → pipeline ingestion → Dataflow cleanup → Spark +enrichment → warehouse publish → semantic refresh. Shipment events supply a +separate live tile. Build a five-minute review that finds risk before the user +does. + +### Review order + +Start with the outcome: newest consumer-visible sales business date/time and +live shipment lag. Compare each with its service objective. Then inspect the +stage boundary timestamps and committed watermarks. Only after locating the +first stale boundary drill into execution and resource evidence. This prevents +spending the whole review on one slow but noncritical activity. + +Use Monitoring hub for supported run overview; pipeline/activity detail for +resolved runs and copy metrics; Dataflow refresh history/logs for query and +destination; Spark application/UI for stages and skew; warehouse query/monitor +for publish; semantic refresh history/summary for tables and partitions; +Eventstream/Eventhouse monitoring for rate, lag, invalid events, ingestion and +queries; and Capacity Metrics for shared throttling/CU context. + +### Morning evidence board + +Record a small operational table: + +```text +dataset | source_ready | ingested_to | curated_to | model_to | consumer_to + | status | run_id | rejects | freshness_lag | owner +``` + +For streaming add input rate, output rate, backlog/lag, late/invalid count, and +destination. For quality add evaluated and failed per critical rule. This table +contains references and aggregates, not raw sensitive payloads. + +### Three observations + +1. Sales ingestion succeeded to 05:40, Dataflow succeeded, but semantic model is + at 04:00. Refresh detail shows incremental partitions excluded the newest + boundary. Correct refresh policy/partition and retest; do not rerun ingestion. +2. Shipment input is 9,000/sec and output 6,000/sec with rising lag, no failure. + Inspect operator/destination and capacity; alert on sustained lag because + status alone is green. +3. Customer rejects rise from 0.02% to 4% after a source release. Stop publish + under the quality contract, preserve last known-good output, investigate + schema/type change, and communicate freshness impact. + +### Alerts and handoff + +Create distinct alerts for missed freshness cutoff, required job failure, +critical quality breach, sustained stream lag, and capacity throttling. Each +contains environment, dataset/item, current/threshold value, time window, +run/request ID, owner and runbook, deduplication, escalation, and recovery. +Suppress maintenance deliberately, not by muting broad channels. + +The review is successful when it identifies the first stale/failing boundary, +distinguishes outcome from execution and resource, preserves correlation IDs, +routes to an accountable owner, and verifies recovery at the consumer. A ticket +closed after rerun is insufficient if the source-to-consumer timestamp or +quality control remains wrong. + ## Exam distinctions - Monitoring hub gives centralized visibility; item detail provides engine-specific evidence. diff --git a/content/published/dp700/chapters/optimize-performance.md b/content/published/dp700/chapters/optimize-performance.md index 7823aaf..4459f92 100644 --- a/content/published/dp700/chapters/optimize-performance.md +++ b/content/published/dp700/chapters/optimize-performance.md @@ -310,6 +310,155 @@ practice, write an A/B optimization record with hypothesis, fixed conditions, three baseline and three changed runs, correctness checksum, p50/p95, bytes scanned, and CU/compute result. +## Optimization scenario drills + +### Fast query, expensive capacity + +A warehouse rewrite lowers median latency from 12 to 4 seconds but doubles CU +consumption and worsens p95 under concurrency. The change is not automatically +an improvement. Recheck result equivalence, scan/data movement, plan, repeated +work, and concurrency. Decide against the stated SLA and cost objective: perhaps +8 seconds at lower CU meets the business need and protects other workloads. + +Report latency distribution and cost together. A local speedup that causes +capacity throttling can make the overall product slower. Optimization ownership +extends beyond the single developer's test query. + +### Small files caused by streaming writes + +A Delta table receives frequent tiny micro-batches and accumulates thousands of +small files per day. Queries spend substantial time listing/planning and launch +many tiny tasks. Measure file distribution and scan/task evidence, then choose a +maintenance cadence or optimized-write strategy that compacts after sufficient +accumulation without running `OPTIMIZE` after every batch. + +Coordinate compaction, V-Order, readers, retention, and vacuum safety. Compare +end-to-end write plus maintenance cost with query benefit. If queries do not read +the table often, aggressive maintenance may cost more than it saves. + +### Parallelism cliff + +A Copy pipeline improves from concurrency 1 to 4, then slows at 16 while source +throttling and sink commit waits rise. Select the measured knee—perhaps 4 or 8— +and use backoff. Tune copy-internal parallelism and ForEach concurrency together +because their product determines pressure. Repartition source work so one giant +partition does not dominate. + +Repeat across representative volume and time, since external source limits can +vary. Validate target counts and file layout; faster transfer that produces a +small-file problem moves the cost downstream. + +### Broadcast assumption fails + +A lookup table was small during development but grows to several gigabytes. +Forced broadcast now causes executor memory pressure and failures. Remove the +assumption or define a guarded size threshold, inspect the adaptive/actual plan, +and choose a partitioned join strategy. Project only required lookup columns and +filter it before considering broadcast. + +Compare shuffle, spill, task distribution, executor memory, duration, and +correct totals. “Dimension” is a logical role, not proof that its physical size +is safe to copy to every executor. + +### Materialization freshness trade-off + +A KQL aggregation over raw events is run hundreds of times per hour. A +materialized view can shift repeated query computation to ingestion/maintenance, +but adds storage, update cost, and freshness/late-data semantics. Measure total +query savings against ingestion overhead and verify supported functions and +correction behavior. + +If users require immediate raw detail and only a few queries use the aggregate, +direct query may be better. If the aggregate is stable and dominant, materialize +with monitoring for health and lag. Cost is moved and amortized, not magically +removed. + +### Scaling after efficiency work + +Capacity Metrics shows sustained throttling across well-designed concurrent +workloads after unnecessary scans, skew, retries, and schedules are addressed. +Scaling or autoscale is now an evidence-backed option. Define expected demand, +budget, success metric, and rollback; compare throttling, latency, throughput, +and CU/cost after the change. + +Scaling is not a failure of engineering when demand exceeds the provisioned +resource. It is a poor first answer when one accidental Cartesian join or +unbounded stream state consumes the capacity. + +## Capstone: optimize without moving the bottleneck + +A Fabric product misses its 07:00 freshness objective. Pipeline elapsed time is +95 minutes, Spark transformation 40 minutes, warehouse publish 20 minutes, and +semantic refresh 25 minutes, but several stages overlap. Interactive reports +also slow during the load. The team proposes a larger capacity immediately. + +### Establish the critical path + +Create a timeline using trigger, queue, source extraction, Copy, Spark stages, +warehouse application, model refresh, and publish. Overlap means durations cannot +simply be added. Identify the earliest time each required output is ready and the +dependency that determines consumer readiness. Collect Capacity Metrics for +throttling and workload overlap, item run IDs, source/sink throughput, Spark UI, +warehouse query/plan, and refresh partition detail. + +Set fixed correctness controls: selected source rows = accepted + rejected, +target distinct business keys, amount totals, late partition count, and +consumer-visible maximum business time. Record three comparable baseline days +and their cache/concurrency/capacity state. + +### Competing hypotheses + +Pipeline evidence shows 6,000 tiny files and source throttling at high copy +parallelism. Spark shows large file-list/task overhead plus one skewed customer +key. Warehouse publication scans the full fact table despite processing one day. +Semantic refresh unnecessarily processes historical partitions. Capacity also +shows brief throttling while all operations overlap. + +These are four hypotheses, not one “Fabric is slow” diagnosis. Test in controlled +increments: + +1. Reduce Copy/ForEach concurrency to the measured throughput knee and consolidate + file boundaries, preserving restartable partitions. +2. Compact/optimize accumulated Delta layout at an evidence-based cadence and + address Spark skew through preaggregation or a targeted strategy. +3. Make warehouse application and query predicates prune the bounded date range; + update relevant statistics and inspect the changed plan. +4. Configure/validate semantic incremental partitions and schedule heavy stages + to reduce unnecessary contention where business dependencies permit. + +Do not apply every change at once. Each experiment retains input version, +configuration, p50/p95 or repeated duration, bytes/files/tasks/shuffle/scan, +capacity/CU, and correctness checksum. + +### Interpret results + +Suppose file consolidation cuts Spark planning/task overhead by 12 minutes; +skew repair removes a 9-minute tail; bounded warehouse processing saves 8 +minutes; incremental semantic refresh saves 10. Scheduling reduces p95 report +latency during the load. End-to-end readiness improves by only 25 minutes because +some savings overlap. Report critical-path improvement, not the sum of local +speedups. + +Capacity now shows no throttling on typical days but p95 seasonal days still +breach SLA. Model seasonal demand and compare further efficiency, schedule, +autoscale, or SKU change. Scaling is now evaluated against a residual measured +resource limit rather than masking small files, skew, full scans, and unnecessary +refresh. + +### Regression and operations + +Add thresholds for file-count/size distribution, Spark skew ratio and spill, +pipeline throughput/throttling, warehouse scan/plan regression, refresh duration, +capacity throttling, and end-to-end freshness. Avoid alerting on one noisy sample; +use appropriate windows and recovery. Maintenance jobs themselves consume +capacity, so schedule and measure compaction/materialization costs. + +The capstone passes when the same source produces identical accepted/rejected +counts and business totals; the critical path meets the objective on repeated +representative runs; p95 interactive performance does not regress; capacity cost +is reported; every maintenance/scale decision has ownership and rollback; and a +future data-volume change can be detected before the original bottleneck returns. + ## Exam distinctions - `OPTIMIZE` compacts Delta files; V-Order changes Parquet layout; partitioning organizes data for pruning. @@ -333,5 +482,6 @@ scanned, and CU/compute result. - [Lakehouse and Delta tables](https://learn.microsoft.com/en-us/fabric/data-engineering/lakehouse-and-delta-tables) - [Delta optimization and V-Order](https://learn.microsoft.com/en-us/fabric/data-engineering/delta-optimization-and-v-order) - [Warehouse performance guidelines](https://learn.microsoft.com/en-us/fabric/data-warehouse/guidelines-warehouse-performance) +- [Spark errors and performance troubleshooting](https://learn.microsoft.com/en-us/fabric/data-engineering/troubleshoot-spark) - [KQL query best practices](https://learn.microsoft.com/en-us/kusto/query/best-practices) - [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/content/published/dp700/chapters/orchestration.md b/content/published/dp700/chapters/orchestration.md index 85c6517..26d6464 100644 --- a/content/published/dp700/chapters/orchestration.md +++ b/content/published/dp700/chapters/orchestration.md @@ -139,9 +139,13 @@ manifest, completion marker, stable-size check, or upstream contract rather than an arbitrary delay. -**Worked trigger.** An event for -`landing/region=CA/business_date=2026-09-01/orders.parquet` passes the URL, -event ID, and modification timestamp to a pipeline. The first activity rejects +**Worked trigger.** An event arrives for this object: + +```text +landing/region=CA/business_date=2026-09-01/orders.parquet +``` + +It passes the URL, event ID, and modification timestamp to a pipeline. The first activity rejects unexpected paths and checks a control table keyed by URL plus version. A valid new object is processed and the key recorded atomically. A duplicate event finds the completed key and exits successfully without inserting rows again. A @@ -224,6 +228,141 @@ draw a parent pipeline with lookup, bounded fan-out, child invocation, failure collection, and a reconciliation trigger; label parameters, variables, and system values. +## Orchestration scenario drills + +### Choose the owner of each responsibility + +A finance process copies eight source tables, applies dimensional logic, and +publishes a quality summary. Use a pipeline to own schedule, dependencies, Copy +activities, parameters, retry classification, and failure routing. Use a +notebook or warehouse SQL procedure for code-first/set-based dimensional logic. +Use Dataflow Gen2 only where its visual Power Query transformations and supported +destinations improve maintainability. Do not translate tested SQL into dozens of +pipeline expressions. + +Define one run contract: business date, full/incremental flag, source upper +bound, target environment, and correlation ID. The pipeline passes only required +typed values; transformations return small status/count outputs. Success requires +quality and reconciliation, not merely every activity turning green. + +### Event burst and duplicate delivery + +Ten thousand object-created events arrive after a source outage, including +duplicates. An unbounded trigger-to-run design overwhelms the source and sink. +Filter expected paths, validate a stable object/version idempotency key, and +control run concurrency. A completed key exits successfully; an in-progress key +prevents unsafe overlap; a failed key is retriable according to target semantics. + +If events can describe partially written objects, require a manifest/completion +marker or other readiness contract. A later scheduled reconciliation compares +expected versus processed objects and submits gaps. Monitor event rate, queued +runs, source/sink throttling, duplicate suppression, failed keys, and freshness. + +### Metadata-driven loading with one bad entity + +A control table lists 40 entities. One row references a removed source column. +The parent lookup returns enabled rows and invokes a parameterized child with +bounded ForEach concurrency. The child validates metadata, extracts, stages, +checks schema, applies target changes, and returns counts. Thirty-nine succeed; +one fails deterministically. + +The parent records each result and fails the required overall run without +reapplying successful entities. Repair the metadata/source contract and rerun +only the failed entity. A blanket retry of the ForEach would waste capacity and +could duplicate outputs if sinks are not idempotent. Retain parent and child run +IDs to correlate the incident. + +### Clock schedule across daylight saving + +A daily job must run after a source closes at 01:30 local time, in a region with +daylight-saving transitions. Specify the business time zone explicitly and +decide behavior for the repeated or missing local hour. Better, use the source's +published completion signal when available and keep a cutoff reconciliation. +Record logical business date separately from UTC trigger timestamp. + +Prevent concurrent runs when the job replaces the same partition; allow parallel +business dates only if target isolation is proven. Alert on “no successful +business date by cutoff,” which catches a missing trigger as well as a failed +run. A simple alert on activity failure cannot detect a run that never started. + +## Capstone: metadata-driven retail ingestion + +A retailer receives daily customer and product snapshots, hourly order files, +and near-real-time shipment events. Build one operating design without forcing +all three sources through the same pattern. + +### Requirements and proposed design + +- Customer and product snapshots close at 02:00 local business time and must be + available before fact processing. +- Order files arrive by region and can be redelivered with the same object + version. All regions must be complete by 05:30. +- Shipment events should appear within five minutes, but correctness by next + morning matters more than never missing an event. +- Development, Test, and Production use different connections; credentials must + not appear in parameters or logs. + +Use a scheduled parent pipeline for snapshot dimensions. It validates the +business date, invokes parameterized child loads, checks uniqueness and counts, +and publishes a dimension-ready control only after both dimensions succeed. Use +event-triggered order ingestion keyed by object URL plus version, with bounded +concurrency and an idempotent target merge. A 04:30 reconciliation pipeline +compares the regional manifest with completed keys. Use Eventstreams or Spark for +continuous shipment processing, retain raw events, and schedule a bounded replay +comparison before the 05:30 cutoff. + +Connections or variable libraries provide environment configuration where +supported; a governed credential mechanism owns secrets. Parameters carry +business date, entity, object reference, source upper bound, and correlation ID. +Variables hold only mutable run state such as collected failure count. System +variables supply pipeline/trigger identifiers. + +### Control tables + +Design three small contracts: + +```text +entity_config(entity, source, target, load_type, watermark_column, enabled) +object_run(object_url, object_version, business_date, status, pipeline_run_id) +batch_control(entity, lower_bound, upper_bound, rows_read, rows_written, + rows_rejected, status, committed_at) +``` + +The parent reads enabled configuration and passes one typed row to each child. +The child selects a fixed interval, stages, validates, writes idempotently, and +commits its watermark after target success. `object_run` enforces duplicate +suppression. Do not use a pipeline variable as the durable watermark; variables +disappear with the run and are unsafe under concurrency. + +### Failure injections + +1. Deliver one order file twice. The second event should find the completed + object/version and exit without another logical write. +2. Make one region's schema incompatible. Other independent regions may finish, + but the cutoff reconciliation reports the missing required region and blocks + readiness. +3. Crash after target merge but before watermark update. A rerun reads the same + range and produces the same target through stable keys. +4. Withhold a shipment event from the trigger path while retaining it in raw + source. The scheduled comparison discovers and replays it. +5. Expire a Test connection. Deployment remains successful, while the runtime + smoke test fails authorization and prevents promotion. + +### Monitoring and acceptance + +Monitor schedule/trigger state, queue and activity duration, source rows/files, +input/output, rejects, watermarks, duplicate suppressions, reconciliation gaps, +stream lag, and consumer-visible freshness. Each failure alert carries +environment, entity/object, run ID, source boundary, first failing activity, and +runbook. A separate cutoff alert catches missing runs. + +The capstone is accepted only when every failure can be replayed from the +smallest safe boundary; secrets remain outside ordinary parameters; duplicate +delivery changes no totals; all control equations reconcile; and Test deployment +plus runtime/data checks pass. This is the central orchestration idea: the +pipeline coordinates evidence-backed state transitions while the selected +compute tool owns transformation. + ## Exam distinctions - A schedule answers *when*; an activity dependency answers *after what*. diff --git a/content/published/dp700/chapters/resolve-errors.md b/content/published/dp700/chapters/resolve-errors.md index 4ca8796..06940b2 100644 --- a/content/published/dp700/chapters/resolve-errors.md +++ b/content/published/dp700/chapters/resolve-errors.md @@ -258,6 +258,147 @@ shortcut path? For practice, write one incident record with time, scope, correlation ID, classification, evidence, correction, rerun boundary, outcome, and regression test. +## Cross-item incident drills + +### The downstream cascade + +A pipeline Copy fails authentication. Its notebook dependency is skipped, the +warehouse procedure never runs, and the semantic refresh later fails because a +staging table is absent. Start at the earliest pipeline authentication failure, +not the final semantic error. Identify the connection identity, credential +state, source permission, and whether rotation or ownership changed. Repair and +rerun from the smallest repeat-safe boundary. + +After recovery, validate copied counts, notebook output, warehouse transaction, +and model freshness. Add credential ownership/rotation monitoring and a pipeline +failure alert. The later errors are useful impact evidence but not separate root +causes. + +### Intermittent versus data-dependent + +A notebook task fails on roughly the same partition every retry. Increasing +automatic retries makes the run longer but not more reliable. Compare failed +task input/key range and exception. If the same malformed record, oversized +group, or serialization path recurs, classify deterministic and isolate/correct +it. If different workers fail with transient service/network evidence, bounded +retry may be appropriate. + +Build a small fixture containing the failing record or skewed key and preserve +it as regression input. A production retry policy should not conceal the +difference between repeatable code/data faults and genuinely transient worker +loss. + +### Permission denied after deployment + +A pipeline worked in Development but fails in Test with access denied. The +definition and parameters deployed, but target connection/identity permission +did not. Compare resolved Test connection, executing identity, gateway/source +ACL, workspace/item permissions, and secret binding. Do not grant Contributor +to make one source read work. + +Apply the narrow connection or data permission, test the allowed operation and +a denied alternate operation, and add the binding/permission check to release +smoke tests. Deployment success and runtime authorization are different gates. + +### Empty KQL result after successful ingestion + +Eventhouse ingestion metrics show accepted events, but the query returns none. +Start with the correct database/table and `take` or a broad known time range. +Inspect ingestion time and parsed event time, then add each `where`, dynamic +parse, and join operator one at a time. A future time-zone conversion or null +event-time filter can eliminate every row. + +Retain the request/query ID and input/result counts at each reduction. If a +broad query is also empty, return to ingestion mapping/table evidence. This +prevents changing a working ingestion path to repair a query-boundary problem. + +### Shortcut works for owner only + +The creator can query an external shortcut, while consumers receive denied +errors. Enumerate creator elevation, consumer workspace/item/OneLake permission, +connection credential mode, and source ACL. Test with a clean representative +consumer rather than impersonating through an owner session. Confirm Tables +versus Files and engine-specific requirements after authorization is understood. + +Do not solve it by sharing source credentials or granting broad workspace +write. Choose the supported least-privilege credential/access design, rotate it +through governed ownership, and preserve a negative test. + +### Repair without losing incident evidence + +During an outage, responders are tempted to edit multiple parameters, delete a +checkpoint, recreate a shortcut, and scale capacity simultaneously. That can +erase the causal trail. Preserve run/request IDs, logs, resolved values, +checkpoint/target metadata, source version, and timestamps first. Change one +hypothesized cause in a safe scope and compare the outcome. + +Emergency restoration can justify a faster roll-forward or rollback, but record +each action and its evidence. After service returns, reproduce the fault safely, +add a regression, update the runbook, and remove temporary excess permission or +capacity. + +## Capstone: one incident, seven surfaces + +At 03:10, a source schema release changes `CustomerId` from integer to text and +renames `eventTime`. The batch pipeline maps the old integer, Dataflow performs +an implicit type conversion, a notebook joins on mismatched types, Eventstream +windows reference the old field, Eventhouse receives some malformed records, a +warehouse procedure encounters conversion errors, and a lakehouse shortcut still +points to valid source data. Several red symptoms share one upstream change but +must be proven, not assumed. + +### Evidence funnel + +Record incident window, deployment/source version, affected workspace/items, +identities, pipeline/Dataflow/Spark/Eventhouse/query IDs, and consumer impact. +Find the earliest boundary: source contract versus pipeline resolved mapping. +Then trace each branch: + +- Pipeline: old mapping and resolved inputs fail conversion. Update the explicit + contract after source ownership confirms the change. +- Dataflow: preview may not include new values; refresh detail identifies the + first conversion or renamed-column step. Use explicit text handling and a + quarantine query. +- Notebook: Spark plan/tasks show join keys with incompatible types. Normalize + once at the validated boundary, not through scattered casts. +- Eventstream: raw input continues, but curated output falls to zero after the + window operator because `eventTime` is absent. Version/normalize before + windowing and replay raw retained events. +- Eventhouse: ingestion results distinguish accepted from mapping/type rejects; + KQL query logs are not the source of ingestion truth. +- T-SQL: preserve error number/message and offending staged values; correct + staging type/mapping before target constraints. +- Shortcut: verify target/path/connection and known query. If it still works, + do not recreate it just because adjacent consumers fail schema expectations. + +### Recovery plan + +Freeze the committed batch watermark if target application did not succeed. +Deploy compatible normalization to Test, run a fixture containing legacy and new +schema versions, and compare explicit dispositions. Replay the bounded batch +range idempotently. Replay streaming raw events from the source/version boundary +using controlled checkpoint/target state. Reconcile source = accepted + +quarantined, warehouse totals, window counts, and consumer freshness. + +Avoid broad retry while deterministic mappings remain wrong. Avoid editing all +seven items independently when a shared schema adapter/contract boundary can +normalize both versions. Keep the last known-good published output until the +corrected version passes controls. + +### Regression and post-incident controls + +Store the breaking records as sanitized fixtures. Add pre-publish schema- +compatibility tests, explicit type/field assertions, Dataflow error-rate checks, +notebook join-key tests, Eventstream schema-version routing, Eventhouse ingestion- +reject alerting, staging constraints, and end-to-end freshness/reconciliation. +Update source change notification and release coordination. + +The incident closes only when every surface is classified as root cause, +consequence, or unaffected; all affected source data has a disposition; replay +is repeat-safe; consumer output is fresh and correct; temporary changes are +removed; and the exact breaking schema change can no longer pass the regression +gate unnoticed. + ## Exam distinctions - Retry transient faults; correct deterministic faults. @@ -280,6 +421,9 @@ and regression test. - [Pipeline troubleshooting guide](https://learn.microsoft.com/en-us/fabric/data-factory/pipeline-troubleshoot-guide) - [Monitor Dataflow Gen2 refreshes](https://learn.microsoft.com/en-us/fabric/data-factory/dataflows-gen2-monitor) +- [Spark errors overview in Microsoft Fabric](https://learn.microsoft.com/en-us/fabric/data-engineering/troubleshoot-spark) - [Manage and monitor a KQL database](https://learn.microsoft.com/en-us/fabric/real-time-intelligence/manage-monitor-database) +- [Real-Time Intelligence overview](https://learn.microsoft.com/en-us/fabric/real-time-intelligence/overview) +- [Troubleshoot Fabric Data Warehouse](https://learn.microsoft.com/en-us/fabric/data-warehouse/troubleshoot-fabric-data-warehouse) - [OneLake shortcuts](https://learn.microsoft.com/en-us/fabric/onelake/onelake-shortcuts) - [Official DP-700 study guide](https://learn.microsoft.com/en-us/credentials/certifications/resources/study-guides/dp-700) diff --git a/content/published/dp700/chapters/security-governance.md b/content/published/dp700/chapters/security-governance.md index ce0f133..7c2eeb9 100644 --- a/content/published/dp700/chapters/security-governance.md +++ b/content/published/dp700/chapters/security-governance.md @@ -318,6 +318,142 @@ Finally, design a five-row access matrix for an administrator, engineer, analyst, report-only consumer, and unassigned user; include at least one denied test for workspace, item, SQL, and OneLake access. +## Security scenario drills + +### Report consumer versus direct lake access + +Nora must view a governed Power BI report but must not browse lakehouse files or +query its SQL endpoint. Share the report/app through the supported consumption +path and provide only the semantic-model/report permissions that path requires. +Do not add Nora as Contributor and do not grant `ReadAll` merely because the +report uses OneLake-backed data. Test report rendering, build/export behavior as +required, direct SQL, direct OneLake, unrelated item discovery, and reshare. + +If the report works but direct SQL fails, that may be the intended least- +privilege outcome. If the report fails, trace the semantic model's connection, +identity, Direct Lake/fallback behavior, model security, and source permissions +instead of granting a broad workspace role. Record the successful consumer path +and every denied alternate route. + +### Regional analysts with protected attributes + +Analysts may see orders only for assigned regions and must not see personal +email; Finance may see all regions and email. Start with group-based identities +and ensure analysts do not inherit Admin, Member, or Contributor. Grant the +appropriate item/data access, implement a region policy using a governed +identity-to-region mapping, and use CLS/object design to remove Email from the +analyst path. DDM may reduce accidental display but is not the primary boundary. + +Test an analyst in one region, an analyst assigned to two regions, Finance, an +unassigned user, a null/unknown region, and Spark/SQL/Direct Lake routes in scope. +Count expected versus actual rows and columns. If an analyst sees everything, +inspect additive OneLake roles and workspace write before rewriting the RLS +predicate. + +### Labeled and certified does not mean authorized + +A curated lakehouse is labeled Highly Confidential and certified as the trusted +source. A new user still cannot open it. This is consistent: the sensitivity +label classifies and may protect supported flows; certification signals that an +authorized governance process trusts the asset; neither grants the user an item +or data permission. Determine the business requirement and grant the narrow +workspace/item/OneLake or endpoint permission only if approved. + +Conversely, a user with access can still mishandle content if the label/export +route is not supported or organizational policy is absent. Test the exact item, +downstream inheritance/export path, and access separately. Review certification +when owner, quality, freshness, security, or documentation changes. + +### Audit an unexpected permission change + +A restricted table became readable to a contractor. Preserve the observation, +identity, time zone, item, endpoint, and current effective memberships. Search +Purview Audit over an appropriate window for share, permission, role, or group- +related operations available to the investigation; export with query criteria. +Compare audit history with current workspace roles, direct item grants, OneLake +roles, SQL/model roles, and Entra group membership. + +Audit can show recorded actions but may not encode the entire present effective- +permission calculation, and absence can reflect retention, ingestion delay, or +filter choice. After removing the unintended grant, re-test denied and approved +users, identify the control gap, and add periodic access review or alerting. Keep +sensitive audit evidence under restricted investigation handling. + +### Shortcut across security boundaries + +A lakehouse shortcut points to an external storage account. The shortcut object, +Fabric item, OneLake role, connection identity, and source ACL all participate. +Granting the user access to see the lakehouse does not necessarily authorize the +target, and a workspace Contributor can be broader than the intended narrow +role. Document whether access uses a shared connection, delegated identity, or +another supported credential route. + +Test an owner, intended reader, denied reader, and expired/rotated connection. +Deleting the shortcut removes the reference, not external data. Sensitivity and +endorsement should be reviewed for the exposed Fabric item, but neither repairs +the underlying authorization path. + +## Capstone: governed analytics workspace + +A health-services analytics team (using synthetic exam data) needs engineers to +build, regional analysts to query only their region, executives to consume one +report, and an external auditor to inspect access-change evidence for a limited +period. Design the boundary before assigning roles. + +Place engineers in an Entra group with Contributor only in the development +workspace. Keep workspace Admin and Member groups small and separate. Analysts +and executives should not become Contributors in Production. Share the required +warehouse/lakehouse or semantic/report item and grant the documented data path: +regional row policy plus protected columns for analysts; report-only consumption +for executives. Give the audit investigator controlled Purview Audit access and +an evidence export location, not broad data authoring. + +Classify sensitive items through published Purview labels and verify supported +inheritance/export routes. Certify the curated semantic model only after owner, +source, quality, freshness, security, documentation, and support criteria pass. +Certification signals trusted reuse; the label communicates sensitivity; the +permission layers enforce access. + +### Expected access matrix + +| Identity | Workspace author | Direct curated data | Other regions | Sensitive column | Report | Audit evidence | +| --- | --- | --- | --- | --- | --- | --- | +| Platform Admin | Yes | Elevated | Elevated | Elevated | Yes | As assigned | +| Engineer | Dev only | Dev/build need | Dev scope | Dev need | Test | No by default | +| Regional analyst | No | Assigned region | Denied | Denied | Yes | No | +| Executive | No | No direct route | No direct route | No direct route | Yes | No | +| Auditor | No | No | No | No | No | Time-bounded investigation | +| Unassigned user | No | Denied | Denied | Denied | Denied | Denied | + +“Elevated” is a warning in the matrix: administrators and Contributors are poor +negative-test identities because workspace write can supersede narrow OneLake +grants. Test each persona with clean membership and record effective groups, +endpoint, action, expected and actual result. + +### Changes and investigation + +Simulate four events: an analyst changes region, an engineer leaves, a label is +changed, and an unintended direct item share is added. Group-driven regional +mapping and engineer access should update through governed identity processes. +The label change follows information-protection policy and review. The direct +share should be discovered through access review and investigated through the +relevant audit activities and current effective-permission state. + +Preserve audit query criteria, UTC/local time interpretation, operation/user/item +identifiers, exported evidence, and investigator. Remove the unintended share, +retest both allowed and denied paths, and add a preventive control. Do not infer +from a missing search result that no change occurred until retention, ingestion +delay, licensing, operation naming, and filters are considered. + +### Acceptance evidence + +The capstone passes when least-privilege groups and owners are documented; +workspace, item, OneLake/SQL/model policies form one traced path; regional and +column tests include denied cases; direct lake access remains unavailable to +report-only users; label and endorsement behavior is verified separately from +authorization; shortcut/source credentials are owned and rotated; and a benign +permission change can be found, explained, remediated, and retested. + ## Exam distinctions - Workspace roles are broad collaboration grants; item permissions are narrower. diff --git a/content/published/dp700/chapters/streaming-data.md b/content/published/dp700/chapters/streaming-data.md index 0a8f179..cebc5c5 100644 --- a/content/published/dp700/chapters/streaming-data.md +++ b/content/published/dp700/chapters/streaming-data.md @@ -299,6 +299,140 @@ How can a KQL query distinguish parse failures from filtered values? How many operational contract listing event ID, event time, allowed lateness, checkpoint, raw retention, sink key, replay method, metrics, and owner. +## Streaming scenario drills + +### Hot live data plus long OneLake history + +Operations needs subsecond KQL over the last two hours; analysts occasionally +query two years of Delta history in OneLake. Ingest live events to a native +Eventhouse table for predictable indexed serving and route raw events to durable +OneLake. Expose historical Delta through a shortcut. If users frequently join a +recent historical window to live data, measure query acceleration for that +window; do not cache two years merely because it exists. + +Define event ID/time and schema consistently across both representations. Test +the handoff boundary so no events are missing or double-counted. Native retention +and cache horizon differ from raw historical retention. An accelerated external +table still cannot replace native policies/features that its documented +limitations exclude. + +### Poison event without stopping the stream + +One producer changes `temperature` from number to object for a single message. +Raw retention receives the original event. Eventstream or Spark validation +routes the malformed event to restricted quarantine with source position, +schema version, reason, and replay reference; valid events continue. A quality +metric and threshold determines whether the incident is warning or stops +publication. + +Do not log the entire payload indiscriminately or retry the same deterministic +event forever. After the producer or transform is corrected, replay the event +using its stable ID and prove the sink has one logical result. Monitor invalid +rate because one tolerated event can become a breaking schema rollout. + +### Hopping-window double counting + +A ten-minute window hops every two minutes. One event can belong to five windows. +This is correct for a moving view, but summing those window totals into a daily +total counts the same event multiple times. Use a nonoverlapping source +aggregation for additive daily totals or compute daily directly from events. + +Hand-calculate boundary timestamps, including an event exactly at the end of a +window, using the engine's interval semantics. Add events arriving within and +beyond the watermark. Validate emitted mode/finality, late-event handling, and +time zone before comparing totals. + +### Stateful Spark restart + +A structured-streaming job uses a unique checkpoint and idempotent Delta merge. +After restart with unchanged compatible logic, it resumes from checkpointed +progress and state. If a developer points a different query at that checkpoint, +state/query incompatibility or incorrect recovery can result. Restore the +intended code or create a controlled new checkpoint and replay range into an +isolated target. + +Compare source offsets, checkpoint, state schema, last committed sink batch, and +target keys. Never delete the checkpoint as a first troubleshooting step. A +checkpoint protects engine progress; the stable merge key protects target +effects. + +### Streaming join state explosion + +Two infinite streams are joined by device ID without time bounds. The engine +must retain unbounded history because any future event could match any prior +event. Add event-time watermarks and a business-valid time-range condition—for +example, readings may match commands within five minutes. Measure state rows and +memory, late/unmatched behavior, and output correctness. + +If the relationship is actually stream-to-small-static reference data, use the +appropriate static/broadcast or lookup pattern instead of a stream-stream join. +The engine decision follows state semantics, not just syntactic availability. + +## Capstone: fleet telemetry in real time + +A fleet emits location-free synthetic engine telemetry: `event_id`, `vehicle_id`, +`event_time`, `metric`, `value`, and `schema_version`. Operations needs five- +minute alerts, engineers need 30-day KQL exploration, and data science needs two +years of replayable Delta history. + +### Architecture + +Use Eventstreams to authenticate the event source, normalize field names/types, +filter supported schema versions, branch raw events to OneLake, route curated +events to Eventhouse, and send a derived condition stream toward Activator where +appropriate. Eventhouse native tables provide low-latency KQL and 30-day +retention/cache aligned to operational use. OneLake holds durable Delta history. +Spark Structured Streaming performs a custom stateful correlation only if the +visual/KQL route cannot express it maintainably. + +Define source partition/offset, stable event ID, event-time UTC contract, allowed +lateness distribution, duplicate horizon, raw retention, schema ownership, +checkpoint, sink key, and replay process. Keep the raw path less transformed so +future logic can be recomputed. + +### Derived products + +Create tumbling five-minute counts and average value by vehicle/metric for +operational tiles. A separate hopping 15-minute window every five minutes can +detect a moving threshold; never sum those overlapping windows for a daily +total. Use a session window only for a question genuinely defined by inactivity. +In KQL, constrain time early, project needed columns, parse dynamic content only +when required, and retain request/query IDs for diagnosis. + +For historical-to-live comparison, expose Delta through a shortcut. Measure +query acceleration for the recent historical period frequently joined to live +data; keep standard external access for infrequent broad history. Choose native +ingestion instead if required policies/features are unsupported on accelerated +external tables. + +### Test sequence + +Inject 20 known events including one duplicate ID, one prior schema version, one +invalid numeric value, two events out of order but within watermark, one beyond +watermark, and events on exact window boundaries. Before execution, calculate +expected raw, curated, quarantine, deduplicated, window, late, and alert counts. +Check each Eventstream boundary and destination, Eventhouse ingestion result, +KQL aggregate, Spark progress/checkpoint if used, and OneLake retained raw count. + +Stop and restart the stateful query with the same checkpoint; output must remain +logically identical. Then run corrected logic from a controlled new checkpoint +over a bounded raw range into an isolated target, reconcile, and publish. Deny +one Eventhouse destination connection and verify raw retention continues and +the incident is localized. + +### Operations and acceptance + +Monitor source/input/output rates, lag/backlog, invalid and late counts, state +size, micro-batch duration, checkpoint progress, destination ingestion, KQL +latency/scanned data, hot-cache coverage, and capacity. Alerts use sustained lag +and data-quality thresholds with owner, runbook, deduplication, and recovery. + +The capstone passes when all 20 events have an explicit disposition; duplicate +delivery produces one logical curated event; window math matches hand results; +restart is repeat-safe; raw replay repairs a logic defect; source and destination +failures are distinguishable; and the two-year history does not force every +operational query to scan two years. + ## Exam distinctions - Native Eventhouse tables ingest and index; shortcuts query supported data in place. diff --git a/content/published/dp700/chapters/workspace-settings.md b/content/published/dp700/chapters/workspace-settings.md index 861cd5d..b69511f 100644 --- a/content/published/dp700/chapters/workspace-settings.md +++ b/content/published/dp700/chapters/workspace-settings.md @@ -257,6 +257,72 @@ unexpected pool, check the workspace default and whether item customization is allowed. If tasks queue after the environment is running, examine worker concurrency and task demand before increasing compute. +## Scenario drills + +### Spark: one workload, different runtime + +A workspace uses the starter pool and the default runtime for every job. One +machine-learning notebook needs a newer runtime, a published library, and more +executor memory. Other notebooks must remain unchanged. The workspace Admin +should first allow item-level compute customization, then the owner should +configure an environment with the required runtime, library, pool/resources, +save and publish it, and attach it to the notebook. Changing only `spark.conf` +does not select the environment runtime or driver/executor allocation. + +The negative tests matter: confirm another notebook still inherits the workspace +default; confirm the target notebook fails predictably if the environment is +unpublished or incompatible; and confirm the scheduled identity can use the +environment and data. If all notebooks instead need the same change, revise the +workspace default rather than creating many identical overrides. + +### Domain: discovery without accidental authorization + +The organization wants all new Sales workspaces discoverable under Sales, with +Sales data stewards managing supported governance settings. Existing Finance +workspaces must not move, and catalog organization must not grant sales staff +access. Create a Sales domain, appoint the appropriate domain administrators and +contributors, configure a default domain scoped to Sales creators/groups, and +delegate only supported governance settings. Existing assigned workspaces remain +assigned; eligible unassigned and new workspaces follow the default mechanism. + +Test catalog placement and authorization separately. A user who can find the +workspace through the domain should still fail to open its items without a +workspace/item/data grant. A contributor assigning a workspace must also +administer that workspace. This is the exam pattern: domains organize and +delegate; security controls authorize. + +### OneLake: compliance logs and historical data + +A compliance workspace needs access diagnostics protected from modification for +180 days. Curated tables are queried hourly, while exported evidence is rarely +read after 30 days. Select a same-capacity diagnostic lakehouse satisfying +network constraints; grant the configuring workspace Admin appropriate +destination access; enable diagnostics; and apply the documented immutability +window. Define privacy, reviewer, and cleanup ownership before protection begins. + +Keep active curated data hot. Use a path-scoped lifecycle rule for eligible +evidence exports after modeling cool/cold minimum retention, access, transaction, +and early-deletion cost. Monitor activation and asynchronous lifecycle timing. +Diagnostics creates evidence, immutability protects it, lifecycle manages its +tier, and a later cleanup process removes it after legal retention. No one +setting performs all four jobs. + +### Airflow: startup objective versus worker capacity + +A production team needs DAGs to begin within a predictable window at 06:00 and +runs eight independent worker tasks. The managed starter pool is acceptable for +development but its availability and configuration do not meet the production +objective. Evaluate a custom pool whose active uptime covers the schedule, +choose node size from task demand, choose extra nodes from justified concurrent +workers, and use autoscaling only when variable demand warrants it. Assign an +owner for pause/resume, cost, and monitoring. + +If the environment takes time to resume, adjust uptime/pool readiness. If it is +running but tasks wait, examine worker concurrency and task duration. If one task +is slow, fix or size for that task rather than adding workers. Finally, recheck +current network limitations from Microsoft Learn before assuming private/VNet +connectivity is supported. + ## Exam distinctions diff --git a/content/published/dp700/coverage.yaml b/content/published/dp700/coverage.yaml index a43957b..df22238 100644 --- a/content/published/dp700/coverage.yaml +++ b/content/published/dp700/coverage.yaml @@ -73,7 +73,7 @@ objectives: chapter_id: lifecycle-management status: complete subtopics: [Git providers, Workspace connection, Branches, Commit and update direction, Conflicts, Supported items, Secrets] - source_ids: [dp700-study-guide, fabric-cicd-overview] + source_ids: [dp700-study-guide, fabric-cicd-overview, fabric-cicd-workflows] gaps: [] evidence: conceptual_model: version-control-model @@ -90,7 +90,7 @@ objectives: chapter_id: lifecycle-management status: complete subtopics: [Declarative schema, Project structure, Build validation, DACPAC, Publish plan, State versus migration, Data-loss safeguards] - source_ids: [dp700-study-guide, fabric-cicd-overview] + source_ids: [dp700-study-guide, fabric-cicd-overview, fabric-cicd-workflows] gaps: [] evidence: conceptual_model: database-projects-model @@ -107,7 +107,7 @@ objectives: chapter_id: lifecycle-management status: complete subtopics: [Stages, Workspace assignment, Item pairing, Comparison, Deployment rules, Dependency binding, Post-deployment checks] - source_ids: [dp700-study-guide, fabric-cicd-overview] + source_ids: [dp700-study-guide, fabric-cicd-overview, fabric-cicd-workflows] gaps: [] evidence: conceptual_model: deployment-pipelines-model @@ -260,7 +260,7 @@ objectives: chapter_id: orchestration status: complete subtopics: [Dataflow Gen2, Pipeline, Notebook, Transformation versus orchestration, Personas, Scale and maintainability] - source_ids: [dp700-study-guide, pipeline-overview] + source_ids: [dp700-study-guide, pipeline-overview, data-factory-overview] gaps: [] evidence: conceptual_model: choose-tool-model @@ -532,7 +532,7 @@ objectives: chapter_id: streaming-data status: complete subtopics: [Native ingestion, Indexed storage, External Delta, Feature support, Latency, Cost, Source availability] - source_ids: [dp700-study-guide, real-time-intelligence-overview] + source_ids: [dp700-study-guide, real-time-intelligence-overview, rti-onelake-shortcuts] gaps: [] evidence: conceptual_model: native-shortcut-comprehensive @@ -736,7 +736,7 @@ objectives: chapter_id: resolve-errors status: complete subtopics: [Driver errors, Executor failures, Spark UI, Runtime, Libraries, Lakehouse attachment, Memory, Skew and bad records] - source_ids: [dp700-study-guide] + source_ids: [dp700-study-guide, spark-troubleshooting] gaps: [] evidence: conceptual_model: resolve-notebook-comprehensive @@ -753,7 +753,7 @@ objectives: chapter_id: resolve-errors status: complete subtopics: [Ingestion failures, Mapping and format, Permissions, Table policies, KQL request IDs, Capacity, Empty results] - source_ids: [dp700-study-guide] + source_ids: [dp700-study-guide, eventhouse-manage-monitor] gaps: [] evidence: conceptual_model: resolve-eventhouse-comprehensive @@ -770,7 +770,7 @@ objectives: chapter_id: resolve-errors status: complete subtopics: [Connector state, Source rate, Transform schema, Event time, Destination health, Poison events, Backpressure] - source_ids: [dp700-study-guide] + source_ids: [dp700-study-guide, real-time-intelligence-overview] gaps: [] evidence: conceptual_model: resolve-eventstream-comprehensive @@ -787,7 +787,7 @@ objectives: chapter_id: resolve-errors status: complete subtopics: [Error numbers, Context and names, Permissions, Conversions, Constraints, Transactions, Deadlocks, Timeouts and plans] - source_ids: [dp700-study-guide] + source_ids: [dp700-study-guide, warehouse-troubleshooting] gaps: [] evidence: conceptual_model: resolve-tsql-comprehensive @@ -889,7 +889,7 @@ objectives: chapter_id: optimize-performance status: complete subtopics: [Spark UI, Partition pruning, Shuffle, Skew, Broadcast joins, Repartition, Coalesce, Cache, Compute sizing] - source_ids: [dp700-study-guide, lakehouse-delta-tables] + source_ids: [dp700-study-guide, lakehouse-delta-tables, spark-troubleshooting] gaps: [] evidence: conceptual_model: optimize-spark-comprehensive diff --git a/docs/dp700-completeness-report.md b/docs/dp700-completeness-report.md new file mode 100644 index 0000000..356760f --- /dev/null +++ b/docs/dp700-completeness-report.md @@ -0,0 +1,99 @@ +# DP-700 comprehensive content completeness report + +**Report date:** September 2, 2026 +**Blueprint:** DP-700 study guide effective July 21, 2026 +**Tracking:** [Issue #36](https://github.com/troyscott/study-reader/issues/36) · +[PR #37](https://github.com/troyscott/study-reader/pull/37) + +## Outcome + +The DP-700 book satisfies the approved comprehensive-content contract. All 54 +measured objectives are represented exactly once in the book manifest and +coverage audit. Every objective is marked complete only after its evidence map +identifies durable chapter blocks for all ten rubric categories and retains no +known content gap. + +| Measure | Result | +| --- | ---: | +| Blueprint domains | 3 | +| Published chapters | 10 | +| Measured objectives | 54 | +| Objectives with complete evidence | 54 | +| Objectives with unresolved gaps | 0 | +| Original chapter words | 35,081 | +| Registered Microsoft Learn sources | 41 | + +## Depth contract + +Every objective maps evidence for: + +1. conceptual model and terminology; +2. prerequisites, responsibilities, and security boundaries; +3. procedure or operational workflow; +4. decisions and trade-offs; +5. a worked example; +6. limitations and failure modes; +7. troubleshooting or monitoring; +8. exam distinctions; +9. active-recall questions; and +10. a scenario or mini-lab. + +The chapters add end-to-end capstones for metadata-driven orchestration, +governed workspaces, a trustworthy sales mart, fleet telemetry, operations +review, multi-surface incident response, and measurement-led optimization. +Dataflow Gen2 receives specific treatment of types and locale, cleaning, merge +versus append, grouping, pivot/unpivot, schema behavior, query folding, +destinations, quality outputs, and failure diagnosis. + +## Sources and originality + +Microsoft Learn remains authoritative for DP-700 and Microsoft Fabric behavior. +The book is original study-oriented writing rather than copied Learn pages. Each +registered source records its canonical Learn URL, retrieval timestamp, +SHA-256 content hash, and affected chapters. A content test rejects any Microsoft +Learn link in a chapter that lacks a provenance record. + +The source set includes the official study guide and direct product material for +workspace configuration, OneLake, CI/CD, security, orchestration, Dataflow Gen2, +dimensional loading, mirroring, Real-Time Intelligence, Structured Streaming, +monitoring, troubleshooting, Delta/V-Order, warehouse performance, and KQL best +practices. + +## Verification evidence + +The final local verification produced: + +- Ruff formatting and lint: passed; +- mypy: passed with no issues across 14 source files; +- pytest: 29 passed; +- statement coverage: 92.68%, above the 90% gate; +- browser-state tests: 5 passed; +- Git diff whitespace validation: passed; +- objective audit: 54 complete, 0 incomplete; +- content-depth gate: 35,081 words, above the approved 35,000-word minimum; and +- source-link provenance gate: passed for every chapter. + +## Responsive reader review + +All ten chapters were loaded through the running FastAPI reader at a 390 × 844 +iPhone-sized viewport. Each chapter title and complete control rendered, all ten +pages stayed within the 390-pixel document width after two long-inline-code +regressions were corrected, and no console errors were present. A representative +1,440 × 900 desktop pass confirmed the table of contents and centered reading +pane remained visible without document overflow. + +The browser sweep validates rendered structure and responsive boundaries. It is +not a substitute for executing Fabric examples against a live Fabric tenant; +the examples and labs are study material grounded in the recorded official +sources. + +## Refresh and preservation boundaries + +Durable Markdown block IDs back objective evidence and reading-position anchors. +The coverage contract makes missing evidence fail validation. Stable book, +chapter, objective, source, and block identifiers allow future source refreshes +to rebuild affected chapters while preserving local progress, bookmarks, notes, +and highlights according to the reader's existing stable-ID design. + +This report supports human review of PR #37. It does not authorize merge; the +pull request remains the review gate until the reviewer approves it. diff --git a/tests/content/test_dp700_outline.py b/tests/content/test_dp700_outline.py index bd9bdc0..802a51d 100644 --- a/tests/content/test_dp700_outline.py +++ b/tests/content/test_dp700_outline.py @@ -1,5 +1,6 @@ """Official DP-700 blueprint and source-registry tests.""" +import re from datetime import date from pathlib import Path @@ -69,12 +70,28 @@ def test_source_registry_records_current_learn_provenance() -> None: assert set(book.sources[0].chapter_ids) == {chapter.id for chapter in book.chapters} +def test_every_chapter_learn_link_has_registered_provenance() -> None: + catalog = BookCatalog(CONTENT_ROOT) + book = catalog.load_book("dp700") + registered_urls = {str(source.url).rstrip("/") for source in book.sources} + + for chapter in book.chapters: + markdown = catalog.load_chapter_markdown(book, chapter) + chapter_urls = { + url.rstrip("/") + for url in re.findall(r"https://learn\.microsoft\.com[^\s)>]+", markdown) + } + assert chapter_urls <= registered_urls + + def test_every_chapter_is_published_substantive_and_directly_sourced() -> None: catalog = BookCatalog(CONTENT_ROOT) book = catalog.load_book("dp700") + total_words = 0 for chapter in book.chapters: markdown = catalog.load_chapter_markdown(book, chapter) + total_words += len(markdown.split()) assert chapter.status == "published" assert len(chapter.source_ids) >= 2 @@ -90,6 +107,8 @@ def test_every_chapter_is_published_substantive_and_directly_sourced() -> None: assert "Planned study work" not in markdown assert "placeholder" not in markdown.lower() + assert total_words >= 35_000 + def test_workspace_settings_retains_representative_technical_depth() -> None: catalog = BookCatalog(CONTENT_ROOT)