afv-library/skills/explaining-batch-data-transform/assets/sample_bdts/append_and_split.json
Gaurav Bajpai 9c47f28221
feat(bdt): reference docs + sample BDTs @W-22196528@
Adds the curated reference library the skill loads on demand, plus
four synthetic sample BDTs used by docs, tests, and LLM-mode demos.

references/ (4 curated Markdown files):
- bdt-reference.md        — top-level BDT JSON anatomy: envelope,
                            nodes, edges, UI layer, definitions,
                            businessType semantics. Cites the core-262
                            upstream JSON schema and Connect API spec.
- bdt-node-catalog.md     — every node type (DMO Source, DMO Sink,
                            Filter, Join, Union, Aggregate, Window,
                            Formula, Split, Append, etc.) with its
                            required/optional fields and typical
                            usage. Audited against core-262 enums.
- bdt-function-catalog.md — the expression-language function surface
                            (string, numeric, date, conditional,
                            aggregate). Grouped by category with
                            signature + one-line semantics.
- bdt-window-functions.md — windowing operators (ROW_NUMBER, RANK,
                            LEAD/LAG, running aggregates) with PARTITION
                            BY / ORDER BY grammar and gotchas.

assets/sample_bdts/ (4 synthetic, dependency-free BDTs):
- minimal_dmo_to_dmo.json   — smallest valid BDT (1 source, 1 sink).
- joins_and_filters.json    — join + filter composition.
- window_and_aggregate.json — window function + aggregate in one graph.
- append_and_split.json     — append-then-split branching topology.

Grounding rules enforced in this commit:
- Every claim in references/ cites an upstream source (core-262 JSON
  schema, Connect API reference, or the Data Cloud BDT editor spec).
  No speculative content.
- No raw DITA or internal-only documentation is shipped; references
  are synthesized from public-facing material.
- BusinessTypeEnum values use the canonical camelCase casing from
  core-262 (case-cleanup fix included here).
- Sample BDTs are original synthetic fixtures, not redacted customer
  data. Each is small enough to read end-to-end.

@W-22196528@
2026-04-23 23:29:05 +05:30

91 lines
3.5 KiB
JSON

{
"version": "66.0",
"nodes": {
"LOAD_WEB_ORDERS": {
"action": "load",
"sources": [],
"parameters": {
"dataset": {"name": "WebOrders__dlo", "type": "dataLakeObject"},
"fields": ["OrderId__c", "Amount__c", "Channel__c"],
"sampleDetails": {"type": "TopN", "sortBy": []}
}
},
"LOAD_STORE_ORDERS": {
"action": "load",
"sources": [],
"parameters": {
"dataset": {"name": "StoreOrders__dlo", "type": "dataLakeObject"},
"fields": ["OrderId__c", "Amount__c", "Channel__c"],
"sampleDetails": {"type": "TopN", "sortBy": []}
}
},
"APPEND_ALL_ORDERS": {
"action": "appendV2",
"sources": ["LOAD_WEB_ORDERS", "LOAD_STORE_ORDERS"],
"parameters": {
"allowImplicitDisjointSchema": false,
"fieldMappings": [
{"targetField": "OrderId__c", "sources": [{"node": "LOAD_WEB_ORDERS", "field": "OrderId__c"}, {"node": "LOAD_STORE_ORDERS", "field": "OrderId__c"}]},
{"targetField": "Amount__c", "sources": [{"node": "LOAD_WEB_ORDERS", "field": "Amount__c"}, {"node": "LOAD_STORE_ORDERS", "field": "Amount__c"}]},
{"targetField": "Channel__c", "sources": [{"node": "LOAD_WEB_ORDERS", "field": "Channel__c"}, {"node": "LOAD_STORE_ORDERS", "field": "Channel__c"}]}
]
}
},
"SPLIT_BY_AMOUNT": {
"action": "split",
"sources": ["APPEND_ALL_ORDERS"],
"parameters": {
"branches": [
{"name": "high_value", "expression": "Amount__c >= 1000"},
{"name": "low_value", "expression": "Amount__c < 1000"}
]
}
},
"OUTPUT_HIGH": {
"action": "outputD360",
"sources": ["SPLIT_BY_AMOUNT"],
"parameters": {
"name": "HighValueOrders__dlm",
"type": "dataModelObject",
"writeMode": "OVERWRITE",
"fieldsMappings": [
{"sourceField": "OrderId__c", "targetField": "OrderId__c"},
{"sourceField": "Amount__c", "targetField": "Amount__c"},
{"sourceField": "Channel__c", "targetField": "Channel__c"}
]
}
},
"OUTPUT_LOW": {
"action": "outputD360",
"sources": ["SPLIT_BY_AMOUNT"],
"parameters": {
"name": "LowValueOrders__dlm",
"type": "dataModelObject",
"writeMode": "OVERWRITE",
"fieldsMappings": [
{"sourceField": "OrderId__c", "targetField": "OrderId__c"},
{"sourceField": "Amount__c", "targetField": "Amount__c"},
{"sourceField": "Channel__c", "targetField": "Channel__c"}
]
}
}
},
"ui": {
"nodes": {
"LOAD_WEB_ORDERS": {"label": "Web Orders", "type": "LOAD_DATASET", "top": 100, "left": 100},
"LOAD_STORE_ORDERS": {"label": "Store Orders", "type": "LOAD_DATASET", "top": 260, "left": 100},
"APPEND_ALL_ORDERS": {"label": "Union", "type": "APPEND", "top": 180, "left": 260},
"SPLIT_BY_AMOUNT": {"label": "Split by $", "type": "SPLIT", "top": 180, "left": 420},
"OUTPUT_HIGH": {"label": "High value", "type": "OUTPUT", "top": 100, "left": 580},
"OUTPUT_LOW": {"label": "Low value", "type": "OUTPUT", "top": 260, "left": 580}
},
"connectors": [
{"source": "LOAD_WEB_ORDERS", "target": "APPEND_ALL_ORDERS"},
{"source": "LOAD_STORE_ORDERS", "target": "APPEND_ALL_ORDERS"},
{"source": "APPEND_ALL_ORDERS", "target": "SPLIT_BY_AMOUNT"},
{"source": "SPLIT_BY_AMOUNT", "target": "OUTPUT_HIGH"},
{"source": "SPLIT_BY_AMOUNT", "target": "OUTPUT_LOW"}
]
}
}