diff --git a/skills/datacloud_code_extension/README.md b/skills/datacloud_code_extension/README.md new file mode 100644 index 0000000..c6216cf --- /dev/null +++ b/skills/datacloud_code_extension/README.md @@ -0,0 +1,409 @@ +# datacloud_code_extension Skill + +## Overview + +A Claude Code skill that provides complete workflow for developing, testing, and deploying custom Python code extensions to Salesforce Data Cloud using the SF CLI plugin. + +## Installation + +The skill is now installed at: +``` +/home/codebuilder/dx-project/.a4drules/skills/datacloud_code_extension/ +``` + +## What It Does + +This skill helps you create Data Cloud Code Extensions through a complete workflow: + +1. **Init** - Create new code extension project with scaffolding +2. **Develop** - Write Python transformation logic +3. **Scan** - Auto-detect permissions and generate config +4. **Run** - Test locally against Data Cloud org +5. **Deploy** - Package and deploy to Data Cloud + +## Usage + +### In Claude Code Conversations + +Simply ask Claude naturally: + +**Initialize a project:** +``` +"Create a new Data Cloud code extension project called employee-transform" +"Initialize a code extension to transform employee data" +``` + +**Test locally:** +``` +"Run the code extension in my-transform directory against afvibe org" +"Test the entrypoint.py file locally" +``` + +**Scan for permissions:** +``` +"Scan the entrypoint.py to generate config" +"Update permissions in config.json" +``` + +**Deploy:** +``` +"Deploy Employee_Upper code extension to afvibe" +"Deploy this transform with package-version 1.0.0" +``` + +### Direct Command Usage + +```bash +# Initialize project +sf data-code-extension init --code-type script + +# Scan for permissions +sf data-code-extension scan ./payload/entrypoint.py + +# Test locally +sf data-code-extension run ./payload/entrypoint.py --target-org + +# Deploy +sf data-code-extension deploy --target-org --name --package-version --description --package-dir +``` + +## Prerequisites + +1. **SF CLI with Plugin** + ```bash + sf plugins install @salesforce/plugin-data-codeextension + ``` + +2. **Python 3.11** + ```bash + python --version # Must be 3.11.x + ``` + +3. **Data Cloud Custom Code SDK** + ```bash + pip install salesforce-data-customcode + ``` + +4. **Docker** (for deploy only) + - Docker Desktop or equivalent + +5. **Authenticated Org** + ```bash + sf org login web --alias + ``` + +## Quick Start + +### Complete End-to-End Example + +```bash +# 1. Create project +mkdir employee-transform && cd employee-transform +sf data-code-extension init . --code-type script + +# 2. Edit payload/entrypoint.py with your transformation + +# 3. Scan for permissions +sf data-code-extension scan ./payload/entrypoint.py + +# 4. Test locally +sf data-code-extension run ./payload/entrypoint.py --target-org afvibe + +# 5. Deploy +sf data-code-extension deploy \ + --target-org afvibe \ + --name Employee_Upper \ + --version 1.0.0 \ + --description "Uppercase employee positions" +``` + +## Command Reference + +### Init +```bash +sf data-code-extension init --code-type +``` +Creates project structure with entrypoint.py, config.json, requirements.txt. + +### Scan +```bash +sf data-code-extension scan [--config ] [--dry-run] [--no-requirements] +``` +Detects read/write permissions and Python dependencies. + +### Run +```bash +sf data-code-extension run --target-org [--config-file ] +``` +Executes transformation locally using real Data Cloud data. + +### Deploy +```bash +sf data-code-extension deploy \ + --target-org \ + --name \ + [--version ] \ + [--description ] \ + [--cpu-size ] \ + [--path ] +``` +Packages and deploys to Data Cloud. + +## Example Transformation + +**Read from DLO, transform, write to DLO:** + +```python +from datacustomcode import Client + +client = Client() + +# Read employee data from DLO +employees = client.read_dlo('Employee__dll') + +# Transform - uppercase position field +employees['position_upper'] = employees['position'].str.upper() + +# Select output columns +output = employees[['id', 'name', 'position_upper']] + +# Write to output DLO +client.write_to_dlo('Employee_Upper__dll', output, 'overwrite') + +print(f"Processed {len(output)} employee records") +``` + +## Project Structure + +After `init`, you'll have: + +``` +my-transform/ +├── payload/ +│ ├── entrypoint.py # Your transformation code +│ ├── config.json # Permissions and configuration +│ └── requirements.txt # Python dependencies +└── README.md +``` + +## Common Operations + +### Read/Write DLOs +```python +# Read +df = client.read_dlo('Employee__dll') + +# Write (modes: 'overwrite', 'append') +client.write_to_dlo('Employee_Upper__dll', df, 'overwrite') +``` + +### Read/Write DMOs +```python +# Read +df = client.read_dmo('EmployeeDMO') + +# Write (modes: 'upsert', 'insert') +client.write_to_dmo('EmployeeDMO', df, 'upsert') +``` + +### Data Transformations +```python +import pandas as pd + +# Filter +active_employees = df[df['status'] == 'Active'] + +# Add computed column +df['full_name'] = df['first_name'] + ' ' + df['last_name'] + +# Aggregate +summary = df.groupby('department').agg({'salary': 'mean'}) + +# Join +merged = employees.merge(departments, on='dept_id') +``` + +## Troubleshooting + +### Plugin Not Found +```bash +sf plugins install @salesforce/plugin-data-codeextension +``` + +### Python SDK Missing +```bash +pip install salesforce-data-customcode +datacustomcode version # Verify +``` + +### Wrong Python Version +```bash +# Use pyenv to manage versions +pyenv install 3.11.0 +pyenv local 3.11.0 +python --version # Verify 3.11.x +``` + +### Docker Not Running +- Start Docker Desktop +- Or: `sudo systemctl start docker` (Linux) + +### Org Not Connected +```bash +sf org login web --alias +sf org list # Verify +``` + +### Config.json Missing +```bash +sf data-code-extension scan ./payload/entrypoint.py +``` + +### DLO Not Found +- Use DLO Schema skill to list DLOs +- Verify DLO name ends with `__dll` +- Check read permissions in config.json + +## CPU Size Selection + +Choose based on data volume: + +| CPU Size | Use Case | Data Volume | +|----------|----------|-------------| +| CPU_L | Small datasets | < 1M records | +| CPU_XL | Medium datasets | 1M-5M records | +| CPU_2XL | Large datasets (default) | 5M-10M records | +| CPU_4XL | Very large datasets | > 10M records | + +## Integration with Other Skills + +### With DLO Schema Skill +``` +1. "Show me all DLOs in afvibe" +2. "Get schema for Employee__dll" +3. "Create a code extension to read Employee__dll and transform it" +``` + +### With Datakit Workflow +``` +1. Create DLO via code extension +2. Map DLO to DMO using datakit workflow +3. Create segments from DMO +``` + +## Example Use Cases + +### 1. Data Enrichment +Read employee data, lookup additional info, write enriched data back. + +### 2. Data Cleansing +Read raw data, standardize formats, remove duplicates, write clean data. + +### 3. Aggregation +Read transaction data, calculate summaries, write aggregated metrics. + +### 4. Multi-Source Join +Read from multiple DLOs, join on keys, write unified view. + +### 5. Data Validation +Read data, check quality rules, write valid records and flag errors. + +## Best Practices + +### Development +1. Always scan after code changes +2. Test locally before deploying +3. Use semantic versioning +4. Add descriptive deployment names + +### Code Quality +1. Add print statements for logging +2. Handle errors with try/except +3. Validate input data types +4. Document transformation logic + +### Performance +1. Choose appropriate CPU size +2. Filter data early in pipeline +3. Select only needed columns +4. Process in batches for large datasets + +### Security +1. Never hardcode credentials +2. Use SF CLI authentication only +3. Validate all input data +4. Limit write permissions in config + +## Files Created + +``` +datacloud_code_extension/ +├── SKILL.md # Complete skill documentation +├── README.md # This file +└── quick-reference.md # Command cheat sheet +``` + +## Resources + +- **SF CLI Plugin**: https://github.com/salesforcecli/plugin-data-code-extension +- **Python SDK**: https://github.com/forcedotcom/datacloud-customcode-python-sdk +- **Data Cloud Docs**: https://help.salesforce.com/s/articleView?id=sf.c360_a_intro.htm +- **SDK on PyPI**: https://pypi.org/project/salesforce-data-customcode/ + +## Command Flow + +``` +┌─────────────────────────────────────────────────────┐ +│ 1. INIT │ +│ sf data-code-extension init my-project │ +│ Creates: entrypoint.py, config.json, requirements │ +└─────────────────────────────────────────────────────┘ + ↓ +┌─────────────────────────────────────────────────────┐ +│ 2. DEVELOP │ +│ Edit payload/entrypoint.py │ +│ Write transformation logic │ +└─────────────────────────────────────────────────────┘ + ↓ +┌─────────────────────────────────────────────────────┐ +│ 3. SCAN │ +│ sf data-code-extension scan --entrypoint ./payload/entrypoint.py│ +│ Updates: config.json, requirements.txt │ +└─────────────────────────────────────────────────────┘ + ↓ +┌─────────────────────────────────────────────────────┐ +│ 4. RUN (Local Test) │ +│ sf data-code-extension run --entrypoint │ +│ ./payload/entrypoint.py --target-org afvibe │ +└─────────────────────────────────────────────────────┘ + ↓ +┌─────────────────────────────────────────────────────┐ +│ 5. DEPLOY │ +│ sf data-code-extension deploy --target-org afvibe │ +│ --name Employee_Upper --package-version 1.0.0 │ +| --description "Upper case Employee position column"│ +│ --package-dir ./payload │ +└─────────────────────────────────────────────────────┘ +``` + +## Version History + +- **v1.0** (2026-03-26) - Initial release + - Complete init/scan/run/deploy workflow + - Comprehensive error handling + - Integration with DLO Schema skill + - Full documentation + +## Support + +For issues or questions: +- SF CLI Plugin: https://github.com/salesforcecli/plugin-data-code-extension/issues +- Python SDK: https://github.com/forcedotcom/datacloud-customcode-python-sdk/issues + +## The Skill is Ready! 🚀 + +Try it now: +``` +"Create a code extension to uppercase employee positions" +"Deploy my transform to Data Cloud" +``` diff --git a/skills/datacloud_code_extension/SKILL.md b/skills/datacloud_code_extension/SKILL.md new file mode 100644 index 0000000..eaeb75e --- /dev/null +++ b/skills/datacloud_code_extension/SKILL.md @@ -0,0 +1,758 @@ +--- +name: datacloud_code_extension +description: Develop and deploy Data Cloud Code Extensions using SF CLI plugin. Use this skill when creating custom Python transformations for Data Cloud, deploying code extensions, or testing data transformations. Supports init, run, scan, and deploy operations. +--- + +# datacloud_code_extension Skill + +## Overview + +This skill provides a complete workflow for developing, testing, and deploying custom Python code extensions to Salesforce Data Cloud. Code extensions allow you to write Python transformations that read from and write to Data Lake Objects (DLOs) and Data Model Objects (DMOs). + +## When to Use + +- User wants to create a new code extension project +- User needs to test a code extension locally +- User wants to scan code for required permissions +- User needs to deploy a code extension to Data Cloud +- User is working with Data Cloud transformations +- User wants to read/write DLO or DMO data programmatically + +## Prerequisites Check + +Before executing any code extension commands, verify prerequisites: + +1. **SF CLI with plugin installed** + ```bash + sf plugins --core | grep data-code-extension + ``` + If not installed: + ```bash + sf plugins install @salesforce/plugin-data-codeextension + ``` + +2. **Python 3.11** + ```bash + python --version # Should show 3.11.x + ``` + +3. **Data Cloud Custom Code SDK** + ```bash + pip list | grep salesforce-data-customcode + ``` + If not installed: + ```bash + pip install salesforce-data-customcode + ``` + +4. **Docker running** (for deploy only) + ```bash + docker ps + ``` + +5. **Authenticated org** + ```bash + sf org display --target-org --json + ``` + +## Skill Workflow + +### Phase 1: Initialize Project + +Create a new code extension project with scaffolding. + +**Commands:** + +For **script-based** code extensions (batch transformations): +```bash +sf data-code-extension script init --package-dir +``` + +For **function-based** code extensions (real-time): +```bash +sf data-code-extension function init --package-dir +``` + +**Required Option:** +- `--package-dir, -p` - Directory path where the package will be created + +**Examples:** +```bash +# Create script project in new directory +sf data-code-extension script init --package-dir ./my-transform + +# Create function project in current directory +sf data-code-extension function init --package-dir . +``` + +**What it creates:** +``` +my-transform/ # ← Project root +├── payload/ # ← CRITICAL: This is what --package-dir must point to for deploy +│ ├── entrypoint.py # Main transformation code +│ ├── requirements.txt # Python dependencies +│ └── config.json # Code extension configuration +└── README.md +``` + +## Directory Context During Workflow + +**IMPORTANT:** Understanding the directory structure is critical for successful deployment. + +After running `init`, your structure looks like: + +``` +my-transform/ # ← Project root (run commands from here) +├── payload/ # ← THIS directory contains deployable code +│ ├── entrypoint.py +│ ├── config.json +│ └── requirements.txt +└── README.md +``` + +**Commands and their directory requirements:** + +| Command | Run From | Path/File Argument | +|---------|----------|-------------------| +| `init` | Parent directory | `` or `.` | +| `scan` | Project root | `./payload/entrypoint.py` | +| `run` | Project root | `./payload/entrypoint.py` | +| `deploy` | Project root | `--package-dir ./payload` (**REQUIRED**) | + +**CRITICAL: The `--package-dir` argument in deploy command MUST point to the `payload` directory, not the project root.** + +### Phase 2: Develop Transformation + +Edit `payload/entrypoint.py` with transformation logic. + +**Script Example (Batch):** +```python +from datacustomcode import Client + +client = Client() + +# Read from DLO +df = client.read_dlo('Employee__dll') + +# Transform data (uppercase position field) +df['position_upper'] = df['position'].str.upper() + +# Write to output DLO +client.write_to_dlo('Employee_Upper__dll', df, 'overwrite') +``` + +**Function Example (Real-time):** +```python +from datacustomcode import FunctionClient + +def transform(event, context): + client = FunctionClient(context) + + # Process incoming record + input_data = event['data'] + + # Transform + output = { + 'name': input_data['name'].upper(), + 'status': 'processed' + } + + return output +``` + +**Common Operations:** +- `client.read_dlo('DLO_Name__dll')` - Read from DLO +- `client.read_dmo('DMO_Name')` - Read from DMO +- `client.write_to_dlo('DLO_Name__dll', df, 'overwrite')` - Write to DLO +- `client.write_to_dmo('DMO_Name', df, 'upsert')` - Write to DMO + +### Phase 3: Scan for Permissions + +Scan the entrypoint file to detect required permissions and generate config.json. + +**Command:** +```bash +sf data-code-extension script scan --entrypoint +``` + +**Example:** +```bash +# Scan and update config.json +sf data-code-extension script scan --entrypoint ./payload/entrypoint.py +``` + +**What it detects:** +- Read permissions for DLOs/DMOs +- Write permissions for DLOs/DMOs +- Python package dependencies +- Updates `config.json` and `requirements.txt` + +**Example config.json:** +```json +{ + "version": "1.0", + "permissions": { + "read": ["Employee__dll"], + "write": ["Employee_Upper__dll"] + }, + "resources": { + "cpu_size": "CPU_2XL" + } +} +``` + +### Phase 4: Validate DLO Schema (Pre-Test Check) + +**CRITICAL: Before running tests locally, validate that all DLOs used in your code exist and have the expected fields.** + +This prevents runtime errors and ensures your transformation will work with the actual Data Cloud schema. + +#### Step 4a: Extract DLOs from config.json + +After scanning, review the generated `config.json` to identify all DLOs: + +```bash +cat payload/config.json +``` + +Look for DLOs in the `permissions` section: +```json +{ + "permissions": { + "read": ["Employee__dll", "Department__dll"], + "write": ["Employee_Upper__dll"] + } +} +``` + +#### Step 4b: Validate Each DLO Schema + +**Use the `datacloud_schema` skill to verify DLOs exist and check field names.** + +For each DLO referenced in your code: + +1. **Verify DLO exists:** + ``` + Ask Claude: "Use datacloud_schema skill to check if Employee__dll exists in afvibe" + ``` + + Or manually: + ```bash + python3 ~/.a4drules/skills/datacloud_schema/scripts/get_dlo_schema.py afvibe Employee__dll + ``` + +2. **Verify field names match:** + + Compare fields used in your `entrypoint.py` against the DLO schema: + + **In your code:** + ```python + df['position_upper'] = df['position'].str.upper() + ``` + + **Verify in schema:** + - Check that `position` field exists in Employee__dll + - Check data type is Text + - Verify you have read permissions + +3. **Check all DLOs:** + - Validate all DLOs in `read` permissions + - Validate all DLOs in `write` permissions + - Check field names match exactly (case-sensitive) + - Verify data types are compatible with operations + +#### Step 4c: Validation Checklist + +Before proceeding to run, ensure: + +- [ ] All DLOs in config.json exist in target org +- [ ] All field names used in code exist in DLO schemas +- [ ] Field data types match your transformation logic +- [ ] Primary key fields are correctly identified +- [ ] Write target DLOs are created and accessible + +**Common Issues to Check:** + +| Issue | Check | Fix | +|-------|-------|-----| +| DLO doesn't exist | Use datacloud_schema skill | Create DLO first or update code | +| Field name typo | Compare code vs. schema | Fix field name in entrypoint.py | +| Wrong data type | Check schema data type | Update transformation logic | +| Missing permissions | Check config.json | Re-run scan | + +**Example Validation Workflow:** + +```bash +# 1. Check what DLOs are used +cat payload/config.json + +# 2. Validate source DLO exists and get schema +python3 ~/.a4drules/skills/datacloud_schema/scripts/get_dlo_schema.py afvibe Employee__dll + +# 3. Verify field 'position' exists in schema output +# Look for: name: position__c (or position) + +# 4. Check target DLO exists +python3 ~/.a4drules/skills/datacloud_schema/scripts/get_dlo_schema.py afvibe Employee_Upper__dll + +# 5. If all checks pass, proceed to run +``` + +### Phase 5: Test Locally + +After validating DLO schemas, run the code extension locally against your Data Cloud org. + +**Command:** +```bash +sf data-code-extension script run --entrypoint --target-org [options] +``` + +**Options:** +- `--target-org, -o` - SF CLI org alias (required) +- `--config-file, -c` - Custom config file path + +**Example:** +```bash +# Run with default config (after schema validation) +sf data-code-extension script run --entrypoint ./payload/entrypoint.py --target-org afvibe + +# Run with custom config +sf data-code-extension script run --entrypoint ./payload/entrypoint.py -o afvibe -c custom-config.json +``` + +**What it does:** +- Executes transformation locally +- Reads/writes data from/to actual Data Cloud org +- Shows execution logs and errors +- Tests logic before deployment + +**Monitor output for:** +- Data read/write operations +- Transformation results +- Errors or warnings +- Execution time + +**If you get errors:** +- Re-validate DLO schemas +- Check field names are exact matches +- Verify data types are compatible +- Review error messages for field/DLO issues + +### Phase 6: Deploy to Data Cloud + +Deploy the code extension to Data Cloud for scheduled or on-demand execution. + +**CRITICAL: You MUST specify `--package-dir ./payload` to point to the payload directory created by init.** + +**Command:** +```bash +sf data-code-extension script deploy --target-org --name --package-dir ./payload --package-version --description [options] +``` + +**Required Options:** +- `--target-org, -o` - SF CLI org alias +- `--name, -n` - Name for code extension deployment +- `--package-dir` - Path to payload directory (**REQUIRED** - must be `./payload` when running from project root) +- `--package-version` - Version string (default: 0.0.1) +- `--description` - Description of code extension + +**Optional Options:** +- `--cpu-size` - CPU size: CPU_L, CPU_XL, CPU_2XL (default), CPU_4XL +- `--function-invoke-opt` - Function invoke options (for function type) +- `--network` - Docker network (default: default) + +**Example from project root:** +```bash +# Basic deployment (MUST include --package-dir ./payload) +sf data-code-extension script deploy \ + --target-org afvibe \ + --name Employee_Upper \ + --package-version 1.0.0 \ + --description "Uppercase employee positions" \ + --package-dir ./payload + +# Full deployment with all options +sf data-code-extension script deploy \ + --target-org afvibe \ + --name Employee_Upper \ + --package-version 1.0.0 \ + --description "Uppercase employee positions" \ + --cpu-size CPU_4XL \ + --package-dir ./payload +``` + +**Example with full path (if not in project root):** +```bash +sf data-code-extension script deploy \ + --target-org afvibe \ + --name Employee_Upper \ + --package-version 1.0.0 \ + --description "Uppercase employee positions" \ + --package-dir /full/path/to/my-transform/payload +``` + +**What it does:** +- Packages code with dependencies using Docker +- Uploads to Data Cloud +- Creates deployment record +- Makes code available for execution in UI + +**After deployment:** +- Navigate to Data Cloud in Salesforce UI +- Go to Data Transforms section +- Find your deployment by name +- Click "Run Now" to execute +- Schedule for recurring execution + +## Complete Workflow Example + +Here's a complete end-to-end example for creating an Employee uppercase transformation: + +```bash +# 1. Create project directory (or use existing directory) +mkdir employee-transform && cd employee-transform + +# 2. Initialize script project +sf data-code-extension script init --package-dir . + +# 3. Edit payload/entrypoint.py (see code below) + +# 4. Scan for permissions +sf data-code-extension script scan --entrypoint ./payload/entrypoint.py + +# 5. Validate DLO schemas (CRITICAL: check before testing) +# Check config.json for DLOs used +cat payload/config.json +# Validate Employee__dll exists and has 'position' field +python3 ~/.a4drules/skills/datacloud_schema/scripts/get_dlo_schema.py afvibe Employee__dll +# Validate Employee_Upper__dll exists (target DLO) +python3 ~/.a4drules/skills/datacloud_schema/scripts/get_dlo_schema.py afvibe Employee_Upper__dll + +# 6. Test locally (after DLO validation passes) +sf data-code-extension script run --entrypoint ./payload/entrypoint.py --target-org afvibe + +# 7. Deploy to Data Cloud (CRITICAL: include --package-dir ./payload) +sf data-code-extension script deploy \ + --target-org afvibe \ + --name Employee_Upper \ + --package-version 1.0.0 \ + --description "Uppercase employee positions" \ + --package-dir ./payload +``` + +**entrypoint.py code:** +```python +from datacustomcode import Client + +client = Client() + +# Read employee data +employees = client.read_dlo('Employee__dll') + +# Transform - uppercase position field +employees['position_upper'] = employees['position'].str.upper() + +# Select relevant columns +output = employees[['id', 'name', 'position_upper']] + +# Write to output DLO +client.write_to_dlo('Employee_Upper__dll', output, 'overwrite') + +print(f"Processed {len(output)} employee records") +``` + +## Error Handling + +### Common Issues and Solutions + +**1. SF CLI Plugin Not Found** +``` +Error: command data-code-extension not found +``` +Solution: +```bash +sf plugins install @salesforce/plugin-data-codeextension +``` + +**2. Python SDK Not Installed** +``` +Error: datacustomcode CLI not found +``` +Solution: +```bash +pip install salesforce-data-customcode +datacustomcode version # Verify +``` + +**3. Wrong Python Version** +``` +Error: Python version mismatch +``` +Solution: +```bash +python --version # Should show 3.11.x +# Use pyenv to manage Python versions +pyenv install 3.11.0 +pyenv local 3.11.0 +``` + +**4. Docker Not Running** +``` +Error: Cannot connect to Docker daemon +``` +Solution: +- Start Docker Desktop +- Or start Docker service: `sudo systemctl start docker` + +**5. Org Not Authenticated** +``` +Error: No org found for alias 'afvibe' +``` +Solution: +```bash +sf org login web --alias afvibe +sf org list # Verify +``` + +**6. Config.json Missing** +``` +Error: config.json not found +``` +Solution: +```bash +sf data-code-extension scan ./payload/entrypoint.py +``` + +**7. DLO Not Found During Run** +``` +Error: DLO 'Employee__dll' not found +``` +Solution: +- Verify DLO exists in org (use DLO Schema skill) +- Check spelling and suffix (__dll) +- Ensure proper read permissions in config.json + +**8. Permission Denied During Write** +``` +Error: Permission denied writing to 'Employee_Upper__dll' +``` +Solution: +- Run scan to update permissions: `sf data-code-extension script scan ./payload/entrypoint.py` +- Verify target DLO exists and is writable +- Check Data Cloud permissions in org + +**9. Deploy Fails - Wrong Directory Path** +``` +Error: Cannot find entrypoint.py or config.json +Error: No such file or directory: './entrypoint.py' +Error: Deploy failed - invalid path +``` +Solution: +**CRITICAL:** Ensure `--package-dir` argument points to the `payload` directory, not the project root. + +```bash +# WRONG - Missing --package-dir argument +sf data-code-extension script deploy -o afvibe -n MyTransform + +# WRONG - Pointing to project root instead of payload +sf data-code-extension script deploy -o afvibe -n MyTransform --package-dir . + +# CORRECT - From project root, point to payload directory, includes package-version +sf data-code-extension script deploy -o afvibe -n MyTransform --package-dir ./payload --package-version 1.0.0 --description "Uppercase employee positions" + +# CORRECT - With full path +sf data-code-extension script deploy -o afvibe -n MyTransform --package-version 1.0.0 --package-dir /full/path/to/project/payload --description "Uppercase employee positions" +``` + +Check your current directory: +```bash +pwd # Should be in project root +ls # Should see 'payload' directory +ls payload/ # Should see entrypoint.py, config.json, requirements.txt +``` + +## Best Practices + +### Development +1. **Always scan before testing**: Run scan after code changes +2. **Test locally first**: Use `run` command before deploying +3. **Use version control**: Git commit after each successful test +4. **Version your deployments**: Use semantic versioning (1.0.0, 1.1.0, etc.) +5. **Deploy from project root with --package-dir ./payload**: ALWAYS specify the payload directory explicitly in deploy commands + +### Code Organization +1. **Keep entrypoint.py focused**: Main transformation logic only +2. **Extract helpers**: Create separate modules for reusable functions +3. **Add logging**: Use print statements for debugging +4. **Handle errors**: Add try/except blocks for data operations + +### Performance +1. **Choose appropriate CPU size**: + - CPU_L: Small datasets (< 1M records) + - CPU_2XL: Medium datasets (1M-10M records) + - CPU_4XL: Large datasets (> 10M records) +2. **Filter early**: Read only needed columns/rows +3. **Batch operations**: Process data in chunks for large datasets + +### Security +1. **No hardcoded credentials**: Use SF CLI authentication only +2. **Validate input data**: Check for nulls and data types +3. **Limit write permissions**: Only grant necessary DLO/DMO access + +## Integration with Other Skills + +**Use with DLO Schema Skill (CRITICAL for validation):** + +The `datacloud_schema` skill is **required** for validating DLOs before testing code extensions. + +**Workflow Integration:** +``` +1. List all DLOs: "Show me all DLOs in afvibe" + → Verify target DLOs exist in org + +2. Get DLO schema: "What's the schema for Employee__dll?" + → Validate field names used in code + +3. Check field types: Review schema output for data types + → Ensure transformation logic is compatible + +4. Validate before test: Use schema validation BEFORE running locally + → Prevents runtime errors from missing fields/DLOs + +5. Create code extension: "Create a code extension to read Employee__dll" + → Design transformation based on actual schema +``` + +**Example Integrated Workflow:** +```bash +# Step 1: Check what DLOs exist +python3 ~/.a4drules/skills/datacloud_schema/scripts/get_dlo_schema.py afvibe + +# Step 2: Get schema for source DLO +python3 ~/.a4drules/skills/datacloud_schema/scripts/get_dlo_schema.py afvibe Employee__dll + +# Step 3: Design code extension based on schema +# (Write entrypoint.py using fields from schema) + +# Step 4: Scan for permissions +sf data-code-extension script scan --entrypoint ./payload/entrypoint.py + +# Step 5: Validate all DLOs referenced in config.json +cat payload/config.json +python3 ~/.a4drules/skills/datacloud_schema/scripts/get_dlo_schema.py afvibe Employee__dll +python3 ~/.a4drules/skills/datacloud_schema/scripts/get_dlo_schema.py afvibe Employee_Upper__dll + +# Step 6: Test locally (after validation passes) +sf data-code-extension script run --entrypoint ./payload/entrypoint.py --target-org afvibe + +# Step 7: Deploy +sf data-code-extension script deploy --target-org afvibe --name Employee_Upper --package-dir ./payload --package-version 1.0.0 --description "Uppercase employee positions" +``` + +**Use with Datakit Workflow:** +``` +1. Create DLO via code extension +2. Map DLO to DMO using datakit workflow +3. Use DMO in segments and activations +``` + +## Output Interpretation + +### Scan Output +``` +Scanning ./payload/entrypoint.py... +Found permissions: + Read: Employee__dll + Write: Employee_Upper__dll +Found dependencies: + pandas==2.0.0 + numpy==1.24.0 +Updated: payload/config.json +Updated: payload/requirements.txt +``` + +### Run Output +``` +Reading from Employee__dll... +Records read: 12 +Transforming data... +Writing to Employee_Upper__dll... +Records written: 12 +Execution completed in 2.3s +``` + +### Deploy Output +``` +Building Docker image... +Packaging dependencies... +Uploading to Data Cloud... +Deployment 'Employee_Upper' created successfully +ID: 2dgXXXXXXXXXXXXXXX +Version: 1.0.0 +Status: ACTIVE +``` + +## Advanced Usage + +### Custom Dependencies +Add to `requirements.txt`: +```txt +pandas==2.0.0 +numpy==1.24.0 +scikit-learn==1.3.0 +``` + +### Multiple DLO Operations +```python +# Read from multiple DLOs +employees = client.read_dlo('Employee__dll') +departments = client.read_dlo('Department__dll') + +# Join data +merged = employees.merge(departments, on='dept_id') + +# Write to multiple outputs +client.write_to_dlo('Employee_Enriched__dll', merged, 'overwrite') +client.write_to_dmo('EmployeeDMO', merged, 'upsert') +``` + +### Conditional Logic +```python +# Read data +df = client.read_dlo('Employee__dll') + +# Apply transformations based on conditions +df['grade'] = df['position'].apply(lambda x: + 'Senior' if 'Director' in x or 'VP' in x else 'Junior' +) + +# Filter and write +senior_employees = df[df['grade'] == 'Senior'] +client.write_to_dlo('Senior_Employees__dll', senior_employees, 'overwrite') +``` + +## Command Reference + +| Command | Purpose | Required Args | +|---------|---------|---------------| +| `script init` | Create new script project | --package-dir | +| `function init` | Create new function project | --package-dir | +| `script scan` | Generate config | entrypoint file | +| `script run` | Test locally | entrypoint file, --target-org | +| `script deploy` | Deploy to Data Cloud | --target-org, --name, --package-dir, --package-version, --description | + +## Resources + +- SF CLI Plugin: https://github.com/salesforcecli/plugin-data-code-extension +- Python SDK: https://github.com/forcedotcom/datacloud-customcode-python-sdk +- Data Cloud Docs: https://help.salesforce.com/s/articleView?id=sf.c360_a_intro.htm +- Python SDK PyPI: https://pypi.org/project/salesforce-data-customcode/ + +## Notes + +- Code extensions run in isolated Python 3.11 environment +- Docker is required only for deployment, not for local testing +- Use SF CLI authentication only (no separate credential files) +- Scan command auto-detects permissions from code +- Local run uses actual Data Cloud data (not mocked) +- Deployments are versioned and can be rolled back in UI diff --git a/skills/datacloud_code_extension/quick-reference.md b/skills/datacloud_code_extension/quick-reference.md new file mode 100644 index 0000000..c41a222 --- /dev/null +++ b/skills/datacloud_code_extension/quick-reference.md @@ -0,0 +1,325 @@ +# Data Cloud Code Extension - Quick Reference + +## Command Cheat Sheet + +### Initialize Project +```bash +# Create script project +sf data-code-extension init --code-type script + +# Create function project +sf data-code-extension init --code-type function + +# Examples +sf data-code-extension init . --code-type script +sf data-code-extension init my-transform --code-type script +``` + +### Scan for Permissions +```bash +# Basic scan +sf data-code-extension scan ./payload/entrypoint.py + +# Preview without saving +sf data-code-extension scan ./payload/entrypoint.py --dry-run + +# Custom config location +sf data-code-extension scan ./payload/entrypoint.py --config ./custom-config.json + +# Skip requirements.txt +sf data-code-extension scan ./payload/entrypoint.py --no-requirements +``` + +### Run Locally +```bash +# Basic run +sf data-code-extension run ./payload/entrypoint.py --target-org + +# With custom config +sf data-code-extension run ./payload/entrypoint.py -o -c custom-config.json + +# Examples +sf data-code-extension run ./payload/entrypoint.py --target-org afvibe +sf data-code-extension run ./payload/entrypoint.py -o afvibe +``` + +### Deploy +```bash +# Minimal deployment (MUST include --path ./payload) +sf data-code-extension deploy \ + --target-org \ + --name \ + --package-version \ + --description "" \ + --path ./payload + +# Full options +sf data-code-extension deploy \ + --target-org \ + --name \ + --package-version \ + --description "" \ + --cpu-size \ + --path ./payload + +# Examples (CRITICAL: Always include --path ./payload) +sf data-code-extension deploy \ + --target-org afvibe \ + --name Employee_Upper \ + --package-version 1.0.0 \ + --description "Uppercase employee positions" \ + --path ./payload + +sf data-code-extension deploy \ + -o afvibe \ + -n Employee_Upper \ + --package-version 1.0.0 \ + --description "Uppercase employee positions" \ + --cpu-size CPU_4XL \ + --path ./payload +``` + +## Common Workflows + +### New Project from Scratch +```bash +# 1. Create directory +mkdir my-transform && cd my-transform + +# 2. Initialize +sf data-code-extension init . --code-type script + +# 3. Edit entrypoint.py +# (Add your transformation code) + +# 4. Scan +sf data-code-extension scan ./payload/entrypoint.py + +# 5. Test +sf data-code-extension run ./payload/entrypoint.py --target-org afvibe + +# 6. Deploy (MUST include --path ./payload) +sf data-code-extension deploy \ + --target-org afvibe \ + --name MyTransform \ + --package-version 1.0.0 \ + --description "Uppercase employee positions" \ + --path ./payload +``` + +### Update Existing Code Extension +```bash +# 1. Edit entrypoint.py + +# 2. Re-scan +sf data-code-extension scan ./payload/entrypoint.py + +# 3. Test +sf data-code-extension run ./payload/entrypoint.py -o afvibe + +# 4. Deploy with new version (include --path ./payload) +sf data-code-extension deploy \ + -o afvibe \ + -n MyTransform \ + --package-version 1.1.0 \ + --description "Uppercase employee positions" \ + --path ./payload +``` + +## Python Code Patterns + +### Read/Write DLO +```python +from datacustomcode import Client + +client = Client() + +# Read +df = client.read_dlo('Employee__dll') + +# Transform +df['new_field'] = df['old_field'].str.upper() + +# Write (modes: 'overwrite', 'append') +client.write_to_dlo('Output__dll', df, 'overwrite') +``` + +### Read/Write DMO +```python +# Read +df = client.read_dmo('EmployeeDMO') + +# Write (modes: 'upsert', 'insert') +client.write_to_dmo('EmployeeDMO', df, 'upsert') +``` + +### Multiple DLO Operations +```python +# Read multiple +employees = client.read_dlo('Employee__dll') +departments = client.read_dlo('Department__dll') + +# Join +merged = employees.merge(departments, on='dept_id') + +# Write multiple +client.write_to_dlo('Enriched__dll', merged, 'overwrite') +client.write_to_dmo('EmployeeDMO', merged, 'upsert') +``` + +### Data Transformations +```python +import pandas as pd + +# Filter +active = df[df['status'] == 'Active'] + +# Computed column +df['full_name'] = df['first'] + ' ' + df['last'] + +# Aggregate +summary = df.groupby('dept')['salary'].mean() + +# Conditional +df['grade'] = df['position'].apply( + lambda x: 'Senior' if 'VP' in x else 'Junior' +) +``` + +## Option Reference + +### --code-type +- `script` - Batch transformation (default) +- `function` - Real-time function + +### --cpu-size +- `CPU_L` - Small datasets (< 1M records) +- `CPU_XL` - Medium datasets (1M-5M) +- `CPU_2XL` - Large datasets (5M-10M) **[default]** +- `CPU_4XL` - Very large (> 10M records) + +### Write Modes +- `overwrite` - Replace all data +- `append` - Add to existing data +- `upsert` - Update or insert (DMO only) +- `insert` - Insert only (DMO only) + +## Troubleshooting Quick Fixes + +```bash +# Plugin not found +sf plugins install @salesforce/plugin-data-codeextension + +# Python SDK missing +pip install salesforce-data-customcode + +# Verify Python version (must be 3.11.x) +python --version + +# Org not connected +sf org login web --alias + +# Config missing +sf data-code-extension scan ./payload/entrypoint.py + +# Docker not running (for deploy) +# Start Docker Desktop +``` + +## File Structure + +``` +my-project/ +├── payload/ +│ ├── entrypoint.py # Main code +│ ├── config.json # Auto-generated permissions +│ └── requirements.txt # Auto-generated dependencies +└── README.md +``` + +## config.json Format + +```json +{ + "version": "1.0", + "permissions": { + "read": ["Employee__dll", "Department__dll"], + "write": ["Enriched__dll"] + }, + "resources": { + "cpu_size": "CPU_2XL" + } +} +``` + +## Common Errors + +| Error | Quick Fix | +|-------|-----------| +| Plugin not found | `sf plugins install @salesforce/plugin-data-codeextension` | +| Python SDK missing | `pip install salesforce-data-customcode` | +| Wrong Python version | Use pyenv to install 3.11.0 | +| Org not connected | `sf org login web --alias ` | +| Config missing | Run scan command | +| DLO not found | Check DLO name, use DLO Schema skill | +| Docker error | Start Docker Desktop | + +## Deployment Checklist + +- [ ] Code written in entrypoint.py +- [ ] Scanned for permissions +- [ ] Tested locally +- [ ] Version number decided +- [ ] Description added +- [ ] CPU size chosen +- [ ] Docker running +- [ ] Org authenticated + +## Next Steps After Deploy + +1. Go to Data Cloud in Salesforce UI +2. Navigate to Code Extensions +3. Find your deployment +4. Click "Run Now" to test +5. Schedule for recurring execution +6. Monitor execution logs + +## Quick Examples + +### Example 1: Simple Transform +```python +from datacustomcode import Client +client = Client() + +df = client.read_dlo('Employee__dll') +df['upper_pos'] = df['position'].str.upper() +client.write_to_dlo('Employee_Upper__dll', df, 'overwrite') +``` + +### Example 2: Filter and Write +```python +from datacustomcode import Client +client = Client() + +df = client.read_dlo('Employee__dll') +managers = df[df['position'].str.contains('Manager')] +client.write_to_dlo('Managers__dll', managers, 'overwrite') +``` + +### Example 3: Join Two DLOs +```python +from datacustomcode import Client +client = Client() + +employees = client.read_dlo('Employee__dll') +departments = client.read_dlo('Department__dll') + +merged = employees.merge(departments, left_on='dept_id', right_on='id') +client.write_to_dlo('Employee_With_Dept__dll', merged, 'overwrite') +``` + +## Resources + +- Plugin: https://github.com/salesforcecli/plugin-data-code-extension +- Python SDK: https://github.com/forcedotcom/datacloud-customcode-python-sdk +- Data Cloud Docs: https://help.salesforce.com/s/articleView?id=sf.c360_a_intro.htm diff --git a/skills/datacloud_schema/SKILL.md b/skills/datacloud_schema/SKILL.md new file mode 100644 index 0000000..8fbae49 --- /dev/null +++ b/skills/datacloud_schema/SKILL.md @@ -0,0 +1,355 @@ +--- +name: datacloud_schema +description: Retrieve Data Lake Object (DLO) and Data Model Object (DMO) schema information from Salesforce Data Cloud using REST APIs. Use this skill when you need to inspect DLO or DMO field definitions, data types, or metadata. Takes org alias and optional DLO/DMO name as parameters. +--- + +# datacloud_schema Skill + + +## Overview + +This skill retrieves Data Lake Object (DLO) and Data Model Object (DMO) schema information from Salesforce Data Cloud using the SSOT REST API. It can list all DLOs or DMOs in an org, or retrieve detailed schema for a specific DLO or DMO. + +## When to Use + +- User wants to see all DLOs or DMOs in a Data Cloud org +- User needs field schema for a specific DLO or DMO +- User is exploring Data Cloud data structures +- User needs to understand DLO or DMO field types and metadata + +## Prerequisites + +- SF CLI installed and authenticated to target org +- Org has Data Cloud enabled +- User has appropriate Data Cloud permissions + +## Skill Execution + +### Parameters + +1. **org_alias** (required): The SF CLI org alias (e.g., 'afvibe', 'myorg') +2. **dlo_name** (optional): Specific DLO developer name (e.g., 'Employee__dll') +3. **dmo_name** (optional): Specific DMO developer name (e.g., 'Individual__dlm') + +### Step 1: Validate SF CLI Authentication + +Before making API calls, verify the org is connected: + +```bash +sf org display --target-org --json +``` + +If not connected, inform user to run: +```bash +sf org login web --alias +``` + +### Step 2a: Execute DLO Schema Script + +Use the Python script to retrieve DLO information: + +**To list all DLOs:** +```bash +python3 ./scripts/get_dlo_schema.py +``` + +**To get specific DLO schema:** +```bash +python3 ./scripts/get_dlo_schema.py +``` + +### Step 2b: Execute DMO Schema Script + +Use the Python script to retrieve DMO information: + +**To list all DMOs:** +```bash +python3 ./scripts/get_dmo_schema.py +``` + +**To get specific DMO schema:** +```bash +python3 ./scripts/get_dmo_schema.py +``` + +### Step 3: Present Results + +Parse and present the results in a user-friendly format: + +**For DLO List:** +- Show DLO name, label, category, and ID +- Indicate total count +- Highlight DLOs with data (totalRecords > 0) + +**For DLO Schema:** +- Show basic info (name, label, category, status) +- List all fields with: + - Field name + - Data type + - Primary key indicator + - Nullable status +- Highlight custom fields (exclude system fields like DataSource__c, cdp_sys_*) +- Show record count if available + +**For DMO List:** +- Show DMO name, label, category, and ID +- Indicate total count + +**For DMO Schema:** +- Show basic info (name, label, category, description) +- List all fields with: + - Field name + - Data type + - Primary key indicator + - Nullable status +- Show dataspace information if available + +### Step 4: Offer Next Steps + +After displaying results, suggest relevant follow-up actions: +- Query data from the DLO +- Create calculated insights +- Build segments +- Set up data streams +- Create DMO mappings + +## API Endpoints Used + +### List All DLOs +``` +GET /services/data/v64.0/ssot/data-lake-objects +``` + +Response structure: +```json +{ + "dataLakeObjects": [ + { + "name": "Employee__dll", + "label": "Employee", + "category": "Profile", + "id": "1dlXXXXXXXXXXXXXXX", + "status": "ACTIVE", + "totalRecords": 12, + "fields": [...] + } + ], + "totalSize": 5 +} +``` + +### Get DLO Schema +``` +GET /services/data/v64.0/ssot/data-lake-objects/{dlo_name} +``` + +Response structure (same as individual object in list response, but wrapped in paginated format). + +### List All DMOs +``` +GET /services/data/v64.0/ssot/data-model-objects +``` + +Response structure: +```json +{ + "dataModelObjects": [ + { + "name": "Individual__dlm", + "label": "Individual", + "category": "Profile", + "id": "0dmXXXXXXXXXXXXXXX", + "fields": [...] + } + ], + "totalSize": 10 +} +``` + +### Get DMO Schema +``` +GET /services/data/v64.0/ssot/data-model-objects/{dmo_name} +``` + +Response structure (same as individual object in list response, but wrapped in paginated format). + +## Error Handling + +**Common Issues:** + +1. **Org not connected** + - Message: "Org not connected" + - Solution: Ask user to authenticate via SF CLI + +2. **DLO not found** + - Message: "DLO 'XYZ__dll' not found" + - Solution: List all DLOs first to verify name + +5. **DMO not found** + - Message: "DMO 'XYZ__dlm' not found" + - Solution: List all DMOs first to verify name + +3. **Permission issues** + - Message: HTTP 403 errors + - Solution: Verify user has Data Cloud permissions + +4. **API version mismatch** + - Current: v64.0 + - Solution: Script can be updated for newer API versions + +## Example Usage + +**Example 1: List all DLOs** +``` +User: "Show me all DLOs in afvibe org" + +Response: +1. Authenticate to afvibe +2. Run: python3 ./scripts/get_dlo_schema.py afvibe +3. Display formatted list of DLOs +``` + +**Example 2: Get specific DLO schema** +``` +User: "Get the schema for Employee__dll in afvibe" + +Response: +1. Authenticate to afvibe +2. Run: python3 ./scripts/get_dlo_schema.py afvibe Employee__dll +3. Display field schema with types and metadata +``` + +**Example 3: Explore DLOs then get schema** +``` +User: "What DLOs exist in myorg and show me the schema for the Employee one" + +Response: +1. List all DLOs in myorg +2. Identify Employee__dll +3. Get detailed schema for Employee__dll +4. Present both results +``` + +**Example 4: List all DMOs** +``` +User: "Show me all DMOs in afvibe org" + +Response: +1. Authenticate to afvibe +2. Run: python3 ./scripts/get_dmo_schema.py afvibe +3. Display formatted list of DMOs +``` + +**Example 5: Get specific DMO schema** +``` +User: "Get the schema for Individual__dlm in afvibe" + +Response: +1. Authenticate to afvibe +2. Run: python3 ./scripts/get_dmo_schema.py afvibe Individual__dlm +3. Display field schema with types and metadata +``` + +**Example 6: Explore DMOs then get schema** +``` +User: "What DMOs exist in myorg and show me the schema for the Individual one" + +Response: +1. List all DMOs in myorg +2. Identify Individual__dlm +3. Get detailed schema for Individual__dlm +4. Present both results +``` + +## Output Format + +### DLO List Output +``` +Found 5 DLOs in org 'afvibe': + +1. DataCustomCodeLogs__dll + Label: DataCustomCodeLogs + Category: Engagement + Records: 233 + +2. Employee__dll + Label: Employee + Category: Profile + Records: 12 + +[...] +``` + +### DLO Schema Output +``` +DLO: Employee__dll +Label: Employee +Category: Profile +Status: ACTIVE +Records: 12 + +Custom Fields: + • id__c (Text) - Primary Key + • name__c (Text) + • position__c (Text) + • manager_id__c (Number) + +System Fields: + • DataSource__c (Text) + • InternalOrganization__c (Text) + • cdp_sys_SourceVersion__c (Text) + +Next steps: +- Query data: SELECT * FROM Employee__dll LIMIT 10 +- Create segment based on position field +- Set up data stream for real-time updates +``` + +### DMO List Output +``` +Found 10 DMOs in org 'afvibe': + +1. Individual__dlm + Label: Individual + Category: Profile + +2. ContactPointEmail__dlm + Label: Contact Point Email + Category: Profile + +[...] +``` + +### DMO Schema Output +``` +DMO: Individual__dlm +Label: Individual +Category: Profile +Description: Represents an individual person + +Fields: + • Id__c (Text) - Primary Key + • FirstName__c (Text) + • LastName__c (Text) + • BirthDate__c (DateTime) + +Next steps: +- Query data: SELECT * FROM Individual__dlm LIMIT 10 +- View DLO mappings to this DMO +- Create calculated insights +``` + +## Notes + +- DLO names always end with `__dll` suffix +- DMO names always end with `__dlm` suffix +- Field names always end with `__c` suffix +- System fields (DataSource__c, KQ_*, cdp_sys_*) are automatically added +- Primary key fields are required for DLO and DMO queries +- API supports pagination (limit/offset) for large result sets + +## Related Skills + +- **datakit_workflow**: For DMO mapping operations +- **datakit_validation**: For validating datakit configurations +- Use this skill before creating DMO mappings to understand source DLO structure diff --git a/skills/datacloud_schema/docs/README.md b/skills/datacloud_schema/docs/README.md new file mode 100644 index 0000000..6259982 --- /dev/null +++ b/skills/datacloud_schema/docs/README.md @@ -0,0 +1,265 @@ +# datacloud_schema Skill + +## Overview + +A Claude Code skill that retrieves Data Lake Object (DLO) and Data Model Object (DMO) schema information from Salesforce Data Cloud using REST APIs. + +## Installation + +The skill is now installed at: +``` +/home/codebuilder/dx-project/.a4drules/skills/datacloud_schema/ +``` + +## Usage + +### Using the Skill in Claude Code + +Simply ask Claude to get DLO or DMO information: + +**List all DLOs:** +``` +"Show me all DLOs in afvibe org" +"List Data Lake Objects in myorg" +``` + +**Get specific DLO schema:** +``` +"Get the schema for Employee__dll in afvibe" +"What fields does the Employee__dll DLO have in myorg?" +``` + +**List all DMOs:** +``` +"Show me all DMOs in afvibe org" +"List Data Model Objects in myorg" +``` + +**Get specific DMO schema:** +``` +"Get the schema for Individual__dlm in afvibe" +"What fields does the Individual__dlm DMO have in myorg?" +``` + +### Direct Script Usage + +You can also run the scripts directly: + +```bash +# List all DLOs +python3 ./scripts/get_dlo_schema.py + +# Get specific DLO schema +python3 ./scripts/get_dlo_schema.py + +# List all DMOs +python3 ./scripts/get_dmo_schema.py + +# Get specific DMO schema +python3 ./scripts/get_dmo_schema.py +``` + +**Examples:** +```bash +# List all DLOs in afvibe org +python3 ./scripts/get_dlo_schema.py afvibe + +# Get Employee__dll schema from afvibe +python3 ./scripts/get_dlo_schema.py afvibe Employee__dll + +# List all DMOs in afvibe org +python3 ./scripts/get_dmo_schema.py afvibe + +# Get Individual__dlm schema from afvibe +python3 ./scripts/get_dmo_schema.py afvibe Individual__dlm +``` + +## Prerequisites + +1. **SF CLI Installed** + ```bash + sf --version + ``` + +2. **Authenticated to Target Org** + ```bash + sf org login web --alias + ``` + +3. **Python 3 and Dependencies** + ```bash + pip install requests pyyaml + ``` + +4. **Data Cloud Enabled** + - Org must have Data Cloud provisioned + - User must have Data Cloud permissions + +## What It Does + +### List All DLOs +- Calls: `GET /services/data/v64.0/ssot/data-lake-objects` +- Returns: All DLOs with name, label, category, ID, record count +- Shows paginated results + +### Get DLO Schema +- Calls: `GET /services/data/v64.0/ssot/data-lake-objects/{dlo_name}` +- Returns: Detailed field schema including: + - Field names and labels + - Data types (Text, Number, DateTime, etc.) + - Primary key indicators + - Nullable status + - Field metadata + +### List All DMOs +- Calls: `GET /services/data/v64.0/ssot/data-model-objects` +- Returns: All DMOs with name, label, category, ID +- Shows paginated results + +### Get DMO Schema +- Calls: `GET /services/data/v64.0/ssot/data-model-objects/{dmo_name}` +- Returns: Detailed field schema including: + - Field names and labels + - Data types (Text, Number, DateTime, etc.) + - Primary key indicators + - Nullable status + - Field metadata + +## API Endpoints + +| Endpoint | Method | Purpose | +|----------|--------|---------| +| `/services/data/v64.0/ssot/data-lake-objects` | GET | List all DLOs | +| `/services/data/v64.0/ssot/data-lake-objects/{name}` | GET | Get DLO schema | +| `/services/data/v64.0/ssot/data-model-objects` | GET | List all DMOs | +| `/services/data/v64.0/ssot/data-model-objects/{name}` | GET | Get DMO schema | + +## Output Format + +### DLO List +``` +Found 5 DLOs in org 'afvibe': + +1. DataCustomCodeLogs__dll + Label: DataCustomCodeLogs + Category: Engagement + Records: 233 + +2. Employee__dll + Label: Employee + Category: Profile + Records: 12 +``` + +### DLO Schema +``` +DLO: Employee__dll +Label: Employee +Category: Profile +Status: ACTIVE +Records: 12 + +Fields (9 total): + • id__c (Text) - Primary Key + • name__c (Text) + • position__c (Text) + • manager_id__c (Number) + • DataSource__c (Text) + • InternalOrganization__c (Text) + [...] +``` + +### DMO List +``` +Found 10 DMOs in org 'afvibe': + +1. Individual__dlm + Label: Individual + Category: Profile + +2. ContactPointEmail__dlm + Label: Contact Point Email + Category: Profile +``` + +### DMO Schema +``` +DMO: Individual__dlm +Label: Individual +Category: Profile + +Fields (8 total): + • Id__c (Text) - Primary Key + • FirstName__c (Text) + • LastName__c (Text) + • BirthDate__c (DateTime) + [...] +``` + +## Files + +``` +datacloud_schema/ +├── SKILL.md # Skill definition and instructions +├── docs/ +│ └── README.md # This file +└── scripts/ + ├── get_dlo_schema.py # Python script for DLO REST APIs + └── get_dmo_schema.py # Python script for DMO REST APIs +``` + +## Troubleshooting + +**Issue: "Org not connected"** +```bash +sf org login web --alias +sf org list # Verify +``` + +**Issue: "Module not found: requests"** +```bash +pip install requests pyyaml +``` + +**Issue: "DLO not found"** +- Verify DLO name ends with `__dll` suffix +- List all DLOs first to confirm exact name + +**Issue: "DMO not found"** +- Verify DMO name ends with `__dlm` suffix +- List all DMOs first to confirm exact name + +**Issue: "Permission denied"** +- Verify user has Data Cloud permissions +- Check org has Data Cloud enabled + +## Related Skills + +- `aisuite_datakit_workflow` - For DMO mapping operations +- `aisuite_datakit_validation` - For validating datakit configs + +## Next Steps After Getting Schema + +1. **Query DLO Data** + ```sql + SELECT * FROM Employee__dll LIMIT 10 + ``` + +2. **Create Segments** + - Use fields to build audience segments + +3. **Set Up Data Streams** + - Create ingestion pipelines + +4. **Create DMO Mappings** + - Map DLO fields to Data Model Objects + +5. **Build Calculated Insights** + - Aggregate data from DLO fields + +## Version + +- **Version**: 1.0 +- **Created**: 2026-03-26 +- **API Version**: v64.0 +- **Python**: 3.9+ diff --git a/skills/datacloud_schema/scripts/get_dlo_schema.py b/skills/datacloud_schema/scripts/get_dlo_schema.py new file mode 100755 index 0000000..43e58ac --- /dev/null +++ b/skills/datacloud_schema/scripts/get_dlo_schema.py @@ -0,0 +1,244 @@ +#!/usr/bin/env python3 +""" +List all Data Lake Objects and retrieve schema for one DLO using REST API. +Uses SF CLI for authentication. +""" + +import subprocess +import json +import sys +import requests + + +def authenticate_to_org(org_alias): + """ + Authenticate to Salesforce org using SF CLI. + + Args: + org_alias: SF CLI org alias (e.g., 'afvibe') + + Returns: + Tuple of (instance_url, access_token, username) + """ + print(f"🔐 Authenticating to Salesforce org '{org_alias}'...") + + try: + result = subprocess.run( + ['sf', 'org', 'display', '--target-org', org_alias, '--json'], + capture_output=True, + text=True, + check=True + ) + + org_data = json.loads(result.stdout) + + if org_data.get('status') != 0: + raise Exception(f"SF CLI returned error: {org_data}") + + org_info = org_data['result'] + + if org_info.get('connectedStatus') != 'Connected': + raise Exception(f"Org '{org_alias}' is not connected. Run: sf org login web --alias {org_alias}") + + instance_url = org_info['instanceUrl'] + access_token = org_info['accessToken'] + username = org_info.get('username', 'Unknown') + + print(f"✅ Authenticated as: {username}") + print(f"📍 Instance: {instance_url}\n") + + return instance_url, access_token, username + + except subprocess.CalledProcessError as e: + raise Exception(f"SF CLI command failed: {e.stderr}") + except (json.JSONDecodeError, KeyError) as e: + raise Exception(f"Failed to parse SF CLI output: {e}") + + +def list_all_dlos(instance_url, access_token, api_version='v64.0'): + """ + List all Data Lake Objects using SSOT REST API. + + Args: + instance_url: Salesforce instance URL + access_token: OAuth access token + api_version: API version (default: v64.0) + + Returns: + List of DLO dictionaries + """ + url = f"{instance_url}/services/data/{api_version}/ssot/data-lake-objects" + + headers = { + 'Authorization': f'Bearer {access_token}', + 'Content-Type': 'application/json', + 'Accept': 'application/json' + } + + print("📋 Fetching all Data Lake Objects...") + response = requests.get(url, headers=headers) + + if response.status_code != 200: + raise Exception(f"API Error: HTTP {response.status_code}\n{response.text[:500]}") + + response_data = response.json() + + # Extract DLO list from paginated response + if isinstance(response_data, dict) and 'dataLakeObjects' in response_data: + dlos = response_data['dataLakeObjects'] + total_size = response_data.get('totalSize', len(dlos)) + print(f"✅ Found {len(dlos)} DLOs (Total: {total_size})\n") + else: + # Fallback if response format is different + dlos = response_data if isinstance(response_data, list) else [] + print(f"✅ Found {len(dlos)} DLOs\n") + + return dlos + + +def get_dlo_schema(instance_url, access_token, dlo_name, api_version='v64.0'): + """ + Get detailed schema for a specific DLO. + + Args: + instance_url: Salesforce instance URL + access_token: OAuth access token + dlo_name: DLO developer name (e.g., 'Employee__dll') + api_version: API version (default: v64.0) + + Returns: + DLO detail dictionary with full schema + """ + url = f"{instance_url}/services/data/{api_version}/ssot/data-lake-objects/{dlo_name}" + + headers = { + 'Authorization': f'Bearer {access_token}', + 'Content-Type': 'application/json', + 'Accept': 'application/json' + } + + print(f"🔍 Fetching schema for DLO: {dlo_name}...") + response = requests.get(url, headers=headers) + + if response.status_code != 200: + raise Exception(f"API Error: HTTP {response.status_code}\n{response.text[:500]}") + + response_data = response.json() + + # Extract DLO from paginated response + if isinstance(response_data, dict) and 'dataLakeObjects' in response_data: + dlos = response_data['dataLakeObjects'] + if dlos: + return dlos[0] # Return first (should be only) DLO + else: + raise Exception(f"DLO '{dlo_name}' not found") + else: + # Fallback if response format is different + return response_data + + +def display_dlo_list(dlos): + """Display summary of all DLOs.""" + print("=" * 80) + print("📊 DATA LAKE OBJECTS") + print("=" * 80) + + for idx, dlo in enumerate(dlos, 1): + print(f"\n{idx}. {dlo.get('name', 'Unknown')}") + print(f" Label: {dlo.get('label', 'N/A')}") + print(f" Category: {dlo.get('category', 'N/A')}") + if 'id' in dlo: + print(f" ID: {dlo['id']}") + + +def display_dlo_schema(dlo_detail): + """Display detailed schema information for a DLO.""" + print("\n" + "=" * 80) + print(f"🔍 SCHEMA DETAILS FOR: {dlo_detail.get('name')}") + print("=" * 80) + + print(f"\n📝 Basic Information:") + print(f" Name: {dlo_detail.get('name')}") + print(f" Label: {dlo_detail.get('label')}") + print(f" Category: {dlo_detail.get('category')}") + print(f" Description: {dlo_detail.get('description', 'N/A')}") + + if 'dataspaceInfo' in dlo_detail: + dataspaces = dlo_detail['dataspaceInfo'] + dataspace_names = [ds.get('name', 'Unknown') for ds in dataspaces] + print(f" Dataspaces: {', '.join(dataspace_names)}") + + # Display field schema + fields = dlo_detail.get('fields', []) + + if fields: + print(f"\n🔧 Fields ({len(fields)} total):") + print("-" * 80) + + # Show all fields with detailed info + for field in fields: + print(f"\n • {field.get('name')}") + print(f" Label: {field.get('label', 'N/A')}") + print(f" Data Type: {field.get('dataType', 'Unknown')}") + print(f" Primary Key: {field.get('isPrimaryKey', False)}") + print(f" Nullable: {field.get('isNullable', True)}") + + if 'length' in field: + print(f" Length: {field['length']}") + if 'precision' in field: + print(f" Precision: {field['precision']}") + if 'scale' in field: + print(f" Scale: {field['scale']}") + else: + print("\n ⚠️ No fields found in schema") + + # Show full JSON (optional, can be commented out) + print("\n" + "=" * 80) + print("📄 FULL SCHEMA (JSON):") + print("=" * 80) + print(json.dumps(dlo_detail, indent=2)) + + +def main(): + """Main execution function.""" + if len(sys.argv) < 2: + print("Usage: python list_dlos_and_schema.py [dlo_name]") + print("\nExamples:") + print(" python list_dlos_and_schema.py afvibe") + print(" python list_dlos_and_schema.py afvibe Employee__dll") + sys.exit(1) + + org_alias = sys.argv[1] + specific_dlo = sys.argv[2] if len(sys.argv) > 2 else None + + try: + # Step 1: Authenticate + instance_url, access_token, username = authenticate_to_org(org_alias) + + # Step 2: List all DLOs + dlos = list_all_dlos(instance_url, access_token) + display_dlo_list(dlos) + + # Step 3: Get schema for a specific DLO + if specific_dlo: + # User specified a DLO name + dlo_detail = get_dlo_schema(instance_url, access_token, specific_dlo) + display_dlo_schema(dlo_detail) + elif dlos: + # Get schema for the first DLO + first_dlo = dlos[0] + dlo_name = first_dlo.get('name') + dlo_detail = get_dlo_schema(instance_url, access_token, dlo_name) + display_dlo_schema(dlo_detail) + else: + print("\n⚠️ No DLOs found in this org") + + print("\n✅ Done!") + + except Exception as e: + print(f"\n❌ Error: {e}") + sys.exit(1) + + +if __name__ == '__main__': + main() diff --git a/skills/datacloud_schema/scripts/get_dmo_schema.py b/skills/datacloud_schema/scripts/get_dmo_schema.py new file mode 100755 index 0000000..5328f85 --- /dev/null +++ b/skills/datacloud_schema/scripts/get_dmo_schema.py @@ -0,0 +1,233 @@ +#!/usr/bin/env python3 +""" +List all Data Model Objects and retrieve schema for one DMO using REST API. +Uses SF CLI for authentication. +""" + +import subprocess +import json +import sys +import requests + + +def authenticate_to_org(org_alias): + """ + Authenticate to Salesforce org using SF CLI. + + Args: + org_alias: SF CLI org alias (e.g., 'afvibe') + + Returns: + Tuple of (instance_url, access_token, username) + """ + print(f"🔐 Authenticating to Salesforce org '{org_alias}'...") + + try: + result = subprocess.run( + ['sf', 'org', 'display', '--target-org', org_alias, '--json'], + capture_output=True, + text=True, + check=True + ) + + org_data = json.loads(result.stdout) + + if org_data.get('status') != 0: + raise Exception(f"SF CLI returned error: {org_data}") + + org_info = org_data['result'] + + if org_info.get('connectedStatus') != 'Connected': + raise Exception(f"Org '{org_alias}' is not connected. Run: sf org login web --alias {org_alias}") + + instance_url = org_info['instanceUrl'] + access_token = org_info['accessToken'] + username = org_info.get('username', 'Unknown') + + print(f"✅ Authenticated as: {username}") + print(f"📍 Instance: {instance_url}\n") + + return instance_url, access_token, username + + except subprocess.CalledProcessError as e: + raise Exception(f"SF CLI command failed: {e.stderr}") + except (json.JSONDecodeError, KeyError) as e: + raise Exception(f"Failed to parse SF CLI output: {e}") + + +def list_all_dmos(instance_url, access_token, api_version='v64.0'): + """ + List all Data Model Objects using SSOT REST API. + + Args: + instance_url: Salesforce instance URL + access_token: OAuth access token + api_version: API version (default: v64.0) + + Returns: + List of DMO dictionaries + """ + url = f"{instance_url}/services/data/{api_version}/ssot/data-model-objects" + + headers = { + 'Authorization': f'Bearer {access_token}', + 'Content-Type': 'application/json', + 'Accept': 'application/json' + } + + print("📋 Fetching all Data Model Objects...") + response = requests.get(url, headers=headers) + + if response.status_code != 200: + raise Exception(f"API Error: HTTP {response.status_code}\n{response.text[:500]}") + + response_data = response.json() + + # Extract DMO list from paginated response + if isinstance(response_data, dict) and 'dataModelObject' in response_data: + dmos = response_data['dataModelObject'] + total_size = response_data.get('totalSize', len(dmos)) + print(f"✅ Found {len(dmos)} DMOs (Total: {total_size})\n") + else: + # Fallback if response format is different + dmos = response_data if isinstance(response_data, list) else [] + print(f"✅ Found {len(dmos)} DMOs\n") + + return dmos + + +def get_dmo_schema(instance_url, access_token, dmo_name, api_version='v64.0'): + """ + Get detailed schema for a specific DMO. + + Args: + instance_url: Salesforce instance URL + access_token: OAuth access token + dmo_name: DMO developer name (e.g., 'Individual__dlm') + api_version: API version (default: v64.0) + + Returns: + DMO detail dictionary with full schema + """ + url = f"{instance_url}/services/data/{api_version}/ssot/data-model-objects/{dmo_name}" + + headers = { + 'Authorization': f'Bearer {access_token}', + 'Content-Type': 'application/json', + 'Accept': 'application/json' + } + + print(f"🔍 Fetching schema for DMO: {dmo_name}...") + response = requests.get(url, headers=headers) + + if response.status_code != 200: + raise Exception(f"API Error: HTTP {response.status_code}\n{response.text[:500]}") + + response_data = response.json() + + # Single DMO endpoint returns the object directly (not wrapped in an array) + return response_data + + +def display_dmo_list(dmos): + """Display summary of all DMOs.""" + print("=" * 80) + print("📊 DATA MODEL OBJECTS") + print("=" * 80) + + for idx, dmo in enumerate(dmos, 1): + print(f"\n{idx}. {dmo.get('name', 'Unknown')}") + print(f" Label: {dmo.get('label', 'N/A')}") + print(f" Category: {dmo.get('category', 'N/A')}") + print(f" Creation Type: {dmo.get('creationType', 'N/A')}") + print(f" Data Space: {dmo.get('dataSpaceName', 'N/A')}") + + +def display_dmo_schema(dmo_detail): + """Display detailed schema information for a DMO.""" + print("\n" + "=" * 80) + print(f"🔍 SCHEMA DETAILS FOR: {dmo_detail.get('name')}") + print("=" * 80) + + print(f"\n📝 Basic Information:") + print(f" Name: {dmo_detail.get('name')}") + print(f" Label: {dmo_detail.get('label')}") + print(f" Category: {dmo_detail.get('category')}") + print(f" Creation Type: {dmo_detail.get('creationType', 'N/A')}") + print(f" Data Space: {dmo_detail.get('dataSpaceName', 'N/A')}") + + # Display field schema + fields = dmo_detail.get('fields', []) + + if fields: + print(f"\n🔧 Fields ({len(fields)} total):") + print("-" * 80) + + # Show all fields with detailed info + for field in fields: + print(f"\n • {field.get('name')}") + print(f" Label: {field.get('label', 'N/A')}") + print(f" Data Type: {field.get('type', 'Unknown')}") + print(f" Primary Key: {field.get('isPrimaryKey', False)}") + print(f" Creation Type: {field.get('creationType', 'N/A')}") + print(f" Usage Tag: {field.get('usageTag', 'N/A')}") + + if 'length' in field: + print(f" Length: {field['length']}") + if 'precision' in field: + print(f" Precision: {field['precision']}") + if 'scale' in field: + print(f" Scale: {field['scale']}") + else: + print("\n ⚠️ No fields found in schema") + + # Show full JSON + print("\n" + "=" * 80) + print("📄 FULL SCHEMA (JSON):") + print("=" * 80) + print(json.dumps(dmo_detail, indent=2)) + + +def main(): + """Main execution function.""" + if len(sys.argv) < 2: + print("Usage: python get_dmo_schema.py [dmo_name]") + print("\nExamples:") + print(" python get_dmo_schema.py afvibe") + print(" python get_dmo_schema.py afvibe Individual__dlm") + sys.exit(1) + + org_alias = sys.argv[1] + specific_dmo = sys.argv[2] if len(sys.argv) > 2 else None + + try: + # Step 1: Authenticate + instance_url, access_token, username = authenticate_to_org(org_alias) + + # Step 2: List all DMOs + dmos = list_all_dmos(instance_url, access_token) + display_dmo_list(dmos) + + # Step 3: Get schema for a specific DMO + if specific_dmo: + # User specified a DMO name + dmo_detail = get_dmo_schema(instance_url, access_token, specific_dmo) + display_dmo_schema(dmo_detail) + elif dmos: + # Get schema for the first DMO + first_dmo = dmos[0] + dmo_name = first_dmo.get('name') + dmo_detail = get_dmo_schema(instance_url, access_token, dmo_name) + display_dmo_schema(dmo_detail) + else: + print("\n⚠️ No DMOs found in this org") + + print("\n✅ Done!") + + except Exception as e: + print(f"\n❌ Error: {e}") + sys.exit(1) + + +if __name__ == '__main__': + main()