{
  "format": "skilldb.starter-bundle",
  "schemaVersion": 1,
  "bundle": {
    "slug": "data-workflow",
    "title": "Data workflow",
    "url": "https://skilldb.dev/bundles/data-workflow"
  },
  "summary": "Plan a repeatable batch import, define what each row means, and make bad data visible.",
  "scope": "For a small analytical data pipeline. Select the simplest suitable storage and runtime; no infrastructure is created.",
  "review": {
    "date": "2026-10-04",
    "revision": "0a89389d0ec3e4f12e694baab71bdb56134eeb7e",
    "scope": "Selection reviewed for topic coverage and overlap. Code examples and task outcomes have not been evaluated.",
    "evaluationStatus": "pending"
  },
  "contentIncluded": false,
  "installationIncluded": false,
  "exportNotice": "Reference manifest only: selection notes, skill IDs, and public catalog links. No skill bodies, executable code, credentials, or host configuration are included. Downloading does not install or enable skills.",
  "rightsNotice": "Source licenses have not been verified for redistribution. Catalog pages provide a public reference; access does not grant redistribution rights. Verify the original license and attribution requirements before copying or packaging skill content.",
  "overlap": "Data Pipeline Architecture covers movement and recovery, Data Modeling covers grain and relationships, and Data Quality covers validation. All discuss contracts; write one shared contract rather than maintaining three versions.",
  "evaluation": {
    "status": "pending",
    "task": "Import a small orders dataset into a reporting table, then repeat the import with duplicate rows, missing fields, and a late correction.",
    "checks": [
      "Define row grain and a repeatable key before implementing transformations.",
      "Run the same import twice and verify that intended totals remain stable.",
      "Make rejected records, freshness, and reconciliation results visible; test a recovery or backfill."
    ]
  },
  "skills": [
    {
      "id": "data-engineering-skills/data-pipeline-architecture.md",
      "title": "Data Pipeline Architecture",
      "role": "Pipeline and recovery plan",
      "rationale": "Covers batch versus streaming choices, idempotency, schema changes, and recoverable failures.",
      "reviewNote": "The scale and latency examples are illustrative. Do not adopt streaming infrastructure or exactly-once claims without testing the actual source and sink.",
      "catalogUrl": "https://skilldb.dev/skills/data-engineering-skills/data-pipeline-architecture",
      "maintainerProvenance": {
        "repository": "latentsmurf/SkillDB",
        "access": "private; maintainer access required",
        "revision": "0a89389d0ec3e4f12e694baab71bdb56134eeb7e",
        "path": "packs/data-engineering-skills/data-pipeline-architecture.md"
      },
      "licenseStatus": "not-verified-for-redistribution"
    },
    {
      "id": "data-engineering-skills/data-modeling.md",
      "title": "Data Modeling",
      "role": "Row grain and relationships",
      "rationale": "Explains grain, keys, and model choices so the pipeline produces data that supports the intended questions.",
      "reviewNote": "Enterprise modeling approaches are alternatives, not a checklist to implement. A small workflow may need only a clearly documented table.",
      "catalogUrl": "https://skilldb.dev/skills/data-engineering-skills/data-modeling",
      "maintainerProvenance": {
        "repository": "latentsmurf/SkillDB",
        "access": "private; maintainer access required",
        "revision": "0a89389d0ec3e4f12e694baab71bdb56134eeb7e",
        "path": "packs/data-engineering-skills/data-modeling.md"
      },
      "licenseStatus": "not-verified-for-redistribution"
    },
    {
      "id": "data-engineering-skills/data-quality.md",
      "title": "Data Quality",
      "role": "Validation and failure visibility",
      "rationale": "Adds completeness, uniqueness, freshness, and reconciliation checks around the pipeline contract.",
      "reviewNote": "Choose thresholds from the dataset and business rules. The example percentages and composite scores are not validated targets for your project.",
      "catalogUrl": "https://skilldb.dev/skills/data-engineering-skills/data-quality",
      "maintainerProvenance": {
        "repository": "latentsmurf/SkillDB",
        "access": "private; maintainer access required",
        "revision": "0a89389d0ec3e4f12e694baab71bdb56134eeb7e",
        "path": "packs/data-engineering-skills/data-quality.md"
      },
      "licenseStatus": "not-verified-for-redistribution"
    }
  ]
}
