name: data-cleaning-pipeline
description: "A safe routine to profile, clean and validate a dataset, with a log of every change."
inputs:
  - "data file"
steps:
  - id: 1
    name: "Profile"
    do: "Report rows, columns, types, missing values and duplicates."
  - id: 2
    name: "Plan"
    do: "List fixes and what each changes."
  - id: 3
    name: "Clean"
    do: "Apply fixes to a copy of the data."
  - id: 4
    name: "Validate"
    do: "Compare counts and spot-check rows."
  - id: 5
    name: "Log"
    do: "Write the change log."
output: "A cleaned file, a script and a change log."
