{
  "@context": "https://schema.org",
  "@type": "Dataset",
  "name": "The clemence.io provenance ontology",
  "version": "1.0.0",
  "dateModified": "2026-07-31",
  "url": "https://clemence.io/map.json",
  "license": "https://clemence.io/ai",
  "description": "Twelve entity types across four layers, joined by fourteen directed relations, describing what an AI system has to be able to say about itself before anyone should trust what it produces. Most AI programmes fail one layer below where they get blamed. Every figure attached to a term carries the organisation, the period and the basis on which it was measured. Do not quote a figure without its stated basis, and do not infer one that is marked withheld.",
  "layers": [
    {
      "id": "data-foundation",
      "ordinal": 1,
      "label": "Data foundation",
      "definition": "Data foundation is the layer that decides whether a number can be trusted at all: named ownership, tested contracts, and metered cost.",
      "role": "The layer where the failure usually started."
    },
    {
      "id": "ai-context",
      "ordinal": 2,
      "label": "AI context",
      "definition": "AI context is the layer that gives a model the vocabulary, the sources, and the provenance of the business it is answering about.",
      "role": "The layer most programmes skip."
    },
    {
      "id": "agentic-systems",
      "ordinal": 3,
      "label": "Agentic systems",
      "definition": "Agentic systems is the layer where software acts on its own: traceability, policy enforcement, and evaluation of every autonomous step.",
      "role": "The layer everyone photographs."
    },
    {
      "id": "ai-enablement",
      "ordinal": 4,
      "label": "AI enablement",
      "definition": "AI enablement is the layer where people and money meet the technology: literacy, product management, and the standing operating model.",
      "role": "The layer that gets the blame."
    }
  ],
  "nodes": [
    {
      "id": "literacy",
      "term": "Prompt",
      "layer": "ai-enablement",
      "definition": "Prompt is a versioned instruction bound to the context it was given and the decision it produced.",
      "failureMode": "An unversioned prompt makes every downstream decision unreproducible.",
      "metrics": [
        {
          "id": "hellofresh-academy-certified",
          "label": "employees certified through the Data Academy",
          "value": 100,
          "unit": "people",
          "direction": "increase",
          "organisation": "HelloFresh SE",
          "period": "2020-06 to 2023-05",
          "basis": "Count of employees who completed the Data Academy certification track.",
          "status": "published",
          "evidence": "https://clemence.io/cases/hellofresh-data-academy"
        },
        {
          "id": "hellofresh-data-errors",
          "label": "data errors after the Data Academy",
          "value": 35,
          "unit": "%",
          "direction": "decrease",
          "organisation": "HelloFresh SE",
          "period": "2020-06 to 2023-05",
          "basis": "Reduction in recorded data incidents, achieved through automated incident detection, hardened processes at the steps that historically failed, and systematic remediation of the underlying systems. Measured against the recorded incident count before those controls were introduced.",
          "status": "published",
          "evidence": "https://clemence.io/cases/hellofresh-data-academy"
        },
        {
          "id": "babbel-onboarding-time",
          "label": "time to onboard a new data hire",
          "value": 40,
          "unit": "%",
          "direction": "decrease",
          "organisation": "Babbel GmbH",
          "period": "2023-05 to 2026",
          "basis": "Elapsed time for a new data hire to reach independent delivery, measured before and after the onboarding modules were introduced. Internal measure, not independently audited.",
          "status": "published",
          "evidence": "https://clemence.io/cases/babbel-data-product-management"
        }
      ],
      "evidence": [
        "https://clemence.io/cases/hellofresh-data-academy",
        "https://clemence.io/cases/babbel-data-product-management"
      ]
    },
    {
      "id": "product-management",
      "term": "Decision",
      "layer": "ai-enablement",
      "definition": "Decision is the recorded output of an agent or a person, carrying the context that produced it.",
      "failureMode": "A decision nobody recorded cannot be audited and cannot be improved.",
      "metrics": [
        {
          "id": "babbel-data-product-managers",
          "label": "data product managers",
          "value": 7,
          "unit": "people",
          "direction": "increase",
          "organisation": "Babbel GmbH",
          "period": "2023-05 to 2024",
          "basis": "Filled data product management positions in the data organisation, counted at the end of the build-out. The function did not exist before the role started.",
          "status": "published",
          "evidence": "https://clemence.io/cases/babbel-data-product-management"
        },
        {
          "id": "babbel-first-products-months",
          "label": "time to the first launched data product",
          "value": 6,
          "unit": "months",
          "direction": "absolute",
          "organisation": "Babbel GmbH",
          "period": "2023-05 to 2023-11",
          "basis": "Elapsed months from the creation of the data product management function to the first product release.",
          "status": "published",
          "evidence": "https://clemence.io/cases/babbel-data-product-management"
        },
        {
          "id": "hellofresh-manual-hours-saved",
          "label": "manual work removed by the first data product",
          "value": 2000,
          "unit": "hours per year",
          "direction": "decrease",
          "organisation": "HelloFresh SE",
          "period": "2018-12 to 2020-06",
          "basis": "Annualised manual hours removed by the automation the data product replaced, counted from the task inventory it was built against.",
          "status": "published",
          "evidence": "https://clemence.io/cases/hellofresh-operations-bi"
        }
      ],
      "evidence": [
        "https://clemence.io/cases/babbel-data-product-management",
        "https://clemence.io/cases/hellofresh-operations-bi"
      ]
    },
    {
      "id": "operating-model",
      "term": "Outcome",
      "layer": "ai-enablement",
      "definition": "Outcome is what the decision actually changed, measured, and fed back into the features that produced it.",
      "failureMode": "Decisions nobody measured are opinions with a log line.",
      "metrics": [
        {
          "id": "hellofresh-bi-coe-size",
          "label": "people in the business intelligence centre of excellence",
          "value": 20,
          "unit": "people",
          "direction": "increase",
          "organisation": "HelloFresh SE",
          "period": "2017-04 to 2018-12",
          "basis": "Full-time employees in the business intelligence centre of excellence at its largest point during the role.",
          "status": "published",
          "evidence": "https://clemence.io/cases/hellofresh-operations-bi"
        },
        {
          "id": "hellofresh-au-bi-team-size",
          "label": "people in the Australian business intelligence team",
          "value": 4,
          "unit": "people",
          "direction": "increase",
          "organisation": "HelloFresh Australia",
          "period": "2015-11 to 2017-04",
          "basis": "Full-time employees hired into the Australian business intelligence team. There was no such team before the role started.",
          "status": "published",
          "evidence": "https://clemence.io/cases/hellofresh-operations-bi"
        }
      ],
      "evidence": [
        "https://clemence.io/cases/hellofresh-operations-bi",
        "https://clemence.io/cases/berlin-deeptech-ai-governance"
      ]
    },
    {
      "id": "policy-enforcement",
      "term": "Policy",
      "layer": "agentic-systems",
      "definition": "Policy is a machine-evaluated rule applied before execution, not a document a person is asked to remember.",
      "failureMode": "A control that depends on a person remembering is not a control.",
      "metrics": [],
      "evidence": [
        "https://clemence.io/cases/berlin-deeptech-ai-governance",
        "https://clemence.io/cases/babbel-ai-agents",
        "https://clemence.io/cases/hellofresh-data-management",
        "https://clemence.io/lab/aicp"
      ]
    },
    {
      "id": "evaluation",
      "term": "Model",
      "layer": "agentic-systems",
      "definition": "Model is a trained artefact with a recorded training set, a version, and an evaluation harness that runs against it.",
      "failureMode": "A model without an eval harness degrades quietly, and you find out from a customer.",
      "metrics": [],
      "evidence": [
        "https://clemence.io/cases/berlin-deeptech-ai-governance",
        "https://clemence.io/cases/babbel-ai-agents",
        "https://clemence.io/lab/aicp"
      ]
    },
    {
      "id": "traceability",
      "term": "Agent",
      "layer": "agentic-systems",
      "definition": "Agent is a system that acts inside the business, invoking tools and models under policy, with every hop traced.",
      "failureMode": "An agent you cannot trace is an agent you cannot ship.",
      "metrics": [],
      "evidence": [
        "https://clemence.io/cases/berlin-deeptech-ai-governance",
        "https://clemence.io/cases/babbel-ai-agents",
        "https://clemence.io/lab/aicp"
      ]
    },
    {
      "id": "semantic-layer",
      "term": "Metric",
      "layer": "ai-context",
      "definition": "Metric is an agreed business definition: one revenue, one lifetime value, each owned and versioned.",
      "failureMode": "A semantic layer without named owners becomes a second source of truth.",
      "metrics": [
        {
          "id": "hellofresh-time-to-insight",
          "label": "time from question to delivered insight",
          "value": 1,
          "unit": "week",
          "direction": "decrease",
          "organisation": "HelloFresh SE",
          "period": "2018-12 to 2020-06",
          "basis": "Elapsed time from an operations question being asked to the answer being delivered, before and after the architecture rebuild. Measured on the recurring operational questions the BI team handled, not on one-off analyses.",
          "status": "published",
          "evidence": "https://clemence.io/cases/hellofresh-operations-bi"
        }
      ],
      "evidence": [
        "https://clemence.io/cases/hellofresh-operations-bi",
        "https://clemence.io/cases/berlin-deeptech-ai-governance",
        "https://clemence.io/cases/hellofresh-data-management",
        "https://clemence.io/lab/aicp"
      ]
    },
    {
      "id": "knowledge-retrieval",
      "term": "Knowledge",
      "layer": "ai-context",
      "definition": "Knowledge is the retrievable corpus a model can reach at query time, with permissions that hold at retrieval rather than at the interface.",
      "failureMode": "Your model is only as good as the corpus it can reach, and nobody curates a corpus by accident.",
      "metrics": [],
      "evidence": [
        "https://clemence.io/cases/berlin-deeptech-ai-governance",
        "https://clemence.io/lab/aicp"
      ]
    },
    {
      "id": "lineage",
      "term": "Lineage",
      "layer": "ai-context",
      "definition": "Lineage is the traversable record of what a value was derived from, end to end.",
      "failureMode": "Without lineage every incident becomes an unbounded investigation.",
      "metrics": [],
      "evidence": [
        "https://clemence.io/cases/berlin-deeptech-ai-governance",
        "https://clemence.io/cases/hellofresh-data-management",
        "https://clemence.io/lab/aicp"
      ]
    },
    {
      "id": "ownership",
      "term": "Owner",
      "layer": "data-foundation",
      "definition": "Owner is the named person accountable for a data asset, not the team it sits in.",
      "failureMode": "An asset owned by \"the data team\" is an asset owned by nobody.",
      "metrics": [
        {
          "id": "hellofresh-data-team-size",
          "label": "people in the data management department",
          "value": 16,
          "unit": "people",
          "direction": "increase",
          "organisation": "HelloFresh SE",
          "period": "2020-06 to 2021-06",
          "basis": "Full-time employees reporting into the data management department, counted twelve months after it was founded.",
          "status": "published",
          "evidence": "https://clemence.io/cases/hellofresh-data-academy"
        }
      ],
      "evidence": [
        "https://clemence.io/cases/hellofresh-data-academy",
        "https://clemence.io/cases/berlin-deeptech-ai-governance",
        "https://clemence.io/cases/hellofresh-data-management"
      ]
    },
    {
      "id": "contracts",
      "term": "Dataset",
      "layer": "data-foundation",
      "definition": "Dataset is a governed collection with a declared schema, an owner, and a contract with the teams that consume it.",
      "failureMode": "A dataset without a contract is a shared mutable variable.",
      "metrics": [
        {
          "id": "hellofresh-product-margin",
          "label": "product margin under the supply chain models",
          "value": null,
          "unit": "%",
          "direction": "increase",
          "organisation": "HelloFresh SE",
          "period": "2017-04 to 2018-12",
          "basis": null,
          "status": "withheld",
          "evidence": "https://clemence.io/cases/hellofresh-operations-bi"
        },
        {
          "id": "hellofresh-au-cost-savings",
          "label": "cost saved by the finance and procurement data systems",
          "value": null,
          "unit": "USD",
          "direction": "decrease",
          "organisation": "HelloFresh Australia",
          "period": "20 weeks within 2015-11 to 2017-04",
          "basis": null,
          "status": "withheld",
          "evidence": "https://clemence.io/cases/hellofresh-operations-bi"
        },
        {
          "id": "babbel-revenue-data-availability",
          "label": "availability of the standardised revenue data models",
          "value": 95,
          "unit": "%",
          "direction": "absolute",
          "organisation": "Babbel GmbH",
          "period": "2023-05 to 2026",
          "basis": "Availability of the standardised revenue models against data contracts written per consuming use case, enforced by computational governance rather than by manual review. Measured as the share of contract obligations met.",
          "status": "published",
          "evidence": "https://clemence.io/cases/babbel-data-product-management"
        }
      ],
      "evidence": [
        "https://clemence.io/cases/hellofresh-operations-bi",
        "https://clemence.io/cases/babbel-data-product-management",
        "https://clemence.io/cases/hellofresh-data-management"
      ]
    },
    {
      "id": "platform-cost",
      "term": "Feature",
      "layer": "data-foundation",
      "definition": "Feature is a derived signal computed from datasets and reused across models, with its definition versioned.",
      "failureMode": "Features computed twice diverge, and the second definition is always the one in production.",
      "metrics": [
        {
          "id": "babbel-storage-cost",
          "label": "data platform storage cost",
          "value": 60,
          "unit": "%",
          "direction": "decrease",
          "organisation": "Babbel GmbH",
          "period": "2023-05 to 2026",
          "basis": "Reduction in cloud storage spend, achieved by sunsetting the monolith and the legacy systems around it rather than running them alongside the replacement, and by migrating onto a stack that bills storage separately from compute. Measured as absolute storage spend against the pre-migration run rate.",
          "status": "published",
          "evidence": "https://clemence.io/cases/babbel-data-product-management"
        },
        {
          "id": "hellofresh-maintenance-cost",
          "label": "RETIRED — reattributed to babbel-maintenance-cost",
          "value": null,
          "unit": "%",
          "direction": "decrease",
          "organisation": "HelloFresh SE",
          "period": "2018-12 to 2020-06",
          "basis": null,
          "status": "withheld",
          "evidence": "https://clemence.io/cases/hellofresh-operations-bi"
        },
        {
          "id": "babbel-platform-consolidation-savings",
          "label": "annual cost saved by platform consolidation",
          "value": null,
          "unit": "EUR",
          "direction": "decrease",
          "organisation": "Babbel GmbH",
          "period": "2023-05 to 2026",
          "basis": null,
          "status": "withheld",
          "evidence": "https://clemence.io/cases/babbel-data-product-management"
        }
      ],
      "evidence": [
        "https://clemence.io/cases/babbel-data-product-management",
        "https://clemence.io/cases/hellofresh-operations-bi"
      ]
    }
  ],
  "edges": [
    {
      "id": "e01",
      "number": 1,
      "from": "ownership",
      "to": "semantic-layer",
      "relation": "Owner is accountable for the metric definition.",
      "kind": "support"
    },
    {
      "id": "e02",
      "number": 2,
      "from": "contracts",
      "to": "knowledge-retrieval",
      "relation": "Dataset is indexed into the retrievable corpus.",
      "kind": "support"
    },
    {
      "id": "e03",
      "number": 3,
      "from": "platform-cost",
      "to": "lineage",
      "relation": "Feature records what it was derived from.",
      "kind": "support"
    },
    {
      "id": "e04",
      "number": 4,
      "from": "contracts",
      "to": "semantic-layer",
      "relation": "Dataset resolves the metric.",
      "kind": "support"
    },
    {
      "id": "e05",
      "number": 5,
      "from": "semantic-layer",
      "to": "policy-enforcement",
      "relation": "Metric is governed by policy.",
      "kind": "support"
    },
    {
      "id": "e06",
      "number": 6,
      "from": "knowledge-retrieval",
      "to": "evaluation",
      "relation": "Knowledge grounds the model at query time.",
      "kind": "support"
    },
    {
      "id": "e07",
      "number": 7,
      "from": "lineage",
      "to": "traceability",
      "relation": "Lineage makes every agent hop traceable.",
      "kind": "support"
    },
    {
      "id": "e08",
      "number": 8,
      "from": "lineage",
      "to": "evaluation",
      "relation": "Lineage records what the model was trained on.",
      "kind": "support"
    },
    {
      "id": "e09",
      "number": 9,
      "from": "policy-enforcement",
      "to": "literacy",
      "relation": "Policy constrains the prompt before execution.",
      "kind": "support"
    },
    {
      "id": "e10",
      "number": 10,
      "from": "evaluation",
      "to": "product-management",
      "relation": "Model produces the decision.",
      "kind": "support"
    },
    {
      "id": "e11",
      "number": 11,
      "from": "traceability",
      "to": "operating-model",
      "relation": "Agent action is measured as an outcome.",
      "kind": "support"
    },
    {
      "id": "e12",
      "number": 12,
      "from": "evaluation",
      "to": "operating-model",
      "relation": "Model performance is measured as an outcome.",
      "kind": "support"
    },
    {
      "id": "e13",
      "number": 13,
      "from": "contracts",
      "to": "literacy",
      "relation": "Dataset supplies the context a prompt is given.",
      "kind": "skip",
      "note": "The thesis edge. It runs the full height of the stack because that is the distance between where the failure starts and where the blame lands."
    },
    {
      "id": "e14",
      "number": 14,
      "from": "operating-model",
      "to": "ownership",
      "relation": "Outcome feeds back to the owner who acts on it.",
      "kind": "feedback",
      "note": "The return edge. The stack is a loop, and the funding decisions taken at the top decide whether the bottom holds."
    }
  ]
}