{
 "meta": {
  "title": "The Agentic Data Engineering Index",
  "edition": "Q4 2026",
  "date": "2026-09-09",
  "publisher": "Data Workers",
  "contact": "hello@dataworkers.io",
  "subtitle": "Every vendor in the agentic data engineering market, including Data Workers, graded on one six-level autonomy scale, one cell at a time, every cell cited.",
  "why": "Buyers keep asking the same question in different words: when this vendor says agent, does the agent alert, suggest, draft, apply with approval, apply on its own with a receipt, or close the loop and verify? The word agent now covers all six. This index grades what each vendor publicly evidences, lane by lane, so the answer can be checked rather than believed.",
  "licence": "The dataset (data.json, served at https://dataworkers.io/agentic-data-engineering-index.json) is released under CC BY 4.0. Cite as: Data Workers, The Agentic Data Engineering Index, Q4 2026.",
  "licence_url": "https://creativecommons.org/licenses/by/4.0/",
  "dataset_url": "https://dataworkers.io/agentic-data-engineering-index.json",
  "dataset_local": "agentic-data-engineering-index-2026-q4/data.json (the competitors.csv survey it was built from is not public)",
  "data_json_local": "agentic-data-engineering-index-2026-q4/data.json",
  "update_cadence": "Quarterly. Cells change between editions only when a cited URL changes or a contest is upheld.",
  "next_edition": "Q1 2027"
 },
 "scale": [
  {
   "level": 0,
   "name": "Alert",
   "short": "Detects and notifies.",
   "definition": "The system detects a condition and notifies a human. The human diagnoses and does everything else."
  },
  {
   "level": 1,
   "name": "Suggest",
   "short": "Explains or recommends; human does everything.",
   "definition": "The agent explains what happened or recommends what to do. It does not produce the change itself."
  },
  {
   "level": 2,
   "name": "Propose",
   "short": "Drafts the change; human applies.",
   "definition": "The agent produces the actual artefact (SQL, model, test, description, pipeline code, review verdict). A human applies it."
  },
  {
   "level": 3,
   "name": "Apply with approval",
   "short": "Agent applies after a named human approves; dry-run; receipt.",
   "definition": "The agent applies the change itself, but only after a named human approves. A dry-run or preview precedes the approval and the change carries a record of what was done."
  },
  {
   "level": 4,
   "name": "Autonomous with receipts",
   "short": "Routine changes applied unattended within policy; every change has an audit receipt and rollback.",
   "definition": "Within a declared policy the agent applies routine changes without waiting for a human. Every change is logged with a receipt (what changed, why, blast radius) and can be rolled back."
  },
  {
   "level": 5,
   "name": "Self-verifying operation",
   "short": "Detects, fixes, verifies the outcome and learns, across systems, with governed escalation.",
   "definition": "The agent detects, fixes, then verifies that the fix produced the intended outcome, learns from the result, operates across systems, and escalates to a human under a governed policy when it cannot verify."
  }
 ],
 "lanes": [
  {
   "key": "incident",
   "name": "Incident resolution",
   "definition": "A pipeline or table breaks (freshness, failed run, wrong numbers). What does the agent do between the alert and the verified fix?"
  },
  {
   "key": "quality",
   "name": "Data quality",
   "definition": "Rules, anomaly detection and remediation of bad or missing data in production tables."
  },
  {
   "key": "schema",
   "name": "Schema and change review",
   "definition": "Reviewing a proposed change (PR, migration, schema evolution) for downstream impact, and acting on the verdict."
  },
  {
   "key": "pipeline",
   "name": "Pipeline build",
   "definition": "Creating or migrating transformation code and pipelines from a request, through validation to deployment."
  },
  {
   "key": "catalog",
   "name": "Catalog and documentation",
   "definition": "Producing and maintaining descriptions, semantic definitions, ownership and lineage in a catalog or context layer."
  },
  {
   "key": "xcloud",
   "name": "Cross-cloud scope",
   "definition": "The level at which the agent operates when the estate spans more than one warehouse or cloud (for example Snowflake and BigQuery in one workspace). Single-platform products are graded not evidenced here, not L0."
  }
 ],
 "grading_rules": [
  "A cell is graded at the highest level for which a public page evidences the mechanism. Where the only basis is a marketing sentence with no documented mechanism, the cell keeps the stated level but is marked claim and rendered hatched.",
  "L3 requires a documented approval gate before the agent applies. Dry-run and receipt strengthen an L3 cell but the gate is the test.",
  "L4 requires documented unattended application AND a documented per-change audit record and rollback. Unattended application with neither is graded as a claim.",
  "Where no page evidences the lane at all, the cell is graded not evidenced, not L0. L0 is a real grade for products that detect and alert.",
  "Citations are vendor docs, pricing or product pages, GitHub, or a file path in the public competitive-intelligence repository (prefixed ci:). Repository paths are used only where the underlying vendor page was captured there and could not be re-fetched this pass.",
  "Each vendor's homepage and up to two docs or pricing pages were fetched on 2026-09-09. Where the fetched page and the repository row disagree, the disagreement is recorded on the vendor card.",
  "Data Workers is graded from its public site (dataworkers.io, llms.txt) and public GitHub only, under the same rules. Internal knowledge was not used."
 ],
 "contest": {
  "how": "Send a URL to hello@dataworkers.io with the vendor, the lane and the level you believe is evidenced. A cell moves when the URL documents the mechanism; a vendor statement that a feature exists is recorded as a claim until the docs show it. Upheld contests are credited in the next edition's changelog.",
  "turnaround": "Contests are reviewed within ten business days. The vendor's own docs outrank third-party coverage; a dated, reachable page outranks a slide."
 },
 "vendors": [
  {
   "id": "upriver",
   "name": "Upriver",
   "scope": "Autonomous data engineering platform with a context engine (living map) and a plan, execute, validate loop under human sign-off.",
   "url": "https://www.upriverdata.com/",
   "category": "Direct peer",
   "publishes": {
    "price": "no",
    "soc2": "yes",
    "soc2_note": "trust.upriverdata.com lists SOC 2 Type 2, GDPR and HIPAA with a SOC 2 Type 2 document available on request.",
    "licence": "Proprietary (nimble_udf repo public; platform closed)",
    "mcp": "No first-party MCP server found; integrates via Claude and Cursor."
   },
   "funding": "$14M seed, June 2026 (Valley Capital Partners, Hetz Ventures).",
   "fetched": [
    "https://www.upriverdata.com/",
    "https://trust.upriverdata.com/",
    "https://www.upriverdata.com/product (404)"
   ],
   "ci_vs_live": "The repository row recorded the trust center as positioning only. The live trust center now lists SOC 2 Type 2 as an available document, so the SOC 2 field moves to yes. The /product path returned 404; product detail is taken from the homepage and the repository teardown.",
   "grades": {
    "incident": {
     "level": 3,
     "basis": "docs",
     "cite": "https://www.upriverdata.com/",
     "note": "Homepage: issues fixed before the business Slacks you; late pipelines, logical errors and slow queries surfaced and fixed. Teardown records a mandatory human sign-off and a per-change validation report (queries run, rows sampled, before and after metrics, verdict). Approval gate plus receipt is L3.",
     "cite2": "ci:teardowns/features/upriver--wave1.md"
    },
    "quality": {
     "level": 3,
     "basis": "docs",
     "cite": "https://www.upriverdata.com/",
     "note": "Flags violations of data quality standards and fixes them under the same sign-off and validation-report loop."
    },
    "schema": {
     "level": 3,
     "basis": "docs",
     "cite": "ci:teardowns/features/upriver--wave1.md",
     "note": "Every change ships with a validation report linked to lineage and waits for sign-off. The report is the receipt; the sign-off is the gate."
    },
    "pipeline": {
     "level": 3,
     "basis": "docs",
     "cite": "https://www.upriverdata.com/",
     "note": "Turn any request into a validated pipeline, with the same plan, execute, validate loop and human sign-off before production."
    },
    "catalog": {
     "level": 2,
     "basis": "claim",
     "cite": "https://www.upriverdata.com/",
     "note": "A living map of the full data environment, continuously updated in the background. Documentation is agent-maintained, but no page shows an approval or audit mechanism for doc writes, so the cell is capped at propose and marked claim."
    },
    "xcloud": {
     "level": null,
     "basis": "none",
     "cite": "https://www.upriverdata.com/",
     "note": "No page lists the supported warehouses. The teardown describes one warehouse plus one orchestrator plus one repo per deployment. Not evidenced."
    }
   }
  },
  {
   "id": "alkera",
   "name": "Alkera",
   "scope": "Data engineering agent in the CLI and IDE with column-level lineage and a living knowledge base.",
   "url": "https://alkera.ai/",
   "category": "Direct peer",
   "publishes": {
    "price": "yes",
    "soc2": "no",
    "soc2_note": "Homepage states pending SOC 2 Type II, ISO 27001, GDPR and HIPAA. trust.alkera.ai resolves (HTTP 200).",
    "licence": "Proprietary (public repo holds DataAgentBench traces only)",
    "mcp": "No MCP server found; distribution is its own CLI and IDE extension with first-party warehouse plugins."
   },
   "funding": "Y Combinator Summer 2026; no priced round disclosed.",
   "fetched": [
    "https://alkera.ai/",
    "https://docs.alkera.ai/",
    "https://pricing.alkera.ai/ (DNS did not resolve; alkera.ai/pricing returns 200)"
   ],
   "ci_vs_live": "The homepage still states Ranked first on DataAgentBench at 83.28 percent; the repository's July reading of the live board placed Alkera second behind Actioneer's Sentinel. The pricing subdomain the homepage links did not resolve on 2026-09-09; the tiers recorded in July ($0, $50, $250, custom) are retained from the repository row and alkera.ai/pricing.",
   "grades": {
    "incident": {
     "level": 3,
     "basis": "docs",
     "cite": "https://alkera.ai/",
     "note": "Correctness and data quality issues are triaged and fixed before they reach a report; destructive work waits for approval. Docs: each action stays within the access and change rules you set. Approval gate evidenced; no receipt or rollback documented."
    },
    "quality": {
     "level": 3,
     "basis": "docs",
     "cite": "https://alkera.ai/",
     "note": "Same triage-and-fix loop with approval on destructive work."
    },
    "schema": {
     "level": 3,
     "basis": "docs",
     "cite": "https://alkera.ai/",
     "note": "Global column lineage so migrations replicate dependencies and no change ships with unknown downstream impact, under the same approval rule. Impact analysis plus gated apply."
    },
    "pipeline": {
     "level": 3,
     "basis": "docs",
     "cite": "https://alkera.ai/",
     "note": "New pipelines ship in hours on the tools you already run, built to your conventions, with dbt and Airflow orchestration and the same approval rule."
    },
    "catalog": {
     "level": 2,
     "basis": "claim",
     "cite": "https://docs.alkera.ai/",
     "note": "Living knowledge base kept current as your pipeline changes, inside the docs you already use. No approval or audit mechanism for doc writes is documented. Capped at propose, marked claim."
    },
    "xcloud": {
     "level": 2,
     "basis": "docs",
     "cite": "https://alkera.ai/",
     "note": "One agent with plugins for Snowflake, Databricks, BigQuery, Redshift, ClickHouse and eight more, plus dbt and Airflow. Multi-warehouse drafting is evidenced; a gated apply spanning two warehouses is not."
    }
   }
  },
  {
   "id": "altimate",
   "name": "Altimate AI",
   "scope": "Agentic data engineering platform: dbt pipeline build, migration, warehouse optimisation; open-source altimate-code harness.",
   "url": "https://altimate.ai/",
   "category": "Direct peer",
   "publishes": {
    "price": "yes",
    "soc2": "yes",
    "soc2_note": "SOC 2 listed on the Enterprise tier; the repository's July pass verified a report and pen-test summary are available under NDA via a named process.",
    "licence": "MIT (altimate-code); platform proprietary",
    "mcp": "Yes; the local extension exposes an MCP endpoint with dbt, SQL, lineage and FinOps tools."
   },
   "funding": "$2M (SEC Form D, 2022); no later round found.",
   "fetched": [
    "https://altimate.ai/",
    "https://altimate.ai/pricing",
    "https://github.com/AltimateAI/altimate-code"
   ],
   "ci_vs_live": "Consistent. Live pricing adds outcome-based pricing wording on the Enterprise tier. altimate-code shows 809 stars.",
   "grades": {
    "incident": {
     "level": null,
     "basis": "none",
     "cite": "https://altimate.ai/",
     "note": "Neither the homepage nor the altimate-code README names incident response or on-call resolution as a capability. Not evidenced."
    },
    "quality": {
     "level": 2,
     "basis": "docs",
     "cite": "https://github.com/AltimateAI/altimate-code",
     "note": "README: data quality validation and PII detection, SQL anti-pattern detection. The harness drafts findings and tests; applying fixes to production data is not documented."
    },
    "schema": {
     "level": 2,
     "basis": "docs",
     "cite": "https://github.com/AltimateAI/altimate-code",
     "note": "Column-level lineage extraction across dialects and schema-aware impact assessment. Produces the assessment; a human acts on it."
    },
    "pipeline": {
     "level": 3,
     "basis": "docs",
     "cite": "https://github.com/AltimateAI/altimate-code",
     "note": "Build, test and deploy production-quality dbt pipelines. Builder mode requires approval before dangerous operations, with DROP DATABASE, DROP SCHEMA and TRUNCATE hard-blocked; Analyst mode is read-only. Gate evidenced in the README."
    },
    "catalog": {
     "level": null,
     "basis": "none",
     "cite": "https://altimate.ai/",
     "note": "Documentation generation exists in the older dbt Power User extension but was not on any page fetched this pass. Not evidenced; contest with a URL."
    },
    "xcloud": {
     "level": 2,
     "basis": "docs",
     "cite": "https://github.com/AltimateAI/altimate-code",
     "note": "Thirteen warehouses in one harness including Snowflake, BigQuery and Databricks; optimisation across all three on the homepage. Drafting across platforms is evidenced; applying across two is not."
    }
   }
  },
  {
   "id": "datus",
   "name": "Datus AI",
   "scope": "Open-source agentic data engineering agent (CLI, VS Code, MCP server and client) with context learning and the Dosi semantic compiler.",
   "url": "https://datus.ai/",
   "category": "Direct peer",
   "publishes": {
    "price": "partial",
    "soc2": "no",
    "soc2_note": "No SOC 2 claim anywhere on the site or pricing page. Enterprise lists governance, sandboxing and approvals.",
    "licence": "Apache-2.0",
    "mcp": "Yes; both MCP server and MCP client."
   },
   "funding": "Undisclosed; no filing found.",
   "fetched": [
    "https://datus.ai/",
    "https://github.com/Datus-ai/Datus-agent",
    "https://datus.ai/pricing/"
   ],
   "ci_vs_live": "Stars moved from about 1,407 in July to 1.7k. The pricing page now lists governance, sandboxing and approvals and long-running agents on the Enterprise tier, neither of which the July row recorded.",
   "grades": {
    "incident": {
     "level": 3,
     "basis": "docs",
     "cite": "https://datus.ai/",
     "note": "Proposed fixes for broken components inside shared workflow threads; writes go through execute_sql with explicit confirmation and statement-level SQL authorisation with AI pre-review (README). Per-statement approval is a gate, so L3, though narrower than a change-level approval with a receipt."
    },
    "quality": {
     "level": 1,
     "basis": "docs",
     "cite": "https://datus.ai/",
     "note": "Watches freshness, schema drift and metric anomalies across pipelines and explains them in the thread. Remediation of the data itself is not documented."
    },
    "schema": {
     "level": 1,
     "basis": "docs",
     "cite": "https://datus.ai/",
     "note": "Schema drift detection and an agent that understands warehouse, lineage and past failures. Impact explanation; no review verdict or gated apply documented."
    },
    "pipeline": {
     "level": 3,
     "basis": "docs",
     "cite": "https://github.com/Datus-ai/Datus-agent",
     "note": "Plan, write, run, validate, deploy, monitor; SQL authoring and validation, pipeline, report and dashboard generation; Airflow scheduling built in; write and DDL statements stop for confirmation."
    },
    "catalog": {
     "level": 2,
     "basis": "docs",
     "cite": "https://github.com/Datus-ai/Datus-agent",
     "note": "Reads schema and SQL history, then generates OSI semantic models; every run and correction settles into context. Drafts semantic definitions; humans ship them."
    },
    "xcloud": {
     "level": 2,
     "basis": "docs",
     "cite": "https://github.com/Datus-ai/Datus-agent",
     "note": "Nineteen-plus dialects via adapters; Dosi compiles one semantic model into SQL for 13-plus dialects. Cross-platform drafting evidenced."
    }
   }
  },
  {
   "id": "ardent",
   "name": "Ardent AI",
   "scope": "Formerly an AI data engineer. Now Postgres database branching for coding agents (clone any Postgres in about six seconds).",
   "url": "https://tryardent.com/",
   "category": "Pivoted (retained for continuity)",
   "publishes": {
    "price": "yes",
    "soc2": "unknown",
    "soc2_note": "No security or compliance statement on the homepage or pricing page.",
    "licence": "Proprietary (DE-Bench benchmark repo is AGPL-3.0)",
    "mcp": "No MCP server; coding agents integrate via ardent-cli."
   },
   "funding": "$2.15M pre-seed, September 2025 (Crane Venture Partners).",
   "fetched": [
    "https://tryardent.com/",
    "https://tryardent.com/pricing"
   ],
   "ci_vs_live": "Consistent with the repository's pivot note. Live pricing: Starter $0 plus compute and storage, Scale $250 per month, Enterprise custom, compute $0.40 per CU-hour, storage $0.70 per GB-month.",
   "grades": {
    "incident": {
     "level": null,
     "basis": "none",
     "cite": "https://tryardent.com/",
     "note": "The product is a database sandbox for other agents. It does not resolve incidents."
    },
    "quality": {
     "level": null,
     "basis": "none",
     "cite": "https://tryardent.com/",
     "note": "Let agents clean, deduplicate and standardize data on an exact copy of production describes infrastructure for a third-party agent, not an agent of its own."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://tryardent.com/",
     "note": "Migration testing on a branch is a capability of the sandbox, not an agentic review."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://tryardent.com/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": null,
     "basis": "none",
     "cite": "https://tryardent.com/",
     "note": "Not evidenced."
    },
    "xcloud": {
     "level": null,
     "basis": "none",
     "cite": "https://tryardent.com/",
     "note": "Postgres only."
    }
   }
  },
  {
   "id": "recce",
   "name": "Recce",
   "scope": "AI data review agent for dbt pull requests: schema diff, data diff, lineage impact, review summary posted to the PR. Open-source core plus Recce Cloud.",
   "url": "https://reccehq.com/",
   "category": "Direct peer (single lane)",
   "publishes": {
    "price": "yes",
    "soc2": "yes",
    "soc2_note": "SOC 2 badge on the homepage and pricing page; trust.reccehq.com returns 200.",
    "licence": "Apache-2.0 (DataRecce/recce)",
    "mcp": "Yes; MCP server and a Claude Code plugin in the docs setup guides."
   },
   "funding": "$4M, April 2025 (Heavybit).",
   "fetched": [
    "https://reccehq.com/",
    "https://reccehq.com/pricing/",
    "https://docs.reccehq.com/"
   ],
   "ci_vs_live": "Consistent. Live pricing: Free $0 with 100 agent reviews per month, Team $250 per month annual or $300 monthly with 1,000 reviews, Enterprise custom with SSO, BYOC and RBAC.",
   "grades": {
    "incident": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.reccehq.com/",
     "note": "Review-time only. No production incident workflow."
    },
    "quality": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.reccehq.com/",
     "note": "One-off questions become automated checks, but they run at PR time. Graded under change review, not production data quality."
    },
    "schema": {
     "level": 2,
     "basis": "docs",
     "cite": "https://docs.reccehq.com/",
     "note": "On every PR the agent runs data diffs, identifies impact to the column level, writes a data review summary explaining what matters and what is safe to ignore, and posts it as a PR comment. It produces the review verdict; the human merges. It never applies changes."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.reccehq.com/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.reccehq.com/",
     "note": "Not evidenced."
    },
    "xcloud": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.reccehq.com/",
     "note": "Works with Snowflake, BigQuery, Redshift and Databricks, one warehouse per project. No cross-platform review evidenced."
    }
   }
  },
  {
   "id": "validio",
   "name": "Validio",
   "scope": "Agentic data quality, observability and lineage: anomaly detection, agentic root-cause analysis, catalog.",
   "url": "https://validio.io/",
   "category": "Observability with agentic RCA",
   "publishes": {
    "price": "no",
    "soc2": "yes",
    "soc2_note": "ISO 27001 and SOC 2 Type II stated on the homepage and pricing page; GDPR and HIPAA also stated.",
    "licence": "Proprietary",
    "mcp": "Not stated."
   },
   "funding": "$30M Series A, March 2026 (Plural); $47M total.",
   "fetched": [
    "https://validio.io/",
    "https://docs.validio.io/",
    "https://validio.io/pricing"
   ],
   "ci_vs_live": "The repository row carries the vendor claim detect and autonomously resolve. Neither the homepage nor the docs fetched this pass documents any remediation mechanism; the evidenced product is detection, agentic root-cause analysis and lineage. The row's claim is therefore not graded.",
   "grades": {
    "incident": {
     "level": 1,
     "basis": "docs",
     "cite": "https://docs.validio.io/",
     "note": "Lineage tracking and incident grouping identify where issues originate, so you can debug across pipelines rather than individual tables. Agentic root-cause analysis; no fix applied."
    },
    "quality": {
     "level": 1,
     "basis": "docs",
     "cite": "https://validio.io/",
     "note": "ML anomaly detection with adaptive thresholds, AI-assisted setup with automatic recommendations. Recommends; does not remediate."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.validio.io/",
     "note": "Schema change detection was not on any page fetched. Not evidenced; contest with a URL."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://validio.io/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": 0,
     "basis": "docs",
     "cite": "https://docs.validio.io/",
     "note": "Cataloging is listed as a platform capability alongside observability, quality and lineage. Nothing shows agent-written documentation. Graded L0 for surfacing assets and alerts."
    },
    "xcloud": {
     "level": 1,
     "basis": "docs",
     "cite": "https://validio.io/",
     "note": "Fifteen-plus platforms including Snowflake, BigQuery, Redshift, Databricks, Kafka, S3 and ADLS; RCA groups incidents across pipelines. Explains across platforms; changes nothing."
    }
   }
  },
  {
   "id": "velum",
   "name": "Velum Labs",
   "scope": "YC W26 profile: the OS for data quality across any stack. Live homepage on 2026-09-09: RouteKit, a local router that sends each coding task to the model that should do it.",
   "url": "https://velum-labs.com/",
   "category": "Pivoted or repositioned (retained for continuity)",
   "publishes": {
    "price": "no",
    "soc2": "unknown",
    "soc2_note": "No security statement on the homepage.",
    "licence": "Unknown",
    "mcp": "Not stated."
   },
   "funding": "Y Combinator Winter 2026 (YC API); no priced round found.",
   "fetched": [
    "https://velum-labs.com/",
    "https://www.ycombinator.com/companies/velum-labs"
   ],
   "ci_vs_live": "Disagreement. The repository row describes data-quality resolution with governed contract write-back, and the YC profile still says data quality. The live homepage sells RouteKit, a model router for coding agents, and does not mention data quality, lineage or contracts. Graded on the live page.",
   "grades": {
    "incident": {
     "level": null,
     "basis": "none",
     "cite": "https://velum-labs.com/",
     "note": "No data engineering product on the live site."
    },
    "quality": {
     "level": null,
     "basis": "none",
     "cite": "https://www.ycombinator.com/companies/velum-labs",
     "note": "The YC profile describes automated data-quality monitoring and enforcement; no product page documents it. Not evidenced."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://velum-labs.com/",
     "note": "Not evidenced."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://velum-labs.com/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": null,
     "basis": "none",
     "cite": "https://velum-labs.com/",
     "note": "Not evidenced."
    },
    "xcloud": {
     "level": null,
     "basis": "none",
     "cite": "https://velum-labs.com/",
     "note": "Not evidenced."
    }
   }
  },
  {
   "id": "mica",
   "name": "Mica AI",
   "scope": "YC S24 profile: replace the humans fixing bad data. Live homepage on 2026-09-09: a platform for publishing, reviewing and installing employee-built AI agents across ChatGPT, Claude and Cursor.",
   "url": "https://usemica.com/",
   "category": "Pivoted or repositioned (retained for continuity)",
   "publishes": {
    "price": "no",
    "soc2": "unknown",
    "soc2_note": "No security statement on the homepage.",
    "licence": "Proprietary",
    "mcp": "Not stated."
   },
   "funding": "Undisclosed; Y Combinator Summer 2024.",
   "fetched": [
    "https://usemica.com/",
    "https://www.ycombinator.com/companies/mica-ai"
   ],
   "ci_vs_live": "Disagreement. The repository row describes an autonomous data-fix agent at about 30 percent of manual cost. The live homepage describes agent sharing and governance (publish, admin review, install, run, one-click rollback of agent versions) with no data pipeline capability. The YC profile still carries the data-fix description. Graded on the live page.",
   "grades": {
    "incident": {
     "level": null,
     "basis": "none",
     "cite": "https://usemica.com/",
     "note": "The YC profile's autonomous resolution of pipeline errors is not on any product page. Not evidenced."
    },
    "quality": {
     "level": null,
     "basis": "none",
     "cite": "https://usemica.com/",
     "note": "Not evidenced."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://usemica.com/",
     "note": "Not evidenced."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://usemica.com/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": null,
     "basis": "none",
     "cite": "https://usemica.com/",
     "note": "Not evidenced."
    },
    "xcloud": {
     "level": null,
     "basis": "none",
     "cite": "https://usemica.com/",
     "note": "Not evidenced."
    }
   }
  },
  {
   "id": "actioneer",
   "name": "Actioneer",
   "scope": "Enterprise AI agent platform for operational work: conversational analytics over approved definitions, monitoring agents on business metrics, operational playbooks with approvers.",
   "url": "https://actioneer.com/",
   "category": "Adjacent (business operations)",
   "publishes": {
    "price": "no",
    "soc2": "unknown",
    "soc2_note": "Footer badges assert SOC 2, ISO/IEC 27001:2022, ISO/IEC 27701:2019 and GDPR. trust.actioneer.com has no DNS record and actioneer.com/security returns 404 (checked 2026-09-09). Asserted, not verifiable from public pages.",
    "licence": "Proprietary (can run on open-source models on-premise)",
    "mcp": "Not mentioned."
   },
   "funding": "$5.4M pre-seed, August 2025, raised as GameRamp (BITKRAFT Ventures).",
   "fetched": [
    "https://actioneer.com/",
    "https://actioneer.com/resources"
   ],
   "ci_vs_live": "The July row recorded a bare SOC 2 certified claim; the live footer adds ISO 27001 and ISO 27701 badges, still with no reachable trust page. Benchmark line now reads #1 on DABstep, SOTA on KramaBench and DataAgentBench.",
   "grades": {
    "incident": {
     "level": 1,
     "basis": "docs",
     "cite": "https://actioneer.com/",
     "note": "Monitoring agents watch metrics and thresholds; the moment something moves, they detect it, trace it to the source, and flag what changed. Business-metric incidents, not pipeline incidents. Playbooks carry a decision to an outcome with a named approver and a recorded reasoning trail, which would be L3 for operational actions, but no page shows a playbook repairing a data pipeline."
    },
    "quality": {
     "level": null,
     "basis": "none",
     "cite": "https://actioneer.com/",
     "note": "Not evidenced."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://actioneer.com/",
     "note": "Not evidenced."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://actioneer.com/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": null,
     "basis": "none",
     "cite": "https://actioneer.com/",
     "note": "Answers are grounded in approved data definitions in a context store; the store is human-approved. Not evidenced as agent-written documentation."
    },
    "xcloud": {
     "level": 1,
     "basis": "docs",
     "cite": "https://actioneer.com/",
     "note": "Snowflake, BigQuery and SaaS sources in one platform; read-only by default with write access scoped per capability. Explains across platforms."
    }
   }
  },
  {
   "id": "nao",
   "name": "Nao Labs",
   "scope": "Open-source analytics agent built for context engineering: text-to-SQL chat over a file-system context, nao MCP, self-host with your own key.",
   "url": "https://getnao.io/",
   "category": "Adjacent (analytics agent)",
   "publishes": {
    "price": "partial",
    "soc2": "yes",
    "soc2_note": "SOC 2 and AICPA badges; SOC 2 Type II reports on the Enterprise tier; compliance.getnao.io returns 200.",
    "licence": "Apache-2.0 with an enterprise carve-out (SSO, RLS, impersonation behind a licence key)",
    "mcp": "Yes; nao MCP reachable from Claude, Codex, Cursor, Slack, Teams, WhatsApp, Telegram, and external MCPs can be added to the agent."
   },
   "funding": "About $500K, July 2025 (Kima Ventures, Sapienta VC, Y Combinator).",
   "fetched": [
    "https://getnao.io/",
    "https://getnao.io/pricing/",
    "https://github.com/getnao/nao"
   ],
   "ci_vs_live": "Consistent. GitHub shows 1.6k stars. Free self-host with your own key; Cloud and Enterprise are custom and unpriced.",
   "grades": {
    "incident": {
     "level": null,
     "basis": "none",
     "cite": "https://getnao.io/",
     "note": "Analytics only. No incident workflow."
    },
    "quality": {
     "level": null,
     "basis": "none",
     "cite": "https://getnao.io/",
     "note": "Not evidenced."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://getnao.io/",
     "note": "Not evidenced."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://getnao.io/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": null,
     "basis": "none",
     "cite": "https://github.com/getnao/nao",
     "note": "Context is human-authored files in git (metadata, docs, tools, MCPs) that the agent reads. The agent does not write documentation. Not evidenced."
    },
    "xcloud": {
     "level": 1,
     "basis": "docs",
     "cite": "https://getnao.io/",
     "note": "BigQuery, Snowflake, Postgres, Databricks, DuckDB, MotherDuck and Redshift; read-only answers with transparent reasoning."
    }
   }
  },
  {
   "id": "ataccama",
   "name": "Ataccama",
   "scope": "Ataccama ONE: data quality, observability, catalog, lineage, governance and MDM, now with a ONE AI Agent (digital data steward) and an MCP server.",
   "url": "https://www.ataccama.com/",
   "category": "Incumbent converging on the category",
   "publishes": {
    "price": "no",
    "soc2": "yes",
    "soc2_note": "Public trust center; the repository's July pass recorded ISO 27001:2022, ISO 9001 and an annual SOC 2 Type II.",
    "licence": "Proprietary",
    "mcp": "Yes; a named platform component distributing trust signals and metadata to agents, copilots and LLMs."
   },
   "funding": "$150M minority growth, 2023 (Bain Capital Tech Opportunities).",
   "fetched": [
    "https://www.ataccama.com/",
    "https://www.ataccama.com/ai",
    "https://www.ataccama.com/platform/mcp",
    "https://www.ataccama.com/pricing"
   ],
   "ci_vs_live": "Consistent. Pricing page describes three dimensions (named users, data objects, active DQ configurations) with no figures. Four pages were fetched for this vendor, one over the pass budget, because the MCP and AI pages are separate.",
   "grades": {
    "incident": {
     "level": null,
     "basis": "none",
     "cite": "https://www.ataccama.com/ai",
     "note": "No incident or on-call workflow on the pages fetched. Data-quality remediation is graded in the quality lane."
    },
    "quality": {
     "level": 4,
     "basis": "claim",
     "cite": "https://www.ataccama.com/ai",
     "note": "The ONE AI Agent autonomously generates data quality rules, suggests where to apply them, finds anomalies, and executes fixes at the source. Unattended execution is stated; no page documents a per-change audit receipt, an approval policy or rollback. Graded at the stated level and marked claim."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://www.ataccama.com/",
     "note": "Not evidenced on the pages fetched."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://www.ataccama.com/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": 2,
     "basis": "docs",
     "cite": "https://www.ataccama.com/ai",
     "note": "Auto-generated asset descriptions and table description generation and editing; auto-generated rules suggested for placement. Drafts that stewards edit."
    },
    "xcloud": {
     "level": 2,
     "basis": "docs",
     "cite": "https://www.ataccama.com/",
     "note": "One platform over Snowflake, Databricks, BigQuery and Redshift; descriptions and rules are drafted across sources. Cross-source unattended fixes inherit the claim mark from the quality lane, so the cross-cloud cell stops at propose."
    }
   }
  },
  {
   "id": "atlan",
   "name": "Atlan",
   "scope": "The context layer for enterprise AI: catalog, governance, Context Engineering Studio with AI-bootstrapped definitions certified by humans, Atlan MCP server.",
   "url": "https://atlan.com/",
   "category": "Incumbent catalog",
   "publishes": {
    "price": "no",
    "soc2": "unknown",
    "soc2_note": "No SOC 2 statement on the homepage, pricing page or Context Engineering Studio page fetched this pass. Atlan publishes a trust portal elsewhere; not verified in this pass.",
    "licence": "Proprietary (MIT-licensed client SDKs)",
    "mcp": "Yes; hosted Atlan MCP server behind OAuth, and Context Repos expose context over MCP."
   },
   "funding": "$105M Series C, May 2024; about $206M total.",
   "fetched": [
    "https://atlan.com/",
    "https://atlan.com/pricing/",
    "https://atlan.com/context-engineering-studio/"
   ],
   "ci_vs_live": "Consistent. Pricing page publishes no figures and routes to sales.",
   "grades": {
    "incident": {
     "level": null,
     "basis": "none",
     "cite": "https://atlan.com/",
     "note": "Not evidenced."
    },
    "quality": {
     "level": null,
     "basis": "none",
     "cite": "https://atlan.com/",
     "note": "Not evidenced on the pages fetched."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://atlan.com/",
     "note": "Impact analysis is a catalog feature elsewhere in the product; not on the pages fetched. Contest with a URL."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://atlan.com/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": 3,
     "basis": "docs",
     "cite": "https://atlan.com/context-engineering-studio/",
     "note": "Description Generator, Term Linkage, Metrics Generator and Ontology Generator draft context; domain experts review AI-bootstrapped definitions, resolve conflicts and approve context for production; git-like versioning with full history, staged rollouts and rollbacks. Draft, named approval, apply, rollback: L3."
    },
    "xcloud": {
     "level": 2,
     "basis": "docs",
     "cite": "https://atlan.com/",
     "note": "Eighty-plus connectors across Snowflake, Databricks, BigQuery, Redshift, BI and business systems in one graph; context drafted across all of them. The approval flow is documented per repo, not shown spanning platforms, so the cross-cloud cell stops at propose."
    }
   }
  },
  {
   "id": "databricks",
   "name": "Databricks (Lakeflow, Genie)",
   "scope": "Lakeflow data engineering with Genie Code agents and Lakeflow Designer, Unity Catalog governance, Genie One and Genie Ontology.",
   "url": "https://www.databricks.com/product/data-engineering",
   "category": "Platform vendor",
   "publishes": {
    "price": "yes",
    "soc2": "yes",
    "soc2_note": "Trust page states certifications and attestations for regulated industries with a compliance sub-page; specific reports not quoted on the page fetched.",
    "licence": "Proprietary platform; Unity Catalog OSS is Apache-2.0",
    "mcp": "Yes; managed MCP and Genie One expose MCP per the repository row (Data+AI Summit, June 2026)."
   },
   "funding": "Private; $134B valuation (Series L, completed February 2026).",
   "fetched": [
    "https://www.databricks.com/product/data-engineering",
    "https://www.databricks.com/trust",
    "https://www.databricks.com/product/lakeflow (404)"
   ],
   "ci_vs_live": "The repository row records Lakeflow ZeroOps (autonomous remediation) from the June summit. No page fetched this pass documents ZeroOps or its approval and audit model, so incident resolution is graded on Genie Code's troubleshooting wording only. Contest with a docs URL.",
   "grades": {
    "incident": {
     "level": 2,
     "basis": "docs",
     "cite": "https://www.databricks.com/product/data-engineering",
     "note": "Genie Code: agents that understand your data and can author, maintain and troubleshoot data pipelines. Drafts the fix in the workspace. ZeroOps unattended remediation is announced but undocumented on the pages fetched."
    },
    "quality": {
     "level": 0,
     "basis": "docs",
     "cite": "https://www.databricks.com/product/data-engineering",
     "note": "Full visibility into pipeline health with real-time metrics; custom alerts guarantee you know exactly when issues occur. Alerting. Declarative expectations are a platform rule, not an agent."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.databricks.com/aws/en/data-engineering/",
     "note": "Schema evolution is a pipeline feature; no agentic change review evidenced."
    },
    "pipeline": {
     "level": 2,
     "basis": "docs",
     "cite": "https://www.databricks.com/product/data-engineering",
     "note": "Genie Code builds pipelines from natural language; Lakeflow Designer is AI-first authoring. The user runs and deploys."
    },
    "catalog": {
     "level": 2,
     "basis": "docs",
     "cite": "https://www.databricks.com/blog/introducing-genie-one-genie-ontology-and-genie-agents",
     "note": "Genie Ontology auto-extracts a learned context layer (snippets with source and definition) on top of Unity Catalog; each snippet is inspectable. Learned drafts consumed by Genie; human curation model not detailed.",
     "cite2": "ci:teardowns/features/databricks--genie-one-ontology-agents-regate-2026-07.md"
    },
    "xcloud": {
     "level": null,
     "basis": "none",
     "cite": "https://www.databricks.com/product/data-engineering",
     "note": "Runs on AWS, Azure and GCP but operates one lakehouse. Not evidenced across warehouses."
    }
   }
  },
  {
   "id": "snowflake",
   "name": "Snowflake (Cortex Code, TensorStax)",
   "scope": "Cortex Code (CoCo): an agent inside Snowflake for data engineering, analytics and ML, in Snowsight, Desktop, CLI and editors. TensorStax's autonomous data engineering agent was acquired in February 2026.",
   "url": "https://www.snowflake.com/en/product/features/cortex-code/",
   "category": "Platform vendor",
   "publishes": {
    "price": "yes",
    "soc2": "yes",
    "soc2_note": "Public trust center at trust.snowflake.com (body not readable by the fetcher); Snowflake is a listed company with published SOC 2 and ISO reports.",
    "licence": "Proprietary; Apache Polaris is Apache-2.0",
    "mcp": "Yes; CoCo CLI implements MCP (preview) and Snowflake managed MCP is GA."
   },
   "funding": "Public company (NYSE: SNOW).",
   "fetched": [
    "https://www.snowflake.com/en/product/features/cortex-code/",
    "https://docs.snowflake.com/en/user-guide/cortex-code/cortex-code",
    "https://trust.snowflake.com/"
   ],
   "ci_vs_live": "Consistent. Neither page names TensorStax; the acquisition is recorded in the repository row (announced 2026-02-04). Docs describe a three-tier approval system for the CLI without detailing the tiers.",
   "grades": {
    "incident": {
     "level": 2,
     "basis": "docs",
     "cite": "https://docs.snowflake.com/en/user-guide/cortex-code/cortex-code",
     "note": "Take actions and answer questions about credit consumption, query performance, governance and user permissions; explain and optimise existing queries. Drafts fixes inside the account; no incident workflow with a gate is documented."
    },
    "quality": {
     "level": null,
     "basis": "none",
     "cite": "https://www.snowflake.com/en/product/features/cortex-code/",
     "note": "Not evidenced on the pages fetched."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://www.snowflake.com/en/product/features/cortex-code/",
     "note": "Not evidenced."
    },
    "pipeline": {
     "level": 3,
     "basis": "docs",
     "cite": "https://docs.snowflake.com/en/user-guide/cortex-code/cortex-code",
     "note": "Create data pipelines, ML models, apps and agents in plain language, with dbt, Airflow and Spark support; generated ML pipelines run in Notebooks. The CLI has a three-tier approval system and inherits Snowflake RBAC. Gate evidenced, tiers undocumented."
    },
    "catalog": {
     "level": 1,
     "basis": "docs",
     "cite": "https://www.snowflake.com/en/product/features/cortex-code/",
     "note": "CoCo reads your catalog, lineage and RBAC policies so generated code references real objects; semantic catalog search built in. Consumes the catalog; writing documentation is not evidenced."
    },
    "xcloud": {
     "level": null,
     "basis": "none",
     "cite": "https://www.snowflake.com/en/product/features/cortex-code/",
     "note": "Snowflake-native. Not evidenced across warehouses."
    }
   }
  },
  {
   "id": "google",
   "name": "Google (BigQuery data engineering agent, Dataplex)",
   "scope": "Gemini-powered data engineering agent in BigQuery Studio and Dataform, plus Dataplex Universal Catalog with a first-party MCP server.",
   "url": "https://docs.cloud.google.com/gemini/data-agents/data-engineering-agent/agent-overview",
   "category": "Platform vendor",
   "publishes": {
    "price": "yes",
    "soc2": "yes",
    "soc2_note": "Google Cloud publishes SOC 2 and ISO reports through its compliance centre; not re-fetched this pass.",
    "licence": "Proprietary",
    "mcp": "Yes; first-party Dataplex MCP server (read and write on data products, assets and aspects) per the repository row."
   },
   "funding": "Alphabet product line.",
   "fetched": [
    "https://cloud.google.com/bigquery/docs/data-engineering-agent (301, target 404)",
    "https://cloud.google.com/dataplex (truncated)",
    "https://docs.cloud.google.com/bigquery/docs/data-engineering-agent-intro (404)"
   ],
   "ci_vs_live": "The BigQuery docs path recorded in the repository moved under docs.cloud.google.com/gemini/data-agents/ (HTTP 200 on 2026-09-09). Evidence is taken from the repository teardown, which captured that page in June, with the live URL cited.",
   "grades": {
    "incident": {
     "level": 2,
     "basis": "docs",
     "cite": "https://docs.cloud.google.com/gemini/data-agents/data-engineering-agent/agent-overview",
     "note": "Failure troubleshooting reads execution logs, root-causes failures and proposes fixes; Planning Mode emits an explicit plan for human review before acting. Proposes; human applies.",
     "cite2": "ci:teardowns/features/bigquery-data-engineering-agents--wave-2.md"
    },
    "quality": {
     "level": 1,
     "basis": "claim",
     "cite": "ci:teardowns/features/bigquery-data-engineering-agents--wave-2.md",
     "note": "Auto-generated data quality scorecards and assertions derived from Dataplex rules are stated as an output of pipeline generation; the teardown marks this medium confidence, not verified verbatim in primary docs. Marked claim."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.cloud.google.com/gemini/data-agents/data-engineering-agent/agent-overview",
     "note": "Not evidenced."
    },
    "pipeline": {
     "level": 2,
     "basis": "docs",
     "cite": "https://docs.cloud.google.com/gemini/data-agents/data-engineering-agent/agent-overview",
     "note": "Natural-language request to Dataform/SQL pipeline code; plans, generates, self-compiles and fixes its own compilation errors, then hands to a human. Explicitly human-gated: it does not deploy on its own."
    },
    "catalog": {
     "level": 2,
     "basis": "docs",
     "cite": "https://cloud.google.com/dataplex",
     "note": "Dataplex Universal Catalog generates metadata and insights and exposes a read-write MCP server; the fetched page was truncated, so the mechanism for human review is not confirmed. Graded propose."
    },
    "xcloud": {
     "level": null,
     "basis": "none",
     "cite": "https://cloud.google.com/dataplex",
     "note": "BigQuery and Dataform only for the agent. Not evidenced across warehouses."
    }
   }
  },
  {
   "id": "wren",
   "name": "Wren AI",
   "scope": "Agentic GenBI on an open context engine (MDL semantics), governed text-to-SQL, GenBI apps, MCP.",
   "url": "https://getwren.ai/",
   "category": "Adjacent (analytics and context layer)",
   "publishes": {
    "price": "yes",
    "soc2": "yes",
    "soc2_note": "SOC 2 Type II ticked on every cloud tier of the pricing page.",
    "licence": "Apache-2.0 (Canner/WrenAI engine; open core)",
    "mcp": "Yes; MCP listed as a first-class consumption pattern."
   },
   "funding": "Last known $3.5M pre-A, March 2022 (Taiwania Capital).",
   "fetched": [
    "https://getwren.ai/",
    "https://getwren.ai/pricing",
    "https://docs.getwren.ai/cp/overview"
   ],
   "ci_vs_live": "Consistent. Live tiers: Free $0 (20 credits), Essential $179 per month, Enterprise $559 per month, Enterprise Plus custom, open source free.",
   "grades": {
    "incident": {
     "level": null,
     "basis": "none",
     "cite": "https://getwren.ai/",
     "note": "Not evidenced."
    },
    "quality": {
     "level": null,
     "basis": "none",
     "cite": "https://getwren.ai/",
     "note": "Not evidenced."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://getwren.ai/",
     "note": "Not evidenced."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://getwren.ai/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.getwren.ai/cp/overview",
     "note": "The context layer (MDL) is human-authored; the agent learns Knowledge, Skills and Memories for answering. Agent-written catalog documentation is not evidenced."
    },
    "xcloud": {
     "level": 1,
     "basis": "docs",
     "cite": "https://getwren.ai/",
     "note": "Twenty-plus connectors including Snowflake, BigQuery, Databricks and Redshift; read-only governed answers with the SQL behind them."
    }
   }
  },
  {
   "id": "openmetadata",
   "name": "OpenMetadata (Collate)",
   "scope": "Open-source catalog repositioned as the open context layer for AI agents: knowledge graph, memory, data quality, lineage, incident manager, MCP server with read and write.",
   "url": "https://open-metadata.org/",
   "category": "Open-source catalog and context layer",
   "publishes": {
    "price": "partial",
    "soc2": "unknown",
    "soc2_note": "No SOC 2 statement on the homepage or README; Collate's cloud may publish one separately.",
    "licence": "Apache-2.0",
    "mcp": "Yes; MCP server for direct agent-to-context integration, read and write on the same auth engine as the APIs."
   },
   "funding": "$10M Series A, July 2025 (Venrock) via Collate.",
   "fetched": [
    "https://open-metadata.org/",
    "https://github.com/open-metadata/OpenMetadata",
    "https://blog.open-metadata.org/announcing-openmetadata-2-0-the-open-context-layer-for-ai-agents-83b8ce8b9dde (403)"
   ],
   "ci_vs_live": "Consistent. GitHub shows 15.2k stars and 130-plus connectors.",
   "grades": {
    "incident": {
     "level": 0,
     "basis": "docs",
     "cite": "https://github.com/open-metadata/OpenMetadata",
     "note": "Incident management and governance workflows, freshness checks and alerts. Human-run incident workflow over alerts."
    },
    "quality": {
     "level": 0,
     "basis": "docs",
     "cite": "https://github.com/open-metadata/OpenMetadata",
     "note": "Tests, profiling, freshness checks and alerts. Collate's test-generating agents were not on the pages fetched (blog returned 403)."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://github.com/open-metadata/OpenMetadata",
     "note": "Not evidenced on the pages fetched."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://open-metadata.org/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": 3,
     "basis": "docs",
     "cite": "https://open-metadata.org/",
     "note": "Agent automation and AI workflows write metadata through the MCP server (read and write); Memory records corrections, decisions, approvals, feedback loops, and an auditable record of every change. Agent writes, approvals and an audit record on catalog changes: L3."
    },
    "xcloud": {
     "level": 1,
     "basis": "docs",
     "cite": "https://github.com/open-metadata/OpenMetadata",
     "note": "One graph over 130-plus connectors with cross-platform lineage. The graph explains across platforms; agentic writes across platforms are not separately evidenced."
    }
   }
  },
  {
   "id": "montecarlo",
   "name": "Monte Carlo",
   "scope": "Data and AI observability with a Monitoring Agent, a Troubleshooting Agent and an MCP and agent toolkit.",
   "url": "https://montecarlo.ai/",
   "category": "Observability incumbent",
   "publishes": {
    "price": "no",
    "soc2": "yes",
    "soc2_note": "trust.montecarlodata.com is linked from the footer (returned 403 to a plain fetch); the repository row records SOC 2 as published.",
    "licence": "Proprietary (apollo-agent collector is open source)",
    "mcp": "Yes; MCP and agent toolkit under Agentic Operations, reachable from Claude Code."
   },
   "funding": "$236M total; $1.6B valuation (2022).",
   "fetched": [
    "https://montecarlo.ai/",
    "https://docs.getmontecarlo.com/",
    "https://docs.getmontecarlo.com/docs/troubleshooting-agent"
   ],
   "ci_vs_live": "montecarlodata.com now redirects to montecarlo.ai. Pricing is request-only. Otherwise consistent.",
   "grades": {
    "incident": {
     "level": 1,
     "basis": "docs",
     "cite": "https://docs.getmontecarlo.com/docs/troubleshooting-agent",
     "note": "Works through hundreds of hypotheses across warehouse query logs, dbt, Airflow, Databricks Workflows and GitHub or GitLab PRs and highlights the most likely cause. Human triggers it and acts on it; no fix applied."
    },
    "quality": {
     "level": 1,
     "basis": "docs",
     "cite": "https://docs.getmontecarlo.com/",
     "note": "Monitoring plan and Monitoring Agent recommend monitors; anomalies alert. Recommends; does not remediate."
    },
    "schema": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.getmontecarlo.com/",
     "note": "Schema-change monitors exist in the product but were not on the pages fetched this pass. Not evidenced; contest with the docs URL."
    },
    "pipeline": {
     "level": null,
     "basis": "none",
     "cite": "https://montecarlo.ai/",
     "note": "Not evidenced."
    },
    "catalog": {
     "level": null,
     "basis": "none",
     "cite": "https://montecarlo.ai/",
     "note": "Not evidenced."
    },
    "xcloud": {
     "level": 1,
     "basis": "docs",
     "cite": "https://docs.getmontecarlo.com/docs/troubleshooting-agent",
     "note": "Root cause traced across warehouse, dbt, Airflow and code hosting in one investigation. Explains across systems."
    }
   }
  },
  {
   "id": "datafold",
   "name": "Datafold",
   "scope": "Data Diff, CI deployment testing on PRs, monitors, a Migration Agent delivered as a service, and a Data Knowledge Graph served over MCP.",
   "url": "https://www.datafold.com/",
   "category": "Adjacent (diff, CI, migration)",
   "publishes": {
    "price": "no",
    "soc2": "yes",
    "soc2_note": "SOC 2 and HIPAA compliance stated on the homepage; security.datafold.com returns 200.",
    "licence": "Proprietary (data-diff OSS repo archived)",
    "mcp": "Yes; Data Diff, monitors and the Data Knowledge Graph exposed via MCP for Claude Code, Cursor and Windsurf."
   },
   "funding": "About $26M total (NEA, Amplify, YC).",
   "fetched": [
    "https://www.datafold.com/",
    "https://docs.datafold.com/welcome",
    "https://www.datafold.com/pricing (redirects to contact)"
   ],
   "ci_vs_live": "Consistent. Pricing redirects to a contact form.",
   "grades": {
    "incident": {
     "level": null,
     "basis": "none",
     "cite": "https://docs.datafold.com/welcome",
     "note": "Not evidenced."
    },
    "quality": {
     "level": 0,
     "basis": "docs",
     "cite": "https://docs.datafold.com/welcome",
     "note": "Monitors send alerts when data diffs fall outside predefined ranges; ML anomaly detection. Alerting."
    },
    "schema": {
     "level": 1,
     "basis": "docs",
     "cite": "https://docs.datafold.com/welcome",
     "note": "Deployment testing diffs every PR against production at value level before merge. Explains what changed; no review verdict is drafted and nothing is applied."
    },
    "pipeline": {
     "level": 2,
     "basis": "docs",
     "cite": "https://docs.datafold.com/welcome",
     "note": "Migration Agent translates legacy code to the target with automated validation and humans reviewing what falls out, delivered as a guaranteed-outcome service. Drafts; humans review."
    },
    "catalog": {
     "level": 1,
     "basis": "docs",
     "cite": "https://docs.datafold.com/welcome",
     "note": "Data Knowledge Graph collects lineage, business logic, usage, BI connections and git history and serves it to agents over MCP. Explains; does not write documentation."
    },
    "xcloud": {
     "level": 1,
     "basis": "docs",
     "cite": "https://docs.datafold.com/welcome",
     "note": "Cross-database diffing compares datasets across warehouses at value level. Explains across platforms."
    }
   }
  },
  {
   "id": "dataworkers",
   "name": "Data Workers",
   "scope": "The Autonomous Agentic Data Platform: Data-Agents Swarm (20 specialised agents), Data Context Wizard (cross-cloud semantic and knowledge graph, provenance-stamped), Autonomous Data-Conductor (detect, diagnose, fix, review, verify with an autonomy dial), Spellbook Data Catalog (agent-written catalog and human control plane, preview). Runs inside Claude Code, Cursor, Codex CLI, OpenCode and any MCP client.",
   "url": "https://dataworkers.io/",
   "category": "Publisher of this index",
   "publishes": {
    "price": "yes",
    "soc2": "no",
    "soc2_note": "Not SOC 2 certified. llms.txt states architected for SOC 2 and that data stays inside your own infrastructure. No trust center.",
    "licence": "Apache-2.0 core (11 agents, 160-plus MCP tools in the community edition)",
    "mcp": "Yes; every agent is exposed as MCP tools; runs in Claude Code, Cursor, Codex CLI, OpenCode, Gemini and any MCP client."
   },
   "funding": "Early stage. No public customer references. Pre-revenue.",
   "fetched": [
    "https://dataworkers.io/",
    "https://dataworkers.io/llms.txt",
    "https://github.com/DataWorkersProject/dataworkers-claw-community"
   ],
   "ci_vs_live": "The community repository shows 12 stars and states that pipeline generation and model training require Pro; the community edition runs on in-memory stubs by default. Published pricing: free core, $7,500 pilot, from $1,000 per month, no metering, unlimited seats. The site carries a benchmark figure without a public run record; this index does not cite it (see Forthcoming).",
   "grades": {
    "incident": {
     "level": 3,
     "basis": "docs",
     "cite": "https://dataworkers.io/llms.txt",
     "note": "Agents draft the change, run it in a sandboxed dry-run, and hand it to a named human for approval; every change ships with a receipt (what changed, why, blast radius across downstream models and dashboards) and one-click rollback. Gate, dry-run and receipt are all stated. The Conductor's unattended mode (autonomy dial) would be L4 but no public page documents the policy under which changes are applied unattended, so L3.",
     "cite2": "https://dataworkers.io/product/autonomous-data-conductor/"
    },
    "quality": {
     "level": 3,
     "basis": "docs",
     "cite": "https://dataworkers.io/llms.txt",
     "note": "Quality monitoring agent detects; fixes go through the same propose, dry-run, approve, receipt path. No public evidence of unattended quality fixes."
    },
    "schema": {
     "level": 3,
     "basis": "docs",
     "cite": "https://dataworkers.io/blog/data-change-review-agent/",
     "note": "Data Change Review and Schema Evolution agents; the receipt carries blast radius across downstream models and dashboards; apply only after a named human approves. The strongest publicly documented lane."
    },
    "pipeline": {
     "level": 2,
     "basis": "docs",
     "cite": "https://github.com/DataWorkersProject/dataworkers-claw-community",
     "note": "The community edition drafts pipelines but states that pipeline generation requires Pro, and the Pro path is not publicly documented beyond the llms.txt claim. Graded on what a stranger can verify: propose."
    },
    "catalog": {
     "level": 2,
     "basis": "docs",
     "cite": "https://dataworkers.io/product/spellbook-data-catalog/",
     "note": "Spellbook is an agent-written catalog with a human control plane, in preview. Agents draft descriptions and the human plane approves; the approval and audit path for catalog writes is not yet documented publicly, and the product is preview. Propose."
    },
    "xcloud": {
     "level": 2,
     "basis": "docs",
     "cite": "https://github.com/DataWorkersProject/dataworkers-claw-community",
     "note": "Fifteen catalog connectors including Snowflake, BigQuery, Databricks, Glue, Hive Metastore and Iceberg feeding one provenance-stamped graph. Cross-cloud drafting is evidenced; a gated apply spanning two warehouses is not shown publicly. Propose, the same cap applied to Alkera, Altimate and Atlan."
    }
   }
  }
 ],
 "self_grading": {
  "headline": "How we graded ourselves",
  "body": "Data Workers was graded last, from dataworkers.io, llms.txt and the public GitHub repository only, by the same two rules that capped every other vendor: an approval gate has to be documented for L3, and unattended application has to come with a documented receipt and rollback for L4. We are not L4 anywhere. The autonomy dial on the Conductor describes an unattended mode, but no public page states the policy under which it applies changes, so it does not count. Pipeline build is graded propose because the community edition gates pipeline generation behind a paid tier a stranger cannot inspect. Spellbook is in preview and is graded propose. Cross-cloud scope is graded propose under the same cap applied to Alkera, Altimate and Atlan: connectors to three warehouses are evidenced, a gated change spanning two of them is not shown.",
  "not_claimed": "No customers, revenue, awards or third-party references are claimed. Data Workers is not SOC 2 certified. The site's benchmark figure is not cited here because no public run record accompanies it.",
  "lowest_rule": "Lowest lanes are listed by level, then alphabetically by lane name."
 },
 "forthcoming": {
  "headline": "Forthcoming from Data Workers",
  "body": "Data Workers intends to publish benchmark results with a public run record, and technical white papers on the receipt format and the cross-cloud context graph. None of that material is cited in this edition and no figure from it appears here. When published, it will be graded under the same rules as every other vendor's docs and the relevant cells will move only if the documents show the mechanism."
 }
}