{
  "tool": "datahub-compliance-posture",
  "generated_at": "2026-08-09T20:56:29Z",
  "run": {
    "run_id": "0406b48c-b438-4b2f-bb2d-d495c1870333",
    "generated_at": "2026-08-09T20:56:29Z",
    "tool_version": "0.1.0",
    "schema_version": "2.0.0",
    "catalog_fingerprint_sha256": "37cd51fd1dd2cce5d653aec6775ac92f6d7d05deaf3215118fc4691da3eff47b",
    "rulepack_manifest_sha256": "91cfaa84aa052371bca2b2efb2c35d2b0e908bf36fd61c34e50efd9c4ad0ddcf",
    "previous_run_urn": null
  },
  "report_pair_sha256": "d11579e72fd154923bcef96684dc92d8c4446e99bbc94dfe330e15d7db8d96a1",
  "catalog": {
    "label": "http://localhost:8080",
    "dataset_count": 9
  },
  "observations": {
    "backup_requirement_coverage": {
      "label": "Documented backup-requirement decision",
      "criterion": "The dataset has a controlled io.obsidiantek.dhcp.backupRequirement value: REQUIRED, NOT_REQUIRED, or CONDITIONAL.",
      "observation_id": "backup_requirement_coverage",
      "observed_count": 7,
      "not_observed_count": 2,
      "total": 9,
      "coverage_ratio": 0.7778,
      "observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_training_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_validation_set,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.clinical_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.encounter_events,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.patient_records,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.claims_billing,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.compliance_reporting,PROD)"
      ],
      "not_observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_inference_inputs,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.consent_preferences,PROD)"
      ]
    },
    "documentation_coverage": {
      "label": "Substantive dataset description",
      "criterion": "The trimmed dataset description is at least 20 characters long.",
      "observation_id": "documentation_coverage",
      "observed_count": 9,
      "not_observed_count": 0,
      "total": 9,
      "coverage_ratio": 1.0,
      "observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_inference_inputs,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_training_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_validation_set,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.clinical_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.consent_preferences,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.encounter_events,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.patient_records,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.claims_billing,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.compliance_reporting,PROD)"
      ],
      "not_observed_assets": []
    },
    "domain_assignment": {
      "label": "Domain assignment",
      "criterion": "The dataset has a non-empty DataHub domain assignment.",
      "observation_id": "domain_assignment",
      "observed_count": 8,
      "not_observed_count": 1,
      "total": 9,
      "coverage_ratio": 0.8889,
      "observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_inference_inputs,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_training_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_validation_set,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.clinical_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.consent_preferences,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.encounter_events,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.patient_records,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.claims_billing,PROD)"
      ],
      "not_observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.compliance_reporting,PROD)"
      ]
    },
    "lineage_presence": {
      "label": "Registered lineage",
      "criterion": "The dataset has at least one registered upstream or downstream lineage edge.",
      "observation_id": "lineage_presence",
      "observed_count": 8,
      "not_observed_count": 1,
      "total": 9,
      "coverage_ratio": 0.8889,
      "observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_inference_inputs,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_training_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_validation_set,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.clinical_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.encounter_events,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.patient_records,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.claims_billing,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.compliance_reporting,PROD)"
      ],
      "not_observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.consent_preferences,PROD)"
      ]
    },
    "ownership_coverage": {
      "label": "Assigned owner",
      "criterion": "The dataset has at least one assigned DataHub owner.",
      "observation_id": "ownership_coverage",
      "observed_count": 8,
      "not_observed_count": 1,
      "total": 9,
      "coverage_ratio": 0.8889,
      "observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_inference_inputs,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_training_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_validation_set,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.clinical_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.consent_preferences,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.encounter_events,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.patient_records,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.claims_billing,PROD)"
      ],
      "not_observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.compliance_reporting,PROD)"
      ]
    },
    "personal_information_retention_coverage": {
      "label": "Retention intent for identified personal information",
      "criterion": "Among datasets with a schema field carrying a recognized PII, PHI, personal, personal-data, or personal-information label, the dataset has a non-empty io.acryl.privacy.retentionTime structured property.",
      "observation_id": "personal_information_retention_coverage",
      "observed_count": 4,
      "not_observed_count": 3,
      "total": 7,
      "coverage_ratio": 0.5714,
      "observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_inference_inputs,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_validation_set,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.consent_preferences,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.patient_records,PROD)"
      ],
      "not_observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_training_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.clinical_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.encounter_events,PROD)"
      ]
    },
    "pii_tag_coverage": {
      "label": "Field sensitivity label",
      "criterion": "At least one schema field has a tag or glossary term whose normalized name includes one of these exact tokens: PHI, HIPAA, PCI, financial, PII, GDPR, personal, sensitive, or confidential; explicitly non-sensitive labels do not count.",
      "observation_id": "pii_tag_coverage",
      "observed_count": 7,
      "not_observed_count": 2,
      "total": 9,
      "coverage_ratio": 0.7778,
      "observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_inference_inputs,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_training_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_validation_set,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.clinical_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.consent_preferences,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.encounter_events,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.patient_records,PROD)"
      ],
      "not_observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.claims_billing,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.compliance_reporting,PROD)"
      ]
    },
    "retention_property_coverage": {
      "label": "Documented retention intent",
      "criterion": "The dataset has a non-empty io.acryl.privacy.retentionTime structured property.",
      "observation_id": "retention_property_coverage",
      "observed_count": 6,
      "not_observed_count": 3,
      "total": 9,
      "coverage_ratio": 0.6667,
      "observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_inference_inputs,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_validation_set,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.consent_preferences,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.patient_records,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.claims_billing,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.compliance_reporting,PROD)"
      ],
      "not_observed_assets": [
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.ai_training_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.clinical_features,PROD)",
        "urn:li:dataset:(urn:li:dataPlatform:showcase,aic.encounter_events,PROD)"
      ]
    }
  },
  "evidence_profiles": [
    {
      "profile": "aicm",
      "name": "CSA AI Controls Matrix (AICM)",
      "profile_version": "1.1.0",
      "standard_version": "AICM v1.1.0",
      "source_url": "https://cloudsecurityalliance.org/artifacts/ai-controls-matrix",
      "source_note": "Selected identifiers and titles are attributed to Cloud Security Alliance; interpretations and evidence statements are project-authored.",
      "objectives": [
        {
          "id": "DSP-03",
          "title": "Data Inventory",
          "interpretation": "A governed data inventory should make in-scope datasets discoverable, organized, and understandable.",
          "evidence_relevance": "Domain assignments and substantive dataset descriptions provide catalog-visible inventory context for auditor review.",
          "limitations": "These observations do not establish that every in-scope data resource is cataloged or that inventory records are accurate and current.",
          "observation_ids": [
            "domain_assignment",
            "documentation_coverage"
          ],
          "remediation": "Assign missing DataHub domains and add substantive dataset descriptions.",
          "datahub_surfaces": [
            "Domains",
            "Dataset descriptions"
          ]
        },
        {
          "id": "DSP-04",
          "title": "Data Classification",
          "interpretation": "Cataloged fields should carry explicit classification labels so reviewers can identify recorded sensitivity decisions.",
          "evidence_relevance": "A recognized sensitivity tag or directly assigned glossary term on at least one field provides catalog-visible evidence that selected data-classification decisions have been recorded for that dataset.",
          "limitations": "This observation does not establish classification completeness or correctness, coverage of data type or criticality, handling requirements, or operating effectiveness.",
          "observation_ids": [
            "pii_tag_coverage"
          ],
          "remediation": "Review unlabeled sensitivity candidates and apply confirmed field-level classifications through the governed catalog process.",
          "datahub_surfaces": [
            "Schema field tags",
            "Glossary terms"
          ]
        },
        {
          "id": "DSP-05",
          "title": "Data Flow Documentation",
          "interpretation": "Data movement should be documented so reviewers can trace material upstream and downstream relationships.",
          "evidence_relevance": "Registered DataHub lineage edges provide catalog-visible data-flow evidence.",
          "limitations": "Lineage presence does not establish that every transformation, transfer, system boundary, or external recipient is represented.",
          "observation_ids": [
            "lineage_presence"
          ],
          "remediation": "Register and verify upstream and downstream lineage for datasets with no catalog-visible edges.",
          "datahub_surfaces": [
            "Lineage"
          ]
        },
        {
          "id": "DSP-06",
          "title": "Data Ownership and Stewardship",
          "interpretation": "Governed data assets should have a catalog-visible accountable party.",
          "evidence_relevance": "Assigned DataHub owners provide direct metadata evidence that a dataset has a named accountable party.",
          "limitations": "Owner assignment does not establish role acceptance, authority, stewardship procedures, or operating effectiveness.",
          "observation_ids": [
            "ownership_coverage"
          ],
          "remediation": "Assign a technical or business owner to each dataset missing one.",
          "datahub_surfaces": [
            "Ownership"
          ]
        },
        {
          "id": "DSP-16",
          "title": "Data Retention and Deletion",
          "interpretation": "Retention intentions should be recorded in a structured form so lifecycle work can be identified and handed off.",
          "evidence_relevance": "A non-empty DataHub retention-time structured property provides evidence of documented retention intent.",
          "limitations": "The property does not prove that a period is legally appropriate, that deletion occurred, or that lifecycle enforcement is operating.",
          "observation_ids": [
            "retention_property_coverage"
          ],
          "remediation": "Record the reviewed retention period in io.acryl.privacy.retentionTime for datasets where it is absent.",
          "datahub_surfaces": [
            "Structured properties",
            "Forms"
          ]
        },
        {
          "id": "DSP-17",
          "title": "Sensitive Data Protection",
          "interpretation": "Sensitive-data review depends on catalog-visible identification of fields that may require protection.",
          "evidence_relevance": "An explicit sensitivity tag or glossary term on at least one field shows that field-level sensitivity labeling is present for that dataset.",
          "limitations": "This observation does not measure classification completeness, validate labels, or establish that any protection control is implemented.",
          "observation_ids": [
            "pii_tag_coverage"
          ],
          "remediation": "Review unlabeled sensitivity candidates and apply confirmed field-level labels through the governed catalog process.",
          "datahub_surfaces": [
            "Schema field tags",
            "Glossary terms"
          ]
        },
        {
          "id": "DSP-20",
          "title": "Data Provenance and Transparency",
          "interpretation": "Reviewers should be able to understand a dataset's stated purpose and trace its catalog-visible origin or downstream use.",
          "evidence_relevance": "Substantive descriptions and registered lineage provide complementary documentation and provenance evidence.",
          "limitations": "These observations do not establish end-to-end provenance completeness, source authenticity, transformation accuracy, or transparency to affected people.",
          "observation_ids": [
            "documentation_coverage",
            "lineage_presence"
          ],
          "remediation": "Document dataset purpose and source context, then register and verify material lineage edges.",
          "datahub_surfaces": [
            "Dataset descriptions",
            "Lineage"
          ]
        }
      ]
    },
    {
      "profile": "gdpr",
      "name": "GDPR",
      "profile_version": "1.0.0",
      "standard_version": "Regulation (EU) 2016/679",
      "source_url": "https://eur-lex.europa.eu/eli/reg/2016/679/oj/eng",
      "source_note": "Article identifiers anchor the official regulation; interpretations and evidence statements are project-authored and are not legal advice.",
      "objectives": [
        {
          "id": "Article 30",
          "title": "Records of processing activities",
          "interpretation": "A processing-activity review needs an accountable inventory with purpose, personal-data context, data flows, and retention intent.",
          "evidence_relevance": "Owners, domains, descriptions, sensitivity labels, lineage, and retention properties provide reusable catalog evidence for assembling or testing parts of a record of processing activities.",
          "limitations": "A DataHub dataset is not a processing activity. These observations do not establish lawful basis, purposes, data-subject categories, recipients, transfers, security measures, completeness, or applicability of Article 30.",
          "observation_ids": [
            "ownership_coverage",
            "domain_assignment",
            "documentation_coverage",
            "pii_tag_coverage",
            "lineage_presence",
            "retention_property_coverage"
          ],
          "remediation": "Improve the catalog evidence, then have privacy counsel or the accountable privacy team reconcile it with the authoritative record of processing activities.",
          "datahub_surfaces": [
            "Ownership",
            "Domains",
            "Dataset descriptions",
            "Schema field tags",
            "Glossary terms",
            "Lineage",
            "Structured properties"
          ]
        },
        {
          "id": "Article 5(1)(e)",
          "title": "Storage limitation",
          "interpretation": "Personal-data review should be able to identify the intended retention period associated with cataloged datasets.",
          "evidence_relevance": "A non-empty retention-time structured property provides reviewable evidence of documented retention intent.",
          "limitations": "The observation does not identify personal data, determine necessity, validate the period, account for exceptions, or prove deletion and enforcement.",
          "observation_ids": [
            "retention_property_coverage"
          ],
          "remediation": "Have the accountable privacy team review and record retention intent for relevant datasets, then verify enforcement outside DHCP.",
          "datahub_surfaces": [
            "Structured properties",
            "Forms"
          ]
        }
      ]
    },
    {
      "profile": "hipaa",
      "name": "HIPAA",
      "profile_version": "1.0.1",
      "standard_version": "45 CFR Parts 160 and 164",
      "source_url": "https://www.hhs.gov/hipaa/for-professionals/index.html",
      "source_note": "Citations anchor official HHS rules and guidance; interpretations and evidence statements are project-authored and are not legal advice.",
      "objectives": [
        {
          "id": "45 CFR 164.308(a)(1)(ii)(A)",
          "title": "Security risk-analysis scope",
          "interpretation": "Risk analysis begins with identifying all locations where electronic protected health information is created, received, maintained, or transmitted.",
          "evidence_relevance": "Catalog ownership, domain, documentation, field sensitivity labeling, and lineage observations can support scoping and data collection for ePHI risk analysis.",
          "limitations": "DHCP does not determine HIPAA applicability, identify all ePHI, assess threats or vulnerabilities, assign risk, evaluate safeguards, or perform the required risk analysis.",
          "observation_ids": [
            "ownership_coverage",
            "domain_assignment",
            "documentation_coverage",
            "pii_tag_coverage",
            "lineage_presence"
          ],
          "remediation": "Review catalog evidence and sensitivity candidates with the HIPAA security team, then incorporate confirmed systems and flows into the formal risk analysis.",
          "datahub_surfaces": [
            "Ownership",
            "Domains",
            "Dataset descriptions",
            "Schema field tags",
            "Glossary terms",
            "Lineage"
          ],
          "reference_url": "https://www.hhs.gov/hipaa/for-professionals/security/guidance/guidance-risk-analysis/index.html"
        },
        {
          "id": "45 CFR 164.502(b) / 164.514(d)",
          "title": "Minimum-necessary review support",
          "interpretation": "Reviewers need to identify categories of protected health information and their stated purpose before assessing minimum-necessary policies and access.",
          "evidence_relevance": "Field sensitivity labels and substantive dataset descriptions provide catalog evidence that can help scope minimum-necessary review.",
          "limitations": "These observations do not establish that data is PHI, whether an exception applies, who has access, the purpose of a use or disclosure, or whether minimum-necessary policies are effective.",
          "observation_ids": [
            "pii_tag_coverage",
            "documentation_coverage"
          ],
          "remediation": "Confirm PHI classifications and purposes with the privacy team, then review role-based access, information categories, exceptions, and disclosure policies in the authoritative systems.",
          "datahub_surfaces": [
            "Schema field tags",
            "Glossary terms",
            "Dataset descriptions"
          ],
          "reference_url": "https://www.hhs.gov/hipaa/for-professionals/privacy/guidance/minimum-necessary-requirement/index.html"
        }
      ]
    },
    {
      "profile": "iso27001",
      "name": "ISO/IEC 27001",
      "profile_version": "1.1.0",
      "standard_version": "ISO/IEC 27001:2022 Annex A",
      "source_url": "https://www.iso.org/standard/27001",
      "source_note": "Control identifiers are used for identification; titles, interpretations, and evidence statements are project-authored paraphrases.",
      "objectives": [
        {
          "id": "A.5.9",
          "title": "Inventory of information and associated assets",
          "interpretation": "Cataloged datasets should carry accountable ownership and sufficient context to support review of the information-asset inventory.",
          "evidence_relevance": "Dataset ownership, domain assignment, and substantive descriptions provide catalog-visible evidence for reviewing inventory records, accountability, and context.",
          "limitations": "These observations do not establish that the inventory covers all information and associated assets, that records are maintained over time, that ownership assignments are appropriate or current, or that the full control is satisfied.",
          "observation_ids": [
            "ownership_coverage",
            "domain_assignment",
            "documentation_coverage"
          ],
          "remediation": "Assign missing owners and domains, and add substantive dataset descriptions.",
          "datahub_surfaces": [
            "Ownership",
            "Domains",
            "Dataset descriptions"
          ]
        },
        {
          "id": "A.5.12",
          "title": "Information-classification records",
          "interpretation": "Cataloged datasets should expose reviewed field-level sensitivity classifications that support information-classification governance.",
          "evidence_relevance": "A recognized sensitivity tag or directly assigned glossary term on at least one field provides catalog-visible evidence that selected classification decisions have been recorded for that dataset.",
          "limitations": "This observation does not establish the organization's classification scheme or criteria, evaluate confidentiality, integrity or availability requirements, measure classification completeness, accuracy or currency, verify resulting handling controls, or satisfy the full control.",
          "observation_ids": [
            "pii_tag_coverage"
          ],
          "remediation": "Review unlabeled sensitivity candidates and record only authorized field-level classifications.",
          "datahub_surfaces": [
            "Schema field tags",
            "Glossary terms"
          ]
        },
        {
          "id": "A.5.13",
          "title": "Labelling of information",
          "interpretation": "Fields with confirmed sensitivity classifications should carry explicit catalog labels so reviewers can see where information labelling has been implemented.",
          "evidence_relevance": "A recognized sensitivity tag or directly assigned glossary term on at least one field provides catalog-visible evidence that field-level labelling is present for that dataset.",
          "limitations": "This observation does not establish an organization-wide labelling procedure, alignment with the adopted classification scheme, label completeness, accuracy or currency, resulting handling rules, or satisfaction of the full control.",
          "observation_ids": [
            "pii_tag_coverage"
          ],
          "remediation": "Review unlabeled sensitivity candidates and apply only confirmed field-level labels.",
          "datahub_surfaces": [
            "Schema field tags",
            "Glossary terms"
          ]
        }
      ]
    },
    {
      "profile": "iso42001",
      "name": "ISO/IEC 42001",
      "profile_version": "1.1.1",
      "standard_version": "ISO/IEC 42001:2023 Annex A",
      "source_url": "https://www.iso.org/standard/81230.html",
      "source_note": "Control identifiers are used for identification; titles, interpretations, and evidence statements are project-authored paraphrases.",
      "objectives": [
        {
          "id": "A.4.3",
          "title": "Documentation of data resources",
          "interpretation": "In-scope AI data resources should carry catalog documentation, provenance relationships, explicit field sensitivity classification, and documented retention intent.",
          "evidence_relevance": "Substantive descriptions can carry intended-use context, registered lineage provides provenance inputs, reviewed field labels expose selected information categories, and retention properties record retention intent.",
          "limitations": "These observations do not identify the complete AI-system data-resource boundary or establish the data's last-update date, machine-learning role, labeling process, quality, disposal policy or enforcement, bias, preparation, or the completeness and accuracy of any documentation.",
          "observation_ids": [
            "documentation_coverage",
            "lineage_presence",
            "pii_tag_coverage",
            "retention_property_coverage"
          ],
          "remediation": "For each in-scope AI data resource, document its intended use and known constraints, register material lineage, apply only reviewed sensitivity labels, and record reviewed retention intent.",
          "datahub_surfaces": [
            "Dataset descriptions",
            "Lineage",
            "Schema field tags",
            "Glossary terms",
            "Structured properties"
          ]
        },
        {
          "id": "A.7.5",
          "title": "Data provenance",
          "interpretation": "Cataloged AI data resources should expose traceable upstream or downstream relationships for provenance review.",
          "evidence_relevance": "Registered DataHub lineage edges provide machine-readable provenance evidence for cataloged datasets.",
          "limitations": "Lineage presence does not prove end-to-end completeness, source authenticity, transformation accuracy, or that a dataset is used by an AI system.",
          "observation_ids": [
            "lineage_presence"
          ],
          "remediation": "Register and independently review material lineage for in-scope AI data resources.",
          "datahub_surfaces": [
            "Lineage"
          ]
        }
      ]
    },
    {
      "profile": "soc2",
      "name": "SOC 2",
      "profile_version": "1.5.0",
      "standard_version": "2017 Trust Services Criteria",
      "source_url": "https://www.aicpa-cima.com/resources/landing/system-and-organization-controls-soc-suite-of-services",
      "source_note": "Criterion identifier is used for identification; title, interpretation, and evidence statements are project-authored paraphrases.",
      "objectives": [
        {
          "id": "CC2.1",
          "title": "Quality information supporting internal control",
          "interpretation": "Cataloged descriptions, lineage, and field-level sensitivity labels support review of asset records, documented data flows, and information classification.",
          "evidence_relevance": "Substantive descriptions, registered lineage, and field-level sensitivity tags or glossary terms provide catalog-visible evidence of asset records, data-flow documentation, and information classification.",
          "limitations": "These observations do not establish inventory completeness, lineage completeness, classification accuracy or completeness, information quality, communication to responsible parties, operating effectiveness, or satisfaction of the full criterion.",
          "observation_ids": [
            "documentation_coverage",
            "lineage_presence",
            "pii_tag_coverage"
          ],
          "remediation": "Add substantive dataset descriptions, register and verify material lineage, and apply only reviewed field-level sensitivity labels where evidence is absent.",
          "datahub_surfaces": [
            "Dataset descriptions",
            "Lineage",
            "Schema field tags",
            "Glossary terms"
          ]
        },
        {
          "id": "A1.2",
          "title": "Backup-requirement decisions",
          "interpretation": "In-scope datasets should carry a governed decision recording whether backup is required.",
          "evidence_relevance": "A controlled dataset Structured Property provides catalog-visible evidence that an accountable backup-requirement decision has been recorded.",
          "limitations": "This observation does not establish that the decision is appropriate or current, that backups run or succeed, that copies are complete, protected, immutable or off-site, that restoration is tested, or that recovery objectives and infrastructure satisfy the full criterion.",
          "observation_ids": [
            "backup_requirement_coverage"
          ],
          "remediation": "Have accountable owners review missing decisions and record REQUIRED, NOT_REQUIRED, or CONDITIONAL; verify backup operation and restoration outside DHCP.",
          "datahub_surfaces": [
            "Structured properties",
            "Forms",
            "Ownership",
            "Dataset documentation"
          ]
        },
        {
          "id": "C1.1",
          "title": "Confidential-information retention review",
          "interpretation": "Confidential-information review should be able to identify cataloged sensitive datasets, their stated purpose or context, and their documented retention intent.",
          "evidence_relevance": "Field-level sensitivity labels, substantive dataset descriptions, and retention Structured Properties provide complementary catalog evidence for identifying confidential-information retention decisions that need review.",
          "limitations": "These observations do not establish that every confidential dataset is identified, that labels or stated purposes are correct, that a retention period is necessary or legally appropriate, that exceptions are handled, or that deletion and lifecycle enforcement operate effectively.",
          "observation_ids": [
            "documentation_coverage",
            "pii_tag_coverage",
            "retention_property_coverage"
          ],
          "remediation": "Review sensitivity labels and stated purpose, then have accountable owners record missing retention intent and verify necessity, exceptions, deletion, and enforcement outside DHCP.",
          "datahub_surfaces": [
            "Dataset descriptions",
            "Schema field tags",
            "Glossary terms",
            "Structured properties",
            "Forms"
          ]
        },
        {
          "id": "C1.2",
          "title": "Confidential-information disposition identification",
          "interpretation": "Cataloged sensitivity labels and documented retention intent provide inputs for identifying confidential datasets that may require disposition review when their retention period ends.",
          "evidence_relevance": "Field-level sensitivity labels and retention Structured Properties provide complementary catalog evidence for assembling an exact confidential-information population whose disposition requirements need accountable review.",
          "limitations": "These observations do not establish that every confidential dataset is identified, that retention metadata is correct or machine-evaluable, that a retention period has ended, that holds or exceptions were considered, that destruction was authorized or performed, or that the full criterion is satisfied.",
          "observation_ids": [
            "pii_tag_coverage",
            "retention_property_coverage"
          ],
          "remediation": "Review sensitivity labels and record missing retention intent, then identify expiration candidates and verify holds, exceptions, authorization, deletion, and destruction evidence outside DHCP.",
          "datahub_surfaces": [
            "Schema field tags",
            "Glossary terms",
            "Structured properties",
            "Forms"
          ]
        },
        {
          "id": "PI1.2",
          "title": "System-input relationship records",
          "interpretation": "Cataloged lineage helps document declared upstream input relationships for in-scope datasets.",
          "evidence_relevance": "Registered upstream and downstream lineage edges provide reviewable catalog records of declared system-input relationships, with raw dataset counts and exact observed and not-observed populations.",
          "limitations": "This observation does not establish that every input or input activity was recorded, that records are complete, accurate, or timely, that transactions or data inputs were validated, that related procedures operated effectively, or that the full criterion is satisfied.",
          "observation_ids": [
            "lineage_presence"
          ],
          "remediation": "Register material upstream input relationships where lineage is absent, then have accountable reviewers verify the relationships and assess input-activity, completeness, accuracy, timeliness, and validation evidence outside DHCP.",
          "datahub_surfaces": [
            "Lineage"
          ]
        },
        {
          "id": "P4.2",
          "title": "Personal-information retention review",
          "interpretation": "Cataloged datasets identified as containing personal information should carry documented retention intent.",
          "evidence_relevance": "Reviewed PII, PHI, personal-data, or personal-information field labels define the catalog-identified population; a retention Structured Property on the same dataset records stated retention intent.",
          "limitations": "This observation does not establish that every personal-information dataset is identified, that labels or retention metadata are complete, accurate or current, that a period is appropriate or legally permitted, that exceptions are handled, that deletion or lifecycle enforcement operates, or that the full criterion is satisfied.",
          "observation_ids": [
            "personal_information_retention_coverage"
          ],
          "remediation": "Review personal-information identification separately, then have accountable owners record missing retention intent and verify purpose, necessity, legal requirements, exceptions, deletion, and lifecycle enforcement outside DHCP.",
          "datahub_surfaces": [
            "Schema field tags",
            "Glossary terms",
            "Structured properties",
            "Forms"
          ]
        }
      ]
    }
  ],
  "claim_boundary": "Observation coverage describes catalog-visible metadata only. Profile mappings explain possible relevance and limitations; DHCP does not determine conformity with any objective.",
  "disclaimer": "Supporting evidence derived from DataHub catalog metadata \u2014 not a legal or compliance determination, audit, assessment, or certification.",
  "auditor_analysis": {
    "text": "Documentation catalog observation coverage is the strongest signal in this digest, with all 9 datasets carrying substantive descriptions, and ownership, domain assignment, and lineage each observed on 8 of 9 datasets. The weakest observations are retention_property_coverage, present on 6 of 9 datasets, and personal_information_retention_coverage, present on only 4 of the 7 datasets identified as carrying personal-information labels. The catalog lookup confirms that aic.compliance_reporting is the single dataset missing ownership, domain assignment, and a PII tag simultaneously, and that aic.ai_training_features, aic.clinical_features, and aic.encounter_events are the datasets whose retention intent has not been recorded in DataHub.\n\nThe most consequential review implications cluster around retention and classification. For GDPR Article 5(1)(e), SOC 2 P4.2, and DSP-16, the 3 datasets without a retention property and the 3 personal-information-labeled datasets without a retention record represent the highest-priority catalog gaps; however, the absence of a structured property does not prove that no retention policy exists outside DataHub, and accountable privacy counsel must confirm whether any period is legally appropriate and whether deletion is enforced.",
    "generated_by": "langchain_agent",
    "grounded_in": "deterministic_observations",
    "context_provider": "DataHub Agent Context Kit",
    "context_access": "read_only",
    "validation": "accepted",
    "model_id": "claude-sonnet-4-6",
    "available_tool_count": 10,
    "tool_call_count": 1,
    "tool_error_count": 0,
    "round_count": 2,
    "repair_count": 0
  }
}