================================================================================
CONCATENATED DOCUMENT
================================================================================
Input Directory: data/d4d_individual/gpt5/AI_READI
Total Files: 7
Extensions: All
Recursive: False
================================================================================

TABLE OF CONTENTS
--------------------------------------------------------------------------------
  1. docs_aireadi_org_docs-2_d4d.yaml
  2. docs_aireadi_org_docs-2_d4d_metadata.yaml
  3. doi_row2_d4d.yaml
  4. doi_row2_d4d_metadata.yaml
  5. doi_row9_d4d.yaml
  6. fairhub_row10_d4d.yaml
  7. fairhub_row13_d4d.yaml
================================================================================

FILE: docs_aireadi_org_docs-2_d4d.yaml
PATH: data/d4d_individual/gpt5/AI_READI/docs_aireadi_org_docs-2_d4d.yaml
SIZE: 9113 bytes
--------------------------------------------------------------------------------

# D4D Metadata extracted from: docs_aireadi_org_docs-2_row8.txt
# Column: AI_READI
# Validation: Download ✅ success
# Relevance: ✅ relevant
# Generated: 2025-09-16 18:05:08

id: AI-READI Dataset
name: AI-READI Dataset
title: AI-READI Flagship Dataset of Type 2 Diabetes
description: >
  The AI-READI dataset consists of data collected from individuals with and
  without Type 2 Diabetes Mellitus (T2DM), harmonized across three data
  collection sites. It was designed with future AI/Machine Learning (AI/ML)
  studies in mind, including recruitment sampling procedures aimed at achieving
  approximately equal distribution of participants across diabetes severity, and
  a multi-domain data acquisition protocol (survey data, physical measurements,
  clinical data, imaging data, wearable device data, etc.). The goal is to
  better understand salutogenesis (the pathway from disease to health) in T2DM.
  Some non-sensitive data will be publicly downloadable upon agreement with a
  license defining permitted uses. The full dataset is accessible via a Data Use
  Agreement (DUA). Public data include survey data, blood and urine lab results,
  fitness activity levels, clinical measurements (e.g., monofilament and
  cognitive function testing), retinal images, ECG, blood sugar levels, and
  environmental variables (e.g., home air quality). Controlled-access data
  include 5-digit ZIP code, sex, race, ethnicity, genetic sequencing data, past
  health records, medications, and traffic and accident reports. Enrollment is
  ongoing; pilot and periodic releases may not achieve balanced distributions
  across groups. Documentation versions correspond to dataset versions and
  include domain-level acquisition and processing details.
language: en
keywords:
  - Type 2 Diabetes
  - T2DM
  - AI
  - Machine Learning
  - multimodal
  - harmonized
  - multi-site
  - survey data
  - clinical data
  - imaging data
  - wearable device data
  - time-series
  - ECG
  - retinal images
  - blood glucose
  - laboratory results
  - environmental data
  - FAIR principles
  - Healthsheet
is_tabular: mixed (tabular and non-tabular modalities)
purposes:
  - response: >
      Better understand salutogenesis (pathway from disease to health) in T2DM
      using a harmonized, multi-domain dataset designed for AI/ML research.
tasks:
  - response: >
      Enable downstream AI/ML analyses across survey, clinical, imaging,
      wearable, and environmental domains related to T2DM.
addressing_gaps:
  - response: >
      Provide a harmonized, multi-site, multi-domain dataset enabling AI/ML
      analyses not feasible with existing sources (e.g., claims or EHR alone),
      with recruitment targeting approximately equal distribution by diabetes
      severity.
instances:
  - representation: Individuals (participants) with and without Type 2 Diabetes Mellitus (T2DM)
    instance_type: Participants and their multi-domain measurements
    data_type: >
      Survey responses; physical and clinical measurements; blood and urine lab
      results; imaging (retinal); physiological signals (ECG); wearable device
      time-series; blood glucose levels; environmental sensor data (e.g., home
      air quality).
subsets:
  - name: Public dataset
    description: >
      Includes survey data, blood and urine lab results, fitness activity
      levels, clinical measurements (e.g., monofilament and cognitive function
      testing), retinal images, ECG, blood sugar levels, and environmental
      variables (e.g., home air quality). Available for public download upon
      agreement with a license defining permitted uses.
  - name: Controlled-access dataset
    description: >
      Includes 5-digit ZIP code, sex, race, ethnicity, genetic sequencing data,
      past health records, medications, and traffic and accident reports.
      Accessible via a Data Use Agreement (DUA).
sampling_strategies:
  - is_sample:
      - Yes (recruited participants across three data collection sites)
  - is_random:
      - No (targeted recruitment to balance diabetes severity)
  - source_data:
      - Participants enrolled via three data collection sites
  - is_representative:
      - Designed for approximate balance by diabetes severity; broader representativeness not claimed
  - representative_verification:
      - Balance targeted during recruitment; enrollment ongoing so pilot release may not achieve balance
  - strategies:
      - Recruitment sampling to achieve approximately equal distribution across diabetes severity
subpopulations:
  - identification:
      - With and without T2DM
      - Diabetes severity strata
    distribution:
      - Recruitment aimed at approximately equal distribution by diabetes severity
      - Pilot and periodic releases may not achieve full balance due to ongoing enrollment
anomalies:
  - description:
      - Ongoing enrollment means early releases may exhibit unbalanced distributions across groups.
external_resources:
  - external_resources:
      - Dataset landing page on the FAIRhub data portal (documentation complements the landing page)
      - Documentation sections with domain-specific standards, metadata, file formats, and example outputs
    archival:
      - Documentation versions correspond to dataset versions
    restrictions:
      - Full dataset requires a Data Use Agreement; some data publicly available under license
confidential_elements:
  - description:
      - Contains protected health information elements under controlled access (e.g., past health records).
content_warnings: []
sensitive_elements:
  - description:
      - Genetic sequencing data (controlled)
      - Past health records and medications (controlled)
      - Demographics (sex, race, ethnicity) and 5-digit ZIP code (controlled)
is_deidentified:
  description:
    - Public data are released under licensing terms; sensitive elements are held under controlled access via DUA to protect participant privacy.
acquisition_methods:
  - description:
      - Harmonized, multi-domain data acquisition across three collection sites
      - Data collected via surveys, clinical exams, imaging devices, wearable sensors, and environmental monitors
    was_directly_observed: Yes (physical/clinical measurements, imaging, wearable, environmental sensors)
    was_reported_by_subjects: Yes (survey data)
    was_inferred_derived: Unspecified
    was_validated_verified: >
      Harmonization across sites; domain-specific validation details provided in the documentation.
collection_mechanisms:
  - description:
      - Hardware devices and sensors (e.g., retinal imaging, ECG, wearables, environmental monitors)
      - Clinical procedures (e.g., monofilament and cognitive testing)
      - Software-driven data capture and manual curation as needed
data_collectors:
  - description:
      - Three data collection sites were involved in recruitment and data acquisition.
collection_timeframes:
  - description:
      - Pilot study phase data included; enrollment is ongoing; periodic data releases planned.
preprocessing_strategies:
  - description:
      - Domain-specific processing and harmonization described in the Dataset Documentation (file formats, standards, metadata, example outputs).
cleaning_strategies:
  - description:
      - Harmonization and processing across three sites; details provided per domain in the documentation.
labeling_strategies:
  - description:
      - Domain-specific labeling/annotation where applicable (e.g., clinical test outputs, imaging outputs), as described in the documentation.
existing_uses: []
use_repository: []
other_tasks: []
future_use_impacts:
  - description:
      - Early-release imbalance across groups may affect AI/ML model performance and fairness; users should account for group balance and distribution shifts.
      - Sensitive elements must be handled under DUA to mitigate privacy risks.
discouraged_uses:
  - description:
      - Not specified; users must adhere to license terms and the Data Use Agreement.
distribution_formats:
  - description:
      - Public dataset downloadable upon agreement with a license
      - Full dataset available via controlled access (DUA)
      - Multiple modalities spanning tabular data, images, and time-series; file formats and standards are documented per domain
distribution_dates: []
license_and_use_terms:
  description:
    - Public data are available for download upon agreement with a license defining permitted uses.
    - Full dataset access is contingent on entering into a Data Use Agreement (controlled access).
ip_restrictions: []
regulatory_restrictions: []
maintainers:
  - description:
      - AI-READI project team (see Documentation site Contact Us and GitHub references)
errata: []
updates:
  description:
    - Periodic updates to data releases are planned as enrollment proceeds; documentation versions align with dataset versions.
retention_limit: {}
version_access:
  description:
    - Separate documentation versions align to dataset versions (e.g., v1.0.0, v2.0.0); users can navigate between versions via the documentation site.
extension_mechanism: {}

================================================================================

FILE: docs_aireadi_org_docs-2_d4d_metadata.yaml
PATH: data/d4d_individual/gpt5/AI_READI/docs_aireadi_org_docs-2_d4d_metadata.yaml
SIZE: 1704 bytes
--------------------------------------------------------------------------------

extraction_metadata:
  timestamp: '2025-09-17T01:05:08.079063Z'
  extraction_id: b98c2be66bfa
input_document:
  filename: docs_aireadi_org_docs-2_row8.txt
  relative_path: docs_aireadi_org_docs-2_row8.txt
  format: txt
  size_bytes: 4226
  sha256_hash: 4c058272f5cb87258523a418d7bdc99cfb402fb9c1f2eeadbbaab7d07828a045
  project_column: AI_READI
output_document:
  filename: docs_aireadi_org_docs-2_d4d.yaml
  relative_path: docs_aireadi_org_docs-2_d4d.yaml
  format: yaml
datasheets_schema:
  version: 1.0.0
  url: https://raw.githubusercontent.com/monarch-initiative/ontogpt/main/src/ontogpt/templates/data_sheets_schema.yaml
  retrieved_at: '2025-09-17T01:05:08.079263Z'
d4d_agent:
  version: 1.0.0
  implementation: pydantic_ai
  wrapper: validated_d4d_wrapper.py
  wrapper_version: 2.0.0
llm_model:
  provider: openai
  model_name: openai:gpt-5
  model_version: gpt-5
  temperature: null
  max_tokens: null
validation_results:
  download_validation:
    success: true
    file_size: 4226
    content_type: .txt
  relevance_validation:
    success: true
    score: 8
    keywords_found:
    - ai-readi
    - diabetes
    - fairhub
    project_indicators:
    - type 2 diabetes
processing_environment:
  platform: Darwin
  python_version: 3.13.4
  processor_architecture: arm64
reproducibility:
  command: python validated_d4d_wrapper.py -i ../downloads_by_column -o ../data/extracted_by_column
  environment_variables:
    OPENAI_API_KEY: required
    ANTHROPIC_API_KEY: not_set
  random_seed: null
provenance:
  extraction_performed_by: validated_d4d_wrapper
  extraction_requested_at: '2025-09-17T01:05:08.079299Z'
  git_commit: null
  notes: D4D extraction using GPT-5 model with validation checks


================================================================================

FILE: doi_row2_d4d.yaml
PATH: data/d4d_individual/gpt5/AI_READI/doi_row2_d4d.yaml
SIZE: 680 bytes
--------------------------------------------------------------------------------

# D4D Metadata extracted from: doi_row2.json
# Column: AI_READI
# Validation: Download ✅ success
# Relevance: ✅ relevant
# Generated: 2025-09-16 18:06:24

id: https://aireadi.org/publications
page: https://aireadi.org/publications
keywords:
  - AI_READI
  - publications
resources:
  - id: doi:10.1038/s42255-024-01165-x
    doi: doi:10.1038/s42255-024-01165-x
    page: https://doi.org/10.1038/s42255-024-01165-x
    keywords:
      - AI_READI
      - publications
      - DOI
  - id: doi:10.1136/bmjopen-2024-097449
    doi: doi:10.1136/bmjopen-2024-097449
    page: https://doi.org/10.1136/bmjopen-2024-097449
    keywords:
      - AI_READI
      - publications
      - DOI

================================================================================

FILE: doi_row2_d4d_metadata.yaml
PATH: data/d4d_individual/gpt5/AI_READI/doi_row2_d4d_metadata.yaml
SIZE: 1577 bytes
--------------------------------------------------------------------------------

extraction_metadata:
  timestamp: '2025-09-17T01:06:24.131832Z'
  extraction_id: 27b918778483
input_document:
  filename: doi_row2.json
  relative_path: doi_row2.json
  format: json
  size_bytes: 243
  sha256_hash: 8713aa46cfac35f60b4bab105a75a8a0e99bace9f192664d6775cc0ddde50d20
  project_column: AI_READI
output_document:
  filename: doi_d4d.yaml
  relative_path: doi_d4d.yaml
  format: yaml
datasheets_schema:
  version: 1.0.0
  url: https://raw.githubusercontent.com/monarch-initiative/ontogpt/main/src/ontogpt/templates/data_sheets_schema.yaml
  retrieved_at: '2025-09-17T01:06:24.132454Z'
d4d_agent:
  version: 1.0.0
  implementation: pydantic_ai
  wrapper: validated_d4d_wrapper.py
  wrapper_version: 2.0.0
llm_model:
  provider: openai
  model_name: openai:gpt-5
  model_version: gpt-5
  temperature: null
  max_tokens: null
validation_results:
  download_validation:
    success: true
    file_size: 243
    content_type: .json
  relevance_validation:
    success: true
    score: 1
    keywords_found:
    - aireadi
    project_indicators: []
processing_environment:
  platform: Darwin
  python_version: 3.13.4
  processor_architecture: arm64
reproducibility:
  command: python validated_d4d_wrapper.py -i ../downloads_by_column -o ../data/extracted_by_column
  environment_variables:
    OPENAI_API_KEY: required
    ANTHROPIC_API_KEY: not_set
  random_seed: null
provenance:
  extraction_performed_by: validated_d4d_wrapper
  extraction_requested_at: '2025-09-17T01:06:24.132565Z'
  git_commit: null
  notes: D4D extraction using GPT-5 model with validation checks


================================================================================

FILE: doi_row9_d4d.yaml
PATH: data/d4d_individual/gpt5/AI_READI/doi_row9_d4d.yaml
SIZE: 311 bytes
--------------------------------------------------------------------------------

# D4D Metadata extracted from: doi_row9.json
# Column: AI_READI
# Validation: Download ✅ success
# Relevance: ⚠️  limited relevance
# Generated: 2025-10-30 19:47:22

id: "doi:10.5281/zenodo.10642459"
doi: "doi:10.5281/zenodo.10642459"
page: "https://doi.org/10.5281/zenodo.10642459"
keywords:
  - AI_READI

================================================================================

FILE: fairhub_row10_d4d.yaml
PATH: data/d4d_individual/gpt5/AI_READI/fairhub_row10_d4d.yaml
SIZE: 395 bytes
--------------------------------------------------------------------------------

# D4D Metadata extracted from: fairhub_row10.json
# Column: AI_READI
# Validation: Download ✅ success
# Relevance: ✅ relevant
# Generated: 2025-10-30 19:48:08

id: "https://fairhub.io/datasets/2"
name: FAIRhub Dataset
title: FAIRhub Dataset 2
description: "Dataset record on FAIRhub with identifier 2. Project category: AI_READI."
page: "https://fairhub.io/datasets/2"
keywords:
  - AI_READI

================================================================================

FILE: fairhub_row13_d4d.yaml
PATH: data/d4d_individual/gpt5/AI_READI/fairhub_row13_d4d.yaml
SIZE: 272 bytes
--------------------------------------------------------------------------------

# D4D Metadata extracted from: fairhub_row13.json
# Column: AI_READI
# Validation: Download ✅ success
# Relevance: ✅ relevant
# Generated: 2025-10-30 19:46:18

id: "https://fairhub.io/datasets/2"
page: "https://fairhub.io/datasets/2"
keywords:
  - AI_READI
  - FAIRhub