diff --git a/.github/workflows/test-synthetic.yaml b/.github/workflows/test-synthetic.yaml new file mode 100644 index 0000000..561fe4a --- /dev/null +++ b/.github/workflows/test-synthetic.yaml @@ -0,0 +1,61 @@ +name: Test Synthetic Corpus + +on: + pull_request: + paths: + - "synthetic/**" + - ".github/workflows/test-synthetic.yaml" + push: + branches: [main] + paths: + - "synthetic/**" + - ".github/workflows/test-synthetic.yaml" + +jobs: + test-synthetic: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.12" + + # The corpus is a build product, not a committed artifact: it is + # regenerated here and checked, rather than diffed against something + # stored in the repo. + - name: Generate raw tables + working-directory: synthetic + run: python generate.py + + - name: Generate transformation specs + working-directory: synthetic + run: python specs.py + + - name: Validate distributions and invariants + working-directory: synthetic + run: python validate.py + + - name: Check specs are well-formed YAML + working-directory: synthetic + run: | + pip install --quiet pyyaml + python - <<'PY' + import glob + import sys + + import yaml + + bad = [] + paths = sorted(glob.glob("specs/*/*.yaml")) + for path in paths: + try: + yaml.safe_load(open(path)) + except yaml.YAMLError as exc: + bad.append(f"{path}: {exc}") + print(f"{len(paths) - len(bad)}/{len(paths)} specs parse") + if bad: + print("\n".join(bad)) + sys.exit(1) + PY diff --git a/pyproject.toml b/pyproject.toml index 72e2269..f220d77 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -33,6 +33,7 @@ target-version = "py312" line-length = 120 include = [ "api/**/*.py", + "synthetic/**/*.py", ] [tool.ruff.lint] diff --git a/synthetic/.gitignore b/synthetic/.gitignore new file mode 100644 index 0000000..4bfca6f --- /dev/null +++ b/synthetic/.gitignore @@ -0,0 +1,7 @@ +# Generated corpus — reproducible from SEED, distributed as release assets +data/ +output/ +# sample/ is committed on purpose — see README +# Vendored upstream schema, fetched not committed +bdchm.yaml +__pycache__/ diff --git a/synthetic/README.md b/synthetic/README.md new file mode 100644 index 0000000..ea4fec3 --- /dev/null +++ b/synthetic/README.md @@ -0,0 +1,134 @@ +# Synthetic corpus + +Two fictional cohorts, structurally faithful to harmonized BDC data, for teams +building against the portal without touching participant data. + +Nothing here derives from real participants. Every coded value — CURIEs, enum +members, units — is taken from RTI's `priority_variables_transform` specs or +from BDCHM itself, so the corpus resolves against the same vocabulary as real +harmonized data. + +## Why it goes through dm-bip + +The generator emits **dbGaP-style raw tables**, not harmonized output, and those +are transformed by dm-bip against BDCHM. Emitting BDCHM directly would be a +second implementation of the transformation, free to drift from the real one in +ways nobody would notice until a portal built on it met real data. + +``` +generate.py -> data/raw/*.txt.gz -> dm-bip map-data -> harmonized BDCHM +specs.py -> specs/*/*.yaml -> ^ +``` + +## Running it + +```bash +python generate.py # raw tables, per study +python specs.py # BDCHM-targeted transformation specs +python validate.py # distributions and invariants +``` + +After the pipeline has run: + +```bash +python schema.py --study study_one # JSONL -> Parquet, typed from BDCHM +``` + +The pipeline emits YAML and JSONL side by side — YAML is what makes the +published corpus readable, JSONL is the machine-facing form with the same +nesting. `schema.py` reads BDCHM for leaf types and the transformation specs for +which slots nest, because the model alone does not decide that: BDCHM gives +`associated_participant` a range of `Participant`, but the spec materialises it +as a uuid5 string while `value_quantity` is nested inline. + +It also reports where the model and the data disagree on cardinality. That is +not hypothetical — BDCHM declares `identity` multivalued and every +transformation spec, RTI's included, emits a scalar. + +Then, from a dm-bip checkout: + +```bash +SYNTH=/path/to/synthetic +make pipeline CONFIG=$SYNTH/pipeline/example_study_one.mk \ + SYNTH_DIR=$SYNTH SYNTH_OUTPUT_DIR=$SYNTH/output/study_one +``` + +BDCHM is fetched rather than vendored, pinned to a release so an upstream change +is a deliberate bump here rather than a silent change in what the pipeline +produces: + +```bash +./fetch-bdchm.sh # currently v1.3.0 +``` + +## What is and isn't committed + +The generator is committed. The corpus is not: it is 14MB of YAML, deterministic +given `SEED`, and would produce a 14MB diff every time the seed or the model +changed. Built corpora are distributed as release assets — 3MB compressed, with +a stable URL and a version, which is what a consuming team needs anyway. + +`sample/` **is** committed — 40KB of hand-selected records covering every +structural feature the corpus claims. It exists so a reviewer, or a team +deciding whether this is the reference data they want, can see the output shape +without running the pipeline. Regenerate it with `python sample.py` after a run. + +## The cohorts + +| | Example Study One | Example Study Two | +|---|---|---| +| Participants | 500 | 500 | +| Visits | 3 | 5 | +| Sex skew | slightly male | slightly female | +| Race | 50/30/10/10 white, black, Asian, American Indian | 60/20/10/10 white, black, Middle Eastern, Native Hawaiian | +| BMI | continuous | categorical | + +5% of Study Two's participants are the same individuals as in Study One. They +share a `dbGaP_Subject_ID`, so harmonization resolves them to one `Person` with +two `Participant` records — the same way real cross-study participation appears. + +One visit per participant is `TELEHEALTH`; the rest are `STUDY_SITE_VISIT`. +Deceased participants stop attending, so nobody is measured after they die. + +## Known shapes that surprise people + +**`cause_of_death` is present for living participants**, multivalued, with a null +cause: + +```yaml +cause_of_death: +- cause: null + order: null + id: 73e3cc59-... +vital_status: OMOP:4230556 +``` + +Test `vital_status`, not the presence of `cause_of_death`. + +This is not an artifact of the synthetic corpus — it is how the real +transformation behaves. RTI's MESA spec derives `cause_of_death` the same way, +with `cause` set to `None` for the living and no mechanism to suppress the +object, because `ClassDerivation` in linkml-map has no conditional emission +(see linkml/linkml-map#187). Real harmonized BDC data therefore carries the same +shape, and the corpus reproduces it deliberately. A portal built against a +tidied-up version would break on real data. + +## Structural features exercised + +Beyond the clinical values, the corpus is meant to exercise the shapes a portal +has to render: + +- `MeasurementObservationSet` with nested systolic and diastolic observations +- `Quantity.operator` for results censored below an assay's detection limit +- `Assay` with `lower_limit_of_detection` / `upper_limit_of_detection` (HDL) +- `associated_assay` referencing a CBC instance (WBC) +- `qualifier` marking a value as an average (BUN) +- `Condition.relationship_to_participant` for family history +- `associated_evidence` on study-record-sourced conditions +- `exposure_status` distinguishing absent from present drug exposures +- Continuous and categorical presentations of the same concept across cohorts + +Coverage against BDCHM's full slot inventory — the percentage of slots appearing +at least once — is intended as the acceptance number for broadening the corpus +beyond this first pass. **That measurement is not built yet.** This pass covers +the brief only. diff --git a/synthetic/fetch-bdchm.sh b/synthetic/fetch-bdchm.sh new file mode 100755 index 0000000..cde993d --- /dev/null +++ b/synthetic/fetch-bdchm.sh @@ -0,0 +1,35 @@ +#!/usr/bin/env bash +# Fetch the pinned BDCHM schema used as the transformation target. +# +# Pinned by commit rather than by tag: git tags are mutable and can be +# retargeted, so a tag alone would let the schema change underneath us. The +# checksum then verifies the bytes, covering the case where the fetch itself +# returns something unexpected. +# +# To bump: change all three values together and re-run the pipeline. +set -euo pipefail + +BDCHM_VERSION="v1.3.0" +BDCHM_COMMIT="84222624fb550e47ce7ec3f6c4a3754d80a1cd16" +BDCHM_SHA256="01af15d50ba1ce3929344a30698cf83776fc10f19902b392cc1fcf7469c10029" + +REPO="RTIInternational/NHLBI-BDC-DMC-HM" +URL="https://raw.githubusercontent.com/${REPO}/${BDCHM_COMMIT}/src/bdchm/schema/bdchm.yaml" + +DEST="$(cd "$(dirname "$0")" && pwd)/bdchm.yaml" +TMP="$(mktemp)" +trap 'rm -f "$TMP"' EXIT + +curl -fsSL "$URL" -o "$TMP" + +ACTUAL="$(sha256sum "$TMP" | cut -d' ' -f1)" +if [ "$ACTUAL" != "$BDCHM_SHA256" ]; then + echo "Checksum mismatch for BDCHM ${BDCHM_VERSION} (${BDCHM_COMMIT})" >&2 + echo " expected ${BDCHM_SHA256}" >&2 + echo " actual ${ACTUAL}" >&2 + exit 1 +fi + +mv "$TMP" "$DEST" +trap - EXIT +echo "Fetched BDCHM ${BDCHM_VERSION} (${BDCHM_COMMIT:0:12}) -> ${DEST}" diff --git a/synthetic/generate.py b/synthetic/generate.py new file mode 100644 index 0000000..0131ac2 --- /dev/null +++ b/synthetic/generate.py @@ -0,0 +1,157 @@ +# ruff: noqa: S311 +""" +Emit the synthetic corpus as dbGaP-style raw tables. + +These are inputs to dm-bip, not harmonized output. Running them through the +pipeline with BDCHM as the target schema is what makes the result structurally +identical to real harmonized data, rather than a second implementation of the +transformation that can quietly drift. + + python synthetic/generate.py [--out DIR] +""" + +import argparse +import gzip +from pathlib import Path + +import population as pop + +CITATION = "Synthetic corpus for portal development. Not derived from participant data." + +# Column layouts. The phv accessions are fictional but well-formed, and are +# what the transformation specs reference. +LAYOUTS = { + "subject": ( + ["dbGaP_Subject_ID", "SUBJECT_ID", "SEX", "RACE", "ETHNICITY", "AGE", "VITAL_STATUS", "DEATH_CAUSE"], + 7, + ), + "visit": (["dbGaP_Subject_ID", "SUBJECT_ID", "VISIT_NUM", "VISIT_TYPE", "AGE_DAYS"], 4), + "clinical": ( + ["dbGaP_Subject_ID", "SUBJECT_ID", "VISIT_NUM", "HEIGHT_CM", "WEIGHT_KG", "BMI", "SBP", "DBP"], + 7, + ), + "labs": ( + ["dbGaP_Subject_ID", "SUBJECT_ID", "VISIT_NUM", "HDL", "HDL_OPERATOR", "BUN", "WBC"], + 6, + ), + "conditions": ( + [ + "dbGaP_Subject_ID", + "SUBJECT_ID", + "HEART_FAILURE", + "HF_CONCEPT", + "FAM_STROKE", + "FS_CONCEPT", + "FS_RELATIVE", + "HYPERTENSION", + "HEART_ATTACK", + "HA_CONCEPT", + "HA_SOURCE", + ], + 10, + ), + "meds": (["dbGaP_Subject_ID", "SUBJECT_ID", "CCB_STATUS", "CCB_CONCEPT"], 3), +} + + +def phv_ids(study, table, count): + """Fictional but well-formed phv accessions, unique per study and table.""" + base = int(study.phs[-3:]) * 100000 + int(study.tables[table][-3:]) * 100 + return [f"phv{base + i:08d}.v1" for i in range(count)] + + +def write_table(out_dir, study, table, rows): + """Write one table in dbGaP raw format, returning its path.""" + columns, phv_count = LAYOUTS[table] + pht = study.tables[table] + path = out_dir / f"{study.phs}.v1.{pht}.v1.p1.c1.ex0_1s.HMB.txt.gz" + + with gzip.open(path, "wt", newline="") as fh: + fh.write(f"# Study accession: {study.phs}.v1.p1\n") + fh.write(f"# Table accession: {pht}\n") + fh.write("# Consent group: All subjects\n") + fh.write(f"# Citation: {CITATION}\n") + fh.write("#\n") + fh.write("##\t" + "\t".join(phv_ids(study, table, phv_count)) + "\n") + fh.write("\t".join(columns) + "\n") + fh.write("\n") + for row in rows: + fh.write("\t".join("" if c is None else str(c) for c in row) + "\n") + return path + + +def build_rows(participants, study): + """Shape the population into the six per-study tables.""" + rows = {k: [] for k in LAYOUTS} + + for p in (x for x in participants if x.study is study): + gid, sid = p.person.dbgap_id, p.subject_id + person = p.person + + rows["subject"].append([ + gid, sid, person.sex, person.race, person.ethnicity, + person.baseline_age, + 1 if person.deceased else 0, + person.death_cause, + ]) + + for visit in p.visits: + rows["visit"].append([gid, sid, visit.number, visit.category, visit.age_days]) + rows["clinical"].append([ + gid, sid, visit.number, + visit.height_cm, visit.weight_kg, + visit.bmi_category if study.bmi_categorical else visit.bmi, + visit.systolic, visit.diastolic, + ]) + rows["labs"].append([ + gid, sid, visit.number, + visit.hdl, visit.hdl_operator, visit.bun, visit.wbc, + ]) + + c = p.conditions + rows["conditions"].append([ + gid, sid, + c["heart_failure"]["status"], c["heart_failure"]["concept"], + c["family_stroke"]["status"], c["family_stroke"]["concept"], + c["family_stroke"]["relationship"], + c["hypertension"]["status"], + c["heart_attack"]["status"], c["heart_attack"]["concept"], + "STUDY_RECORD" if c["heart_attack"]["from_study_record"] else "SELF_REPORT", + ]) + + rows["meds"].append([ + gid, sid, + "PRESENT" if p.ccb else "ABSENT", + p.ccb, + ]) + + return rows + + +def main(): + """Generate the raw tables for both studies.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--out", + type=Path, + default=Path(__file__).parent / "data" / "raw", + help="directory to write the dbGaP-style tables into", + ) + args = parser.parse_args() + + _, participants = pop.build() + + written = [] + for study, slug in ((pop.STUDY_ONE, "study_one"), (pop.STUDY_TWO, "study_two")): + out_dir = args.out / slug + out_dir.mkdir(parents=True, exist_ok=True) + rows = build_rows(participants, study) + for table in LAYOUTS: + written.append((study.name, table, write_table(out_dir, study, table, rows[table]), len(rows[table]))) + + for name, table, path, count in written: + print(f"{name:20} {table:12} {count:6} rows {path.name}") + + +if __name__ == "__main__": + main() diff --git a/synthetic/pipeline/example_study_one.mk b/synthetic/pipeline/example_study_one.mk new file mode 100644 index 0000000..336a311 --- /dev/null +++ b/synthetic/pipeline/example_study_one.mk @@ -0,0 +1,19 @@ +# Pipeline config for the synthetic Example Study One cohort. +# Target schema is BDCHM itself, so the harmonized output is structurally +# identical to real harmonized data rather than a simplified stand-in. + +DM_RAW_SOURCE := $(SYNTH_DIR)/data/raw/study_one +DM_SCHEMA_NAME := ExampleStudyOne +DM_OUTPUT_DIR := $(or $(SYNTH_OUTPUT_DIR),output/ExampleStudyOne) +DM_INPUT_DIR := $(DM_OUTPUT_DIR)/prepared +DM_TRANS_SPEC_DIR := $(SYNTH_DIR)/specs/example_study_one +DM_MAPPING_SPEC := $(DM_TRANS_SPEC_DIR) +DM_MAP_TARGET_SCHEMA := $(SYNTH_DIR)/bdchm.yaml +DM_MAPPING_PREFIX := SYNTH1 +DM_MAPPING_POSTFIX := -data +DM_MAP_STRICT := false + +# YAML stays primary because it is what makes the published corpus readable. +# JSONL is emitted alongside it as the machine-facing form: it preserves the +# same nesting and is what the Parquet build reads. +DM_MAP_OUTPUT_TYPE := yaml jsonl diff --git a/synthetic/pipeline/example_study_two.mk b/synthetic/pipeline/example_study_two.mk new file mode 100644 index 0000000..7a49b54 --- /dev/null +++ b/synthetic/pipeline/example_study_two.mk @@ -0,0 +1,19 @@ +# Pipeline config for the synthetic Example Study Two cohort. +# Target schema is BDCHM itself, so the harmonized output is structurally +# identical to real harmonized data rather than a simplified stand-in. + +DM_RAW_SOURCE := $(SYNTH_DIR)/data/raw/study_two +DM_SCHEMA_NAME := ExampleStudyTwo +DM_OUTPUT_DIR := $(or $(SYNTH_OUTPUT_DIR),output/ExampleStudyTwo) +DM_INPUT_DIR := $(DM_OUTPUT_DIR)/prepared +DM_TRANS_SPEC_DIR := $(SYNTH_DIR)/specs/example_study_two +DM_MAPPING_SPEC := $(DM_TRANS_SPEC_DIR) +DM_MAP_TARGET_SCHEMA := $(SYNTH_DIR)/bdchm.yaml +DM_MAPPING_PREFIX := SYNTH2 +DM_MAPPING_POSTFIX := -data +DM_MAP_STRICT := false + +# YAML stays primary because it is what makes the published corpus readable. +# JSONL is emitted alongside it as the machine-facing form: it preserves the +# same nesting and is what the Parquet build reads. +DM_MAP_OUTPUT_TYPE := yaml jsonl diff --git a/synthetic/population.py b/synthetic/population.py new file mode 100644 index 0000000..5a4826d --- /dev/null +++ b/synthetic/population.py @@ -0,0 +1,350 @@ +# ruff: noqa: S311 +""" +The population model. + +Builds people, their study participation, visits, measurements, conditions and +drug exposures. Emits nothing — see generate.py for the dbGaP-style writers. + +Everything is deterministic given SEED, so the corpus need not be committed: +the same seed reproduces it byte for byte. +""" + +import random +from dataclasses import dataclass, field + +import vocab as v + +SEED = 20260807 + +# 0.5% of individual cell values are nulled, per the brief. +NULL_RATE = 0.005 +# 10% of measurements land outside the normal range. +PATHOLOGY_RATE = 0.10 +# Measurements co-occur with the conditions that explain them, so that selecting +# a cohort produces a distribution that differs from the corpus as a whole. +# Illustrative: the direction of each association is real, the strength is not +# taken from any literature and must not be read as a finding. +COMORBID_PATHOLOGY_RATE = 0.55 +COMORBIDITY = { + "heart_failure": ("systolic", "diastolic", "bmi", "hdl", "bun"), + "hypertension": ("systolic", "diastolic"), + "heart_attack": ("systolic", "hdl"), +} +# Exactly this many HDL results fall below the assay's lower limit. +HDL_BELOW_LLOD = 5 + +HDL_LLOD = 5.0 +HDL_ULOD = 150.0 + + +@dataclass +class Study: + """A synthetic cohort. Accessions are fictional but well-formed.""" + + name: str + phs: str + size: int + visits: int + male_fraction: float + race_mix: dict + bmi_categorical: bool + tables: dict + + +STUDY_ONE = Study( + name="Example Study One", + phs="phs000101", + size=500, + visits=3, + male_fraction=0.54, + race_mix={v.WHITE: 0.50, v.BLACK: 0.30, v.ASIAN: 0.10, v.AMERICAN_INDIAN: 0.10}, + bmi_categorical=False, + tables={ + "subject": "pht000111", + "visit": "pht000112", + "clinical": "pht000113", + "labs": "pht000114", + "conditions": "pht000115", + "meds": "pht000116", + }, +) + +STUDY_TWO = Study( + name="Example Study Two", + phs="phs000102", + size=500, + visits=5, + male_fraction=0.46, + race_mix={ + v.WHITE: 0.60, + v.BLACK: 0.20, + v.MIDDLE_EASTERN: 0.10, + v.NATIVE_HAWAIIAN: 0.10, + }, + bmi_categorical=True, + tables={ + "subject": "pht000121", + "visit": "pht000122", + "clinical": "pht000123", + "labs": "pht000124", + "conditions": "pht000125", + "meds": "pht000126", + }, +) + + +@dataclass +class Person: + """An individual. Shared across studies when they enrol in both.""" + + dbgap_id: int + sex: str + race: str + ethnicity: str + deceased: bool + death_cause: str | None + baseline_age: int + + +@dataclass +class Participant: + """A person's enrolment in one study.""" + + person: Person + subject_id: int + study: Study + visits: list = field(default_factory=list) + conditions: dict = field(default_factory=dict) + ccb: str | None = None + + +@dataclass +class Visit: + """One study visit and the measurements taken at it.""" + + number: int + category: str + age_days: int + height_cm: float | None = None + weight_kg: float | None = None + bmi: float | None = None + bmi_category: str | None = None + systolic: float | None = None + diastolic: float | None = None + hdl: float | None = None + hdl_operator: str | None = None + bun: float | None = None + wbc: float | None = None + + +def _weighted(rng, mapping): + return rng.choices(list(mapping), weights=list(mapping.values()), k=1)[0] + + +def _maybe_null(rng, value): + """Sprinkle nulls at NULL_RATE, per the brief.""" + return None if rng.random() < NULL_RATE else value + + +def _measure(rng, normal, pathological, rate=PATHOLOGY_RATE): + """Draw from the normal range, or the pathological one `rate` of the time.""" + lo, hi = pathological if rng.random() < rate else normal + return round(rng.uniform(lo, hi), 1) + + +def _elevated(conditions): + """Which measures this participant's conditions make abnormal more often.""" + if not conditions: + return frozenset() + measures = set() + for name, linked in COMORBIDITY.items(): + entry = conditions.get(name) + if entry and entry["status"] in (v.PRESENT, v.HISTORICAL): + measures.update(linked) + return frozenset(measures) + + +def _make_person(rng, dbgap_id, study): + sex = v.MALE if rng.random() < study.male_fraction else v.FEMALE + race = _weighted(rng, study.race_mix) + + # 10% of white and 5% of black participants are Hispanic. + hispanic_chance = {v.WHITE: 0.10, v.BLACK: 0.05}.get(race, 0.0) + ethnicity = v.HISPANIC if rng.random() < hispanic_chance else v.NOT_HISPANIC + + deceased = rng.random() < 0.10 + return Person( + dbgap_id=dbgap_id, + sex=sex, + race=race, + ethnicity=ethnicity, + deceased=deceased, + death_cause=rng.choice(v.TOP_DEATH_CAUSES) if deceased else None, + baseline_age=rng.randint(45, 78), + ) + + +def _make_visits(rng, study, person): + """ + One TELEHEALTH visit per participant; the rest are site visits. + + Deceased participants stop attending: visits are truncated at a random + point so nobody is measured after they die. + """ + telehealth_at = rng.randrange(study.visits) + count = study.visits + if person.deceased: + count = rng.randint(1, study.visits) + + visits = [] + for i in range(count): + category = v.TELEHEALTH if i == telehealth_at else v.STUDY_SITE_VISIT + visits.append( + Visit( + number=i + 1, + category=category, + age_days=(person.baseline_age + i * 2) * 365, + ) + ) + return visits + + +def _fill_measurements(rng, study, visit, conditions=None): + elevated = _elevated(conditions) + + def rate(measure): + return COMORBID_PATHOLOGY_RATE if measure in elevated else PATHOLOGY_RATE + + visit.height_cm = _maybe_null(rng, _measure(rng, (150, 190), (135, 205))) + visit.weight_kg = _maybe_null(rng, _measure(rng, (55, 95), (38, 145))) + + bmi = _measure(rng, (18.5, 24.9), (25.0, 41.0), rate("bmi")) + if study.bmi_categorical: + if bmi < 18.5: + category = "underweight" + elif bmi < 25.0: + category = "normal weight" + else: + category = "over weight" + visit.bmi_category = _maybe_null(rng, category) + else: + visit.bmi = _maybe_null(rng, bmi) + + sys_rate, dia_rate = rate("systolic"), rate("diastolic") + visit.systolic = _maybe_null(rng, _measure(rng, (105, 132), (141, 178), sys_rate)) + visit.diastolic = _maybe_null(rng, _measure(rng, (68, 84), (91, 108), dia_rate)) + visit.hdl = _maybe_null(rng, _measure(rng, (40, 72), (22, 39), rate("hdl"))) + visit.bun = _maybe_null(rng, _measure(rng, (7, 20), (21, 46), rate("bun"))) + visit.wbc = _maybe_null(rng, _measure(rng, (4.5, 11.0), (1.8, 19.5), rate("wbc"))) + + +def _make_conditions(rng): + """Condition statuses follow the proportions in the brief.""" + conditions = {} + + r = rng.random() + conditions["heart_failure"] = { + "status": v.PRESENT if r < 0.15 else v.ABSENT, + "concept": rng.choice(v.HEART_FAILURE), + } + + r = rng.random() + if r < 0.35: + status = v.PRESENT + elif r < 0.95: + status = v.ABSENT + else: + status = v.UNKNOWN + conditions["family_stroke"] = { + "status": status, + "concept": rng.choice(v.STROKE), + "relationship": rng.choice([v.NATURAL_FATHER, v.NATURAL_MOTHER]), + } + + r = rng.random() + if r < 0.10: + status = v.PRESENT + elif r < 0.20: + status = v.HISTORICAL + else: + status = v.ABSENT + conditions["hypertension"] = {"status": status, "concept": v.HYPERTENSION} + + present = rng.random() < 0.10 + conditions["heart_attack"] = { + "status": v.PRESENT if present else v.ABSENT, + "concept": rng.choice(v.HEART_ATTACK), + # Half of those with an MI have it from the study record, with an ECG + # in evidence; the rest self-report. + "from_study_record": present and rng.random() < 0.5, + } + return conditions + + +def build(): + """Build both cohorts, with 5% of Study Two's participants shared with One.""" + rng = random.Random(SEED) + + people = {} + participants = [] + + next_dbgap = 900001 + next_subject = {STUDY_ONE.name: 100001, STUDY_TWO.name: 200001} + + for study in (STUDY_ONE, STUDY_TWO): + shared = [] + if study is STUDY_TWO: + one = [p for p in participants if p.study is STUDY_ONE] + shared = rng.sample(one, k=int(round(study.size * 0.05))) + + for i in range(study.size): + if i < len(shared): + # Same individual, so the same dbGaP_Subject_ID and therefore + # the same Person after harmonisation — but a new study-local + # SUBJECT_ID, and therefore a distinct Participant. + person = shared[i].person + else: + person = _make_person(rng, next_dbgap, study) + people[next_dbgap] = person + next_dbgap += 1 + + participant = Participant( + person=person, + subject_id=next_subject[study.name], + study=study, + ) + next_subject[study.name] += 1 + + # Conditions first: measurements are drawn conditional on them, so + # that an abnormal result and the diagnosis explaining it co-occur. + participant.conditions = _make_conditions(rng) + + participant.visits = _make_visits(rng, study, person) + for visit in participant.visits: + _fill_measurements(rng, study, visit, participant.conditions) + + if rng.random() < 0.20: + participant.ccb = rng.choice(v.CALCIUM_CHANNEL_BLOCKERS) + + participants.append(participant) + + _apply_hdl_limits(rng, participants) + return people, participants + + +def _apply_hdl_limits(rng, participants): + """ + Force exactly HDL_BELOW_LLOD results below the assay's lower limit. + + These carry a '<' operator on the Quantity rather than a plain value, which + is how BDCHM represents a censored result. + """ + candidates = [ + (p, visit) + for p in participants + for visit in p.visits + if visit.hdl is not None + ] + for _, visit in rng.sample(candidates, k=HDL_BELOW_LLOD): + visit.hdl = HDL_LLOD + visit.hdl_operator = "<" diff --git a/synthetic/sample.py b/synthetic/sample.py new file mode 100644 index 0000000..2e5f70d --- /dev/null +++ b/synthetic/sample.py @@ -0,0 +1,123 @@ +""" +Extract a small, feature-demonstrating sample of harmonized output. + +The corpus itself is a build product and stays out of git, but a reviewer — or +a team deciding whether this is the reference data they want — should be able to +see what the pipeline produces without running it. This picks records that +between them exercise every structural feature the corpus claims to cover, +rather than the first few of each class. + +Run after the pipeline: + + python synthetic/sample.py +""" + +import argparse +from pathlib import Path + +import yaml + +# Each entry: output class, how many to take, and what makes a record interesting. +SELECTORS = { + # cause_of_death is multivalued, and is emitted even for the living — with + # a null cause — so presence of the entry does not mean deceased. + "Person": [ + ("deceased, with cause of death", + lambda d: any(c.get("cause") for c in d.get("cause_of_death") or [])), + ("living, note the empty cause_of_death entry", + lambda d: not any(c.get("cause") for c in d.get("cause_of_death") or [])), + ], + "Participant": [ + ("enrolment, linked to a Person", lambda d: d.get("associated_person")), + ], + "Demography": [ + ("Hispanic ethnicity", lambda d: d.get("ethnicity") == "OMOP:38003563"), + ("non-Hispanic", lambda d: d.get("ethnicity") == "OMOP:38003564"), + ], + "Visit": [ + ("telehealth", lambda d: d.get("visit_category") == "TELEHEALTH"), + ("study site visit", lambda d: d.get("visit_category") == "STUDY_SITE_VISIT"), + ], + "MeasurementObservationSet": [ + ("blood pressure, systolic and diastolic in one set", + lambda d: len(d.get("observations", [])) == 2), + ], + "MeasurementObservation": [ + ("HDL below the assay's lower limit, carrying the operator", + lambda d: d.get("value_quantity", {}).get("operator")), + ("HDL with assay detection limits", + lambda d: d.get("associated_assay", {}).get("lower_limit_of_detection")), + ("WBC referencing the CBC assay", + lambda d: d.get("associated_assay", {}).get("method") == "MMO:0000533"), + ("BUN qualified as an average", lambda d: d.get("qualifier")), + ("BMI as a continuous value", + lambda d: d.get("observation_type") == "OMOP:3038553" + and d.get("value_quantity", {}).get("value_decimal") is not None), + ("BMI as a category", + lambda d: d.get("observation_type") == "OMOP:3038553" + and d.get("value_quantity", {}).get("value_concept") is not None), + ], + "Condition": [ + ("family history, with the relative recorded", + lambda d: d.get("relationship_to_participant", "").startswith("OMOP:")), + ("about oneself", lambda d: d.get("relationship_to_participant") == "ONESELF"), + ], + "DrugExposure": [ + ("taking a calcium channel blocker", lambda d: d.get("drug_concept")), + ("not taking one", lambda d: not d.get("drug_concept")), + ], +} + + +def load(path): + """Read a multi-document YAML file, skipping empties.""" + with open(path) as fh: + return [d for d in yaml.safe_load_all(fh) if d] + + +def pick(records, selectors): + """Take the first record matching each selector, keeping the labels.""" + chosen = [] + for label, predicate in selectors: + for record in records: + try: + if predicate(record): + chosen.append((label, record)) + break + except (AttributeError, TypeError): + continue + return chosen + + +def main(): + """Write the sample, drawing from both studies.""" + parser = argparse.ArgumentParser(description=__doc__) + here = Path(__file__).parent + parser.add_argument("--output", type=Path, default=here / "output") + parser.add_argument("--out", type=Path, default=here / "sample") + args = parser.parse_args() + args.out.mkdir(parents=True, exist_ok=True) + + studies = [("study_one", "SYNTH1"), ("study_two", "SYNTH2")] + written = 0 + + for cls, selectors in SELECTORS.items(): + blocks = [] + for study, prefix in studies: + path = args.output / study / "mapped-data" / f"{prefix}-{cls}--data.yaml" + if not path.exists(): + continue + for label, record in pick(load(path), selectors): + blocks.append(f"# {study}: {label}\n" + yaml.safe_dump(record, sort_keys=True)) + + if blocks: + target = args.out / f"{cls}.yaml" + target.write_text("---\n".join(blocks)) + written += 1 + print(f"{cls:26} {len(blocks):2} records {target.stat().st_size / 1024:5.1f} KB") + + print(f"\n{written} files written to {args.out}") + + +if __name__ == "__main__": + main() diff --git a/synthetic/sample/Condition.yaml b/synthetic/sample/Condition.yaml new file mode 100644 index 0000000..0faa180 --- /dev/null +++ b/synthetic/sample/Condition.yaml @@ -0,0 +1,31 @@ +# study_one: family history, with the relative recorded +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +condition_concept: MONDO:0005264 +condition_provenance: PATIENT_SELF-REPORTED_CONDITION +condition_status: ABSENT +id: 1bacbc56-4aaf-5017-b70c-42fa03f9d48b +relationship_to_participant: OMOP:4321888 +--- +# study_one: about oneself +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +condition_concept: MONDO:0005068 +condition_provenance: PATIENT_SELF-REPORTED_CONDITION +condition_status: PRESENT +id: 4058ed2d-92d4-5b8a-af0e-aee32c2c9960 +relationship_to_participant: ONESELF +--- +# study_two: family history, with the relative recorded +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +condition_concept: MONDO:0013792 +condition_provenance: PATIENT_SELF-REPORTED_CONDITION +condition_status: UNKNOWN +id: 87bc5b14-516d-598a-a33d-0bd3d7b92615 +relationship_to_participant: OMOP:4321888 +--- +# study_two: about oneself +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +condition_concept: MONDO:0005068 +condition_provenance: PATIENT_SELF-REPORTED_CONDITION +condition_status: ABSENT +id: 67332483-bbf9-57fd-98de-9f54e47bcd9e +relationship_to_participant: ONESELF diff --git a/synthetic/sample/Demography.yaml b/synthetic/sample/Demography.yaml new file mode 100644 index 0000000..b9c45d5 --- /dev/null +++ b/synthetic/sample/Demography.yaml @@ -0,0 +1,31 @@ +# study_one: Hispanic ethnicity +associated_participant: dd37092b-1515-5c8b-aebf-18507b1f98f6 +ethnicity: OMOP:38003563 +id: 27107a6a-5d26-54ac-90cd-c243d4102d3a +race: +- OMOP:8527 +sex: OMOP:8507 +--- +# study_one: non-Hispanic +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +ethnicity: OMOP:38003564 +id: 0dc199a5-8a48-58f1-be03-90ac898be070 +race: +- OMOP:8516 +sex: OMOP:8507 +--- +# study_two: Hispanic ethnicity +associated_participant: 390f76f9-c667-5487-a1b9-ecd8b5fe9927 +ethnicity: OMOP:38003563 +id: 021e4f47-6504-5725-b1e0-d79450ce064a +race: +- OMOP:8527 +sex: OMOP:8532 +--- +# study_two: non-Hispanic +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +ethnicity: OMOP:38003564 +id: 18e9e092-8ad5-5414-b9e8-677ff7d886d1 +race: +- OMOP:8515 +sex: OMOP:8532 diff --git a/synthetic/sample/DrugExposure.yaml b/synthetic/sample/DrugExposure.yaml new file mode 100644 index 0000000..9ec6e0b --- /dev/null +++ b/synthetic/sample/DrugExposure.yaml @@ -0,0 +1,27 @@ +# study_one: taking a calcium channel blocker +associated_participant: e805802b-da55-5d52-9db3-810ad3ed2058 +drug_concept: RxCUI:17767 +exposure_provenance: PATIENT_SELF_REPORTED_MEDICATION +exposure_status: PRESENT +id: 5f34642a-0636-5a84-a778-a3ba54d8f45f +--- +# study_one: not taking one +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +drug_concept: null +exposure_provenance: PATIENT_SELF_REPORTED_MEDICATION +exposure_status: ABSENT +id: f9592e79-ad0d-5e45-b130-a5823ecc447a +--- +# study_two: taking a calcium channel blocker +associated_participant: 0a0a6844-cd9d-5401-95b0-f7dfa5d910db +drug_concept: NDFRT:N0000175421 +exposure_provenance: PATIENT_SELF_REPORTED_MEDICATION +exposure_status: PRESENT +id: 5cd32ad9-48ef-570f-a1bb-c7ef61bbc57c +--- +# study_two: not taking one +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +drug_concept: null +exposure_provenance: PATIENT_SELF_REPORTED_MEDICATION +exposure_status: ABSENT +id: db1378b7-26b4-5bad-b1e6-2c1d28a4c562 diff --git a/synthetic/sample/MeasurementObservation.yaml b/synthetic/sample/MeasurementObservation.yaml new file mode 100644 index 0000000..8d8a406 --- /dev/null +++ b/synthetic/sample/MeasurementObservation.yaml @@ -0,0 +1,154 @@ +# study_one: HDL below the assay's lower limit, carrying the operator +associated_assay: + id: ddcb07cc-0d54-5828-93a8-495a3f8a3f07 + lower_limit_of_detection: + id: c6473842-2d99-5ae5-9b6e-ad28ae6871a3 + unit: mg/dL + value_decimal: 5.0 + method: MMO:0000133 + upper_limit_of_detection: + id: b0655aa0-8a6e-5719-809d-e5678af1e9e6 + unit: mg/dL + value_decimal: 150.0 +associated_participant: 9e9b7cbd-07b5-5ea9-9912-6cc95a1e8ca5 +associated_visit: 7b35a86e-1813-56e3-97d2-0fbb66803db0 +id: b01f532f-d4cd-5dc7-b494-79b5876a6e8c +observation_type: OMOP:3007070 +value_quantity: + id: be2ab2d1-121b-547f-aae9-c647ddf3d69f + operator: < + unit: mg/dL + value_decimal: 5.0 +--- +# study_one: HDL with assay detection limits +associated_assay: + id: ddcb07cc-0d54-5828-93a8-495a3f8a3f07 + lower_limit_of_detection: + id: c6473842-2d99-5ae5-9b6e-ad28ae6871a3 + unit: mg/dL + value_decimal: 5.0 + method: MMO:0000133 + upper_limit_of_detection: + id: b0655aa0-8a6e-5719-809d-e5678af1e9e6 + unit: mg/dL + value_decimal: 150.0 +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +associated_visit: 75122f58-efd2-559d-9eac-2619199795aa +id: 5c2378dc-1d98-5cf0-9af3-2161df597017 +observation_type: OMOP:3007070 +value_quantity: + id: 16ec110d-a004-5269-a4f0-33574ed63357 + operator: null + unit: mg/dL + value_decimal: 58.4 +--- +# study_one: WBC referencing the CBC assay +associated_assay: + id: 25bf7403-bd81-5764-9e90-8af5246333ff + method: MMO:0000533 +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +associated_visit: 75122f58-efd2-559d-9eac-2619199795aa +id: 520eee1a-926b-5bc0-80fa-fa7aec68871c +observation_type: OMOP:3000905 +value_quantity: + id: 071da59d-a293-55d0-a8ab-141a15d99ab3 + unit: 10*3/uL + value_decimal: 10.4 +--- +# study_one: BUN qualified as an average +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +associated_visit: 75122f58-efd2-559d-9eac-2619199795aa +id: 74e323bb-6fa2-5a03-b8d4-01d1b37713bb +observation_type: OMOP:3013682 +qualifier: AVERAGE +value_quantity: + id: 54f1a4f3-26f9-532c-af11-0f9a032fcc7e + unit: mg/dL + value_decimal: 18.4 +--- +# study_one: BMI as a continuous value +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +associated_visit: 75122f58-efd2-559d-9eac-2619199795aa +id: dcfddca5-aea6-55ea-a3f5-05c8ea5e87b6 +observation_type: OMOP:3038553 +value_quantity: + id: 69a41584-68c4-5f4c-ab51-419b2f2416f7 + unit: kg/m2 + value_decimal: 21.9 +--- +# study_two: HDL below the assay's lower limit, carrying the operator +associated_assay: + id: 1533b5fa-b7d7-5d34-a075-88950c88b9b3 + lower_limit_of_detection: + id: 7994b9e9-3e5a-5f29-a258-5873bfa4ba3f + unit: mg/dL + value_decimal: 5.0 + method: MMO:0000133 + upper_limit_of_detection: + id: cd40e959-ee21-55fb-bcd2-a6a84fdae679 + unit: mg/dL + value_decimal: 150.0 +associated_participant: f82b92bc-cc81-5d6d-9fcc-e2b1df68b733 +associated_visit: ca601664-8b89-5304-8f58-200d2ccaa4a2 +id: e5e4b790-b5ed-57ee-8679-9c75b4af3960 +observation_type: OMOP:3007070 +value_quantity: + id: 31829a46-4891-541c-b0ae-437badfdb600 + operator: < + unit: mg/dL + value_decimal: 5.0 +--- +# study_two: HDL with assay detection limits +associated_assay: + id: 1533b5fa-b7d7-5d34-a075-88950c88b9b3 + lower_limit_of_detection: + id: 7994b9e9-3e5a-5f29-a258-5873bfa4ba3f + unit: mg/dL + value_decimal: 5.0 + method: MMO:0000133 + upper_limit_of_detection: + id: cd40e959-ee21-55fb-bcd2-a6a84fdae679 + unit: mg/dL + value_decimal: 150.0 +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +associated_visit: 22bf721d-ce80-5f08-b1c5-d19857b89b15 +id: 417b93ad-af07-509f-861b-451ae970154f +observation_type: OMOP:3007070 +value_quantity: + id: 9c071689-64ca-58df-b23b-1c664ee4336f + operator: null + unit: mg/dL + value_decimal: 63.7 +--- +# study_two: WBC referencing the CBC assay +associated_assay: + id: c1a9ff74-d562-5bd5-b97f-15a2ed7a8e57 + method: MMO:0000533 +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +associated_visit: 22bf721d-ce80-5f08-b1c5-d19857b89b15 +id: 4965a51d-170d-5b2b-a818-05af41b3424c +observation_type: OMOP:3000905 +value_quantity: + id: a91f6e17-efb3-5168-b09b-0a66df0abe43 + unit: 10*3/uL + value_decimal: 9.9 +--- +# study_two: BUN qualified as an average +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +associated_visit: 22bf721d-ce80-5f08-b1c5-d19857b89b15 +id: d5c1a7cf-457f-5274-b8f5-41644eb4b1c5 +observation_type: OMOP:3013682 +qualifier: AVERAGE +value_quantity: + id: 6e3ab74d-99e2-5419-984a-29361d647941 + unit: mg/dL + value_decimal: 17.9 +--- +# study_two: BMI as a category +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +associated_visit: 22bf721d-ce80-5f08-b1c5-d19857b89b15 +id: 6b2572f5-a244-504c-9f4f-7dae5a26ceb0 +observation_type: OMOP:3038553 +value_quantity: + id: f427eb8f-c038-5c83-9ea3-ffa4586a6728 + value_concept: normal weight diff --git a/synthetic/sample/MeasurementObservationSet.yaml b/synthetic/sample/MeasurementObservationSet.yaml new file mode 100644 index 0000000..15b032b --- /dev/null +++ b/synthetic/sample/MeasurementObservationSet.yaml @@ -0,0 +1,47 @@ +# study_one: blood pressure, systolic and diastolic in one set +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +associated_visit: 75122f58-efd2-559d-9eac-2619199795aa +id: f443b266-2d72-5429-8df4-43dfaf1bcc12 +observations: +- associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a + body_position: SITTING_POSITION + body_site: Right arm + id: daae1547-43eb-5db9-88a0-1fc442022951 + observation_type: OMOP:4152194 + value_quantity: + id: 7fe894d1-919d-573a-a17d-d36adc78f486 + unit: mm[Hg] + value_decimal: 164.3 +- associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a + body_position: SITTING_POSITION + body_site: Right arm + id: ecee49c5-c3c5-540b-b723-f43b78f05656 + observation_type: OMOP:4154790 + value_quantity: + id: f2c228df-56c8-56a0-946d-d8c06f69d7e0 + unit: mm[Hg] + value_decimal: 82.6 +--- +# study_two: blood pressure, systolic and diastolic in one set +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +associated_visit: 22bf721d-ce80-5f08-b1c5-d19857b89b15 +id: e9806428-dd0d-5ea0-b9d4-a7b21d8156c5 +observations: +- associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb + body_position: SITTING_POSITION + body_site: Right arm + id: 2d648e6e-ea07-5879-a4f8-569930bc20d0 + observation_type: OMOP:4152194 + value_quantity: + id: d7463d3b-6bbb-53a5-9ee8-d61c723822bd + unit: mm[Hg] + value_decimal: 128.2 +- associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb + body_position: SITTING_POSITION + body_site: Right arm + id: f3d69c77-bf43-5e06-9185-ce6cac66e8de + observation_type: OMOP:4154790 + value_quantity: + id: cbe14372-7714-55f0-a567-db4b1735f728 + unit: mm[Hg] + value_decimal: 94.6 diff --git a/synthetic/sample/Participant.yaml b/synthetic/sample/Participant.yaml new file mode 100644 index 0000000..97a6123 --- /dev/null +++ b/synthetic/sample/Participant.yaml @@ -0,0 +1,11 @@ +# study_one: enrolment, linked to a Person +associated_person: ddbe1e3e-e49d-590e-97f5-2e5e2102cbd0 +id: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +identity: '100001' +member_of_research_study: Example Study One +--- +# study_two: enrolment, linked to a Person +associated_person: b9733cac-ed81-52b8-bb35-ac8f1fccb21f +id: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +identity: '200001' +member_of_research_study: Example Study Two diff --git a/synthetic/sample/Person.yaml b/synthetic/sample/Person.yaml new file mode 100644 index 0000000..f547a0a --- /dev/null +++ b/synthetic/sample/Person.yaml @@ -0,0 +1,39 @@ +# study_one: deceased, with cause of death +cause_of_death: +- cause: ICD10CM:R99 + id: 90723e22-0b8f-5ab4-b1bd-8e697cd52e1e + order: 1 +id: ddbe1e3e-e49d-590e-97f5-2e5e2102cbd0 +identity: '900001' +species: NCBITaxon:9606 +vital_status: OMOP:434489 +--- +# study_one: living, note the empty cause_of_death entry +cause_of_death: +- cause: null + id: 73e3cc59-34b6-5c7e-aee4-df4f2adf0d96 + order: null +id: 567cf7d7-a287-5577-82b2-1feaca5df345 +identity: '900002' +species: NCBITaxon:9606 +vital_status: OMOP:4230556 +--- +# study_two: deceased, with cause of death +cause_of_death: +- cause: ICD10CM:R99 + id: c810c2ab-9b85-58f1-a7ba-009eccd3b843 + order: 1 +id: 79153ada-4607-5077-a30d-93cc240aa7df +identity: '900441' +species: NCBITaxon:9606 +vital_status: OMOP:434489 +--- +# study_two: living, note the empty cause_of_death entry +cause_of_death: +- cause: null + id: c92da1ee-347b-585a-a12c-bda32ec3c201 + order: null +id: b9733cac-ed81-52b8-bb35-ac8f1fccb21f +identity: '900093' +species: NCBITaxon:9606 +vital_status: OMOP:4230556 diff --git a/synthetic/sample/Visit.yaml b/synthetic/sample/Visit.yaml new file mode 100644 index 0000000..b8797ee --- /dev/null +++ b/synthetic/sample/Visit.yaml @@ -0,0 +1,27 @@ +# study_one: telehealth +age_at_visit_start: 28105 +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +id: 7d2ad563-3512-5d32-8737-82e10cfcac51 +visit_category: TELEHEALTH +visit_provenance: CASE_REPORT_FORM +--- +# study_one: study site visit +age_at_visit_start: 27375 +associated_participant: ded1d538-866f-5c18-8aa2-0b24f1f08a5a +id: 75122f58-efd2-559d-9eac-2619199795aa +visit_category: STUDY_SITE_VISIT +visit_provenance: CASE_REPORT_FORM +--- +# study_two: telehealth +age_at_visit_start: 22630 +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +id: 17f3fa62-5837-571d-9e3a-5b089b4c1d9d +visit_category: TELEHEALTH +visit_provenance: CASE_REPORT_FORM +--- +# study_two: study site visit +age_at_visit_start: 21170 +associated_participant: e9d16f51-a014-5b89-bb71-d3550b9a7cbb +id: 22bf721d-ce80-5f08-b1c5-d19857b89b15 +visit_category: STUDY_SITE_VISIT +visit_provenance: CASE_REPORT_FORM diff --git a/synthetic/schema.py b/synthetic/schema.py new file mode 100644 index 0000000..0644152 --- /dev/null +++ b/synthetic/schema.py @@ -0,0 +1,252 @@ +""" +Build Parquet from harmonized JSONL, with column types derived from BDCHM. + +Why not let DuckDB infer: inference reads the data, so a column that happens to +be all-null in one run types differently in the next, and consumers see the +schema shift under them. Deriving from BDCHM makes the schema a property of the +model rather than of the sample. + +Why BDCHM alone is not enough: whether a slot arrives as a nested object or as a +reference is decided by the *transformation spec*, not the model. BDCHM gives +`associated_participant` a range of `Participant`, but the spec materialises it +as a uuid5 string, while `value_quantity` is nested inline. So this reads both — +BDCHM for leaf types, the specs for which slots nest. + + python schema.py --study study_one +""" + +import argparse +import json +from pathlib import Path + +import duckdb +import yaml + +# LinkML builtin ranges to DuckDB types. Identifiers stay VARCHAR rather than +# UUID: they are uuid5 today, but nothing in the model requires that, and a +# stricter type would break the day one is not. +TYPE_MAP = { + "string": "VARCHAR", + "uriorcurie": "VARCHAR", + "uri": "VARCHAR", + "curie": "VARCHAR", + "ncname": "VARCHAR", + "integer": "BIGINT", + "decimal": "DOUBLE", + "double": "DOUBLE", + "float": "DOUBLE", + "boolean": "BOOLEAN", + "date": "DATE", + "datetime": "TIMESTAMP", + "time": "TIME", +} + +unmapped = set() +mismatches = [] + + +class Model: + """BDCHM, indexed for slot lookup with inheritance.""" + + def __init__(self, path): + """Load and index the model.""" + schema = yaml.safe_load(open(path)) + self.classes = schema.get("classes", {}) + self.slots = schema.get("slots", {}) + self.enums = set(schema.get("enums", {})) + self.types = set(schema.get("types", {})) + + def slot(self, class_name, slot_name): + """ + Find a slot on a class or any ancestor. + + BDCHM defines slots three ways and a class can use all of them: inline + under `attributes`, by reference under `slots` to a top-level definition, + and refined per class under `slot_usage`. A definition is assembled from + whichever apply, with slot_usage taking precedence — that is where + cardinality and range are often narrowed for a particular class. + """ + merged = {} + seen = set() + current = class_name + chain = [] + while current and current in self.classes and current not in seen: + seen.add(current) + chain.append(self.classes[current]) + current = self.classes[current].get("is_a") + + # Walk from the most distant ancestor down so subclasses win. + for cls in reversed(chain): + if slot_name in (cls.get("slots") or []): + merged.update(self.slots.get(slot_name) or {}) + attrs = cls.get("attributes") or {} + if slot_name in attrs: + merged.update(attrs[slot_name] or {}) + usage = (cls.get("slot_usage") or {}).get(slot_name) + if usage: + merged.update(usage) + + # A slot referenced only through slot_usage still inherits the + # top-level definition, which is where multivalued usually lives. + if merged and slot_name in self.slots: + base = dict(self.slots[slot_name]) + base.update(merged) + merged = base + + return merged or None + + def duckdb_type(self, class_name, slot_name, nested=None): + """Resolve the DuckDB type for one slot. `nested` maps slot -> nested spec.""" + slot = self.slot(class_name, slot_name) + rng = (slot or {}).get("range") + multivalued = bool((slot or {}).get("multivalued")) + + if nested: + inner = self.struct_type(nested["class"], nested["slots"]) + return f"{inner}[]" if multivalued else inner + + if rng in TYPE_MAP: + typ = TYPE_MAP[rng] + elif rng in self.enums or rng in self.classes: + # Enums serialise as their permissible value; class-ranged slots that + # are not nested by the spec arrive as identifier strings. + typ = "VARCHAR" + else: + if rng: + unmapped.add(f"{class_name}.{slot_name} ({rng})") + typ = "VARCHAR" + + return f"{typ}[]" if multivalued else typ + + def struct_type(self, class_name, slots): + """Compose a STRUCT for a nested object, recursing where it nests further.""" + fields = [] + for name, child in slots.items(): + fields.append(f'"{name}" {self.duckdb_type(class_name, name, child)}') + return "STRUCT(" + ", ".join(fields) + ")" + + +def spec_shape(spec_dir): + """ + Read the transform specs to learn which slots nest, and into what. + + Returns {target_class: {slot: None | {class, slots, multivalued}}}. + """ + + def walk(class_derivations): + shape = {} + for cls, cd in (class_derivations or {}).items(): + slots = {} + for name, sd in (cd.get("slot_derivations") or {}).items(): + nested = sd.get("class_derivations") + if nested: + inner_cls, inner_cd = next(iter(nested[0].items())) + slots[name] = { + "class": inner_cls, + "slots": walk({inner_cls: inner_cd})[inner_cls], + } + else: + slots[name] = None + shape[cls] = slots + return shape + + merged = {} + for path in sorted(Path(spec_dir).glob("*.yaml")): + for doc in yaml.safe_load(open(path)) or []: + for cls, slots in walk(doc.get("class_derivations")).items(): + merged.setdefault(cls, {}).update(slots) + return merged + + +def reconcile(columns, jsonl_path, class_name): + """ + Compare the model-derived cardinality against what the data actually holds. + + Where they disagree the data wins, because the goal is a loadable artifact — + but every disagreement is reported. These are real findings: BDCHM declares + `identity` multivalued while the transformation specs, RTI's included, emit a + scalar. A schema derived strictly from the model would not load real + harmonized data either. + """ + with open(jsonl_path) as fh: + first = json.loads(fh.readline() or "{}") + + for name, typ in list(columns.items()): + if name not in first or first[name] is None: + continue + is_list = isinstance(first[name], list) + declared_list = typ.endswith("[]") + if is_list and not declared_list: + columns[name] = f"{typ}[]" + mismatches.append(f"{class_name}.{name}: model says single, data is a list") + elif declared_list and not is_list: + columns[name] = typ[:-2] + mismatches.append(f"{class_name}.{name}: model says multivalued, data is scalar") + return columns + + +def build(model, shape, jsonl_path, parquet_path, class_name): + """Convert one class's JSONL to Parquet under an explicit schema.""" + slots = shape[class_name] + columns = {n: model.duckdb_type(class_name, n, child) for n, child in slots.items()} + columns = reconcile(columns, jsonl_path, class_name) + + # A struct literal, not JSON: DuckDB wants {'col': 'TYPE'}, and the double + # quotes around nested field names are literal inside the single-quoted + # type string. + literal = ", ".join(f"'{name}': '{typ}'" for name, typ in columns.items()) + + con = duckdb.connect() + # noqa justification: every interpolated value originates in BDCHM, the + # specs, or our own paths — none of it is user input. DuckDB's columns + # argument takes a struct literal, which cannot be parameterised. + sql = ( + f"COPY (SELECT * FROM read_json('{jsonl_path}', columns := {{{literal}}})) " # noqa: S608 + f"TO '{parquet_path}' (FORMAT PARQUET)" + ) + con.execute(sql) + return columns + + +def main(): + """Build Parquet for every class the study produced.""" + here = Path(__file__).parent + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--study", default="study_one") + parser.add_argument("--output", type=Path, default=here / "output") + parser.add_argument("--schema", type=Path, default=here / "bdchm.yaml") + args = parser.parse_args() + + slug = "example_" + args.study + model = Model(args.schema) + shape = spec_shape(here / "specs" / slug) + + mapped = args.output / args.study / "mapped-data" + dest = args.output / args.study / "parquet" + dest.mkdir(parents=True, exist_ok=True) + + for jsonl in sorted(mapped.glob("*.jsonl")): + class_name = jsonl.stem.split("-")[1] + if class_name not in shape: + print(f" skip {class_name}: not in specs") + continue + target = dest / f"{class_name}.parquet" + build(model, shape, jsonl, target, class_name) + print( + f"{class_name:26} {jsonl.stat().st_size / 1024:8.1f} KB jsonl" + f" -> {target.stat().st_size / 1024:7.1f} KB parquet" + ) + + if unmapped: + print("\nslots with no mapped range, defaulted to VARCHAR:") + for u in sorted(unmapped): + print(f" {u}") + + if mismatches: + print("\nmodel and data disagree on cardinality (data used):") + for m in mismatches: + print(f" {m}") + + +if __name__ == "__main__": + main() diff --git a/synthetic/specs.py b/synthetic/specs.py new file mode 100644 index 0000000..ec83de1 --- /dev/null +++ b/synthetic/specs.py @@ -0,0 +1,454 @@ +""" +Emit BDCHM-targeted transformation specs for the synthetic cohorts. + +The specs are generated rather than hand-written because both studies share a +structure and differ only in accessions — generating them from the same column +layout the raw tables use means the phv references cannot drift out of sync. + +Patterned on RTI's priority_variables_transform specs: the same uuid5 identity +scheme, the same slot names, the same coded values. + + python synthetic/specs.py [--out DIR] +""" + +import argparse +from pathlib import Path + +import generate as g +import population as pop +import vocab as v + +PERSON_NS = "https://w3id.org/bdchm/Person" +PARTICIPANT_NS = "https://w3id.org/bdchm/Participant" +VISIT_NS = "https://w3id.org/bdchm/Visit" +OBSERVATION_NS = "https://w3id.org/bdchm/MeasurementObservation" +QUANTITY_NS = "https://w3id.org/bdchm/Quantity" +ASSAY_NS = "https://w3id.org/bdchm/Assay" + + +def uid(namespace, key): + """Build a uuid5 derivation expression. Nested objects need explicit ids.""" + return f"'uuid5(\"{namespace}\", {key})'" + + +def row_key(study, table, suffix): + """Build a key unique per source row, for identifying nested objects.""" + subject = phv(study, table, "SUBJECT_ID") + columns, _ = g.LAYOUTS[table] + if "VISIT_NUM" in columns: + num = phv(study, table, "VISIT_NUM") + return f'str({{{subject}}}) + ":{study.name}:" + str({{{num}}}) + ":{suffix}"' + return f'str({{{subject}}}) + ":{study.name}:{suffix}"' + + +def phv(study, table, column): + """Look up the phv accession for a column, by name.""" + columns, count = g.LAYOUTS[table] + ids = g.phv_ids(study, table, count) + return ids[columns.index(column) - 1].split(".")[0] + + +def ident(study, table): + """Build the uuid5 expressions that tie records back to participant and visit.""" + subject = phv(study, table, "SUBJECT_ID") + return subject, f"'uuid5(\"{PARTICIPANT_NS}\", str({{{subject}}}) + \":{study.name}\")'" + + +def visit_ref(study, table): + """Build the uuid5 reference to a row's visit.""" + subject = phv(study, table, "SUBJECT_ID") + num = phv(study, table, "VISIT_NUM") + return f"'uuid5(\"{VISIT_NS}\", str({{{subject}}}) + \":{study.name} VISIT \" + str({{{num}}}))'" + + +def person_and_participant(study): + """Build the Person and Participant derivations.""" + t = study.tables["subject"] + subject = phv(study, "subject", "SUBJECT_ID") + vital = phv(study, "subject", "VITAL_STATUS") + cause = phv(study, "subject", "DEATH_CAUSE") + return f"""- class_derivations: + Person: + populated_from: {t} + slot_derivations: + id: + expr: 'uuid5("{PERSON_NS}", str({{dbGaP_Subject_ID}}))' + identity: + expr: 'str({{dbGaP_Subject_ID}})' + species: + value: NCBITaxon:9606 + vital_status: + expr: 'case(({{{vital}}} == 0, "OMOP:4230556"), ({{{vital}}} == 1, "OMOP:434489"))' + cause_of_death: + class_derivations: + - CauseOfDeath: + populated_from: {t} + slot_derivations: + id: + expr: {uid(v.CAUSE_OF_DEATH_NS, 'str({dbGaP_Subject_ID})')} + cause: + populated_from: {cause} + order: + expr: 'case(({{{vital}}} == 1, 1))' +- class_derivations: + Participant: + populated_from: {t} + slot_derivations: + id: + expr: 'uuid5("{PARTICIPANT_NS}", str({{{subject}}}) + ":{study.name}")' + identity: + expr: 'str({{{subject}}})' + member_of_research_study: + value: '{study.name}' + associated_person: + expr: 'uuid5("{PERSON_NS}", str({{dbGaP_Subject_ID}}))' +""" + + +def demography(study): + """Build the Demography derivation.""" + t = study.tables["subject"] + _, participant = ident(study, "subject") + return f"""- class_derivations: + Demography: + populated_from: {t} + slot_derivations: + id: + expr: {uid(v.DEMOGRAPHY_NS, row_key(study, "subject", "demography"))} + associated_participant: + expr: {participant} + sex: + populated_from: {phv(study, "subject", "SEX")} + race: + populated_from: {phv(study, "subject", "RACE")} + ethnicity: + populated_from: {phv(study, "subject", "ETHNICITY")} +""" + + +def visits(study): + """Build the Visit derivation.""" + t = study.tables["visit"] + subject = phv(study, "visit", "SUBJECT_ID") + num = phv(study, "visit", "VISIT_NUM") + return f"""- class_derivations: + Visit: + populated_from: {t} + slot_derivations: + id: + expr: 'uuid5("{VISIT_NS}", str({{{subject}}}) + ":{study.name} VISIT " + str({{{num}}}))' + associated_participant: + expr: 'uuid5("{PARTICIPANT_NS}", str({{{subject}}}) + ":{study.name}")' + visit_category: + populated_from: {phv(study, "visit", "VISIT_TYPE")} + age_at_visit_start: + populated_from: {phv(study, "visit", "AGE_DAYS")} + visit_provenance: + value: {v.VISIT_PROVENANCE} +""" + + +def blood_pressure(study): + """Systolic and diastolic as two observations inside one set, as in ARIC.""" + t = study.tables["clinical"] + _, participant = ident(study, "clinical") + visit = visit_ref(study, "clinical") + sbp = phv(study, "clinical", "SBP") + dbp = phv(study, "clinical", "DBP") + + def observation(concept, source, label): + return f""" - MeasurementObservation: + populated_from: {t} + slot_derivations: + id: + expr: {uid(OBSERVATION_NS, row_key(study, "clinical", label))} + associated_participant: + expr: {participant} + observation_type: + value: {concept} + body_position: + value: "SITTING_POSITION" + body_site: + value: "Right arm" + value_quantity: + class_derivations: + - Quantity: + populated_from: {t} + slot_derivations: + id: + expr: {uid(QUANTITY_NS, row_key(study, "clinical", label))} + value_decimal: + populated_from: {source} + unit: + value: "mm[Hg]" +""" + + return f"""- class_derivations: + MeasurementObservationSet: + populated_from: {t} + slot_derivations: + id: + expr: {uid(v.OBSERVATION_SET_NS, row_key(study, "clinical", "bp"))} + associated_participant: + expr: {participant} + associated_visit: + expr: {visit} + observations: + class_derivations: +{observation("OMOP:4152194", sbp, "systolic")}{observation("OMOP:4154790", dbp, "diastolic")}""" + + +def simple_measure(study, table, column, concept, unit, extra="", categorical=False): + """ + Build a single MeasurementObservation with a Quantity value. + + Categorical measurements go to value_concept rather than value_decimal — + value_decimal is typed decimal, and the real specs (see fam_income) put + label strings in value_concept for this case. + """ + t = study.tables[table] + _, participant = ident(study, table) + visit = visit_ref(study, table) + key = row_key(study, table, column) + if categorical: + value_slot = "value_concept" + unit_line = "" + else: + value_slot = "value_decimal" + unit_line = f' unit:\n value: "{unit}"\n' + return f"""- class_derivations: + MeasurementObservation: + populated_from: {t} + slot_derivations: + id: + expr: {uid(OBSERVATION_NS, key)} + associated_participant: + expr: {participant} + associated_visit: + expr: {visit} + observation_type: + value: {concept} +{extra} value_quantity: + class_derivations: + - Quantity: + populated_from: {t} + slot_derivations: + id: + expr: {uid(QUANTITY_NS, key)} + {value_slot}: + populated_from: {phv(study, table, column)} +{unit_line}""" + + +def hdl(study): + """HDL carries an operator for censored results and an assay with limits.""" + t = study.tables["labs"] + operator = phv(study, "labs", "HDL_OPERATOR") + _, participant = ident(study, "labs") + visit = visit_ref(study, "labs") + return f"""- class_derivations: + MeasurementObservation: + populated_from: {t} + slot_derivations: + id: + expr: {uid(OBSERVATION_NS, row_key(study, "labs", "HDL"))} + associated_participant: + expr: {participant} + associated_visit: + expr: {visit} + observation_type: + value: OMOP:3007070 + associated_assay: + class_derivations: + - Assay: + populated_from: {t} + slot_derivations: + id: + expr: {uid(ASSAY_NS, f'"{study.name}:HDL"')} + method: + value: {v.ASSAY_METHOD_HDL} + lower_limit_of_detection: + class_derivations: + - Quantity: + populated_from: {t} + slot_derivations: + id: + expr: {uid(QUANTITY_NS, f'"{study.name}:HDL:LLOD"')} + value_decimal: + value: {pop.HDL_LLOD} + unit: + value: "mg/dL" + upper_limit_of_detection: + class_derivations: + - Quantity: + populated_from: {t} + slot_derivations: + id: + expr: {uid(QUANTITY_NS, f'"{study.name}:HDL:ULOD"')} + value_decimal: + value: {pop.HDL_ULOD} + unit: + value: "mg/dL" + value_quantity: + class_derivations: + - Quantity: + populated_from: {t} + slot_derivations: + id: + expr: {uid(QUANTITY_NS, row_key(study, "labs", "HDL"))} + value_decimal: + populated_from: {phv(study, "labs", "HDL")} + operator: + populated_from: {operator} + unit: + value: "mg/dL" +""" + + +def wbc(study): + """WBC references a CBC assay instance through associated_assay.""" + t = study.tables["labs"] + _, participant = ident(study, "labs") + visit = visit_ref(study, "labs") + return f"""- class_derivations: + MeasurementObservation: + populated_from: {t} + slot_derivations: + id: + expr: {uid(OBSERVATION_NS, row_key(study, "labs", "WBC"))} + associated_participant: + expr: {participant} + associated_visit: + expr: {visit} + observation_type: + value: OMOP:3000905 + associated_assay: + class_derivations: + - Assay: + populated_from: {t} + slot_derivations: + id: + expr: {uid(ASSAY_NS, f'"{study.name}:CBC"')} + method: + value: {v.ASSAY_METHOD_WBC} + value_quantity: + class_derivations: + - Quantity: + populated_from: {t} + slot_derivations: + id: + expr: {uid(QUANTITY_NS, row_key(study, "labs", "WBC"))} + value_decimal: + populated_from: {phv(study, "labs", "WBC")} + unit: + value: "10*3/uL" +""" + + +def condition(study, label, status_col, concept_col, provenance, relationship=None): + """One Condition. relationship defaults to ONESELF for self-reported history.""" + t = study.tables["conditions"] + _, participant = ident(study, "conditions") + relationship_line = ( + f" populated_from: {relationship}" + if relationship + else f" value: {v.ONESELF}" + ) + concept = ( + f" populated_from: {phv(study, 'conditions', concept_col)}" + if concept_col + else " value: HP:0000822" + ) + return f"""- class_derivations: + Condition: + populated_from: {t} + slot_derivations: + id: + expr: {uid(v.CONDITION_NS, row_key(study, "conditions", label))} + associated_participant: + expr: {participant} + condition_concept: +{concept} + condition_status: + populated_from: {phv(study, "conditions", status_col)} + condition_provenance: + value: {provenance} + relationship_to_participant: +{relationship_line} +""" + + +def drug_exposure(study): + """Build the DrugExposure derivation.""" + t = study.tables["meds"] + _, participant = ident(study, "meds") + return f"""- class_derivations: + DrugExposure: + populated_from: {t} + slot_derivations: + id: + expr: {uid(v.DRUG_EXPOSURE_NS, row_key(study, "meds", "ccb"))} + associated_participant: + expr: {participant} + drug_concept: + populated_from: {phv(study, "meds", "CCB_CONCEPT")} + exposure_status: + populated_from: {phv(study, "meds", "CCB_STATUS")} + exposure_provenance: + value: PATIENT_SELF_REPORTED_MEDICATION +""" + + +def build(study): + """All specs for one study, keyed by filename.""" + bmi_unit = "" if study.bmi_categorical else "kg/m2" + specs = { + "person_participant": person_and_participant(study), + "demography": demography(study), + "visit": visits(study), + "blood_pressure": blood_pressure(study), + "height": simple_measure(study, "clinical", "HEIGHT_CM", "OMOP:3036277", "cm"), + "weight": simple_measure(study, "clinical", "WEIGHT_KG", "OMOP:3025315", "kg"), + "bmi": simple_measure( + study, "clinical", "BMI", "OMOP:3038553", bmi_unit, + categorical=study.bmi_categorical, + ), + "hdl": hdl(study), + # BUN is reported as an average, which the qualifier slot records. + "bun": simple_measure( + study, "labs", "BUN", "OMOP:3013682", "mg/dL", + extra=' qualifier:\n value: "AVERAGE"\n', + ), + "wbc": wbc(study), + "cond_heart_failure": condition( + study, "heart_failure", "HEART_FAILURE", "HF_CONCEPT", v.SELF_REPORTED_CONDITION), + "cond_family_stroke": condition( + study, "family_stroke", "FAM_STROKE", "FS_CONCEPT", v.SELF_REPORTED_CONDITION, + relationship=phv(study, "conditions", "FS_RELATIVE"), + ), + "cond_hypertension": condition( + study, "hypertension", "HYPERTENSION", None, v.SELF_REPORTED_CONDITION), + "cond_heart_attack": condition( + study, "heart_attack", "HEART_ATTACK", "HA_CONCEPT", v.SELF_REPORTED_CONDITION), + "drug_exposure": drug_exposure(study), + } + return specs + + +def main(): + """Write specs for both studies.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--out", type=Path, default=Path(__file__).parent / "specs") + args = parser.parse_args() + + for study in (pop.STUDY_ONE, pop.STUDY_TWO): + slug = study.name.lower().replace(" ", "_") + out = args.out / slug + out.mkdir(parents=True, exist_ok=True) + for name, body in build(study).items(): + (out / f"{name}.yaml").write_text(body) + print(f"{study.name}: {len(build(study))} specs -> {out}") + + +if __name__ == "__main__": + main() diff --git a/synthetic/specs/example_study_one/blood_pressure.yaml b/synthetic/specs/example_study_one/blood_pressure.yaml new file mode 100644 index 0000000..9bb3b4e --- /dev/null +++ b/synthetic/specs/example_study_one/blood_pressure.yaml @@ -0,0 +1,60 @@ +- class_derivations: + MeasurementObservationSet: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservationSet", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":bp")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111300}) + ":Example Study One")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10111300}) + ":Example Study One VISIT " + str({phv10111301}))' + observations: + class_derivations: + - MeasurementObservation: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":systolic")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111300}) + ":Example Study One")' + observation_type: + value: OMOP:4152194 + body_position: + value: "SITTING_POSITION" + body_site: + value: "Right arm" + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":systolic")' + value_decimal: + populated_from: phv10111305 + unit: + value: "mm[Hg]" + - MeasurementObservation: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":diastolic")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111300}) + ":Example Study One")' + observation_type: + value: OMOP:4154790 + body_position: + value: "SITTING_POSITION" + body_site: + value: "Right arm" + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":diastolic")' + value_decimal: + populated_from: phv10111306 + unit: + value: "mm[Hg]" diff --git a/synthetic/specs/example_study_one/bmi.yaml b/synthetic/specs/example_study_one/bmi.yaml new file mode 100644 index 0000000..3002d9a --- /dev/null +++ b/synthetic/specs/example_study_one/bmi.yaml @@ -0,0 +1,23 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":BMI")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111300}) + ":Example Study One")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10111300}) + ":Example Study One VISIT " + str({phv10111301}))' + observation_type: + value: OMOP:3038553 + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":BMI")' + value_decimal: + populated_from: phv10111304 + unit: + value: "kg/m2" diff --git a/synthetic/specs/example_study_one/bun.yaml b/synthetic/specs/example_study_one/bun.yaml new file mode 100644 index 0000000..74fb89a --- /dev/null +++ b/synthetic/specs/example_study_one/bun.yaml @@ -0,0 +1,25 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000114 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10111400}) + ":Example Study One:" + str({phv10111401}) + ":BUN")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111400}) + ":Example Study One")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10111400}) + ":Example Study One VISIT " + str({phv10111401}))' + observation_type: + value: OMOP:3013682 + qualifier: + value: "AVERAGE" + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000114 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10111400}) + ":Example Study One:" + str({phv10111401}) + ":BUN")' + value_decimal: + populated_from: phv10111404 + unit: + value: "mg/dL" diff --git a/synthetic/specs/example_study_one/cond_family_stroke.yaml b/synthetic/specs/example_study_one/cond_family_stroke.yaml new file mode 100644 index 0000000..e9f0aea --- /dev/null +++ b/synthetic/specs/example_study_one/cond_family_stroke.yaml @@ -0,0 +1,16 @@ +- class_derivations: + Condition: + populated_from: pht000115 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Condition", str({phv10111500}) + ":Example Study One:family_stroke")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111500}) + ":Example Study One")' + condition_concept: + populated_from: phv10111504 + condition_status: + populated_from: phv10111503 + condition_provenance: + value: PATIENT_SELF-REPORTED_CONDITION + relationship_to_participant: + populated_from: phv10111505 diff --git a/synthetic/specs/example_study_one/cond_heart_attack.yaml b/synthetic/specs/example_study_one/cond_heart_attack.yaml new file mode 100644 index 0000000..b9e4774 --- /dev/null +++ b/synthetic/specs/example_study_one/cond_heart_attack.yaml @@ -0,0 +1,16 @@ +- class_derivations: + Condition: + populated_from: pht000115 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Condition", str({phv10111500}) + ":Example Study One:heart_attack")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111500}) + ":Example Study One")' + condition_concept: + populated_from: phv10111508 + condition_status: + populated_from: phv10111507 + condition_provenance: + value: PATIENT_SELF-REPORTED_CONDITION + relationship_to_participant: + value: ONESELF diff --git a/synthetic/specs/example_study_one/cond_heart_failure.yaml b/synthetic/specs/example_study_one/cond_heart_failure.yaml new file mode 100644 index 0000000..5bfeeb3 --- /dev/null +++ b/synthetic/specs/example_study_one/cond_heart_failure.yaml @@ -0,0 +1,16 @@ +- class_derivations: + Condition: + populated_from: pht000115 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Condition", str({phv10111500}) + ":Example Study One:heart_failure")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111500}) + ":Example Study One")' + condition_concept: + populated_from: phv10111502 + condition_status: + populated_from: phv10111501 + condition_provenance: + value: PATIENT_SELF-REPORTED_CONDITION + relationship_to_participant: + value: ONESELF diff --git a/synthetic/specs/example_study_one/cond_hypertension.yaml b/synthetic/specs/example_study_one/cond_hypertension.yaml new file mode 100644 index 0000000..060428d --- /dev/null +++ b/synthetic/specs/example_study_one/cond_hypertension.yaml @@ -0,0 +1,16 @@ +- class_derivations: + Condition: + populated_from: pht000115 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Condition", str({phv10111500}) + ":Example Study One:hypertension")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111500}) + ":Example Study One")' + condition_concept: + value: HP:0000822 + condition_status: + populated_from: phv10111506 + condition_provenance: + value: PATIENT_SELF-REPORTED_CONDITION + relationship_to_participant: + value: ONESELF diff --git a/synthetic/specs/example_study_one/demography.yaml b/synthetic/specs/example_study_one/demography.yaml new file mode 100644 index 0000000..2eda7ac --- /dev/null +++ b/synthetic/specs/example_study_one/demography.yaml @@ -0,0 +1,14 @@ +- class_derivations: + Demography: + populated_from: pht000111 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Demography", str({phv10111100}) + ":Example Study One:demography")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111100}) + ":Example Study One")' + sex: + populated_from: phv10111101 + race: + populated_from: phv10111102 + ethnicity: + populated_from: phv10111103 diff --git a/synthetic/specs/example_study_one/drug_exposure.yaml b/synthetic/specs/example_study_one/drug_exposure.yaml new file mode 100644 index 0000000..b4e0939 --- /dev/null +++ b/synthetic/specs/example_study_one/drug_exposure.yaml @@ -0,0 +1,14 @@ +- class_derivations: + DrugExposure: + populated_from: pht000116 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/DrugExposure", str({phv10111600}) + ":Example Study One:ccb")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111600}) + ":Example Study One")' + drug_concept: + populated_from: phv10111602 + exposure_status: + populated_from: phv10111601 + exposure_provenance: + value: PATIENT_SELF_REPORTED_MEDICATION diff --git a/synthetic/specs/example_study_one/hdl.yaml b/synthetic/specs/example_study_one/hdl.yaml new file mode 100644 index 0000000..7374b82 --- /dev/null +++ b/synthetic/specs/example_study_one/hdl.yaml @@ -0,0 +1,56 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000114 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10111400}) + ":Example Study One:" + str({phv10111401}) + ":HDL")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111400}) + ":Example Study One")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10111400}) + ":Example Study One VISIT " + str({phv10111401}))' + observation_type: + value: OMOP:3007070 + associated_assay: + class_derivations: + - Assay: + populated_from: pht000114 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Assay", "Example Study One:HDL")' + method: + value: MMO:0000133 + lower_limit_of_detection: + class_derivations: + - Quantity: + populated_from: pht000114 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", "Example Study One:HDL:LLOD")' + value_decimal: + value: 5.0 + unit: + value: "mg/dL" + upper_limit_of_detection: + class_derivations: + - Quantity: + populated_from: pht000114 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", "Example Study One:HDL:ULOD")' + value_decimal: + value: 150.0 + unit: + value: "mg/dL" + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000114 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10111400}) + ":Example Study One:" + str({phv10111401}) + ":HDL")' + value_decimal: + populated_from: phv10111402 + operator: + populated_from: phv10111403 + unit: + value: "mg/dL" diff --git a/synthetic/specs/example_study_one/height.yaml b/synthetic/specs/example_study_one/height.yaml new file mode 100644 index 0000000..f44ebb2 --- /dev/null +++ b/synthetic/specs/example_study_one/height.yaml @@ -0,0 +1,23 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":HEIGHT_CM")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111300}) + ":Example Study One")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10111300}) + ":Example Study One VISIT " + str({phv10111301}))' + observation_type: + value: OMOP:3036277 + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":HEIGHT_CM")' + value_decimal: + populated_from: phv10111302 + unit: + value: "cm" diff --git a/synthetic/specs/example_study_one/person_participant.yaml b/synthetic/specs/example_study_one/person_participant.yaml new file mode 100644 index 0000000..b3bf582 --- /dev/null +++ b/synthetic/specs/example_study_one/person_participant.yaml @@ -0,0 +1,35 @@ +- class_derivations: + Person: + populated_from: pht000111 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Person", str({dbGaP_Subject_ID}))' + identity: + expr: 'str({dbGaP_Subject_ID})' + species: + value: NCBITaxon:9606 + vital_status: + expr: 'case(({phv10111105} == 0, "OMOP:4230556"), ({phv10111105} == 1, "OMOP:434489"))' + cause_of_death: + class_derivations: + - CauseOfDeath: + populated_from: pht000111 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/CauseOfDeath", str({dbGaP_Subject_ID}))' + cause: + populated_from: phv10111106 + order: + expr: 'case(({phv10111105} == 1, 1))' +- class_derivations: + Participant: + populated_from: pht000111 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111100}) + ":Example Study One")' + identity: + expr: 'str({phv10111100})' + member_of_research_study: + value: 'Example Study One' + associated_person: + expr: 'uuid5("https://w3id.org/bdchm/Person", str({dbGaP_Subject_ID}))' diff --git a/synthetic/specs/example_study_one/visit.yaml b/synthetic/specs/example_study_one/visit.yaml new file mode 100644 index 0000000..a7d38a1 --- /dev/null +++ b/synthetic/specs/example_study_one/visit.yaml @@ -0,0 +1,14 @@ +- class_derivations: + Visit: + populated_from: pht000112 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10111200}) + ":Example Study One VISIT " + str({phv10111201}))' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111200}) + ":Example Study One")' + visit_category: + populated_from: phv10111202 + age_at_visit_start: + populated_from: phv10111203 + visit_provenance: + value: CASE_REPORT_FORM diff --git a/synthetic/specs/example_study_one/wbc.yaml b/synthetic/specs/example_study_one/wbc.yaml new file mode 100644 index 0000000..ef3c46e --- /dev/null +++ b/synthetic/specs/example_study_one/wbc.yaml @@ -0,0 +1,32 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000114 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10111400}) + ":Example Study One:" + str({phv10111401}) + ":WBC")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111400}) + ":Example Study One")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10111400}) + ":Example Study One VISIT " + str({phv10111401}))' + observation_type: + value: OMOP:3000905 + associated_assay: + class_derivations: + - Assay: + populated_from: pht000114 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Assay", "Example Study One:CBC")' + method: + value: MMO:0000533 + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000114 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10111400}) + ":Example Study One:" + str({phv10111401}) + ":WBC")' + value_decimal: + populated_from: phv10111405 + unit: + value: "10*3/uL" diff --git a/synthetic/specs/example_study_one/weight.yaml b/synthetic/specs/example_study_one/weight.yaml new file mode 100644 index 0000000..a5e78c4 --- /dev/null +++ b/synthetic/specs/example_study_one/weight.yaml @@ -0,0 +1,23 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":WEIGHT_KG")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10111300}) + ":Example Study One")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10111300}) + ":Example Study One VISIT " + str({phv10111301}))' + observation_type: + value: OMOP:3025315 + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000113 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10111300}) + ":Example Study One:" + str({phv10111301}) + ":WEIGHT_KG")' + value_decimal: + populated_from: phv10111303 + unit: + value: "kg" diff --git a/synthetic/specs/example_study_two/blood_pressure.yaml b/synthetic/specs/example_study_two/blood_pressure.yaml new file mode 100644 index 0000000..043fbff --- /dev/null +++ b/synthetic/specs/example_study_two/blood_pressure.yaml @@ -0,0 +1,60 @@ +- class_derivations: + MeasurementObservationSet: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservationSet", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":bp")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212300}) + ":Example Study Two")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10212300}) + ":Example Study Two VISIT " + str({phv10212301}))' + observations: + class_derivations: + - MeasurementObservation: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":systolic")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212300}) + ":Example Study Two")' + observation_type: + value: OMOP:4152194 + body_position: + value: "SITTING_POSITION" + body_site: + value: "Right arm" + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":systolic")' + value_decimal: + populated_from: phv10212305 + unit: + value: "mm[Hg]" + - MeasurementObservation: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":diastolic")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212300}) + ":Example Study Two")' + observation_type: + value: OMOP:4154790 + body_position: + value: "SITTING_POSITION" + body_site: + value: "Right arm" + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":diastolic")' + value_decimal: + populated_from: phv10212306 + unit: + value: "mm[Hg]" diff --git a/synthetic/specs/example_study_two/bmi.yaml b/synthetic/specs/example_study_two/bmi.yaml new file mode 100644 index 0000000..97e28fb --- /dev/null +++ b/synthetic/specs/example_study_two/bmi.yaml @@ -0,0 +1,21 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":BMI")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212300}) + ":Example Study Two")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10212300}) + ":Example Study Two VISIT " + str({phv10212301}))' + observation_type: + value: OMOP:3038553 + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":BMI")' + value_concept: + populated_from: phv10212304 diff --git a/synthetic/specs/example_study_two/bun.yaml b/synthetic/specs/example_study_two/bun.yaml new file mode 100644 index 0000000..4b26615 --- /dev/null +++ b/synthetic/specs/example_study_two/bun.yaml @@ -0,0 +1,25 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000124 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10212400}) + ":Example Study Two:" + str({phv10212401}) + ":BUN")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212400}) + ":Example Study Two")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10212400}) + ":Example Study Two VISIT " + str({phv10212401}))' + observation_type: + value: OMOP:3013682 + qualifier: + value: "AVERAGE" + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000124 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10212400}) + ":Example Study Two:" + str({phv10212401}) + ":BUN")' + value_decimal: + populated_from: phv10212404 + unit: + value: "mg/dL" diff --git a/synthetic/specs/example_study_two/cond_family_stroke.yaml b/synthetic/specs/example_study_two/cond_family_stroke.yaml new file mode 100644 index 0000000..c51161a --- /dev/null +++ b/synthetic/specs/example_study_two/cond_family_stroke.yaml @@ -0,0 +1,16 @@ +- class_derivations: + Condition: + populated_from: pht000125 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Condition", str({phv10212500}) + ":Example Study Two:family_stroke")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212500}) + ":Example Study Two")' + condition_concept: + populated_from: phv10212504 + condition_status: + populated_from: phv10212503 + condition_provenance: + value: PATIENT_SELF-REPORTED_CONDITION + relationship_to_participant: + populated_from: phv10212505 diff --git a/synthetic/specs/example_study_two/cond_heart_attack.yaml b/synthetic/specs/example_study_two/cond_heart_attack.yaml new file mode 100644 index 0000000..e1d3965 --- /dev/null +++ b/synthetic/specs/example_study_two/cond_heart_attack.yaml @@ -0,0 +1,16 @@ +- class_derivations: + Condition: + populated_from: pht000125 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Condition", str({phv10212500}) + ":Example Study Two:heart_attack")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212500}) + ":Example Study Two")' + condition_concept: + populated_from: phv10212508 + condition_status: + populated_from: phv10212507 + condition_provenance: + value: PATIENT_SELF-REPORTED_CONDITION + relationship_to_participant: + value: ONESELF diff --git a/synthetic/specs/example_study_two/cond_heart_failure.yaml b/synthetic/specs/example_study_two/cond_heart_failure.yaml new file mode 100644 index 0000000..16afe07 --- /dev/null +++ b/synthetic/specs/example_study_two/cond_heart_failure.yaml @@ -0,0 +1,16 @@ +- class_derivations: + Condition: + populated_from: pht000125 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Condition", str({phv10212500}) + ":Example Study Two:heart_failure")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212500}) + ":Example Study Two")' + condition_concept: + populated_from: phv10212502 + condition_status: + populated_from: phv10212501 + condition_provenance: + value: PATIENT_SELF-REPORTED_CONDITION + relationship_to_participant: + value: ONESELF diff --git a/synthetic/specs/example_study_two/cond_hypertension.yaml b/synthetic/specs/example_study_two/cond_hypertension.yaml new file mode 100644 index 0000000..b980ae9 --- /dev/null +++ b/synthetic/specs/example_study_two/cond_hypertension.yaml @@ -0,0 +1,16 @@ +- class_derivations: + Condition: + populated_from: pht000125 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Condition", str({phv10212500}) + ":Example Study Two:hypertension")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212500}) + ":Example Study Two")' + condition_concept: + value: HP:0000822 + condition_status: + populated_from: phv10212506 + condition_provenance: + value: PATIENT_SELF-REPORTED_CONDITION + relationship_to_participant: + value: ONESELF diff --git a/synthetic/specs/example_study_two/demography.yaml b/synthetic/specs/example_study_two/demography.yaml new file mode 100644 index 0000000..a1b5068 --- /dev/null +++ b/synthetic/specs/example_study_two/demography.yaml @@ -0,0 +1,14 @@ +- class_derivations: + Demography: + populated_from: pht000121 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Demography", str({phv10212100}) + ":Example Study Two:demography")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212100}) + ":Example Study Two")' + sex: + populated_from: phv10212101 + race: + populated_from: phv10212102 + ethnicity: + populated_from: phv10212103 diff --git a/synthetic/specs/example_study_two/drug_exposure.yaml b/synthetic/specs/example_study_two/drug_exposure.yaml new file mode 100644 index 0000000..4d5392e --- /dev/null +++ b/synthetic/specs/example_study_two/drug_exposure.yaml @@ -0,0 +1,14 @@ +- class_derivations: + DrugExposure: + populated_from: pht000126 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/DrugExposure", str({phv10212600}) + ":Example Study Two:ccb")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212600}) + ":Example Study Two")' + drug_concept: + populated_from: phv10212602 + exposure_status: + populated_from: phv10212601 + exposure_provenance: + value: PATIENT_SELF_REPORTED_MEDICATION diff --git a/synthetic/specs/example_study_two/hdl.yaml b/synthetic/specs/example_study_two/hdl.yaml new file mode 100644 index 0000000..cccf6c0 --- /dev/null +++ b/synthetic/specs/example_study_two/hdl.yaml @@ -0,0 +1,56 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000124 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10212400}) + ":Example Study Two:" + str({phv10212401}) + ":HDL")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212400}) + ":Example Study Two")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10212400}) + ":Example Study Two VISIT " + str({phv10212401}))' + observation_type: + value: OMOP:3007070 + associated_assay: + class_derivations: + - Assay: + populated_from: pht000124 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Assay", "Example Study Two:HDL")' + method: + value: MMO:0000133 + lower_limit_of_detection: + class_derivations: + - Quantity: + populated_from: pht000124 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", "Example Study Two:HDL:LLOD")' + value_decimal: + value: 5.0 + unit: + value: "mg/dL" + upper_limit_of_detection: + class_derivations: + - Quantity: + populated_from: pht000124 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", "Example Study Two:HDL:ULOD")' + value_decimal: + value: 150.0 + unit: + value: "mg/dL" + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000124 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10212400}) + ":Example Study Two:" + str({phv10212401}) + ":HDL")' + value_decimal: + populated_from: phv10212402 + operator: + populated_from: phv10212403 + unit: + value: "mg/dL" diff --git a/synthetic/specs/example_study_two/height.yaml b/synthetic/specs/example_study_two/height.yaml new file mode 100644 index 0000000..fdca786 --- /dev/null +++ b/synthetic/specs/example_study_two/height.yaml @@ -0,0 +1,23 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":HEIGHT_CM")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212300}) + ":Example Study Two")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10212300}) + ":Example Study Two VISIT " + str({phv10212301}))' + observation_type: + value: OMOP:3036277 + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":HEIGHT_CM")' + value_decimal: + populated_from: phv10212302 + unit: + value: "cm" diff --git a/synthetic/specs/example_study_two/person_participant.yaml b/synthetic/specs/example_study_two/person_participant.yaml new file mode 100644 index 0000000..d7d1780 --- /dev/null +++ b/synthetic/specs/example_study_two/person_participant.yaml @@ -0,0 +1,35 @@ +- class_derivations: + Person: + populated_from: pht000121 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Person", str({dbGaP_Subject_ID}))' + identity: + expr: 'str({dbGaP_Subject_ID})' + species: + value: NCBITaxon:9606 + vital_status: + expr: 'case(({phv10212105} == 0, "OMOP:4230556"), ({phv10212105} == 1, "OMOP:434489"))' + cause_of_death: + class_derivations: + - CauseOfDeath: + populated_from: pht000121 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/CauseOfDeath", str({dbGaP_Subject_ID}))' + cause: + populated_from: phv10212106 + order: + expr: 'case(({phv10212105} == 1, 1))' +- class_derivations: + Participant: + populated_from: pht000121 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212100}) + ":Example Study Two")' + identity: + expr: 'str({phv10212100})' + member_of_research_study: + value: 'Example Study Two' + associated_person: + expr: 'uuid5("https://w3id.org/bdchm/Person", str({dbGaP_Subject_ID}))' diff --git a/synthetic/specs/example_study_two/visit.yaml b/synthetic/specs/example_study_two/visit.yaml new file mode 100644 index 0000000..b26d1f0 --- /dev/null +++ b/synthetic/specs/example_study_two/visit.yaml @@ -0,0 +1,14 @@ +- class_derivations: + Visit: + populated_from: pht000122 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10212200}) + ":Example Study Two VISIT " + str({phv10212201}))' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212200}) + ":Example Study Two")' + visit_category: + populated_from: phv10212202 + age_at_visit_start: + populated_from: phv10212203 + visit_provenance: + value: CASE_REPORT_FORM diff --git a/synthetic/specs/example_study_two/wbc.yaml b/synthetic/specs/example_study_two/wbc.yaml new file mode 100644 index 0000000..6f34c7a --- /dev/null +++ b/synthetic/specs/example_study_two/wbc.yaml @@ -0,0 +1,32 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000124 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10212400}) + ":Example Study Two:" + str({phv10212401}) + ":WBC")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212400}) + ":Example Study Two")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10212400}) + ":Example Study Two VISIT " + str({phv10212401}))' + observation_type: + value: OMOP:3000905 + associated_assay: + class_derivations: + - Assay: + populated_from: pht000124 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Assay", "Example Study Two:CBC")' + method: + value: MMO:0000533 + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000124 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10212400}) + ":Example Study Two:" + str({phv10212401}) + ":WBC")' + value_decimal: + populated_from: phv10212405 + unit: + value: "10*3/uL" diff --git a/synthetic/specs/example_study_two/weight.yaml b/synthetic/specs/example_study_two/weight.yaml new file mode 100644 index 0000000..930ac0d --- /dev/null +++ b/synthetic/specs/example_study_two/weight.yaml @@ -0,0 +1,23 @@ +- class_derivations: + MeasurementObservation: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/MeasurementObservation", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":WEIGHT_KG")' + associated_participant: + expr: 'uuid5("https://w3id.org/bdchm/Participant", str({phv10212300}) + ":Example Study Two")' + associated_visit: + expr: 'uuid5("https://w3id.org/bdchm/Visit", str({phv10212300}) + ":Example Study Two VISIT " + str({phv10212301}))' + observation_type: + value: OMOP:3025315 + value_quantity: + class_derivations: + - Quantity: + populated_from: pht000123 + slot_derivations: + id: + expr: 'uuid5("https://w3id.org/bdchm/Quantity", str({phv10212300}) + ":Example Study Two:" + str({phv10212301}) + ":WEIGHT_KG")' + value_decimal: + populated_from: phv10212303 + unit: + value: "kg" diff --git a/synthetic/validate.py b/synthetic/validate.py new file mode 100644 index 0000000..42c9a37 --- /dev/null +++ b/synthetic/validate.py @@ -0,0 +1,154 @@ +""" +Check the corpus against the brief and against its own internal consistency. + +Distribution checks use a tolerance because sampling variation is expected and +wanted — the brief says as much. Invariant checks are exact: a violation there +is a bug, not noise. + + python synthetic/validate.py +""" + +import sys +from collections import Counter + +import population as pop +import vocab as v + +TOLERANCE = 0.05 + +failures = [] +notes = [] + + +def near(label, actual, expected, tolerance=TOLERANCE): + """Record a distribution check that allows sampling variation.""" + ok = abs(actual - expected) <= tolerance + notes.append(f"{'ok ' if ok else 'FAIL'} {label}: {actual:.3f} (expected ~{expected:.3f})") + if not ok: + failures.append(label) + + +def exact(label, actual, expected): + """Record an invariant check that must hold exactly.""" + ok = actual == expected + notes.append(f"{'ok ' if ok else 'FAIL'} {label}: {actual} (expected {expected})") + if not ok: + failures.append(label) + + +def fraction(items, predicate): + """Return the proportion of items satisfying the predicate.""" + items = list(items) + return sum(1 for x in items if predicate(x)) / len(items) if items else 0.0 + + +def main(): + """Run every check and report.""" + people, participants = pop.build() + + for study in (pop.STUDY_ONE, pop.STUDY_TWO): + group = [p for p in participants if p.study is study] + label = study.name + + exact(f"{label}: participants", len(group), study.size) + near(f"{label}: male", fraction(group, lambda p: p.person.sex == v.MALE), study.male_fraction) + + for race, expected in study.race_mix.items(): + near(f"{label}: race {race}", fraction(group, lambda p, r=race: p.person.race == r), expected) + + near(f"{label}: deceased", fraction(group, lambda p: p.person.deceased), 0.10) + + # Hispanic ethnicity is conditioned on race, so check it that way. + white = [p for p in group if p.person.race == v.WHITE] + black = [p for p in group if p.person.race == v.BLACK] + near(f"{label}: hispanic|white", fraction(white, lambda p: p.person.ethnicity == v.HISPANIC), 0.10) + near(f"{label}: hispanic|black", fraction(black, lambda p: p.person.ethnicity == v.HISPANIC), 0.05) + + # Visit structure + max_visits = max(len(p.visits) for p in group) + exact(f"{label}: max visits", max_visits, study.visits) + bad_telehealth = [ + p for p in group + if sum(1 for x in p.visits if x.category == v.TELEHEALTH) > 1 + ] + exact(f"{label}: participants with >1 telehealth visit", len(bad_telehealth), 0) + + # Nobody is measured after they die: visit ages must increase and the + # deceased attend fewer visits than the living. + non_monotonic = [ + p for p in group + if [x.age_days for x in p.visits] != sorted(x.age_days for x in p.visits) + ] + exact(f"{label}: non-monotonic visit ages", len(non_monotonic), 0) + + near(f"{label}: on a calcium channel blocker", fraction(group, lambda p: p.ccb), 0.20) + + # Condition proportions + near(f"{label}: heart failure PRESENT", + fraction(group, lambda p: p.conditions["heart_failure"]["status"] == v.PRESENT), 0.15) + near(f"{label}: family stroke PRESENT", + fraction(group, lambda p: p.conditions["family_stroke"]["status"] == v.PRESENT), 0.35) + near(f"{label}: family stroke UNKNOWN", + fraction(group, lambda p: p.conditions["family_stroke"]["status"] == v.UNKNOWN), 0.05) + near(f"{label}: hypertension PRESENT", + fraction(group, lambda p: p.conditions["hypertension"]["status"] == v.PRESENT), 0.10) + near(f"{label}: hypertension HISTORICAL", + fraction(group, lambda p: p.conditions["hypertension"]["status"] == v.HISTORICAL), 0.10) + near(f"{label}: heart attack PRESENT", + fraction(group, lambda p: p.conditions["heart_attack"]["status"] == v.PRESENT), 0.10) + + mi = [p for p in group if p.conditions["heart_attack"]["status"] == v.PRESENT] + near(f"{label}: MI from study record", + fraction(mi, lambda p: p.conditions["heart_attack"]["from_study_record"]), 0.50, tolerance=0.15) + + # Cross-study: 5% of Study Two shares an identity with Study One. + one_ids = {p.person.dbgap_id for p in participants if p.study is pop.STUDY_ONE} + two = [p for p in participants if p.study is pop.STUDY_TWO] + shared = [p for p in two if p.person.dbgap_id in one_ids] + near("shared participants", len(shared) / len(two), 0.05, tolerance=0.005) + + # A shared individual must be one Person but two Participants. + subject_ids = {p.subject_id for p in participants} + exact("distinct study-local subject ids", len(subject_ids), len(participants)) + exact("distinct people", len(people), len(participants) - len(shared)) + + # Exactly five censored HDL results, each carrying the operator. + censored = [x for p in participants for x in p.visits if x.hdl_operator] + exact("HDL results below LLOD", len(censored), pop.HDL_BELOW_LLOD) + exact("censored results carrying '<'", sum(1 for x in censored if x.hdl_operator == "<"), pop.HDL_BELOW_LLOD) + + # Null sprinkling across measured cells. + cells = [ + getattr(x, field) + for p in participants + for x in p.visits + for field in ("height_cm", "weight_kg", "systolic", "diastolic", "hdl", "bun", "wbc") + ] + near("null rate", sum(1 for c in cells if c is None) / len(cells), pop.NULL_RATE, tolerance=0.002) + + # The four calcium channel blockers should be evenly divided. Guard the + # denominator: a generation change that produced no exposures should fail a + # check, not crash the run partway through. + ccb = Counter(p.ccb for p in participants if p.ccb) + total = sum(ccb.values()) + exact("any calcium channel blocker exposures", total > 0, True) + # Only ~200 participants are exposed, so a quarter-share carries a standard + # error near 0.03. At 2 sd, one of these four checks fails on roughly one + # seed in six; 3 sd still catches a genuinely skewed choice. + if total: + for concept in v.CALCIUM_CHANNEL_BLOCKERS: + near(f"CCB share {concept}", ccb[concept] / total, 0.25, tolerance=0.09) + + print("\n".join(notes)) + print() + if failures: + print(f"{len(failures)} check(s) failed:") + for f in failures: + print(f" - {f}") + return 1 + print(f"All {len(notes)} checks passed.") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/synthetic/vocab.py b/synthetic/vocab.py new file mode 100644 index 0000000..e67b744 --- /dev/null +++ b/synthetic/vocab.py @@ -0,0 +1,98 @@ +# ruff: noqa: S311 +""" +Coded values used by the synthetic corpus. + +Every CURIE here is attested in RTI's priority_variables_transform specs or in +BDCHM itself, so the corpus resolves against the same vocabulary as real +harmonized data. Nothing in this file is invented. +""" + +# Demography — BDCHM RaceEnum / EthnicityEnum / SexEnum meanings +MALE = "OMOP:8507" +FEMALE = "OMOP:8532" + +WHITE = "OMOP:8527" +BLACK = "OMOP:8516" +ASIAN = "OMOP:8515" +AMERICAN_INDIAN = "OMOP:8657" +MIDDLE_EASTERN = "OMOP:38003615" +NATIVE_HAWAIIAN = "OMOP:8557" + +HISPANIC = "OMOP:38003563" +NOT_HISPANIC = "OMOP:38003564" + +# Vital status, as coded in MESA-ingest/person.yaml +ALIVE = "OMOP:4230556" +DECEASED = "OMOP:434489" + +SPECIES_HUMAN = "NCBITaxon:9606" + +# The three most frequent ICD10CM causes of death across the real specs. +TOP_DEATH_CAUSES = [ + "ICD10CM:I20-I25", # ischaemic heart disease + "ICD10CM:R99", # ill-defined / unknown cause + "ICD10CM:I00-I99", # circulatory system, unspecified +] + +# Conditions — from hist_heart_failure, fam_stroke, hypertension, hist_mi +HEART_FAILURE = ["MONDO:0005009", "MONDO:0005252"] +STROKE = [ + "MONDO:0005099", + "MONDO:0005264", + "MONDO:0006809", + "MONDO:0011057", + "MONDO:0013792", +] +HYPERTENSION = "HP:0000822" +HEART_ATTACK = ["MONDO:0005068", "MONDO:0006803"] + +# Observation types — from blood_pressure.yaml and the lab specs +SYSTOLIC = "OMOP:4152194" +DIASTOLIC = "OMOP:4154790" + +# Drug concepts — all four attested in tak_calchanblk specs +CALCIUM_CHANNEL_BLOCKERS = [ + "ATC:C08", + "NDFRT:N0000175421", + "RxCUI:7417", + "RxCUI:17767", +] + +# Family relationship — BDCHM FamilyRelationshipEnum meanings. +# The brief said "FATHER or MOTHER"; the model spells these NATURAL_*. +NATURAL_FATHER = "OMOP:4321888" +NATURAL_MOTHER = "OMOP:4277283" + +# Provenance. Note the hyphen in the condition value — it is in the enum. +SELF_REPORTED_CONDITION = "PATIENT_SELF-REPORTED_CONDITION" +SELF_REPORTED_MEDICATION = "PATIENT_SELF_REPORTED_MEDICATION" +STUDY_RECORD = "STUDY_RECORD" + +# Status values — BDCHM HistoricalStatusEnum +PRESENT = "PRESENT" +ABSENT = "ABSENT" +UNKNOWN = "UNKNOWN" +HISTORICAL = "HISTORICAL" + +# Visit categories — BDCHM VisitTypeEnum +STUDY_SITE_VISIT = "STUDY_SITE_VISIT" +TELEHEALTH = "TELEHEALTH" + +# Visit provenance — BDCHM VisitProvenanceEnum. The real specs leave this unset +# even though it is required; a reference corpus should populate it. +VISIT_PROVENANCE = "CASE_REPORT_FORM" + +# Assay methods — BDCHM AssayMethodEnum resolves through MMO. No MMO term is +# attested in the real specs, so these were looked up in MMO directly rather +# than invented. +ASSAY_METHOD_HDL = "MMO:0000133" # serum high-density lipoprotein-cholesterol measurement test +ASSAY_METHOD_WBC = "MMO:0000533" # white blood cell counting method + +# Relationship of a condition to the participant. ONESELF has no OMOP meaning +# in BDCHM — it is a bare permissible value. +ONESELF = "ONESELF" +CONDITION_NS = "https://w3id.org/bdchm/Condition" +DEMOGRAPHY_NS = "https://w3id.org/bdchm/Demography" +DRUG_EXPOSURE_NS = "https://w3id.org/bdchm/DrugExposure" +OBSERVATION_SET_NS = "https://w3id.org/bdchm/MeasurementObservationSet" +CAUSE_OF_DEATH_NS = "https://w3id.org/bdchm/CauseOfDeath"