From 5ba858fc608fa629fa0b8c2fd0528861b3f15095 Mon Sep 17 00:00:00 2001 From: amc-corey-cox <69321580+amc-corey-cox@users.noreply.github.com> Date: Thu, 10 Sep 2026 10:37:21 -0500 Subject: [PATCH 1/2] Prefix synthetic raw tables with SYNTHETIC so they cannot be mistaken for dbGaP exports --- synthetic/README.md | 9 +++++++-- synthetic/generate.py | 13 ++++++++++++- 2 files changed, 19 insertions(+), 3 deletions(-) diff --git a/synthetic/README.md b/synthetic/README.md index 86114be..511e01e 100644 --- a/synthetic/README.md +++ b/synthetic/README.md @@ -16,10 +16,15 @@ second implementation of the transformation, free to drift from the real one in ways nobody would notice until a portal built on it met real data. ``` -generate.py -> data/raw/*.txt.gz -> dm-bip map-data -> harmonized BDCHM -specs.py -> specs/*/*.yaml -> ^ +generate.py -> data/raw/SYNTHETIC.*.txt.gz -> dm-bip map-data -> harmonized BDCHM +specs.py -> specs/*/*.yaml -> ^ ``` +The raw tables follow the dbGaP naming convention exactly apart from a +`SYNTHETIC.` prefix. Without it a file lifted out of its directory is +indistinguishable from a controlled-access export, and a file is identified by +its name rather than by the citation line in its header. + ## Running it ```bash diff --git a/synthetic/generate.py b/synthetic/generate.py index ee0c136..d872c7e 100644 --- a/synthetic/generate.py +++ b/synthetic/generate.py @@ -18,6 +18,17 @@ CITATION = "Synthetic corpus for portal development. Not derived from participant data." +# Every raw table is named so it cannot be mistaken for a controlled-access +# dbGaP export. The header already says so, but a file that escapes its +# directory is identified by its name, not by its contents, and these otherwise +# follow the dbGaP naming convention exactly. +# +# A prefix rather than a suffix: it is what a directory listing sorts on and +# shows first, and it survives the truncation that hides the middle of a long +# name. dm-bip finds the table accession with an unanchored search for +# `pht[0-9]+`, so the prefix does not disturb the pipeline. +SYNTHETIC_MARKER = "SYNTHETIC" + # Column layouts. The phv accessions are fictional but well-formed, and are # what the transformation specs reference. LAYOUTS = { @@ -93,7 +104,7 @@ def write_table(out_dir, study, table, rows): """Write one table in dbGaP raw format, returning its path.""" columns, phv_count = LAYOUTS[table] pht = study.tables[table] - path = out_dir / f"{study.phs}.v1.{pht}.v1.p1.c1.ex0_1s.HMB.txt.gz" + path = out_dir / f"{SYNTHETIC_MARKER}.{study.phs}.v1.{pht}.v1.p1.c1.ex0_1s.HMB.txt.gz" with gzip.open(path, "wt", newline="") as fh: fh.write(f"# Study accession: {study.phs}.v1.p1\n") From 4d22e20fb4bda02de781a86f0e8e66bd05ec004b Mon Sep 17 00:00:00 2001 From: amc-corey-cox <69321580+amc-corey-cox@users.noreply.github.com> Date: Thu, 10 Sep 2026 10:48:49 -0500 Subject: [PATCH 2/2] Address PR review feedback --- synthetic/README.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/synthetic/README.md b/synthetic/README.md index 511e01e..0fb8e35 100644 --- a/synthetic/README.md +++ b/synthetic/README.md @@ -16,8 +16,8 @@ second implementation of the transformation, free to drift from the real one in ways nobody would notice until a portal built on it met real data. ``` -generate.py -> data/raw/SYNTHETIC.*.txt.gz -> dm-bip map-data -> harmonized BDCHM -specs.py -> specs/*/*.yaml -> ^ +generate.py -> data/raw/*/SYNTHETIC.*.txt.gz -> dm-bip map-data -> harmonized BDCHM +specs.py -> specs/*/*.yaml -> ^ ``` The raw tables follow the dbGaP naming convention exactly apart from a