{
 "release": "gi-promoter-atlas-2026-07-31",
 "release_version": "1.0.0",
 "generated_utc": "2026-08-16T03:08:27Z",
 "title": "GI Promoter Atlas -- natural promoters, the schema-enforced artifact of record",
 "distribution_status": "THE SCHEMA-ENFORCED ARTIFACT OF RECORD, NOT A PUBLICATION. This tree is the packaged, checksummed and schema-validated form of what shipped, rebuilt from the ranked artifacts by one deterministic command. It is not published anywhere and nothing here depends on its being published; it exists so that what may be claimed is enforced by a machine -- required fields, required caveats, a privacy scan that must be able to fail -- rather than asserted by a sentence in a document. The research-use-only and not-wet-lab-validated statements below are claim rules and are correct regardless of who reads them.",
 "research_use_only": "RESEARCH USE ONLY. Every number in this release is a prediction from the g0-expression model on checkpoint 20260523. Nothing here has been validated in a wet lab, in any cell type, at any length. The margins are model outputs on a checkpoint with a measured compressed dynamic range, computed on natural genomic TSS windows -- not on AAV cassettes, which are out of distribution for this model. Do not use these sequences or rankings in a clinical, diagnostic or therapeutic decision.",
 "licence_note": "Licensed CC BY 4.0 -- the data, the tables, the documentation and the pages built from them. The pipeline that produced this tree is proprietary and unpublished; no licence is granted over it. Attribution asks for the release id and the model checkpoint named in this manifest. Research use only: every number here is a model prediction and nothing has been measured in a wet lab -- that is a statement about what this is, not a condition of the licence, which adds no field-of-use restriction. The underlying genome annotation is GENCODE / Ensembl and NCBI MANE under their own terms, and is redistributed here as coordinates and identifiers rather than sequence; the predictions are outputs of the Genomic Intelligence g0-expression model.",
 "relationship_to_the_generated_design_release": {
  "this_release": "natural human promoters found by screening the genome. Every entry is a real human TSS window, and the model was trained on those windows -- so a high rank here is partly recall of training data. See limitations.training_set_membership and recall_check.json.",
  "the_other_release": "generated designs -- `gi-promoter-atlas-designs-2026-08-05`, packaged as the sibling tree `releases/generated/`. 30 model-designed 600 bp promoter modules for two of these pairs (A1 and A2), scored in a fixed genomic scaffold. No design is an unmodified natural promoter, so no memorisation asterisk attaches to them.",
  "rule": "The two are separate releases with separate file names and separate limitation blocks. Do not merge them into one table or one ranking -- they carry different caveats. The designs are 600 bp modules scored in one named scaffold and their margins are only comparable within that scaffold; the natural candidates are 9,198 bp genomic windows. A ranking that mixed them would compare two different measurements."
 },
 "model": {
  "model_id": "g0-expression",
  "revision": "v1",
  "checkpoint": "20260523",
  "dna_encoder": "AIRI-Institute/moderngena-large",
  "dna_encoder_revision": "01de8202d4cf5e79d87d3e4f71f3e51ae923392b",
  "unit": "ln(quantile-normalised TPM + 1)",
  "unit_note": "Natural log, and the training targets were quantile-normalised -- which is why magnitudes are compressed. Compare deltas, never absolutes. expression_tpm in the API response is exp(expression) - 1 and inherits the same normalisation."
 },
 "window_recipe": {
  "assembly": "GRCh38",
  "annotation_release": "GENCODE release 50 (Ensembl 116)",
  "annotation_url": "https://ftp.ebi.ac.uk/pub/databases/gencode/Gencode_human/release_50/gencode.v50.annotation.gtf.gz",
  "annotation_sha256": "83fba3e9b03f0b8c958f3595c6c350adc55f468abf8b0e47b6d5284cfe13a453",
  "genome_fasta_url": "https://ftp.ebi.ac.uk/pub/databases/gencode/Gencode_human/release_50/GRCh38.primary_assembly.genome.fa.gz",
  "genome_fasta_sha256": "b760d18dbb651dd14dfc290083371b3ef3bff122d43a9cefb13ca4ecf38f05ca",
  "mane_release": "NCBI MANE v1.4",
  "mane_url": "https://ftp.ncbi.nlm.nih.gov/refseq/MANE/MANE_human/release_1.4/MANE.GRCh38.v1.4.summary.txt.gz",
  "mane_sha256": "4b3992457556e302a5e47e18a305bb4763718377696ca37d5a6b34df7db630d2",
  "canonical_transcript_rule": [
   "1. mane_select -- the transcript tagged MANE_Select in GENCODE V50 and present in NCBI MANE v1.4. Used for 142 of the 148 entries in this release.",
   "2. ensembl_canonical -- for genes with no MANE Select transcript, the transcript tagged Ensembl_canonical. Used for 6 entries.",
   "No entry in this release resolved past rule 2; the ambiguous-canonical rule never fired across all 20,107 genes."
  ],
  "tss_definition": "The 5' end of the canonical transcript in gene-sense orientation: the GTF transcript 'start' on the + strand, 'end' on the - strand.",
  "window_bp": 9198,
  "tss_offset_0based": 4599,
  "coordinates": "window.start and window.end are 1-based inclusive GRCh38 coordinates on the FORWARD strand, so end - start + 1 == 9198 for every entry.",
  "orientation": "The submitted sequence is GENE-SENSE. For a - strand gene, take the forward-strand span and reverse-complement it. After that step the TSS sits at 0-based offset 4599 in the submitted string for both strands.",
  "why_it_matters": "The model's input window is exactly 9,198 bp, TSS-centered, gene-sense. The server accepts 500-500,000 bp and returns a confidently wrong number for a mis-sized or mis-centered window, silently. Assert the length and the TSS offset client-side. Using gene-level rather than canonical-transcript start/end puts you kilobases off (HBB by 2,324 bp; ACTB by 33,301 bp). A 1 bp shift is not negligible either: it moves a low-expression prediction by ~0.10 log(TPM+1).",
  "verification": "The offline window extraction was checked byte-for-byte against the Ensembl REST API on 6 control genes and a 798-gene stratified random sample: zero mismatches."
 },
 "score_unit": {
  "unit": "ln(quantile-normalised TPM + 1)",
  "note": "Natural log, and the training targets were quantile-normalised -- which is why magnitudes are compressed. Compare deltas, never absolutes. expression_tpm in the API response is exp(expression) - 1 and inherits the same normalisation."
 },
 "pairs": [
  {
   "code": "A1",
   "pair_id": "cardiomyocyte__vs__hepatocyte",
   "label": "Ventricular cardiomyocyte ON / Hepatocyte (liver parenchymal cell) OFF",
   "on_context_id": "cardiomyocyte_ventricular",
   "off_context_id": "hepatocyte_primary",
   "rationale": "Cardiac gene therapy delivered by IV-systemic AAV: the capsid loads the liver, so the promoter carries the de-targeting.",
   "file": "pairs/A1_cardiomyocyte__vs__hepatocyte.json",
   "passed_both_gates": 618,
   "provisional_within_noise": 190,
   "records_in_release": 27,
   "genome_wide_median_margin": 0.7446,
   "on_target_floor": 4.5,
   "specificity_lost_on_trim": 1,
   "margin_substantially_reduced_on_trim": 4
  },
  {
   "code": "A2",
   "pair_id": "skeletal_muscle_myofiber__vs__hepatocyte",
   "label": "Skeletal muscle myofiber / myotube ON / Hepatocyte (liver parenchymal cell) OFF",
   "on_context_id": "skeletal_myofiber",
   "off_context_id": "hepatocyte_primary",
   "rationale": "Neuromuscular gene therapy: same liver de-targeting problem, myofiber payload.",
   "file": "pairs/A2_skeletal_muscle_myofiber__vs__hepatocyte.json",
   "passed_both_gates": 1920,
   "provisional_within_noise": 1501,
   "records_in_release": 28,
   "genome_wide_median_margin": 0.2471,
   "on_target_floor": 1.8203,
   "specificity_lost_on_trim": 1,
   "margin_substantially_reduced_on_trim": 8
  },
  {
   "code": "A3",
   "pair_id": "cardiomyocyte__vs__skeletal_muscle_myofiber",
   "label": "Ventricular cardiomyocyte ON / Skeletal muscle myofiber / myotube OFF",
   "on_context_id": "cardiomyocyte_ventricular",
   "off_context_id": "skeletal_myofiber",
   "rationale": "Cardiac-restricted designs that must also stay off skeletal muscle -- the within-lineage contrast the standard MCK-derived elements do not achieve.",
   "file": "pairs/A3_cardiomyocyte__vs__skeletal_muscle_myofiber.json",
   "passed_both_gates": 38,
   "provisional_within_noise": 31,
   "records_in_release": 26,
   "genome_wide_median_margin": 0.457,
   "on_target_floor": 4.5,
   "specificity_lost_on_trim": 4,
   "margin_substantially_reduced_on_trim": 4
  },
  {
   "code": "A4",
   "pair_id": "cns_neuron__vs__hepatocyte",
   "label": "CNS neuron (cortical / forebrain excitatory / projection neuron) ON / Hepatocyte (liver parenchymal cell) OFF",
   "on_context_id": "cns_neuron",
   "off_context_id": "hepatocyte_primary",
   "rationale": "CNS gene therapy where systemic exposure reaches the liver.",
   "file": "pairs/A4_cns_neuron__vs__hepatocyte.json",
   "passed_both_gates": 1450,
   "provisional_within_noise": 554,
   "records_in_release": 30,
   "genome_wide_median_margin": 0.5,
   "on_target_floor": 3.5,
   "specificity_lost_on_trim": 4,
   "margin_substantially_reduced_on_trim": 22
  },
  {
   "code": "A5",
   "pair_id": "cns_neuron__vs__astrocyte",
   "label": "CNS neuron (cortical / forebrain excitatory / projection neuron) ON / Astrocyte OFF",
   "on_context_id": "cns_neuron",
   "off_context_id": "astrocyte",
   "rationale": "Neuron-restricted CNS designs that must stay off astrocytes -- a within-lineage contrast inside the brain.",
   "file": "pairs/A5_cns_neuron__vs__astrocyte.json",
   "passed_both_gates": 96,
   "provisional_within_noise": 142,
   "records_in_release": 29,
   "genome_wide_median_margin": 0.1719,
   "on_target_floor": 3.5,
   "specificity_lost_on_trim": 11,
   "margin_substantially_reduced_on_trim": 12
  }
 ],
 "withdrawn": [
  {
   "code": "A6",
   "label": "Retinal pigment epithelium ON / Hepatocyte OFF",
   "reason": "Failed its own go/no-go. RPE65's own promoter does not clear the reportable margin threshold against liver -- and RPE65 is the gene whose loss causes the disease treated by the only approved RPE-directed gene therapy, so if any RPE promoter should have been easy to recover, it was this one. (That therapy delivers RPE65 as the transgene under a general-purpose viral promoter; it does not use the RPE65 promoter. The point here is the gene's standing in this cell type, not its construct.) One of five RPE markers is reportable. A coarse cross-organ contrast that should have been easy."
  },
  {
   "code": "B2",
   "label": "Photoreceptor ON / Retinal pigment epithelium OFF",
   "reason": "0 of 9 RPE marker genes separate RPE from photoreceptor. The two retinal contexts are not resolvable from each other on this checkpoint, so no ranked list in either direction is defensible."
  }
 ],
 "deferred": [
  {
   "code": "A7",
   "pair_id": "astrocyte__vs__hepatocyte",
   "label": "Astrocyte ON / Hepatocyte (liver parenchymal cell) OFF",
   "reason": "Ranked 2026-08-14 and held out of this release until it carries what every shipped pair carries: a row in the recall check, a per-pair verdict in the held-out re-analysis (which covers A1-A5 only), and a stated rationale. Its ranking exists internally; nothing about it is published here, and no number in this release includes it."
  }
 ],
 "counts": {
  "pairs_shipped": 5,
  "pairs_withdrawn": 2,
  "pairs_deferred": 1,
  "records_with_full_provenance": 140,
  "rows_in_candidates_tsv": 140,
  "rows_in_shortlists_tsv": 4122,
  "rows_in_excluded_tsv": 600,
  "contexts": 5
 },
 "files": [
  {
   "path": "README.md",
   "format": "markdown",
   "what": "human-readable release notes: what is here, the limitations, how to regenerate a number"
  },
  {
   "path": "index.html",
   "format": "html",
   "what": "self-contained browsable page; no build step, no external requests"
  },
  {
   "path": "MANIFEST.json",
   "format": "json",
   "schema": "schemas/natural_release.schema.json",
   "what": "this file -- release-level provenance and the file inventory"
  },
  {
   "path": "candidates.jsonl",
   "format": "jsonl",
   "schema": "schemas/natural_promoter_record.schema.json",
   "what": "one fully self-contained record per shortlisted candidate and reference promoter: every number with everything needed to regenerate it"
  },
  {
   "path": "candidates.tsv",
   "format": "tsv",
   "what": "the same rows, flat, for spreadsheets and dataframes"
  },
  {
   "path": "shortlists.tsv",
   "format": "tsv",
   "what": "EVERY candidate clearing both gates in every pair -- the full database, one row each. candidates.jsonl provenances the top 25 per pair; this is the rest of them"
  },
  {
   "path": "excluded.tsv",
   "format": "tsv",
   "what": "the 120 windows excluded by the sequence filters, with their scores"
  },
  {
   "path": "contexts.json",
   "format": "json",
   "what": "the context strings verbatim, with their measured prompt-noise floors"
  },
  {
   "path": "withdrawn.json",
   "format": "json",
   "what": "the pairs that failed and why"
  },
  {
   "path": "recall_check.json",
   "format": "json",
   "what": "the genome-wide recall evidence AND the training-split caveat that qualifies it, under training_split_caveat -- neither half is quotable without the other"
  },
  {
   "path": "pairs/*.json",
   "format": "json",
   "schema": "schemas/natural_pair.schema.json",
   "what": "per-pair record: gates, controls, trim procedure, statistics, and every record for that pair"
  },
  {
   "path": "CHECKSUMS.txt",
   "format": "text",
   "what": "SHA-256 of every other file in the release"
  }
 ],
 "regenerate": {
  "command": "PYTHONPATH=src python3 scripts/export_natural.py",
  "deterministic": true,
  "network_required": false,
  "note": "Rebuilds the whole release from the ranked artifacts. Only MANIFEST.generated_utc and the checksums that depend on it change between runs on unchanged inputs."
 },
 "recall_evidence_summary": "Genome-wide ranking over 19,987 windows recovered TNNT2 (cTnT) at rank 4, CKM (MCK) at 24, ALB at 1 and GFAP at 2. Five of six attempted pairs passed; the sixth is in withdrawn.json. ALL FOUR of those promoters were in the model's training data, as were 118 of the 125 provenanced top-25 rows (94.4%, against a 21.0% held-out base rate), so this is not on its own a generalisation result. The held-out-only re-ranking is: four of the five held-out field standards (ENO2, MYL1, DES, SNAP25) still clear both gates, and MYBPC3 is the top held-out gene in both cardiac pairs -- strong for A2 and A4, clean for A1 after adjusting for the held-out pool's composition, and not demonstrated for A3 or A5. See recall_check.json and limitations.training_set_membership.",
 "limitations": [
  {
   "id": "not_wet_lab_validated",
   "severity": "critical",
   "text": "NOTHING IN THIS RELEASE IS WET-LAB VALIDATED. Every margin, rank and cassette recommendation is a model prediction. No candidate has been cloned, transfected, transduced or measured. Treat the whole database as a hypothesis generator."
  },
  {
   "id": "training_set_membership",
   "severity": "critical",
   "text": "THE RANKING IS PARTLY RECALL OF TRAINING DATA. The model was trained on the TSS window of essentially every human gene paired with its measured expression, over a declared train/test/validation split -- and this release ranks those same windows. The provenanced top-25 tables are 118 of 125 training genes (94.4%) against a 21.0% held-out base rate, and ten of the eleven promoters the recall evidence highlights are training genes. A high rank here is therefore NOT on its own evidence that the model generalises rather than remembers. What is: re-ranking only the 4,053 held-out genes, with the identical gates and no score changed, recovers four of the five field-standard promoters that are held out -- ENO2 (held-out rank 17 in A4, 11 in A5), MYL1 (3, A2), DES (60, A2) and SNAP25 (101, A4), all clearing both gates -- and MYBPC3, a held-out cardiac gene, is the top held-out hit in both cardiac pairs. Per pair: held-out genes clear both gates at the training rate in A2 (1.17x) and A4 (1.09x), and in A1 once the held-out pool's depletion of cardiac genes is adjusted for (95 observed vs 85.1 expected, p = 0.87); in A3 and A5 they do not (~6x and ~2x deficits after the same adjustment). A3 IS NOT VALIDATED: one held-out gene of 4,053 clears both gates, and none of A3's four reference promoters is held out, so no field standard can test it on this split. Both statements hold at once: the limitation is real and so is the generalisation result. Generated sequences cannot have been memorised at all; they are packaged separately, see relationship_to_the_generated_design_release."
  },
  {
   "id": "compressed_dynamic_range",
   "severity": "critical",
   "text": "The 20260523 checkpoint compresses magnitudes. Most real genes occupy 0.03-0.9 ln(TPM+1) and the smallest reportable margin threshold, 0.80, is close to the full width of that band. In that band a selectivity claim is barely resolvable from prompting noise alone. A larger margin is better evidence than a smaller one; neither is a measurement."
  },
  {
   "id": "baselines_differ_across_pairs",
   "severity": "critical",
   "text": "Contexts sit at different genome-wide baselines, so RAW MARGINS ARE NOT COMPARABLE ACROSS PAIRS. Median score over the 19,987 rankable windows: cardiomyocyte 2.17, CNS neuron 2.05, astrocyte 1.67, myofiber 1.38, hepatocyte 0.90. A margin against hepatocyte therefore starts with a ~1.3 log-unit head start that has nothing to do with the candidate. In A1 THE MEDIAN GENE ALREADY SCORES +0.745 -- a raw margin of 0.8 in A1 is what a random gene achieves. Use percentile_margin when comparing across pairs, and read every margin against the pair's genome_wide_median_margin."
  },
  {
   "id": "cassettes_out_of_distribution",
   "severity": "critical",
   "text": "AAV CASSETTES ARE OUT OF DISTRIBUTION FOR THIS MODEL. It was trained on genomic TSS windows of real genes in real chromatin context. A [promoter]->[payload] construct inside an AAV backbone, episomal and unchromatinised, is not that. The trimmed fragments in this release are natural genomic fragments scored as bare DNA -- a step toward a cassette, not a cassette. The go/no-go on cassette encoding has not been run."
  },
  {
   "id": "hsyn1_fails_its_own_floor",
   "severity": "high",
   "text": "hSYN1, the standard neuron-specific AAV promoter, FAILS THE ON-TARGET EXPRESSION FLOOR in both neuronal pairs. SYN1 scores 2.469 under the cns_neuron context against a floor of 3.50. Its A4 margin (+2.419) is reportable and its direction is right, but the field's standard neuronal promoter is not in our shortlist and the model ranks STMN2 and ENO2 above it. We cannot show the model is wrong. The gate was not moved to accommodate it; SYN1 is carried in both pair files so that this is visible rather than absent."
  },
  {
   "id": "trimming_destroys_selectivity",
   "severity": "high",
   "text": "TRIMMING TO A DELIVERABLE CASSETTE LENGTH DESTROYS SELECTIVITY FOR A SUBSTANTIAL FRACTION OF CANDIDATES. Per-pair counts of the shortlisted candidates where no deliverable-length fragment clears its own margin floor (specificity_lost_on_trim): A1 1/25, A2 1/25, A3 4/25, A4 4/25, A5 11/25. Separately, the best fragment retains under half the full-window margin for A1 4/25, A2 8/25, A3 4/25, A4 22/25, A5 12/25 (margin_substantially_reduced_on_trim). Read both flags before quoting a recommended_cassette. In A4 the selectivity localises to the first 1,533 bp DOWNSTREAM of the TSS for 24 of 25 candidates, which is exactly the region a conventional upstream cassette discards."
  },
  {
   "id": "margin_is_not_evidence",
   "severity": "high",
   "text": "A margin is not evidence on its own, at any size. ACTB scores +2.031 HepG2-K562 where the measured truth is -0.568 -- wrong sign, and a spurious margin larger than most real hits. The control panel checks a handful of genes; it cannot tell you the model is right about the next one."
  },
  {
   "id": "single_off_target",
   "severity": "medium",
   "text": "Off-target coverage is one cell type per pair. A candidate quiet against hepatocyte may be loud in a tissue not screened here. Macrophage, vascular endothelium, microglia and oligodendrocyte are not in these numbers."
  },
  {
   "id": "fragment_scores_within_gene_only",
   "severity": "medium",
   "text": "Fragment scores are within-gene, within-context comparisons only. A 600 bp input is shorter than anything the model was trained on and the length effect is large: the median on-target score drops 1.6 (2,500 bp) to 5.6 (500 bp) log units purely from shortening the input. All fragment gates are therefore re-derived at the fragment's own length."
  },
  {
   "id": "rank_is_not_biological_quality",
   "severity": "medium",
   "text": "Rank is a ranking of predicted margin under two specific context strings, not a ranking of biological quality. Change the strings -- even to a defensible paraphrase -- and low scores move by ~60% of their own value. The exact strings used are in every record and in contexts.json."
  },
  {
   "id": "contexts_are_pseudobulk_profiles",
   "severity": "medium",
   "text": "The context strings are pseudobulk cell-type expression profiles flattened to '<field> is <value>.' clauses, not promoter assays. A prediction under cardiomyocyte_ventricular is the expected profile of that annotated cell population, not a measurement of a promoter."
  },
  {
   "id": "effective_window_is_asymmetric",
   "severity": "medium",
   "text": "DNA is tokenized at ~6.25 bp/token, so a 9,198 bp window is ~1,400-1,500 tokens against ~1,022 usable slots. The model sees all 4,599 bp upstream of the TSS but only roughly 1,600-2,160 bp downstream, and how much survives varies with the sequence's compressibility. A tile scored beyond that boundary shows the model sequence the full-window score never saw."
  },
  {
   "id": "not_localised_means_unknown",
   "severity": "medium",
   "text": "selectivity_locus = 'not_localised' means no single 1,533 bp tile reproduced the contrast, i.e. we could not locate the active element -- not that it is proximal. It is the commonest verdict in A3 (14/25) and A5 (12/25). Do not read a recommended_cassette on such a candidate as 'the active element is in there'."
  },
  {
   "id": "excluded_windows",
   "severity": "low",
   "text": "The model reads N as real sequence. 120 of 20,107 windows are excluded -- 104 duplicate sequence (mostly pseudoautosomal genes annotated twice, a double-counting hazard in any ranked list), 11 assembly gap, 5 chrM contig-edge padding. All 120 are listed in excluded.tsv with their scores rather than dropped."
  },
  {
   "id": "single_model",
   "severity": "medium",
   "text": "Every number comes from one model. There is no cross-model concordance and no independent measured evidence (ENCODE/SCREEN cCRE overlap, FANTOM5 CAGE, lentiMPRA) attached to any entry in this release. Model agreement would in any case be largely correlated error; only measured data can overrule a model."
  }
 ]
}
