Skip to content

just_dna_enricher.cli

just_dna_enricher.cli

Command-line front door for the enricher (Typer) — the network tier's user-facing command.

just-dna-enricher enrich spec/ --strict --offline just-dna-enricher frequencies spec/ # pass 2: allele frequency (online only) just-dna-enricher gene-metrics spec/ # pass 3: gene constraint (offline capable) just-dna-enricher literature spec/ # pass 4: citations (online only) just-dna-enricher gene-validity spec/ --source gencc # curated gene-disease assertions (online) just-dna-enricher assertions spec/ # ClinVar call + review tier (offline capable) just-dna-enricher enrich-and-compile spec/ out/ --frequencies --gene-metrics just-dna-enricher gnomad constraint build --download --out gnomad_constraint/ # [dev] just-dna-enricher upload out/coronary --repo just-dna-seq/annotators # [dev]

enrich_

enrich_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Fail unless every variant resolves.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Cache-only: never touch the network.",
    ),
    ensembl_cache: Path | None = typer.Option(
        None,
        "--ensembl-cache",
        help="Explicit Ensembl cache dir/.duckdb.",
    ),
    clinvar_cache: Path | None = typer.Option(
        None,
        "--clinvar-cache",
        help="Explicit ClinVar snapshot dir.",
    ),
    pubmind_cache: Path | None = typer.Option(
        None,
        "--pubmind-cache",
        help="Built PubMind snapshot dir (from `pubmind build`) — the second authority in the clinical-significance concordance check. Omit it and $JUST_DNA_PUBMIND_CACHE is read; with neither, PubMind's leg reads unchecked rather than agreement.",
    ),
    use_clinvar: bool = typer.Option(
        True,
        "--clinvar/--no-clinvar",
        help="Use the ClinVar link (after the Ensembl cache).",
    ),
    use_gnomad: bool = typer.Option(
        True,
        "--gnomad/--no-gnomad",
        help="Use the gnomAD link (last, after live Ensembl).",
    ),
    mint_vrs: bool = typer.Option(
        True,
        "--vrs/--no-vrs",
        help="Mint GA4GH VRS allele ids onto resolved rows.",
    ),
    verify_ref: bool = typer.Option(
        True,
        "--verify-ref/--no-verify-ref",
        help="Check each authored ref against the reference sequence and report disagreements.",
    ),
    verify_clinsig: bool = typer.Option(
        True,
        "--verify-clinsig/--no-verify-clinsig",
        help="Check each authored clin_sig against the ClinVar snapshot's own (warns, never fails).",
    ),
    verify_rsids: bool = typer.Option(
        True,
        "--verify-rsids/--no-verify-rsids",
        help="Check each authored rsID against dbSNP for merges/withdrawals (online only).",
    ),
    verify_datasets: bool = typer.Option(
        True,
        "--verify-datasets/--no-verify-datasets",
        help="Check each release recorded in sources.csv against the one that source publishes now, and report the gap. One request per source, and the cheap question to put before --rederive: it tells you whether re-asking every subject is worth the run.",
    ),
    keep_par_twin: bool = typer.Option(
        False,
        "--keep-par-twin",
        help="Record both contigs of a pseudoautosomal locus. Default keeps only the X spelling, which is the one every annotation source uses and the only one a hard-masked GRCh38 analysis set can match.",
    ),
    rederive: bool = typer.Option(
        False,
        "--rederive",
        help="Re-ask every source about every subject, including the ones already recorded, and report which of them changed value. An ordinary run gap-fills and never re-asks, so a source that quietly revised an answer moves nothing you could notice.",
    ),
    keep_staging: bool = typer.Option(
        False,
        "--keep-staging",
        help="Leave the staged answers beside resolution.csv after a successful run. They are removed by default; a killed run leaves them either way, and the next run resumes from them.",
    ),
) -> None

Resolve a spec's variants into resolution.csv beside the spec. Exit 1 in strict mode if unresolved.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("enrich")
def enrich_(  # `enrich` command; function name avoids shadowing the imported enrich()
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    strict: bool = typer.Option(False, "--strict/--best-effort", help="Fail unless every variant resolves."),
    offline: bool = typer.Option(False, "--offline", help="Cache-only: never touch the network."),
    ensembl_cache: Path | None = typer.Option(
        None, "--ensembl-cache", help="Explicit Ensembl cache dir/.duckdb."
    ),
    clinvar_cache: Path | None = typer.Option(None, "--clinvar-cache", help="Explicit ClinVar snapshot dir."),
    pubmind_cache: Path | None = typer.Option(
        None,
        "--pubmind-cache",
        help=(
            "Built PubMind snapshot dir (from `pubmind build`) — the second authority in the "
            "clinical-significance concordance check. Omit it and $JUST_DNA_PUBMIND_CACHE is read; "
            "with neither, PubMind's leg reads unchecked rather than agreement."
        ),
    ),
    use_clinvar: bool = typer.Option(
        True, "--clinvar/--no-clinvar", help="Use the ClinVar link (after the Ensembl cache)."
    ),
    use_gnomad: bool = typer.Option(
        True, "--gnomad/--no-gnomad", help="Use the gnomAD link (last, after live Ensembl)."
    ),
    mint_vrs: bool = typer.Option(
        True, "--vrs/--no-vrs", help="Mint GA4GH VRS allele ids onto resolved rows."
    ),
    verify_ref: bool = typer.Option(
        True,
        "--verify-ref/--no-verify-ref",
        help="Check each authored ref against the reference sequence and report disagreements.",
    ),
    verify_clinsig: bool = typer.Option(
        True,
        "--verify-clinsig/--no-verify-clinsig",
        help="Check each authored clin_sig against the ClinVar snapshot's own (warns, never fails).",
    ),
    verify_rsids: bool = typer.Option(
        True,
        "--verify-rsids/--no-verify-rsids",
        help="Check each authored rsID against dbSNP for merges/withdrawals (online only).",
    ),
    verify_datasets: bool = typer.Option(
        True,
        "--verify-datasets/--no-verify-datasets",
        help="Check each release recorded in sources.csv against the one that source publishes now, "
        "and report the gap. One request per source, and the cheap question to put before "
        "--rederive: it tells you whether re-asking every subject is worth the run.",
    ),
    keep_par_twin: bool = typer.Option(
        False,
        "--keep-par-twin",
        help="Record both contigs of a pseudoautosomal locus. Default keeps only the X spelling, "
        "which is the one every annotation source uses and the only one a hard-masked GRCh38 "
        "analysis set can match.",
    ),
    rederive: bool = typer.Option(
        False,
        "--rederive",
        help="Re-ask every source about every subject, including the ones already recorded, and "
        "report which of them changed value. An ordinary run gap-fills and never re-asks, so a "
        "source that quietly revised an answer moves nothing you could notice.",
    ),
    keep_staging: bool = typer.Option(
        False,
        "--keep-staging",
        help="Leave the staged answers beside resolution.csv after a successful run. They are "
        "removed by default; a killed run leaves them either way, and the next run resumes "
        "from them.",
    ),
) -> None:
    """Resolve a spec's variants into resolution.csv beside the spec. Exit 1 in strict mode if unresolved."""
    try:
        result = enrich(
            spec_dir,
            mode=_mode(strict),
            offline=offline,
            ensembl_cache=ensembl_cache,
            clinvar_cache=clinvar_cache,
            pubmind_cache=pubmind_cache,
            use_clinvar=use_clinvar,
            use_gnomad=use_gnomad,
            mint_vrs=mint_vrs,
            verify_ref=verify_ref,
            verify_clinsig=verify_clinsig,
            verify_rsids=verify_rsids,
            verify_datasets=verify_datasets,
            keep_par_twin=keep_par_twin,
            rederive=rederive,
            keep_staging=keep_staging,
        )
    except EnrichmentError as exc:
        typer.secho(f"ENRICH FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    # The path the pass actually wrote, not `spec_dir / <name>` (RM99) — a module keeping its
    # sidecars under `derived/` is written there, and a guess sends the author to a file that
    # did not change.
    typer.secho(
        f"enriched: {sidecar_path(spec_dir, 'resolution.csv', error=EnrichmentError)}",
        fg=typer.colors.GREEN,
    )
    typer.echo(
        f"rows: {len(result.rows)}  fully_resolved: {result.fully_resolved}  sources: {result.sources}"
    )
    if result.unresolved:
        typer.secho(f"  unresolved: {result.unresolved}", fg=typer.colors.YELLOW, err=True)
    if result.par_twins_dropped:
        # Cyan, not yellow: this is not a finding about the module. It is the table being half the size
        # an author might expect, said out loud so the selection is never silent.
        typer.secho(
            f"  pseudoautosomal: kept the X spelling of {len(result.par_twins_dropped)} locus/loci; "
            f"left out "
            + ", ".join(f"{r} {c}:{p}" for r, c, p in result.par_twins_dropped)
            + " (--keep-par-twin records both)",
            fg=typer.colors.CYAN,
        )
    # Grouped by cause, not one line per row: a systematic mistake (every coordinate shifted one base)
    # produces a finding per variant, and thousands of them hide both the shared cause and every other
    # thing the run reported. Red rather than yellow even in best_effort — this is authored data
    # contradicting the genome, a different and worse thing than a variant the chain could not find.
    for line in summarize_ref_mismatches(result.ref_mismatches):
        typer.secho(f"  ref mismatch: {line}", fg=typer.colors.RED, err=True)
    # Why the ref disagrees, when GRCh37 explains it (RM48). Printed right under the mismatch it
    # diagnoses, because a wrong build is a different remedy from a wrong cell: one row is edited, a
    # whole module is re-authored from rs-numbers.
    for line in summarize_build_diagnoses(result.build_diagnoses):
        typer.secho(f"  old-assembly coordinate: {line}", fg=typer.colors.RED, err=True)
    # Unconditional on `ref_mismatches`, because gating it on them made it unreachable: offline
    # skips the reference check too, so the list is always empty in exactly the runs this notice is
    # about. Which is the point worth saying — an offline run checked neither, and silence here would
    # read as "checked, all clear" (S4).
    if result.build_not_diagnosed == "skipped_offline":
        typer.secho(
            "  reference-allele check and wrong-build diagnosis not run: --offline (both need a "
            "live sequence service, and neither has a local equivalent)",
            fg=typer.colors.CYAN,
        )
    for stale in result.stale_rsids:
        typer.secho(f"  stale rsid: {stale}", fg=typer.colors.YELLOW, err=True)
    for conflict in result.clin_sig_conflicts:
        # Yellow, not red, and in every mode: this is a disagreement between two opinions, not a row
        # contradicting a fact. An opposed call still deserves the author's attention.
        label = "clin_sig conflict" if conflict.opposed else "clin_sig differs"
        typer.secho(f"  {label}: {conflict}", fg=typer.colors.YELLOW, err=True)
    # Say when the check did not run. Silence here reads as "checked, all clear" — which is the one
    # thing it must never mean (S4). `not_requested` is the author's own `--no-verify-clinsig` and
    # needs no echo back.
    if result.clin_sig_not_checked and result.clin_sig_not_checked != "not_requested":
        reason = {
            "no_snapshot": "no ClinVar snapshot this run",
            "unusable_snapshot": "the ClinVar snapshot is present but not queryable",
        }.get(result.clin_sig_not_checked, result.clin_sig_not_checked)
        typer.secho(f"  clin_sig cross-check not run: {reason}", fg=typer.colors.CYAN)
    # One aggregated line, never one per row: on a drafted panel this is thousands of comparisons and
    # what the author has to know is the split. The conflicts themselves printed above, individually,
    # because those are the rows that need answering.
    if result.clin_sig_comparison is not None:
        typer.secho(f"  clin_sig: {result.clin_sig_comparison}", fg=typer.colors.CYAN)
    # `None` means nobody re-derived and there is nothing to say; an empty list means every recorded
    # subject was re-asked and none moved, which prints nothing either — a comparison whose empty
    # result is the normal case would be announcing a zero as evidence. Only a real difference prints.
    for drift in result.rederived or ():
        typer.secho(f"  re-derived: {drift}", fg=typer.colors.YELLOW, err=True)
    # RM85. One line per superseded release rather than an aggregate: `sources.csv` carries one row
    # per (source, layer), so this is a handful of lines on the largest module — the collapse rule is
    # for the per-variant passes, where a systematic mistake produces thousands. Yellow, because an
    # author has something to do about it; the legs nobody could ask are cyan, since that is not a
    # finding about the module, and they are aggregated by reason because a reason repeats.
    if result.dataset_currency is not None:
        for superseded in result.dataset_currency.behind:
            typer.secho(f"  dataset moved on: {superseded}", fg=typer.colors.YELLOW, err=True)
        for line in unchecked_sentences(result.dataset_currency):
            typer.secho(f"  dataset currency: {line}", fg=typer.colors.CYAN)
        # Silence would read as "checked, all clear" (S4). `not_requested` is the author's own
        # `--no-verify-datasets` and needs no echo back.
        if (
            result.dataset_currency.not_checked is not None
            and result.dataset_currency.not_checked != "nothing_to_check"
        ):
            typer.secho(
                f"  dataset currency not checked: {result.dataset_currency.not_checked} — no source "
                f"was asked which release it publishes, so every recorded dataset is unchecked "
                f"rather than current",
                fg=typer.colors.CYAN,
            )

frequencies_

frequencies_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Fail unless every resolved allele has a frequency.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="No-op with a warning: gnomAD frequency has no offline snapshot.",
    ),
    populations: str | None = typer.Option(
        None,
        "--populations",
        help="Comma-separated ancestry groups to keep (e.g. 'global' for one row per allele). Default: all.",
    ),
    dataset: str | None = typer.Option(
        None,
        "--dataset",
        help="Override the dataset label recorded on each row.",
    ),
) -> None

Fill frequencies.csv from the coordinates already in resolution.csv (pass 2, online only).

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("frequencies")
def frequencies_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    strict: bool = typer.Option(
        False, "--strict/--best-effort", help="Fail unless every resolved allele has a frequency."
    ),
    offline: bool = typer.Option(
        False, "--offline", help="No-op with a warning: gnomAD frequency has no offline snapshot."
    ),
    populations: str | None = typer.Option(
        None,
        "--populations",
        help="Comma-separated ancestry groups to keep (e.g. 'global' for one row per allele). Default: all.",
    ),
    dataset: str | None = typer.Option(
        None, "--dataset", help="Override the dataset label recorded on each row."
    ),
) -> None:
    """Fill frequencies.csv from the coordinates already in resolution.csv (pass 2, online only)."""
    from just_dna_enricher.gnomad import FREQUENCY_DATASET_LABEL

    groups = [p.strip() for p in populations.split(",") if p.strip()] if populations else None
    try:
        result = enrich_frequencies(
            spec_dir,
            mode=_mode(strict),
            offline=offline,
            populations=groups,
            dataset=dataset or FREQUENCY_DATASET_LABEL,
        )
    except FrequencyEnrichmentError as exc:
        typer.secho(f"FREQUENCIES FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if result.skipped_offline:
        typer.secho("skipped: --offline (gnomAD frequency has no offline snapshot)", fg=typer.colors.YELLOW)
        return
    # The path the pass actually wrote, not `spec_dir / <name>` (RM99) — a module keeping its
    # sidecars under `derived/` is written there, and a guess sends the author to a file that
    # did not change.
    typer.secho(
        f"frequencies: {sidecar_path(spec_dir, 'frequencies.csv', error=FrequencyEnrichmentError)}",
        fg=typer.colors.GREEN,
    )
    typer.echo(f"rows: {len(result.rows)}  alleles covered: {len(result.covered)}  sources: {result.sources}")
    if result.missing:
        typer.secho(f"  no gnomAD frequency: {result.missing}", fg=typer.colors.YELLOW, err=True)

gene_metrics_

gene_metrics_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Fail unless every gene has constraint metrics.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Snapshot only: never touch the network.",
    ),
    constraint_cache: Path | None = typer.Option(
        None,
        "--constraint-cache",
        help="Explicit gnomAD constraint snapshot dir.",
    ),
) -> None

Fill gene_metrics.csv for the genes variants.csv mentions (pass 3, snapshot then live API).

With no local snapshot the v4.1 one is downloaded from HuggingFace first, exactly as enrich provisions the Ensembl and ClinVar snapshots — --offline is what turns that off, and then the pass is snapshot-only. Reaching the live API instead means v2.1.1 numbers, which the row's dataset records; provisioning is what keeps a plain install on v4.1.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("gene-metrics")
def gene_metrics_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    strict: bool = typer.Option(
        False, "--strict/--best-effort", help="Fail unless every gene has constraint metrics."
    ),
    offline: bool = typer.Option(False, "--offline", help="Snapshot only: never touch the network."),
    constraint_cache: Path | None = typer.Option(
        None, "--constraint-cache", help="Explicit gnomAD constraint snapshot dir."
    ),
) -> None:
    """Fill gene_metrics.csv for the genes variants.csv mentions (pass 3, snapshot then live API).

    With no local snapshot the v4.1 one is downloaded from HuggingFace first, exactly as `enrich`
    provisions the Ensembl and ClinVar snapshots — `--offline` is what turns that off, and then the pass
    is snapshot-only. Reaching the live API instead means **v2.1.1** numbers, which the row's `dataset`
    records; provisioning is what keeps a plain install on v4.1.
    """
    try:
        result = enrich_gene_metrics(
            spec_dir,
            mode=_mode(strict),
            offline=offline,
            constraint_cache=constraint_cache,
        )
    except GeneMetricsEnrichmentError as exc:
        typer.secho(f"GENE METRICS FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    # The path the pass actually wrote, not `spec_dir / <name>` (RM99) — a module keeping its
    # sidecars under `derived/` is written there, and a guess sends the author to a file that
    # did not change.
    typer.secho(
        f"gene metrics: {sidecar_path(spec_dir, 'gene_metrics.csv', error=GeneMetricsEnrichmentError)}",
        fg=typer.colors.GREEN,
    )
    typer.echo(f"rows: {len(result.rows)}  genes covered: {len(result.covered)}  sources: {result.sources}")
    if result.missing:
        typer.secho(f"  no gnomAD constraint: {result.missing}", fg=typer.colors.YELLOW, err=True)

dosage_

dosage_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Fail unless every gene is ClinGen-curated.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="No-op with a warning: ClinGen's curation list is a live download with no snapshot.",
    ),
    url: str = typer.Option(
        DEFAULT_CLINGEN_URL,
        "--url",
        help="ClinGen gene-curation list URL.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use: unstated | non-commercial | commercial. ClinGen is CC0, so no declaration is refused here — it is recorded into sources.csv beside the rows it justifies.",
    ),
) -> None

Add ClinGen dosage-sensitivity rows to gene_metrics.csv (haploinsufficiency/triplosensitivity).

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("dosage")
def dosage_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    strict: bool = typer.Option(
        False, "--strict/--best-effort", help="Fail unless every gene is ClinGen-curated."
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="No-op with a warning: ClinGen's curation list is a live download with no snapshot.",
    ),
    url: str = typer.Option(DEFAULT_CLINGEN_URL, "--url", help="ClinGen gene-curation list URL."),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=(
            "Declared use: unstated | non-commercial | commercial. ClinGen is CC0, so no declaration "
            "is refused here — it is recorded into sources.csv beside the rows it justifies."
        ),
    ),
) -> None:
    """Add ClinGen dosage-sensitivity rows to gene_metrics.csv (haploinsufficiency/triplosensitivity)."""
    try:
        result = enrich_dosage_sensitivity(
            spec_dir, mode=_mode(strict), declared_use=_use(use), offline=offline, url=url
        )
    except ClinGenError as exc:
        typer.secho(f"DOSAGE FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if result.skipped_offline:
        typer.secho(
            "skipped: --offline (ClinGen's curation list has no offline snapshot)",
            fg=typer.colors.YELLOW,
        )
        return
    # The path the pass actually wrote, not `spec_dir / <name>` (RM99) — a module keeping its
    # sidecars under `derived/` is written there, and a guess sends the author to a file that
    # did not change.
    typer.secho(
        f"dosage sensitivity: {sidecar_path(spec_dir, 'gene_metrics.csv', error=ClinGenError)}",
        fg=typer.colors.GREEN,
    )
    typer.echo(f"dataset: {result.dataset}  genes curated: {len(result.covered)}")
    if result.missing:
        # ClinGen curates a subset by design, so this is information rather than a problem.
        typer.secho(f"  not in the ClinGen curation list: {result.missing}", fg=typer.colors.YELLOW)

gene_validity_

gene_validity_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    source: str = typer.Option(
        CLINGEN_VALIDITY_SOURCE,
        "--source",
        help="Which submitter to read: clingen (expert panels) or gencc (an aggregate of nineteen).",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Fail unless every gene carries a curated assertion.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="No-op with a warning: neither ClinGen nor GenCC publishes an offline snapshot.",
    ),
    url: str | None = typer.Option(
        None,
        "--url",
        help="Override the submitter's export URL.",
    ),
) -> None

Fill gene_validity.csv with curated gene-disease assertions for the genes variants.csv names.

One row per (gene, disease, mode of inheritance, submitter) — the source's own grain. Mode of inheritance is in the key because 59 ClinGen (gene, disease) pairs carry two curations that differ only there, and submitter is in it because GenCC publishes the disagreement between submitters, which is the thing it exists to publish.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("gene-validity")
def gene_validity_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    source: str = typer.Option(
        CLINGEN_VALIDITY_SOURCE,
        "--source",
        help="Which submitter to read: clingen (expert panels) or gencc (an aggregate of nineteen).",
    ),
    strict: bool = typer.Option(
        False, "--strict/--best-effort", help="Fail unless every gene carries a curated assertion."
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="No-op with a warning: neither ClinGen nor GenCC publishes an offline snapshot.",
    ),
    url: str | None = typer.Option(None, "--url", help="Override the submitter's export URL."),
) -> None:
    """Fill gene_validity.csv with curated gene-disease assertions for the genes variants.csv names.

    One row per (gene, disease, mode of inheritance, submitter) — the source's own grain. Mode of
    inheritance is in the key because 59 ClinGen (gene, disease) pairs carry two curations that differ
    only there, and `submitter` is in it because GenCC publishes the disagreement between submitters,
    which is the thing it exists to publish.
    """
    try:
        result = enrich_gene_validity(spec_dir, source=source, mode=_mode(strict), offline=offline, url=url)
    except GeneValidityError as exc:
        typer.secho(f"GENE VALIDITY FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if result.skipped_offline:
        typer.secho(
            "skipped: --offline (no gene-validity submitter publishes an offline snapshot)",
            fg=typer.colors.YELLOW,
        )
        return
    # The path the pass actually wrote, not `spec_dir / <name>` — a module keeping its sidecars under
    # `derived/` (RM49) is written there, and printing a guess sends the author to a file that is not
    # the one that changed.
    typer.secho(
        f"gene validity: {sidecar_path(spec_dir, 'gene_validity.csv', error=GeneValidityError)}",
        fg=typer.colors.GREEN,
    )
    typer.echo(f"dataset: {result.dataset}  rows: {len(result.rows)}  genes curated: {len(result.covered)}")
    if result.missing:
        # Both submitters curate a subset by design, so this is information rather than a problem.
        typer.secho(f"  no {source} assertion: {result.missing}", fg=typer.colors.YELLOW)
    if result.unmapped:
        typer.secho(
            f"  wordings this release does not model (kept verbatim in classification_raw): "
            f"{result.unmapped}",
            fg=typer.colors.YELLOW,
            err=True,
        )

gwas_

gwas_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Severity ladder for findings; see the pass docstring.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="No-op with a warning: this pass reads the REST API, not a snapshot.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use recorded on the licence row: unstated|non-commercial|commercial.",
    ),
    study_facts: bool = typer.Option(
        True,
        "--study-facts/--no-study-facts",
        help="Follow each association's study and trait links. Costs 2 requests per association; measured at 382 requests for one real module. Off keeps effects, drops pmid/trait/ancestry PERMANENTLY for the rows it writes: the merge is keyed on association_id, so a later run with study facts on skips those rows rather than back-filling. Delete gwas_effects.csv to re-derive them.",
    ),
) -> None

Fill gwas_effects.csv with the GWAS Catalog's published effect sizes for this module's rsIDs.

One row per published association, not per variant — a well-studied variant carries dozens across different traits and papers. It does NOT fill weight: an authored weight is the author's model of the finding, and no tool writes one. The two sit side by side and a consumer picks.

Reads effect_unit verbatim, including the Catalog's uninformative "unit", because a beta whose scale is unknown must not look like one whose scale is shared. An association the Catalog published without establishing which allele carries the effect keeps a null effect_allele and is counted in the manifest, never dropped.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("gwas")
def gwas_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    strict: bool = typer.Option(
        False, "--strict/--best-effort", help="Severity ladder for findings; see the pass docstring."
    ),
    offline: bool = typer.Option(
        False, "--offline", help="No-op with a warning: this pass reads the REST API, not a snapshot."
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use recorded on the licence row: unstated|non-commercial|commercial.",
    ),
    study_facts: bool = typer.Option(
        True,
        "--study-facts/--no-study-facts",
        help="Follow each association's study and trait links. Costs 2 requests per association; "
        "measured at 382 requests for one real module. Off keeps effects, drops pmid/trait/ancestry "
        "PERMANENTLY for the rows it writes: the merge is keyed on association_id, so a later run "
        "with study facts on skips those rows rather than back-filling. Delete gwas_effects.csv to "
        "re-derive them.",
    ),
) -> None:
    """Fill gwas_effects.csv with the GWAS Catalog's published effect sizes for this module's rsIDs.

    One row per published association, not per variant — a well-studied variant carries dozens across
    different traits and papers. It does NOT fill `weight`: an authored weight is the author's model
    of the finding, and no tool writes one. The two sit side by side and a consumer picks.

    Reads `effect_unit` verbatim, including the Catalog's uninformative "unit", because a beta whose
    scale is unknown must not look like one whose scale is shared. An association the Catalog
    published without establishing which allele carries the effect keeps a null `effect_allele` and is
    counted in the manifest, never dropped.
    """
    try:
        result = enrich_gwas(
            spec_dir, mode=_mode(strict), offline=offline, declared_use=_use(use), study_facts=study_facts
        )
    except GwasError as exc:
        typer.secho(f"GWAS FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if result.skipped_offline:
        typer.secho(
            "skipped: --offline (the GWAS Catalog pass has no offline snapshot)", fg=typer.colors.YELLOW
        )
        return
    # Both halves of the request budget, because the pass computed them and an operator spending
    # somebody else's rate limit is the person who needs the number.
    typer.secho(
        f"gwas: {len(result.rows)} row(s) for {len(result.covered)} variant(s), "
        f"{len(result.missing)} with no published association; "
        f"{result.requests_made} request(s), {result.requests_saved} saved by caching, "
        f"{result.p_value_underflows} p-value(s) below float64 range",
        fg=typer.colors.GREEN,
    )
    # The path the pass actually wrote, never `spec_dir / <name>` — a module keeping its sidecars
    # under `derived/` is written there, and a guess sends the author to the wrong file.
    typer.secho(
        f"gwas effects: {sidecar_path(spec_dir, 'gwas_effects.csv', error=GwasError)}",
        fg=typer.colors.GREEN,
    )

assertions_

assertions_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Fail unless every resolved allele has a ClinVar record.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Snapshot only: never touch the network.",
    ),
    clinvar_cache: Path | None = typer.Option(
        None,
        "--clinvar-cache",
        help="Explicit ClinVar snapshot directory.",
    ),
) -> None

Fill clinical_assertions.csv from the coordinates already in resolution.csv.

Records what ClinVar says about each allele and how much review sits behind it — the star rating a compiled module previously discarded, so a one-star single submission and a practice guideline stopped being the same claim. Offline-capable: with a snapshot provisioned this pass never touches the network, and with none reachable it is a no-op rather than a failure.

It records; it does not adjudicate. Whether the module's own clin_sig agrees with ClinVar's is the enrich cross-check's question, and that one warns in both modes on purpose.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("assertions")
def assertions_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    strict: bool = typer.Option(
        False, "--strict/--best-effort", help="Fail unless every resolved allele has a ClinVar record."
    ),
    offline: bool = typer.Option(False, "--offline", help="Snapshot only: never touch the network."),
    clinvar_cache: Path | None = typer.Option(
        None, "--clinvar-cache", help="Explicit ClinVar snapshot directory."
    ),
) -> None:
    """Fill clinical_assertions.csv from the coordinates already in resolution.csv.

    Records what ClinVar says about each allele **and how much review sits behind it** — the star
    rating a compiled module previously discarded, so a one-star single submission and a practice
    guideline stopped being the same claim. Offline-capable: with a snapshot provisioned this pass
    never touches the network, and with none reachable it is a no-op rather than a failure.

    It records; it does not adjudicate. Whether the module's own clin_sig agrees with ClinVar's is the
    `enrich` cross-check's question, and that one warns in both modes on purpose.
    """
    try:
        result = enrich_clinical_assertions(
            spec_dir,
            mode=_mode(strict),
            offline=offline,
            clinvar_cache=clinvar_cache,
        )
    except ClinicalAssertionError as exc:
        typer.secho(f"ASSERTIONS FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if result.skipped_no_snapshot:
        typer.secho(
            "skipped: no ClinVar snapshot reachable (provision one with `clinvar pull`)",
            fg=typer.colors.YELLOW,
        )
        return
    typer.secho(
        "clinical assertions: "
        f"{sidecar_path(spec_dir, 'clinical_assertions.csv', error=ClinicalAssertionError)}",
        fg=typer.colors.GREEN,
    )
    typer.echo(f"dataset: {result.dataset}  rows: {len(result.rows)}  alleles covered: {len(result.covered)}")
    if result.missing:
        typer.secho(f"  no ClinVar record: {result.missing}", fg=typer.colors.YELLOW)
    if result.off_build:
        typer.secho(
            f"  not on {ASSERTION_GENOME_BUILD}, so never queried: {result.off_build}",
            fg=typer.colors.YELLOW,
            err=True,
        )

literature_

literature_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Fail if a cited PMID does not resolve.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="No-op with a warning: there is no offline PubMed snapshot.",
    ),
    check_fulltext: bool = typer.Option(
        True,
        "--fulltext/--no-fulltext",
        help="Also match provenance quotes against fulltext, falling back to the abstract.",
    ),
    check_doi: bool = typer.Option(
        True,
        "--doi/--no-doi",
        help="Also confirm the authored DOI resolves in Crossref (covers preprints/books).",
    ),
) -> None

Fill literature.csv from a module's citations (pass 4, online only).

studies.csv is one citation site of several: a pmid on a binning row grounds the threshold it sits on, and one on a pharm_variants.csv row grounds that row's own drug and genotype claim. The pass reads every site, so a module citing only from those tables is enriched rather than refused.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("literature")
def literature_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    strict: bool = typer.Option(
        False, "--strict/--best-effort", help="Fail if a cited PMID does not resolve."
    ),
    offline: bool = typer.Option(
        False, "--offline", help="No-op with a warning: there is no offline PubMed snapshot."
    ),
    check_fulltext: bool = typer.Option(
        True,
        "--fulltext/--no-fulltext",
        help="Also match provenance quotes against fulltext, falling back to the abstract.",
    ),
    check_doi: bool = typer.Option(
        True,
        "--doi/--no-doi",
        help="Also confirm the authored DOI resolves in Crossref (covers preprints/books).",
    ),
) -> None:
    """Fill literature.csv from a module's citations (pass 4, online only).

    `studies.csv` is one citation site of several: a `pmid` on a binning row grounds the threshold it
    sits on, and one on a `pharm_variants.csv` row grounds that row's own drug and genotype claim. The
    pass reads every site, so a module citing only from those tables is enriched rather than refused.
    """
    try:
        result = enrich_literature(
            spec_dir,
            mode=_mode(strict),
            offline=offline,
            check_fulltext=check_fulltext,
            check_doi=check_doi,
        )
    except LiteratureEnrichmentError as exc:
        typer.secho(f"LITERATURE FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if result.skipped_offline:
        typer.secho("skipped: --offline (PubMed/Europe PMC have no offline snapshot)", fg=typer.colors.YELLOW)
        return
    # The path the pass actually wrote, not `spec_dir / <name>` (RM99) — a module keeping its
    # sidecars under `derived/` is written there, and a guess sends the author to a file that
    # did not change.
    typer.secho(
        f"literature: {sidecar_path(spec_dir, 'literature.csv', error=LiteratureEnrichmentError)}",
        fg=typer.colors.GREEN,
    )
    # The citations the module makes, not the rows the sidecar holds: merge-not-clobber keeps a row
    # for a citation the author has since deleted, and counting it here would put a number in front
    # of the author that nothing else in the run agrees with.
    typer.echo(f"citations: {len(result.cited)}  {result.coverage}")
    typer.echo(
        f"quotes: {result.quotes_found}/{result.quotes_authored} found, {result.quotes_unchecked} not checked"
    )
    if result.quotes_unexamined:
        # Split out rather than left inside "not checked": an article whose text could not be read
        # and a quote nobody went looking for have different remedies, and only this one is the
        # author's to clear. Calling it "not checkable" was false for an open-access article whose
        # fulltext the previous run read.
        typer.secho(
            f"  {result.quotes_unexamined} authored quote(s) were never looked up: literature.csv "
            f"pins their citation at another quote count — re-run the pass online to re-derive it",
            fg=typer.colors.YELLOW,
        )
    if result.missing:
        # Red: a citation that does not resolve is a defect in the module, not a coverage gap.
        typer.secho(f"  PubMed has no record of: {result.missing}", fg=typer.colors.RED, err=True)
    if result.doi_missing:
        typer.secho(f"  Crossref has no record of: {result.doi_missing}", fg=typer.colors.RED, err=True)
    for conflict in result.doi_conflicts:
        typer.secho(f"  doi conflict: {conflict}", fg=typer.colors.RED, err=True)
    # Printed for the same reason and in the same place: a cross-check that only ever speaks under
    # `--strict` is invisible in the mode almost every author runs, and the two identifiers naming
    # different articles is exactly the case the schema's PMC guard cannot see (RM50).
    for conflict in result.pmcid_conflicts:
        typer.secho(f"  pmcid conflict: {conflict}", fg=typer.colors.RED, err=True)
    # Off the tally, which knows which citations the module quotes *now*: the sidecar keeps a row for
    # a citation the author has since dropped, and its pinned `quotes_authored` would have gone on
    # naming publisher text this module no longer carries.
    noncommercial = result.noncommercial_quoted
    if noncommercial:
        # Yellow, not red, and never a non-zero exit: quoting for comment or research is often fine,
        # and the format is not the tier that adjudicates copyright (the `clin_sig` precedent).
        typer.secho(
            f"  quoted under a non-commercial licence: {noncommercial} — the passage is publisher "
            f"text in this module's annotation layer",
            fg=typer.colors.YELLOW,
        )
    # Yellow and never an exit code, for the same reason: what the author wrote is a quote, and
    # whether a title is an acceptable locator for their claim is theirs to decide. What the tool can
    # say is that `quotes_found` establishes nothing here — a title always appears in its own
    # fulltext, so this is the one shape the quote check cannot fail on (S54).
    if result.titles_as_quotes:
        typer.secho(
            f"  provenance_quote is the article's own title: {result.titles_as_quotes} — a title "
            f"appears in its own fulltext, so quotes_found cannot fail on it and establishes nothing "
            f"about whether the claim is in the paper. Replace it with the passage the claim rests on",
            fg=typer.colors.YELLOW,
        )

pgx_

pgx_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Fail on an allele-function discrepancy.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Snapshots only: never reach PharmVar or CPIC live.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use: unstated | non-commercial | commercial. Sources that forbid sale are SKIPPED when unstated and REFUSED when commercial.",
    ),
    use_pharmvar: bool = typer.Option(
        True,
        "--pharmvar/--no-pharmvar",
        help="Consult PharmVar (needs PHARMVAR_API_KEY).",
    ),
    use_cpic: bool = typer.Option(
        True,
        "--cpic/--no-cpic",
        help="Consult CPIC (open, no key).",
    ),
    cpic_cache: Path | None = typer.Option(
        None,
        "--cpic-cache",
        help="Explicit CPIC snapshot dir.",
    ),
    pharmvar_cache: Path | None = typer.Option(
        None,
        "--pharmvar-cache",
        help="Explicit PharmVar snapshot dir.",
    ),
) -> None

Cross-check star-allele tables against PharmVar/CPIC and record terms into sources.csv.

Snapshot first, live second (RM38). A built snapshot serves the check without egress and without spending a shared per-IP budget; --offline says snapshot-only, and a leg with neither is skipped with a reason rather than silently passing.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("pgx")
def pgx_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    strict: bool = typer.Option(
        False, "--strict/--best-effort", help="Fail on an allele-function discrepancy."
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Snapshots only: never reach PharmVar or CPIC live.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=(
            "Declared use: unstated | non-commercial | commercial. Sources that forbid sale are "
            "SKIPPED when unstated and REFUSED when commercial."
        ),
    ),
    use_pharmvar: bool = typer.Option(
        True, "--pharmvar/--no-pharmvar", help="Consult PharmVar (needs PHARMVAR_API_KEY)."
    ),
    use_cpic: bool = typer.Option(True, "--cpic/--no-cpic", help="Consult CPIC (open, no key)."),
    cpic_cache: Path | None = typer.Option(None, "--cpic-cache", help="Explicit CPIC snapshot dir."),
    pharmvar_cache: Path | None = typer.Option(
        None, "--pharmvar-cache", help="Explicit PharmVar snapshot dir."
    ),
) -> None:
    """Cross-check star-allele tables against PharmVar/CPIC and record terms into sources.csv.

    Snapshot first, live second (RM38). A built snapshot serves the check without egress and without
    spending a shared per-IP budget; `--offline` says snapshot-only, and a leg with neither is skipped
    with a reason rather than silently passing.
    """
    try:
        result = enrich_pgx(
            spec_dir,
            mode=_mode(strict),
            offline=offline,
            declared_use=_use(use),
            use_pharmvar=use_pharmvar,
            use_cpic=use_cpic,
            cpic_cache=cpic_cache,
            pharmvar_cache=pharmvar_cache,
        )
    except LicenseRefusal as exc:
        typer.secho(f"REFUSED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    except PgxEnrichmentError as exc:
        typer.secho(f"PGX FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if result.rows:
        # The file the pass actually wrote, not a guessed name — the module may carry either spelling.
        typer.secho(f"sources: {sources_path(spec_dir, error=PgxEnrichmentError)}", fg=typer.colors.GREEN)
    typer.echo(f"sources recorded: {len(result.rows)}  declared use: {result.declared_use}")
    if result.recorded_use:
        typer.echo(
            "  declared in the licence table by an earlier run: "
            + ", ".join(f"{s}={u}" for s, u in sorted(result.recorded_use.items()))
        )
    if result.routes:
        typer.echo("  routes: " + ", ".join(f"{s}={r}" for s, r in sorted(result.routes.items())))
    for reason in result.skipped_offline:
        typer.secho(f"  {reason}", fg=typer.colors.YELLOW, err=True)
    for reason in result.skipped:
        typer.secho(f"  skipped: {reason}", fg=typer.colors.YELLOW, err=True)
    for warning in result.warnings:
        typer.secho(f"  {warning}", fg=typer.colors.YELLOW, err=True)
    # Warns in BOTH modes on purpose: PharmVar and CPIC are different expert panels and genuinely
    # disagree, so failing would make the format arbitrate between its own authorities.
    for conflict in result.conflicts:
        typer.secho(f"  allele-function difference: {conflict}", fg=typer.colors.YELLOW, err=True)

clinpgx_build_

clinpgx_build_(
    out_dir: Path = typer.Option(
        repro_out("clinpgx"),
        "--out",
        file_okay=False,
        help="Snapshot output directory.",
    ),
    zip_path: Path | None = typer.Option(
        None,
        "--zip",
        help=f"An existing {CURRENT_ARCHIVE.archive} (else downloaded).",
    ),
    url: str = typer.Option(
        DEFAULT_CLINPGX_URL,
        "--url",
        help="ClinPGx bulk download URL.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use: unstated | non-commercial | commercial.",
    ),
) -> None

Download + build the ClinPGx snapshot (dev surface; needs polars).

Source code in enricher/src/just_dna_enricher/cli.py
@clinpgx_app.command("build")
def clinpgx_build_(
    out_dir: Path = typer.Option(
        repro_out("clinpgx"), "--out", file_okay=False, help="Snapshot output directory."
    ),
    zip_path: Path | None = typer.Option(
        None,
        "--zip",
        help=f"An existing {CURRENT_ARCHIVE.archive} (else downloaded).",
    ),
    url: str = typer.Option(DEFAULT_CLINPGX_URL, "--url", help="ClinPGx bulk download URL."),
    use: str = typer.Option(
        "unstated", "--use", help="Declared use: unstated | non-commercial | commercial."
    ),
) -> None:
    """Download + build the ClinPGx snapshot (dev surface; needs polars)."""
    declared = _use(use)
    try:
        # The terms are accepted when the data is TAKEN, so the gate runs before the download.
        reason = check_declared_use(CLINPGX_TERMS, declared)
    except LicenseRefusal as exc:
        typer.secho(f"REFUSED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if reason is not None:
        typer.secho(f"SKIPPED: {reason}", fg=typer.colors.YELLOW, err=True)
        raise typer.Exit(code=1)
    source_sha: str | None = None
    try:
        if zip_path is None:
            # Named for what `--url` actually points at, so a mirror or a retired name is visible on
            # disk rather than filed under whatever this lane used to download.
            filename = Path(urlparse(url).path).name or CURRENT_ARCHIVE.archive
            zip_path, source_sha = download_clinpgx_zip(Path(out_dir) / filename, url)
        result = build_clinpgx_snapshot(zip_path, out_dir, source_url=url, source_sha256=source_sha)
    except (ClinPgxArchiveError, OSError) as exc:
        typer.secho(f"CLINPGX BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"clinpgx snapshot: {result.parquet_path}", fg=typer.colors.GREEN)
    typer.echo(
        f"rows: {result.row_count}  annotations: {result.annotation_count}  "
        f"genes: {len(result.genes)}  release: {result.created_date}"
    )
    typer.echo(f"licence pinned: {result.license_sha256}")

clinpgx_check_

clinpgx_check_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    snapshot: Path | None = typer.Option(
        None,
        "--snapshot",
        help="Explicit ClinPGx snapshot dir. Omit it and the cache is used, or one is downloaded.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Use a local snapshot only: never download one.",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Fail on a stale evidence level.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use: unstated | non-commercial | commercial.",
    ),
) -> None

Cross-check pharm_variants.csv against the ClinPGx snapshot.

The snapshot no longer has to be handed over by hand (RM38): explicit path → $JUST_DNA_CLINPGX_CACHE / the default cache → downloaded from HuggingFace. --offline stops at the second step.

Source code in enricher/src/just_dna_enricher/cli.py
@clinpgx_app.command("check")
def clinpgx_check_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    snapshot: Path | None = typer.Option(
        None,
        "--snapshot",
        help="Explicit ClinPGx snapshot dir. Omit it and the cache is used, or one is downloaded.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Use a local snapshot only: never download one.",
    ),
    strict: bool = typer.Option(False, "--strict/--best-effort", help="Fail on a stale evidence level."),
    use: str = typer.Option(
        "unstated", "--use", help="Declared use: unstated | non-commercial | commercial."
    ),
) -> None:
    """Cross-check pharm_variants.csv against the ClinPGx snapshot.

    The snapshot no longer has to be handed over by hand (RM38): explicit path → `$JUST_DNA_CLINPGX_CACHE`
    / the default cache → downloaded from HuggingFace. `--offline` stops at the second step.
    """
    try:
        result = enrich_clinpgx(
            spec_dir,
            mode=_mode(strict),
            declared_use=_use(use),
            snapshot=snapshot,
            offline=offline,
        )
    except LicenseRefusal as exc:
        typer.secho(f"REFUSED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    except ClinPgxEnrichmentError as exc:
        typer.secho(f"CLINPGX FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.echo(f"dataset: {result.dataset}  sources recorded: {len(result.rows)}")
    for warning in result.warnings:
        typer.secho(f"  {warning}", fg=typer.colors.YELLOW, err=True)
    for conflict in result.conflicts:
        typer.secho(f"  evidence-level difference: {conflict}", fg=typer.colors.YELLOW, err=True)

draft_

draft_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    gene: list[str] = typer.Option(
        ...,
        "--gene",
        help="Gene to draft from CPIC (repeatable).",
    ),
    drug: list[str] = typer.Option(
        [],
        "--drug",
        help="Also draft CPIC's prescribing recommendations for this drug (repeatable).",
    ),
    allele: list[str] = typer.Option(
        [],
        "--allele",
        help="Draft only these star alleles, in all three tables (repeatable; `*1` is always kept). A caller emits a bounded allele set, and n alleles is n(n+1)/2 pairs — CYP2D6 is 16,290 diplotypes unfiltered. Requires a single --gene, since a star name is gene-scoped.",
    ),
    population: str | None = typer.Option(
        None,
        "--population",
        help="Draft only this CPIC clinical context (e.g. 'NVI'). Default: every context, as rows.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use: unstated | non-commercial | commercial. CPIC forbids sale, so a draft is SKIPPED when unstated and REFUSED when commercial.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Draft from a built CPIC snapshot only; never reach CPIC live.",
    ),
    cpic_cache: Path | None = typer.Option(
        None,
        "--cpic-cache",
        help="Explicit CPIC snapshot dir.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Report what would be added; write nothing.",
    ),
) -> None

Draft PGx tables for one or more genes from CPIC — appends rows, never overwrites one.

Re-runnable and additive, so a multi-gene module is built up a gene at a time. A row whose key is already in the file is reported, never replaced: what CPIC now says about a row you already wrote is a finding for pgx, not an edit for this command to make.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("draft")
def draft_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    gene: list[str] = typer.Option(..., "--gene", help="Gene to draft from CPIC (repeatable)."),
    drug: list[str] = typer.Option(
        [],
        "--drug",
        help="Also draft CPIC's prescribing recommendations for this drug (repeatable).",
    ),
    allele: list[str] = typer.Option(
        [],
        "--allele",
        help=(
            "Draft only these star alleles, in all three tables (repeatable; `*1` is always kept). "
            "A caller emits a bounded allele set, and n alleles is n(n+1)/2 pairs — CYP2D6 is 16,290 "
            "diplotypes unfiltered. Requires a single --gene, since a star name is gene-scoped."
        ),
    ),
    population: str | None = typer.Option(
        None,
        "--population",
        help="Draft only this CPIC clinical context (e.g. 'NVI'). Default: every context, as rows.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=(
            "Declared use: unstated | non-commercial | commercial. CPIC forbids sale, so a draft is "
            "SKIPPED when unstated and REFUSED when commercial."
        ),
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Draft from a built CPIC snapshot only; never reach CPIC live.",
    ),
    cpic_cache: Path | None = typer.Option(None, "--cpic-cache", help="Explicit CPIC snapshot dir."),
    dry_run: bool = typer.Option(False, "--dry-run", help="Report what would be added; write nothing."),
) -> None:
    """Draft PGx tables for one or more genes from CPIC — appends rows, never overwrites one.

    Re-runnable and additive, so a multi-gene module is built up a gene at a time. A row whose key is
    already in the file is reported, never replaced: what CPIC now says about a row you already wrote
    is a finding for `pgx`, not an edit for this command to make.
    """
    declared = _use(use)
    if allele and len(gene) != 1:
        # `*2` in CYP2C9 and `*2` in CYP2C19 are different alleles of different genes, so one set
        # applied across several genes would filter each by a name that means something else there.
        # Drafting is per-gene and re-runnable by design — run the command once per gene.
        typer.secho(
            f"--allele needs exactly one --gene (got {len(gene)}): a star-allele name means a "
            f"different allele in each gene. Draft one gene at a time; the command is additive.",
            fg=typer.colors.RED,
            err=True,
        )
        raise typer.Exit(code=2)
    total_added = 0
    for name in gene:
        try:
            result = draft_gene(
                spec_dir,
                name,
                drugs=drug,
                alleles=allele,
                population=population,
                declared_use=declared,
                dry_run=dry_run,
                offline=offline,
                cpic_cache=cpic_cache,
            )
        except (CpicError, *_DRAFT_PRECONDITION_ERRORS) as exc:
            typer.secho(f"DRAFT FAILED ({name}): {exc}", fg=typer.colors.RED, err=True)
            raise typer.Exit(code=1) from exc
        if result.skipped:
            for warning in result.warnings:
                typer.secho(f"  skipped: {warning}", fg=typer.colors.YELLOW, err=True)
            continue
        typer.secho(f"{name}:", fg=typer.colors.GREEN)
        for report in result.reports:
            typer.echo(f"  {report}")
            for outcome in report.differs:
                typer.secho(f"    {outcome}", fg=typer.colors.YELLOW)
        for warning in result.warnings:
            typer.secho(f"  warning: {warning}", fg=typer.colors.YELLOW, err=True)
        total_added += result.added
    verb = "would add" if dry_run else "added"
    typer.echo(f"{verb} {total_added} row(s) across {len(gene)} gene(s) in {spec_dir}")

template_

template_(
    kind: str = typer.Argument(
        ...,
        help="Authored CSV to emit a header for, e.g. repeat_alleles.csv",
    ),
) -> None

Print a header-only CSV for one authored table kind, generated from the live models.

Kept working here, but just-dna-compiler template is canonical: this needs no network, and an author who installed only the tier that owns the CSV shape should not have to add the network tier to get a header. See just-dna-compiler stub for a template with rows to replace.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("template")
def template_(
    kind: str = typer.Argument(..., help="Authored CSV to emit a header for, e.g. repeat_alleles.csv"),
) -> None:
    """Print a header-only CSV for one authored table kind, generated from the live models.

    Kept working here, but `just-dna-compiler template` is canonical: this needs no network, and an
    author who installed only the tier that owns the CSV shape should not have to add the network
    tier to get a header. See `just-dna-compiler stub` for a template with rows to replace.
    """
    try:
        typer.echo(blank_template(kind), nl=False)
        reqs = authoring_requirements(kind)
    except DraftError as exc:
        typer.secho(str(exc), fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"required: {', '.join(reqs['always'])}", fg=typer.colors.BLUE, err=True)
    for group in reqs["any_of"]:
        typer.secho(f"and one of: {' + '.join(group)}", fg=typer.colors.BLUE, err=True)
    if reqs["defaulted"]:
        # Without this line the command gave actively wrong advice: these columns are not "required",
        # so they were never listed, yet an empty cell arrives as None and fails on type.
        shown = ", ".join(f"{k}={v}" for k, v in reqs["defaulted"].items())
        typer.secho(f"must not be left empty (defaults): {shown}", fg=typer.colors.YELLOW, err=True)

check_identifiers_

check_identifiers_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Exit 1 if any identifier is stale.",
    ),
    traits: bool = typer.Option(
        True,
        "--traits/--no-traits",
        help="Check trait_efo_id against OLS4.",
    ),
    genes: bool = typer.Option(
        True,
        "--genes/--no-genes",
        help="Check gene symbols against HGNC.",
    ),
    pgs: bool = typer.Option(
        True,
        "--pgs/--no-pgs",
        help="Check pgs_id against the PGS Catalog, and the two authored cells beside it.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use: unstated | non-commercial | commercial. A PGS score licensed for academic research only bars sale, so a module citing one compiles ONLY with a declaration — and this flag is the one the compile's own refusal tells you to re-run with.",
    ),
) -> None

Report obsolete trait terms, retired gene symbols and unrecognised PGS accessions (online).

Writes no authored cell, and records that the question was put. Unlike the rsID check (whose verdict lands on resolution.csv), these are module-level identifiers with no sidecar column to record, and filling one from the registry being asked about it would make the comparison vacuous — see hints.REDUNDANCY_BEARING. What this does write is verification.json: an attestation that the five checks ran and over how many rows, never a value. A consumer holding the artifact has no other way to tell "asked and clean" from "never asked" (RM45/RM72).

The PGS leg also writes sources.csv (RM163), and that is not an exception to the sentence above: the Catalog's license is a field on each score record and it varies, so a module carrying an academic-research-use-only score must not compile claiming the generic terms. The rows are the terms, never a value in an authored cell.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("check-identifiers")
def check_identifiers_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    strict: bool = typer.Option(False, "--strict/--best-effort", help="Exit 1 if any identifier is stale."),
    traits: bool = typer.Option(True, "--traits/--no-traits", help="Check trait_efo_id against OLS4."),
    genes: bool = typer.Option(True, "--genes/--no-genes", help="Check gene symbols against HGNC."),
    pgs: bool = typer.Option(
        True,
        "--pgs/--no-pgs",
        help="Check pgs_id against the PGS Catalog, and the two authored cells beside it.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=(
            "Declared use: unstated | non-commercial | commercial. A PGS score licensed for academic "
            "research only bars sale, so a module citing one compiles ONLY with a declaration — and "
            "this flag is the one the compile's own refusal tells you to re-run with."
        ),
    ),
) -> None:
    """Report obsolete trait terms, retired gene symbols and unrecognised PGS accessions (online).

    **Writes no authored cell, and records that the question was put.** Unlike the rsID check (whose
    verdict lands on resolution.csv), these are module-level identifiers with no sidecar column to
    record, and filling one from the registry being asked about it would make the comparison vacuous
    — see `hints.REDUNDANCY_BEARING`. What this does write is `verification.json`: an attestation that
    the five checks ran and over how many rows, never a value. A consumer holding the artifact has no
    other way to tell "asked and clean" from "never asked" (RM45/RM72).

    The PGS leg also writes `sources.csv` (RM163), and that is not an exception to the sentence above:
    the Catalog's `license` is a field on each score record and it varies, so a module carrying an
    academic-research-use-only score must not compile claiming the generic terms. The rows are the
    terms, never a value in an authored cell.
    """
    try:
        # `spec_dir=` rather than loading the rows here (RM41). This command was the workspace's own
        # evidence that the row-taking form leaves every caller reaching for a private loader.
        report = check_identifiers(
            spec_dir=spec_dir,
            check_traits=traits,
            check_genes=genes,
            check_pgs=pgs,
            declared_use=use,
        )
    except ValueError as exc:
        # A module whose rows will not load: nothing is attested, because there are no bytes for an
        # attestation to bind to and no question was reached.
        typer.secho(f"{exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    except IdentifierUnavailable as exc:
        # The registry never answered, which is `unreachable` rather than an absence (S20) — and it is
        # the run on which a reader most needs the record, since the report is empty. `check-acmg`
        # records the same thing through `AcmgListUnavailable`; without this the promise two lines up
        # would be false exactly when it matters.
        #
        # This read `except httpx.HTTPError` until RM101, and it only ever fired because
        # `OntologyClient` leaked its transport library's exception — the very defect RM97 set out to
        # end. So the leak was not merely unnoticed here, it was **load-bearing**: repairing the
        # client without this line would have turned the attestation off silently, on exactly the run
        # the comment above says needs it most. The comment already named the right shape one clause
        # over; the type now matches it.
        _attest_on_the_way_out(
            identifier_unreachable(check_traits=traits, check_genes=genes, check_pgs=pgs, detail=str(exc)),
            spec_dir,
        )
        typer.secho(f"IDENTIFIER CHECK FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    except IdentifierCheckError as exc:
        # **After the `IdentifierUnavailable` arm, and the order is load-bearing** — that class is a
        # subclass of this one, so catching the parent first would swallow every unreachable-registry
        # run into this message (`@client-exception-contract`). What reaches here is the PGS leg's
        # licence write failing on the module's own layout, which is neither a stale identifier nor a
        # source that would not answer.
        #
        # Nothing is attested, and that is deliberate rather than an omission: the registries *did*
        # answer, so `unreachable` would be a false record, and the attestation is written through the
        # same sidecar resolver that has just refused — so it would fail again for the same reason and
        # replace this sentence with a worse one.
        typer.secho(f"CHECKED, BUT THE TERMS WERE NOT RECORDED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    # **The guard is the roster, not a filename.** This command opened with `if not (spec_dir /
    # "variants.csv").exists(): return "nothing to check"`, on the reasoning that such a module "has no
    # gene, trait_efo_id or row for these checks to have an opinion about". Nine tables carry each
    # column and four of them are the PGx kinds a module with no `variants.csv` is built out of, so
    # that sentence was false and the command exited 0 having asked nothing — S86's unreadable zero
    # surviving one level above the function that repaired it, on this repo's own
    # `cyp2c19_star_alleles`, which names CYP2C19 on every row it has.
    #
    # No attestation here, which is the one half of the old guard that was right: with no id-bearing
    # table present there is no question to record having put, and minting a nonce would create a
    # `verification.json` on a module that never asked for one. Both checks switched off is a
    # different state and keeps its existing path below, where it is recorded as `not_requested`.
    #
    # **And it must not fire while a table was present and unreadable.** That is the same defect one
    # step further out: a module whose only id-bearing table will not parse read *nothing*, so both
    # halves of the condition were true and the command exited 0 with "nothing to check" — never
    # printing the unreadable-table warning below, and never attesting. The reason a question was not
    # put is exactly what a reader needs on that run.
    #
    # **`and`, not `or`, across the three flags** — a check the author switched off is a record they
    # asked for, and `not_requested` is written on the path below. The guard is for the module that
    # carries no identifier at all, which is a different absence from one the author chose.
    if (
        traits
        and genes
        and pgs
        and not (report.trait_tables_read or report.gene_tables_read or report.pgs_tables_read)
        and not report.unreadable_tables
    ):
        typer.secho(
            "no table carrying trait ids, gene symbols or PGS accessions — nothing to check",
            fg=typer.colors.YELLOW,
        )
        return
    # **The count names the tables it is out of (S86).** `traits checked: 0` used to say two things —
    # the module declares no trait, and its traits are in a table the roster never read — and a reader
    # took the second for the first, which is how a retired CURIE ships with every gate green.
    typer.echo(
        f"traits checked: {len(report.traits)}"
        f" (from {len(report.trait_tables_read)} table(s): {', '.join(report.trait_tables_read) or 'none'})"
        f"  genes checked: {len(report.genes)}"
        f" (from {len(report.gene_tables_read)} table(s): {', '.join(report.gene_tables_read) or 'none'})"
        f"  PGS accessions checked: {len(report.pgs)}"
        f" (from {len(report.pgs_tables_read)} table(s): {', '.join(report.pgs_tables_read) or 'none'})"
    )
    if report.pgs:
        # The comparison's own three numbers, never recomputed here: compared, drifted, withheld. A
        # count beside a check is one that can disagree with it, and then the terminal and the
        # attestation give two accounts of one run.
        comparison = report.pgs_metadata
        typer.echo(
            f"  PGS metadata cells compared: {len(comparison.compared)} of "
            f"{len(comparison.authored)} authored"
            f" ({len(comparison.drift)} disagree, {len(comparison.withheld)} withheld)"
            + (f"  [PGS Catalog release {report.pgs_release}]" if report.pgs_release else "")
        )
    for name, why in sorted(report.unreadable_tables.items()):
        # Only the tables that exist and would not parse. An absent optional table is every module's
        # normal shape and would bury this line in noise.
        typer.secho(
            f"  {name} carries identifiers and could not be read ({why}) — its ids were NOT checked",
            fg=typer.colors.YELLOW,
            err=True,
        )
    for finding in [
        *report.stale_traits,
        *report.stale_genes,
        *report.gene_loci,
        *report.stale_pgs,
        *report.pgs_metadata.drift,
    ]:
        typer.secho(f"  {finding}", fg=typer.colors.YELLOW, err=True)
    for sentence in identifier_pgs_withheld(report.pgs_metadata):
        # Never silently, for `gene_loci_not_checked`'s reason one axis over: a cell this check looked
        # at and could not settle is neither a finding nor a clean comparison, and a reader who cannot
        # see it reads the agreement count as covering every authored cell.
        typer.secho(f"  {sentence}", fg=typer.colors.YELLOW)
    if report.gene_loci_not_checked:
        # Never silently: an empty conflict list means "nothing disagreed" and "never compared", and
        # a reader who cannot tell them apart is being told a check passed that was never put (S24).
        typer.secho(
            f"  gene/chromosome agreement not checked: {report.gene_loci_not_checked}",
            fg=typer.colors.YELLOW,
        )
    # One call for all five records: the proof-of-work binds the whole document, so a per-check write
    # would pay it five times for one guarantee. Before the strict exit below, because the check DID
    # run — the exit code is presentation, and an attestation withheld on it would make the record
    # depend on which flag the author passed.
    try:
        record_verification(
            identifier_records(report, check_traits=traits, check_genes=genes, check_pgs=pgs),
            spec_dir,
            error=EnrichmentError,
        )
    except EnrichmentError as exc:
        # The check did not fail — the report is above and it is complete. What failed is the
        # attestation, so the message says which, in the `vrs mint` shape.
        typer.secho(f"CHECKED, BUT NOT ATTESTED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if report.clean:
        # **"Current" out of nothing is the same unreadable zero one level up (S86).** With both
        # checks off, or with every id-bearing table absent, `report.clean` is vacuously true and the
        # green line asserted a pass over a question nobody put. It says what it read instead.
        looked_at = len(report.trait_tables_read) + len(report.gene_tables_read) + len(report.pgs_tables_read)
        if not report.traits and not report.genes and not report.pgs:
            typer.secho(
                "no identifiers were checked"
                + (
                    f" — {looked_at} table(s) read and none carries a trait id, gene symbol or PGS accession"
                    if looked_at
                    else " — no table carrying identifiers was read"
                ),
                fg=typer.colors.YELLOW,
            )
        elif report.metadata_disagrees:
            # **Checked, and something differs — but the exit code stays 0 even under `--strict`.**
            # Every identifier the registries were asked about is current; what disagrees is a
            # curated cell against a source's own summary, which is the shape the strict gate
            # deliberately does not arbitrate (`@a-source-recuring-is-not-a-strict-matter`). The
            # green line is withheld all the same, because a difference was reported above.
            typer.secho(
                "all identifiers current, but a source disagrees with an authored cell above",
                fg=typer.colors.YELLOW,
            )
        else:
            typer.secho("all identifiers current", fg=typer.colors.GREEN)
    else:
        # **The verdict's own reasons, in both modes.** Every finding behind them is already printed
        # above, so this is the one line that says which of them the exit code turns on — and under
        # `--best-effort` it is the only place the run states that it failed at all. `tables_unreadable`
        # is why this branch is now reachable with nothing stale: a table carrying ids that will not
        # parse was reported above and then exited 0 under `--strict` beneath *all identifiers current*
        # (RM235).
        typer.secho(f"identifier check: {report.clean}", fg=typer.colors.RED)
        if strict:
            raise typer.Exit(code=1)

check_acmg_

check_acmg_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Exit 1 if any acmg_sf disagrees.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="No network. Needs --sf-list, else nothing is checked.",
    ),
    url: str = typer.Option(
        DEFAULT_ACMG_URL,
        "--url",
        help="ACMG secondary-findings page URL (fallback).",
    ),
    sf_list: Path | None = typer.Option(
        None,
        "--sf-list",
        exists=True,
        file_okay=False,
        help="Built ACMG SF snapshot (see `acmg build`). Preferred: NCBI's page still serves v3.2. Omit it and a snapshot in $JUST_DNA_ACMG_CACHE (or the shared cache base) is used; the page is scraped only when neither is there.",
    ),
) -> None

Check each row's acmg_sf against the ACMG secondary-findings list (reports only).

Writes no authored cell, and records that the question was put — the same two halves as check-identifiers. acmg_sf is an authored cell this asks a registry about, not a fact this pass contributes, and filling it here would break the check (see hints.REDUNDANCY_BEARING). The verification.json record is an attestation, never a value: it says the list was consulted and over how many rows, which is the one thing a downstream reader cannot reconstruct from the artifact (RM45/RM72).

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("check-acmg")
def check_acmg_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    strict: bool = typer.Option(False, "--strict/--best-effort", help="Exit 1 if any acmg_sf disagrees."),
    offline: bool = typer.Option(
        False, "--offline", help="No network. Needs --sf-list, else nothing is checked."
    ),
    url: str = typer.Option(DEFAULT_ACMG_URL, "--url", help="ACMG secondary-findings page URL (fallback)."),
    sf_list: Path | None = typer.Option(
        None,
        "--sf-list",
        exists=True,
        file_okay=False,
        help=(
            "Built ACMG SF snapshot (see `acmg build`). Preferred: NCBI's page still serves "
            "v3.2. Omit it and a snapshot in $JUST_DNA_ACMG_CACHE (or the shared cache base) "
            "is used; the page is scraped only when neither is there."
        ),
    ),
) -> None:
    """Check each row's `acmg_sf` against the ACMG secondary-findings list (reports only).

    **Writes no authored cell, and records that the question was put** — the same two halves as
    `check-identifiers`. `acmg_sf` is an authored cell this asks a registry about, not a fact this
    pass contributes, and filling it here would break the check (see `hints.REDUNDANCY_BEARING`). The
    `verification.json` record is an attestation, never a value: it says the list was consulted and
    over how many rows, which is the one thing a downstream reader cannot reconstruct from the
    artifact (RM45/RM72).
    """
    if not (spec_dir / "variants.csv").exists():
        # Nothing attested — see `check-identifiers`: `acmg_sf` is a `variants.csv` column, so with no
        # such file the check does not apply and there is no claim to have an opinion about.
        typer.secho("no variants.csv — nothing to check", fg=typer.colors.YELLOW)
        return
    try:
        report = verify_acmg_sf(
            spec_dir=spec_dir, mode=_mode(strict), offline=offline, url=url, snapshot_dir=sf_list
        )
    except AcmgListUnavailable as exc:
        # No list was obtained, so the check applies and did not run — the one failure here that is a
        # skip. The reason travels on the exception (`unreachable` for a request that never answered,
        # `no_reference` for a source that was there and carried no readable list), decided where the
        # failure happened rather than sniffed out of the message.
        _attest_on_the_way_out(
            [skipped("acmg_secondary_findings", exc.skip, detail=str(exc), source="acmg")],
            spec_dir,
        )
        typer.secho(f"ACMG CHECK FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    except AcmgSfError as exc:
        # Everything else: a `strict` refusal, or a `variants.csv` that will not load. Nothing is
        # attested — the strict path read the list and got an answer, so recording a skip would say the
        # question was never put on the one run where it was put and answered badly; and a module whose
        # rows will not load has no bytes for an attestation to bind to.
        typer.secho(f"ACMG CHECK FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    version = f"ACMG SF v{report.version}" if report.version else "not consulted"
    typer.echo(f"{version}: {report.checked}/{len(report.verdicts)} row(s) checked")
    for warning in report.warnings:
        typer.secho(f"  {warning}", fg=typer.colors.YELLOW, err=True)
    # Unverifiable disagreements are printed like mismatches and excluded from the exit code: the
    # module may be right and the list old. They are the loud half of the stale-list fix.
    for gene, rows, message in AcmgReport.by_gene(report.unverifiable):
        typer.secho(
            f"  unverifiable: {gene} ({len(rows)} row(s), first at {rows[0]}): {message}",
            fg=typer.colors.YELLOW,
            err=True,
        )
    # Grouped by gene: every verdict is a statement about a gene, so a per-row list prints one
    # sentence once per variant in it.
    for gene, rows, message in AcmgReport.by_gene(report.notes):
        typer.secho(f"  note: {gene} ({len(rows)} row(s)): {message}", fg=typer.colors.CYAN)
    for gene, rows, message in AcmgReport.by_gene(report.mismatches):
        typer.secho(
            f"  {gene} ({len(rows)} row(s), first at {rows[0]}): {message}", fg=typer.colors.YELLOW, err=True
        )
    # After the report, for `check-identifiers`' reason: the check ran and its answer is above, so an
    # attestation that cannot be written must not take the answer down with it. `vrs mint`'s shape —
    # the message says which of the two failed, because saying "the check failed" would be false.
    try:
        record_verification([acmg_record(report)], spec_dir, error=EnrichmentError)
    except EnrichmentError as exc:
        typer.secho(f"CHECKED, BUT NOT ATTESTED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    # No `and report.version` guard any more: the property carries the reason a run cannot certify
    # instead of the caller remembering to test for one (RM234, retrofitted to a `Verdict`).
    if report.clean:
        # **A pass over nothing is not a pass, and this is the arm where that can happen.** `offline`
        # is an error and lands in the verdict; `nothing_to_check` — the list was read and the module
        # states no `acmg_sf` cell — is not, so it arrives here truthy. Saying *every stated acmg_sf
        # agrees* over zero stated cells is the S86 shape the sibling command already guards.
        if not report.checked:
            typer.secho(
                f"no acmg_sf cell to check — list {report.version} read, 0 row(s) state one",
                fg=typer.colors.YELLOW,
            )
        else:
            typer.secho(
                f"every stated acmg_sf agrees with the list ({report.checked} row(s) checked)",
                fg=typer.colors.GREEN,
            )
    else:
        typer.secho(f"acmg check: {report.clean}", fg=typer.colors.RED)

enrich_and_compile

enrich_and_compile(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    output_dir: Path = typer.Argument(
        ...,
        file_okay=False,
        help="Output dir for parquet + manifest.json",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Fail unless every variant resolves.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Cache-only: never touch the network.",
    ),
    ensembl_cache: Path | None = typer.Option(
        None,
        "--ensembl-cache",
        help="Explicit Ensembl cache dir/.duckdb.",
    ),
    clinvar_cache: Path | None = typer.Option(
        None,
        "--clinvar-cache",
        help="Explicit ClinVar snapshot dir.",
    ),
    use_clinvar: bool = typer.Option(
        True,
        "--clinvar/--no-clinvar",
        help="Use the ClinVar link (after the Ensembl cache).",
    ),
    use_gnomad: bool = typer.Option(
        True,
        "--gnomad/--no-gnomad",
        help="Use the gnomAD link (last, after live Ensembl).",
    ),
    frequencies: bool = typer.Option(
        False,
        "--frequencies",
        help="Also run the frequency pass (writes frequencies.csv).",
    ),
    gene_metrics: bool = typer.Option(
        False,
        "--gene-metrics",
        help="Also run the gene-constraint pass (writes gene_metrics.csv).",
    ),
) -> None

Enrich, then compile from the produced resolution.csv (offline, deterministic). Exit 1 on failure.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("enrich-and-compile")
def enrich_and_compile(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    output_dir: Path = typer.Argument(..., file_okay=False, help="Output dir for parquet + manifest.json"),
    strict: bool = typer.Option(False, "--strict/--best-effort", help="Fail unless every variant resolves."),
    offline: bool = typer.Option(False, "--offline", help="Cache-only: never touch the network."),
    ensembl_cache: Path | None = typer.Option(
        None, "--ensembl-cache", help="Explicit Ensembl cache dir/.duckdb."
    ),
    clinvar_cache: Path | None = typer.Option(None, "--clinvar-cache", help="Explicit ClinVar snapshot dir."),
    use_clinvar: bool = typer.Option(
        True, "--clinvar/--no-clinvar", help="Use the ClinVar link (after the Ensembl cache)."
    ),
    use_gnomad: bool = typer.Option(
        True, "--gnomad/--no-gnomad", help="Use the gnomAD link (last, after live Ensembl)."
    ),
    frequencies: bool = typer.Option(
        False, "--frequencies", help="Also run the frequency pass (writes frequencies.csv)."
    ),
    gene_metrics: bool = typer.Option(
        False, "--gene-metrics", help="Also run the gene-constraint pass (writes gene_metrics.csv)."
    ),
) -> None:
    """Enrich, then compile from the produced resolution.csv (offline, deterministic). Exit 1 on failure."""
    try:
        enrich(
            spec_dir,
            mode=_mode(strict),
            offline=offline,
            ensembl_cache=ensembl_cache,
            clinvar_cache=clinvar_cache,
            use_clinvar=use_clinvar,
            use_gnomad=use_gnomad,
        )
        # The sidecar passes run between enrich and compile so one command produces every input the
        # compile then consumes. Each is opt-in: a frequency pass costs real requests against a
        # 10-per-minute budget, so it must never be something a plain compile does by surprise.
        if frequencies:
            enrich_frequencies(spec_dir, mode=_mode(strict), offline=offline)
        if gene_metrics:
            enrich_gene_metrics(spec_dir, mode=_mode(strict), offline=offline)
    except (EnrichmentError, FrequencyEnrichmentError, GeneMetricsEnrichmentError) as exc:
        typer.secho(f"ENRICH FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    # Compile consumes the just-written resolution.csv (path 1); no reference, no network.
    result = compile_module(spec_dir, output_dir, ensembl_cache=None, strict=strict)
    for w in result.warnings:
        typer.secho(f"  warning: {w}", fg=typer.colors.YELLOW, err=True)
    if not result.success:
        for e in result.errors:
            typer.secho(f"  error: {e}", fg=typer.colors.RED, err=True)
        typer.secho(f"COMPILE FAILED: {spec_dir}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1)
    typer.secho(f"compiled: {output_dir}", fg=typer.colors.GREEN)
    typer.echo(f"digest: {result.manifest.artifact.digest if result.manifest else '?'}")

upload_

upload_(
    module_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Compiled module directory (at least one annotation parquet + manifest.json).",
    ),
    repo_id: str | None = typer.Option(
        None,
        "--repo",
        help="Target HF dataset (owner/name). Default: just-dna-seq/annotators.",
    ),
    name: str | None = typer.Option(
        None,
        "--name",
        help="Module name under data/<name>/ (and data/<name>/v<version>/) in the repo. Default: the directory basename.",
    ),
    commit_message: str | None = typer.Option(
        None,
        "--message",
        "-m",
        help="Commit message. Default: 'Add <name> module'.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded without contacting HuggingFace.",
    ),
    force: bool = typer.Option(
        False,
        "--force",
        help="Overwrite data/<name>/v<version>/ even when it already holds a different artifact. Without this the publish refuses; the flat path is always overwritten.",
    ),
) -> None

Upload a compiled module to a HuggingFace dataset collection (publisher/dev surface).

Writes data//, which keeps meaning "latest", and — when the manifest states a version — data//v/ under it, in that order, as two commits. With no version, the flat path alone, and the reason why.

Refuses when the versioned path already holds a different artifact (compare by artifact.digest), unless --force. The flat path means latest and is overwritten either way.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("upload")
def upload_(
    module_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Compiled module directory (at least one annotation parquet + manifest.json).",
    ),
    repo_id: str | None = typer.Option(
        None,
        "--repo",
        help="Target HF dataset (owner/name). Default: just-dna-seq/annotators.",
    ),
    name: str | None = typer.Option(
        None,
        "--name",
        help=(
            "Module name under data/<name>/ (and data/<name>/v<version>/) in the repo. "
            "Default: the directory basename."
        ),
    ),
    commit_message: str | None = typer.Option(
        None,
        "--message",
        "-m",
        help="Commit message. Default: 'Add <name> module'.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded without contacting HuggingFace.",
    ),
    force: bool = typer.Option(
        False,
        "--force",
        help=(
            "Overwrite data/<name>/v<version>/ even when it already holds a different artifact. "
            "Without this the publish refuses; the flat path is always overwritten."
        ),
    ),
) -> None:
    """Upload a compiled module to a HuggingFace dataset collection (publisher/dev surface).

    Writes data/<name>/, which keeps meaning "latest", and — when the
    manifest states a version — data/<name>/v<version>/ under it, in
    that order, as two commits. With no version, the flat path alone,
    and the reason why.

    Refuses when the versioned path already holds a different artifact
    (compare by artifact.digest), unless --force. The flat path means
    latest and is overwritten either way.
    """
    from just_dna_enricher.upload import PublishCollisionError, plan_upload, upload_module

    module_name = name or module_dir.name
    if dry_run:
        plan = plan_upload(module_dir, module_name, repo_id)
        typer.echo(f"Would upload to {plan.repo_id} at {plan.path_in_repo}/:")
        for f in plan.files:
            typer.echo(f"  • {f}")
        if plan.versioned_path_in_repo is not None:
            typer.echo(f"…and the same files to {plan.versioned_path_in_repo}/")
        else:
            typer.secho(
                f"no versioned copy: {plan.version_unknown_reason}",
                fg=typer.colors.YELLOW,
                err=True,
            )
        return

    try:
        plan = upload_module(
            module_dir,
            module_name,
            repo_id=repo_id,
            commit_message=commit_message,
            force=force,
        )
    except PublishCollisionError as exc:
        # Its own branch, and its own word: this module is publishable and the remote already has
        # this version. "UPLOAD FAILED" beside the three `plan_upload` refusals would read as
        # "your module is broken", which is the opposite of what happened.
        typer.secho(f"ALREADY PUBLISHED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    except (FileNotFoundError, PermissionError, ImportError) as exc:
        typer.secho(f"UPLOAD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(
        f"uploaded: {module_name} → {plan.repo_id}/{plan.path_in_repo} ({len(plan.files)} files)",
        fg=typer.colors.GREEN,
    )
    if plan.versioned_path_in_repo is not None:
        typer.secho(
            f"uploaded: {module_name} → {plan.repo_id}/{plan.versioned_path_in_repo} "
            f"({len(plan.files)} files)",
            fg=typer.colors.GREEN,
        )
    else:
        typer.secho(
            f"no versioned copy: {plan.version_unknown_reason}",
            fg=typer.colors.YELLOW,
            err=True,
        )

acmg_build_

acmg_build_(
    workbook: Path = typer.Argument(
        ...,
        exists=True,
        dir_okay=False,
        help="ACMG SF supplementary workbook (.xlsx), downloaded by you.",
    ),
    out: Path = typer.Option(
        repro_out("acmg_sf"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes acmg_sf.csv + release.json).",
    ),
    source_url: str | None = typer.Option(
        None,
        "--source-url",
        help="Where the workbook came from, recorded in release.json.",
    ),
    doi: str | None = typer.Option(
        None,
        "--doi",
        help="DOI of the statement the workbook accompanies, recorded in release.json.",
    ),
) -> None

Convert ACMG's SF workbook into the snapshot check-acmg --sf-list reads.

Why this exists: NCBI's page serves v3.2 and ACMG published v3.3 in June 2025, so the live scrape reports correctly authored rows as wrong. Nothing is downloaded here — the workbook is ACMG/Elsevier supplementary material and the author supplies their own copy, which is the same inject-only shape every other reference in this repo uses.

Source code in enricher/src/just_dna_enricher/cli.py
@acmg_app.command("build")
def acmg_build_(
    workbook: Path = typer.Argument(
        ...,
        exists=True,
        dir_okay=False,
        help="ACMG SF supplementary workbook (.xlsx), downloaded by you.",
    ),
    out: Path = typer.Option(
        repro_out("acmg_sf"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes acmg_sf.csv + release.json).",
    ),
    source_url: str | None = typer.Option(
        None,
        "--source-url",
        help="Where the workbook came from, recorded in release.json.",
    ),
    doi: str | None = typer.Option(
        None,
        "--doi",
        help="DOI of the statement the workbook accompanies, recorded in release.json.",
    ),
) -> None:
    """Convert ACMG's SF workbook into the snapshot `check-acmg --sf-list` reads.

    Why this exists: NCBI's page serves **v3.2** and ACMG published **v3.3** in June 2025, so the live
    scrape reports correctly authored rows as wrong. Nothing is downloaded here — the workbook is
    ACMG/Elsevier supplementary material and the author supplies their own copy, which is the same
    inject-only shape every other reference in this repo uses.
    """
    from just_dna_enricher.acmg_build import build_acmg_snapshot

    try:
        sf_list = build_acmg_snapshot(workbook, out, source_url=source_url, doi=doi)
    except (AcmgSfError, ImportError) as exc:
        typer.secho(f"BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"built: {out}", fg=typer.colors.GREEN)
    typer.echo(
        f"ACMG SF v{sf_list.version}: {len(sf_list.genes)} genes over {len(sf_list.findings)} "
        f"gene-condition rows"
    )
    added = sorted({f.gene for f in sf_list.findings if f.since_version == sf_list.version})
    if added:
        typer.echo(f"  first listed in v{sf_list.version}: {', '.join(added)}")

clinvar_build_

clinvar_build_(
    vcf: Path | None = typer.Option(
        None,
        "--vcf",
        exists=True,
        dir_okay=False,
        help="Local ClinVar VCF (.vcf.gz). Omit and pass --download to fetch from NCBI.",
    ),
    download: bool = typer.Option(
        False,
        "--download",
        help="Download the NCBI ClinVar GRCh38 VCF into --out first.",
    ),
    out: Path = typer.Option(
        repro_out("clinvar"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/*.parquet + release.json).",
    ),
) -> None

Convert a ClinVar VCF into the per-chromosome parquet snapshot the resolver reads.

Source code in enricher/src/just_dna_enricher/cli.py
@clinvar_app.command("build")
def clinvar_build_(
    vcf: Path | None = typer.Option(
        None,
        "--vcf",
        exists=True,
        dir_okay=False,
        help="Local ClinVar VCF (.vcf.gz). Omit and pass --download to fetch from NCBI.",
    ),
    download: bool = typer.Option(
        False,
        "--download",
        help="Download the NCBI ClinVar GRCh38 VCF into --out first.",
    ),
    out: Path = typer.Option(
        repro_out("clinvar"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/*.parquet + release.json).",
    ),
) -> None:
    """Convert a ClinVar VCF into the per-chromosome parquet snapshot the resolver reads."""
    from just_dna_enricher.clinvar_build import build_snapshot, download_clinvar_vcf

    if vcf is None and not download:
        typer.secho("Provide --vcf PATH or --download.", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1)
    try:
        source_vcf = vcf if vcf is not None else download_clinvar_vcf(out / "clinvar.vcf.gz")
        result = build_snapshot(source_vcf, out)
    except (FileNotFoundError, ImportError) as exc:
        typer.secho(f"BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"built: {result.out_dir}", fg=typer.colors.GREEN)
    typer.echo(
        f"records: {result.record_count}  chromosomes: {len(result.chromosomes)}  "
        f"clinvar_file_date: {result.clinvar_file_date}"
    )
    typer.secho(
        f"  skipped: non-ACGT {result.skipped_non_acgt}, too-long {result.skipped_too_long}, "
        f"off-target chrom {result.skipped_bad_chrom}",
        fg=typer.colors.YELLOW,
    )

clinvar_publish_

clinvar_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/*.parquet + release.json).",
    ),
    repo_id: str | None = typer.Option(
        None,
        "--repo",
        help="Target HF dataset (owner/name). Default: just-dna-seq/clinvar.",
    ),
    commit_message: str | None = typer.Option(
        None, "--message", "-m", help="Commit message."
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded. Reads the repo's file list; uploads nothing.",
    ),
) -> None

Create-or-update the dataset repo and upload the built ClinVar snapshot (publisher/dev).

Source code in enricher/src/just_dna_enricher/cli.py
@clinvar_app.command("publish")
def clinvar_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/*.parquet + release.json).",
    ),
    repo_id: str | None = typer.Option(
        None,
        "--repo",
        help="Target HF dataset (owner/name). Default: just-dna-seq/clinvar.",
    ),
    commit_message: str | None = typer.Option(None, "--message", "-m", help="Commit message."),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded. Reads the repo's file list; uploads nothing.",
    ),
) -> None:
    """Create-or-update the dataset repo and upload the built ClinVar snapshot (publisher/dev)."""
    from just_dna_enricher.upload import (
        OrphanedSidecarError,
        check_publish_orphans_no_sidecar,
        plan_reference_snapshot,
        publish_reference_snapshot,
    )

    if dry_run:
        plan = plan_reference_snapshot(snapshot_dir, repo_id)
        # A rehearsal that skips the check the real thing refuses on is a different operation. This
        # is the one command that can reach the published ClinVar repo from a snapshot built without
        # its citations half, so the dry run reads the repo rather than promising a refused publish.
        try:
            check_publish_orphans_no_sidecar(plan)
        except OrphanedSidecarError as exc:
            typer.secho(f"WOULD BE REFUSED: {exc}", fg=typer.colors.RED, err=True)
            raise typer.Exit(code=1) from exc
        typer.echo(f"Would upload to {plan.repo_id}:")
        for f in plan.files:
            typer.echo(f"  • {f}")
        return
    try:
        plan = publish_reference_snapshot(snapshot_dir, repo_id, commit_message=commit_message)
    except (FileNotFoundError, PermissionError, ImportError, OrphanedSidecarError) as exc:
        typer.secho(f"PUBLISH FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(
        f"published: {snapshot_dir} → {plan.repo_id} ({len(plan.files)} files)",
        fg=typer.colors.GREEN,
    )

cache_status_

cache_status_() -> None

Say which snapshots are present, where, and which release each holds.

Reads only: nothing is downloaded, so this is safe on a machine with no network and it is the first thing to run when a pass reports that a source was skipped.

Source code in enricher/src/just_dna_enricher/cli.py
@cache_app.command("status")
def cache_status_() -> None:
    """Say which snapshots are present, where, and which release each holds.

    Reads only: nothing is downloaded, so this is safe on a machine with no network and it is the first
    thing to run when a pass reports that a source was skipped.
    """
    # Rendered from `lane_status`, the registry's own projection, so a consumer serving the same
    # answer over HTTP reads the same function rather than re-deriving this loop (S91, RM204).
    for status in lane_status():
        lane = status.lane
        if status.state == "absent":
            # The lane's own command, taken from the registry rather than composed from its name:
            # two lanes are not `<name> build` (`clinpgx build-labels`, `gnomad constraint build`)
            # and a convention that holds for ten of twelve prints two commands nobody can run.
            how = "`cache pull`" if lane.ensure is not None else f"`{lane.build_command}`"
            typer.secho(
                f"  {lane.name:13} absent   — {lane.serves}; provision with {how}",
                fg=typer.colors.YELLOW,
            )
            continue
        if status.state == "occupied":
            # Neither present nor absent: something is at the place the lane looks and it is not a
            # snapshot. A pull or build will be refused here (`prepare` never deletes), so the line
            # says what to do instead of pointing at a command that will decline.
            typer.secho(
                f"  {lane.name:13} occupied {status.looked_in} holds no {lane.name} snapshot; "
                f"move it aside (or `cache prune --only {lane.name}` if it is a retired file)",
                fg=typer.colors.RED,
            )
            continue
        # Present and unreadable is not the same as absent, and a provenance failure is not a data
        # failure — the snapshot is still usable, so this says so instead of hiding it.
        label = status.release or ("(unreadable release.json)" if status.release_unreadable else "")
        size = f"{(status.size_bytes or 0) / 1e6:.1f} MB"
        typer.secho(f"  {lane.name:13} present  {status.path}  {label}  {size}", fg=typer.colors.GREEN)

cache_pull_

cache_pull_(
    only: list[str] = typer.Option(
        [],
        "--only",
        help="Pull just these caches (repeatable). Default: every publishable one.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use for the licence-gated snapshots. They forbid sale, so they are SKIPPED when unstated and REFUSED when commercial — downloading is taking the data.",
    ),
) -> None

Download the published parquet snapshots from HuggingFace into the local caches.

The provisioning step a hosted deployment runs once, so no pass ever reaches a source live per request. Already-complete caches are trusted without touching the network, so this is re-runnable and cheap; a truncated file is removed and refetched.

A lane with nothing published says so and names its reason, which is a field on the registry rather than a comment: PharmVar's and PubMind's are refusals, ACMG's and MANE's are permissions nobody has established. Build those with cache rebuild.

Source code in enricher/src/just_dna_enricher/cli.py
@cache_app.command("pull")
def cache_pull_(
    only: list[str] = typer.Option(
        [],
        "--only",
        help="Pull just these caches (repeatable). Default: every publishable one.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=(
            "Declared use for the licence-gated snapshots. They forbid sale, so they are SKIPPED "
            "when unstated and REFUSED when commercial — downloading is taking the data."
        ),
    ),
) -> None:
    """Download the published parquet snapshots from HuggingFace into the local caches.

    The provisioning step a hosted deployment runs **once**, so no pass ever reaches a source live per
    request. Already-complete caches are trusted without touching the network, so this is re-runnable
    and cheap; a truncated file is removed and refetched.

    A lane with nothing published says so and names its reason, which is a field on the registry
    rather than a comment: PharmVar's and PubMind's are refusals, ACMG's and MANE's are permissions
    nobody has established. Build those with `cache rebuild`.
    """
    declared = _use(use)
    wanted, lanes = _selected(only)
    failures = 0
    for lane in lanes:
        if lane.ensure is None:
            if lane.name in wanted:  # asked for by name, so say why it is not coming
                typer.secho(f"  {lane.name}: {lane.unpublished}", fg=typer.colors.YELLOW, err=True)
            continue
        if lane.terms is not None:
            # The terms are accepted when the data is TAKEN, and a download is taking it.
            try:
                reason = check_declared_use(lane.terms, declared)
            except LicenseRefusal as exc:
                typer.secho(f"  {lane.name}: REFUSED — {exc}", fg=typer.colors.RED, err=True)
                failures += 1
                continue
            if reason is not None:
                typer.secho(f"  {lane.name}: skipped — {reason}", fg=typer.colors.YELLOW, err=True)
                continue
        try:
            path = lane.ensure()
        except SnapshotNotPublished as exc:
            # Asked, and absent. Not a failure and not counted as one: three lanes gained an
            # `ensure_*` before anyone created their repos, and a provisioning command that exits 1
            # on a fresh machine because a snapshot has never been published is reporting the state
            # of the world, not an error (`@unreachable-not-absent`).
            typer.secho(f"  {lane.name}: not published yet — {exc}", fg=typer.colors.YELLOW, err=True)
            continue
        except Exception as exc:
            typer.secho(f"  {lane.name}: FAILED — {exc}", fg=typer.colors.RED, err=True)
            failures += 1
            continue
        typer.secho(f"  {lane.name}: {path}", fg=typer.colors.GREEN)
    if failures:
        raise typer.Exit(code=1)
    pullable = sorted(lane.name for lane in CACHE_LANES if lane.ensure is not None)
    typer.echo(f"caches available: {', '.join(pullable)}. Run `cache status` to confirm.")

cache_prepare_

cache_prepare_(
    only: list[str] = typer.Option(
        [],
        "--only",
        help="Prepare just these caches (repeatable). Default: every one.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=f"Declared use: one of {sorted(VALID_DECLARED_USE)}.",
    ),
    pin: list[str] = typer.Option(
        [],
        "--pin",
        help="lane=release, repeatable, for the lanes that are built rather than pulled.",
    ),
    source: list[str] = typer.Option(
        [],
        "--source",
        help="lane=path, repeatable: build from a file you already hold.",
    ),
) -> None

Leave this machine with every cache it can have — pull what is published, build what is not.

The complement of cache pull, and the one command a deployment actually wants. pull fetches the published snapshots and stops; four lanes are not published for recorded reasons — PharmVar's personal key, PubMind's absent terms, NCBI's policy over MANE, ACMG's supplementary material — so a machine that only pulled is missing four caches and the checks that read them skip themselves. This runs each lane by the route it has.

The route is a property of the lane, never a flag. A published lane pulls, because building it would spend an operator's bandwidth re-deriving bytes somebody already made; an unpublished one builds, because that is the only route there will ever be. Asking for the choice would be asking an operator to restate the licensing story.

A cache that is already present is left alone, exactly as cache pull leaves one alone, so this is idempotent and cheap to re-run. Re-cutting a snapshot that exists is cache rebuild, which writes somewhere else on purpose — a build straight into a live cache is visible half-done to anything reading it, and a short parquet still has a footer.

The Python counterpart is just_dna_enricher.caches.prepare_caches, which this calls.

Source code in enricher/src/just_dna_enricher/cli.py
@cache_app.command("prepare")
def cache_prepare_(
    only: list[str] = typer.Option(
        [],
        "--only",
        help="Prepare just these caches (repeatable). Default: every one.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=f"Declared use: one of {sorted(VALID_DECLARED_USE)}.",
    ),
    pin: list[str] = typer.Option(
        [],
        "--pin",
        help="lane=release, repeatable, for the lanes that are built rather than pulled.",
    ),
    source: list[str] = typer.Option(
        [],
        "--source",
        help="lane=path, repeatable: build from a file you already hold.",
    ),
) -> None:
    """Leave this machine with every cache it can have — pull what is published, build what is not.

    **The complement of `cache pull`, and the one command a deployment actually wants.** `pull`
    fetches the published snapshots and stops; four lanes are not published *for recorded reasons* —
    PharmVar's personal key, PubMind's absent terms, NCBI's policy over MANE, ACMG's supplementary
    material — so a machine that only pulled is missing four caches and the checks that read them
    skip themselves. This runs each lane by the route it has.

    **The route is a property of the lane, never a flag.** A published lane pulls, because building
    it would spend an operator's bandwidth re-deriving bytes somebody already made; an unpublished
    one builds, because that is the only route there will ever be. Asking for the choice would be
    asking an operator to restate the licensing story.

    **A cache that is already present is left alone**, exactly as `cache pull` leaves one alone, so
    this is idempotent and cheap to re-run. Re-cutting a snapshot that exists is `cache rebuild`,
    which writes somewhere else on purpose — a build straight into a live cache is visible half-done
    to anything reading it, and a short parquet still has a footer.

    The Python counterpart is `just_dna_enricher.caches.prepare_caches`, which this calls.
    """
    _, lanes = _selected(only)
    outcomes = prepare_caches(
        lanes,
        declared_use=_use(use),
        pins=_pairs(pin, "--pin"),
        sources={k: Path(v).expanduser() for k, v in _pairs(source, "--source", must_exist=True).items()},
    )
    for outcome in outcomes:
        colour = {
            True: typer.colors.GREEN,
            False: typer.colors.RED,
            None: typer.colors.YELLOW,
        }[outcome.ready]
        typer.secho(
            f"  {outcome.lane:11} {outcome.label:10} {outcome.detail}",
            fg=colour,
            err=outcome.ready is not True,
        )
    ready = [o for o in outcomes if o.ready is True]
    failed = [o for o in outcomes if o.ready is False]
    typer.echo(
        f"{len(ready)} of {len(outcomes)} cache(s) ready "
        f"({sum(o.route == 'pulled' for o in ready)} pulled, "
        f"{sum(o.route == 'built' for o in ready)} built, "
        f"{sum(o.route == 'present' for o in ready)} already there). "
        f"Run `cache status` to confirm."
    )
    if failed:
        raise typer.Exit(code=1)

cache_prune_

cache_prune_(
    only: list[str] = typer.Option(
        [],
        "--only",
        help="Prune just these caches (repeatable). Default: every published one.",
    ),
    yes: bool = typer.Option(
        False,
        "--yes",
        help="Delete without asking. Without it this prints the plan and stops.",
    ),
) -> None

Say what a published snapshot repo carries that its lane is not made of, and offer to delete it.

Deletion is never a side effect of publishing, and this is the command that makes that affordable (RM186). A published repo accumulates: the publisher adds and does not remove, so a layout change leaves the old spelling in place, and just-dna-seq/clinvar still carries the 159 MB single-file clinvar.parquet from before the per-chromosome split. Provisioning already refuses to download it — the glob is what defends this tier — but any consumer globbing data/*.parquet, the dataset viewer included, still gets two schemas under one relation.

Nothing here is a sweep. A file is a candidate only if the lane's own glob excludes it or a LayoutShift declares it retired; README.md, .gitattributes, release.json, LICENSE.txt and sidecar directories are never touched. Without --yes this reads and prints and does nothing else, which is the mode to run first.

Source code in enricher/src/just_dna_enricher/cli.py
@cache_app.command("prune")
def cache_prune_(
    only: list[str] = typer.Option(
        [],
        "--only",
        help="Prune just these caches (repeatable). Default: every published one.",
    ),
    yes: bool = typer.Option(
        False,
        "--yes",
        help="Delete without asking. Without it this prints the plan and stops.",
    ),
) -> None:
    """Say what a published snapshot repo carries that its lane is not made of, and offer to delete it.

    **Deletion is never a side effect of publishing, and this is the command that makes that
    affordable** (RM186). A published repo accumulates: the publisher adds and does not remove, so a
    layout change leaves the old spelling in place, and `just-dna-seq/clinvar` still carries the
    159 MB single-file `clinvar.parquet` from before the per-chromosome split. Provisioning already
    refuses to download it — the glob is what defends this tier — but any consumer globbing
    `data/*.parquet`, the dataset viewer included, still gets two schemas under one relation.

    **Nothing here is a sweep.** A file is a candidate only if the lane's own glob excludes it or a
    `LayoutShift` declares it retired; `README.md`, `.gitattributes`, `release.json`, `LICENSE.txt`
    and sidecar directories are never touched. Without `--yes` this reads and prints and does nothing
    else, which is the mode to run first.
    """
    from just_dna_enricher.download import SNAPSHOT_FILE_GLOBS
    from just_dna_enricher.locations import SNAPSHOT_DATA_DIRNAME
    from just_dna_enricher.upload import plan_prune, prune_repo

    _, lanes = _selected(only)
    planned = 0
    for lane in lanes:
        if lane.publish_repo is None:
            typer.secho(
                f"  {lane.name:13} skipped  — {lane.unpublished or 'published elsewhere'}",
                fg=typer.colors.YELLOW,
            )
            continue
        glob = SNAPSHOT_FILE_GLOBS.get(lane.name)
        if glob is None:
            # STRchive's snapshot is one JSON at the repo root: there is no `data/` for a file to be
            # outside of, so there is nothing this command can name. Said rather than skipped
            # silently, because "prune found nothing" and "prune cannot look" are different answers.
            typer.secho(
                f"  {lane.name:13} n/a      — this snapshot has no {SNAPSHOT_DATA_DIRNAME}/ "
                f"to be made of anything",
                fg=typer.colors.YELLOW,
            )
            continue
        try:
            plan = plan_prune(lane.publish_repo, glob)
        except Exception as exc:
            typer.secho(
                f"  {lane.name:13} FAILED   — could not read {lane.publish_repo}: {exc}",
                fg=typer.colors.RED,
                err=True,
            )
            continue
        if not plan.candidates:
            typer.secho(f"  {lane.name:13} clean    {lane.publish_repo}", fg=typer.colors.GREEN)
            continue
        planned += len(plan.candidates)
        typer.secho(
            f"  {lane.name:13} {len(plan.candidates)} file(s), {plan.total_bytes / 1e6:.1f} MB in "
            f"{lane.publish_repo}",
            fg=typer.colors.YELLOW,
        )
        for candidate in plan.candidates:
            size = "" if candidate.size is None else f" ({candidate.size / 1e6:.1f} MB)"
            typer.echo(f"      • {candidate.path}{size} — {candidate.reason}")
        if not yes:
            continue
        deleted = prune_repo(plan)
        typer.secho(f"      deleted {deleted} file(s) from {lane.publish_repo}", fg=typer.colors.GREEN)
    if planned and not yes:
        typer.echo("Nothing was deleted. Re-run with --yes to remove the files listed above.")

cache_rebuild_

cache_rebuild_(
    out: Path = typer.Option(
        Path(CACHES_DIRNAME),
        "--out",
        file_okay=False,
        help="Base directory. Each lane is built into <base>/<lane>/, never in place. The default is under data/, which this workspace git-ignores wholesale.",
    ),
    only: list[str] = typer.Option(
        [],
        "--only",
        help="Rebuild just these caches (repeatable). Default: every one that can be.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=f"Declared use: one of {sorted(VALID_DECLARED_USE)}.",
    ),
    pin: list[str] = typer.Option(
        [],
        "--pin",
        help="lane=release, repeatable. e.g. --pin mane=1.5 --pin civic=2026-08-01.",
    ),
    source: list[str] = typer.Option(
        [],
        "--source",
        help="lane=path, repeatable: build from a file you already hold instead of downloading. Required for acmg; the offline off-switch for clinvar, constraint, clinpgx, drug_labels, pubmind and strchive. mane and civic take three files each and refuse it.",
    ),
    publish: bool = typer.Option(
        False,
        "--publish",
        help="Also upload each rebuilt snapshot to its HuggingFace repo.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="With --publish: show what would be uploaded, send nothing.",
    ),
) -> None

Rebuild every cache this tier builds — acquire, convert, and optionally publish (RM176).

The one endpoint over eleven builders. Each per-lane X build command stays, and this calls the same download_*/build_* functions they do, so there is one conversion algorithm with two callers rather than two that have to agree. What differs is only flag plumbing: a per-lane command offers the local-file inputs an operator holds, and a rebuild pass by definition holds none.

Every lane is built into <base>/<lane>/, never in place over a resolved cache. A rebuild takes minutes and an enrich reading a half-written snapshot mid-flight would see a real but incomplete table — the failure a resolver cannot detect, because a short parquet is still a parquet. Point the caches at the new base when the run is done, or copy each directory across.

An outcome is three-valued. ACMG needs a workbook that is Elsevier supplementary material, PharmVar a personal key, CIViC a release date to pin — none of those is a failure, and a nightly rebuild reporting errors for them would be reporting the licences working as designed. They are printed as not run, with the reason, and the exit code counts only real failures.

Source code in enricher/src/just_dna_enricher/cli.py
@cache_app.command("rebuild")
def cache_rebuild_(
    out: Path = typer.Option(
        Path(CACHES_DIRNAME),
        "--out",
        file_okay=False,
        help=(
            "Base directory. Each lane is built into <base>/<lane>/, never in place. The default is "
            "under data/, which this workspace git-ignores wholesale."
        ),
    ),
    only: list[str] = typer.Option(
        [],
        "--only",
        help="Rebuild just these caches (repeatable). Default: every one that can be.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=f"Declared use: one of {sorted(VALID_DECLARED_USE)}.",
    ),
    pin: list[str] = typer.Option(
        [],
        "--pin",
        help="lane=release, repeatable. e.g. --pin mane=1.5 --pin civic=2026-08-01.",
    ),
    source: list[str] = typer.Option(
        [],
        "--source",
        help=(
            "lane=path, repeatable: build from a file you already hold instead of downloading. "
            "Required for acmg; the offline off-switch for clinvar, constraint, clinpgx, "
            "drug_labels, pubmind and strchive. mane and civic take three files each and refuse it."
        ),
    ),
    publish: bool = typer.Option(
        False,
        "--publish",
        help="Also upload each rebuilt snapshot to its HuggingFace repo.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="With --publish: show what would be uploaded, send nothing.",
    ),
) -> None:
    """Rebuild every cache this tier builds — acquire, convert, and optionally publish (RM176).

    **The one endpoint over eleven builders.** Each per-lane `X build` command stays, and this calls
    the same `download_*`/`build_*` functions they do, so there is one conversion algorithm with two
    callers rather than two that have to agree. What differs is only flag plumbing: a per-lane command
    offers the local-file inputs an operator holds, and a rebuild pass by definition holds none.

    **Every lane is built into `<base>/<lane>/`, never in place over a resolved cache.** A rebuild
    takes minutes and an `enrich` reading a half-written snapshot mid-flight would see a real but
    incomplete table — the failure a resolver cannot detect, because a short parquet is still a
    parquet. Point the caches at the new base when the run is done, or copy each directory across.

    **An outcome is three-valued.** ACMG needs a workbook that is Elsevier supplementary material,
    PharmVar a personal key, CIViC a release date to pin — none of those is a failure, and a nightly
    rebuild reporting errors for them would be reporting the licences working as designed. They are
    printed as *not run*, with the reason, and the exit code counts only real failures.
    """
    declared = _use(use)
    _, lanes = _selected(only)
    pins = _pairs(pin, "--pin")
    sources = _pairs(source, "--source", must_exist=True)

    outcomes: list[RebuildOutcome] = []
    for lane in lanes:
        request = RebuildRequest(
            out_dir=out / lane.name,
            declared_use=declared,
            pin=pins.get(lane.name),
            source=Path(sources[lane.name]).expanduser() if lane.name in sources else None,
            parents=parents_from_rebuild_dir(lane, out),
        )
        outcome = rebuild_lane(lane, request)
        outcomes.append(outcome)
        colour = {
            True: typer.colors.GREEN,
            False: typer.colors.RED,
            None: typer.colors.YELLOW,
        }[outcome.built]
        typer.secho(
            f"  {lane.name:13} {outcome.label:8} {outcome.detail}",
            fg=colour,
            err=outcome.built is not True,
        )
        if outcome.built and publish:
            _publish_rebuilt(lane, outcome, dry_run=dry_run)

    built = [o for o in outcomes if o.built is True]
    failed = [o for o in outcomes if o.built is False]
    not_run = [o for o in outcomes if o.built is None]
    typer.echo(
        f"rebuilt {len(built)}, failed {len(failed)}, not run {len(not_run)} "
        f"of {len(outcomes)} lane(s) into {out}"
    )
    if failed:
        raise typer.Exit(code=1)

cpic_build_

cpic_build_(
    out_dir: Path = typer.Option(
        repro_out("cpic"),
        "--out",
        file_okay=False,
        help="Snapshot output directory.",
    ),
    endpoint: str = typer.Option(
        DEFAULT_CPIC_ENDPOINT,
        "--endpoint",
        help="CPIC PostgREST base URL.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use: unstated | non-commercial | commercial.",
    ),
) -> None

Fetch CPIC whole into data/*.parquet + release.json (dev surface; needs polars).

No gene filter, deliberately: the whole database is ~120k narrow rows, and a snapshot covering only the genes the operator thought of answers "CPIC has nothing" for the next one.

Source code in enricher/src/just_dna_enricher/cli.py
@cpic_app.command("build")
def cpic_build_(
    out_dir: Path = typer.Option(
        repro_out("cpic"), "--out", file_okay=False, help="Snapshot output directory."
    ),
    endpoint: str = typer.Option(DEFAULT_CPIC_ENDPOINT, "--endpoint", help="CPIC PostgREST base URL."),
    use: str = typer.Option(
        "unstated", "--use", help="Declared use: unstated | non-commercial | commercial."
    ),
) -> None:
    """Fetch CPIC whole into `data/*.parquet` + release.json (dev surface; needs polars).

    No gene filter, deliberately: the whole database is ~120k narrow rows, and a snapshot covering only
    the genes the operator thought of answers "CPIC has nothing" for the next one.
    """
    declared = _use(use)
    try:
        # The terms are accepted when the data is TAKEN, so the gate runs before the fetch.
        reason = check_declared_use(CPIC_TERMS, declared)
    except LicenseRefusal as exc:
        typer.secho(f"REFUSED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if reason is not None:
        typer.secho(f"SKIPPED: {reason}", fg=typer.colors.YELLOW, err=True)
        raise typer.Exit(code=1)
    try:
        result = build_cpic_snapshot(out_dir, endpoint=endpoint)
    except (CpicError, CpicBuildError) as exc:
        typer.secho(f"CPIC BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"cpic snapshot: {result.out_dir / 'data'}", fg=typer.colors.GREEN)
    typer.echo(f"genes: {result.gene_count}  rows: {result.total_rows}  dataset: {result.dataset}")
    for name, count in sorted(result.row_counts.items()):
        typer.echo(f"  {name}: {count}")

cpic_publish_

cpic_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/*.parquet + release.json).",
    ),
    repo: str = typer.Option(
        DEFAULT_CPIC_REPO_ID,
        "--repo",
        help="Target HuggingFace dataset repo (owner/name).",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded; send nothing.",
    ),
    commit_message: str | None = typer.Option(
        None, "--message", "-m", help="Commit message."
    ),
) -> None

Create-or-update the dataset repo and upload the built CPIC snapshot (publisher/dev).

Publishable because CPIC's recorded terms permit redistribution — CC BY-SA grants sharing under share-alike plus attribution, which sources.csv carries. PharmVar has no equivalent command, and that is the design rather than an omission.

Source code in enricher/src/just_dna_enricher/cli.py
@cpic_app.command("publish")
def cpic_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/*.parquet + release.json).",
    ),
    repo: str = typer.Option(
        DEFAULT_CPIC_REPO_ID,
        "--repo",
        help="Target HuggingFace dataset repo (owner/name).",
    ),
    dry_run: bool = typer.Option(False, "--dry-run", help="Show what would be uploaded; send nothing."),
    commit_message: str | None = typer.Option(None, "--message", "-m", help="Commit message."),
) -> None:
    """Create-or-update the dataset repo and upload the built CPIC snapshot (publisher/dev).

    Publishable because CPIC's recorded terms permit redistribution — CC BY-SA grants sharing under
    share-alike plus attribution, which `sources.csv` carries. PharmVar has no equivalent command, and
    that is the design rather than an omission.
    """
    from just_dna_enricher.upload import plan_reference_snapshot, publish_reference_snapshot

    if dry_run:
        plan = plan_reference_snapshot(snapshot_dir, repo)
        typer.echo(f"would upload {len(plan.files)} file(s) to {plan.repo_id}: {plan.files}")
        return
    plan = publish_reference_snapshot(snapshot_dir, repo, commit_message=commit_message)
    typer.secho(
        f"published: {snapshot_dir} → {plan.repo_id} ({len(plan.files)} files)",
        fg=typer.colors.GREEN,
    )

clinpgx_publish_

clinpgx_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/*.parquet + release.json).",
    ),
    repo: str = typer.Option(
        DEFAULT_CLINPGX_REPO_ID,
        "--repo",
        help="Target HuggingFace dataset repo (owner/name).",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded; send nothing.",
    ),
    commit_message: str | None = typer.Option(
        None, "--message", "-m", help="Commit message."
    ),
) -> None

Publish a built ClinPGx snapshot so clinpgx check can provision it (publisher/dev).

LICENSE.txt travels with the parquet — the terms ClinPGx ships inside its own archive are what license_sha256 pins, and a published snapshot without them pins nothing for whoever downloads it.

Source code in enricher/src/just_dna_enricher/cli.py
@clinpgx_app.command("publish")
def clinpgx_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/*.parquet + release.json).",
    ),
    repo: str = typer.Option(
        DEFAULT_CLINPGX_REPO_ID,
        "--repo",
        help="Target HuggingFace dataset repo (owner/name).",
    ),
    dry_run: bool = typer.Option(False, "--dry-run", help="Show what would be uploaded; send nothing."),
    commit_message: str | None = typer.Option(None, "--message", "-m", help="Commit message."),
) -> None:
    """Publish a built ClinPGx snapshot so `clinpgx check` can provision it (publisher/dev).

    `LICENSE.txt` travels with the parquet — the terms ClinPGx ships inside its own archive are what
    `license_sha256` pins, and a published snapshot without them pins nothing for whoever downloads it.
    """
    from just_dna_enricher.upload import plan_reference_snapshot, publish_reference_snapshot

    if dry_run:
        plan = plan_reference_snapshot(snapshot_dir, repo)
        typer.echo(f"would upload {len(plan.files)} file(s) to {plan.repo_id}: {plan.files}")
        return
    plan = publish_reference_snapshot(snapshot_dir, repo, commit_message=commit_message)
    typer.secho(
        f"published: {snapshot_dir} → {plan.repo_id} ({len(plan.files)} files)",
        fg=typer.colors.GREEN,
    )

pharmvar_build_

pharmvar_build_(
    out_dir: Path = typer.Option(
        repro_out("pharmvar"),
        "--out",
        file_okay=False,
        help="Snapshot output directory.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use: unstated | non-commercial | commercial.",
    ),
) -> None

Fetch PharmVar whole into data/*.parquet + release.json (dev surface; needs polars + a key).

There is no pharmvar publish, and there will not be. The data is pulled under a key PharmVar's terms §2 make personal and non-transferable, and no axis SourceTerms records covers passing that on — an unestablished permission is not a permission. Build your own; point at it with $JUST_DNA_PHARMVAR_CACHE or --pharmvar-cache.

Source code in enricher/src/just_dna_enricher/cli.py
@pharmvar_app.command("build")
def pharmvar_build_(
    out_dir: Path = typer.Option(
        repro_out("pharmvar"), "--out", file_okay=False, help="Snapshot output directory."
    ),
    use: str = typer.Option(
        "unstated", "--use", help="Declared use: unstated | non-commercial | commercial."
    ),
) -> None:
    """Fetch PharmVar whole into `data/*.parquet` + release.json (dev surface; needs polars + a key).

    **There is no `pharmvar publish`, and there will not be.** The data is pulled under a key
    PharmVar's terms §2 make personal and non-transferable, and no axis `SourceTerms` records covers
    passing that on — an unestablished permission is not a permission. Build your own; point at it with
    `$JUST_DNA_PHARMVAR_CACHE` or `--pharmvar-cache`.
    """
    declared = _use(use)
    try:
        reason = check_declared_use(PHARMVAR_TERMS, declared)
    except LicenseRefusal as exc:
        typer.secho(f"REFUSED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if reason is not None:
        typer.secho(f"SKIPPED: {reason}", fg=typer.colors.YELLOW, err=True)
        raise typer.Exit(code=1)
    try:
        result = build_pharmvar_snapshot(out_dir)
    except PharmVarError as exc:
        typer.secho(f"PHARMVAR BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"pharmvar snapshot: {result.out_dir / 'data'}", fg=typer.colors.GREEN)
    typer.echo(
        f"genes: {result.gene_count}  alleles: {result.allele_count}  "
        f"variants: {result.variant_count} on {result.genome_build}  dataset: {result.dataset}"
    )
    typer.secho(
        "  operator-built and inject-only: do not publish or pass this snapshot on (terms §2).",
        fg=typer.colors.YELLOW,
    )

civic_build_

civic_build_(
    release: str | None = typer.Option(
        None,
        "--release",
        help="Dated CIViC release to download, e.g. 01-Aug-2026. A DATED release, never the nightly: a snapshot that cannot name its input is one nothing can reproduce.",
    ),
    evidence: Path | None = typer.Option(
        None,
        "--evidence",
        exists=True,
        dir_okay=False,
        help="Local ClinicalEvidenceSummaries.tsv. Use instead of --release to build offline.",
    ),
    variants: Path | None = typer.Option(
        None,
        "--variants",
        exists=True,
        dir_okay=False,
        help="Local VariantSummaries.tsv.",
    ),
    profiles: Path | None = typer.Option(
        None,
        "--profiles",
        exists=True,
        dir_okay=False,
        help="Local MolecularProfileSummaries.tsv.",
    ),
    out: Path = typer.Option(
        repro_out("civic"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/civic.parquet + release.json).",
    ),
    submitted: bool = typer.Option(
        False,
        "--submitted",
        help="Also read the release's civic_accepted_and_submitted.vcf, so evidence a curator entered but no editor signed off joins the snapshot. Over 01-Aug-2026 that widens the direction corpus from 507 rows on 270 variants to 1149 on 397, adding 642 submitted rows and 127 variants, and every row carries the status CIViC gave it. Dated and pinnable like the TSVs, so the build stays reproducible.",
    ),
    vcf: Path | None = typer.Option(
        None,
        "--vcf",
        exists=True,
        dir_okay=False,
        help="Local civic_accepted_and_submitted.vcf. Use with the local TSV flags to build offline.",
    ),
) -> None

Reduce a dated CIViC release to the parquet snapshot the direction-axis drafter reads.

The bulk release, not the GraphQL API, and the two are not interchangeable. Every row of ClinicalEvidenceSummaries.tsv is status accepted; the API defaults to NON_REJECTED and serves roughly 2.35x as many evidence items. A snapshot has to be reproducible from a pinned input, so this reads the dated files and records the basis in release.json.

--submitted widens that basis without leaving the dated release (RM169). CIViC publishes <date>-civic_accepted_and_submitted.vcf beside the TSVs, so unreviewed evidence is pinnable too and no API read is needed. The TSVs stay primary — the VCF cannot carry a variant with no GRCh37 coordinate, which is exactly the class whose identity had to be read out of its name — and the VCF supplies the curation status, the submitted evidence, and the identity for the 112 variants VariantSummaries.tsv (itself accepted-only) does not describe. Those rows are stamped identity_derivation="vcf_csq", and nothing is placed from the VCF's own GRCh37 position.

There is no --use flag. CIViC is CC0 on every axis, so a declared-use gate would permit every build unconditionally, and a flag feeding a gate that never gates is a flag that does nothing (@acquisition-gate-is-not-a-read-gate).

Source code in enricher/src/just_dna_enricher/cli.py
@civic_app.command("build")
def civic_build_(
    release: str | None = typer.Option(
        None,
        "--release",
        help=(
            "Dated CIViC release to download, e.g. 01-Aug-2026. A DATED release, never the nightly: "
            "a snapshot that cannot name its input is one nothing can reproduce."
        ),
    ),
    evidence: Path | None = typer.Option(
        None,
        "--evidence",
        exists=True,
        dir_okay=False,
        help="Local ClinicalEvidenceSummaries.tsv. Use instead of --release to build offline.",
    ),
    variants: Path | None = typer.Option(
        None,
        "--variants",
        exists=True,
        dir_okay=False,
        help="Local VariantSummaries.tsv.",
    ),
    profiles: Path | None = typer.Option(
        None,
        "--profiles",
        exists=True,
        dir_okay=False,
        help="Local MolecularProfileSummaries.tsv.",
    ),
    out: Path = typer.Option(
        repro_out("civic"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/civic.parquet + release.json).",
    ),
    submitted: bool = typer.Option(
        False,
        "--submitted",
        help=(
            "Also read the release's civic_accepted_and_submitted.vcf, so evidence a curator entered "
            "but no editor signed off joins the snapshot. Over 01-Aug-2026 that widens the direction "
            "corpus from 507 rows on 270 variants to 1149 on 397, adding 642 submitted rows and 127 "
            "variants, and every row carries the status CIViC gave it. Dated and pinnable like the "
            "TSVs, so the build stays reproducible."
        ),
    ),
    vcf: Path | None = typer.Option(
        None,
        "--vcf",
        exists=True,
        dir_okay=False,
        help="Local civic_accepted_and_submitted.vcf. Use with the local TSV flags to build offline.",
    ),
) -> None:
    """Reduce a dated CIViC release to the parquet snapshot the direction-axis drafter reads.

    **The bulk release, not the GraphQL API, and the two are not interchangeable.** Every row of
    `ClinicalEvidenceSummaries.tsv` is status `accepted`; the API defaults to `NON_REJECTED` and
    serves roughly 2.35x as many evidence items. A snapshot has to be reproducible from a pinned
    input, so this reads the dated files and records the basis in `release.json`.

    **`--submitted` widens that basis without leaving the dated release (RM169).** CIViC publishes
    `<date>-civic_accepted_and_submitted.vcf` beside the TSVs, so unreviewed evidence is pinnable too
    and no API read is needed. The TSVs stay primary — the VCF cannot carry a variant with no GRCh37
    coordinate, which is exactly the class whose identity had to be read out of its name — and the VCF
    supplies the curation status, the submitted evidence, and the identity for the 112 variants
    `VariantSummaries.tsv` (itself accepted-only) does not describe. Those rows are stamped
    `identity_derivation="vcf_csq"`, and nothing is placed from the VCF's own GRCh37 position.

    **There is no `--use` flag.** CIViC is CC0 on every axis, so a declared-use gate would permit
    every build unconditionally, and a flag feeding a gate that never gates is a flag that does
    nothing (`@acquisition-gate-is-not-a-read-gate`).
    """
    from just_dna_enricher.civic_build import (
        CIVIC_EVIDENCE_FILE,
        CIVIC_PROFILE_FILE,
        CIVIC_VARIANT_FILE,
        build_snapshot,
        civic_release_url,
        download_civic_file,
    )
    from just_dna_enricher.civic_vcf import CIVIC_VCF_FILE

    if vcf is not None and submitted:
        raise typer.BadParameter(
            "pass --vcf to read a local VCF or --submitted to download the release's one, not both. "
            "Two sources for one input is a build whose provenance nothing can state."
        )
    local = (evidence, variants, profiles)
    if release is None and not all(local):
        raise typer.BadParameter(
            "pass --release to download a dated release, or all three of --evidence, --variants "
            "and --profiles to build from local files. Two of the three is not a build: the "
            "evidence file carries the claims, the variant file the identities, and the profile "
            "file is what tells a combination genotype from a dangling reference."
        )
    shas: dict[str, str | None] = {}
    vcf_path = vcf
    if all(local):
        evidence_path, variant_path, profile_path = local
    else:
        out.mkdir(parents=True, exist_ok=True)
        paths = []
        for filename in (CIVIC_EVIDENCE_FILE, CIVIC_VARIANT_FILE, CIVIC_PROFILE_FILE):
            got = download_civic_file(out / filename, civic_release_url(release, filename))
            shas[filename] = got.sha256
            paths.append(got.path)
        evidence_path, variant_path, profile_path = paths
        if submitted:
            got = download_civic_file(out / CIVIC_VCF_FILE, civic_release_url(release, CIVIC_VCF_FILE))
            shas[CIVIC_VCF_FILE] = got.sha256
            vcf_path = got.path

    result = build_snapshot(
        evidence_path,
        variant_path,
        profile_path,
        out,
        release=release,
        evidence_sha256=shas.get(CIVIC_EVIDENCE_FILE),
        variant_sha256=shas.get(CIVIC_VARIANT_FILE),
        profile_sha256=shas.get(CIVIC_PROFILE_FILE),
        vcf=vcf_path,
        vcf_sha256=shas.get(CIVIC_VCF_FILE),
    )
    typer.echo(f"Wrote {result.parquet_file} ({result.record_count} rows, {result.variants} variants)")
    typer.echo(f"  status basis: {result.status_basis}")
    if result.status_counts:
        typer.echo("  " + " · ".join(f"{k} {v}" for k, v in sorted(result.status_counts.items())))
    typer.echo(f"  dataset: {result.dataset or 'unknown (no --release named)'}")
    typer.echo(f"  read {result.input_rows} evidence rows; dropped:")
    for reason, count in result.dropped.items():
        typer.echo(f"    {reason:24s} {count}")
    typer.echo(f"  identity: {result.identity_derivations}")
    if result.withheld_direction:
        typer.echo(
            f"  {result.withheld_direction} row(s) kept with no direction: the source refuted a "
            f"claim rather than making one, and a refutation is not the opposite claim."
        )
    if result.dropped["unresolvable_identity"]:
        typer.echo(
            f"  {result.dropped['unresolvable_identity']} row(s) dropped for carrying neither an "
            f"rsID nor a GRCh38 accession; {result.unresolvable_with_caid} of those variants do "
            f"carry a ClinGen CAID, so they stay addressable by a later identity pass."
        )

civic_citations_

civic_citations_(
    spec: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory.",
    ),
    snapshot: Path | None = typer.Option(
        None,
        "--snapshot",
        exists=True,
        file_okay=False,
        help="CIViC snapshot to map authored rows through. Default: the provisioned cache.",
    ),
    variant_id: list[int] = typer.Option(
        [],
        "--variant-id",
        help="Ask about a CIViC variant id directly, repeatable. Its citations ground the MODULE rather than a variant, which is the only route to a record CIViC publishes no identity for — variant 1955 is the case this exists for.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Do not fetch. Every subject is recorded as not-asked; no row is written.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Report what would be appended, write nothing.",
    ),
) -> None

Append the citations a CIViC variant carries that the dated bulk release cannot reach (RM160).

Why this is not part of civic build. The builder reads a dated release and is byte- reproducible from it; civic reproduce proves it by building twice. The wider basis RM169 adopted comes from a VCF, and a VCF record needs a POS — so submitted evidence attached to a variant with no GRCh37 coordinate is published on exactly one surface, the GraphQL API, which has no release to pin. The read is also one request per variant by construction, because evidenceItems takes a single variantId. Batching it into the builder is the first repair anyone proposes and it is exactly the reproducibility bargain this shape refused.

A recovered citation lands in studies.csv; literature.csv is derived from those PMIDs by the literature command, and an article row nothing cites is dropped from the artifact. CIViC's own curation status rides in confidence/confidence_unit, unconverted, so an accepted row and a submitted row are not the same row. Appending only — a second run over an unchanged API adds nothing — and enrich re-asks later and reports what has moved since.

CIViC is CC0, so there is no --use flag: a declared-use gate would permit every call unconditionally, and a flag feeding a gate that never gates is a flag that does nothing.

Source code in enricher/src/just_dna_enricher/cli.py
@civic_app.command("citations")
def civic_citations_(
    spec: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory."),
    snapshot: Path | None = typer.Option(
        None,
        "--snapshot",
        exists=True,
        file_okay=False,
        help="CIViC snapshot to map authored rows through. Default: the provisioned cache.",
    ),
    variant_id: list[int] = typer.Option(
        [],
        "--variant-id",
        help=(
            "Ask about a CIViC variant id directly, repeatable. Its citations ground the MODULE "
            "rather than a variant, which is the only route to a record CIViC publishes no identity "
            "for — variant 1955 is the case this exists for."
        ),
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Do not fetch. Every subject is recorded as not-asked; no row is written.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Report what would be appended, write nothing.",
    ),
) -> None:
    """Append the citations a CIViC variant carries that the dated bulk release cannot reach (RM160).

    **Why this is not part of `civic build`.** The builder reads a dated release and is byte-
    reproducible from it; `civic reproduce` proves it by building twice. The wider basis RM169 adopted
    comes from a VCF, and a VCF record needs a POS — so submitted evidence attached to a variant with
    no GRCh37 coordinate is published on exactly one surface, the GraphQL API, which has no release to
    pin. The read is also **one request per variant by construction**, because `evidenceItems` takes a
    single `variantId`. Batching it into the builder is the first repair anyone proposes and it is
    exactly the reproducibility bargain this shape refused.

    A recovered citation lands in `studies.csv`; `literature.csv` is derived from those PMIDs by the
    `literature` command, and an article row nothing cites is dropped from the artifact. CIViC's own
    curation status rides in `confidence`/`confidence_unit`, unconverted, so an accepted row and a
    submitted row are not the same row. Appending only — a second run over an unchanged API adds
    nothing — and `enrich` re-asks later and reports what has moved since.

    CIViC is CC0, so there is no `--use` flag: a declared-use gate would permit every call
    unconditionally, and a flag feeding a gate that never gates is a flag that does nothing.
    """
    from just_dna_enricher.civic_citations import (
        CivicCitationsError,
        draft_civic_citations,
        read_module,
    )
    from just_dna_enricher.enrich import spec_genome_build
    from just_dna_enricher.locations import resolve_civic_reference

    reference = snapshot if snapshot is not None else resolve_civic_reference()
    try:
        variants, resolution = read_module(spec, genome_build=spec_genome_build(spec))
        result = draft_civic_citations(
            spec,
            variants=variants,
            resolution_rows=resolution,
            reference=reference,
            requested=variant_id,
            offline=offline,
            dry_run=dry_run,
        )
    except (CivicCitationsError, *_DRAFT_PRECONDITION_ERRORS) as exc:
        typer.secho(f"CIVIC CITATIONS FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc

    typer.echo(
        f"  {len(result.subjects)} CIViC variant(s) in scope, {result.asked} asked; "
        f"{result.citations_seen} PubMed citation(s) seen"
    )
    if result.unmapped_rows:
        typer.echo(
            f"  {result.unmapped_rows} authored row(s) resolved to a locus no CIViC variant matched "
            f"— name one with --variant-id if CIViC publishes no identity for it"
        )
    for reason, count in result.withheld.items():
        if count:
            typer.echo(f"  withheld {reason:22s} {count}")
    if result.confidence_withheld:
        typer.echo(
            f"  {result.confidence_withheld} row(s) written with no confidence: CIViC states more "
            f"than one status for the items citing that paper"
        )
    for report in result.reports:
        typer.echo(f"  {report}")
    for warning in result.warnings:
        typer.secho(f"  warning: {warning}", fg=typer.colors.YELLOW, err=True)
    verb = "would add" if dry_run else "added"
    breakdown = ", ".join(f"{r.csv_name} {len(r.added)}" for r in result.reports) or "nothing"
    typer.secho(f"{verb}: {breakdown} — in {spec}", fg=typer.colors.GREEN)

civic_publish_

civic_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/civic.parquet + release.json).",
    ),
    repo_id: str | None = typer.Option(
        None,
        "--repo",
        help="Target HF dataset (owner/name). Default: just-dna-seq/civic.",
    ),
    commit_message: str | None = typer.Option(
        None, "--message", "-m", help="Commit message."
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded without contacting HuggingFace.",
    ),
) -> None

Upload a built CIViC snapshot to a HuggingFace dataset repo (publisher/dev).

This one does not refuse, and the contrast with pubmind publish is the point. CIViC's content is CC0 1.0 — a public-domain dedication with no share-alike, no bar on sale and attribution requested rather than required — so there is no permission to establish before passing the bytes on. PubMind's command exists in order to say no because its terms are unstated; PharmVar's cache is unpublishable because its terms forbid it. Nothing here is in either position.

What the snapshot carries is a derivation of CIViC's release, not a copy of it: the germline direction rows, placed on GRCh38 through identifiers CIViC itself publishes. release.json records which dated release it came from and the accepted status basis, so a consumer can tell what they are looking at without re-deriving it.

The ClinGen Allele Registry's answers are NOT in here, and that is deliberate rather than an oversight: the registry states no terms, and a lookup performed at draft time is a read, while baking its responses into a published file would be redistribution of bytes nobody has established we may pass on. The snapshot carries the CAID; resolving it stays the consumer's own fetch.

Source code in enricher/src/just_dna_enricher/cli.py
@civic_app.command("publish")
def civic_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/civic.parquet + release.json).",
    ),
    repo_id: str | None = typer.Option(
        None,
        "--repo",
        help="Target HF dataset (owner/name). Default: just-dna-seq/civic.",
    ),
    commit_message: str | None = typer.Option(None, "--message", "-m", help="Commit message."),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded without contacting HuggingFace.",
    ),
) -> None:
    """Upload a built CIViC snapshot to a HuggingFace dataset repo (publisher/dev).

    **This one does not refuse, and the contrast with `pubmind publish` is the point.** CIViC's content
    is CC0 1.0 — a public-domain dedication with no share-alike, no bar on sale and attribution
    requested rather than required — so there is no permission to establish before passing the bytes
    on. PubMind's command exists in order to say no because its terms are unstated; PharmVar's cache is
    unpublishable because its terms forbid it. Nothing here is in either position.

    What the snapshot carries is a *derivation* of CIViC's release, not a copy of it: the germline
    direction rows, placed on GRCh38 through identifiers CIViC itself publishes. `release.json` records
    which dated release it came from and the `accepted` status basis, so a consumer can tell what they
    are looking at without re-deriving it.

    **The ClinGen Allele Registry's answers are NOT in here**, and that is deliberate rather than an
    oversight: the registry states no terms, and a lookup performed at draft time is a read, while
    baking its responses into a published file would be redistribution of bytes nobody has established
    we may pass on. The snapshot carries the CAID; resolving it stays the consumer's own fetch.
    """
    from just_dna_enricher.upload import (
        DEFAULT_CIVIC_REPO_ID,
        plan_reference_snapshot,
        publish_reference_snapshot,
    )

    repo_id = repo_id or DEFAULT_CIVIC_REPO_ID
    if dry_run:
        plan = plan_reference_snapshot(snapshot_dir, repo_id)
        typer.echo(f"Would upload to {plan.repo_id}:")
        for f in plan.files:
            typer.echo(f"  • {f}")
        return
    try:
        plan = publish_reference_snapshot(snapshot_dir, repo_id, commit_message=commit_message)
    except (FileNotFoundError, PermissionError, ImportError) as exc:
        typer.secho(f"PUBLISH FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"Published {len(plan.files)} file(s) to {plan.repo_id}", fg=typer.colors.GREEN)

civic_reproduce_

civic_reproduce_(
    release: str = typer.Option(
        "01-Aug-2026",
        "--release",
        help="Dated CIViC release to reproduce, e.g. 01-Aug-2026.",
    ),
    out: Path = typer.Option(
        repro_out("civic_reproduce"),
        "--out",
        file_okay=False,
        help="Working directory. The release files and two independent builds land here. The default is under data/, which this workspace git-ignores wholesale.",
    ),
    keep: bool = typer.Option(
        False,
        "--keep",
        help="Leave the downloaded release files in place for inspection.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Skip the reference cross-check. The build and determinism checks still run.",
    ),
    submitted: bool = typer.Option(
        False,
        "--submitted",
        help="Reproduce the wider basis: also download the release's civic_accepted_and_submitted.vcf and build with it, so the submitted rows and their coordinates go through every check below rather than only the accepted ones.",
    ),
) -> None

Build the CIViC snapshot from a dated release and check it, end to end.

Five checks, and the third is the one worth the network. The first two are about us; the third is about whether the coordinates we produced are real.

  1. The release downloads and its bytes are recorded — a sha256 per file, so a rerun that disagrees is a finding about the source rather than a mystery.
  2. Two independent builds are byte-identical (Principle 7). A parquet has no inherent row order, so this is the check that the sort is doing its job.
  3. Every placed coordinate is cross-checked against the GRCh38 reference sequence. This is the external validation: the snapshot's positions come from RefSeq accessions inside ClinVar HGVS, and this asks an unrelated service (refget/seqrepo) whether the reference base at each of those positions is what we wrote. A wrong-build or off-by-one placement fails here and nowhere else.
  4. The drop registry closes — every input row kept or counted, an equality over a walked set.
  5. The published file list is exactly what the publisher would upload.

Exits non-zero if any check fails, so it is usable in CI.

Source code in enricher/src/just_dna_enricher/cli.py
@civic_app.command("reproduce")
def civic_reproduce_(
    release: str = typer.Option(
        "01-Aug-2026",
        "--release",
        help="Dated CIViC release to reproduce, e.g. 01-Aug-2026.",
    ),
    out: Path = typer.Option(
        repro_out("civic_reproduce"),
        "--out",
        file_okay=False,
        help=(
            "Working directory. The release files and two independent builds land here. The default "
            "is under data/, which this workspace git-ignores wholesale."
        ),
    ),
    keep: bool = typer.Option(
        False,
        "--keep",
        help="Leave the downloaded release files in place for inspection.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Skip the reference cross-check. The build and determinism checks still run.",
    ),
    submitted: bool = typer.Option(
        False,
        "--submitted",
        help=(
            "Reproduce the wider basis: also download the release's "
            "civic_accepted_and_submitted.vcf and build with it, so the submitted rows and their "
            "coordinates go through every check below rather than only the accepted ones."
        ),
    ),
) -> None:
    """Build the CIViC snapshot from a dated release and check it, end to end.

    **Five checks, and the third is the one worth the network.** The first two are about us; the third
    is about whether the coordinates we produced are real.

    1. **The release downloads and its bytes are recorded** — a sha256 per file, so a rerun that
       disagrees is a finding about the source rather than a mystery.
    2. **Two independent builds are byte-identical** (Principle 7). A parquet has no inherent row
       order, so this is the check that the sort is doing its job.
    3. **Every placed coordinate is cross-checked against the GRCh38 reference sequence.** This is the
       external validation: the snapshot's positions come from RefSeq accessions inside ClinVar HGVS,
       and this asks an unrelated service (refget/seqrepo) whether the reference base at each of those
       positions is what we wrote. A wrong-build or off-by-one placement fails here and nowhere else.
    4. **The drop registry closes** — every input row kept or counted, an equality over a walked set.
    5. **The published file list is exactly what the publisher would upload.**

    Exits non-zero if any check fails, so it is usable in CI.
    """
    from just_dna_format.resolution import ResolutionRow

    from just_dna_enricher.civic_build import (
        CIVIC_COLUMNS,
        CIVIC_EVIDENCE_FILE,
        CIVIC_PROFILE_FILE,
        CIVIC_VARIANT_FILE,
        build_snapshot,
        civic_release_url,
        download_civic_file,
    )
    from just_dna_enricher.civic_vcf import CIVIC_VCF_FILE
    from just_dna_enricher.locations import RELEASE_FILENAME, SNAPSHOT_LICENSE_FILENAME
    from just_dna_enricher.sequences import SequenceProxy, verify_reference_alleles
    from just_dna_enricher.upload import DEFAULT_CIVIC_REPO_ID, plan_reference_snapshot

    failures: list[str] = []

    def check(ok: bool, label: str, detail: str = "") -> None:
        typer.secho(
            f"  {'PASS' if ok else 'FAIL'}  {label}{(' — ' + detail) if detail else ''}",
            fg=typer.colors.GREEN if ok else typer.colors.RED,
        )
        if not ok:
            failures.append(label)

    out.mkdir(parents=True, exist_ok=True)
    typer.echo(f"CIViC release {release}")

    # 1 ── the release, with its bytes recorded
    typer.echo("\n1. Downloading the dated release")
    paths, shas = [], {}
    wanted = [CIVIC_EVIDENCE_FILE, CIVIC_VARIANT_FILE, CIVIC_PROFILE_FILE]
    if submitted:
        # Hashed like every other input, because the whole point of reading it from the dated release
        # rather than the API is that its bytes can be pinned.
        wanted.append(CIVIC_VCF_FILE)
    for filename in wanted:
        got = download_civic_file(out / filename, civic_release_url(release, filename))
        paths.append(got.path)
        shas[filename] = got.sha256
        typer.echo(f"     {filename:38s} {got.sha256[:16]}…  {got.path.stat().st_size:>9,} bytes")
    check(all(shas.values()), "every input file hashed")
    vcf_path = paths.pop() if submitted else None

    # 2 ── two builds, byte for byte
    typer.echo("\n2. Building twice")
    first = build_snapshot(
        *paths,
        out / "build-a",
        release=release,
        evidence_sha256=shas[CIVIC_EVIDENCE_FILE],
        variant_sha256=shas[CIVIC_VARIANT_FILE],
        profile_sha256=shas[CIVIC_PROFILE_FILE],
        vcf=vcf_path,
        vcf_sha256=shas.get(CIVIC_VCF_FILE),
    )
    second = build_snapshot(*paths, out / "build-b", release=release, vcf=vcf_path)
    typer.echo(f"     {first.record_count} rows on {first.variants} variants ({first.status_basis})")
    if first.status_counts:
        typer.echo("     " + " · ".join(f"{k} {v}" for k, v in sorted(first.status_counts.items())))
    check(
        first.parquet_file.read_bytes() == second.parquet_file.read_bytes(),
        "a rebuild is byte-identical (P7)",
    )

    # 4 ── the accounting (run before the slow check, so a broken build fails fast)
    total_dropped = sum(first.dropped.values())
    check(
        first.record_count + total_dropped == first.input_rows,
        "the drop registry accounts for every input row",
        f"{first.input_rows} = {first.record_count} kept + {total_dropped} dropped",
    )
    for reason, count in first.dropped.items():
        typer.echo(f"     dropped {reason:24s} {count:>6,}")

    # 5 ── what would be published
    plan = plan_reference_snapshot(first.out_dir, DEFAULT_CIVIC_REPO_ID)
    expected = {f"data/{first.parquet_file.name}", RELEASE_FILENAME, SNAPSHOT_LICENSE_FILENAME}
    check(
        set(plan.files) == expected,
        "the publish plan is data + release.json + LICENSE",
        ", ".join(sorted(plan.files)),
    )

    frame = pl.read_parquet(first.parquet_file) if (pl := _polars()) else None
    if frame is not None:
        check(list(frame.columns) == list(CIVIC_COLUMNS), "the emitted column order is the fixed one")

    # 3 ── the external check, and the reason this command needs a network
    typer.echo("\n3. Cross-checking placed coordinates against the GRCh38 reference")
    if offline or frame is None:
        typer.secho(
            "     SKIPPED (--offline) — a check that did not run is not a check that passed",
            fg=typer.colors.YELLOW,
        )
    else:
        placed = frame.filter(pl.col("chrom").is_not_null() & pl.col("ref").is_not_null())
        rows = [
            ResolutionRow(
                variant_key=f"{r['chrom']}:{r['start']}:{r['ref']}",
                status="resolved",
                chrom=r["chrom"],
                start=r["start"],
                ref=r["ref"],
            )
            for r in placed.iter_rows(named=True)
        ]
        result = verify_reference_alleles(rows, sequences=SequenceProxy())
        if result.not_checked:
            typer.secho(
                f"     SKIPPED — {result.not_checked}. A check that could not run is not a check "
                f"that passed.",
                fg=typer.colors.YELLOW,
            )
        else:
            # `subjects` rather than `len(rows)`: a row the service answered nothing about is outside
            # the denominator rather than inside it with a clean bill, and reporting the wider number
            # would claim coverage the check did not have.
            check(
                not result.mismatches,
                "every placed ref matches the GRCh38 reference sequence",
                f"{result.subjects} of {len(rows)} coordinate(s) read, {len(result.mismatches)} mismatch(es)",
            )
            for m in result.mismatches[:5]:
                typer.secho(f"     {m}", fg=typer.colors.RED)

    if not keep:
        for path in paths:
            path.unlink(missing_ok=True)

    typer.echo("")
    if failures:
        typer.secho(
            f"REPRODUCTION FAILED: {len(failures)} check(s) — {'; '.join(failures)}",
            fg=typer.colors.RED,
            err=True,
        )
        raise typer.Exit(code=1)
    typer.secho(f"Reproduced {release}: {first.record_count} rows, all checks passed", fg=typer.colors.GREEN)

pubmind_build_

pubmind_build_(
    table: Path | None = typer.Option(
        None,
        "--table",
        exists=True,
        dir_okay=False,
        help="Local hg38_pubmind_db.txt.gz. Omit and pass --download to fetch it from ANNOVAR.",
    ),
    download: bool = typer.Option(
        False,
        "--download",
        help="Download the ANNOVAR-distributed PubMind table into --out first.",
    ),
    out: Path = typer.Option(
        repro_out("pubmind"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/pubmind.parquet + release.json).",
    ),
) -> None

Reduce the ANNOVAR-distributed PubMind table to the parquet snapshot the checks read.

There is no --use flag, and its absence is the design. PubMind's data terms could not be established, and unknown terms warn rather than gate (commercial_use=None never taints a module). A declared-use gate here would refuse every build unconditionally, and a flag feeding a gate that never gates is a flag that does nothing. What the unknown terms do gate is publishing a snapshot or a module carrying these bytes — see pubmind publish, which refuses.

Source code in enricher/src/just_dna_enricher/cli.py
@pubmind_app.command("build")
def pubmind_build_(
    table: Path | None = typer.Option(
        None,
        "--table",
        exists=True,
        dir_okay=False,
        help="Local hg38_pubmind_db.txt.gz. Omit and pass --download to fetch it from ANNOVAR.",
    ),
    download: bool = typer.Option(
        False,
        "--download",
        help="Download the ANNOVAR-distributed PubMind table into --out first.",
    ),
    out: Path = typer.Option(
        repro_out("pubmind"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/pubmind.parquet + release.json).",
    ),
) -> None:
    """Reduce the ANNOVAR-distributed PubMind table to the parquet snapshot the checks read.

    **There is no `--use` flag, and its absence is the design.** PubMind's data terms could not be
    established, and unknown terms warn rather than gate (`commercial_use=None` never taints a
    module). A declared-use gate here would refuse every build unconditionally, and a flag feeding a
    gate that never gates is a flag that does nothing. What the unknown terms do gate is *publishing*
    a snapshot or a module carrying these bytes — see `pubmind publish`, which refuses.
    """
    if table is None and not download:
        typer.secho("Provide --table PATH or --download.", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1)
    fetched = None
    # `PubMindBuildError` covers the download too: it translates `httpx` into `PubMindUnavailable`, a
    # subclass, so this one arm catches a moved bulk URL as well as a malformed table. Without that,
    # ANNOVAR rotating the file — the rot the source's own lack of a cadence invites — printed a
    # traceback instead of this line.
    try:
        if table is None:
            fetched = download_pubmind_table(out / "hg38_pubmind_db.txt.gz")
        result = build_pubmind_snapshot(
            fetched.path if fetched is not None else table,
            out,
            source_url=fetched.url if fetched is not None else None,
            source_sha256=fetched.sha256 if fetched is not None else None,
            source_etag=fetched.etag if fetched is not None else None,
            source_last_modified=fetched.last_modified if fetched is not None else None,
        )
    except (FileNotFoundError, ImportError, PubMindBuildError) as exc:
        typer.secho(f"PUBMIND BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"pubmind snapshot: {result.parquet_file}", fg=typer.colors.GREEN)
    typer.echo(
        f"kept: {result.record_count} of {result.input_rows} row(s)  "
        f"dataset: {result.dataset}  "
        + "  ".join(f"{name}: {count}" for name, count in sorted(result.derivations.items()))
    )
    typer.secho(
        "  dropped: " + ", ".join(f"{name} {count}" for name, count in result.dropped.items()),
        fg=typer.colors.YELLOW,
    )
    typer.secho(
        f"  {result.multi_pvid_keys} of {result.allele_keys} coordinate(s) carry several PVIDs "
        f"(worst {result.max_pvids_per_key}), and {result.contested_keys} of those disagree. Every "
        f"PVID is kept as its own row: choosing a winner would be an ordering nobody defined.",
        fg=typer.colors.YELLOW,
    )
    typer.secho(
        "  data terms unestablished: operator-built and inject-only, never published.",
        fg=typer.colors.YELLOW,
    )

pubmind_publish_

pubmind_publish_() -> None

Refuse to publish the PubMind snapshot, and say why.

The command exists in order to refuse. A missing one reads as an oversight somebody will helpfully add, and the reason belongs where a reader looks for it rather than only in a design document.

Source code in enricher/src/just_dna_enricher/cli.py
@pubmind_app.command("publish")
def pubmind_publish_() -> None:
    """Refuse to publish the PubMind snapshot, and say why.

    The command exists in order to refuse. A missing one reads as an oversight somebody will helpfully
    add, and the reason belongs where a reader looks for it rather than only in a design document.
    """
    typer.secho(f"REFUSED: {PUBMIND_PUBLISH_REFUSAL}", fg=typer.colors.RED, err=True)
    raise typer.Exit(code=1)

constraint_build_

constraint_build_(
    tsv: Path | None = typer.Option(
        None,
        "--tsv",
        exists=True,
        dir_okay=False,
        help="Local gnomAD constraint metrics TSV. Omit and pass --download to fetch it.",
    ),
    download: bool = typer.Option(
        False,
        "--download",
        help="Download the gnomAD v4.1 constraint TSV (95.5 MB) into --out first.",
    ),
    out: Path = typer.Option(
        repro_out("gnomad_constraint"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/gnomad_constraint.parquet + release.json).",
    ),
) -> None

Reduce the per-transcript constraint TSV to the gene-level parquet the resolver reads.

Source code in enricher/src/just_dna_enricher/cli.py
@constraint_app.command("build")
def constraint_build_(
    tsv: Path | None = typer.Option(
        None,
        "--tsv",
        exists=True,
        dir_okay=False,
        help="Local gnomAD constraint metrics TSV. Omit and pass --download to fetch it.",
    ),
    download: bool = typer.Option(
        False,
        "--download",
        help="Download the gnomAD v4.1 constraint TSV (95.5 MB) into --out first.",
    ),
    out: Path = typer.Option(
        repro_out("gnomad_constraint"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/gnomad_constraint.parquet + release.json).",
    ),
) -> None:
    """Reduce the per-transcript constraint TSV to the gene-level parquet the resolver reads."""
    from just_dna_enricher.constraint_build import build_snapshot, download_constraint_tsv

    if tsv is None and not download:
        typer.secho("Provide --tsv PATH or --download.", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1)
    try:
        source = (
            tsv if tsv is not None else download_constraint_tsv(out / "gnomad.v4.1.constraint_metrics.tsv")
        )
        result = build_snapshot(source, out)
    except (FileNotFoundError, ImportError) as exc:
        typer.secho(f"BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"built: {result.out_dir}", fg=typer.colors.GREEN)
    typer.echo(f"genes: {result.gene_count}  from transcript rows: {result.source_rows}")
    if result.unresolved_genes:
        typer.secho(
            f"  dropped (no MANE/canonical Ensembl row): {len(result.unresolved_genes)}",
            fg=typer.colors.YELLOW,
        )

constraint_publish_

constraint_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/*.parquet + release.json).",
    ),
    repo_id: str | None = typer.Option(
        None,
        "--repo",
        help="Target HF dataset (owner/name). Default: just-dna-seq/gnomad_constraint.",
    ),
    commit_message: str | None = typer.Option(
        None, "--message", "-m", help="Commit message."
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded without contacting HuggingFace.",
    ),
) -> None

Create-or-update the dataset repo and upload the built constraint snapshot (publisher/dev).

Source code in enricher/src/just_dna_enricher/cli.py
@constraint_app.command("publish")
def constraint_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/*.parquet + release.json).",
    ),
    repo_id: str | None = typer.Option(
        None,
        "--repo",
        help="Target HF dataset (owner/name). Default: just-dna-seq/gnomad_constraint.",
    ),
    commit_message: str | None = typer.Option(None, "--message", "-m", help="Commit message."),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded without contacting HuggingFace.",
    ),
) -> None:
    """Create-or-update the dataset repo and upload the built constraint snapshot (publisher/dev)."""
    from just_dna_enricher.upload import (
        DEFAULT_CONSTRAINT_REPO_ID,
        plan_reference_snapshot,
        publish_reference_snapshot,
    )

    target = repo_id or DEFAULT_CONSTRAINT_REPO_ID
    if dry_run:
        plan = plan_reference_snapshot(snapshot_dir, target)
        typer.echo(f"Would upload to {plan.repo_id}:")
        for f in plan.files:
            typer.echo(f"  • {f}")
        return
    try:
        plan = publish_reference_snapshot(snapshot_dir, target, commit_message=commit_message)
    except (FileNotFoundError, PermissionError, ImportError) as exc:
        typer.secho(f"PUBLISH FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(
        f"published: {snapshot_dir} → {plan.repo_id} ({len(plan.files)} files)",
        fg=typer.colors.GREEN,
    )

vrs_mint_

vrs_mint_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Substitutions only: indels need the reference sequence, which means a network call.",
    ),
) -> None

Stamp ga4gh:VA.… allele ids onto resolution.csv (substitutions offline, indels online).

Source code in enricher/src/just_dna_enricher/cli.py
@vrs_app.command("mint")
def vrs_mint_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Substitutions only: indels need the reference sequence, which means a network call.",
    ),
) -> None:
    """Stamp ga4gh:VA.… allele ids onto resolution.csv (substitutions offline, indels online)."""
    from just_dna_compiler.compiler import load_csv_rows
    from just_dna_format.resolution import ResolutionRow

    from just_dna_enricher.enrich import _write_resolution_csv
    from just_dna_enricher.vrs import mint_resolution_rows

    path = sidecar_path(spec_dir, "resolution.csv", error=EnrichmentError)
    if not path.exists():
        typer.secho(f"no resolution.csv in {spec_dir} — run `enrich` first.", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1)
    rows, errors, _ = load_csv_rows(path, ResolutionRow, path.name)
    if errors:
        typer.secho(f"{path.name} is invalid: {errors[0]}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1)
    result = mint_resolution_rows(rows, offline=offline)
    _write_resolution_csv(rows, path)
    typer.secho(f"minted: {path}", fg=typer.colors.GREEN)
    typer.echo(
        f"stdlib: {result.minted_stdlib}  normalized: {result.minted_normalized}  "
        f"unmintable: {result.skipped_unmintable}  already present: {result.already_present}"
    )
    typer.echo(f"coverage: {result.identified}/{result.alleles} allele(s) carry a ga4gh:VA. id")
    for line in result.coverage_warnings():
        typer.secho(f"  warning: {line}", fg=typer.colors.YELLOW, err=True)
    for mismatch in result.mismatches:
        typer.secho(f"  mismatch: {mismatch}", fg=typer.colors.YELLOW, err=True)
    try:
        record_verification([_mint_record(result)], spec_dir, error=EnrichmentError)
    except EnrichmentError as exc:
        # The mint did **not** fail — the ids are on disk and the line above says so. What failed is
        # the attestation, and the only way this raises is a module carrying `verification.json` in
        # both legal places, which is a defect in the module's layout rather than in this run. Naming
        # the mint here would tell the author the opposite of what happened.
        typer.secho(f"MINTED, BUT NOT ATTESTED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc

hint_variant_

hint_variant_(
    rsid: str | None = typer.Option(
        None, "--rsid", help="dbSNP id to look up."
    ),
    chrom: str | None = typer.Option(
        None, "--chrom", help="Chromosome (with --start)."
    ),
    start: int | None = typer.Option(
        None,
        "--start",
        help="1-based position (with --chrom).",
    ),
    ref: str | None = typer.Option(
        None,
        "--ref",
        help="Reference allele, for an allele-exact lookup.",
    ),
    alts: str | None = typer.Option(
        None,
        "--alts",
        help="Alt allele(s), comma-separated.",
    ),
    ambiguity: bool = typer.Option(
        False,
        "--ambiguity",
        help="Warn when the answer is not unique.",
    ),
    frequencies: bool = typer.Option(
        False,
        "--frequencies",
        help="Add gnomAD populations (paced: ~6s).",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Snapshots only; never touch the network.",
    ),
    ensembl_cache: Path | None = typer.Option(
        None,
        "--ensembl-cache",
        help="Explicit Ensembl cache.",
    ),
    clinvar_cache: Path | None = typer.Option(
        None,
        "--clinvar-cache",
        help="Explicit ClinVar snapshot.",
    ),
    pubmind_cache: Path | None = typer.Option(
        None,
        "--pubmind-cache",
        help="Explicit PubMind snapshot (see `pubmind build`); $JUST_DNA_PUBMIND_CACHE otherwise.",
    ),
    as_json: bool = typer.Option(
        False,
        "--json",
        help="Emit the full machine answer.",
    ),
) -> None

Validity, coordinates, alleles, populations and clinical calls for one variant.

Nothing is decided for you: a one-to-many rsID returns every locus and a position matching several rsIDs returns every candidate. The coordinate is reported, never written into variants.csv — resolution puts it in resolution.csv, which is where it belongs.

Source code in enricher/src/just_dna_enricher/cli.py
@hint_app.command("variant")
def hint_variant_(
    rsid: str | None = typer.Option(None, "--rsid", help="dbSNP id to look up."),
    chrom: str | None = typer.Option(None, "--chrom", help="Chromosome (with --start)."),
    start: int | None = typer.Option(None, "--start", help="1-based position (with --chrom)."),
    ref: str | None = typer.Option(None, "--ref", help="Reference allele, for an allele-exact lookup."),
    alts: str | None = typer.Option(None, "--alts", help="Alt allele(s), comma-separated."),
    ambiguity: bool = typer.Option(False, "--ambiguity", help="Warn when the answer is not unique."),
    frequencies: bool = typer.Option(False, "--frequencies", help="Add gnomAD populations (paced: ~6s)."),
    offline: bool = typer.Option(False, "--offline", help="Snapshots only; never touch the network."),
    ensembl_cache: Path | None = typer.Option(None, "--ensembl-cache", help="Explicit Ensembl cache."),
    clinvar_cache: Path | None = typer.Option(None, "--clinvar-cache", help="Explicit ClinVar snapshot."),
    pubmind_cache: Path | None = typer.Option(
        None,
        "--pubmind-cache",
        help="Explicit PubMind snapshot (see `pubmind build`); $JUST_DNA_PUBMIND_CACHE otherwise.",
    ),
    as_json: bool = typer.Option(False, "--json", help="Emit the full machine answer."),
) -> None:
    """Validity, coordinates, alleles, populations and clinical calls for one variant.

    Nothing is decided for you: a one-to-many rsID returns every locus and a position matching
    several rsIDs returns every candidate. The coordinate is reported, never written into
    `variants.csv` — resolution puts it in `resolution.csv`, which is where it belongs.
    """
    if rsid is None and (chrom is None or start is None):
        typer.secho("give --rsid, or --chrom and --start", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1)
    hint = lookup_variant(
        rsid=rsid,
        chrom=chrom,
        start=start,
        ref=ref,
        alts=alts,
        ambiguity=ambiguity,
        frequencies=frequencies,
        offline=offline,
        ensembl_cache=ensembl_cache,
        clinvar_cache=clinvar_cache,
        pubmind_cache=pubmind_cache,
    )
    if as_json:
        typer.echo(
            json.dumps(
                {
                    "rsid": hint.rsid,
                    "rsid_status": str(hint.rsid_status) if hint.rsid_status else None,
                    "loci": hint.loci,
                    "rsid_candidates": hint.rsid_candidates,
                    "populations": hint.populations,
                    "clin_sig": hint.clin_sig,
                    "pubmind": hint.pubmind,
                    "vrs_id": hint.vrs_id,
                    "ambiguous": hint.ambiguous,
                    "advisory": as_report_rows(hint),
                    "findings": [f"{f.level}: {f.message}" for f in hint.findings],
                },
                indent=2,
                default=str,
            )
        )
        return
    for locus in hint.loci:
        typer.echo(f"locus\t{locus['chrom']}:{locus['start']}\t{locus.get('ref')}>{locus.get('alts')}")
    for candidate in hint.rsid_candidates:
        typer.echo(f"rsid_candidate\t{candidate}")
    for population in hint.populations:
        af = population.get("allele_frequency")
        # The allele leads the row (S108, RM255): a multi-allelic locus answers with one row per
        # ancestry group *per allele*, and without it two different claims render identically.
        typer.echo(
            f"population\t{population.get('allele')}\t{population.get('population')}"
            f"\tAC={population.get('allele_count')}"
            f"\tAN={population.get('allele_number')}\tAF={'' if af is None else f'{af:.6g}'}"
        )
    # One line per PubMind record, never a rolled-up verdict: several records can describe one
    # position and their calls are allowed to differ, which is the finding rather than untidiness.
    for record in hint.pubmind:
        typer.echo(
            f"pubmind\t{record.get('clin_sig')}\t{record.get('pvid')}"
            f"\tconfidence={record.get('confidence')}\tderivation={record.get('derivation')}"
        )
    _echo_hint(hint)

hint_recover_

hint_recover_(
    chrom: str = typer.Option(
        ...,
        "--chrom",
        help="Chromosome of the old coordinate.",
    ),
    start: int = typer.Option(
        ...,
        "--start",
        help=f"1-based {GRCH37_BUILD} position.",
    ),
    ref: str | None = typer.Option(
        None,
        "--ref",
        help="Reference allele, to narrow the answer.",
    ),
    alts: str | None = typer.Option(
        None,
        "--alts",
        help="Alt allele(s), comma-separated.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Skip the lookup and say so.",
    ),
    as_json: bool = typer.Option(
        False,
        "--json",
        help="Emit the full machine answer.",
    ),
) -> None

Which rs-number GRCh37 dbSNP records at an hg19/GRCh37 coordinate.

For a paper that predates GRCh38. Author the rs-number, not a converted position: an rs-number resolves into a coordinate the compiler can cross-examine, where a lifted-over position becomes the row's only witness to itself. Nothing is written — the rs-number is the row's identity, and a machine filling one migrates variant_key with no authored edit anywhere.

Source code in enricher/src/just_dna_enricher/cli.py
@hint_app.command("recover")
def hint_recover_(
    chrom: str = typer.Option(..., "--chrom", help="Chromosome of the old coordinate."),
    start: int = typer.Option(..., "--start", help=f"1-based {GRCH37_BUILD} position."),
    ref: str | None = typer.Option(None, "--ref", help="Reference allele, to narrow the answer."),
    alts: str | None = typer.Option(None, "--alts", help="Alt allele(s), comma-separated."),
    offline: bool = typer.Option(False, "--offline", help="Skip the lookup and say so."),
    as_json: bool = typer.Option(False, "--json", help="Emit the full machine answer."),
) -> None:
    """Which rs-number GRCh37 dbSNP records at an hg19/GRCh37 coordinate.

    For a paper that predates GRCh38. Author the **rs-number**, not a converted position: an
    rs-number resolves into a coordinate the compiler can cross-examine, where a lifted-over position
    becomes the row's only witness to itself. Nothing is written — the rs-number is the row's
    identity, and a machine filling one migrates `variant_key` with no authored edit anywhere.
    """
    hint = lookup_old_assembly(chrom=chrom, start=start, ref=ref, alts=alts, offline=offline)
    if as_json:
        typer.echo(
            json.dumps(
                {
                    "chrom": hint.recovery.chrom,
                    "start": hint.recovery.start,
                    "genome_build": GRCH37_BUILD,
                    "outcome": hint.recovery.outcome,
                    "rsids": hint.recovery.rsids,
                    "candidates": hint.recovery.candidates,
                    "advisory": as_report_rows(hint),
                    "findings": [f"{f.level}: {f.message}" for f in hint.findings],
                },
                indent=2,
                default=str,
            )
        )
        return
    for candidate in hint.recovery.candidates:
        typer.echo(
            f"candidate\t{candidate['rsid']}\t{GRCH37_BUILD} {hint.recovery.chrom}:"
            f"{candidate['start']}-{candidate['end']}\t{'/'.join(candidate['alleles'])}"
        )
    _echo_hint(hint)

hint_citation_

hint_citation_(
    pmid: str | None = typer.Option(
        None, "--pmid", help="PubMed id to check."
    ),
    doi: str | None = typer.Option(
        None,
        "--doi",
        help="DOI to check (the one you authored).",
    ),
    pmcid: str | None = typer.Option(
        None,
        "--pmcid",
        help="PubMed Central id (PMC…) to resolve to the PubMed id tables key on.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Skip the check and say so.",
    ),
    as_json: bool = typer.Option(
        False,
        "--json",
        help="Emit the full machine answer.",
    ),
) -> None

Does this citation exist, and is it the paper you meant?

A paywall hides the fulltext, never the PubMed record, so existence is answerable for paywalled work; Crossref covers what PubMed does not index at all. Every answer is tri-state — unknown means the registry could not be asked, which is not the same as "no such paper".

Existence is not identity. PMIDs are densely allocated, so a recalled or invented number is very likely to be a real record for a different article, and pmid_exists alone cannot catch a fabricated citation. The title, journal, year and first author come back in the same response and are printed for exactly that comparison (S12).

--pmcid goes the other way. Every pmid in the schema keys on the PubMed id — studies.csv, a binning row's and a pharm_variants.csv row's alike — and a curator holding only a PMC… id had no route to it: the schema refused the cell and named no remedy. This resolves it and then asks PubMed which paper that is. The id is reported, never written: filling pmid from NCBI would make the existence check compare NCBI with itself.

Source code in enricher/src/just_dna_enricher/cli.py
@hint_app.command("citation")
def hint_citation_(
    pmid: str | None = typer.Option(None, "--pmid", help="PubMed id to check."),
    doi: str | None = typer.Option(None, "--doi", help="DOI to check (the one you authored)."),
    pmcid: str | None = typer.Option(
        None, "--pmcid", help="PubMed Central id (PMC…) to resolve to the PubMed id tables key on."
    ),
    offline: bool = typer.Option(False, "--offline", help="Skip the check and say so."),
    as_json: bool = typer.Option(False, "--json", help="Emit the full machine answer."),
) -> None:
    """Does this citation exist, and **is it the paper you meant**?

    A paywall hides the fulltext, never the PubMed record, so existence is answerable for paywalled
    work; Crossref covers what PubMed does not index at all. Every answer is tri-state — `unknown`
    means the registry could not be asked, which is not the same as "no such paper".

    **Existence is not identity.** PMIDs are densely allocated, so a recalled or invented number is
    very likely to be a real record for a different article, and `pmid_exists` alone cannot catch a
    fabricated citation. The title, journal, year and first author come back in the same response and
    are printed for exactly that comparison (S12).

    **`--pmcid` goes the other way.** Every `pmid` in the schema keys on the PubMed id — `studies.csv`,
    a binning row's and a `pharm_variants.csv` row's alike — and a curator holding only a `PMC…` id
    had no route to it: the schema refused the cell and named no remedy. This resolves it and then asks PubMed which paper that is. The id is **reported,
    never written**: filling `pmid` from NCBI would make the existence check compare NCBI with itself.
    """
    if pmid is None and doi is None and pmcid is None:
        typer.secho("give --pmid, --doi or --pmcid", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1)
    hint = lookup_citation(pmid=pmid, doi=doi, pmcid=pmcid, offline=offline)
    if as_json:
        typer.echo(
            json.dumps(
                {
                    "pmid": hint.pmid,
                    "doi": hint.doi,
                    "pmid_exists": hint.pmid_exists,
                    "doi_exists": hint.doi_exists,
                    "registry_doi": hint.registry_doi,
                    "pmcid": hint.pmcid,
                    "open_access": hint.open_access,
                    "abstract_available": hint.abstract_available,
                    "title": hint.title,
                    "journal": hint.journal,
                    "year": hint.year,
                    "first_author": hint.first_author,
                    "advisory": as_report_rows(hint),
                    "findings": [f"{f.level}: {f.message}" for f in hint.findings],
                },
                indent=2,
                default=str,
            )
        )
        return
    for label, value in (("pmid_exists", hint.pmid_exists), ("doi_exists", hint.doi_exists)):
        typer.echo(f"{label}\t{'unknown' if value is None else value}")
    if hint.pmcid:
        typer.echo(f"pmcid\t{hint.pmcid}")
    # Printed unconditionally when known: the whole point is that a caller reading prose can compare
    # what the record says against the paper they had in mind.
    for label, value in (
        ("title", hint.title),
        ("journal", hint.journal),
        ("year", hint.year),
        ("first_author", hint.first_author),
    ):
        if value:
            typer.echo(f"{label}\t{value}")
    _echo_hint(hint)

hint_trait_

hint_trait_(
    curie: str = typer.Argument(
        ..., help="Trait CURIE, e.g. EFO_0004340"
    ),
) -> None

Is this trait id current, obsolete, or unknown?

Source code in enricher/src/just_dna_enricher/cli.py
@hint_app.command("trait")
def hint_trait_(curie: str = typer.Argument(..., help="Trait CURIE, e.g. EFO_0004340")) -> None:
    """Is this trait id current, obsolete, or unknown?"""
    typer.echo(str(lookup_trait(curie)))

hint_gene_

hint_gene_(
    symbol: str = typer.Argument(
        ..., help="Gene symbol, e.g. MTHFR"
    ),
) -> None

Is this gene symbol approved or retired?

Source code in enricher/src/just_dna_enricher/cli.py
@hint_app.command("gene")
def hint_gene_(symbol: str = typer.Argument(..., help="Gene symbol, e.g. MTHFR")) -> None:
    """Is this gene symbol approved or retired?"""
    typer.echo(str(lookup_gene(symbol)))

draft_clinpgx_

draft_clinpgx_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    snapshot: Path = typer.Option(
        ...,
        "--snapshot",
        exists=True,
        file_okay=False,
        help="Built ClinPGx snapshot (see `clinpgx build`). Inject-only; nothing is downloaded.",
    ),
    drug: list[str] = typer.Option(
        [],
        "--drug",
        help="Only annotations naming this drug (repeatable).",
    ),
    gene: list[str] = typer.Option(
        [],
        "--gene",
        help="Only annotations naming this gene (repeatable).",
    ),
    min_evidence_level: str | None = typer.Option(
        None,
        "--min-evidence-level",
        help="Keep annotations at least this strong: 1A|1B|2A|2B|3|4.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use: unstated | non-commercial | commercial. ClinPGx forbids sale.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Report what would be added; write nothing.",
    ),
) -> None

Draft pharm_variants.csv rows from the ClinPGx snapshot — appends, never overwrites a row.

Narrow with --drug and re-run as the module grows. A row already in the file is reported, never replaced: drift against ClinPGx is clinpgx check's finding, not this command's edit to make.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("draft-clinpgx")
def draft_clinpgx_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    snapshot: Path = typer.Option(
        ...,
        "--snapshot",
        exists=True,
        file_okay=False,
        help="Built ClinPGx snapshot (see `clinpgx build`). Inject-only; nothing is downloaded.",
    ),
    drug: list[str] = typer.Option([], "--drug", help="Only annotations naming this drug (repeatable)."),
    gene: list[str] = typer.Option([], "--gene", help="Only annotations naming this gene (repeatable)."),
    min_evidence_level: str | None = typer.Option(
        None, "--min-evidence-level", help="Keep annotations at least this strong: 1A|1B|2A|2B|3|4."
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use: unstated | non-commercial | commercial. ClinPGx forbids sale.",
    ),
    dry_run: bool = typer.Option(False, "--dry-run", help="Report what would be added; write nothing."),
) -> None:
    """Draft pharm_variants.csv rows from the ClinPGx snapshot — appends, never overwrites a row.

    Narrow with --drug and re-run as the module grows. A row already in the file is reported, never
    replaced: drift against ClinPGx is `clinpgx check`'s finding, not this command's edit to make.
    """
    try:
        result = draft_pharm_variants(
            spec_dir,
            snapshot=snapshot,
            genes=gene,
            drugs=drug,
            min_evidence_level=min_evidence_level,
            declared_use=_use(use),
            dry_run=dry_run,
        )
    except (ClinPgxEnrichmentError, DraftError) as exc:
        typer.secho(f"DRAFT FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if result.skipped:
        for warning in result.warnings:
            typer.secho(f"  skipped: {warning}", fg=typer.colors.YELLOW, err=True)
        return
    for report in result.reports:
        typer.echo(f"  {report}")
        for outcome in report.differs:
            typer.secho(f"    {outcome}", fg=typer.colors.YELLOW)
    for warning in result.warnings:
        typer.secho(f"  warning: {warning}", fg=typer.colors.YELLOW, err=True)
    verb = "would add" if dry_run else "added"
    typer.secho(f"{verb} {result.added} row(s) in {spec_dir}", fg=typer.colors.GREEN)

draft_panel_

draft_panel_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    gene: list[str] = typer.Option(
        [],
        "--gene",
        help="Gene to draft rows for (repeatable). Required for every source but mitomap-miss, whose increment is asked for as a whole and where --gene only filters.",
    ),
    source: str = typer.Option(
        "clinvar",
        "--source",
        help="Which authority to draft the calls from: clinvar (the default); pubmind — an LLM's reading of the literature, which needs an operator-built snapshot and still reads the ClinVar one for its gene attribution; civic — curated cancer interpretations, which writes the DIRECTION axis rather than clin_sig and needs a `civic build` snapshot; or mitomap-miss — the curated mtDNA calls MITOMAP publishes and the ClinVar cache does not, which needs a `mitomap miss` snapshot.",
    ),
    mitomap_miss_cache: Path | None = typer.Option(
        None,
        "--mitomap-miss-cache",
        exists=True,
        file_okay=False,
        help="Built MITOMAP-miss snapshot (see `mitomap miss`). Only read under --source mitomap-miss; omit it and $JUST_DNA_MITOMAP_MISS_CACHE is used.",
    ),
    civic_cache: Path | None = typer.Option(
        None,
        "--civic-cache",
        exists=True,
        file_okay=False,
        help="Built CIViC snapshot (see `civic build`). Only read under --source civic; omit it and $JUST_DNA_CIVIC_CACHE is used.",
    ),
    snapshot: Path | None = typer.Option(
        None,
        "--snapshot",
        exists=True,
        file_okay=False,
        help="Built ClinVar snapshot (see `clinvar build`). Omit it and the cache is used, or the published snapshot downloaded — the citations table comes with it, which is what a panel needs to compile. Read for its gene attribution under --source pubmind, which publishes no gene column of its own.",
    ),
    pubmind_cache: Path | None = typer.Option(
        None,
        "--pubmind-cache",
        exists=True,
        file_okay=False,
        help="Built PubMind snapshot (see `pubmind build`), for --source pubmind. Omit it and $JUST_DNA_PUBMIND_CACHE is read; there is no published one to download.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Use a local snapshot only: never download one.",
    ),
    download: bool = typer.Option(
        True,
        "--download/--no-download",
        help="Provision the published snapshot when no local one is found. Fetching it is this command's only network use, so --no-download coincides with --offline today; it is a separate switch because it says 'do not go and get one', not 'make no request'.",
    ),
    clin_sig: str | None = typer.Option(
        None,
        "--clin-sig",
        help="Comma-separated calls to include. Default: pathogenic,likely_pathogenic.",
    ),
    min_review_stars: int = typer.Option(
        2,
        "--min-review-stars",
        min=0,
        max=4,
        help="Review-status floor, --source clinvar only. 2 = multiple submitters, no conflicts.",
    ),
    max_citations: int = typer.Option(
        3,
        "--max-citations",
        min=0,
        help="Study rows to draft per variant from ClinVar's literature links. 0 disables. --source clinvar only: PubMind's channel carries no PMID.",
    ),
    min_confidence: int = typer.Option(
        DEFAULT_MIN_CONFIDENCE,
        "--min-confidence",
        min=0,
        max=3,
        help="Evidence-depth floor, --source pubmind only. PubMind's confidence counts how much of the literature spoke, 0-3; 1 means more than a single mention.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use (ClinVar is public domain).",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Report what would be added; write nothing.",
    ),
) -> None

Draft a gene panel's variants.csv rows from an authority — appends, never overwrites a row.

The drafted rows carry a genotype placeholder, so the module will not compile until you decide what each finding is about. That is deliberate: ClinVar publishes alleles, and whether carrying one is a carrier state or an affected one follows from the condition's inheritance mode, which the source does not say. Rows land in their gene's block, and a re-run leaves anything already there — stub or filled — exactly as it is.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("draft-panel")
def draft_panel_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    gene: list[str] = typer.Option(
        [],
        "--gene",
        help=(
            "Gene to draft rows for (repeatable). Required for every source but mitomap-miss, whose "
            "increment is asked for as a whole and where --gene only filters."
        ),
    ),
    source: str = typer.Option(
        "clinvar",
        "--source",
        help=(
            "Which authority to draft the calls from: clinvar (the default); pubmind — an LLM's "
            "reading of the literature, which needs an operator-built snapshot and still reads the "
            "ClinVar one for its gene attribution; civic — curated cancer interpretations, which "
            "writes the DIRECTION axis rather than clin_sig and needs a `civic build` snapshot; or "
            "mitomap-miss — the curated mtDNA calls MITOMAP publishes and the ClinVar cache does "
            "not, which needs a `mitomap miss` snapshot."
        ),
    ),
    mitomap_miss_cache: Path | None = typer.Option(
        None,
        "--mitomap-miss-cache",
        exists=True,
        file_okay=False,
        help=(
            "Built MITOMAP-miss snapshot (see `mitomap miss`). Only read under "
            "--source mitomap-miss; omit it and $JUST_DNA_MITOMAP_MISS_CACHE is used."
        ),
    ),
    civic_cache: Path | None = typer.Option(
        None,
        "--civic-cache",
        exists=True,
        file_okay=False,
        help=(
            "Built CIViC snapshot (see `civic build`). Only read under --source civic; omit it and "
            "$JUST_DNA_CIVIC_CACHE is used."
        ),
    ),
    snapshot: Path | None = typer.Option(
        None,
        "--snapshot",
        exists=True,
        file_okay=False,
        help=(
            "Built ClinVar snapshot (see `clinvar build`). Omit it and the cache is used, or the "
            "published snapshot downloaded — the citations table comes with it, which is what a panel "
            "needs to compile. Read for its gene attribution under --source pubmind, which publishes "
            "no gene column of its own."
        ),
    ),
    pubmind_cache: Path | None = typer.Option(
        None,
        "--pubmind-cache",
        exists=True,
        file_okay=False,
        help=(
            "Built PubMind snapshot (see `pubmind build`), for --source pubmind. Omit it and "
            "$JUST_DNA_PUBMIND_CACHE is read; there is no published one to download."
        ),
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Use a local snapshot only: never download one.",
    ),
    download: bool = typer.Option(
        True,
        "--download/--no-download",
        help="Provision the published snapshot when no local one is found. Fetching it is this "
        "command's only network use, so --no-download coincides with --offline today; it is a "
        "separate switch because it says 'do not go and get one', not 'make no request'.",
    ),
    clin_sig: str | None = typer.Option(
        None,
        "--clin-sig",
        help="Comma-separated calls to include. Default: pathogenic,likely_pathogenic.",
    ),
    min_review_stars: int = typer.Option(
        2,
        "--min-review-stars",
        min=0,
        max=4,
        help="Review-status floor, --source clinvar only. 2 = multiple submitters, no conflicts.",
    ),
    max_citations: int = typer.Option(
        3,
        "--max-citations",
        min=0,
        help="Study rows to draft per variant from ClinVar's literature links. 0 disables. "
        "--source clinvar only: PubMind's channel carries no PMID.",
    ),
    min_confidence: int = typer.Option(
        DEFAULT_MIN_CONFIDENCE,
        "--min-confidence",
        min=0,
        max=3,
        help="Evidence-depth floor, --source pubmind only. PubMind's confidence counts how much of "
        "the literature spoke, 0-3; 1 means more than a single mention.",
    ),
    use: str = typer.Option("unstated", "--use", help="Declared use (ClinVar is public domain)."),
    dry_run: bool = typer.Option(False, "--dry-run", help="Report what would be added; write nothing."),
) -> None:
    """Draft a gene panel's variants.csv rows from an authority — appends, never overwrites a row.

    The drafted rows carry a **genotype placeholder**, so the module will not compile until you decide
    what each finding is about. That is deliberate: ClinVar publishes alleles, and whether carrying one
    is a carrier state or an affected one follows from the condition's inheritance mode, which the
    source does not say. Rows land in their gene's block, and a re-run leaves anything already there —
    stub or filled — exactly as it is.
    """
    resolved = _panel_source(source)
    if resolved is None:
        typer.secho(
            f"--source {source!r} is not an authority this command drafts from. "
            f"Known: {', '.join(sorted(PANEL_SOURCES))}.",
            fg=typer.colors.RED,
            err=True,
        )
        raise typer.Exit(code=1)
    source = resolved
    if not gene and source in _GENE_SCOPED_PANEL_SOURCES:
        # Refused rather than defaulted to "every gene": ClinVar's snapshot is 4.4M records and CIViC's
        # is a cancer corpus, so an unfiltered draft from either is not a panel, it is the source.
        typer.secho(
            f"--source {source} drafts a gene panel and needs at least one --gene. Only "
            f"--source {MITOMAP_MISS_SOURCE} is asked for as a whole, because its snapshot IS the "
            f"increment.",
            fg=typer.colors.RED,
            err=True,
        )
        raise typer.Exit(code=2)
    calls = frozenset(c.strip() for c in clin_sig.split(",") if c.strip()) if clin_sig else None
    # A dial belonging to the other authority, set to something other than its default, is named
    # rather than silently ignored: a run that honoured neither the flag nor the author's expectation
    # is the failure this reports before it happens.
    for dial, value, default, belongs in (
        ("--min-review-stars", min_review_stars, 2, "clinvar"),
        ("--max-citations", max_citations, 3, "clinvar"),
        ("--min-confidence", min_confidence, DEFAULT_MIN_CONFIDENCE, "pubmind"),
    ):
        if value != default and source != belongs:
            typer.secho(
                f"  warning: {dial} is a --source {belongs} dial and does nothing under --source {source}",
                fg=typer.colors.YELLOW,
                err=True,
            )
    # `--clin-sig` is the third dial belonging elsewhere, and it is named outside the loop above
    # because it is not merely inert under --source civic: CIViC's germline clinical-significance
    # calls are five in all with none benign-class, which is exactly why that provider writes
    # `direction` instead. Once, not once per dial — a warning emitted from inside the loop printed
    # three times for one condition.
    if calls and source == "civic":
        typer.secho(
            "  warning: --clin-sig does nothing under --source civic, which drafts the "
            "direction axis (risk/protective) rather than clinical significance",
            fg=typer.colors.YELLOW,
            err=True,
        )
    # And the fourth, for its own reason again: MITOMAP's increment is already filtered to the five
    # documented VCEP classes, and everything outside them is *withheld* rather than assigned. A
    # `--clin-sig` there would narrow a set this command has no way to widen.
    if calls and source == MITOMAP_MISS_SOURCE:
        typer.secho(
            f"  warning: --clin-sig does nothing under --source {MITOMAP_MISS_SOURCE}, which drafts "
            f"exactly the five documented ClinGen mtDNA VCEP classes and withholds everything else",
            fg=typer.colors.YELLOW,
            err=True,
        )
    try:
        if source == MITOMAP_MISS_SOURCE:
            result = draft_panel_from_mitomap_miss(
                spec_dir,
                gene,
                snapshot=mitomap_miss_cache,
                declared_use=_use(use),
                dry_run=dry_run,
            )
        elif source == "civic":
            result = draft_panel_from_civic(
                spec_dir,
                gene,
                snapshot=civic_cache,
                declared_use=_use(use),
                offline=offline,
                dry_run=dry_run,
            )
        elif source == "pubmind":
            result = draft_gene_panel_from_pubmind(
                spec_dir,
                gene,
                snapshot=snapshot,
                pubmind_snapshot=pubmind_cache,
                offline=offline,
                download=download,
                **({"clin_sig": calls} if calls else {}),
                min_confidence=min_confidence,
                declared_use=_use(use),
                dry_run=dry_run,
            )
        else:
            result = draft_gene_panel(
                spec_dir,
                gene,
                snapshot=snapshot,
                offline=offline,
                download=download,
                **({"clin_sig": calls} if calls else {}),
                min_review_stars=min_review_stars,
                max_citations=max_citations,
                declared_use=_use(use),
                dry_run=dry_run,
            )
    except (
        ClinVarDraftError,
        PubMindDraftError,
        MitomapDraftError,
        *_DRAFT_PRECONDITION_ERRORS,
    ) as exc:
        typer.secho(f"DRAFT FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    # Only the ClinVar result carries `skipped`; `PubMindDraftResult` has no such field. A
    # `getattr(..., False)` default reads "this type cannot skip" and "this run did not skip" as the
    # same answer, so a skip path added to the PubMind provider would go dark with nothing failing.
    # Ask whether the field exists, then read it.
    if hasattr(result, "skipped") and result.skipped:
        for warning in result.warnings:
            typer.secho(f"  skipped: {warning}", fg=typer.colors.YELLOW, err=True)
        return
    for report in result.reports:
        typer.echo(f"  {report}")
    for warning in result.warnings:
        typer.secho(f"  warning: {warning}", fg=typer.colors.YELLOW, err=True)
    verb = "would add" if dry_run else "added"
    # Per table, never a rolled-up total. The draft writes `variants.csv` AND `studies.csv`, so a single
    # number matches neither file — `ClinVarDraftResult.added` says as much in its own docstring.
    breakdown = ", ".join(f"{r.csv_name} {len(r.added)}" for r in result.reports) or "nothing"
    typer.secho(f"{verb}: {breakdown} — in {spec_dir}", fg=typer.colors.GREEN)

clinvar_citations_

clinvar_citations_(
    out: Path = typer.Option(
        ...,
        "--out",
        file_okay=False,
        help="Existing ClinVar snapshot dir.",
    ),
    citations_txt: Path | None = typer.Option(
        None,
        "--citations",
        exists=True,
        dir_okay=False,
        help="Local var_citations.txt.",
    ),
    download: bool = typer.Option(
        False,
        "--download",
        help="Fetch var_citations.txt first.",
    ),
    url: str = typer.Option(
        DEFAULT_CITATIONS_URL,
        "--url",
        help="Source for --download.",
    ),
) -> None

Add ClinVar's literature links to a snapshot: data/citations.parquet ([dev], needs polars).

Separate from clinvar build because ClinVar publishes citations separately from the VCF — which is precisely why a drafted gene panel could not compile without this: studies.csv is mandatory and the VCF carries no PMIDs. Written beside the snapshot, so an existing cache keeps its bytes.

Source code in enricher/src/just_dna_enricher/cli.py
@clinvar_app.command("citations")
def clinvar_citations_(
    out: Path = typer.Option(..., "--out", file_okay=False, help="Existing ClinVar snapshot dir."),
    citations_txt: Path | None = typer.Option(
        None, "--citations", exists=True, dir_okay=False, help="Local var_citations.txt."
    ),
    download: bool = typer.Option(False, "--download", help="Fetch var_citations.txt first."),
    url: str = typer.Option(DEFAULT_CITATIONS_URL, "--url", help="Source for --download."),
) -> None:
    """Add ClinVar's literature links to a snapshot: `data/citations.parquet` (\\[dev], needs polars).

    Separate from `clinvar build` because ClinVar publishes citations separately from the VCF — which
    is precisely why a drafted gene panel could not compile without this: `studies.csv` is mandatory
    and the VCF carries no PMIDs. Written beside the snapshot, so an existing cache keeps its bytes.
    """
    if citations_txt is None and not download:
        typer.secho("give --citations, or --download", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1)
    source_sha: str | None = None
    if citations_txt is None:
        path, source_sha = download_var_citations(out / "var_citations.txt", url=url)
    else:
        path = citations_txt
    try:
        result = build_citations(path, out, source_url=url, source_sha256=source_sha)
    except (ImportError, RuntimeError) as exc:
        typer.secho(f"CITATIONS BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(
        f"wrote {result.row_count} PubMed citation link(s) under {out / CITATIONS_DIRNAME}",
        fg=typer.colors.GREEN,
    )
    if result.release_updated:
        # ClinVar publishes citations on its own cadence, so a snapshot can carry two releases — the
        # block says which, and `clinvar publish` now ships the table with the data.
        typer.echo(f"  recorded the citations provenance in {out / RELEASE_FILENAME}")
    else:
        typer.secho(
            f"  could not record the citations provenance in {out / RELEASE_FILENAME} — the snapshot "
            f"will not say which citations release it carries",
            fg=typer.colors.YELLOW,
            err=True,
        )

litvar_coverage_

litvar_coverage_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="No network; every locus is recorded as unchecked.",
    ),
    quiet: bool = typer.Option(
        False,
        "--quiet",
        help="Only the tier summary, not a line per locus.",
    ),
) -> None

Report LitVar's literature coverage per locus, naming the tier that answered.

It answers which papers discuss an allele that is already identified. It does not answer which allele a name meant — those read as the same question and are not. Measured against the two hardest records in this repository (CIViC 1955 and 2131, four candidate alleles with registered CAIDs), the index returns no node for any of them, because PubTator3 mines titles and abstracts and those alleles live in a table inside a paywalled paper. Do not reach for this to recover an identity.

Writes no row and no sources.csv entry: nothing here reaches a module's tables, so the module does not use this source. What it does write is verification.json — an attestation that the question was put, over how many loci, and at which tier each was answered.

Source code in enricher/src/just_dna_enricher/cli.py
@litvar_app.command("coverage")
def litvar_coverage_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    offline: bool = typer.Option(
        False, "--offline", help="No network; every locus is recorded as unchecked."
    ),
    quiet: bool = typer.Option(False, "--quiet", help="Only the tier summary, not a line per locus."),
) -> None:
    """Report LitVar's literature coverage per locus, naming the tier that answered.

    **It answers *which papers discuss an allele that is already identified*. It does not answer
    *which allele a name meant*** — those read as the same question and are not. Measured against the
    two hardest records in this repository (CIViC 1955 and 2131, four candidate alleles with
    registered CAIDs), the index returns no node for any of them, because PubTator3 mines titles and
    abstracts and those alleles live in a table inside a paywalled paper. Do not reach for this to
    recover an identity.

    Writes no row and no `sources.csv` entry: nothing here reaches a module's tables, so the module
    does not *use* this source. What it does write is `verification.json` — an attestation that the
    question was put, over how many loci, and at which tier each was answered.
    """
    try:
        report = check_literature_coverage(spec_dir, offline=offline)
    except (ValueError, LitvarError) as exc:
        typer.secho(f"LITERATURE COVERAGE FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if not report.loci:
        # No attestation, for `check-identifiers`' reason: with no rsID-bearing row there is no
        # question to record having put, and minting a nonce would create a `verification.json` on a
        # module that never asked for one.
        typer.secho("no authored table names an rsID — nothing to ask LitVar about", fg=typer.colors.YELLOW)
        return
    typer.echo(
        f"loci: {len(report.loci)}"
        f" (from {len(report.tables_read)} table(s): {', '.join(report.tables_read) or 'none'})"
    )
    for name, why in sorted(report.tables_not_read.items()):
        if why != "not present":
            typer.secho(
                f"  {name} names rsIDs and could not be read ({why}) — its loci were NOT asked about",
                fg=typer.colors.YELLOW,
                err=True,
            )
    if not quiet:
        for locus in report.loci:
            typer.echo(f"  {coverage_reason(locus)}")
            if locus.tier == "allele":
                # Both numbers, side by side and labelled. 328 and 3,945 are both true about
                # rs429358 and only one of them is about the allele in the module.
                typer.echo(
                    f"    allele-resolved: {locus.allele_pmids} paper(s); position node: "
                    f"{locus.position_pmids}; on the position node and no allele node: "
                    f"{locus.position_only_pmids}"
                )
            elif locus.tier == "position":
                typer.echo(
                    f"    position node: {locus.position_pmids} paper(s); of those, "
                    f"{locus.position_only_pmids} sit on no allele node"
                )
    for tier in ("allele", "position", "absent", "unchecked"):
        typer.echo(f"{tier}: {len(report.at(tier))}")
    typer.echo(f"papers on a position node that no allele node claims: {report.position_only_residue}")
    if report.degraded:
        typer.secho(
            f"allele-level questions answered position-level: "
            f"{', '.join(locus.rsid for locus in report.degraded)}",
            fg=typer.colors.YELLOW,
        )
    try:
        record_verification(litvar_records(report), spec_dir, error=EnrichmentError)
    except EnrichmentError as exc:
        typer.secho(f"CHECKED, BUT NOT ATTESTED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc

litvar_gene_

litvar_gene_(
    gene: str = typer.Argument(
        ..., help="Gene symbol, e.g. HFE"
    ),
) -> None

List every node LitVar holds under a gene symbol, grouped by tier. Writes nothing.

This is the endpoint that serves line-delimited Python repr() rather than JSON, and the tier split is the reason to look: of 588 HFE nodes on 2026-09-01, 220 are rsID-only, 69 carry a ClinGen allele id, exactly one is the gene node, and the remaining 298 are unnormalized protein strings — text a miner saw, not an identity anything should join on.

Source code in enricher/src/just_dna_enricher/cli.py
@litvar_app.command("gene")
def litvar_gene_(
    gene: str = typer.Argument(..., help="Gene symbol, e.g. HFE"),
) -> None:
    """List every node LitVar holds under a gene symbol, grouped by tier. Writes nothing.

    This is the endpoint that serves line-delimited Python `repr()` rather than JSON, and the tier
    split is the reason to look: of 588 HFE nodes on 2026-09-01, 220 are rsID-only, 69 carry a
    ClinGen allele id, exactly one is the gene node, and the remaining 298 are unnormalized protein
    strings — text a miner saw, not an identity anything should join on.
    """
    try:
        nodes = LitvarClient().gene_nodes(gene)
    except LitvarError as exc:
        typer.secho(f"LITVAR GENE LOOKUP FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if not nodes:
        typer.secho(f"LitVar holds no node under {gene}", fg=typer.colors.YELLOW)
        return
    for tier in ("clingen", "rsid", "gene", "mention"):
        at_tier = [node for node in nodes if node.tier == tier]
        papers = sum(node.pmid_count or 0 for node in at_tier)
        typer.echo(f"{tier}: {len(at_tier)} node(s), {papers} paper-node link(s)")
    typer.secho(
        "a `mention` node is an unnormalized protein string under a gene id — it names no allele, "
        "and nothing here should be joined on as an identity",
        fg=typer.colors.YELLOW,
    )

mane_build_

mane_build_(
    download: bool = typer.Option(
        False,
        "--download",
        help="Fetch the release from NCBI. Without --release the newest version is discovered from current/README_versions.txt and then pinned to its versioned directory.",
    ),
    release: str | None = typer.Option(
        None,
        "--release",
        help="MANE version to pin, e.g. 1.5. Resolves to release_<version>/, never current/.",
    ),
    summary: Path | None = typer.Option(
        None,
        "--summary",
        exists=True,
        dir_okay=False,
        help="Local MANE.GRCh38.v<ver>.summary.txt.gz. Use instead of --download to build offline.",
    ),
    changed: Path | None = typer.Option(
        None,
        "--changed",
        exists=True,
        dir_okay=False,
        help="Local MANE.GRCh38.v<ver>.changed_select_accessions.txt.gz.",
    ),
    not_in_mane: Path | None = typer.Option(
        None,
        "--not-in-mane",
        exists=True,
        dir_okay=False,
        help="Local MANE.GRCh38.v<ver>.protein_coding_genes_not_in_mane.txt.gz.",
    ),
    versions: Path | None = typer.Option(
        None,
        "--versions",
        exists=True,
        dir_okay=False,
        help="Local README_versions.txt. Optional, and the only way an offline build can name its release: a filename is never parsed for one.",
    ),
    out: Path = typer.Option(
        repro_out("mane"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/*.parquet + release.json).",
    ),
) -> None

Reduce one pinned MANE release to the three parquet tables the numbering frame needs.

All three files, in one pass. The summary is the frame; changed_select_accessions is the currency check and Update_Affects_CDS is the numbering-frame axis stated by the source; the negative roster is what makes "MANE has no answer for this gene" distinguishable from "nobody asked", with the reason attached. Shipping the cache without the thing that notices it going stale is the defect this command exists to close.

There is no --offline flag: the off-switch is passing the local files instead of --download. And there is no --use flag, because NCBI states a policy rather than a licence — every gating axis is unknown, and check_declared_use returns a skip for an unknown whatever the declaration says, so the gate would silently skip every build. A flag feeding a gate that never gates is a flag that does nothing (@acquisition-gate-is-not-a-read-gate).

MANE is the default, not the answer. A gene with two rows carries two CDS numbering frames and MANE_status says which; a gene with one row says nothing about the isoforms MANE does not carry, so a pass treating this table as an oracle would be wrong in a way the table cannot report.

Source code in enricher/src/just_dna_enricher/cli.py
@mane_app.command("build")
def mane_build_(
    download: bool = typer.Option(
        False,
        "--download",
        help=(
            "Fetch the release from NCBI. Without --release the newest version is discovered from "
            "current/README_versions.txt and then pinned to its versioned directory."
        ),
    ),
    release: str | None = typer.Option(
        None,
        "--release",
        help="MANE version to pin, e.g. 1.5. Resolves to release_<version>/, never current/.",
    ),
    summary: Path | None = typer.Option(
        None,
        "--summary",
        exists=True,
        dir_okay=False,
        help="Local MANE.GRCh38.v<ver>.summary.txt.gz. Use instead of --download to build offline.",
    ),
    changed: Path | None = typer.Option(
        None,
        "--changed",
        exists=True,
        dir_okay=False,
        help="Local MANE.GRCh38.v<ver>.changed_select_accessions.txt.gz.",
    ),
    not_in_mane: Path | None = typer.Option(
        None,
        "--not-in-mane",
        exists=True,
        dir_okay=False,
        help="Local MANE.GRCh38.v<ver>.protein_coding_genes_not_in_mane.txt.gz.",
    ),
    versions: Path | None = typer.Option(
        None,
        "--versions",
        exists=True,
        dir_okay=False,
        help=(
            "Local README_versions.txt. Optional, and the only way an offline build can name its "
            "release: a filename is never parsed for one."
        ),
    ),
    out: Path = typer.Option(
        repro_out("mane"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/*.parquet + release.json).",
    ),
) -> None:
    """Reduce one pinned MANE release to the three parquet tables the numbering frame needs.

    **All three files, in one pass.** The summary is the frame; `changed_select_accessions` is the
    currency check and `Update_Affects_CDS` is the numbering-frame axis stated by the source; the
    negative roster is what makes "MANE has no answer for this gene" distinguishable from "nobody
    asked", with the reason attached. Shipping the cache without the thing that notices it going
    stale is the defect this command exists to close.

    **There is no `--offline` flag**: the off-switch is passing the local files instead of
    `--download`. And there is no `--use` flag, because NCBI states a policy rather than a licence —
    every gating axis is unknown, and `check_declared_use` returns a *skip* for an unknown whatever
    the declaration says, so the gate would silently skip every build. A flag feeding a gate that
    never gates is a flag that does nothing (`@acquisition-gate-is-not-a-read-gate`).

    **MANE is the default, not the answer.** A gene with two rows carries two CDS numbering frames
    and `MANE_status` says which; a gene with one row says nothing about the isoforms MANE does not
    carry, so a pass treating this table as an oracle would be wrong in a way the table cannot report.
    """
    from just_dna_enricher.mane_build import (
        MANE_TABLES,
        MANE_VERSIONS_FILENAME,
        ManeBuildError,
        build_snapshot,
        discover_current_release,
        download_mane_file,
        mane_release_url,
        mane_versions_url,
    )

    local = {
        "summary": summary,
        "changed_select_accessions": changed,
        "protein_coding_genes_not_in_mane": not_in_mane,
    }
    given = {name: path for name, path in local.items() if path is not None}
    if download and given:
        raise typer.BadParameter(
            "pass --download to fetch a release or the local file flags to build from files you "
            "already hold, not both. Two sources for one input is a build whose provenance nothing "
            "can state."
        )
    if not download and len(given) != len(local):
        raise typer.BadParameter(
            "pass --download, or all three of --summary, --changed and --not-in-mane. Two of the "
            "three is not a snapshot: the summary is the frame, the changed-accession list is what "
            "notices the frame moving, and the negative roster is what tells a gene MANE excluded "
            "from a gene nobody asked about."
        )
    if release is not None and not download:
        raise typer.BadParameter(
            "--release pins which directory to fetch from, so it only means something with "
            "--download. To name the release of files you already hold, pass their "
            "README_versions.txt as --versions — that is the source's own statement, where a bare "
            "--release would be ours about somebody else's bytes."
        )
    if versions is not None and download:
        raise typer.BadParameter(
            "--versions names the release of files you already hold; a --download build fetches the "
            "release's own README_versions.txt and would ignore yours. Drop one — a flag that is "
            "silently overwritten is worse than a flag that is refused."
        )

    downloads: dict = {}
    versions_file = versions
    try:
        if download:
            pinned = release or discover_current_release()
            out.mkdir(parents=True, exist_ok=True)
            fetched_versions = download_mane_file(out / MANE_VERSIONS_FILENAME, mane_versions_url(pinned))
            versions_file = fetched_versions.path
            for table in MANE_TABLES:
                filename = f"MANE.GRCh38.v{pinned}.{table.source_suffix}"
                downloads[table.name] = download_mane_file(
                    out / filename, mane_release_url(pinned, table.source_suffix)
                )
            inputs = {name: got.path for name, got in downloads.items()}
        else:
            pinned = None
            inputs = given
        result = build_snapshot(
            inputs, out, versions_file=versions_file, release=pinned, downloads=downloads or None
        )
    except (FileNotFoundError, ImportError, ManeBuildError) as exc:
        typer.secho(f"MANE BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc

    typer.secho(f"mane snapshot: {result.out_dir / 'data'}", fg=typer.colors.GREEN)
    typer.echo(
        f"dataset: {result.dataset or 'unknown release'}  "
        + "  ".join(f"{name}: {count}" for name, count in result.rows.items())
    )
    typer.echo(
        "  MANE_status: "
        + ", ".join(f"{status} {count}" for status, count in sorted(result.mane_status_counts.items()))
    )
    typer.secho(
        "  changed MANE Select accessions: "
        + ", ".join(
            f"Update_Affects_CDS {token} {count}"
            for token, count in sorted(result.update_affects_cds_counts.items())
        )
        + " — a gene absent from that table has had a stable numbering frame.",
        fg=typer.colors.YELLOW,
    )
    typer.secho(
        "  excluded genes: "
        + ", ".join(f"{reason} {count}" for reason, count in sorted(result.excluded_reasons.items())),
        fg=typer.colors.YELLOW,
    )
    # The withheld cells reach the terminal too, not only `release.json`: a residue an operator
    # cannot see is one nobody looks for. Printed only when there is one — the measured zeros are in
    # `release.json`, where a reader who wants to know that nothing was withheld can find them.
    withheld = {
        **{f"gene id ({table})": n for table, n in result.unparsable_gene_id.items() if n},
        **({"coordinate": result.unparsable_coordinate} if result.unparsable_coordinate else {}),
        **(
            {"Update_Affects_CDS": result.unparsable_update_affects_cds}
            if result.unparsable_update_affects_cds
            else {}
        ),
    }
    if withheld:
        typer.secho(
            "  withheld cells (the row was kept): "
            + ", ".join(f"{label} {count}" for label, count in sorted(withheld.items())),
            fg=typer.colors.YELLOW,
        )
    typer.secho(
        "  MANE is the default, not the answer: it shows a gene's second clinical transcript where "
        "there is one, and says nothing about isoforms it does not carry.",
        fg=typer.colors.YELLOW,
    )

strchive_build_

strchive_build_(
    out: Path = typer.Option(
        repro_out("strchive"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes STRchive-loci.json + release.json).",
    ),
    catalogue: Path | None = typer.Option(
        None,
        "--catalogue",
        exists=True,
        dir_okay=False,
        help="A STRchive-loci.json you already have. Without it the file is downloaded.",
    ),
    release: str | None = typer.Option(
        None,
        "--release",
        help="Upstream release tag to pin, e.g. v2.26.0. Without it, the default branch, unlabelled.",
    ),
) -> None

Fetch (or copy in) the STRchive catalogue and record its provenance beside it.

Pin a release: the default branch moves, so a comparison whose reference is "whatever was there that afternoon" cannot be re-run, and only a pinned build gets a dataset label the verification record can name.

Source code in enricher/src/just_dna_enricher/cli.py
@strchive_app.command("build")
def strchive_build_(
    out: Path = typer.Option(
        repro_out("strchive"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes STRchive-loci.json + release.json).",
    ),
    catalogue: Path | None = typer.Option(
        None,
        "--catalogue",
        exists=True,
        dir_okay=False,
        help="A STRchive-loci.json you already have. Without it the file is downloaded.",
    ),
    release: str | None = typer.Option(
        None,
        "--release",
        help="Upstream release tag to pin, e.g. v2.26.0. Without it, the default branch, unlabelled.",
    ),
) -> None:
    """Fetch (or copy in) the STRchive catalogue and record its provenance beside it.

    Pin a release: the default branch moves, so a comparison whose reference is "whatever was there
    that afternoon" cannot be re-run, and only a pinned build gets a `dataset` label the verification
    record can name.
    """
    from just_dna_enricher.strchive_build import build_strchive_snapshot

    try:
        result = build_strchive_snapshot(out, catalogue=catalogue, release=release)
    except (StrchiveError, OSError) as exc:
        typer.secho(f"STRCHIVE BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"built: {result.catalogue_file}", fg=typer.colors.GREEN)
    typer.echo(f"  {result.locus_count} locus/loci, sha256 {result.source_sha256}")
    if result.dataset:
        typer.echo(f"  release {result.dataset}")
    else:
        typer.secho(
            "  no --release was pinned, so this snapshot carries no release label and the check "
            "will not be able to say which version it compared against",
            fg=typer.colors.YELLOW,
            err=True,
        )

strchive_publish_

strchive_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (STRchive-loci.json + release.json).",
    ),
    repo: str = typer.Option(
        DEFAULT_STRCHIVE_REPO_ID,
        "--repo",
        help="Target HuggingFace dataset repo (owner/name).",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded; send nothing.",
    ),
    commit_message: str | None = typer.Option(
        None, "--message", "-m", help="Commit message."
    ),
) -> None

Create-or-update the dataset repo and upload the built STRchive catalogue (publisher/dev).

Publishable on the source's own terms: STRchive is MIT, which grants redistribution outright. This lane had no publish command because it grew from a check rather than from a cache, not because anything withheld the permission — the same distinction the roster draws between CIViC's absent ensure_* (a gap) and PharmVar's (a refusal).

Publish a pinned build. An unlabelled snapshot carries no dataset, so whoever pulls it can run the comparison and cannot say which release they compared against — build with --release first, and this refuses nothing but says so.

Source code in enricher/src/just_dna_enricher/cli.py
@strchive_app.command("publish")
def strchive_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (STRchive-loci.json + release.json).",
    ),
    repo: str = typer.Option(
        DEFAULT_STRCHIVE_REPO_ID,
        "--repo",
        help="Target HuggingFace dataset repo (owner/name).",
    ),
    dry_run: bool = typer.Option(False, "--dry-run", help="Show what would be uploaded; send nothing."),
    commit_message: str | None = typer.Option(None, "--message", "-m", help="Commit message."),
) -> None:
    """Create-or-update the dataset repo and upload the built STRchive catalogue (publisher/dev).

    Publishable on the source's own terms: STRchive is MIT, which grants redistribution outright.
    This lane had no publish command because it grew from a check rather than from a cache, not
    because anything withheld the permission — the same distinction the roster draws between CIViC's
    absent `ensure_*` (a gap) and PharmVar's (a refusal).

    **Publish a pinned build.** An unlabelled snapshot carries no `dataset`, so whoever pulls it can
    run the comparison and cannot say which release they compared against — build with `--release`
    first, and this refuses nothing but says so.
    """
    from just_dna_enricher.upload import plan_reference_snapshot, publish_reference_snapshot

    if not (read_release(snapshot_dir) or {}).get("dataset"):
        typer.secho(
            "  this snapshot carries no release label, so everyone who pulls it inherits a "
            "comparison that cannot name its own reference. Rebuild with `strchive build --release`.",
            fg=typer.colors.YELLOW,
            err=True,
        )
    try:
        if dry_run:
            plan = plan_reference_snapshot(snapshot_dir, repo, payload=STRCHIVE_CATALOGUE_FILENAME)
            typer.echo(f"would upload {len(plan.files)} file(s) to {plan.repo_id}: {plan.files}")
            return
        plan = publish_reference_snapshot(
            snapshot_dir,
            repo,
            commit_message=commit_message,
            payload=STRCHIVE_CATALOGUE_FILENAME,
        )
    except (FileNotFoundError, PermissionError, ImportError) as exc:
        typer.secho(f"PUBLISH FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(
        f"published: {snapshot_dir} → {plan.repo_id} ({len(plan.files)} files)",
        fg=typer.colors.GREEN,
    )

mitomap_build_

mitomap_build_(
    out: Path = typer.Option(
        repro_out("mitomap"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/mitomap-*.parquet + release.json).",
    ),
    dump: Path | None = typer.Option(
        None,
        "--dump",
        exists=True,
        dir_okay=False,
        help="A mitomap.dump.sql.gz you already have. Without it the dump is downloaded — the data surface answers plain curl, unlike the web surface. A local dump carries no Last-Modified, so that snapshot is honestly unlabelled.",
    ),
    url: str = typer.Option(
        DEFAULT_MITOMAP_URL,
        "--url",
        help="Source URL for the dump (used only when --dump is absent).",
    ),
) -> None

Cut the two curated mtDNA variant tables, their citations and the references out of the dump.

602 mmutation rows and 494 rtmutation rows out of 6.76 million lines. The snapshot records every count this build computes — rows per table, the dump's own per-table edit dates, how much of reference.nlmid is a PMID, the alleles that cannot be spelled as VCF and the brackets that are not a documented VCEP class — because a number computed and dropped is one every reader has to recompute.

Source code in enricher/src/just_dna_enricher/cli.py
@mitomap_app.command("build")
def mitomap_build_(
    out: Path = typer.Option(
        repro_out("mitomap"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/mitomap-*.parquet + release.json).",
    ),
    dump: Path | None = typer.Option(
        None,
        "--dump",
        exists=True,
        dir_okay=False,
        help=(
            "A mitomap.dump.sql.gz you already have. Without it the dump is downloaded — the data "
            "surface answers plain curl, unlike the web surface. A local dump carries no "
            "Last-Modified, so that snapshot is honestly unlabelled."
        ),
    ),
    url: str = typer.Option(
        DEFAULT_MITOMAP_URL,
        "--url",
        help="Source URL for the dump (used only when --dump is absent).",
    ),
) -> None:
    """Cut the two curated mtDNA variant tables, their citations and the references out of the dump.

    602 `mmutation` rows and 494 `rtmutation` rows out of 6.76 million lines. The snapshot records
    every count this build computes — rows per table, the dump's own per-table edit dates, how much of
    `reference.nlmid` is a PMID, the alleles that cannot be spelled as VCF and the brackets that are
    not a documented VCEP class — because a number computed and dropped is one every reader has to
    recompute.
    """
    from just_dna_enricher.mitomap_build import (
        build_snapshot,
        download_mitomap_dump,
    )

    try:
        if dump is not None:
            result = build_snapshot(dump, out, source_url=f"file://{dump.resolve()}")
        else:
            fetched = download_mitomap_dump(out / "mitomap.dump.sql.gz", url)
            result = build_snapshot(
                fetched.path,
                out,
                source_url=fetched.url,
                source_sha256=fetched.sha256,
                source_last_modified=fetched.last_modified,
            )
    except (MitomapError, ImportError, OSError) as exc:
        typer.secho(f"MITOMAP BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"built: {result.out_dir}", fg=typer.colors.GREEN)
    for name, count in result.rows.items():
        typer.echo(
            f"  {name} {count} rows, curated through {result.edit_dates.get(name) or 'an undated pass'}"
        )
    typer.echo(
        f"  {result.citation_links} citation links from {result.reference_rows} references "
        f"({result.references_without_nlmid} state no nlmid, "
        f"{result.references_not_a_pmid} state something that is not a PMID)"
    )
    if result.unmintable:
        typer.echo(
            "  alleles that cannot be spelled as VCF: "
            + ", ".join(f"{reason} {count}" for reason, count in result.unmintable.items())
        )
    if result.withheld_brackets:
        typer.secho(
            "  brackets withheld as undocumented (never mapped onto clin_sig): "
            + ", ".join(f"{token} {count}" for token, count in result.withheld_brackets.items()),
            fg=typer.colors.YELLOW,
        )
    if result.dataset:
        typer.echo(f"  release {result.dataset}")
    else:
        typer.secho(
            "  the dump states no edit_date for one of its variant tables, so this snapshot has no "
            "release label and the miss lane built from it cannot name the MITOMAP release it "
            "compared",
            fg=typer.colors.YELLOW,
            err=True,
        )

mitomap_miss_

mitomap_miss_(
    out: Path = typer.Option(
        repro_out("mitomap_miss"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/mitomap_miss.parquet + release.json).",
    ),
    mitomap_cache: Path | None = typer.Option(
        None,
        "--mitomap-cache",
        exists=True,
        file_okay=False,
        help="Built MITOMAP snapshot (see `mitomap build`). Omit it and $JUST_DNA_MITOMAP_CACHE is used.",
    ),
    clinvar_cache: Path | None = typer.Option(
        None,
        "--clinvar-cache",
        exists=True,
        file_okay=False,
        help="Built ClinVar snapshot (see `clinvar build`). Omit it and $JUST_DNA_CLINVAR_CACHE is used.",
    ),
) -> None

Join MITOMAP against the ClinVar chrMT parquet and write the increment (RM171).

A derived lane, not a download. Its acquire stage is both parents being on disk, and a parent that is absent is reported as could-not-run rather than as an empty increment — a miss set computed without ClinVar would say MITOMAP publishes a thousand alleles nobody else has, from a comparison that never ran.

Exact (start, ref, alt) on chrMT, upper-cased both sides, no position-level fallback. Four buckets, and draft-panel --source mitomap-miss writes only one of them.

Source code in enricher/src/just_dna_enricher/cli.py
@mitomap_app.command("miss")
def mitomap_miss_(
    out: Path = typer.Option(
        repro_out("mitomap_miss"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/mitomap_miss.parquet + release.json).",
    ),
    mitomap_cache: Path | None = typer.Option(
        None,
        "--mitomap-cache",
        exists=True,
        file_okay=False,
        help="Built MITOMAP snapshot (see `mitomap build`). Omit it and $JUST_DNA_MITOMAP_CACHE is used.",
    ),
    clinvar_cache: Path | None = typer.Option(
        None,
        "--clinvar-cache",
        exists=True,
        file_okay=False,
        help="Built ClinVar snapshot (see `clinvar build`). Omit it and $JUST_DNA_CLINVAR_CACHE is used.",
    ),
) -> None:
    """Join MITOMAP against the ClinVar chrMT parquet and write the increment (RM171).

    **A derived lane, not a download.** Its acquire stage is both parents being on disk, and a parent
    that is absent is reported as could-not-run rather than as an empty increment — a miss set
    computed without ClinVar would say MITOMAP publishes a thousand alleles nobody else has, from a
    comparison that never ran.

    Exact `(start, ref, alt)` on chrMT, upper-cased both sides, no position-level fallback. Four
    buckets, and `draft-panel --source mitomap-miss` writes only one of them.
    """
    from just_dna_enricher.mitomap_miss_build import build_miss_snapshot

    parents = {}
    missing = []
    for name, explicit in (("mitomap", mitomap_cache), ("clinvar", clinvar_cache)):
        found = explicit or LANES_BY_NAME[name].resolve()
        if found is None:
            missing.append(name)
        else:
            parents[name] = found
    if missing:
        typer.secho(
            f"MITOMAP MISS NOT RUN: derived from mitomap and clinvar, and "
            f"{', '.join(missing)} {'is' if len(missing) == 1 else 'are'} not on disk. Provision "
            f"with `cache prepare --only {' --only '.join(missing)}`, or name one with "
            f"--mitomap-cache/--clinvar-cache. An increment computed without a parent is not an "
            f"empty increment.",
            fg=typer.colors.YELLOW,
            err=True,
        )
        raise typer.Exit(code=2)
    try:
        result = build_miss_snapshot(parents["mitomap"], parents["clinvar"], out)
    except (MitomapError, ImportError, OSError) as exc:
        typer.secho(f"MITOMAP MISS FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(f"built: {result.parquet_file}", fg=typer.colors.GREEN)
    for table, counts in result.buckets_by_table.items():
        typer.echo(f"  {table}: " + ", ".join(f"{k} {v}" for k, v in counts.items()))
    typer.echo(
        f"  against {result.clinvar_keys} distinct ClinVar chrMT alleles "
        f"({result.parents['clinvar'].get('clinvar_file_date') or 'an undated snapshot'})"
    )
    if result.rated_miss_by_class:
        typer.echo(
            "  rated misses by class: " + ", ".join(f"{k} {v}" for k, v in result.rated_miss_by_class.items())
        )
    if result.rated_miss_indels:
        typer.secho(
            f"  {result.rated_miss_indels} of {result.rated_misses} rated miss(es) key on an indel. "
            f"The join is exact and neither side is left-aligned here, so one of those is an "
            f"absence or a difference of anchor and this lane cannot tell you which.",
            fg=typer.colors.YELLOW,
            err=True,
        )
    if result.withheld_in_miss:
        typer.secho(
            "  missing rows whose only rating is an undocumented bracket, counted and never mapped: "
            + ", ".join(f"{k} {v}" for k, v in result.withheld_in_miss.items()),
            fg=typer.colors.YELLOW,
        )
    if result.unmintable:
        typer.echo(
            "  rows the join has no key for (prose in an allele column, no event, or a `:` deletion "
            "whose bases rCRS does not have): " + ", ".join(f"{k} {v}" for k, v in result.unmintable.items())
        )

mitomap_publish_

mitomap_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/ + release.json).",
    ),
    repo: str = typer.Option(
        DEFAULT_MITOMAP_REPO_ID,
        "--repo",
        help="Target HuggingFace dataset repo (owner/name).",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded; send nothing.",
    ),
    commit_message: str | None = typer.Option(
        None, "--message", "-m", help="Commit message."
    ),
) -> None

Create-or-update the dataset repo and upload the built MITOMAP snapshot (publisher/dev).

Publishable on the source's own terms: CC BY 3.0, with commercial and clinical use stated free and attribution the one condition — which the snapshot's SourceRow carries. Only the parent lane publishes. The derived miss snapshot pins two parent digests, so a pulled copy would be an increment whose own currency check cannot be run by whoever pulled it; it is rebuilt locally from the parents instead, which is cheaper than the download and cannot be stale.

Source code in enricher/src/just_dna_enricher/cli.py
@mitomap_app.command("publish")
def mitomap_publish_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/ + release.json).",
    ),
    repo: str = typer.Option(
        DEFAULT_MITOMAP_REPO_ID,
        "--repo",
        help="Target HuggingFace dataset repo (owner/name).",
    ),
    dry_run: bool = typer.Option(False, "--dry-run", help="Show what would be uploaded; send nothing."),
    commit_message: str | None = typer.Option(None, "--message", "-m", help="Commit message."),
) -> None:
    """Create-or-update the dataset repo and upload the built MITOMAP snapshot (publisher/dev).

    Publishable on the source's own terms: CC BY 3.0, with commercial and clinical use stated free and
    attribution the one condition — which the snapshot's `SourceRow` carries. **Only the parent lane
    publishes.** The derived miss snapshot pins two parent digests, so a pulled copy would be an
    increment whose own currency check cannot be run by whoever pulled it; it is rebuilt locally from
    the parents instead, which is cheaper than the download and cannot be stale.
    """
    from just_dna_enricher.upload import plan_reference_snapshot, publish_reference_snapshot

    if not (read_release(snapshot_dir) or {}).get("dataset"):
        typer.secho(
            "  this snapshot carries no release label (it was built from a local dump), so everyone "
            "who pulls it inherits a comparison that cannot name its own MITOMAP release.",
            fg=typer.colors.YELLOW,
            err=True,
        )
    try:
        if dry_run:
            plan = plan_reference_snapshot(snapshot_dir, repo)
            typer.echo(f"would upload {len(plan.files)} file(s) to {plan.repo_id}: {plan.files}")
            return
        plan = publish_reference_snapshot(snapshot_dir, repo, commit_message=commit_message)
    except (FileNotFoundError, PermissionError, ImportError) as exc:
        typer.secho(f"PUBLISH FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(
        f"published: {snapshot_dir} → {plan.repo_id} ({len(plan.files)} files)",
        fg=typer.colors.GREEN,
    )

draft_repeats_

draft_repeats_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    genes: list[str] = typer.Option(
        [],
        "--gene",
        "-g",
        help="Restrict to these genes. Repeatable; omit for every catalogue locus.",
    ),
    catalogue: Path | None = typer.Option(
        None,
        "--catalogue",
        exists=True,
        help="Built STRchive snapshot directory (see `strchive build`), or a STRchive-loci.json. Omit it and $JUST_DNA_STRCHIVE_CACHE (or the shared cache base) is used.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=f"Declared use: one of {sorted(VALID_DECLARED_USE)}.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Report what would be added; write nothing.",
    ),
) -> None

Draft repeat_alleles.csv identity rows from STRchive — appends, never overwrites a row.

The bands are not drafted, and that is the design rather than a limitation. A drafted row carries the gene, the motif as the catalogue spells it, the trait CURIE where the locus names exactly one disease, and a conclusion placeholder — so the table cannot compile until a human has filled in what each band means. measure_min/measure_max stay empty: run check-repeat-bands once you have written them and it will report where the catalogue disagrees.

The catalogue's coordinates, ref_copies and locus_structure have no authored column to land in; the run counts them and says so rather than dropping them silently.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("draft-repeats")
def draft_repeats_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    genes: list[str] = typer.Option(
        [],
        "--gene",
        "-g",
        help="Restrict to these genes. Repeatable; omit for every catalogue locus.",
    ),
    catalogue: Path | None = typer.Option(
        None,
        "--catalogue",
        exists=True,
        help=(
            "Built STRchive snapshot directory (see `strchive build`), or a STRchive-loci.json. "
            "Omit it and $JUST_DNA_STRCHIVE_CACHE (or the shared cache base) is used."
        ),
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=f"Declared use: one of {sorted(VALID_DECLARED_USE)}.",
    ),
    dry_run: bool = typer.Option(False, "--dry-run", help="Report what would be added; write nothing."),
) -> None:
    """Draft repeat_alleles.csv identity rows from STRchive — appends, never overwrites a row.

    **The bands are not drafted, and that is the design rather than a limitation.** A drafted row
    carries the gene, the motif as the catalogue spells it, the trait CURIE where the locus names
    exactly one disease, and a `conclusion` placeholder — so the table cannot compile until a human
    has filled in what each band means. `measure_min`/`measure_max` stay empty: run
    `check-repeat-bands` once you have written them and it will report where the catalogue disagrees.

    The catalogue's coordinates, `ref_copies` and `locus_structure` have no authored column to land
    in; the run counts them and says so rather than dropping them silently.
    """
    try:
        result = draft_repeat_loci(
            spec_dir,
            genes,
            catalogue=catalogue,
            declared_use=_use(use),
            dry_run=dry_run,
        )
    except (StrchiveError, StrchiveDraftError, DraftError) as exc:
        typer.secho(f"DRAFT FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc

    for warning in result.warnings:
        typer.secho(f"  {warning}", fg=typer.colors.YELLOW, err=True)
    if result.skipped:
        raise typer.Exit(code=1)
    label = f" ({result.dataset})" if result.dataset else ""
    verb = "would add" if dry_run else "added"
    typer.secho(
        f"{verb} {result.drafted} row(s) from {result.candidates} strchive locus/loci{label}",
        fg=typer.colors.GREEN,
    )
    if result.drafted:
        typer.echo(
            "  every drafted row needs its bands and its conclusion written; the placeholder is "
            "what stops the module compiling until they are"
        )

check_repeat_bands_

check_repeat_bands_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    catalogue: Path | None = typer.Option(
        None,
        "--catalogue",
        exists=True,
        help="Built STRchive snapshot directory (see `strchive build`), or a STRchive-loci.json.",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Carried into the report. A band difference NEVER fails, in either mode.",
    ),
) -> None

Compare a module's repeat_alleles.csv bands against STRchive's, and report the differences.

Writes no authored cell and never fails on a difference. Where a catalogue and an expert author draw a repeat threshold in different places, both are claims by an authority, and a compile that refused would make this format pick the winner — the rule the ClinVar clin_sig and PGx allele-function checks already follow. --strict is accepted so the flag means one thing across the tier, and it changes nothing here but the mode recorded in the report.

The catalogue's pathogenic_max is reported as its own finding and is never written: it is the longest allele the literature records, not a clinical ceiling, and a module that imported it would silently answer nothing at all for a longer one.

Source code in enricher/src/just_dna_enricher/cli.py
@app.command("check-repeat-bands")
def check_repeat_bands_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    catalogue: Path | None = typer.Option(
        None,
        "--catalogue",
        exists=True,
        help="Built STRchive snapshot directory (see `strchive build`), or a STRchive-loci.json.",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Carried into the report. A band difference NEVER fails, in either mode.",
    ),
) -> None:
    """Compare a module's `repeat_alleles.csv` bands against STRchive's, and report the differences.

    **Writes no authored cell and never fails on a difference.** Where a catalogue and an expert
    author draw a repeat threshold in different places, both are claims by an authority, and a compile
    that refused would make this format pick the winner — the rule the ClinVar `clin_sig` and PGx
    allele-function checks already follow. `--strict` is accepted so the flag means one thing across
    the tier, and it changes nothing here but the mode recorded in the report.

    The catalogue's `pathogenic_max` is reported as its own finding and is never written: it is the
    longest allele the literature records, not a clinical ceiling, and a module that imported it would
    silently answer nothing at all for a longer one.
    """
    try:
        result = check_repeat_bands(spec_dir, catalogue=catalogue, mode=_mode(strict))
    except StrchiveError as exc:
        typer.secho(f"REPEAT-BAND CHECK FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc

    for warning in result.warnings:
        typer.secho(f"  {warning}", fg=typer.colors.YELLOW, err=True)
    if not result.compared and not result.withheld:
        # Three different ways to get here and only one of them is "no table": the warnings above
        # already name the other two (no catalogue was provisioned; the table is only `unresolved`
        # sentinels), so saying "no repeat_alleles.csv" unconditionally would print a false diagnosis
        # over a true one.
        if not result.warnings:
            typer.secho("no repeat_alleles.csv — nothing to check", fg=typer.colors.YELLOW)
        raise typer.Exit(code=0)
    label = f" ({result.dataset})" if result.dataset else ""
    typer.echo(f"compared {len(result.compared)} bin group(s) against strchive{label}")
    for _key, reason in result.withheld:
        typer.secho(f"  withheld: {reason}", fg=typer.colors.CYAN)
    for finding in result.findings:
        typer.secho(f"  {finding}", fg=typer.colors.YELLOW, err=True)
    if result.compared and not result.findings:
        typer.secho("every compared band matches the catalogue", fg=typer.colors.GREEN)

clinpgx_build_labels_

clinpgx_build_labels_(
    out_dir: Path = typer.Option(
        repro_out("drug_labels"),
        "--out",
        file_okay=False,
        help="Snapshot output directory.",
    ),
    zip_path: Path | None = typer.Option(
        None,
        "--zip",
        exists=True,
        dir_okay=False,
        help="A drugLabels.zip you already have. Without it the archive is downloaded.",
    ),
    url: str = typer.Option(
        DEFAULT_DRUG_LABELS_URL,
        "--url",
        help="ClinPGx bulk download URL.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=f"Declared use: one of {sorted(VALID_DECLARED_USE)}.",
    ),
) -> None

Download + build the regulator drug-label snapshot (dev surface; needs polars).

A second archive from a source this tier already adopted, with its own release.json: ClinPGx publishes at least twelve downloads on this endpoint and they do not refresh in lockstep, so the label snapshot is dated from its own CREATED_*.txt rather than from the annotation lane's.

There is no --offline: a builder's off-switch is passing --zip instead of downloading.

Source code in enricher/src/just_dna_enricher/cli.py
@clinpgx_app.command("build-labels")
def clinpgx_build_labels_(
    out_dir: Path = typer.Option(
        repro_out("drug_labels"), "--out", file_okay=False, help="Snapshot output directory."
    ),
    zip_path: Path | None = typer.Option(
        None,
        "--zip",
        exists=True,
        dir_okay=False,
        help="A drugLabels.zip you already have. Without it the archive is downloaded.",
    ),
    url: str = typer.Option(DEFAULT_DRUG_LABELS_URL, "--url", help="ClinPGx bulk download URL."),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=f"Declared use: one of {sorted(VALID_DECLARED_USE)}.",
    ),
) -> None:
    """Download + build the regulator drug-label snapshot (dev surface; needs polars).

    A **second** archive from a source this tier already adopted, with its own `release.json`: ClinPGx
    publishes at least twelve downloads on this endpoint and they do not refresh in lockstep, so the
    label snapshot is dated from its own `CREATED_*.txt` rather than from the annotation lane's.

    There is no `--offline`: a builder's off-switch is passing `--zip` instead of downloading.
    """
    from just_dna_enricher.drug_labels_build import (
        build_drug_label_snapshot,
        download_drug_labels_zip,
    )

    declared = _use(use)
    try:
        # The terms are accepted when the data is TAKEN, so the gate runs before the download.
        reason = check_declared_use(CLINPGX_TERMS, declared)
    except LicenseRefusal as exc:
        typer.secho(f"REFUSED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if reason is not None:
        typer.secho(f"SKIPPED: {reason}", fg=typer.colors.YELLOW, err=True)
        raise typer.Exit(code=1)

    source_sha: str | None = None
    try:
        if zip_path is None:
            zip_path, source_sha = download_drug_labels_zip(Path(out_dir) / "drugLabels.zip", url)
        result = build_drug_label_snapshot(zip_path, out_dir, source_url=url, source_sha256=source_sha)
    except (DrugLabelError, OSError) as exc:
        typer.secho(f"DRUG-LABEL BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc

    typer.secho(f"drug-label snapshot: {result.parquet_path}", fg=typer.colors.GREEN)
    typer.echo(
        f"labels: {result.label_count}  regulators: {', '.join(result.regulators)}  "
        f"release: {result.created_date or 'undated'}"
    )
    typer.echo(f"testing levels stated: {', '.join(result.testing_levels)}")
    typer.echo(f"licence pinned: {result.license_sha256}")
    if result.dataset is None:
        typer.secho(
            "  the archive carried no CREATED_<date>.txt, so this snapshot has no release label and "
            "the check will not be able to say which version it compared against",
            fg=typer.colors.YELLOW,
            err=True,
        )

clinpgx_check_labels_

clinpgx_check_labels_(
    spec_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    snapshot: Path | None = typer.Option(
        None,
        "--snapshot",
        exists=True,
        file_okay=False,
        help="Built drug-label snapshot directory (see `clinpgx build-labels`). Omit it and $JUST_DNA_DRUG_LABELS_CACHE (or the shared cache base) is used.",
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Carried into the report. A label difference NEVER fails, in either mode.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=f"Declared use: one of {sorted(VALID_DECLARED_USE)}.",
    ),
) -> None

Compare a module's gene/allele/drug claims against the drug labels five regulators publish.

Two join tiers, reported apart, and the tier belongs to the question. What do the agencies say about this gene and this medicine is the gene-tier subject; …and this star allele or rsID is the allele-tier one. A label naming both answers both, because they are two questions rather than one asked twice, and a gene-level agreement is not an allele-level agreement.

Writes no authored cell and never fails on a difference. Five agencies genuinely disagree with each other — clopidogrel and CYP2C19 is Actionable PGx at four of them and Informative PGx at the EMA — and a compile that refused would make this format pick the winner. A blank Testing Level is a third of the file and is reported as unknown, never as No Clinical PGx.

Source code in enricher/src/just_dna_enricher/cli.py
@clinpgx_app.command("check-labels")
def clinpgx_check_labels_(
    spec_dir: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    snapshot: Path | None = typer.Option(
        None,
        "--snapshot",
        exists=True,
        file_okay=False,
        help=(
            "Built drug-label snapshot directory (see `clinpgx build-labels`). Omit it and "
            "$JUST_DNA_DRUG_LABELS_CACHE (or the shared cache base) is used."
        ),
    ),
    strict: bool = typer.Option(
        False,
        "--strict/--best-effort",
        help="Carried into the report. A label difference NEVER fails, in either mode.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=f"Declared use: one of {sorted(VALID_DECLARED_USE)}.",
    ),
) -> None:
    """Compare a module's gene/allele/drug claims against the drug labels five regulators publish.

    **Two join tiers, reported apart, and the tier belongs to the question.** *What do the agencies
    say about this gene and this medicine* is the gene-tier subject; *…and this star allele or rsID*
    is the allele-tier one. A label naming both answers both, because they are two questions rather
    than one asked twice, and a gene-level agreement is not an allele-level agreement.

    **Writes no authored cell and never fails on a difference.** Five agencies genuinely disagree with
    each other — clopidogrel and CYP2C19 is `Actionable PGx` at four of them and `Informative PGx` at
    the EMA — and a compile that refused would make this format pick the winner. A blank `Testing
    Level` is a third of the file and is reported as unknown, never as `No Clinical PGx`.
    """
    try:
        result = check_drug_labels(
            spec_dir,
            snapshot=snapshot,
            mode=_mode(strict),
            declared_use=_use(use),
        )
    except LicenseRefusal as exc:
        typer.secho(f"REFUSED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    except DrugLabelError as exc:
        typer.secho(f"DRUG-LABEL CHECK FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc

    for warning in result.warnings:
        typer.secho(f"  {warning}", fg=typer.colors.YELLOW, err=True)
    if not result.compared and not result.withheld:
        raise typer.Exit(code=0)
    label = f" ({result.dataset})" if result.dataset else ""
    tiers = ", ".join(f"{count} at the {tier} tier" for tier, count in result.tier_subjects.items())
    typer.echo(
        f"compared {len(result.compared)} claim(s) against {len(result.regulators)} regulator(s)"
        f"{label} — {tiers}"
    )
    if result.unstated_labels:
        typer.secho(
            f"  {len(result.unstated_labels)} label(s) state no testing level: counted as unknown, never "
            f"as a negative",
            fg=typer.colors.CYAN,
        )
    if result.verdicts:
        typer.secho(f"  {arm_summary(result)}", fg=typer.colors.CYAN)
    for _subject, note in result.withheld:
        typer.secho(f"  withheld: {note}", fg=typer.colors.CYAN)
    for finding in result.findings:
        typer.secho(f"  {finding}", fg=typer.colors.YELLOW, err=True)
    if result.compared and not result.findings:
        typer.secho("every compared claim agrees with the labels that reached it", fg=typer.colors.GREEN)

clinpgx_publish_labels_

clinpgx_publish_labels_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/drug_labels.parquet + LICENSE.txt + release.json).",
    ),
    repo: str = typer.Option(
        DEFAULT_DRUG_LABELS_REPO_ID,
        "--repo",
        help="Target HuggingFace dataset repo (owner/name).",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded; send nothing.",
    ),
    commit_message: str | None = typer.Option(
        None, "--message", "-m", help="Commit message."
    ),
) -> None

Publish a built drug-label snapshot so clinpgx check-labels can provision it (publisher/dev).

A second repo rather than a second table in just-dna-seq/clinpgx, for the reason the builder already gives its own release.json: the two ClinPGx archives do not refresh in lockstep, and one repo holding both would date the pair from whichever was published last.

Same grounds as clinpgx publish — CC BY-SA permits redistribution, forbids sale, and requires attribution, which sources.csv carries. LICENSE.txt travels with the parquet: a share-alike snapshot whose terms did not travel pins nothing for whoever downloads it.

Source code in enricher/src/just_dna_enricher/cli.py
@clinpgx_app.command("publish-labels")
def clinpgx_publish_labels_(
    snapshot_dir: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Built snapshot directory (data/drug_labels.parquet + LICENSE.txt + release.json).",
    ),
    repo: str = typer.Option(
        DEFAULT_DRUG_LABELS_REPO_ID,
        "--repo",
        help="Target HuggingFace dataset repo (owner/name).",
    ),
    dry_run: bool = typer.Option(False, "--dry-run", help="Show what would be uploaded; send nothing."),
    commit_message: str | None = typer.Option(None, "--message", "-m", help="Commit message."),
) -> None:
    """Publish a built drug-label snapshot so `clinpgx check-labels` can provision it (publisher/dev).

    A **second** repo rather than a second table in `just-dna-seq/clinpgx`, for the reason the builder
    already gives its own `release.json`: the two ClinPGx archives do not refresh in lockstep, and one
    repo holding both would date the pair from whichever was published last.

    Same grounds as `clinpgx publish` — CC BY-SA permits redistribution, forbids sale, and requires
    attribution, which `sources.csv` carries. `LICENSE.txt` travels with the parquet: a share-alike
    snapshot whose terms did not travel pins nothing for whoever downloads it.
    """
    from just_dna_enricher.upload import plan_reference_snapshot, publish_reference_snapshot

    try:
        if dry_run:
            plan = plan_reference_snapshot(snapshot_dir, repo)
            typer.echo(f"would upload {len(plan.files)} file(s) to {plan.repo_id}: {plan.files}")
            return
        plan = publish_reference_snapshot(snapshot_dir, repo, commit_message=commit_message)
    except (FileNotFoundError, PermissionError, ImportError) as exc:
        typer.secho(f"PUBLISH FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if SNAPSHOT_LICENSE_FILENAME not in plan.files:
        typer.secho(
            f"  no {SNAPSHOT_LICENSE_FILENAME} in this snapshot, so `license_sha256` pins nothing "
            f"for whoever pulls it. Rebuild with `clinpgx build-labels`.",
            fg=typer.colors.YELLOW,
            err=True,
        )
    typer.secho(
        f"published: {snapshot_dir} → {plan.repo_id} ({len(plan.files)} files)",
        fg=typer.colors.GREEN,
    )

atlas_generate_

atlas_generate_(
    refetch: bool = typer.Option(
        False,
        "--refetch",
        help="Re-download the pinned sources even if they are already on disk and match.",
    ),
) -> None

Fetch the pinned Atlas .proto sources and generate the gRPC bindings from them.

The sources are not vendored: the repository carries a commit id and a sha256 per file, and a file that does not match its pin is refused rather than used (RM196). Needs grpcio-tools, which is in \[dev] and deliberately not in \[atlas] — the runtime imports the bindings without it. A released wheel carries both the sources and the bindings already, so this is a checkout command.

Source code in enricher/src/just_dna_enricher/cli.py
@atlas_app.command("generate")
def atlas_generate_(
    refetch: bool = typer.Option(
        False,
        "--refetch",
        help="Re-download the pinned sources even if they are already on disk and match.",
    ),
) -> None:
    """Fetch the pinned Atlas `.proto` sources and generate the gRPC bindings from them.

    The sources are not vendored: the repository carries a commit id and a sha256 per file, and a
    file that does not match its pin is refused rather than used (RM196). Needs `grpcio-tools`,
    which is in `\\[dev]` and deliberately not in `\\[atlas]` — the runtime imports the bindings without
    it. A released wheel carries both the sources and the bindings already, so this is a checkout
    command.
    """
    from just_dna_enricher import atlas_protos

    try:
        if refetch:
            atlas_protos.fetch_protos(force=True)
        out = atlas_protos.generate()
    except atlas_protos.ProtoFetchError as exc:
        typer.secho(f"GENERATE FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    except ImportError as exc:
        typer.secho(
            f"GENERATE FAILED: grpcio-tools is not installed ({exc}). It is build-time only and "
            "lives in the [dev] group: `uv sync` from a checkout, or `pip install grpcio-tools`.",
            fg=typer.colors.RED,
            err=True,
        )
        raise typer.Exit(code=1) from exc
    typer.secho(f"bindings written to {out}", fg=typer.colors.GREEN)

alphagenome_avi_build_

alphagenome_avi_build_(
    input_: Path = typer.Option(
        ...,
        "--input",
        exists=True,
        dir_okay=False,
        help="The extracted alphagenome_variant_impact_score_snvs.tsv.gz (its .tbi must be beside it). Required, and there is no default URL: acquisition is yours, under your own acceptance of the AlphaGenome Services Additional Terms.",
    ),
    out: Path = typer.Option(
        repro_out("alphagenome_avi"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/alphagenome_avi-*.parquet, avi_knots.parquet, release.json, LICENSE.txt).",
    ),
    contig: list[str] = typer.Option(
        None,
        "--contig",
        help="Build only these contigs, repeatable. Omit for every contig the .tbi index knows.",
    ),
    workers: int = typer.Option(
        12,
        "--workers",
        min=1,
        help="How many contigs to read at once. Twelve ran 24 contigs in 41-46 minutes; one takes about four times as long.",
    ),
    no_hash: bool = typer.Option(
        False,
        "--no-hash",
        help="Skip the source sha256. It is a few minutes over 88.5 GB; release.json then records null, which is unknown rather than unpinned.",
    ),
) -> None

Build the AVI snapshot from a local copy of the artifact.

raw_score is stored as Int32 at a scale of 105 — exactly lossless, since the artifact prints at most five decimals — and PHRED is not** stored: it is a rank, a function of raw_score, and the 466 KB knot table beside the data reconstructs it while also carrying the per-value ambiguity interval a threshold has to be checked against.

Source code in enricher/src/just_dna_enricher/cli.py
@alphagenome_app.command("build")
def alphagenome_avi_build_(
    input_: Path = typer.Option(
        ...,
        "--input",
        exists=True,
        dir_okay=False,
        help=(
            "The extracted alphagenome_variant_impact_score_snvs.tsv.gz (its .tbi must be beside "
            "it). Required, and there is no default URL: acquisition is yours, under your own "
            "acceptance of the AlphaGenome Services Additional Terms."
        ),
    ),
    out: Path = typer.Option(
        repro_out("alphagenome_avi"),
        "--out",
        file_okay=False,
        help="Output snapshot directory (writes data/alphagenome_avi-*.parquet, avi_knots.parquet, release.json, LICENSE.txt).",
    ),
    contig: list[str] = typer.Option(
        None,
        "--contig",
        help="Build only these contigs, repeatable. Omit for every contig the .tbi index knows.",
    ),
    workers: int = typer.Option(
        12,
        "--workers",
        min=1,
        help="How many contigs to read at once. Twelve ran 24 contigs in 41-46 minutes; one takes about four times as long.",
    ),
    no_hash: bool = typer.Option(
        False,
        "--no-hash",
        help="Skip the source sha256. It is a few minutes over 88.5 GB; release.json then records null, which is unknown rather than unpinned.",
    ),
) -> None:
    """Build the AVI snapshot from a local copy of the artifact.

    `raw_score` is stored as `Int32` at a scale of 10**5 — exactly lossless, since the artifact
    prints at most five decimals — and `PHRED` is **not** stored: it is a rank, a function of
    `raw_score`, and the 466 KB knot table beside the data reconstructs it while also carrying the
    per-value ambiguity interval a threshold has to be checked against.
    """
    try:
        result = build_alphagenome_snapshot(
            input_,
            out,
            contigs=list(contig) if contig else None,
            workers=workers,
            hash_source=not no_hash,
        )
    except AlphaGenomeBuildError as exc:
        typer.secho(f"BUILD FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    typer.secho(
        f"built {result.rows:,} rows over {len(result.contigs)} contig(s) into {out}",
        fg=typer.colors.GREEN,
    )
    typer.echo(
        f"  {result.knots:,} knots, {result.negative_rows:,} negative scores, "
        f"{result.zero_rows:,} genuine zeros (absence is row-absence, never a zero)"
    )
    if result.dataset is None:
        typer.secho(
            "  the artifact's own timestamp could not be read, so release.json records no dataset. "
            "The Output Terms pin the applicable version to the date the Output was generated, so "
            "that date is worth recovering before the snapshot is relied on.",
            fg=typer.colors.YELLOW,
            err=True,
        )
    typer.echo(
        "  AVI is Permissive Use — commercial and non-commercial (RM195, and the download page that "
        "says so is pinned in docs/vendor/). Sharing is the narrower question: redistribution rests "
        "on reading an open publication as prohibition 1's 'open source release', which is recorded "
        "as a reading rather than quoted from a clause."
    )

alphagenome_check_

alphagenome_check_(
    spec: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    reference: Path | None = typer.Option(
        None,
        "--reference",
        exists=True,
        file_okay=False,
        help="An AVI snapshot directory. Omit to use $JUST_DNA_ALPHAGENOME_AVI_CACHE.",
    ),
    threshold: float | None = typer.Option(
        None,
        "--threshold",
        help="A PHRED cut to check the module's variants against. Without one the pass is entirely offline: there is no question the local artifact cannot answer.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Never reach the Atlas. Straddling variants are recorded as nobody-asked, not as decided.",
    ),
    refinement_cap: int = typer.Option(
        DEFAULT_REFINEMENT_CAP,
        "--refinement-cap",
        min=1,
        help="Refuse rather than refine more than this many variants over the network in one run.",
    ),
    strict: bool = typer.Option(
        False,
        "--strict",
        help="Carried for the report; see the docstring.",
    ),
) -> None

Cross-check a module's variants against AlphaGenome's AVI scores. Reports, never repairs.

The local snapshot answers most of it. The Atlas is asked only where the knot table says the local data genuinely cannot decide — a threshold falling inside a printed score's PHRED interval — and that set is computed offline, before any request is spent.

Source code in enricher/src/just_dna_enricher/cli.py
@alphagenome_app.command("check")
def alphagenome_check_(
    spec: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    reference: Path | None = typer.Option(
        None,
        "--reference",
        exists=True,
        file_okay=False,
        help="An AVI snapshot directory. Omit to use $JUST_DNA_ALPHAGENOME_AVI_CACHE.",
    ),
    threshold: float | None = typer.Option(
        None,
        "--threshold",
        help=(
            "A PHRED cut to check the module's variants against. Without one the pass is entirely "
            "offline: there is no question the local artifact cannot answer."
        ),
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="Never reach the Atlas. Straddling variants are recorded as nobody-asked, not as decided.",
    ),
    refinement_cap: int = typer.Option(
        DEFAULT_REFINEMENT_CAP,
        "--refinement-cap",
        min=1,
        help="Refuse rather than refine more than this many variants over the network in one run.",
    ),
    strict: bool = typer.Option(False, "--strict", help="Carried for the report; see the docstring."),
) -> None:
    """Cross-check a module's variants against AlphaGenome's AVI scores. Reports, never repairs.

    The local snapshot answers most of it. The Atlas is asked only where the knot table says the
    local data genuinely cannot decide — a threshold falling inside a printed score's PHRED interval
    — and that set is computed offline, before any request is spent.
    """
    try:
        client = None
        if threshold is not None and not offline:
            client = _atlas_client_or_none()
        result = check_variant_impact(
            spec,
            reference=reference,
            client=client,
            threshold=threshold,
            mode="strict" if strict else "best_effort",
            offline=offline,
            refinement_cap=refinement_cap,
        )
    except VariantImpactError as exc:
        typer.secho(f"CHECK FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc

    for note in result.warnings:
        typer.secho(f"  {note}", fg=typer.colors.YELLOW, err=True)
    for finding in result.findings:
        typer.secho(f"  {finding}", fg=typer.colors.YELLOW)
    if result.straddling:
        typer.echo(
            f"  {len(result.straddling)} variant(s) sit inside a knot spanning PHRED "
            f"{threshold}; {len(result.refined)} refined against the Atlas"
        )
    by_reason: dict[str, int] = {}
    for _, reason in result.unanswered:
        by_reason[reason] = by_reason.get(reason, 0) + 1
    if by_reason:
        # Grouped by reason rather than listed per row (`@ref-mismatch-causes`): four histories with
        # four remedies, and a flat list of variants hides which one a reader is looking at.
        typer.echo("  no answer: " + ", ".join(f"{n} {reason}" for reason, n in sorted(by_reason.items())))
    typer.secho(
        f"checked {result.subjects} variant(s): {len(result.decided)} decided locally, "
        f"{len(result.findings)} finding(s)",
        fg=typer.colors.GREEN,
    )

alphagenome_expression_

alphagenome_expression_(
    spec: Path = typer.Argument(
        ...,
        exists=True,
        file_okay=False,
        help="Module spec directory",
    ),
    gene: str = typer.Option(
        ...,
        "--gene",
        help="HGNC symbol. REQUIRED even with an explicit interval: the server-side gene filter is not an optimisation, and an unfiltered interval query is refused before it is sent.",
    ),
    chrom: str | None = typer.Option(
        None,
        "--chrom",
        help="Contig of an explicit interval. Wins over the gene's MANE span.",
    ),
    start: int | None = typer.Option(
        None,
        "--start",
        min=0,
        help="1-based start of that interval.",
    ),
    end: int | None = typer.Option(
        None,
        "--end",
        min=0,
        help="1-based end of that interval.",
    ),
    min_score: float | None = typer.Option(
        None,
        "--min-score",
        help="Keep only pairs whose magnitude reaches this. Distal scores run ~10x lower than scores at the gene, so a flat bar keeps the proximal rows and looks like it filtered on effect.",
    ),
    max_rows: int = typer.Option(
        DEFAULT_MAX_ROWS,
        "--max-rows",
        min=1,
        help="Refuse rather than write more rows than this. Raising it is a deliberate act.",
    ),
    offline: bool = typer.Option(
        False,
        "--offline",
        help="No-op with a warning: this pass reads the Atlas, not a snapshot.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Report what would be written without writing it.",
    ),
    use: str = typer.Option(
        "unstated",
        "--use",
        help="Declared use recorded on the licence row: unstated|non-commercial|commercial. AlphaGenome Output is NON-COMMERCIAL ONLY, so an undeclared run writes nothing and says so — pass --use non-commercial.",
    ),
) -> None

Fill expression_effects.csv with AlphaGenome's per-gene expression effects for one gene.

Example — and the --use is not decoration, an undeclared run is a no-op:

just-dna-enricher alphagenome expression ./my_module --gene TBX1 --use non-commercial

One row per (variant, gene): which way the variant moves that gene's predicted expression, how many of the 371 tissue tracks agree, and how far it sits from the gene. The interval is the gene's MANE span widened by the model's measured +/-512 kb attribution horizon, unless --chrom/--start/--end supply one; either way the gene names the server-side filter, and the MANE lane is still consulted for the distance, which an explicit interval cannot supply.

A whole gene is ~3.3 M SNVs at the measured 1,091 SNVs/s — about 50 minutes — and the cost is printed before the query runs rather than discovered during it.

Source code in enricher/src/just_dna_enricher/cli.py
@alphagenome_app.command("expression")
def alphagenome_expression_(
    spec: Path = typer.Argument(..., exists=True, file_okay=False, help="Module spec directory"),
    gene: str = typer.Option(
        ...,
        "--gene",
        help=(
            "HGNC symbol. REQUIRED even with an explicit interval: the server-side gene filter is "
            "not an optimisation, and an unfiltered interval query is refused before it is sent."
        ),
    ),
    chrom: str | None = typer.Option(
        None, "--chrom", help="Contig of an explicit interval. Wins over the gene's MANE span."
    ),
    start: int | None = typer.Option(None, "--start", min=0, help="1-based start of that interval."),
    end: int | None = typer.Option(None, "--end", min=0, help="1-based end of that interval."),
    min_score: float | None = typer.Option(
        None,
        "--min-score",
        help=(
            "Keep only pairs whose magnitude reaches this. Distal scores run ~10x lower than scores "
            "at the gene, so a flat bar keeps the proximal rows and looks like it filtered on effect."
        ),
    ),
    max_rows: int = typer.Option(
        DEFAULT_MAX_ROWS,
        "--max-rows",
        min=1,
        help="Refuse rather than write more rows than this. Raising it is a deliberate act.",
    ),
    offline: bool = typer.Option(
        False, "--offline", help="No-op with a warning: this pass reads the Atlas, not a snapshot."
    ),
    dry_run: bool = typer.Option(False, "--dry-run", help="Report what would be written without writing it."),
    use: str = typer.Option(
        "unstated",
        "--use",
        help=(
            "Declared use recorded on the licence row: unstated|non-commercial|commercial. "
            "AlphaGenome Output is NON-COMMERCIAL ONLY, so an undeclared run writes nothing and "
            "says so — pass --use non-commercial."
        ),
    ),
) -> None:
    """Fill expression_effects.csv with AlphaGenome's per-gene expression effects for one gene.

    Example — and the `--use` is not decoration, an undeclared run is a no-op:

        just-dna-enricher alphagenome expression ./my_module --gene TBX1 --use non-commercial

    One row per (variant, gene): which way the variant moves that gene's predicted expression, how
    many of the 371 tissue tracks agree, and how far it sits from the gene. The interval is the
    gene's MANE span widened by the model's measured +/-512 kb attribution horizon, unless
    --chrom/--start/--end supply one; either way the gene names the server-side filter, and the MANE
    lane is still consulted for the distance, which an explicit interval cannot supply.

    A whole gene is ~3.3 M SNVs at the measured 1,091 SNVs/s — about 50 minutes — and the cost is
    printed before the query runs rather than discovered during it.
    """
    try:
        result = enrich_expression(
            spec,
            gene,
            chrom=chrom,
            start=start,
            end=end,
            min_score=min_score,
            max_rows=max_rows,
            declared_use=_use(use),
            offline=offline,
            write=not dry_run,
        )
    except (ExpressionError, EnrichmentError) as exc:
        typer.secho(f"EXPRESSION FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc

    for note in result.warnings:
        typer.secho(f"  {note}", fg=typer.colors.YELLOW, err=True)
    if result.skipped:
        return
    if result.interval:
        typer.echo(f"interval: {result.interval[0]}:{result.interval[1]}-{result.interval[2]}")
    # The path the pass actually wrote, not `spec / <name>` — a module keeping its sidecars under
    # `derived/` (RM49) is written there, and printing a guess sends the author to the wrong file.
    typer.secho(
        f"expression effects: {sidecar_path(spec, EXPRESSION_SIDECAR, error=ExpressionError)}",
        fg=typer.colors.GREEN,
    )
    typer.echo(
        f"dataset: {result.dataset}  scored: {result.candidates}  written: {result.written}  "
        f"rows: {len(result.rows)}"
    )
    if result.withheld:
        # Grouped by reason, never a bare total: a withhold that cannot say which kind it was is the
        # absence the roster exists to prevent.
        typer.secho(
            "  withheld: " + ", ".join(f"{n} {reason}" for reason, n in sorted(result.withheld.items())),
            fg=typer.colors.YELLOW,
        )
    if result.span is None and result.written:
        typer.secho(
            "  distance_to_gene is null on every row: no MANE span was available, so this table "
            "cannot be thresholded by distance.",
            fg=typer.colors.YELLOW,
        )

alphagenome_publish_

alphagenome_publish_(
    snapshot: Path | None = typer.Argument(
        None,
        exists=True,
        file_okay=False,
        help="The built snapshot directory. Omit to use the resolved cache ($JUST_DNA_ALPHAGENOME_AVI_CACHE, then the cache base), falling back to where `alphagenome build` writes.",
    ),
    repo: str | None = typer.Option(
        None,
        "--repo",
        help=f"Target HF dataset. Default: {DEFAULT_ALPHAGENOME_AVI_REPO_ID}.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded. Reads the repo's file list; sends nothing.",
    ),
    message: str | None = typer.Option(
        None, "--message", "-m", help="Commit message."
    ),
) -> None

Publish the AVI snapshot to HuggingFace.

Its own command because no other one can reach this lane (RM202). cache rebuild --publish walks lanes that have a rebuild adapter, and this lane cannot have one — its source is behind an eligibility gate, so there is nothing for an unattended rebuild to fetch. RM198 gave the lane a publish_repo and left it unreachable.

The upload is two commits and, above 5 GB, goes through the resumable uploader (RM199): the payload first, then release.json — the description must never arrive before the bytes it describes.

Source code in enricher/src/just_dna_enricher/cli.py
@alphagenome_app.command("publish")
def alphagenome_publish_(
    snapshot: Path | None = typer.Argument(
        None,
        exists=True,
        file_okay=False,
        help=(
            "The built snapshot directory. Omit to use the resolved cache "
            "($JUST_DNA_ALPHAGENOME_AVI_CACHE, then the cache base), falling back to where "
            "`alphagenome build` writes."
        ),
    ),
    repo: str | None = typer.Option(
        None,
        "--repo",
        help=f"Target HF dataset. Default: {DEFAULT_ALPHAGENOME_AVI_REPO_ID}.",
    ),
    dry_run: bool = typer.Option(
        False,
        "--dry-run",
        help="Show what would be uploaded. Reads the repo's file list; sends nothing.",
    ),
    message: str | None = typer.Option(None, "--message", "-m", help="Commit message."),
) -> None:
    """Publish the AVI snapshot to HuggingFace.

    **Its own command because no other one can reach this lane** (RM202). `cache rebuild --publish`
    walks lanes that have a `rebuild` adapter, and this lane cannot have one — its source is behind an
    eligibility gate, so there is nothing for an unattended rebuild to fetch. RM198 gave the lane a
    `publish_repo` and left it unreachable.

    The upload is two commits and, above 5 GB, goes through the resumable uploader (RM199): the
    payload first, then `release.json` — the description must never arrive before the bytes it
    describes.
    """
    from just_dna_enricher.upload import plan_reference_snapshot, publish_reference_snapshot

    target = repo or DEFAULT_ALPHAGENOME_AVI_REPO_ID
    if snapshot is None:
        snapshot = _resolve_avi_snapshot()
        if snapshot is None:
            typer.secho(
                "PUBLISH FAILED: no AVI snapshot found. Looked at "
                f"$JUST_DNA_ALPHAGENOME_AVI_CACHE, the cache base, and "
                f"{repro_out('alphagenome_avi').resolve()}. Build one with `alphagenome build "
                "--input <the artifact you downloaded>`, or pass the directory explicitly.",
                fg=typer.colors.RED,
                err=True,
            )
            raise typer.Exit(code=1)
        typer.echo(f"  using {snapshot}")
    try:
        if dry_run:
            plan = plan_reference_snapshot(snapshot, target)
            typer.echo(f"would upload {len(plan.files)} file(s) to {plan.repo_id}:")
            for name in plan.files:
                size = (snapshot / name).stat().st_size if (snapshot / name).is_file() else 0
                typer.echo(f"  {name}  ({size / 1e9:.2f} GB)" if size > 1e8 else f"  {name}")
            typer.echo(
                "  release.json is sent last, in its own commit — a description that arrives "
                "before its bytes describes a snapshot nobody has."
            )
            return
        plan = publish_reference_snapshot(snapshot, target, commit_message=message)
    except (FileNotFoundError, PermissionError, ImportError, ValueError) as exc:
        typer.secho(f"PUBLISH FAILED: {exc}", fg=typer.colors.RED, err=True)
        raise typer.Exit(code=1) from exc
    if KNOT_FILENAME not in plan.files:
        typer.secho(
            f"  no {KNOT_FILENAME} in this snapshot — a puller would hold scores they cannot rank, "
            "because PHRED is not stored. Rebuild with `alphagenome build`.",
            fg=typer.colors.YELLOW,
            err=True,
        )
    typer.secho(
        f"published: {snapshot} → {plan.repo_id} ({len(plan.files)} files)",
        fg=typer.colors.GREEN,
    )