From 0a801c009d45ac6ba8dfe82be8df141a45da667e Mon Sep 17 00:00:00 2001 From: JK Date: Tue, 16 Jun 2026 11:09:48 -0400 Subject: [PATCH] Polish linked public card docs --- README.md | 9 +- docs/SPACEBIOBENCH_CLAIM_REGISTER.md | 395 +++---------------- docs/SPACEBIOBENCH_EVALUATION_CARD.md | 174 +++----- docs/SPACEBIOBENCH_PORTFOLIO_BRIEF.md | 136 +++---- docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md | 225 ++++------- docs/SPACEBIOBENCH_SYSTEM_CARD.md | 277 ++++--------- docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md | 136 +++---- scripts/validate_public_docs_consistency.py | 69 ++++ 8 files changed, 422 insertions(+), 999 deletions(-) diff --git a/README.md b/README.md index b63209a..fab25bd 100644 --- a/README.md +++ b/README.md @@ -40,8 +40,8 @@ mission-held-out validation, and transparent release boundaries. | Hugging Face dataset | Public processed fold package | Download selected LOMO feature matrices and result artifacts | [HF dataset card](docs/hf_dataset_card.md) | | v9 public bulk | Metadata catalog | Task catalog, source inventory, and baseline summaries | [v9 HF-style card](docs/v9_hf_dataset_card.md) | -For linked methods, evaluation, and release cards, start with the -[SpaceBio-Bench card pack](docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md). +For linked methods, evaluation, and release-status notes, start with the +[SpaceBio-Bench public documentation map](docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md). For machine-readable release status, see [release/release_manifest.json](release/release_manifest.json). @@ -159,10 +159,11 @@ scripts/ Data, evaluation, upload, validation, and figure scripts | Need | Document | |---|---| | Public result source | [docs/CANONICAL_RESULTS_V7_1.md](docs/CANONICAL_RESULTS_V7_1.md) | -| Methods, evaluation, and release cards | [docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md](docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md) | +| Public documentation map | [docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md](docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md) | | System scope | [docs/SPACEBIOBENCH_SYSTEM_CARD.md](docs/SPACEBIOBENCH_SYSTEM_CARD.md) | | Evaluation interpretation | [docs/SPACEBIOBENCH_EVALUATION_CARD.md](docs/SPACEBIOBENCH_EVALUATION_CARD.md) | -| Release readiness | [docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md](docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md) | +| Release status | [docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md](docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md) | +| Public statement guide | [docs/SPACEBIOBENCH_CLAIM_REGISTER.md](docs/SPACEBIOBENCH_CLAIM_REGISTER.md) | | Hugging Face dataset card source | [docs/hf_dataset_card.md](docs/hf_dataset_card.md) | | v9 metadata catalog card source | [docs/v9_hf_dataset_card.md](docs/v9_hf_dataset_card.md) | | Contributing and submissions | [CONTRIBUTING.md](CONTRIBUTING.md) and [docs/submission_format.md](docs/submission_format.md) | diff --git a/docs/SPACEBIOBENCH_CLAIM_REGISTER.md b/docs/SPACEBIOBENCH_CLAIM_REGISTER.md index 057020f..1f7bb89 100644 --- a/docs/SPACEBIOBENCH_CLAIM_REGISTER.md +++ b/docs/SPACEBIOBENCH_CLAIM_REGISTER.md @@ -1,362 +1,75 @@ --- -title: SpaceBio-Bench Claim Register -page_type: evidence_register -status: public_review_ready -last_reviewed: 2026-06-05 -claim_boundary: benchmark_claim_register_draft_no_new_release_claim +title: SpaceBio-Bench Public Statement Guide +page_type: public_statement_guide +status: public_ready +last_reviewed: 2026-06-16 --- -# SpaceBio-Bench Claim Register +# SpaceBio-Bench Public Statement Guide ## Purpose -This register ties major SpaceBio-Bench claims to support level, source files, -confidence, allowed language, and blocked language. It is designed to prevent -mixed-surface claims across v1-v7, v8, and v9. +This guide gives concise, public-ready wording for common SpaceBio-Bench +statements. Use it when writing README text, dataset cards, release notes, +abstracts, talks, or issue responses. -The register is a documentation control, not a new release artifact and not a -source of new benchmark results. +## Preferred Wording -Branch note: on the default `main` branch, v9-specific evidence paths such as -`v9/...` and `docs/V9_*` refer to the curated public bulk metadata-alpha subset -included in this repository. Payload matrices and draft extension lanes are -excluded from the current public-review path. +| Topic | Use this wording | Context | +|---|---|---| +| Project name | SpaceBio-Bench / GeneLab Benchmark | SpaceBio-Bench is the forward-looking platform name; GeneLab Benchmark is the historical public name | +| Core task | Mission-held-out spaceflight transcriptomics benchmark | The independence unit is the held-out mission | +| v7.1 result surface | Canonical historical result surface for v1-v7 | Use for published result summaries and citation context | +| v7.1.2 patch | Documentation, public-card, citation, and metadata patch over canonical v7.1 results | The patch does not add new benchmark result generation | +| Hugging Face package | Processed public fold package for selected LOMO tasks | Use for direct dataset downloads | +| v9 public bulk | Public bulk metadata catalog | Use for task manifests, source records, fold indexes, audit summaries, and reference baselines | +| Foundation-model result summary | Tested gene-expression foundation models underperform tuned classical baselines on small-n bulk RNA-seq mission shift | Keep tied to the documented v7.1 result surface | +| Classical baseline summary | PCA-LR is the strongest 8-tissue gene-level baseline in v4, with mean AUROC 0.776 | Use with v7.1 canonical result context | +| Held-out validation | Thymus RR-23 AUROC 0.905; skin RR-7 AUROC 0.885 | Use with the v7.1 held-out validation context | +| OSDR source data | Source data are derived from public NASA OSDR studies | Cite the relevant OSDR study pages | -## Evidence Vocabulary +## Result Interpretation -- `primary`: local source file, generated artifact, official external page, or - primary paper. -- `project-synthesis`: synthesis from multiple local primary sources. -- `inference`: reasonable interpretation from primary sources; should be - phrased cautiously. -- `blocked`: not currently supported by the evidence boundary. -- `future`: plausible future claim after named blockers are resolved. +- Report task, fold, tissue, method, feature surface, and release surface. +- Pair pooled summaries with per-task or per-fold rows. +- Treat benchmark scores as evidence for the stated benchmark task. +- Keep dataset, model, and release labels aligned across README, HF card, + citation metadata, and release notes. -## Reader Summary +## Scope Language -The current public-safe summary is: +Use scope language that is precise and compact: -- v1-v7 is the canonical historical benchmark result surface. -- v7.1.2 is a documentation, public-card, and metadata patch over - the canonical v7.1 result surface, not a new result release. -- v8 is an incubating translational extension and should not be mixed into - v7.1 claims. -- v9 public bulk is a metadata-only alpha with explicit payload blockers. -- Current v9 baselines are scaffold anchors, not final leaderboard entries. -- Clinical, crew-health, intervention, countermeasure, and Mars-regime claims - are out of scope for the current evidence boundary. +- "The public benchmark evaluates mission-held-out transcriptomics + generalization." +- "The v9 public bulk surface is a metadata catalog." +- "The Hugging Face dataset provides processed public fold packages." +- "Clinical, crew-health, countermeasure, and operational-readiness use cases + are outside the current benchmark scope." -## Current Supported Claim Cards +## Citation And Acknowledgment -### SBB-C001 - Cross-Mission Benchmark Surface +Use the GitHub `CITATION.cff` metadata for the software and benchmark citation. +For source data, cite NASA OSDR and the individual OSDR studies used in the +analysis. -- Claim: v1-v7 GeneLab Benchmark evaluates cross-mission generalization of - mouse spaceflight transcriptomic signatures. -- Support: primary. -- Sources: `docs/CANONICAL_RESULTS_V7_1.md`; `docs/hf_dataset_card.md`. -- Confidence: high. -- Use: "cross-mission mouse spaceflight transcriptomics benchmark". -- Avoid: "clinical astronaut health predictor". +Recommended OSDR acknowledgment: -### SBB-C002 - v7.1.2 Patch Boundary +> Data are courtesy of the NASA Open Science Data Repository. -- Claim: v7.1.2 is a documentation, public-card, and metadata patch - over the canonical v7.1 result surface, not a new result-generation release. -- Support: primary. -- Sources: `README.md`; `docs/CANONICAL_RESULTS_V7_1.md`; - `docs/hf_dataset_card.md`. -- Confidence: high. -- Use: "v7.1.2 documentation/card/metadata patch; no new benchmark result - generation". -- Avoid: "v7.1.2 adds new benchmark results". +OSDR resource citation: -### SBB-C003 - v4 Multi-Method Scope +Gebre S G, Scott R T, Saravia-Butler A M, Lopez D K, Sanders L M, and Costes S +V. 2024. NASA Open Science Data Repository: Open Science for Life in Space. +Nucleic Acids Research 53(D1): D1697-D1710. +https://doi.org/10.1093/nar/gkae1116 -- Claim: the v4 multi-method surface covers 8 tissues, 8 classifiers, and 4 - feature types for 256 evaluations. -- Support: primary. -- Sources: `docs/CANONICAL_RESULTS_V7_1.md`. -- Confidence: high. -- Use: "v4 multi-method evaluation: 8 tissues x 8 classifiers x 4 feature - types". -- Avoid: "all methods were evaluated on every later v8/v9 task". +## Companion Documents -### SBB-C004 - Foundation-Model Snapshot Boundary - -- Claim: foundation-model and text-LLM rows are mixed-surface snapshots, not a - single uniform 8-tissue FM leaderboard. -- Support: primary. -- Sources: `docs/CANONICAL_RESULTS_V7_1.md`. -- Confidence: high. -- Use: "benchmark-surface summary with subset notes". -- Avoid: "single uniform 8-tissue foundation-model leaderboard". - -### SBB-C005 - Classical Baseline Comparison - -- Claim: current gene-expression FMs in the canonical v7.1 snapshot do not - automatically outperform tuned classical baselines under small-n bulk RNA-seq - shift. -- Support: primary. -- Sources: `docs/CANONICAL_RESULTS_V7_1.md`. -- Confidence: high. -- Use: "do not automatically outperform tuned classical baselines". -- Avoid: "foundation models fail at space biology". - -### SBB-C006 - v8 Boundary - -- Claim: v8 SpaceMed is an incubating translational extension and should not be - mixed into v7.1 benchmark claims. -- Support: primary. -- Sources: `docs/CANONICAL_RESULTS_V7_1.md`; - `docs/V8_BETA_RELEASE_PLAN_2026_05_10.md`. -- Confidence: high. -- Use: "incubating translational extension". -- Avoid: "v8 proves countermeasure efficacy". - -### SBB-C007 - v9 Metadata-Alpha Boundary - -- Claim: v9 public bulk is a metadata-only alpha snapshot, not a frozen payload - release. -- Support: primary. -- Sources: `docs/v9_hf_dataset_card.md`; - `docs/V9_PUBLIC_BULK_ALPHA_METADATA_SNAPSHOT_DECISION.md`; - `docs/V9_PUBLIC_BULK_ALPHA_CARD_DATAPACKAGE_BOUNDARY_UPDATE.md`. -- Confidence: high. -- Use: "SpaceBio-Bench v9 public bulk metadata alpha". -- Avoid: "frozen v9 public benchmark release". - -### SBB-C008 - v9 Public Bulk Inventory - -- Claim: v9 public bulk currently includes 8 generated public bulk LOMO task - manifests, 6 tissue contexts, 22 deduplicated public OSDR source rows, 33 - fold definitions, 24 baseline runs, and 21 draft Data Package resources. -- Support: primary. -- Sources: `docs/v9_hf_dataset_card.md`; `v9/datapackage.draft.json`. -- Confidence: high. -- Use: counts with a "current public bulk draft" qualifier. -- Avoid: using counts as a frozen DOI release inventory. - -### SBB-C009 - Checksum Evidence Boundary - -- Claim: OSDR API and checksum-manifest evidence has been parsed for all 22 - public bulk source rows, but local payload-level hash verification is still - pending. -- Support: primary. -- Sources: `docs/v9_hf_dataset_card.md`; `v9/source_checksum_audit.csv`; - `docs/V9_PUBLIC_BULK_ALPHA_METADATA_SNAPSHOT_DECISION.md`. -- Confidence: high. -- Use: "checksum-manifest evidence parsed; payload hashing pending". -- Avoid: "locally hash-verified payload bundle". - -### SBB-C010 - Baseline Status - -- Claim: v9 public bulk baselines validate the scaffold workflow and provide - anchors, but are not tuned leaderboard endpoints. -- Support: primary. -- Sources: `docs/v9_hf_dataset_card.md`; - `v9/reports/bulk_lomo_baseline_summary.csv`. -- Confidence: high. -- Use: "scaffold baselines". -- Avoid: "final model rankings". - -### SBB-C011 - Per-Task Reporting - -- Claim: per-task and per-fold reporting should accompany pooled summaries - because pooled averages can hide mission or tissue failures. -- Support: primary/project-synthesis. -- Sources: `docs/v9_hf_dataset_card.md`; `docs/CANONICAL_RESULTS_V7_1.md`. -- Confidence: high. -- Use: "report per-task results, not only pooled averages". -- Avoid: "single pooled score fully characterizes the method". - -### SBB-C012 - Mission Confounding - -- Claim: mission labels can conflate biological spaceflight signal with - vehicle, hardware, protocol, tissue handling, time, and processing effects. -- Support: primary/project-synthesis. -- Sources: `docs/v9_hf_dataset_card.md`; `docs/SPACEBIOBENCH_SYSTEM_CARD.md`. -- Confidence: medium-high. -- Use: "mission-shift benchmark with known confounding risks". -- Avoid: "pure microgravity effect estimator". - -### SBB-C013 - Unsupported Operational Claims - -- Claim: public bulk tasks do not support clinical, crew-health, - countermeasure, intervention, or Mars-regime claims. -- Support: primary. -- Sources: `docs/v9_hf_dataset_card.md`; `docs/CANONICAL_RESULTS_V7_1.md`. -- Confidence: high. -- Use: "benchmark evidence, not biological mechanism or operational - recommendation". -- Avoid: "astronaut health-risk or countermeasure recommendation". - -### SBB-C014 - OSDR Credit - -- Claim: OSDR and individual OSDR datasets should be credited and cited for - downstream analyses. -- Support: primary. -- Sources: `docs/v9_hf_dataset_card.md`; NASA OSDR FAQ. -- Confidence: high. -- Use: "Data are courtesy of the NASA Open Science Data Repository" plus - dataset-specific citations. -- Avoid: hand-written substitute citations without checking OSDR study pages. - -### SBB-C015 - Dataset-Card Role - -- Claim: dataset cards should document contents, context, intended use, - creation, responsible use, and limitations. -- Support: primary. -- Sources: Hugging Face dataset-card docs; Datasheets for Datasets. -- Confidence: high. -- Use: "dataset card as human-facing responsible-use surface". -- Avoid: "README with only download commands is sufficient". - -### SBB-C016 - System Card Versus Model Card - -- Claim: model cards are appropriate for individual trained models or adapters, - but SpaceBio-Bench itself should be documented as a benchmark/system card. -- Support: primary/inference. -- Sources: Model Cards paper; Hugging Face model-card docs; - `docs/SPACEBIOBENCH_SYSTEM_CARD.md`. -- Confidence: high. -- Use: "benchmark/system card for the project; model cards for individual - baselines". -- Avoid: "the full benchmark is a model card". - -## Release Readiness And Future Claim Cards - -### SBB-C017 - Frozen Payload Requirements - -- Claim: a future frozen payload release should add machine-readable payload - manifests and verification reports. -- Support: primary/inference. -- Sources: BagIt RFC 8493; `docs/v9_hf_dataset_card.md`; - `docs/V9_PUBLIC_BULK_ALPHA_METADATA_SNAPSHOT_DECISION.md`. -- Confidence: high. -- Use: "payload-level SHA-256 manifest before frozen payload language". -- Avoid: "payload freeze without payload-level hashes". - -### SBB-C018 - Research-Object Provenance - -- Claim: a future citable research-object release should add RO-Crate or - equivalent provenance metadata. -- Support: primary/inference. -- Sources: RO-Crate technical overview; `docs/V9_LONG_RUN_OPERATING_PROTOCOL.md`; - `docs/v9_hf_dataset_card.md`. -- Confidence: medium-high. -- Use: "future RO-Crate export for research-object provenance". -- Avoid: "current v9 alpha is already a complete citable RO-Crate release". - -### SBB-C019 - DOI-Oriented Metadata - -- Claim: DataCite-style metadata is useful for future DOI-oriented release - planning. -- Support: primary/inference. -- Sources: DataCite Metadata Schema; `docs/V9_LONG_RUN_OPERATING_PROTOCOL.md`. -- Confidence: medium-high. -- Use: "align release metadata with DataCite fields before DOI/archive release". -- Avoid: "DOI release ready without creator, version, related identifier, - license, and resource type review". - -### SBB-C020 - NIST AI RMF Lens - -- Claim: NIST AI RMF can inform documentation structure, but SpaceBio-Bench is - not a deployed AI product. -- Support: primary/inference. -- Sources: NIST AI RMF; `docs/SPACEBIOBENCH_SYSTEM_CARD.md`. -- Confidence: medium. -- Use: "use Govern/Map/Measure/Manage as documentation lenses". -- Avoid: "NIST compliance claim". - -### SBB-C021 - Evaluation Reading Order - -- Claim: evaluation should be interpreted through task, fold, source, - payload-boundary, and run-manifest evidence before pooled summaries. -- Support: primary/project-synthesis. -- Sources: `docs/SPACEBIOBENCH_EVALUATION_CARD.md`; - `docs/v9_hf_dataset_card.md`; `v9/task_data_index.csv`. -- Confidence: high. -- Use: "read per-task and per-fold metrics before pooled means". -- Avoid: "single pooled mean fully establishes benchmark performance". - -### SBB-C022 - v9 Release Tier - -- Claim: v9 public bulk currently satisfies a metadata-alpha tier, not a - frozen-payload or DOI/archive tier. -- Support: primary. -- Sources: `docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md`; - `docs/V9_PUBLIC_BULK_ALPHA_CARD_DATAPACKAGE_BOUNDARY_UPDATE.md`; - `v9/reports/public_bulk_alpha_snapshot_decision/snapshot_decision_summary.csv`. -- Confidence: high. -- Use: "metadata alpha with explicit payload blockers". -- Avoid: "frozen-payload or archive-ready release". - -## Blocked Or Future Claims - -### Frozen v9 Public Bulk Payload Release - -- Current blocker: payload mirroring and payload-level hash verification are - pending. -- Evidence needed: local payload mirror, SHA-256 manifest, verification report, - and release `datapackage.json`. - -### DOI Or Archive-Ready Release - -- Current blocker: metadata alpha status and license/citation review are - pending. -- Evidence needed: DataCite-aligned metadata, final license, dataset-specific - OSDR citations, and archive manifest. - -### Complete RO-Crate Research Object - -- Current blocker: RO-Crate export has not yet been created. -- Evidence needed: `ro-crate-metadata.json`, entity graph, workflow links, and - provenance links. - -### Foundation-Model Leaderboard - -- Current blocker: adapter validation and matched evaluation surfaces are - incomplete. -- Evidence needed: matched task inputs, adapter cards, run manifests, per-task - metrics, and leakage checks. - -### Biological Mechanism Proof - -- Current blocker: benchmark scores are not mechanistic validation. -- Evidence needed: independent biological validation, mechanistic assays, and - matched causal analysis. - -### Countermeasure Or Intervention Recommendation - -- Current blocker: v8/v9 diagnostic claims do not validate interventions. -- Evidence needed: controlled intervention evidence, safety review, and - translational validation. - -### Crew-Health Or Clinical Decision Support - -- Current blocker: public benchmark scope excludes clinical recommendations. -- Evidence needed: controlled human-data review, clinical validation, and - institutional review. - -## Maintenance Rules - -- Update this register whenever public-facing result counts, release status, - or allowed language changes. -- Do not promote a claim from `inference` to `primary` without adding the local - or external primary source. -- Keep v7.1 result claims, v8 translational hypotheses, and v9 alpha scaffold - claims in separate rows. -- When in doubt, prefer allowed wording that names the release surface and - status explicitly. - -## External Sources - -- Model Cards for Model Reporting: https://arxiv.org/abs/1810.03993 -- Hugging Face model cards: https://huggingface.co/docs/hub/main/model-cards -- Hugging Face dataset cards: https://huggingface.co/docs/datasets/v2.7.0/en/dataset_card -- Datasheets for Datasets: https://www.microsoft.com/en-us/research/uploads/prod/2019/01/1803.09010.pdf -- NIST AI RMF: https://www.nist.gov/itl/ai-risk-management-framework -- NASA OSDR FAQ: https://science.nasa.gov/reference/osdr-faq/ -- RO-Crate: https://www.researchobject.org/ro-crate/technical_overview -- BagIt RFC 8493: https://www.rfc-editor.org/info/rfc8493/ -- DataCite Metadata Schema: https://schema.datacite.org/ +- [Public documentation map](SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md) +- [System card](SPACEBIOBENCH_SYSTEM_CARD.md) +- [Evaluation card](SPACEBIOBENCH_EVALUATION_CARD.md) +- [Release status card](SPACEBIOBENCH_RELEASE_READINESS_CARD.md) +- [Canonical v7.1 results](CANONICAL_RESULTS_V7_1.md) +- [v7.1 Hugging Face dataset card](hf_dataset_card.md) +- [v9 metadata catalog card](v9_hf_dataset_card.md) diff --git a/docs/SPACEBIOBENCH_EVALUATION_CARD.md b/docs/SPACEBIOBENCH_EVALUATION_CARD.md index 5b87159..dfdddec 100644 --- a/docs/SPACEBIOBENCH_EVALUATION_CARD.md +++ b/docs/SPACEBIOBENCH_EVALUATION_CARD.md @@ -1,72 +1,53 @@ --- title: SpaceBio-Bench Evaluation Card page_type: evaluation_card -status: public_review_ready -last_reviewed: 2026-06-05 -claim_boundary: benchmark_evaluation_card_draft_no_new_result_claim +status: public_ready +last_reviewed: 2026-06-16 --- # SpaceBio-Bench Evaluation Card -## Evaluation Purpose +## Purpose -This card documents how SpaceBio-Bench evaluations should be interpreted. It -separates task validity, fold structure, metric reporting, baseline status, and -claim boundaries so that benchmark scores are not overread as biological -mechanism, translational readiness, or model superiority claims. - -This card does not introduce new results. It summarizes evaluation evidence -already recorded in the v7.1 canonical result surface and the v9 public bulk -metadata-alpha scaffold. - -Branch note: on the default `main` branch, v9-specific evidence paths such as -`v9/...` and `docs/V9_*` refer to the curated public bulk metadata-alpha subset -included in this repository. Payload matrices and draft extension lanes are -excluded from the current public-review path. +This card explains how to read SpaceBio-Bench evaluations. It separates task +definition, fold structure, metric reporting, baseline rows, and pooled +summaries so readers can understand what a score means. ## Evaluation Surfaces -| Surface | Current status | Evaluation unit | Result boundary | -|---|---|---|---| -| v1-v7 / v7.1 canonical surface | Canonical historical result surface | Tissue, method, feature type, held-out validation, and FM snapshot rows | Cross-mission transcriptomics benchmark summary | -| v9 public bulk alpha | Metadata-only alpha scaffold | Task manifest, LOMO fold, baseline run, prediction row, run manifest | Workflow and provenance evidence, not leaderboard ranking | -| v9 draft extension lanes | Diagnostic or feasibility scaffolds | Asset inventory, metric spec, payload audit, draft task manifest, or diagnostic run | Draft-only; no public benchmark score claim | +| Surface | Evaluation unit | Public interpretation | +|---|---|---| +| v1-v7 / v7.1 canonical surface | Tissue, method, feature type, held-out validation, and foundation-model snapshot rows | Historical cross-mission transcriptomics benchmark summary | +| Hugging Face public fold package | Public LOMO fold with matrices, labels, metadata, and selected genes | Processed data package for reproducible task-level evaluation | +| v9 public bulk catalog | Task manifest, LOMO fold, baseline run, prediction row, metric file, and run record | Metadata catalog and reference baseline surface | ## Evaluation Flow -| Stage | Evidence to inspect | Interpretation control | +| Stage | Evidence to inspect | Why it matters | |---|---|---| -| 1. Source inventory | OSDR accessions, tissue labels, mission labels, access status, and checksum-manifest evidence | Confirms the public data source before interpreting any score | -| 2. Task manifest | Task id, tissue, feature namespace, source ids, label map, and metric ids | Defines what the evaluation is actually testing | -| 3. Held-out mission fold | Train/test mission split, row counts, and selected-gene counts | Keeps mission-held-out validation separate from random-split performance | -| 4. Prediction and metric files | Baseline or submitted predictions, task/fold ids, AUROC, macro-F1, balanced accuracy, calibration | Ties every metric to a concrete task and fold surface | -| 5. Per-task interpretation | Tissue-specific and fold-specific behavior | Prevents pooled means from hiding failures or confounding | -| 6. Pooled summary | Aggregate result only after task/fold checks | Allows navigation-level summaries with mission, tissue, baseline, and payload caveats | -| 7. Claim register language | Allowed, blocked, and future-only wording | Converts evaluation evidence into release-safe public claims | +| 1. Source rows | OSDR accessions, tissue labels, mission labels, and access status | Identifies the public data behind a task | +| 2. Task manifest | Task ID, tissue, feature namespace, label map, and metric IDs | Defines what the evaluation is testing | +| 3. Held-out mission fold | Train/test mission split, row counts, and selected-gene counts | Separates mission-held-out validation from random-split performance | +| 4. Prediction and metric files | Predictions, task/fold IDs, AUROC, macro-F1, balanced accuracy, and calibration | Ties every score to a concrete task and fold | +| 5. Per-task interpretation | Tissue-specific and fold-specific behavior | Preserves mission and tissue variability | +| 6. Pooled summary | Aggregate metrics after task and fold checks | Gives a navigation-level summary | -The evaluation flow is intentionally claim-aware. A score is first interpreted -at the task and fold level, then summarized only with caveats about mission, -tissue, payload, baseline, and release-surface boundaries. +## v9 Public Bulk Evaluation Unit -## Current v9 Public Bulk Evaluation Unit +The v9 public bulk catalog organizes mission-held-out classification tasks for +public mouse bulk RNA-seq sources. -The current v9 public bulk lane evaluates mission-held-out classification tasks -for public mouse bulk RNA-seq sources. The unit hierarchy is: - -| Unit | Meaning | Evidence | +| Unit | Meaning | Public files | |---|---|---| -| Task | Tissue-specific bulk LOMO task with source ids, missions, feature namespace, and metric ids | `v9/task_manifest_index.csv`; `v9/task_manifests/*.json` | +| Task | Tissue-specific bulk LOMO task with missions, labels, feature namespace, and metric IDs | `v9/task_manifest_index.csv`; `v9/task_manifests/*.json` | | Fold | One held-out mission with train/test row counts and selected-gene counts | `v9/task_data_index.csv` | | Baseline run | One baseline family evaluated on one task | `v9/reports/bulk_lomo_baseline_summary.csv` | -| Prediction file | Per-sample prediction output for a task/baseline run | per-baseline `predictions.csv` | +| Prediction file | Per-sample prediction output for a task and baseline run | per-baseline `predictions.csv` | | Metrics file | Task-level metric output for a baseline run | per-baseline `metrics.json` | -| Run manifest | Provenance for a baseline execution | per-baseline `run_manifest.json` | +| Run record | Execution metadata for a baseline run | per-baseline `run_manifest.json` | ## Current Task Inventory -The v9 public bulk metadata-alpha scaffold currently indexes eight generated -bulk LOMO task manifests and 33 fold definitions: - | Task | Tissue | Variant | Missions | Folds | Sources | |---|---|---:|---:|---:|---:| | `A1_liver_bulk_lomo` | liver | canonical | 6 | 6 | 6 | @@ -78,37 +59,28 @@ bulk LOMO task manifests and 33 fold definitions: | `A5_skin_bulk_lomo` | skin | canonical | 3 | 3 | 4 | | `A6_eye_bulk_lomo` | eye | canonical | 3 | 3 | 3 | -These tasks are part of a metadata-only alpha snapshot. They do not imply a -frozen payload release. - ## Metrics -Current v9 public bulk task manifests list these metric ids: - | Metric | Interpretation | Reporting note | |---|---|---| | `macro_f1` | Class-balanced F1 summary across labels | Report per task; small folds can be unstable | | `balanced_accuracy` | Mean sensitivity across classes | Useful under class imbalance | -| `auroc` | Rank-based discrimination between flight/control labels | Can look strong even when calibration or fold reliability is weak | -| `calibration_error` | Probability calibration diagnostic | Lower is better; compare only across matched output formats | +| `auroc` | Rank-based discrimination between flight/control labels | Compare with calibration and per-fold behavior | +| `calibration_error` | Probability calibration diagnostic | Lower is better; compare matched output formats | | `mission_discrimination` | Diagnostic for mission-correlated structure where embeddings are available | Not emitted for every baseline family | -Metric reporting should include the task id, fold family, baseline id, variant, -and whether the result comes from a canonical result surface, metadata alpha, or -draft diagnostic lane. - -## Current v9 Baseline Status +Metric reports should include task ID, fold ID, baseline or method ID, variant, +and release surface. -The current v9 public bulk scaffold includes 24 evaluated baseline rows across -three simple baseline families: +## Current v9 Baseline Rows -| Baseline family | Rows | Current interpretation | +| Baseline family | Rows | Interpretation | |---|---:|---| | `logistic_regression_l2` | 8 | Simple linear workflow baseline | -| `nearest_centroid` | 8 | Simple centroid-based diagnostic baseline | +| `nearest_centroid` | 8 | Simple centroid-based baseline | | `pca_logistic_regression` | 8 | PCA feature-compression plus logistic baseline | -Mean metrics across the eight task rows are recorded in the v9 dataset card: +Mean metrics across the eight task rows: | Baseline | Macro-F1 | Balanced accuracy | AUROC | Calibration error | Mission discrimination | |---|---:|---:|---:|---:|---:| @@ -116,79 +88,45 @@ Mean metrics across the eight task rows are recorded in the v9 dataset card: | Nearest centroid | 0.5383 | 0.5685 | 0.6321 | 0.1132 | 0.8733 | | PCA logistic regression | 0.5353 | 0.5619 | 0.6447 | 0.3747 | 0.9190 | -These are scaffold baselines. They validate the task/evaluation plumbing and -provide anchors for future methods, but they are not tuned leaderboard -endpoints. - -## What This Enables - -The evaluation card gives readers a compact way to check whether a score is -being read at the right level: task, fold, baseline run, or pooled summary. It -also makes clear when a number is a workflow sanity check rather than a model -ranking. +These rows provide reproducible reference anchors. Use per-task rows together +with pooled means when comparing methods. ## Reading Order For Results -Read results in this order: - -1. Release surface and claim boundary. +1. Release surface. 2. Task and fold definition. -3. Source provenance and payload verification status. +3. Source rows and access status. 4. Per-fold and per-task metrics. -5. Baseline run manifest and prediction count. -6. Pooled summary, only after the above checks. +5. Baseline run record and prediction count. +6. Pooled summary. -Do not start with a pooled mean alone. Pooled summaries can hide fold-level -failure, tissue-specific instability, mission-label confounding, or variant -differences. +Starting from the task and fold keeps tissue-specific behavior, mission-label +structure, and variant differences visible. ## Leakage And Confounding Controls -Current controls: - - Mission-held-out fold definitions are explicit in `v9/task_data_index.csv`. - Task-source relationships are explicit in `v9/task_manifest_index.csv`. -- v7.1 public wording requires subset notes when comparing mixed FM surfaces. -- v9 public bulk alpha separates public bulk tasks from draft organoid, - single-cell, and multispecies lanes. -- v9 alpha documents that payload-level hash verification is pending. - -Controls that remain future or lane-specific: - -- Payload-level hash verification for every distributed fold matrix. -- Release-time revalidation of feature-selection and preprocessing leakage - controls for any frozen v9 payload bundle. -- Adapter-specific validation for foundation models and virtual-cell models. -- Human or biological validation for mechanistic interpretation claims. +- v7.1 public wording includes subset notes for mixed foundation-model + comparisons. +- v9 public bulk tasks are separated from extension workspaces for other + modalities. +- Feature selection and preprocessing should be checked within each training + split for any newly packaged fold. ## Failure Modes To Watch | Failure mode | Why it matters | Mitigation | |---|---|---| -| Pooled-score overclaim | A high mean can hide poor tissue or mission performance | Always report per-task metrics | -| Mixed-surface FM comparison | Rows may come from different tissues, adapters, or versions | Label the reported surface for every row | -| Mission confounding | Mission labels can encode hardware, vehicle, age, protocol, or processing | Treat mission shift as benchmark pressure, not pure biology | -| Payload-boundary drift | Metadata alpha can be mistaken for frozen payload release | Keep payload hash status in every release-facing card | -| Extension-lane leakage | Draft organoid, single-cell, or multispecies lanes can be pulled into public bulk claims | Keep release target and claim boundary explicit | -| Biological interpretation overreach | Classification performance is not mechanistic proof | Separate benchmark evidence from mechanistic validation | - -## Boundary Summary +| Pooled-score overread | A high mean can hide weak tissue or mission rows | Report per-task metrics | +| Mixed-surface comparison | Rows can come from different tissues, adapters, or versions | Label the release surface for every row | +| Mission confounding | Mission labels can encode hardware, vehicle, age, protocol, or processing | Interpret mission shift as the benchmark pressure | +| Biological overinterpretation | Classification performance is not mechanistic validation | Separate benchmark scores from mechanistic follow-up | -The current evaluation surfaces should not be read as evidence for: - -- A frozen v9 payload release. -- A state-of-the-art model leaderboard. -- A uniform 8-tissue foundation-model comparison. -- Clinical, crew-health, countermeasure, intervention, or Mars-regime validity. -- Biological mechanism proof from classifier performance. -- Payload-level integrity beyond the currently documented checksum-manifest - evidence. - -## Reproducibility Evidence - -Local evidence: +## Reproducibility Files - `docs/CANONICAL_RESULTS_V7_1.md` +- `docs/hf_dataset_card.md` - `docs/v9_hf_dataset_card.md` - `v9/task_manifest_index.csv` - `v9/task_data_index.csv` @@ -198,9 +136,3 @@ Local evidence: - `v9/reports/nearest_centroid/bulk_lomo_summary.csv` - `v9/reports/sklearn_baselines/bulk_lomo_summary.csv` - `v9/datapackage.draft.json` - -Companion cards: - -- `docs/SPACEBIOBENCH_SYSTEM_CARD.md` -- `docs/SPACEBIOBENCH_CLAIM_REGISTER.md` -- `docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md` diff --git a/docs/SPACEBIOBENCH_PORTFOLIO_BRIEF.md b/docs/SPACEBIOBENCH_PORTFOLIO_BRIEF.md index 4be3291..bee74be 100644 --- a/docs/SPACEBIOBENCH_PORTFOLIO_BRIEF.md +++ b/docs/SPACEBIOBENCH_PORTFOLIO_BRIEF.md @@ -1,130 +1,114 @@ --- title: SpaceBio-Bench Portfolio Brief page_type: portfolio_brief -status: public_review_ready -last_reviewed: 2026-06-05 -claim_boundary: portfolio_brief_no_new_release_claim +status: public_ready +last_reviewed: 2026-06-16 --- # SpaceBio-Bench Portfolio Brief ## One-Sentence Summary -SpaceBio-Bench is a mission-held-out transcriptomics benchmark and transparency -package for public NASA OSDR space-biology data, designed to evaluate model -generalization under mission shift while keeping provenance, evaluation scope, -release readiness, and claim boundaries explicit. - -Public branch note: `main` gives the portfolio-facing entry point and includes -a curated v9 public bulk metadata-alpha evidence subset under `v9/`. Payload -matrices and draft extension lanes are outside this public-review path. +SpaceBio-Bench is a mission-held-out transcriptomics benchmark for public NASA +OSDR space-biology data, built to evaluate whether AI/ML and foundation-model +methods generalize biological signatures across missions. ## Public Artifact Links | Artifact | Link | Review purpose | |---|---|---| -| GitHub repository | [jang1563/GeneLab_benchmark](https://github.com/jang1563/GeneLab_benchmark) | Source tree, transparency cards, tests, release history | -| Hugging Face dataset | [jang1563/genelab-benchmark](https://huggingface.co/datasets/jang1563/genelab-benchmark) | Live dataset card, feature-matrix package, and public metadata | -| Transparency card pack | [SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md](SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md) | System, evaluation, release-readiness, and claim-boundary navigation | +| GitHub repository | [jang1563/GeneLab_benchmark](https://github.com/jang1563/GeneLab_benchmark) | Source tree, documentation, tests, and release history | +| Hugging Face dataset | [jang1563/genelab-benchmark](https://huggingface.co/datasets/jang1563/genelab-benchmark) | Public fold package and dataset card | +| Public documentation map | [SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md](SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md) | System, evaluation, release-status, and statement-guide navigation | ## What I Built - A cross-mission benchmark surface for public space-biology transcriptomics, - with canonical v1-v7 result documentation and a v7.1.2 - public-card/metadata patch over the v7.1 result surface. -- A curated v9 public bulk metadata-alpha evidence subset with 8 task - manifests, 33 fold definitions, 22 public OSDR source rows, 24 scaffold - baseline runs, and a draft Frictionless Data Package descriptor. -- A transparency card pack that separates system scope, evaluation - interpretation, release readiness, and allowed/blocked claims. -- A live Hugging Face dataset card for the public feature-matrix package, plus - a companion v9 metadata-alpha card that keeps payload-release boundaries - explicit. -- A claim register that prevents mixed-surface overclaiming across v1-v7 - results, the v8 translational extension, and v9 draft surfaces. + with canonical v1-v7 result documentation and a v7.1.2 public-card/metadata + patch. +- A live Hugging Face dataset card and processed public fold package for + selected leave-one-mission-out tasks. +- A v9 public bulk metadata catalog with 8 task manifests, 33 fold definitions, + 22 public OSDR source rows, 24 reference baseline runs, and a Frictionless + Data Package descriptor. +- A public documentation map that connects system scope, evaluation + interpretation, release status, and recommended public wording. ## Why It Matters -Space-biology transcriptomics is small-sample, mission-confounded, and easy to -overinterpret. A model can appear strong under a pooled score while failing on a -specific tissue, mission, or evaluation surface. SpaceBio-Bench treats that as -an evaluation-design problem, not just a modeling problem. +Space-biology transcriptomics is small-sample and mission-shifted. A model can +look strong in a pooled summary while failing on a specific tissue, mission, or +task variant. SpaceBio-Bench treats that as an evaluation-design problem: every +result is tied back to a task, fold, tissue, feature surface, and release +surface. The project demonstrates how to build a benchmark that is technically useful -and release-disciplined: tasks are explicit, source rows are traceable, baseline -runs are documented, and unsupported claims are blocked before they become -public release language. +and scientifically legible: tasks are explicit, source rows are traceable, +baseline runs are documented, and public language stays aligned with the +underlying release surface. ## Technical Depth | Area | Evidence | |---|---| | Benchmark design | Mission-held-out LOMO task manifests and fold definitions | -| Bioinformatics scope | Public OSDR-derived mouse bulk RNA-seq task scaffold | -| Evaluation engineering | Baseline predictions, metrics, run manifests, and pooled-summary caveats | -| Provenance | Source inventory plus OSDR API and checksum-manifest evidence | -| Packaging | Live Hugging Face dataset card, companion v9 metadata-alpha card, and draft Data Package descriptor | -| Release discipline | Metadata-alpha boundary, payload blocker tracking, and readiness tiers | -| Scientific communication | System card, evaluation card, release readiness card, and claim register | +| Bioinformatics scope | Public OSDR-derived mouse bulk RNA-seq tasks | +| Evaluation engineering | Baseline predictions, metrics, run records, and pooled-summary interpretation | +| Data access | Hugging Face public fold package plus GitHub documentation | +| Metadata packaging | v9 source inventory, checksum audit, and Data Package descriptor | +| Scientific communication | Public system card, evaluation card, release status card, and statement guide | -## Responsible Release Discipline +## Public Release Discipline -Current v9 public bulk status is intentionally limited: +The public release surface is intentionally separated into: -- It is a metadata-only alpha, not a frozen payload release. -- Payload-level hash verification remains pending. -- Baselines are scaffold anchors, not a state-of-the-art leaderboard. -- Current evidence does not support clinical, crew-health, countermeasure, - intervention, Mars-regime, or biological-mechanism claims. +- v7.1 canonical historical results; +- v7.1.2 documentation, card, citation, and metadata updates; +- the Hugging Face processed fold package; +- the v9 public bulk metadata catalog; +- development workspaces for future benchmark extensions. -This boundary is not a weakness of the project. It is part of the contribution: -the repository shows which claims are supported now, which claims are blocked, -and what evidence would be needed before stronger wording is appropriate. +That separation keeps benchmark scores, dataset access, metadata catalogs, and +extension work easy to inspect without flattening them into one headline. ## Role-Relevant Signal -This project is most directly relevant to roles involving: +This project is most relevant to roles involving: -- AI evaluation and benchmark design. -- Bioinformatics and space-biology data infrastructure. -- Scientific ML / foundation-model evaluation in biology. -- AI safety evaluation, especially claim-boundary and release-readiness work. -- Research engineering for reproducible, auditable scientific artifacts. -- Technical communication for responsible biological AI evaluation systems. +- AI evaluation and benchmark design; +- bioinformatics and space-biology data infrastructure; +- scientific ML and biological foundation-model evaluation; +- AI safety evaluation for biological-domain claims; +- research engineering for reproducible scientific artifacts; +- technical communication for biological AI benchmarks. ## Fast Review Path For a quick review, read these in order: -1. [Transparency Card Pack](SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md) +1. [Public Documentation Map](SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md) 2. [System Card](SPACEBIOBENCH_SYSTEM_CARD.md) 3. [Evaluation Card](SPACEBIOBENCH_EVALUATION_CARD.md) -4. [Release Readiness Card](SPACEBIOBENCH_RELEASE_READINESS_CARD.md) -5. [Claim Register](SPACEBIOBENCH_CLAIM_REGISTER.md) -6. [v9 Metadata-Alpha Dataset Card](v9_hf_dataset_card.md) +4. [Release Status Card](SPACEBIOBENCH_RELEASE_READINESS_CARD.md) +5. [Public Statement Guide](SPACEBIOBENCH_CLAIM_REGISTER.md) +6. [v9 Metadata Catalog Card](v9_hf_dataset_card.md) 7. [Canonical v7.1 Results](CANONICAL_RESULTS_V7_1.md) 8. [Live Hugging Face Dataset](https://huggingface.co/datasets/jang1563/genelab-benchmark) -For model-card or system-card writing roles, the strongest signal is the -combination of the system card, evaluation card, release readiness card, and -claim register: they show how the project separates intended use, evaluation -scope, readiness tier, and allowed versus blocked language. - ## Resume-Ready Summary -Built SpaceBio-Bench, a mission-held-out space-biology transcriptomics benchmark -and transparency package using public NASA OSDR data. The project defines -task/fold manifests, scaffold baselines, provenance and checksum evidence, -Hugging Face dataset-card documentation, release-readiness tiers, and a claim -register to prevent unsupported model, clinical, or payload-release claims. +Built SpaceBio-Bench, a mission-held-out space-biology transcriptomics +benchmark using public NASA OSDR data. The project defines task/fold manifests, +processed public fold packages, reference baselines, checksum-audit summaries, +Hugging Face dataset-card documentation, release metadata, and public +interpretation guides for evaluating biological AI methods under mission shift. ## Cover-Letter Variant SpaceBio-Bench reflects the kind of evaluation work I want to keep doing: building biological AI benchmarks that are technically useful, scientifically -careful, and release-disciplined. Beyond model scores, the project documents -task boundaries, source provenance, baseline interpretation, payload-readiness -blockers, and allowed versus unsupported claims. That combination is especially -important for biological AI evaluation, where overclaiming can happen quickly -when small datasets, mission confounding, and foundation-model comparisons are -compressed into a single headline number. +careful, and easy for other researchers to inspect. Beyond model scores, the +project documents task definitions, source coverage, baseline interpretation, +dataset access, and release status so that small datasets, mission confounding, +and foundation-model comparisons are not compressed into a single headline +number. diff --git a/docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md b/docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md index 69d5545..39ec5a5 100644 --- a/docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md +++ b/docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md @@ -1,171 +1,80 @@ --- -title: SpaceBio-Bench Release Readiness Card -page_type: release_readiness_card -status: public_review_ready -last_reviewed: 2026-06-05 -claim_boundary: benchmark_release_readiness_card_draft_no_release_approval +title: SpaceBio-Bench Release Status Card +page_type: release_status_card +status: public_ready +last_reviewed: 2026-06-16 --- -# SpaceBio-Bench Release Readiness Card +# SpaceBio-Bench Release Status Card -## Card Purpose +## Purpose -This card defines release tiers, readiness gates, blocked claims, and required -evidence for SpaceBio-Bench artifacts. It is designed to keep documentation, -metadata-alpha scaffolds, diagnostic lanes, frozen payload releases, and -archive-ready releases separate without making a stronger release claim than -the evidence supports. +This card summarizes the current public status of the SpaceBio-Bench release +surfaces and the additional work needed for larger dataset or archive releases. -This card does not approve a new release. It records the conditions a surface -must satisfy before its public wording can become stronger. +## Current Public Status -Branch note: on the default `main` branch, v9-specific evidence paths such as -`v9/...` and `docs/V9_*` refer to the curated public bulk metadata-alpha subset -included in this repository. Payload matrices and draft extension lanes are -excluded from the current public-review path. - -## Release Tiers - -| Tier | Meaning | Current examples | Public claim status | -|---|---|---|---| -| `historical_result_surface` | Existing result surface with canonical documentation | v1-v7 GeneLab Benchmark; v7.1 canonical result surface; v7.1.2 documentation/card/metadata patch | Allowed when using canonical v7.1 result language and v7.1.2 patch qualifiers | -| `metadata_alpha` | Public metadata, provenance, task, and baseline scaffold without frozen payload mirror | v9 public bulk alpha | Allowed only with metadata-only and no-payload qualifiers | -| `diagnostic_alpha` | Draft lane used for feasibility, metric, payload, or asset diagnostics | v9 single-cell, organoid, multispecies draft lanes | Draft-only; no public leaderboard or frozen release claim | -| `frozen_payload_release` | Future payload bundle with local mirror and verified payload hashes | Not yet active for v9 public bulk | Blocked until payload gates pass | -| `doi_archive_release` | Future citable archive release with final metadata, license, citations, and integrity manifest | Not yet active | Blocked until archive gates pass | -| `excluded` | Artifact or claim that should not enter public benchmark release | controlled human sequence data; unsupported countermeasure claims; Mars-regime predictions | Not releasable under current scope | +| Surface | Public status | Current wording | +|---|---|---| +| v7.1 GeneLab Benchmark | Canonical historical result surface | Use for v1-v7 results and citation context | +| v7.1.2 public patch | Active documentation and metadata patch | Use for README, HF card, citation, Zenodo, and release metadata updates | +| Hugging Face public fold package | Active processed dataset package | Use for selected LOMO fold downloads and public task access | +| v9 public bulk | Metadata catalog | Use for task manifests, source records, fold indexes, audit summaries, and reference baseline rows | +| Extension workspaces | Development workspaces | Keep separate from the v7.1 public result surface | -## Current Readiness State +## Active Release Checks -| Area | Status | Evidence | -|---|---|---| -| v7.1 canonical result surface | Active as public result source of truth | `docs/CANONICAL_RESULTS_V7_1.md` | -| v9 public bulk task registry | Pass for metadata alpha | `v9/task_manifest_index.csv` | -| v9 public bulk fold registry | Pass for metadata alpha | `v9/task_data_index.csv` | -| v9 public bulk source inventory | Pass for metadata alpha | `v9/source_inventory.csv` | -| OSDR checksum-manifest evidence | Pass for metadata alpha | `v9/source_checksum_audit.csv` | -| Payload-level local hash verification | Blocker for frozen payload release | `freeze_ready_source_count=0` in snapshot decision | -| Baseline output evidence | Pass for metadata alpha | `v9/reports/bulk_lomo_baseline_summary.csv` | -| Dataset-card alpha boundary | Applied | `docs/v9_hf_dataset_card.md`; `docs/V9_PUBLIC_BULK_ALPHA_CARD_DATAPACKAGE_BOUNDARY_UPDATE.md` | -| Draft Data Package alpha boundary | Applied | `v9/datapackage.draft.json`; `docs/V9_PUBLIC_BULK_ALPHA_CARD_DATAPACKAGE_BOUNDARY_UPDATE.md` | -| Paper archive metadata | Release-candidate ready | `CITATION.cff`; `.zenodo.json`; `docs/RELEASE_ARCHIVE_CARD.md`; `docs/RELEASE_ARCHIVE_MANIFEST.md` | - -## Metadata Alpha Requirements - -A `metadata_alpha` release surface may be described publicly only when: - -- Task manifests are generated and indexed. -- Fold definitions are indexed. -- Source inventory is present. -- Source provenance or checksum-manifest evidence is present. -- Baseline outputs, if reported, have predictions, metrics, and run manifests. -- Dataset card states metadata-only alpha status. -- Draft Data Package or equivalent descriptor states release status and payload - boundary. -- Payload-release blockers are explicit. -- Draft extension lanes are excluded from the core alpha scope unless their - own claim boundary is named. - -The current v9 public bulk lane satisfies this tier after -`V9-BULK-ALPHA-003`, with the boundary -`metadata_only_public_bulk_alpha_no_payload_release`. - -## Frozen Payload Release Requirements - -A future `frozen_payload_release` requires: - -- Local payload mirror for release-target fold matrices. -- Payload-level SHA-256 manifest for every distributed payload file. -- Verification report showing all expected payload hashes pass. -- Release `datapackage.json`, not only `datapackage.draft.json`. -- Dataset card language updated from metadata alpha to frozen payload release. -- License and reuse status reviewed. -- Dataset-specific OSDR citations filled from OSDR study pages. -- Regression checks for row counts, fold definitions, selected-gene counts, - source ids, and no local path leaks. - -Until these gates pass, avoid all frozen-payload wording. - -## DOI Or Archive Release Requirements - -A future `doi_archive_release` requires everything in -`frozen_payload_release`, plus: - -- Final release manifest. -- DataCite-aligned metadata: title, creators, contributors, version, - identifiers, related identifiers, resource type, license, and descriptions. -- RO-Crate or equivalent research-object metadata linking data, code, task - manifests, reports, and provenance. -- Archive deposit target selected, such as Zenodo or another approved - repository. -- Citation and acknowledgement text reviewed for OSDR and individual upstream - datasets. -- Changelog entry and maintenance policy. - -## Diagnostic Alpha Requirements - -`diagnostic_alpha` lanes may be useful internally or for reviewer discussion, -but each must keep its own boundary. A diagnostic lane should include: - -- A named claim boundary. -- Source inventory or asset inventory. -- Metric or payload contract if scores are discussed. -- Explicit statement that the lane is not a leaderboard. -- Explicit statement that the lane is not a frozen payload release. -- Owner action or blocker list for promotion. - -Examples include single-cell asset inventory, organoid diagnostic surfaces, and -multispecies feasibility or interaction-task scaffolds. - -## Excluded Or Blocked Claims - -Do not promote these claims under the current readiness state: - -| Claim | Reason | +| Area | Public file | |---|---| -| Frozen v9 public benchmark release | Payload verification has not passed | -| Frozen v9 payload mirror | Local payload mirror and hash manifest are pending | -| DOI/archive-ready release | Final metadata, license, citation, and archive manifest are pending | -| State-of-the-art leaderboard | Baselines are scaffold anchors, not tuned endpoints | -| Uniform foundation-model leaderboard | Current FM rows are mixed-surface or adapter-dependent | -| Countermeasure recommendation | Current evidence is not intervention validation | -| Crew-health or clinical decision support | Current scope excludes clinical recommendations | -| Mars-regime prediction | Current tasks are not validated for Mars-regime extrapolation | - -## Readiness Rules - -- Move a surface to a stronger tier only when its evidence files pass the tier - requirements. -- Preserve the old tier label in changelogs when a surface moves tiers. -- Do not use a stronger tier label in slide, paper, README, or dataset-card - text before the readiness card and claim register are updated. -- When a count changes, update the source artifact, dataset card, system card, - and claim register in the same edit batch. -- If an artifact is diagnostic-only, its filename or report header should say - so. - -## Correction Notes - -If a release-facing artifact overstates the current boundary: - -1. Correct the artifact. -2. Add a note in the relevant decision or readiness document. -3. Update the claim register if the allowed or blocked wording changed. -4. If external users may have seen the artifact, add a public correction note - in the release manifest or changelog for that surface. +| v7.1 result source | `docs/CANONICAL_RESULTS_V7_1.md` | +| v7.1 dataset card source | `docs/hf_dataset_card.md` | +| Citation metadata | `CITATION.cff`; `.zenodo.json` | +| Release manifest | `release/release_manifest.json` | +| v9 task registry | `v9/task_manifest_index.csv` | +| v9 fold registry | `v9/task_data_index.csv` | +| v9 source inventory | `v9/source_inventory.csv` | +| v9 checksum audit | `v9/source_checksum_audit.csv` | +| v9 baseline summaries | `v9/reports/bulk_lomo_baseline_summary.csv` | +| v9 metadata catalog card | `docs/v9_hf_dataset_card.md` | + +## For A Larger v9 Payload Release + +A larger v9 payload release would need: + +- release-target fold matrices packaged as a public payload bundle; +- payload-level SHA-256 manifest for distributed files; +- verification report for expected payload hashes; +- release `datapackage.json` or equivalent descriptor; +- reviewed license and source reuse language; +- dataset-specific OSDR citations; +- regression checks for row counts, fold definitions, selected-gene counts, + source IDs, and local-path hygiene. + +## For A DOI Or Archive Release + +A DOI or archive-oriented release would additionally need: + +- final release manifest; +- DataCite-aligned metadata; +- related identifiers for repository, dataset, and archive records; +- final resource type, license, contributors, and descriptions; +- research-object metadata such as RO-Crate when appropriate; +- archive deposit target such as Zenodo; +- reviewed citation and acknowledgment text. + +## Public Reporting Rules + +- Keep v7.1 result statements tied to the canonical v7.1 result document. +- Describe v7.1.2 as a documentation, card, citation, and metadata patch. +- Describe v9 public bulk as a metadata catalog. +- Pair pooled metrics with per-task or per-fold context. +- Update README, HF card, citation, release manifest, and public docs together + when release labels or counts change. ## Companion Documents -- `docs/SPACEBIOBENCH_SYSTEM_CARD.md` -- `docs/SPACEBIOBENCH_EVALUATION_CARD.md` -- `docs/SPACEBIOBENCH_CLAIM_REGISTER.md` -- `docs/RELEASE_ARCHIVE_CARD.md` -- `docs/RELEASE_ARCHIVE_MANIFEST.md` -- `docs/RELEASE_ARCHIVE_CHECKLIST.md` -- `docs/CANONICAL_RESULTS_V7_1.md` -- `docs/v9_hf_dataset_card.md` -- `docs/V9_PUBLIC_BULK_ALPHA_METADATA_SNAPSHOT_DECISION.md` -- `docs/V9_PUBLIC_BULK_ALPHA_CARD_DATAPACKAGE_BOUNDARY_UPDATE.md` -- `v9/reports/public_bulk_alpha_gap_matrix/public_bulk_alpha_gap_summary.csv` -- `v9/reports/public_bulk_alpha_snapshot_decision/snapshot_decision_summary.csv` -- `v9/datapackage.draft.json` +- [System card](SPACEBIOBENCH_SYSTEM_CARD.md) +- [Evaluation card](SPACEBIOBENCH_EVALUATION_CARD.md) +- [Public statement guide](SPACEBIOBENCH_CLAIM_REGISTER.md) +- [Canonical v7.1 results](CANONICAL_RESULTS_V7_1.md) +- [v9 metadata catalog card](v9_hf_dataset_card.md) diff --git a/docs/SPACEBIOBENCH_SYSTEM_CARD.md b/docs/SPACEBIOBENCH_SYSTEM_CARD.md index 98cae26..3ef3327 100644 --- a/docs/SPACEBIOBENCH_SYSTEM_CARD.md +++ b/docs/SPACEBIOBENCH_SYSTEM_CARD.md @@ -1,243 +1,106 @@ --- title: SpaceBio-Bench System Card page_type: system_card -status: public_review_ready -last_reviewed: 2026-06-05 -claim_boundary: benchmark_system_card_draft_no_new_release_claim +status: public_ready +last_reviewed: 2026-06-16 --- # SpaceBio-Bench System Card -## Card Purpose +## Purpose -This card documents SpaceBio-Bench as a benchmark system, not as a trained -model. It summarizes the project's data surfaces, task definitions, evaluation -logic, provenance controls, release boundaries, and known limitations. - -This card does not replace the Hugging Face dataset cards, task manifests, -release manifests, or model-specific documentation. Individual trained models -or adapters should receive separate model cards when they become release -artifacts. - -Branch note: on the default `main` branch, v9-specific evidence paths such as -`v9/...` and `docs/V9_*` refer to the curated public bulk metadata-alpha subset -included in this repository. Payload matrices and draft extension lanes are -excluded from the current public-review path. +This card describes SpaceBio-Bench as a benchmark system. It summarizes the +public result surfaces, task structure, data sources, evaluation logic, and +scope notes that help readers interpret the repository. ## System Summary -SpaceBio-Bench is a mission-held-out transcriptomics benchmark scaffold for -public space-biology data. It organizes public OSDR-derived expression tasks -into task manifests, fold definitions, baseline outputs, provenance metadata, -and release-boundary documents. +SpaceBio-Bench evaluates whether AI/ML and foundation-model methods generalize +spaceflight transcriptomic signatures across missions. The core design is +mission-held-out validation: train on one set of missions, hold out a mission, +and test whether the method recognizes the flight-vs-ground signal in the +held-out context. -The project currently has multiple surfaces with different maturity levels: +## Public Surfaces -| Surface | Current status | Primary boundary | +| Surface | Current public role | Entry point | |---|---|---| -| v1-v7 GeneLab Benchmark | Canonical historical result surface | Cross-mission mouse transcriptomics benchmark and result synthesis | -| v7.1.2 documentation/card/metadata patch | Documentation, public-card, and metadata patch over v7.1 | No new benchmark result generation | -| v8 SpaceMed | Incubating translational extension | Hypothesis and diagnostic work only; do not mix into v7.1 claims | -| v9 public bulk | Metadata-only alpha snapshot | Task/source/provenance scaffold; not a frozen payload release | -| v9 single-cell, organoid, multispecies | Draft diagnostic lanes | Asset, payload, metric, or feasibility scaffolds depending on lane | - -## System Boundary Map - -| Boundary layer | Evidence entering the layer | What the current card allows | What remains blocked | -|---|---|---|---| -| Source layer | Public NASA OSDR sources, source inventory rows, OSDR API evidence, checksum-manifest evidence | Public source/provenance claims with accession-level traceability | Claims about private, controlled, or non-public human sequence data | -| Task layer | Task manifests, fold definitions, held-out mission labels, feature namespaces | Mission-held-out benchmark task claims when task and fold ids are named | Treating mission labels as pure biology or operational readiness evidence | -| Result layer | Baseline runs, metric files, prediction rows, v7.1 canonical result summaries | Benchmark and workflow-evidence claims tied to the correct release surface | Mixed-surface leaderboard, model-superiority, or biological mechanism claims | -| Transparency layer | System card, evaluation card, release readiness card, and claim register | Allowed benchmark claims with explicit scope and caveats | Blocked clinical, crew-health, countermeasure, and Mars-regime claims | -| v9 metadata-alpha layer | Public bulk task/source/provenance scaffold and baseline anchors | Metadata-alpha and scaffold-baseline language | Frozen payload release claims until payload hashing and release gates pass | - -This map shows the boundary the cards enforce: benchmark evidence can support -task, fold, metric, provenance, and release-readiness claims, but it cannot -support clinical, crew-health, countermeasure, intervention, or Mars-regime -claims under the current release. - -## Intended Uses - -SpaceBio-Bench is intended for: - -- Evaluating method generalization under mission-held-out spaceflight or analog - transcriptomics shift. -- Comparing classical ML, gene-expression foundation-model adapters, and other - biological AI methods when inputs, tasks, and result surfaces are explicitly - matched. -- Stress-testing whether benchmark claims remain grounded in source accessions, - task manifests, fold definitions, and run manifests. -- Developing provenance-aware benchmark packaging for public OSDR-derived - biological data. -- Supporting reviewer-facing analysis where per-task results, claim boundaries, - and upstream data citations are visible. - -## What This Enables - -The card pack is meant to make the benchmark easier to inspect before a reader -trusts a headline number. It exposes what is being evaluated, which result -surface a claim belongs to, which files support that claim, and which release -conditions remain unfinished. - -## Out Of Scope - -SpaceBio-Bench should not be used to claim: - -- Astronaut health-risk prediction. -- Clinical or crew-health recommendations. -- Operational readiness for space missions. -- Countermeasure, intervention, or dosing recommendations. -- Mars-regime point predictions. -- A frozen v9 payload release before local payload mirroring and payload-level - hash verification are complete. -- A uniform foundation-model leaderboard when compared rows come from different - task subsets, tissues, adapters, or evaluation surfaces. +| v7.1 GeneLab Benchmark | Canonical historical result surface for v1-v7 | [CANONICAL_RESULTS_V7_1.md](CANONICAL_RESULTS_V7_1.md) | +| v7.1.2 public patch | Documentation, citation, metadata, and access update over v7.1 | [README.md](../README.md) | +| Hugging Face dataset | Processed public fold package for selected LOMO tasks | [hf_dataset_card.md](hf_dataset_card.md) | +| v9 public bulk | Metadata catalog for task manifests, source rows, fold indexes, audit summaries, and baseline outputs | [v9_hf_dataset_card.md](v9_hf_dataset_card.md) | +| Extension workspaces | Development areas for additional biological settings and modalities | versioned workspace docs under `docs/` and `v*/` | ## System Components -| Component | Role | Evidence | +| Component | Role | Public files | |---|---|---| -| Task manifests | Define benchmark tasks, folds, labels, sources, and metric ids | `v9/task_manifests/*.json`; `v9/task_manifest_index.csv` | -| Fold data index | Tracks train/test files, held-out missions, row counts, and gene counts | `v9/task_data_index.csv` | -| Source inventory | Maps OSDR accessions, GLDS prefixes, tissues, missions, and access status | `v9/source_inventory.csv` | -| Source checksum audit | Records OSDR API and checksum-manifest evidence | `v9/source_checksum_audit.csv`; `v9/source_checksum_audit.json` | -| Draft Data Package | Machine-readable descriptor for metadata and output resources | `v9/datapackage.draft.json` | -| Baseline outputs | Validate task/evaluation workflow, not model superiority | `v9/reports/bulk_lomo_baseline_summary.csv` and per-baseline reports | -| Dataset cards | Human-facing dataset and release-boundary summaries | `docs/hf_dataset_card.md`; `docs/v9_hf_dataset_card.md` | -| Canonical result doc | Public-facing v7.1 result and scope source of truth | `docs/CANONICAL_RESULTS_V7_1.md` | -| Evaluation card | Task, metric, baseline, and result-interpretation controls | `docs/SPACEBIOBENCH_EVALUATION_CARD.md` | -| Release readiness card | Release tiers, readiness gates, and blocked release claims | `docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md` | -| Claim register | Claim, support, confidence, and blocked wording control | `docs/SPACEBIOBENCH_CLAIM_REGISTER.md` | - -## Evaluation Philosophy +| Task manifests | Define tissues, missions, labels, feature namespaces, and metrics | `v9/task_manifests/*.json`; `v9/task_manifest_index.csv` | +| Fold data index | Tracks held-out missions, row counts, selected-gene counts, and fold paths | `v9/task_data_index.csv` | +| OSDR source inventory | Maps OSDR accessions, GLDS prefixes, tissues, missions, and access status | `v9/source_inventory.csv` | +| Checksum audit | Summarizes OSDR API and checksum-manifest parsing | `v9/source_checksum_audit.csv` | +| Data package descriptor | Lists metadata and output resources for the v9 catalog | `v9/datapackage.draft.json` | +| Baseline outputs | Provide reference workflow rows for the v9 catalog | `v9/reports/bulk_lomo_baseline_summary.csv` | +| Dataset cards | Explain public dataset and metadata-catalog surfaces | `docs/hf_dataset_card.md`; `docs/v9_hf_dataset_card.md` | +| Canonical result document | Summarizes v7.1 public result scope and headline results | `docs/CANONICAL_RESULTS_V7_1.md` | -SpaceBio-Bench treats mission shift as the main evaluation pressure. The core -public bulk task shape is leave-one-mission-out classification within a tissue -context: train on some missions, hold out one mission, and evaluate whether the -method generalizes to the held-out mission. - -Evaluation should be read at task and fold level before pooled summaries. -Pooled averages are useful for navigation, but they can hide tissue-specific, -mission-specific, or label-source-specific failure modes. - -Current v9 public bulk baselines are scaffold baselines. They validate the -benchmark workflow and provide comparison anchors, but they are not tuned -leaderboard endpoints and should not be used for biological superiority claims. - -## Data And Provenance Boundary +## Intended Uses -The v9 public bulk lane is currently a metadata-only alpha snapshot. It includes -task manifests, source inventory, OSDR checksum-manifest evidence, -alpha-boundary decision tables, baseline outputs, and a draft Frictionless Data -Package descriptor. +SpaceBio-Bench is intended for: -The current boundary is: +- evaluating generalization under mission-held-out transcriptomics shift; +- comparing methods on fixed task, fold, and feature definitions; +- auditing which public OSDR accessions support each task; +- reporting task-level and fold-level behavior before pooled summaries; +- developing public benchmark packaging for OSDR-derived biological data. -- OSDR API and checksum-manifest evidence exists for all 22 public bulk source - rows. -- The package does not include a frozen local payload mirror. -- Payload-level hash verification for every distributed fold matrix remains - pending. -- The descriptor is `v9/datapackage.draft.json`, not a release - `datapackage.json`. +## Scope Notes -The v9 dataset card records: +The public benchmark surfaces focus on task definitions, processed public fold +packages, baseline outputs, and result interpretation. They do not provide +clinical recommendations, crew-health guidance, countermeasure selection, +dosing guidance, or Mars-regime prediction. -- `spacebio_bench:release_status = metadata_alpha_not_frozen` -- `spacebio_bench:alpha_snapshot_status = metadata_only_alpha_snapshot` -- `spacebio_bench:claim_boundary = metadata_only_public_bulk_alpha_no_payload_release` -- `spacebio_bench:payload_release_allowed = false` -- `spacebio_bench:payload_verification_status = checksum_manifests_parsed_payloads_not_hashed` +The v9 public bulk surface is a metadata catalog. It is separate from archived +v9 fold-matrix payload bundles and from development workspaces for other +modalities. -## Results Boundary +## Evaluation Philosophy -The v7.1 canonical result surface is the current source of truth for public -v1-v7 scope accounting, headline result tables, foundation-model comparison -language, held-out validation, and submission-safe claim wording. +Mission shift is the main evaluation pressure. Pooled averages are useful for +orientation, but the benchmark should be read at the tissue, task, fold, and +mission level before drawing conclusions from aggregate metrics. -The v9 public bulk baseline results are a separate alpha scaffold surface. They -should be described as baseline workflow evidence, not as final model ranking or -biological mechanism evidence. +Foundation-model and adapter comparisons require matched task inputs, +adapters, tissues, and evaluation surfaces. The v7.1 result document records +the current public summary for these comparisons. ## Known Limitations - Mouse bulk RNA-seq is not a complete representation of space biology. -- Mission labels can conflate spaceflight exposure, vehicle, hardware, age, +- Mission labels can combine spaceflight exposure, vehicle, hardware, age, protocol, tissue handling, processing, and batch effects. -- Some task labels include analog or special mission labels that should not be - treated as interchangeable with ISS or deep-space exposure. +- Some task labels include analog or special mission labels such as MHU-2 and + OSD-397. - Bulk RNA-seq does not resolve cell-type-specific effects. -- Legacy processed fold matrices may reflect earlier preprocessing choices. -- Current v9 public bulk packaging is not payload-frozen. -- Foundation-model comparisons need adapter-specific validation and matched - evaluation surfaces. -- Strong benchmark scores do not prove biological mechanism, translational - validity, or operational deployment readiness. - -## Responsible Use - -Users should: - -- Cite OSDR and the individual upstream OSDR datasets used in a downstream - analysis. -- Report per-task and per-fold metrics alongside any pooled summary. -- Keep v7.1 result claims separate from v8 translational and v9 metadata-alpha - claims. -- State whether a result is from a canonical result surface, a metadata alpha, - a diagnostic lane, or a draft feasibility lane. -- Treat intervention, countermeasure, clinical, and crew-health claims as out - of scope unless a future release explicitly validates them. - -## Release And Integrity Controls - -Current controls: - -- Explicit claim-boundary tables in v9 public bulk alpha reports. -- Source inventory and source checksum audit artifacts. -- Draft Data Package metadata with release-status and payload-boundary fields. -- Canonical v7.1 result document to prevent mixed-surface claims. -- Human-facing dataset cards for v7.1 and v9 public bulk alpha. - -Readiness controls: - -- Maintain the release readiness card with release tiers such as - `metadata_alpha`, `diagnostic_alpha`, `frozen_payload_release`, and - `doi_release`. -- Maintain the evaluation card that separates task validity, leakage controls, - baseline status, and result interpretation. -- Add a payload-level SHA-256 manifest before any frozen v9 payload language. -- Add RO-Crate metadata for research-object citation and provenance. -- Add BagIt-style payload manifest checks if a distributable payload bundle is - created. +- Legacy processed fold matrices can reflect earlier preprocessing choices. +- Baseline rows are reference workflow comparisons rather than tuned leaderboard + endpoints. + +## Recommended Reporting + +- Cite NASA OSDR and the individual upstream OSDR datasets used in an analysis. +- Report per-task and per-fold metrics alongside pooled summaries. +- State the release surface used for each result. +- Use the public statement guide for concise release, dataset, and result + wording: [SPACEBIOBENCH_CLAIM_REGISTER.md](SPACEBIOBENCH_CLAIM_REGISTER.md). ## Companion Documents -Local evidence: - -- `docs/CANONICAL_RESULTS_V7_1.md` -- `docs/hf_dataset_card.md` -- `docs/v9_hf_dataset_card.md` -- `docs/V9_PUBLIC_BULK_ALPHA_METADATA_SNAPSHOT_DECISION.md` -- `docs/V9_PUBLIC_BULK_ALPHA_CARD_DATAPACKAGE_BOUNDARY_UPDATE.md` -- `docs/SPACEBIOBENCH_EVALUATION_CARD.md` -- `docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md` -- `docs/SPACEBIOBENCH_CLAIM_REGISTER.md` -- `v9/datapackage.draft.json` -- `v9/source_inventory.csv` -- `v9/source_checksum_audit.csv` -- `v9/task_manifest_index.csv` -- `v9/task_data_index.csv` - -External best-practice anchors: - -- Model Cards for Model Reporting: https://arxiv.org/abs/1810.03993 -- Hugging Face model cards: https://huggingface.co/docs/hub/main/model-cards -- Hugging Face dataset cards: https://huggingface.co/docs/datasets/v2.7.0/en/dataset_card -- Datasheets for Datasets: https://www.microsoft.com/en-us/research/uploads/prod/2019/01/1803.09010.pdf -- NIST AI RMF: https://www.nist.gov/itl/ai-risk-management-framework -- NASA OSDR FAQ: https://science.nasa.gov/reference/osdr-faq/ -- NASA OSDR Biological Data API: https://visualization.osdr.nasa.gov/biodata/api/ -- FAIR principles: https://www.nature.com/articles/sdata201618 -- RO-Crate: https://www.researchobject.org/ro-crate/technical_overview -- BagIt RFC 8493: https://www.rfc-editor.org/info/rfc8493/ -- DataCite Metadata Schema: https://schema.datacite.org/ +- [Public documentation map](SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md) +- [Evaluation card](SPACEBIOBENCH_EVALUATION_CARD.md) +- [Release status card](SPACEBIOBENCH_RELEASE_READINESS_CARD.md) +- [Public statement guide](SPACEBIOBENCH_CLAIM_REGISTER.md) +- [Canonical v7.1 results](CANONICAL_RESULTS_V7_1.md) +- [v7.1 Hugging Face dataset card](hf_dataset_card.md) +- [v9 metadata catalog card](v9_hf_dataset_card.md) diff --git a/docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md b/docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md index 7f2190e..0f75f1e 100644 --- a/docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md +++ b/docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md @@ -1,110 +1,62 @@ --- -title: SpaceBio-Bench Transparency Card Pack -page_type: transparency_card_pack -status: public_review_ready -last_reviewed: 2026-06-05 -claim_boundary: transparency_card_pack_no_new_release_claim +title: SpaceBio-Bench Public Documentation Map +page_type: public_documentation_map +status: public_ready +last_reviewed: 2026-06-16 --- -# SpaceBio-Bench Transparency Card Pack +# SpaceBio-Bench Public Documentation Map -This card pack is the public-facing map for SpaceBio-Bench scope, evaluation -interpretation, release readiness, and claim boundaries. It is meant to help a -reader understand what the benchmark currently supports before relying on a -headline score or release label. - -This pack does not introduce new benchmark results or approve a new release. - -Branch note: on the default `main` branch, this pack is the public entry -surface. The repository includes a curated v9 public bulk metadata-alpha -evidence subset under `v9/` and `docs/V9_*`. Payload matrices and draft -extension lanes remain outside this public-review path. +This map points readers to the main public documentation for SpaceBio-Bench: +system scope, evaluation interpretation, release status, and recommended public +wording. It is a navigation layer, not a new benchmark result surface. ## Current Public Summary -- v1-v7 / v7.1 is the canonical historical result surface for the original - GeneLab Benchmark. -- v7.1.2 is the public-card, metadata, and metadata patch over that canonical - result surface. It does not introduce new result generation. -- v8 is an incubating translational extension and should not be mixed into - v7.1 benchmark claims. -- v9 public bulk is a metadata-only alpha: task/source/provenance metadata and - scaffold baselines are present, but payload-level hash verification remains - pending. -- v9 single-cell, organoid, and multispecies lanes remain diagnostic or draft - surfaces unless a lane-specific release boundary says otherwise. -- Current scores should be read as benchmark evidence, not biological mechanism - proof, clinical guidance, countermeasure evidence, or mission-readiness - evidence. +- **v7.1 GeneLab Benchmark** is the canonical historical result surface for + v1-v7 results. +- **v7.1.2** is the public-card and metadata patch over v7.1. It updates + documentation, citation metadata, release metadata, and access guidance + without adding new result generation. +- **Hugging Face dataset** provides processed public fold packages for selected + mission-held-out tasks. +- **v9 public bulk** is a metadata catalog for public bulk RNA-seq task + manifests, source records, fold indexes, checksum-audit summaries, and + reference baselines. +- Extension workspaces for single-cell, organoid, multispecies, and + translational analyses remain separate from the v7.1 public result surface. ## Three-Minute Review Map -| Review step | Open this | What to verify | +| Step | Open this | What it gives you | |---|---|---| -| 1 | [Portfolio brief](SPACEBIOBENCH_PORTFOLIO_BRIEF.md) | Project contribution, role-relevant signal, and concise application summary | -| 2 | [System card](SPACEBIOBENCH_SYSTEM_CARD.md) | Benchmark scope, data surfaces, provenance boundary, and out-of-scope claims | -| 3 | [Evaluation card](SPACEBIOBENCH_EVALUATION_CARD.md) | Task, fold, metric, baseline, and pooled-summary interpretation | -| 4 | [Release readiness card](SPACEBIOBENCH_RELEASE_READINESS_CARD.md) | Release tier, evidence gates, and blockers for stronger public wording | -| 5 | [Claim register](SPACEBIOBENCH_CLAIM_REGISTER.md) | Allowed wording, blocked wording, support level, and future-only claims | -| Cross-check | [Canonical v7.1 results](CANONICAL_RESULTS_V7_1.md) and [v9 dataset card draft](v9_hf_dataset_card.md) | Whether a statement belongs to the canonical result surface, metadata-alpha scaffold, or a future release lane | - -The intended reading order is portfolio brief first, then the system card, -evaluation card, release readiness card, and claim register. This keeps the -project understandable without collapsing benchmark results, metadata-alpha -scaffolds, and unsupported translational claims into one headline. +| 1 | [Portfolio brief](SPACEBIOBENCH_PORTFOLIO_BRIEF.md) | One-page project framing and contribution summary | +| 2 | [System card](SPACEBIOBENCH_SYSTEM_CARD.md) | Benchmark surfaces, components, intended use, and scope notes | +| 3 | [Evaluation card](SPACEBIOBENCH_EVALUATION_CARD.md) | How to read tasks, folds, metrics, baselines, and pooled summaries | +| 4 | [Release status card](SPACEBIOBENCH_RELEASE_READINESS_CARD.md) | Current public status for v7.1, v7.1.2, HF, and v9 catalog surfaces | +| 5 | [Public statement guide](SPACEBIOBENCH_CLAIM_REGISTER.md) | Preferred wording for common public statements | +| Cross-check | [Canonical v7.1 results](CANONICAL_RESULTS_V7_1.md), [v7.1 HF card](hf_dataset_card.md), and [v9 metadata catalog card](v9_hf_dataset_card.md) | The source document for a result, dataset, or catalog statement | ## Card Index -| Card | What it answers | File | +| Card | Question | File | |---|---|---| | Portfolio Brief | Why does this project matter as a research and portfolio artifact? | [docs/SPACEBIOBENCH_PORTFOLIO_BRIEF.md](SPACEBIOBENCH_PORTFOLIO_BRIEF.md) | -| System Card | What is SpaceBio-Bench, what surfaces exist, and what is out of scope? | [docs/SPACEBIOBENCH_SYSTEM_CARD.md](SPACEBIOBENCH_SYSTEM_CARD.md) | +| System Card | What is SpaceBio-Bench and what surfaces exist? | [docs/SPACEBIOBENCH_SYSTEM_CARD.md](SPACEBIOBENCH_SYSTEM_CARD.md) | | Evaluation Card | How should task, fold, baseline, metric, and pooled results be interpreted? | [docs/SPACEBIOBENCH_EVALUATION_CARD.md](SPACEBIOBENCH_EVALUATION_CARD.md) | -| Release Readiness Card | Which release tier is each surface in, and what gates block stronger wording? | [docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md](SPACEBIOBENCH_RELEASE_READINESS_CARD.md) | -| Claim Register | Which claims are supported, blocked, or future-only? | [docs/SPACEBIOBENCH_CLAIM_REGISTER.md](SPACEBIOBENCH_CLAIM_REGISTER.md) | -| Release Archive Card | What would be included in a paper-supporting archive and what remains blocked? | [docs/RELEASE_ARCHIVE_CARD.md](RELEASE_ARCHIVE_CARD.md) | -| Release Archive Manifest | Which files and metadata surfaces form the archive candidate? | [docs/RELEASE_ARCHIVE_MANIFEST.md](RELEASE_ARCHIVE_MANIFEST.md) | -| Release Archive Checklist | Which final DOI, tag, citation, and checksum gates remain? | [docs/RELEASE_ARCHIVE_CHECKLIST.md](RELEASE_ARCHIVE_CHECKLIST.md) | -| v9 Metadata-Alpha Dataset Card | What does the v9 public bulk metadata alpha contain? | [docs/v9_hf_dataset_card.md](v9_hf_dataset_card.md) | +| Release Status Card | What is active now, and what would be needed for a larger release? | [docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md](SPACEBIOBENCH_RELEASE_READINESS_CARD.md) | +| Public Statement Guide | What wording should be used for public descriptions? | [docs/SPACEBIOBENCH_CLAIM_REGISTER.md](SPACEBIOBENCH_CLAIM_REGISTER.md) | | v7.1 Canonical Results | What is the locked public result and scope source for v1-v7? | [docs/CANONICAL_RESULTS_V7_1.md](CANONICAL_RESULTS_V7_1.md) | - -## Publication Surface Guidance - -For GitHub: - -- Link this pack from the root README and v9 README. -- Keep the four cards in `docs/`. -- Present the pack as transparency documentation, not as a new release. -- On `main`, treat v9-specific evidence paths as the curated public bulk - metadata-alpha subset only; do not infer payload or extension-lane release. - -For Hugging Face: - -- Keep the dataset card concise. -- Add a short "Transparency and Release Boundary" section linking back to this - GitHub card pack. -- Do not paste the full card pack into the Hugging Face README unless the full - docs directory is also published there. - -## Current Readiness Labels - -| Surface | Recommended label | Stronger wording currently blocked | -|---|---|---| -| v1-v7 / v7.1 | canonical historical result surface | New v7.1 result generation | -| v8 SpaceMed | incubating translational extension | Countermeasure or intervention recommendation | -| v9 public bulk | metadata-only public bulk alpha | Frozen payload release; DOI/archive-ready release | -| v9 extension lanes | diagnostic or draft lane | Public leaderboard; frozen payload release | - -## Minimum Public-Update Checklist - -- Root README links to this card pack. -- Root README links to the portfolio brief. -- v9 README links to this card pack. -- v9 Hugging Face-style dataset card has a concise transparency section. -- `SPACEBIOBENCH_RELEASE_READINESS_CARD.md` is used as the release-readiness - surface. -- `RELEASE_ARCHIVE_CARD.md`, `RELEASE_ARCHIVE_MANIFEST.md`, and - `RELEASE_ARCHIVE_CHECKLIST.md` are used before DOI/tag deposition. -- No public-facing text claims a frozen v9 payload release, DOI/archive release, - state-of-the-art leaderboard, clinical use, crew-health guidance, - countermeasure recommendation, or Mars-regime prediction. +| v7.1 HF Dataset Card | What is in the public processed fold package? | [docs/hf_dataset_card.md](hf_dataset_card.md) | +| v9 Metadata Catalog Card | What does the v9 public bulk metadata catalog contain? | [docs/v9_hf_dataset_card.md](v9_hf_dataset_card.md) | + +## Public Update Checklist + +- Keep the root README, Hugging Face card, citation metadata, and release + manifest aligned on release labels and titles. +- Keep v7.1 result statements tied to + [CANONICAL_RESULTS_V7_1.md](CANONICAL_RESULTS_V7_1.md). +- Keep v9 catalog statements tied to + [v9_hf_dataset_card.md](v9_hf_dataset_card.md) and the files under `v9/`. +- Report per-task and per-fold metrics alongside pooled summaries. +- Cite NASA OSDR and the individual OSDR datasets used in downstream analyses. diff --git a/scripts/validate_public_docs_consistency.py b/scripts/validate_public_docs_consistency.py index 2d4159b..3cfa830 100644 --- a/scripts/validate_public_docs_consistency.py +++ b/scripts/validate_public_docs_consistency.py @@ -15,6 +15,12 @@ README = REPO_ROOT / "README.md" HF_CARD = REPO_ROOT / "docs" / "hf_dataset_card.md" V9_HF_CARD = REPO_ROOT / "docs" / "v9_hf_dataset_card.md" +DOC_MAP = REPO_ROOT / "docs" / "SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md" +PORTFOLIO_BRIEF = REPO_ROOT / "docs" / "SPACEBIOBENCH_PORTFOLIO_BRIEF.md" +SYSTEM_CARD = REPO_ROOT / "docs" / "SPACEBIOBENCH_SYSTEM_CARD.md" +EVALUATION_CARD = REPO_ROOT / "docs" / "SPACEBIOBENCH_EVALUATION_CARD.md" +RELEASE_STATUS_CARD = REPO_ROOT / "docs" / "SPACEBIOBENCH_RELEASE_READINESS_CARD.md" +STATEMENT_GUIDE = REPO_ROOT / "docs" / "SPACEBIOBENCH_CLAIM_REGISTER.md" V9_README = REPO_ROOT / "v9" / "README.md" V9_REPORTS_README = REPO_ROOT / "v9" / "reports" / "README.md" CITATION = REPO_ROOT / "CITATION.cff" @@ -65,6 +71,12 @@ def validate_public_docs() -> list[str]: readme = README.read_text() hf_card = HF_CARD.read_text() v9_hf_card = V9_HF_CARD.read_text() + doc_map = DOC_MAP.read_text() + portfolio_brief = PORTFOLIO_BRIEF.read_text() + system_card = SYSTEM_CARD.read_text() + evaluation_card = EVALUATION_CARD.read_text() + release_status_card = RELEASE_STATUS_CARD.read_text() + statement_guide = STATEMENT_GUIDE.read_text() v9_readme = V9_README.read_text() v9_reports_readme = V9_REPORTS_README.read_text() citation = CITATION.read_text() @@ -97,6 +109,9 @@ def validate_public_docs() -> list[str]: require_contains(errors, "README.md", readme, "CONTRIBUTING.md") require_contains(errors, "README.md", readme, "docs/submission_format.md") require_contains(errors, "README.md", readme, "SpaceBio-Bench / GeneLab Benchmark: Mission-Held-Out") + require_contains(errors, "README.md", readme, "public documentation map") + require_contains(errors, "README.md", readme, "Release status") + require_contains(errors, "README.md", readme, "Public statement guide") require_absent(errors, "README.md", readme, "Version: v7.0 (2026-04-12)") require_absent(errors, "README.md", readme, "Status: **v1–v7 Complete**") @@ -171,6 +186,60 @@ def validate_public_docs() -> list[str]: ): require_absent(errors, "v9 public docs", v9_public_text, forbidden) + public_card_text = "\n".join( + [ + doc_map, + portfolio_brief, + system_card, + evaluation_card, + release_status_card, + statement_guide, + ] + ) + for label, text, expected in ( + ( + "docs/SPACEBIOBENCH_TRANSPARENCY_CARD_PACK.md", + doc_map, + "# SpaceBio-Bench Public Documentation Map", + ), + ( + "docs/SPACEBIOBENCH_PORTFOLIO_BRIEF.md", + portfolio_brief, + "# SpaceBio-Bench Portfolio Brief", + ), + ("docs/SPACEBIOBENCH_SYSTEM_CARD.md", system_card, "# SpaceBio-Bench System Card"), + ( + "docs/SPACEBIOBENCH_EVALUATION_CARD.md", + evaluation_card, + "# SpaceBio-Bench Evaluation Card", + ), + ( + "docs/SPACEBIOBENCH_RELEASE_READINESS_CARD.md", + release_status_card, + "# SpaceBio-Bench Release Status Card", + ), + ( + "docs/SPACEBIOBENCH_CLAIM_REGISTER.md", + statement_guide, + "# SpaceBio-Bench Public Statement Guide", + ), + ): + require_contains(errors, label, text, expected) + for forbidden in ( + "metadata-alpha", + "metadata alpha", + "Public Bulk Metadata Alpha", + "claim boundary", + "blocked wording", + "future-only", + "provenance boundary", + "Provenance And Integrity", + "release-readiness blockers", + "should not be used", + "not a frozen", + ): + require_absent(errors, "linked public cards", public_card_text, forbidden) + zenodo_text = json.dumps(zenodo, sort_keys=True) require_contains(errors, ".zenodo.json", zenodo.get("title", ""), "SpaceBio-Bench") require_contains(errors, ".zenodo.json", zenodo.get("version", ""), "7.1.2")