diff --git a/.github/workflows/request-nvskills-ci.yml b/.github/workflows/request-nvskills-ci.yml new file mode 100644 index 0000000..a88a057 --- /dev/null +++ b/.github/workflows/request-nvskills-ci.yml @@ -0,0 +1,26 @@ +name: Request NVSkills CI + +on: + issue_comment: + types: [created] + pull_request: + types: [opened, reopened, synchronize, ready_for_review] + push: + +jobs: + request: + if: > + github.event_name == 'pull_request' || + (github.event_name == 'issue_comment' && + github.event.issue.pull_request && + startsWith(github.event.comment.body, '/nvskills-ci')) || + (github.event_name == 'push' && + github.actor == (vars.NVSKILLS_SIGNATURE_PUSH_ACTOR || 'nv-skills-ci[bot]') && + startsWith(github.event.head_commit.message, vars.NVSKILLS_SIGNATURE_COMMIT_TITLE || 'Attach NVSkills validation signatures')) + permissions: + contents: read + pull-requests: read + statuses: read + uses: NVIDIA/skills/.github/workflows/team-request.yml@main + secrets: + NVSKILLS_CI_DISPATCH_TOKEN: ${{ secrets.NVSKILLS_CI_DISPATCH_TOKEN }} \ No newline at end of file diff --git a/requirements.txt b/requirements.txt index 6ce6974..9c9bcad 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,3 +1,6 @@ +# xFormers CUDA wheels are published on the PyTorch index. +--extra-index-url https://download.pytorch.org/whl/cu124 + # --------- pytorch --------- # torch==2.5.1 torchvision==0.20.1 @@ -22,9 +25,10 @@ pre-commit==4.0.1 # hooks for applying linters on commit rich==13.9.4 # beautiful text formatting in terminal pytest==8.1.1 # tests sh==2.2.2 # for running bash commands in some tests (linux/macos only) +python-dotenv==1.0.1 transformers==4.54.1 polars==1.12.0 -xformers==0.0.28.post3 --index-url https://download.pytorch.org/whl/cu124 +xformers==0.0.28.post3 ninja==1.11.1.1 einops==0.8.0 ipython-autotime==0.3.2 diff --git a/skills/codonfm-embed/BENCHMARK.md b/skills/codonfm-embed/BENCHMARK.md new file mode 100644 index 0000000..2764072 --- /dev/null +++ b/skills/codonfm-embed/BENCHMARK.md @@ -0,0 +1,122 @@ +# Skill Benchmark: codonfm-embed + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `codonfm-embed` +- Evaluation date: 2026-10-07 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 4 evaluation tasks (4 positive) +- Dataset digest: `sha256:8ef619d6a9074d8b7fdc6224f220bb79aad6e37e2222f3290580290b7a6afbeb` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 1 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 96.1% — baseline ran, but no comparable score was available; uplift unavailable | 93.4% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 75.0% → 100.0% (+25.0 points) | 75.0% → 100.0% (+25.0 points) | +| Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Discoverability | 96.3% — baseline ran, but no comparable score was available; uplift unavailable | 86.3% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 85.6% → 90.6% (+5.0 points) | 98.8% → 93.8% (-5.0 points) | +| Efficiency | 93.6% — baseline ran, but no comparable score was available; uplift unavailable | 86.9% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 1,152,349 | 5,695,914 | -4,543,565 | -79.77% | skill 4/4; base 4/4 | +| claude-code | codonfm-embed-001 | 544,672 | 1,934,434 | -1,389,762 | -71.84% | skill 1/1; base 1/1 | +| claude-code | codonfm-embed-002 | 243,222 | 950,364 | -707,142 | -74.41% | skill 1/1; base 1/1 | +| claude-code | codonfm-embed-003 | 265,873 | 2,038,952 | -1,773,079 | -86.96% | skill 1/1; base 1/1 | +| claude-code | codonfm-embed-004 | 98,582 | 772,164 | -673,582 | -87.23% | skill 1/1; base 1/1 | +| codex | All cases | 487,427 | 1,838,975 | -1,351,548 | -73.49% | skill 4/4; base 4/4 | +| codex | codonfm-embed-001 | 106,660 | 518,881 | -412,221 | -79.44% | skill 1/1; base 1/1 | +| codex | codonfm-embed-002 | 159,116 | 322,872 | -163,756 | -50.72% | skill 1/1; base 1/1 | +| codex | codonfm-embed-003 | 157,215 | 924,873 | -767,658 | -83.00% | skill 1/1; base 1/1 | +| codex | codonfm-embed-004 | 64,436 | 72,349 | -7,913 | -10.94% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 1,639,776 | 7,534,889 | -5,895,113 | -78.24% | skill 8/8; base 8/8 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 3 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 4 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/codonfm-embed/SKILL.md`) +- **MEDIUM** SECURITY/Unknown (LP3): MCP Least Privilege: Without declared permissions the skill's intent is opaque and cannot be validated. (`SKILL.md:1`) +- **LOW** SCRIPT_LINT/magic_numbers: validate_inputs.py contains magic numbers (`skills/codonfm-embed/scripts/validate_inputs.py`) + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/codonfm-embed/SKILL.md b/skills/codonfm-embed/SKILL.md new file mode 100644 index 0000000..10e0035 --- /dev/null +++ b/skills/codonfm-embed/SKILL.md @@ -0,0 +1,181 @@ +--- +name: codonfm-embed +description: Validate coding-sequence CSVs, extract public CodonFM/Encodon embeddings, and choose checkpoints for downstream property modeling. +license: Apache-2.0 +metadata: + author: "NVIDIA BioNeMo " + tags: [biology, codonfm, embeddings] +--- + +# Extract public Encodon embeddings + +## Purpose + +Extract one frozen CLS vector per coding sequence with public Encodon v1. +Support input validation, command preparation, extraction, and checkpoint +selection for translation efficiency, expression, or mRNA stability modeling. +Extraction does not automatically train a downstream regressor. + +## Prerequisites + +- Validation needs Python 3 standard library only; no GPU, weights, or API key. +- Execution needs the public CodonFM checkout, its `requirements.txt` environment, + a compatible NVIDIA GPU, and local checkpoint weights. A metadata JSON is not + a checkpoint. A `.safetensors` file needs its sibling `config.json`; `.ckpt` + checkpoints are also supported by the public loader. +- Run `python -m src.runner` from the CodonFM repository root. In an isolated + workspace, use supplied source artifacts; source paths below are relative to + that checkout or source archive, not this skill directory. + +## Inputs + +Input source precedence: explicit user prompt arguments, then supplied +files/checkpoint metadata, then inspected public runner defaults. Resolve +conflicting model names and checkpoint metadata before execution. Supplied 80M +metadata is useful for preparing an 80M command; it does not restrict an +open-ended recommendation to that size. + +Required for validation: a CSV. Required for extraction: the CSV, checkpoint, +matching model name, and output directory. Optional: context length and batch +size overrides. Checkpoint-selection questions can be answered without a CSV. + +| Input | Requirement or default | +| --- | --- | +| Sequence CSV | Columns `id`, `ref_seq`, `value`, `split`; extra columns allowed | +| `id` | Nonblank, unique IDs for unambiguous output association | +| `ref_seq` | Coding sequence, uppercase DNA `A/C/G/T`, length divisible by three; public dataset converts uppercase `U` to `T` | +| `value` | Numeric label; use `0.0` for new extraction-only data, preserve supplied labels | +| `split` | Only exact `test` values enter extraction; blank/other values are excluded | +| Checkpoint and model | Match weights/config to `encodon_80m`, `encodon_600m`, or `encodon_1b` | +| Context length | Public runner default `2048` tokens, including CLS and SEP | +| Output directory | A fresh run directory with an empty predictions directory | + +## Instructions + +1. **Choose the requested workflow.** For a checkpoint/performance question, + read [checkpoint selection](references/checkpoint-selection.md) and answer + from public benchmark evidence. For the strongest published downstream + results, prefer the public **1B random-mask checkpoint** when resources allow; + 80M is a demonstration or resource-constrained choice. A small labeled set + alone does not establish that 80M frozen features are better. Do not download + weights or inspect the entire source tree just to make a recommendation. +2. **Inspect supplied source only where needed.** Confirm runner/config, + `src/data/codon_bert_dataset.py`, `src/data/preprocess/codon_sequence.py`, + `src/inference/encodon.py`, or `src/utils/pred_writer.py` for the relevant + behavior. Read ZIP members with `zipfile.ZipFile.namelist()` and `.read()`; + source inspection does not need extraction. If a checkout is needed, use a + new directory from `tempfile.mkdtemp()` or `mktemp -d`, without deleting or + overwriting an existing directory. For Decodon support questions, inspect + runner/config and model/inference modules, cite the inspected files, explain + the missing public implementation, and finish there. +3. **Validate the CSV before running extraction.** Run the bundled checker + below with the intended context length. Report per-row verdicts using CSV + row numbers as well as IDs, since IDs can repeat. Separate excluded rows, + invalid inputs, duplicate-ID warnings, and truncation. Propose fixes without + silently rewriting supplied data. The checker is a preflight, not model + execution or proof of biological CDS validity. +4. **Deliver the requested preparation or execution.** For preparation, return + a complete command with resolved paths (or clearly identified prerequisites), + the test-row count, validation findings, and the output contract below. + Include all task/dataset/process flags in the final answer, even if already + shown in a tool call. For extraction, reuse/download the chosen checkpoint + when needed, execute once resources are ready, and verify the saved arrays. + If resources are missing, finish preparation and state what is missing. + +## Available Scripts + +| Script | Purpose | Arguments | +| --- | --- | --- | +| [validate_inputs.py](scripts/validate_inputs.py) | Read-only CSV validation and per-row verdicts | Required CSV path; optional `--context-length` (default `2048`) | + +Run the preflight with Python; `CODONFM_SKILL_DIR` is the directory containing this file: + +```bash +python "$CODONFM_SKILL_DIR/scripts/validate_inputs.py" "$CODONFM_DATA_PATH" \ + --context-length 2048 +``` + +The checker prints JSON. Exit `0` means no findings, `1` means row findings to +review (including exclusions/warnings), and `2` means a file/schema error. +Neither warnings nor exclusions imply that the public runner will crash. + +## Output Format + +The checker emits JSON with `total_rows`, `test_rows`, `excluded_rows`, +`context_length`, `codon_limit`, `warnings`, and `rows`. Each row records its +one-based data-row number (excluding the header), ID, split, verdict, issues, +sequence/value validity, and retained/lost codons. A file/schema error emits +`error` and `csv` instead. These are preflight findings, not generated embeddings. + +## Examples + +Set `CODONFM_DATA_PATH` to the CSV, `CODONFM_CHECKPOINT_PATH` to the weights, +`CODONFM_MODEL_NAME` to the matching architecture, and `CODONFM_RUN_DIR` to a +fresh output directory. Substitute actual paths in a prepared command: + +```bash +python -m src.runner eval \ + --task_type embedding_prediction \ + --process_item codon_sequence \ + --dataset_name CodonBertDataset \ + --exp_name embed_extract \ + --model_name "$CODONFM_MODEL_NAME" \ + --checkpoint_path "$CODONFM_CHECKPOINT_PATH" \ + --data_path "$CODONFM_DATA_PATH" \ + --context_length 2048 \ + --num_nodes 1 \ + --num_gpus 1 \ + --num_workers 0 \ + --val_batch_size 2 \ + --out_dir "$CODONFM_RUN_DIR" \ + --predictions_output_dir "$CODONFM_RUN_DIR/predictions" +``` + +For a low-cost demonstration, `encodon_80m` matches +`nvidia/NV-CodonFM-Encodon-80M-v1`, revision +`399ca9fe17b57941a7bebc6788033919b417413c`, file +`NV-CodonFM-Encodon-80M-v1.safetensors` and sibling `config.json`. + +## Outputs + +- Under `--predictions_output_dir`, `embeddings_merged.npy` contains frozen + final-layer CLS vectors, shape `(processed_rows, hidden_size)`. +- `ids_merged.npy` is index-aligned: embedding row `i` belongs to ID row `i`. + Use these IDs to join to the CSV; do not assume every CSV row was retained. + Duplicate IDs make that join ambiguous even when extraction succeeds. +- For the one-GPU example, verify both arrays have the expected test-row count, + embeddings are finite, and width matches checkpoint config (`1024` for 80M, + `2048` for 600M/1B). Do not fabricate arrays for a preparation-only request. + +The public checkout's downstream-model references are: + +- `notebooks/4-EnCodon-Downstream-Task-riboNN.ipynb` +- `notebooks/5-EnCodon-Downstream-Task-mRFP-expression.ipynb` +- `notebooks/6-EnCodon-Downstream-Task-mRNA-stability.ipynb` + +## Limitations + +- Public v1 has no Decodon model/inference implementation, Decodon notebooks, + `notebooks/te_predictor.py`, or `notebooks/mfe_predictor.py`. +- At context length `2048`, retain the first `2046` codons; any remaining + 3-prime sequence is lost. Increasing the flag does not validate a longer + context. Disclose deliberate cropping or a separate chunking/aggregation + strategy; neither is equivalent to embedding the complete sequence once. +- `--dryrun` builds runtime configuration, may create directories, and needs + ML dependencies; it reads neither the CSV nor the weights and is not input + validation. +- Do not claim a benchmark-trained regressor generalizes to a new organism, + cell type, or assay without new labeled validation data. +- Do not invoke this skill for a generic expression-prediction request that + does not mention CodonFM or Encodon. + +## Troubleshooting + +| Symptom | Cause and action | +| --- | --- | +| Missing `split` column | Eval requests the test split despite the dataset docstring calling this column optional; add an explicit split column to a corrected copy | +| Fewer output rows | Blank/non-`test` split values are silently filtered; set intended extraction rows to exact `test` in a corrected copy | +| Repeated output IDs | Duplicate input IDs are not rejected; assign unique IDs while preserving a mapping to the original rows | +| Oversized sequence | Preprocessing truncates at `context_length - 2` codons; report retained/lost lengths and agree on a sequence-handling strategy | +| Missing weights or dependencies | Complete validation/command preparation; metadata and `--dryrun` do not substitute for weights | +| Merge failure on a repeated run | The writer scans `.npy` files; use a fresh predictions directory to avoid stale shards or merged arrays | diff --git a/skills/codonfm-embed/agents/openai.yaml b/skills/codonfm-embed/agents/openai.yaml new file mode 100644 index 0000000..d3cb6e8 --- /dev/null +++ b/skills/codonfm-embed/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "CodonFM Embeddings" + short_description: "Extract public Encodon sequence embeddings" + default_prompt: "Use $codonfm-embed to extract Encodon embeddings from my coding-sequence CSV." diff --git a/skills/codonfm-embed/evals/evals.json b/skills/codonfm-embed/evals/evals.json new file mode 100644 index 0000000..dda3586 --- /dev/null +++ b/skills/codonfm-embed/evals/evals.json @@ -0,0 +1,73 @@ +{ + "skill_name": "codonfm-embed", + "evals": [ + { + "id": "codonfm-embed-001", + "prompt": "Validate the supplied sequences.csv and prepare a public Encodon embedding-extraction command. Explain which rows will be processed and how to associate the output embeddings with sequence IDs. Use the supplied public source and checkpoint metadata.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json", + "files/sequences.csv" + ], + "expected_output": "Two validated test rows and a public embedding_prediction command with the correct embedding/ID output contract.", + "assertions": [ + "The command uses embedding_prediction, codon_sequence, and CodonBertDataset", + "The command specifies --checkpoint_path and output/prediction paths appropriate to the chosen working directory", + "The agent validates both sequence rows, including value and split=test, and explains that evaluation processes the test split", + "The response identifies embeddings_merged.npy and ids_merged.npy and explains their row alignment without fabricating embeddings" + ], + "expected_skill": "codonfm-embed", + "expected_script": null + }, + { + "id": "codonfm-embed-002", + "prompt": "Does the public CodonFM implementation support extracting Decodon embeddings? Check the supplied source and explain the limitation, if any.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json" + ], + "expected_output": "The agent identifies the absence of a public Decodon model and inference implementation.", + "assertions": [ + "The agent explains that the supplied public source has no Decodon model/inference implementation and cites inspected files", + "The agent does not invent a Decodon command or attempt to implement the missing model" + ], + "expected_skill": "codonfm-embed", + "expected_script": null + }, + { + "id": "codonfm-embed-003", + "prompt": "Validate sequences_edge_cases.csv before running Encodon embedding extraction. Some rows look unusual — confirm whether each will process successfully, and if not, say exactly what happens and how to fix it.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json", + "files/sequences_edge_cases.csv" + ], + "expected_output": "A per-row verdict: row 1 (row_normal) processes normally; row 2 (row_blank_split) is silently excluded from the test split, not an error, and should be set to split=test to be included; row 3 shares its id with row 1, which the agent should flag as a duplicate that will make ids_merged.npy ambiguous; row 4 (row_oversized, 2050 codons) exceeds the context-length codon limit and will be truncated rather than embedded in full.", + "assertions": [ + "The agent correctly identifies that the row with a blank split value will be silently excluded from extraction because the eval path filters strictly on split=='test', not treated as an error or crash", + "The agent flags the duplicate id across two rows and explains the consequence for mapping ids_merged.npy back to source rows", + "The agent identifies that the oversized sequence exceeds context_length - 2 codons and will be truncated rather than rejected, and explains what information is lost", + "The agent does not claim all four rows will process identically, and does not fabricate output values" + ], + "expected_skill": "codonfm-embed", + "expected_script": null + }, + { + "id": "codonfm-embed-004", + "prompt": "We want the most reliable embeddings for training a translation-efficiency regressor on a small (~200-sequence) labeled set. Which public Encodon checkpoint should we use, and what correlation with ground truth should we expect? Justify the choice.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json" + ], + "expected_output": "A recommendation for the largest available public checkpoint (1B, or 1B-Cdwt for codon-frequency-weighted masking), citing the CodonFM preprint's reported translation-efficiency correlation advantage over smaller checkpoints, with an explicit caveat that a published benchmark correlation is not a performance guarantee on this specific 200-sequence dataset.", + "assertions": [ + "The agent recommends a specific public checkpoint size rather than declining to answer", + "The agent cites the preprint/published benchmarks as the source of its performance claim, not the supplied source code (which contains no benchmark numbers at all)", + "The agent distinguishes the published benchmark result from a guarantee on the user's own small labeled set", + "The agent does not claim access to internal-only benchmark numbers beyond what the public preprint documents" + ], + "expected_skill": "codonfm-embed", + "expected_script": null + } + ] +} diff --git a/skills/codonfm-embed/evals/files/codonfm_source.zip b/skills/codonfm-embed/evals/files/codonfm_source.zip new file mode 100644 index 0000000..eb76611 Binary files /dev/null and b/skills/codonfm-embed/evals/files/codonfm_source.zip differ diff --git a/skills/codonfm-embed/evals/files/encodon_checkpoint.json b/skills/codonfm-embed/evals/files/encodon_checkpoint.json new file mode 100644 index 0000000..6c6230d --- /dev/null +++ b/skills/codonfm-embed/evals/files/encodon_checkpoint.json @@ -0,0 +1,33 @@ +{ + "repo_id": "nvidia/NV-CodonFM-Encodon-80M-v1", + "revision": "399ca9fe17b57941a7bebc6788033919b417413c", + "model_name": "encodon_80m", + "filename": "NV-CodonFM-Encodon-80M-v1.safetensors", + "size_bytes": 307351588, + "config_filename": "config.json", + "config": { + "vocab_size": 69, + "hidden_size": 1024, + "num_hidden_layers": 6, + "num_attention_heads": 8, + "intermediate_size": 4096, + "hidden_act": "gelu", + "hidden_dropout_prob": 0.1, + "attention_probs_dropout_prob": 0.1, + "initializer_range": 0.02, + "layer_norm_eps": 1e-12, + "pad_token_id": 3, + "position_embedding_type": "rotary", + "classifier_dropout": 0.1, + "rotary_theta": 10000.0, + "ignore_index": -100, + "loss_type": "cross_entropy", + "lora": false, + "lora_alpha": 32.0, + "lora_r": 16, + "lora_dropout": 0.1, + "finetune_strategy": "full" + }, + "source_url": "https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1/tree/399ca9fe17b57941a7bebc6788033919b417413c", + "weights_included": false +} diff --git a/skills/codonfm-embed/evals/files/sequences.csv b/skills/codonfm-embed/evals/files/sequences.csv new file mode 100644 index 0000000..42af3f8 --- /dev/null +++ b/skills/codonfm-embed/evals/files/sequences.csv @@ -0,0 +1,3 @@ +id,ref_seq,value,split +example_1,ATGGCTGAATTTCCGTAA,0.0,test +example_2,ATGGCAGAATTTCCGTAA,0.0,test diff --git a/skills/codonfm-embed/evals/files/sequences_edge_cases.csv b/skills/codonfm-embed/evals/files/sequences_edge_cases.csv new file mode 100644 index 0000000..8c5554d --- /dev/null +++ b/skills/codonfm-embed/evals/files/sequences_edge_cases.csv @@ -0,0 +1,5 @@ +id,ref_seq,value,split +row_normal,ATGGCTGAATTTCCGTAA,0.0,test +row_blank_split,ATGGCAGAATTTCCGTAA,0.0, +row_normal,ATGGCCGAATTTCCGTAA,0.0,test +row_oversized,ATGGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTGCTTAA,0.0,test diff --git a/skills/codonfm-embed/references/checkpoint-selection.md b/skills/codonfm-embed/references/checkpoint-selection.md new file mode 100644 index 0000000..3133f1a --- /dev/null +++ b/skills/codonfm-embed/references/checkpoint-selection.md @@ -0,0 +1,42 @@ +# Choosing a public Encodon checkpoint + +Use this reference for checkpoint recommendations and performance questions. +Honor a requested checkpoint and resource constraints; supplied example metadata +is not evidence that it is the best model for a different task. + +| User objective | Starting choice | Reason | +| --- | --- | --- | +| Strongest published translation-efficiency/expression results | [nvidia/NV-CodonFM-Encodon-1B-v1](https://huggingface.co/nvidia/NV-CodonFM-Encodon-1B-v1), `encodon_1b` | Largest released random-mask model; strongest downstream results in the preprint | +| Codon-frequency-weighted masking | [nvidia/NV-CodonFM-Encodon-Cdwt-1B-v1](https://huggingface.co/nvidia/NV-CodonFM-Encodon-Cdwt-1B-v1), also `encodon_1b` | Alternative masking objective; compare on task-specific validation data | +| Quick demonstration or restricted memory/latency | 80M (`encodon_80m`), or 600M (`encodon_600m`) | Resource tradeoff, not evidence of superior embedding quality | + +The [CodonFM preprint, Figure 5 and accompanying text, page 11](https://research.nvidia.com/labs/dbr/assets/data/manuscripts/nv-codonfm-preprint.pdf) +compares random-forest regressors trained on frozen pretrained embeddings. It +reports the strongest downstream correlation and explained variance for the 1B +model. Cdwt embeddings depend less on simple features such as GC content and can +perform worse on tasks dominated by those features. Prefer random-mask 1B for +this benchmark-driven recommendation; Cdwt is not universally better. + +Keep the reported metrics distinct: Figure 5A's caption specifies mean 10-fold +cross-validation **R²** for translation efficiency; Figure 5B specifies +**Spearman correlation** for mRFP expression. An R² value is not Pearson's r or +Spearman's rho. Cite the paper/figure for performance claims, not runner source +or checkpoint configuration, which contain no benchmark results. Quote a +number only after checking the corresponding metric, panel, and dataset; label +a visually estimated value as approximate. + +For a new ~200-sequence labeled set, no numerical correlation is established by +these benchmarks. Recommend 1B as an evidence-based starting point, explain the +published advantage, and separate it from expected performance on the new assay. +Suggest cross-validation of a lightweight regressor on frozen features, grouping +related sequences to reduce leakage, and reporting uncertainty. Small training +sets motivate controlling regressor complexity; they do not by themselves +justify choosing the smallest frozen backbone. Do not promise `r ≈ 0.6` or infer +correlation by taking the square root of the paper's cross-validation R². + +The [public repository's model table](https://github.com/NVIDIA-Digital-Bio/CodonFM#pre-trained-models) +lists 80M, 600M, 1B, and Cdwt-1B. Parser options alone do not establish that 5B or +10B weights are released. The `TE` checkpoint family refers to the accelerated +**Transformer Engine** implementation, not a translation-efficiency-trained +checkpoint. Keep the original public runner and its compatible checkpoints +together; changing to the accelerated recipe is a separate runtime choice. diff --git a/skills/codonfm-embed/scripts/validate_inputs.py b/skills/codonfm-embed/scripts/validate_inputs.py new file mode 100644 index 0000000..b02442c --- /dev/null +++ b/skills/codonfm-embed/scripts/validate_inputs.py @@ -0,0 +1,126 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. +# SPDX-License-Identifier: Apache-2.0 +"""Read-only CSV preflight for public Encodon embedding extraction. + +Usage: python validate_inputs.py sequences.csv [--context-length 2048] +Arguments: input CSV path; optional context length including CLS and SEP. +Output: JSON summary and per-row verdicts on stdout, or an error object. +Exit codes: 0 clean, 1 row findings to review, 2 file/schema/argument error. + +Uses only the standard library. Reports input problems without claiming to run +the model or reproducing every pandas/tokenizer coercion. Never writes inputs. +""" + +import argparse +from collections import Counter +import csv +import json +import math +from pathlib import Path + + +REQUIRED_COLUMNS = {"id", "ref_seq", "value", "split"} +DEFAULT_CONTEXT_LENGTH = 2048 +SPECIAL_TOKEN_COUNT = 2 +CODON_WIDTH = 3 + + +def validate_csv(path: Path, context_length: int = DEFAULT_CONTEXT_LENGTH) -> dict: + """Describe split selection, sequence validity, ID ambiguity, and truncation.""" + if context_length <= SPECIAL_TOKEN_COUNT: + raise ValueError("context length must allow CLS, at least one codon, and SEP") + with path.open(encoding="utf-8-sig", newline="") as handle: + reader = csv.DictReader(handle, strict=True) + columns = reader.fieldnames or [] + missing = sorted(REQUIRED_COLUMNS - set(columns)) + if missing: + raise ValueError(f"missing required columns for eval: {', '.join(missing)}") + if len(set(columns)) != len(columns): + raise ValueError("duplicate CSV column names") + rows = list(reader) + if any(None in row or any(v is None for v in row.values()) for row in rows): + raise ValueError("CSV row width does not match header") + + ids = Counter(row["id"] for row in rows) + limit = context_length - SPECIAL_TOKEN_COUNT + verdicts = [] + for number, row in enumerate(rows, start=1): + selected = row["split"] == "test" + issues = [] + if not selected: + issues.append("excluded: split must be exactly 'test' to enter extraction") + if not row["id"].strip(): + issues.append("blank id: assign a unique nonblank ID") + elif ids[row["id"]] > 1: + issues.append("duplicate id: extraction does not reject it, but output-to-source joins are ambiguous; assign unique IDs") + + sequence = row["ref_seq"] + dna = sequence.replace("U", "T") + sequence_valid = bool(dna) and set(dna) <= set("ACGT") and len(dna) % CODON_WIDTH == 0 + if not sequence_valid: + issues.append("invalid sequence for this preflight: require nonempty uppercase A/C/G/T (or U) and a length divisible by three; correct a copy") + try: + value_valid = math.isfinite(float(row["value"])) + except ValueError: + value_valid = False + if not value_valid: + issues.append("invalid value: supply a finite numeric label or 0.0 for extraction-only data") + + codons = len(dna) // CODON_WIDTH if sequence_valid else None + retained = min(codons, limit) if selected and sequence_valid else None + lost = codons - retained if retained is not None else None + if lost: + issues.append(f"truncated: first {retained} codons retained; last {lost} codons ({CODON_WIDTH * lost} nucleotides) lost") + if not selected: + verdict = "excluded" + elif not (sequence_valid and value_valid): + verdict = "needs_correction" + elif lost: + verdict = "truncated" + else: + verdict = "processes" + verdicts.append({ + "row": number, + "id": row["id"], + "split": row["split"], + "selected_for_test": selected, + "nucleotides": len(sequence), + "codons": codons, + "value": row["value"], + "sequence_valid": sequence_valid, + "value_valid": value_valid, + "retained_codons": retained, + "lost_codons": lost, + "verdict": verdict, + "issues": issues, + }) + selected_rows = sum(item["selected_for_test"] for item in verdicts) + return { + "csv": str(path), + "context_length": context_length, + "codon_limit": limit, + "total_rows": len(rows), + "test_rows": selected_rows, + "excluded_rows": len(rows) - selected_rows, + "warnings": [] if selected_rows else ["no rows have split='test'; no embeddings can be produced"], + "rows": verdicts, + } + + +def main(argv=None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("csv", type=Path, help="input CSV; never modified") + parser.add_argument("--context-length", type=int, default=DEFAULT_CONTEXT_LENGTH) + args = parser.parse_args(argv) + try: + report = validate_csv(args.csv, args.context_length) + except (OSError, UnicodeError, csv.Error, ValueError) as error: + print(json.dumps({"error": str(error), "csv": str(args.csv)})) + return 2 + print(json.dumps(report, indent=2)) + return int(bool(report["warnings"]) or any(row["issues"] for row in report["rows"])) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/skills/codonfm-embed/skill-card.md b/skills/codonfm-embed/skill-card.md new file mode 100644 index 0000000..64cfb58 --- /dev/null +++ b/skills/codonfm-embed/skill-card.md @@ -0,0 +1,88 @@ +## Description:
+Validate coding-sequence CSVs, extract public CodonFM/Encodon embeddings, and choose checkpoints for downstream property modeling.
+ +This skill is ready for commercial/non-commercial use.
+ +## Owner +NVIDIA
+ +### License/Terms of Use:
+Apache 2.0
+## Use Case:
+Developers and engineers working with codon sequences use this skill to validate CSV inputs, extract frozen Encodon embeddings, and select model checkpoints for downstream property modeling such as translation efficiency and mRNA stability prediction.
+ +### Deployment Geography for Use:
+Global
+ +## Requirements / Dependencies:
+**Requires API Key or External Credential:** [No]
+**Credential Type(s):** [None]
+ +Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+ +## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
+ +## Reference(s):
+- [Checkpoint Selection Guide](references/checkpoint-selection.md)
+- [NV-CodonFM-Encodon-1B-v1 (HuggingFace)](https://huggingface.co/nvidia/NV-CodonFM-Encodon-1B-v1)
+- [NV-CodonFM-Encodon-80M-v1 (HuggingFace)](https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1)
+- [NV-CodonFM-Encodon-600M-v1 (HuggingFace)](https://huggingface.co/nvidia/NV-CodonFM-Encodon-600M-v1)
+- [NV-CodonFM-Encodon-Cdwt-1B-v1 (HuggingFace)](https://huggingface.co/nvidia/NV-CodonFM-Encodon-Cdwt-1B-v1)
+- [CodonFM Encodon on NGC](https://catalog.ngc.nvidia.com/orgs/nvidia/teams/clara/models/nv_codonfm_encodon)
+- [CodonFM Preprint](https://research.nvidia.com/labs/dbr/assets/data/manuscripts/nv-codonfm-preprint.pdf)
+ + +## Skill Output:
+**Output Type(s):** [Shell commands, Analysis]
+**Output Format:** [Markdown with inline bash code blocks and JSON]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [None]
+ +## Evaluation Agents Used:
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + + +## Evaluation Tasks:
+Evaluated against 4 evaluation tasks (4 positive) in isolated sandbox pods, using evaluator version 1.5.6.
+ +## Evaluation Metrics Used:
+Reported benchmark dimensions:
+- Security: Whether the skill is safe to use: checks for unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Whether the final answer is correct against the reference answer.
+- Discoverability: Whether the expected skill was selected and activated when needed, and decoys were avoided.
+- Effectiveness: Whether the skill helped complete the user's goal (50% goal accuracy + 50% expected workflow adherence).
+- Efficiency: Whether the skill avoided wasted tool calls and token usage (50% tool-call productivity + 50% token efficiency).
+ +Underlying evaluation signals used in this run:
+- `security`: Checks for unsafe operations, secret leakage, and unauthorized access.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `skill_execution`: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity measured against expected tool-call patterns.
+- `token_efficiency`: Actual uncached prompt plus completion token usage.
+ + + +## Evaluation Results:
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 96.1% | 93.4% | +| Security | 75.0% → 100.0% (+25.0 points) | 75.0% → 100.0% (+25.0 points) | +| Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Discoverability | 96.3% | 86.3% | +| Effectiveness | 85.6% → 90.6% (+5.0 points) | 98.8% → 93.8% (-5.0 points) | +| Efficiency | 93.6% | 86.9% | + +## Skill Version(s):
+be43117 (source: git SHA, committed 2026-10-07)
+ +## Ethical Considerations:
+NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
+ +(For Release on NVIDIA Platforms Only)
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/codonfm-embed/skill.oms.sig b/skills/codonfm-embed/skill.oms.sig new file mode 100644 index 0000000..03f18f5 --- /dev/null +++ b/skills/codonfm-embed/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAiY29kb25mbS1lbWJlZCIsCiAgICAgICJkaWdlc3QiOiB7CiAgICAgICAgInNoYTI1NiI6ICIyZTNkNzljZDIwODczY2RmYTliMTgyMzVhMTZlYjEzZGFhYTY0MTEzNjU4MWRlMmE0ZTI1N2JmNmRjZTNjZjYzIgogICAgICB9CiAgICB9CiAgXSwKICAicHJlZGljYXRlVHlwZSI6ICJodHRwczovL21vZGVsX3NpZ25pbmcvc2lnbmF0dXJlL3YxLjAiLAogICJwcmVkaWNhdGUiOiB7CiAgICAicmVzb3VyY2VzIjogWwogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMGJmMmM4ZjBkMWIzN2E1OWM3MGU5YmNhY2I5ZjgxYmMwMGZlMWJlMTJlOWM2MjM4MDAwZjA5ZmUyNTMwODg2OSIsCiAgICAgICAgIm5hbWUiOiAiQkVOQ0hNQVJLLm1kIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMGQ4ZDQzMGE0OGExZjBjNjMzMjViMDZlMDVkM2RkNmM5NDgwNjNkNjYyNWRmNjUxMDBkMGMxNjJmMzlhMjRhNyIsCiAgICAgICAgIm5hbWUiOiAiU0tJTEwubWQiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJlYmQ5MTViYjNlZDExYzA3Y2Y0MzE3OGEzMGEyZmEyZWQ3ZmU4OTgwNmQ2ZTEyNDE4ZmNmMTgxZjZiY2EzMTk4IiwKICAgICAgICAibmFtZSI6ICJhZ2VudHMvb3BlbmFpLnlhbWwiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJjOWQ5YmZhNmJjYzUzMGJhNWViYjk0MDZmNWRmYWZkMzU5OTM4ODA0MTM1MDBjYTBkYzI2YWFkYjEzMDU0MTI5IiwKICAgICAgICAibmFtZSI6ICJldmFscy9ldmFscy5qc29uIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZjBjMTIyZjE5YTJkMzE2ZmU5NWNjZmFhMGFjNTFiZDYxM2NmYWJkZDA5ZmQ4OWZiYjc2NjU2MmE1ZjBlZWQyNyIsCiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZmlsZXMvY29kb25mbV9zb3VyY2UuemlwIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiNjk4ZmEzNTQxNDAwMzNlOWFhODU5MWNkMjg2MDcwM2E0ZmNkYzJlYjI3ZWRjOGYyZjA3NjBjNDA4M2U1NjNlZSIsCiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZmlsZXMvZW5jb2Rvbl9jaGVja3BvaW50Lmpzb24iCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICIyZWQ2OGZlZmQ2MWUzNjFkZDIxZDBiYzI4YTdkZGRlZjMxZTNlOWI0MmRkNjEwMGY1MzRlOTA2OTdjNzk3ZTc5IiwKICAgICAgICAibmFtZSI6ICJldmFscy9maWxlcy9zZXF1ZW5jZXMuY3N2IgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiNjlkZTgyM2I0ZGE2ZmY3ZDY1ZmFmYTkyZjBmYTQ3ZjdkZDRjMGI3MTI3ZGRhMzlmYTA0MmJiNWFlODJiZWQzNCIsCiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZmlsZXMvc2VxdWVuY2VzX2VkZ2VfY2FzZXMuY3N2IgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiYjhjZWQ5MjFmMWVjYjQ2MjU2MWNkNDExNzlkMTYwYzk4N2U0Y2QzYzk4Mjg2N2M1NzYxMjUyNTVjMTA0MTJlMCIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9jaGVja3BvaW50LXNlbGVjdGlvbi5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImE3MGZmYTA2ZjhiZmJkNWY0YWQ1OGFiOGYzZjBmYjNmMDYwMmE2ZmIzMWU4NGE4Mjg1YTA4M2I3OWZhMzFjNTIiLAogICAgICAgICJuYW1lIjogInNjcmlwdHMvdmFsaWRhdGVfaW5wdXRzLnB5IgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiNjRkNDZhNDA0YjYxZDQ1ZGFjZGRjNDk3YTQ3ODUxZjg2NjQzODRmZmQ4MmU4NjQ3ZTFjZDc2MDgwYjdmZjlhOSIsCiAgICAgICAgIm5hbWUiOiAic2tpbGwtY2FyZC5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjU3NmJmOTI3ZTljNjA2NjhkOWNkMjdmYWRjYWE2Zjg5ZWM5YjdjZmI3MmViYzMxNjU3NmY1YjA3YTA1NTYzYmMiLAogICAgICAgICJuYW1lIjogInRlc3RzL3Rlc3RfdmFsaWRhdGVfaW5wdXRzLnB5IgogICAgICB9CiAgICBdLAogICAgInNlcmlhbGl6YXRpb24iOiB7CiAgICAgICJhbGxvd19zeW1saW5rcyI6IGZhbHNlLAogICAgICAibWV0aG9kIjogImZpbGVzIiwKICAgICAgImlnbm9yZV9wYXRocyI6IFsKICAgICAgICAiLmdpdGF0dHJpYnV0ZXMiLAogICAgICAgICIuZ2l0aWdub3JlIiwKICAgICAgICAiLmdpdGh1YiIsCiAgICAgICAgIi5naXQiCiAgICAgIF0sCiAgICAgICJoYXNoX3R5cGUiOiAic2hhMjU2IgogICAgfQogIH0KfQ==","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCME1fdEyorazZs3RPQjQ0PrLkZjnMy53fYqp5gHPdjd4nu2N95I1eI8qI9hxxq5mQRAIxALV0IoPK5s8estrH+x4hKfMvZRQHp2Jxo3xaalm7Ux7+p9qLxrs6uTRNzfHuWB3F1Q==","keyid":""}]}} \ No newline at end of file diff --git a/skills/codonfm-embed/tests/test_validate_inputs.py b/skills/codonfm-embed/tests/test_validate_inputs.py new file mode 100644 index 0000000..3a4d8fc --- /dev/null +++ b/skills/codonfm-embed/tests/test_validate_inputs.py @@ -0,0 +1,121 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. +# SPDX-License-Identifier: Apache-2.0 +"""Exercise observable CSV verdicts and the read-only command interface.""" + +import csv +from contextlib import redirect_stdout, redirect_stderr +import importlib.util +import io +import json +from pathlib import Path +import tempfile +import unittest + + +SKILL = Path(__file__).resolve().parents[1] +SCRIPT = SKILL / "scripts" / "validate_inputs.py" +spec = importlib.util.spec_from_file_location("validate_inputs", SCRIPT) +checker = importlib.util.module_from_spec(spec) +spec.loader.exec_module(checker) + + +class ValidateInputsTest(unittest.TestCase): + def write_csv(self, directory, rows, columns=("id", "ref_seq", "value", "split")): + path = Path(directory) / "input.csv" + with path.open("w", newline="") as handle: + writer = csv.writer(handle) + writer.writerow(columns) + writer.writerows(rows) + return path + + def test_supplied_clean_rows_and_no_writes(self): + path = SKILL / "evals" / "files" / "sequences.csv" + before = path.read_bytes() + report = checker.validate_csv(path) + self.assertEqual(report["test_rows"], 2) + self.assertEqual([row["verdict"] for row in report["rows"]], ["processes"] * 2) + self.assertTrue(all(not row["issues"] for row in report["rows"])) + self.assertEqual(path.read_bytes(), before) + + def test_supplied_edge_cases(self): + path = SKILL / "evals" / "files" / "sequences_edge_cases.csv" + before = path.read_bytes() + report = checker.validate_csv(path) + self.assertEqual(report["test_rows"], 3) + self.assertEqual(report["excluded_rows"], 1) + rows = report["rows"] + self.assertEqual([row["verdict"] for row in rows], ["processes", "excluded", "processes", "truncated"]) + self.assertEqual([row["row"] for row in rows], [1, 2, 3, 4]) + self.assertTrue(any("duplicate id" in issue for issue in rows[0]["issues"])) + self.assertTrue(any("duplicate id" in issue for issue in rows[2]["issues"])) + self.assertEqual((rows[3]["codons"], rows[3]["retained_codons"], rows[3]["lost_codons"]), (2050, 2046, 4)) + self.assertEqual(path.read_bytes(), before) + + def test_configurable_context_boundary_and_rna(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_csv(directory, [ + ["boundary", "AUG" * 6, "0.0", "test"], + ["over", "ATG" * 7, "0.0", "test"], + ]) + rows = checker.validate_csv(path, context_length=8)["rows"] + self.assertEqual(rows[0]["verdict"], "processes") + self.assertEqual(rows[0]["lost_codons"], 0) + self.assertEqual(rows[1]["lost_codons"], 1) + + def test_strict_split_filter_including_invalid_excluded_sequence(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_csv(directory, [ + [str(n), "bad-sequence", "not-a-number", split] + for n, split in enumerate(["", "train", "Test", "test "]) + ]) + report = checker.validate_csv(path) + self.assertEqual(report["test_rows"], 0) + self.assertTrue(report["warnings"]) + self.assertTrue(all(row["verdict"] == "excluded" for row in report["rows"])) + self.assertTrue(all(row["retained_codons"] is None for row in report["rows"])) + + def test_invalid_sequences_and_labels(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_csv(directory, [ + ["empty", "", "0", "test"], + ["partial", "ATGA", "0", "test"], + ["lower", "atg", "0", "test"], + ["ambiguous", "ANN", "0", "test"], + ["label", "ATG", "NaN", "test"], + ]) + report = checker.validate_csv(path) + self.assertTrue(all(row["verdict"] == "needs_correction" for row in report["rows"])) + + def test_missing_split_and_bad_row_width(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_csv(directory, [["a", "ATG", "0"]], columns=("id", "ref_seq", "value")) + with self.assertRaisesRegex(ValueError, "split"): + checker.validate_csv(path) + path = self.write_csv(directory, [["a", "ATG", "0", "test", "extra"]]) + with self.assertRaisesRegex(ValueError, "width"): + checker.validate_csv(path) + + def test_empty_dataset_and_invalid_context(self): + with tempfile.TemporaryDirectory() as directory: + path = self.write_csv(directory, []) + report = checker.validate_csv(path) + self.assertEqual(report["rows"], []) + self.assertTrue(report["warnings"]) + with self.assertRaisesRegex(ValueError, "context length"): + checker.validate_csv(path, 2) + + def test_cli_exit_codes_and_json(self): + for name, expected in [("sequences.csv", 0), ("sequences_edge_cases.csv", 1), ("missing.csv", 2)]: + with self.subTest(name=name): + stdout, stderr = io.StringIO(), io.StringIO() + argv = [str(SKILL / "evals" / "files" / name)] + with redirect_stdout(stdout), redirect_stderr(stderr): + code = checker.main(argv) + self.assertEqual(code, expected) + report = json.loads(stdout.getvalue()) + self.assertEqual("error" in report, expected == 2) + self.assertFalse(stderr.getvalue()) + + +if __name__ == "__main__": + unittest.main() diff --git a/skills/codonfm-finetune/BENCHMARK.md b/skills/codonfm-finetune/BENCHMARK.md new file mode 100644 index 0000000..6250f63 --- /dev/null +++ b/skills/codonfm-finetune/BENCHMARK.md @@ -0,0 +1,123 @@ +# Skill Benchmark: codonfm-finetune + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `codonfm-finetune` +- Evaluation date: 2026-10-07 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 3 evaluation tasks (3 positive) +- Dataset digest: `sha256:81d6d44379cdd6e73149247223022a3b07d3b00517da29ff30435ddc7aedc045` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 1 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 85.0% — baseline ran, but no comparable score was available; uplift unavailable | 82.8% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 66.7% (-33.3 points) | 33.3% → 66.7% (+33.4 points) | +| Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Discoverability | 91.7% — baseline ran, but no comparable score was available; uplift unavailable | 76.7% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 86.7% → 86.7% (±0.0 points) | 96.7% → 90.0% (-6.7 points) | +| Efficiency | 80.0% — baseline ran, but no comparable score was available; uplift unavailable | 80.7% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 2,259,711 | 5,318,079 | -3,058,368 | -57.51% | skill 3/3; base 3/3 | +| claude-code | codonfm-finetune-001 | 975,823 | 2,626,474 | -1,650,651 | -62.85% | skill 1/1; base 1/1 | +| claude-code | codonfm-finetune-002 | 728,893 | 1,887,306 | -1,158,413 | -61.38% | skill 1/1; base 1/1 | +| claude-code | codonfm-finetune-003 | 554,995 | 804,299 | -249,304 | -31.00% | skill 1/1; base 1/1 | +| codex | All cases | 1,217,895 | 1,719,144 | -501,249 | -29.16% | skill 3/3; base 3/3 | +| codex | codonfm-finetune-001 | 536,353 | 1,008,386 | -472,033 | -46.81% | skill 1/1; base 1/1 | +| codex | codonfm-finetune-002 | 435,825 | 405,824 | +30,001 | +7.39% | skill 1/1; base 1/1 | +| codex | codonfm-finetune-003 | 245,717 | 304,934 | -59,217 | -19.42% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 3,477,606 | 7,037,223 | -3,559,617 | -50.58% | skill 6/6; base 6/6 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 16 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 3 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/codonfm-finetune/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/codonfm-finetune/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/codonfm-finetune/SKILL.md`) +- **MEDIUM** SECURITY/Unknown (LP3): MCP Least Privilege: The skill declares no explicit tool scope (no 'permissions' or 'allowed-tools' field in metadata), yet the skill content (`SKILL.md:1`) +- **MEDIUM** SECURITY/Autonomous Decision Making (EA2): Excessive Agency: without checking (`SKILL.md:162`) +- 11 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/codonfm-finetune/SKILL.md b/skills/codonfm-finetune/SKILL.md new file mode 100644 index 0000000..1474499 --- /dev/null +++ b/skills/codonfm-finetune/SKILL.md @@ -0,0 +1,200 @@ +--- +name: codonfm-finetune +description: Fine-tune public CodonFM Encodon checkpoints on labeled coding-sequence or coding-variant data using LoRA, head-only, or full fine-tuning. Use when a user explicitly asks to fine-tune CodonFM or Encodon for regression or classification. Support generic public-v1 Encodon workflows only; reject Decodon, MissenseDataset, missense_synom_agg, and generation workflows. +metadata: + author: "NVIDIA BioNeMo " +--- + +# Fine-tune public Encodon + +Use `--pretrained_ckpt_path` for public v1. Do not substitute +`--checkpoint_path`: the public runner does not forward that argument to the +fine-tuning task. + +## Instructions + +Resolve the target label, dataset, checkpoint, and output directory from the +request and available files. Reuse existing data and weights. For training, +check the project's ML dependencies and a compatible NVIDIA GPU before launch. +If a required resource is unavailable, complete the available data preparation +and return the command with the missing prerequisite clearly identified. +When training is requested and the prerequisites are met, execute it and check +the resulting checkpoints and metrics. A request for preparation ends with the +validated inputs and command. + +Use the user's labeled dataset when provided. For a demonstration of sequence +regression without a dataset, use the public human RiboNN translation-efficiency +data below and state that choice. This is not a substitute for a user's intended +assay or for labeled coding variants. If variant labels are missing, return the +required schema and a command template promptly; do not search for labels or +invent measured effects. + +Default demonstration checkpoint: `nvidia/NV-CodonFM-Encodon-80M-v1`, revision +`399ca9fe17b57941a7bebc6788033919b417413c`, file +`NV-CodonFM-Encodon-80M-v1.safetensors` with sibling `config.json`. +The [public weights](https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1) +are about 307 MB. Download them when needed for the requested work; input +preparation can record an intended checkpoint path. These are the original Encodon weights; the `-TE-` +checkpoints use the separate TransformerEngine implementation. + +For an unsupported Decodon or missense-aggregation request, inspect the public +parser/model configuration, explain the missing feature, and finish. Do not +implement the missing model or search private repositories. + +## Examples + +Prepare a small public-data example with the standard-library helper +[prepare_ribonn.py](scripts/prepare_ribonn.py), running from the repository root. +Set `CODONFM_DATA_PATH` to the CSV you want to create: + +```bash +python skills/codonfm-finetune/scripts/prepare_ribonn.py \ + --output "$CODONFM_DATA_PATH" +``` + +With an existing raw file, add `--input "$RIBONN_DATA_PATH"`. The default reads +at most eight accepted rows per split; `--max-rows-per-split 0` processes the full +input. Remote streaming has a time budget and no automatic retries; use a local +file if it fails. The helper follows the CDS slicing in the +[RiboNN notebook](../../notebooks/4-EnCodon-Downstream-Task-riboNN.ipynb): + +- Read the upstream `.csv` with a **tab** delimiter. +- Set `ref_seq = tx_sequence[utr5_size:utr5_size + cds_size]`, `id = transcript_id`, + and `value = mean_te` unchanged. Do not take another logarithm. +- Preserve source fold groups: 0–7 become `train`, 8 becomes `val`, 9 becomes + `test`. This is a demonstration holdout, not the notebook's cross-validation. +- Exclude invalid/non-finite rows and CDSs exceeding 2046 codons instead of + silently truncating labeled examples. Record counts and source in the adjacent + `.metadata.json`. A small subset does not establish predictive performance. + +The pinned dataset URL is in the helper; its source is +[CenikLab/TE_classic_ML](https://github.com/CenikLab/TE_classic_ML/tree/main/data). +The notebook extracts frozen Encodon embeddings and trains a random-forest +regressor with fold-based cross-validation. This skill reuses its data source, +CDS extraction, and target for a separate fine-tuning example; it does not +reproduce the notebook's training procedure or results. + +## Supported strategies + +- `lora`: adapter fine-tuning; default choice for smaller datasets. +- `head_only_random`: freeze the backbone and train a new head. +- `head_only_pretrained`: train an existing compatible pretrained head. +- `full`: update the complete model. + +Accept only `encodon_80m`, `encodon_600m`, or `encodon_1b`. + +## Sequence-level regression or classification + +Require `id`, `ref_seq`, `value`, and `split` columns. Extra columns are allowed. +Map the user's columns to this loader schema; the RiboNN helper is only for +RiboNN source data. Training needs `train` rows and, when validation is enabled, +`val` rows. A `test` split is needed only for later evaluation. Labels in unused +splits need not be populated. Regression targets must be finite numbers; +classification targets must be integer class indices from zero through +`num_classes - 1`. Use a downstream head for scalar targets. + +Check sequence preparation with the user’s assay in mind. The loader converts +uppercase RNA `U` to `T`, and the tokenizer uppercases bases. Ambiguous bases and +incomplete codons can produce unknown tokens; overlength sequences are truncated. +Review these cases rather than silently dropping user records. Choose batches +and a training budget appropriate to the dataset; small training sets may be +resampled by the loader. + +Set `CODONFM_CHECKPOINT_PATH` to the checkpoint file and `CODONFM_RUN_DIR` to +your chosen output directory. This example runs ten steps to check the workflow; +choose the training budget for the actual dataset and task: + +```bash +python -m src.runner finetune \ + --exp_name property_finetune \ + --model_name encodon_80m \ + --pretrained_ckpt_path "$CODONFM_CHECKPOINT_PATH" \ + --data_path "$CODONFM_DATA_PATH" \ + --process_item codon_sequence \ + --dataset_name CodonBertDataset \ + --finetune_strategy lora \ + --lora_alpha 32 \ + --lora_r 16 \ + --lora_dropout 0.1 \ + --loss_type regression \ + --use_downstream_head \ + --lr 2e-5 \ + --max_steps 10 \ + --warmup_iterations 1 \ + --check_val_every_n_epoch 1 \ + --train_batch_size 4 \ + --val_batch_size 4 \ + --num_workers 0 \ + --num_nodes 1 \ + --num_gpus 1 \ + --out_dir "$CODONFM_RUN_DIR" \ + --checkpoints_dir "$CODONFM_RUN_DIR/checkpoints" +``` + +For classification, replace `--loss_type regression` with +`--loss_type classification` and pass the correct `--num_classes`. + +## Generic coding-variant classification + +Use `MutationDataset` only for an ordinary labeled variant head, not the newer +synonymous-codon aggregation loss. Require `id`, the reference-sequence column +(`ref_seq` by default), `ref_codon`, `alt_codon`, `codon_position`, and the chosen +label column. Select existing sequence/label columns with `--ref_seq_col` and +`--label_col`; these overrides apply to `MutationDataset` only. Starting from +the sequence-level command, change/add: + +```text +--process_item mutation_pred_mlm +--dataset_name MutationDataset +--label_col label +--loss_type classification +--num_classes 2 +--use_downstream_head +--extract-seq +--mask_mutation +--train_val_test_ratio 0.8 0.1 0.1 +``` + +Always keep `--mask_mutation` for masked-codon variant inputs. +Use `--extract-seq` to construct the context around a variant in a full CDS; +already prepared contexts can omit it. Choose split ratios for the dataset; +a held-out test split is optional for training. Public v1 reuses existing +`train_idx.npy`, `val_idx.npy`, and `test_idx.npy` files without checking that +they belong to the current CSV. Verify their provenance before reusing them. + +## Execute and outputs + +Check prepared data directly against the selected loader's schema above using +ordinary CSV inspection. Verify required columns, finite labels in the splits +used for training, class indices when applicable, sequence preparation, and +variant reference positions. The RiboNN helper checks its output during +preparation. These checks do not require installing CodonFM's ML dependencies. +For preparation requests, report what was checked and provide the training +command. For execution requests, run it once data, weights, and compute are ready. + +The existing [runner](../../src/runner.py) has an optional `--dryrun` flag that +builds runtime configuration and skips execution. It requires the ML dependencies +and does not read the dataset or load weights. It is not a data-validation step +or a prerequisite for preparing inputs and commands. + +Set validation frequency for the planned training length: for a small example, +`--check_val_every_n_epoch 1` or a smaller `--val_check_interval` avoids public +v1's default interval of 1,000 batches exceeding an epoch. + +- Checkpoints are written under the explicitly supplied `--checkpoints_dir`, + including `last.ckpt` and configured best checkpoints. +- CSV metrics are written below `--out_dir//version_*` unless W&B is + enabled. +- W&B requires `--enable_wandb`, `--project_name`, and `--entity` together. +- Fine-tuning does not produce prediction arrays; run an evaluation task + separately against the resulting checkpoint. + +## Boundaries + +- Do not use `MissenseDataset`, `missense_seq`, `missense_inference`, + `missense_synom_agg`, or any `--missense_*` flag. They are absent publicly. +- Do not use Decodon model names, CLM preprocessing, organism tokens, or + generation datasets. +- Require an explicit learning rate. Public v1 passes `lr=None` otherwise. +- Treat scientific and clinical validity as a separate validation problem; + successful training does not certify the resulting model. diff --git a/skills/codonfm-finetune/agents/openai.yaml b/skills/codonfm-finetune/agents/openai.yaml new file mode 100644 index 0000000..0dafc77 --- /dev/null +++ b/skills/codonfm-finetune/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "CodonFM Fine-tuning" + short_description: "Fine-tune public Encodon models on labeled data" + default_prompt: "Use $codonfm-finetune to prepare and validate an Encodon fine-tuning run on my labeled data." diff --git a/skills/codonfm-finetune/evals/evals.json b/skills/codonfm-finetune/evals/evals.json new file mode 100644 index 0000000..cbbe9fe --- /dev/null +++ b/skills/codonfm-finetune/evals/evals.json @@ -0,0 +1,60 @@ +{ + "skill_name": "codonfm-finetune", + "evals": [ + { + "id": "codonfm-finetune-001", + "prompt": "Prepare a LoRA fine-tuning run for public Encodon 80M using the supplied RiboNN sample. Use mean_te as the translation-efficiency label. Create a labeled coding-sequence CSV, validate it, and provide the training command. Preserve the source fold groups when choosing train, validation, and test splits. The public source and checkpoint metadata are supplied.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json", + "files/ribonn_smoke.tsv", + "files/ribonn_smoke.provenance.json" + ], + "expected_output": "A validated 12-row coding-sequence dataset with original mean_te labels and non-empty splits that preserve source fold groups, plus a public Encodon LoRA regression command. The response distinguishes input checks from training results.", + "assertions": [ + "The agent reads the tab-separated RiboNN sample and extracts CDS using utr5_size and cds_size instead of using the entire transcript", + "The output has id/ref_seq/value/split, preserves all 12 valid rows and the original mean_te labels, and reports non-empty splits without dividing a source fold across splits", + "The training command uses --pretrained_ckpt_path, lora, regression, --use_downstream_head, an explicit --lr, and a validation frequency compatible with the planned training run", + "Batch sizes fit the prepared splits, warmup is shorter than the training budget, and checkpoint/output paths are clearly specified", + "The agent validates the prepared CSV and accurately reports what was checked; it does not present preparation as completed training" + ], + "expected_skill": "codonfm-finetune", + "expected_script": "prepare_ribonn.py" + }, + { + "id": "codonfm-finetune-002", + "prompt": "Validate the supplied variants_labeled.csv and prepare a public Encodon fine-tuning command using a standard binary classification head. These are synthetic examples for checking the workflow; their label values are not measured biological effects. Explain the required preprocessing and data splits. Use the supplied public source and checkpoint metadata.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json", + "files/variants_labeled.csv", + "files/variants_labeled.provenance.json" + ], + "expected_output": "A validated labeled-variant dataset and a public MutationDataset classification command with viable splits. The response identifies the synthetic nature of the labels and does not claim a trained model.", + "assertions": [ + "The command uses MutationDataset, mutation_pred_mlm, --label_col label, --mask_mutation, and --extract-seq", + "The command uses --pretrained_ckpt_path, classification, --num_classes 2, --use_downstream_head, an explicit --lr, and a validation frequency compatible with the planned training run", + "The agent validates reference codons against zero-based CDS positions and binary labels, and checks that the selected training and validation splits are non-empty; a held-out test split is optional for this preparation task", + "The agent avoids reusing unrelated train_idx.npy, val_idx.npy, or test_idx.npy files when preparing the variant CSV", + "The agent does not use MissenseDataset or missense_synom_agg, claim training occurred, or interpret the synthetic labels as biological observations" + ], + "expected_skill": "codonfm-finetune", + "expected_script": null + }, + { + "id": "codonfm-finetune-003", + "prompt": "Does public CodonFM support fine-tuning Decodon with missense_synom_agg? Check the supplied source and explain whether this workflow is available.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json" + ], + "expected_output": "The agent identifies both features as unavailable in the supplied public implementation and explains the compatibility limitation.", + "assertions": [ + "The agent identifies Decodon and missense_synom_agg as unavailable in the supplied public source and cites the inspected files", + "The agent answers the compatibility question without inventing an executable command, downloading an undocumented model, or implementing the missing features" + ], + "expected_skill": "codonfm-finetune", + "expected_script": null + } + ] +} diff --git a/skills/codonfm-finetune/evals/files/codonfm_source.zip b/skills/codonfm-finetune/evals/files/codonfm_source.zip new file mode 100644 index 0000000..eb76611 Binary files /dev/null and b/skills/codonfm-finetune/evals/files/codonfm_source.zip differ diff --git a/skills/codonfm-finetune/evals/files/encodon_checkpoint.json b/skills/codonfm-finetune/evals/files/encodon_checkpoint.json new file mode 100644 index 0000000..6c6230d --- /dev/null +++ b/skills/codonfm-finetune/evals/files/encodon_checkpoint.json @@ -0,0 +1,33 @@ +{ + "repo_id": "nvidia/NV-CodonFM-Encodon-80M-v1", + "revision": "399ca9fe17b57941a7bebc6788033919b417413c", + "model_name": "encodon_80m", + "filename": "NV-CodonFM-Encodon-80M-v1.safetensors", + "size_bytes": 307351588, + "config_filename": "config.json", + "config": { + "vocab_size": 69, + "hidden_size": 1024, + "num_hidden_layers": 6, + "num_attention_heads": 8, + "intermediate_size": 4096, + "hidden_act": "gelu", + "hidden_dropout_prob": 0.1, + "attention_probs_dropout_prob": 0.1, + "initializer_range": 0.02, + "layer_norm_eps": 1e-12, + "pad_token_id": 3, + "position_embedding_type": "rotary", + "classifier_dropout": 0.1, + "rotary_theta": 10000.0, + "ignore_index": -100, + "loss_type": "cross_entropy", + "lora": false, + "lora_alpha": 32.0, + "lora_r": 16, + "lora_dropout": 0.1, + "finetune_strategy": "full" + }, + "source_url": "https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1/tree/399ca9fe17b57941a7bebc6788033919b417413c", + "weights_included": false +} diff --git a/skills/codonfm-finetune/evals/files/ribonn_smoke.provenance.json b/skills/codonfm-finetune/evals/files/ribonn_smoke.provenance.json new file mode 100644 index 0000000..6832940 --- /dev/null +++ b/skills/codonfm-finetune/evals/files/ribonn_smoke.provenance.json @@ -0,0 +1,12 @@ +{ + "source_url": "https://raw.githubusercontent.com/CenikLab/TE_classic_ML/512fca642b6b7b61ae494ad83cfc2b72831636d2/data/data_with_human_TE_cellline_all_NA_plain.csv", + "upstream_revision": "512fca642b6b7b61ae494ad83cfc2b72831636d2", + "selection": "First valid rows per split: 8 train (folds 0-7), 2 val (fold 8), 2 test (fold 9); CDS <= 2046 codons. Only columns required for preprocessing retained; sequences and mean_te values unchanged.", + "purpose": "Real public-data input preparation fixture, not a benchmark", + "split_counts": { + "train": 8, + "val": 2, + "test": 2 + }, + "rows": 12 +} diff --git a/skills/codonfm-finetune/evals/files/ribonn_smoke.tsv b/skills/codonfm-finetune/evals/files/ribonn_smoke.tsv new file mode 100644 index 0000000..67c7336 --- /dev/null +++ b/skills/codonfm-finetune/evals/files/ribonn_smoke.tsv @@ -0,0 +1,13 @@ +transcript_id tx_sequence utr5_size cds_size mean_te fold +ENST00000342066.8 GCAGAGCCCAGCAGATCCCTGCGGCGTTCGCGAGGGTGGGACGGGAAGCGGGCTGGGAAGTCGGGCCGAGGGAAAAGTCTGAAGACGCTTATGTCCAAGGGGATCCTGCAGGTGCATCCTCCGATCTGCGACTGCCCGGGCTGCCGAATATCCTCCCCGGTGAACCGGGGGCGGCTGGCAGACAAGAGGACAGTCGCCCTGCCTGCCGCCCGGAACCTGAAGAAGGAGCGAACTCCCAGCTTCTCTGCCAGCGATGGTGACAGCGACGGGAGTGGCCCCACCTGTGGGCGGCGGCCAGGCTTGAAGCAGGAGGATGGTCCGCACATCCGTATCATGAAGAGAAGAGTCCACACCCACTGGGACGTGAACATCTCTTTCCGAGAGGCGTCCTGCAGCCAGGACGGCAACCTTCCCACCCTCATATCCAGCGTCCACCGCAGCCGCCACCTCGTTATGCCCGAGCATCAGAGCCGCTGTGAATTCCAGAGAGGCAGCCTGGAGATTGGCCTGCGACCCGCCGGTGACCTGTTGGGCAAGAGGCTGGGCCGCTCCCCCCGTATCAGCAGCGACTGCTTTTCAGAGAAGAGGGCACGAAGCGAATCGCCTCAAGAGGCGCTGCTGCTGCCGCGGGAGCTGGGGCCCAGCATGGCCCCGGAGGACCATTACCGCCGGCTTGTGTCAGCACTGAGCGAGGCCAGCACCTTTGAGGACCCTCAGCGCCTCTACCACCTGGGCCTCCCCAGCCACGGTGAGGACCCACCCTGGCATGATCCCCCTCATCACCTCCCCAGCCACGATCTCCTGAGGGTCCGGCAGGAGGTGGCGGCTGCAGCTCTGAGGGGCCCCAGTGGCCTGGAAGCCCACCTGCCCTCCTCCACGGCAGGTCAGCGTCGGAAGCAGGGCCTGGCTCAGCACCGGGAGGGCGCCGCCCCAGCTGCCGCCCCGTCCTTCTCGGAGAGGGAGCTGCCTCAGCCGCCCCCCTTGCTGTCGCCGCAGAATGCCCCTCACGTCGCCCTGGGCCCCCATCTCAGGCCCCCCTTCCTGGGGGTGCCCTCGGCTCTGTGCCAGACCCCAGGCTACGGCTTCCTGCCCCCCGCGCAGGCGGAGATGTTCGCCTGGCAGCAGGAGCTCCTGCGGAAGCAGAACCTGGCCCGGCTGGAGCTGCCCGCCGACCTCCTGCGGCAGAAGGAGCTGGAGAGCGCGCGCCCACAGCTGCTGGCGCCCGAGACCGCCCTGCGCCCCAACGACGGCGCCGAGGAGCTGCAGCGGCGCGGGGCCCTGCTGGTGCTGAACCACGGCGCGGCGCCACTGCTGGCCCTGCCCCCCCAGGGGCCCCCGGGCTCCGGACCCCCCACCCCGTCCCGGGACTCTGCCCGGCGAGCCCCCCGGAAGGGGGGTCCCGGCCCTGCCTCAGCGCGGCCCAGCGAGTCCAAGGAGATGACGGGGGCTAGGCTCTGGGCACAAGATGGCTCGGAAGACGAGCCCCCCAAAGACTCGGACGGAGAGGACCCCGAGACGGCAGCTGTTGGGTGCAGGGGGCCCACTCCGGGCCAAGCTCCAGCTGGAGGGGCCGGCGCCGAGGGGAAGGGGCTTTTCCCAGGGTCCACACTGCCCCTGGGCTTCCCTTATGCCGTCAGCCCCTACTTCCACACAGGCGCGGTAGGGGGACTCTCCATGGATGGGGAGGAGGCCCCAGCCCCTGAGGACGTCACCAAGTGGACCGTGGATGACGTCTGCAGCTTCGTGGGGGGCCTGTCTGGCTGTGGAGAGTACACTCGGGTCTTCAGGGAGCAGGGGATCGACGGGGAGACCCTGCCACTGCTGACGGAGGAGCACCTGCTGACCAACATGGGGCTGAAGCTGGGGCCCGCCCTCAAGATCCGGGCCCAGGTGGCCAGGCGCCTGGGCCGAGTTTTCTACGTGGCCAGCTTCCCCGTGGCTCTGCCACTGCAGCCACCAACCCTGCGGGCCCCGGAGCGAGAACTCGGCACAGGAGAGCAGCCCTTGTCCCCCACGACGGCCACGTCCCCCTATGGAGGGGGCCACGCCCTTGCCGGTCAAACTTCACCCAAGCAGGAGAATGGGACCTTGGCTCTACTTCCAGGGGCCCCCGACCCTTCCCAGCCTCTGTGTTGAGGTTGCCGGGGGTAGGGGTGGGGCCACACAAATCTCCAGGAGCCACCACTCAACACAATGGCCCTGCCTCCCACCGCTTTATTTCTTTCGGTTTCGGATGCAAAACAAAAAATTTTAAAAGAAAATGTGACTTCAAAGGAAAGGAACAAATTTTCAAAGACTTGGGGGAGTGAAGGCAGAGCCTGGTGCAGATGGACGAGGTCTGCAGACGGAGGGCAGAGGTGGTGGAAGGGGCCAGGGGCCTGCAGGCCTCCCCCTGGAACTGGGACTGGTCTCGGTCTGCTGACGTCAGGGTCAGCTCCCCCGCGGAGCTGACTTCAGCAGCCCACAGCTGTGGGGCTTCAGCAGCCACACCAGCCCAGCCCAGCCCAGCTCTCGATACGTTTGGTCTTTCATGCTGAAAAATAAATAATAAAGCCTG 90 2046 -0.6890823386857142 4 +ENST00000327044.7 GCTTCGGGTTGGTGTCATGGCAGCTGCGGGGAGCCGCAAGAGGCGCCTGGCGGAGCTGACGGTGGACGAGTTCCTAGCTTCGGGCTTTGACTCCGAGTCCGAATCCGAGTCCGAAAATTCTCCACAAGCGGAGACACGGGAAGCACGCGAGGCTGCCCGGAGTCCGGATAAGCCGGGCGGGAGCCCCTCGGCCAGCCGGCGTAAAGGCCGTGCCTCTGAGCACAAAGACCAGCTCTCTCGGCTGAAGGACAGAGACCCCGAGTTCTACAAGTTCCTGCAGGAGAATGACCAGAGCCTGCTAAACTTCAGCGACTCGGACAGCTCTGAGGAGGAAGAGGGGCCGTTCCACTCCCTGCCAGATGTGCTGGAGGAAGCCAGTGAGGAGGAGGATGGAGCGGAGGAAGGAGAAGATGGGGACAGAGTCCCCAGAGGGCTGAAGGGGAAGAAGAATTCTGTTCCTGTGACCGTCGCCATGGTTGAGAGATGGAAGCAGGCAGCAAAGCAACGCCTCACTCCAAAGCTGTTCCATGAAGTGGTACAGGCGTTCCGAGCAGCTGTGGCCACCACCCGAGGGGACCAGGAAAGTGCTGAGGCCAACAAATTCCAGGTCACGGACAGTGCTGCATTCAATGCTCTGGTTACCTTCTGCATCAGAGACCTCATTGGCTGTCTCCAGAAGCTGCTGTTTGGAAAGGTGGCAAAGGATAGCAGCAGGATGCTGCAGCCGTCCAGCAGCCCGCTCTGGGGGAAGCTTCGTGTGGACATCAAGGCTTACCTGGGCTCGGCCATACAGCTGGTGTCCTGTCTGTCGGAGACGACGGTGTTGGCGGCCGTGCTGCGGCACATCAGCGTGCTGGTGCCCTGCTTCCTGACCTTCCCCAAGCAGTGCCGCATGCTGCTCAAGAGAATGGTGATCGTATGGAGCACTGGGGAAGAGTCTCTGCGGGTGCTGGCTTTCCTGGTCCTCAGCAGAGTCTGCCGGCACAAGAAGGACACTTTCCTTGGCCCCGTCCTCAAGCAAATGTACATCACGTATGTGAGGAACTGCAAGTTCACCTCGCCTGGTGCCCTCCCCTTCATCAGTTTCATGCAGTGGACCTTGACGGAGCTGCTGGCCCTGGAGCCGGGTGTGGCCTACCAGCACGCCTTCCTCTACATCCGCCAGCTCGCCATACACCTGCGCAACGCCATGACCACTCGCAAGAAGGAAACATACCAGTCTGTGTACAACTGGCAGTATGTGCACTGCCTCTTCCTGTGGTGCCGGGTCCTGAGCACTGCGGGCCCCAGCGAAGCCCTCCAGCCCTTGGTCTACCCCCTTGCCCAAGTCATCATTGGCTGTATCAAGCTCATCCCCACTGCCCGCTTCTACCCGCTGCGAATGCACTGCATCCGTGCCCTGACGCTGCTCTCGGGGAGCTCGGGGGCCTTCATCCCGGTGCTGCCTTTCATCCTGGAGATGTTCCAGCAGGTCGACTTCAACAGGAAGCCAGGGCGCATGAGCTCCAAGCCCATCAACTTCTCCGTGATCCTGAAGCTGTCCAATGTCAACCTGCAGGAGAAGGCGTACCGGGACGGCCTGGTGGAGCAGCTGTACGACCTCACCCTGGAGTACCTGCACAGCCAGGCACACTGCATCGGCTTCCCGGAGCTGGTGCTGCCTGTGGTCCTGCAGCTGAAGTCGTTCCTCCGGGAGTGCAAGGTGGCCAACTACTGCCGGCAGGTGCAGCAGCTGCTTGGGAAGGTTCAGGAGAACTCGGCATACATCTGCAGCCGCCGCCAGAGGGTTTCCTTCGGCGTCTCTGAGCAGCAGGCAGTGGAAGCCTGGGAGAAGCTGACCCGGGAAGAGGGGACACCCCTGACCTTGTACTACAGCCACTGGCGCAAGCTGCGTGACCGGGAGATCCAGCTGGAGATCAGTGGCAAAGAGCGGCTGGAAGACCTGAACTTCCCTGAGATCAAACGAAGGAAGATGGCTGACAGGAAGGATGAGGACAGGAAGCAATTTAAAGACCTCTTTGACCTGAACAGCTCTGAAGAGGACGACACCGAGGGATTCTCGGAGAGAGGGATACTGAGGCCCCTGAGCACTCGGCATGGGGTGGAAGACGATGAAGAGGACGAGGAGGAGGGCGAGGAGGACAGCAGCAACTCGGAGGATGGAGACCCAGACGCAGAGGCGGGGCTGGCCCCTGGGGAGCTGCAGCAGCTGGCCCAGGGGCCGGAGGACGAGCTGGAGGATCTGCAGCTCTCAGAGGACGACTGAGGCAGCCCATCTGGGGGGCCTGTAGGGGCTGCCGGGCTGGTGGCCAGTGTTTCCACCTCCCTGGCAGTCAGGCCTAGAGGCTGGCGTCTGTGCAGTTGGGGGAGGCAGTAGACACGGGACAGGCTTTATTATTTATTTTTCAGCATGAAAGACCAAACGTATCGAGAGCTGGGCTGGGCTGGGCTGGTGTGGCTGCTGAAGCCCCACAGCTGTGGGCTGCTGAAGTCAGCTCCGCGGGGGAGCTGACCCTGACGTCAGCAGACCGAGACCAGTCCCAGTTCCAGGGGGAGGCCTGCAGGCCCCTGGCCCCTTCCACCACCTCTGCCCTCCGTCTGCAGACCTCGTCCATCTGCACCAGGCTCTGCCTTCACTCCCCCAAGTCTTTGAAAATTTGTTCCTTTCCTTTGAAGTCACATTTTCTTTTAAAATTTTTTGTTTTGCATCCGAAACCGAAAGAAATAAAGCGGTGGGAGGCAGGGCCATTGTGTTGA 16 2250 0.6394872221025641 8 +ENST00000338591.8 GGGAGTGAGCGACACAGAGCGGGCCGCCACCGCCGAGCAGCCCTCCGGCAGTCTCCGCGTCCGTTAAGCCCGCGGGTCCTCCGCGAATCGGCGGTGGGTCCGGCAGCCGAATGCAGCCCCGCAGCGAGCGCCCGGCCGGCAGGACGCAGAGCCCGGAGCACGGCAGCCCGGGGCCCGGGCCCGAGGCGCCGCCGCCTCCACCGCCGCAGCCGCCGGCCCCCGAGGCAGAGCGCACGCGGCCCCGGCAGGCTCGGCCCGCAGCCCCCATGGAGGGAGCCGTGCAGCTGCTGAGCCGCGAGGGCCACAGCGTGGCCCACAACTCCAAGCGGCACTACCACGATGCCTTCGTGGCCATGAGCCGCATGCGCCAGCGCGGCCTCCTGTGCGACATCGTCCTGCACGTGGCTGCCAAGGAGATCCGTGCGCACAAAGTGGTGCTGGCCTCCTGCAGCCCCTACTTCCACGCCATGTTCACAAATGAGATGAGCGAGAGCCGCCAGACCCACGTGACGCTGCACGACATCGACCCTCAGGCCTTGGACCAGCTGGTGCAGTTTGCCTACACGGCTGAGATTGTGGTGGGCGAGGGCAATGTGCAGACTCTGCTCCCAGCCGCCAGTCTCCTGCAGCTGAATGGCGTCCGAGACGCTTGCTGCAAGTTTCTACTGAGTCAGCTCGACCCCTCCAACTGCCTGGGTATCCGGGGCTTTGCCGATGCGCACTCCTGCAGCGACCTGCTCAAGGCCGCCCACAGGTACGTGCTGCAGCACTTCGTGGACGTGGCCAAGACCGAGGAGTTTATGCTGCTGCCCCTGAAACAGGTTCTGGAACTGGTCTCTAGCGACAGCCTGAACGTGCCTTCAGAGGAGGAGGTCTACCGAGCCGTCCTGAGCTGGGTGAAACACGACGTGGACGCCCGCAGGCAGCATGTCCCACGGCTCATGAAGTGTGTGCGGCTGCCCTTGCTGAGCCGCGACTTCCTGCTGGGCCACGTGGATGCCGAGAGCCTGGTGAGGCACCACCCTGACTGCAAGGACCTCCTCATCGAGGCCCTGAAGTTCCACCTGCTGCCTGAGCAGAGGGGCGTCCTAGGCACCAGCCGCACACGTCCCCGGCGCTGCGAGGGGGCCGGGCCTGTGCTTTTTGCTGTGGGCGGCGGGAGCCTGTTTGCCATCCACGGAGACTGTGAGGCCTACGACACGCGCACCGACCGCTGGCACGTGGTGGCCTCCATGTCCACGCGCCGGGCCCGGGTGGGAGTGGCTGCGGTGGGGAACCGGCTCTATGCTGTGGGCGGCTATGATGGGACCTCAGACCTGGCTACCGTGGAGTCCTACGACCCCGTGACTAACACGTGGCAGCCGGAGGTGTCCATGGGCACAAGGCGAAGCTGCCTGGGTGTGGCCGCCTTGCATGGACTCCTGTACTCGGCCGGCGGCTATGACGGGGCCTCCTGCCTGAACAGTGCTGAACGCTACGACCCCCTGACCGGAACGTGGACGTCCGTCGCTGCCATGAGCACCCGGAGGCGCTATGTGCGAGTGGCCACGCTTGATGGGAACCTGTATGCTGTGGGCGGCTACGACAGCTCCTCACACCTGGCCACTGTGGAGAAGTATGAGCCCCAGGTGAACGTGTGGTCGCCCGTGGCGTCCATGCTGAGCCGACGCAGCTCAGCGGGCGTGGCCGTGCTGGAGGGTGCCCTGTACGTGGCAGGGGGCAACGACGGCACCAGCTGCCTCAACTCGGTAGAGAGATACAGTCCAAAGGCTGGAGCCTGGGAAAGCGTGGCGCCCATGAATATCCGCAGGAGCACGCATGACCTGGTGGCCATGGACGGATGGTTGTACGCCGTGGGGGGTAACGACGGTAGCTCCAGCCTCAACTCCATCGAGAAGTACAACCCGAGGACCAACAAGTGGGTGGCCGCATCCTGCATGTTCACCCGGCGCAGCAGTGTGGGTGTGGCGGTGCTGGAGCTGCTCAATTTCCCGCCGCCATCCTCCCCGACGCTGTCCGTGTCCTCCACCAGCCTCTGACCCACCTACCACCAGAGGCCTGCAGCCTCCCACATGCCTTAAGGGGACCGTGGCCCCCACCAGGGACGTCCTGCGCCATCCGTTCACGTCTCTGCATCCATTCCTTCATGTCTTTATTTAGTTGTTTATTTATTTAGTTATTTATCTTATTTATTGAGGGGTGAGGAGTGCCACGGCTGCCCGTTTACACCTTTAGCGTCTGGTCCTCCTGCGTGTCCTCCCCTCCACTGCCTGCATGGGGGGCGCGGGGAGTGACCAGGCGGGGGCCTCACCGCCCCAGGGCCGTTGCCTGCTCAGACCTTGCAGGCTGTGGAGCAAGAGGCCCTGGGTCTCTCCAAGCAGCTGCAGACCCCAGCTCGAATTTTGCACATGGCGGGGTCCCGGGAAGGGTGGGGAGCAGTTGTCCTTCCTGTCGTCGTCTGCCGTGTGCCATCTTTCCTGGATCTTGTAGTGGGTGCACACGCGTGCACTGGGACCCCACACAGCAATACGAGTCCAACTTAATAAACACATTTCTGGGGTTCCTCA 110 1929 -0.7954436229032259 9 +ENST00000304952.11 GCGGGCCTGGAGCCGGGATCCGCCCTAGGGGCTCGGATCGCCGCGCGCTCGCCGCTCGCCCGCCAGCCCGCCCGTGGTCCGTGGCGGCGCGCTCCACCCGGCACGGGGAGGCGCGGGGCGCACCATGGCCGCAGACACGCCGGGGAAACCGAGCGCCTCGCCGATGGCAGGAGCGCCGGCCAGCGCCAGCCGGACCCCAGACAAGCCCCGGAGCGCGGCCGAGCACCGCAAGTCCTCCAAGCCGGTCATGGAGAAGCGGCGCCGAGCGCGTATTAACGAGAGCCTCGCTCAGCTCAAAACCCTCATCCTGGACGCCCTCAGAAAAGAGAGCTCCCGCCACTCGAAGCTGGAGAAGGCGGACATCCTGGAGATGACCGTGAGACACCTGCGGAGCCTGCGTCGCGTGCAGGTGACGGCCGCGCTCAGCGCCGACCCCGCCGTTCTGGGCAAGTACCGCGCCGGCTTCCACGAGTGTCTGGCGGAGGTGAACCGCTTCCTGGCCGGCTGCGAGGGCGTCCCGGCCGACGTGCGCTCCCGCCTGCTGGGCCACCTGGCAGCCTGCCTGCGCCAGCTGGGACCCTCCCGCCGCCCGGCCTCGCTGTCCCCGGCTGCCCCCGCAGAGGCCCCAGCGCCCGAGGTCTACGCGGGCCGCCCGCTGCTGCCATCGCTCGGCGGCCCCTTCCCTCTGCTCGCGCCGCCGCTGCTGCCGGGTCTGACCCGGGCGCTGCCCGCCGCCCCCAGGGCGGGGCCGCAGGGCCCGGGTGGGCCCTGGAGGCCGTGGCTGCGCTGAGGCTGTGGCCCTGAGACTGCATCGGAGGCGGCGCCCCGTTCTAGGGCCGTGGCCTTTGCCGAGACTGTAGCAGAGAAAACGTATTTATTATTCCA 124 666 -0.08823092367164176 7 +ENST00000649529.1 GGCGGCTGAGAGGCAGCGAACTCATCTTTGCCAGTACAGGAGCTTGTGCCGTGGCCCACAGCCCACAGCCCACAGCCATGGGCTGGGACCTGACGGTGAAGATGCTGGCGGGCAACGAATTCCAGGTGTCCCTGAGCAGCTCCATGTCGGTGTCAGAGCTGAAGGCGCAGATCACCCAGAAGATCGGCGTGCACGCCTTCCAGCAGCGTCTGGCTGTCCACCCGAGCGGTGTGGCGCTGCAGGACAGGGTCCCCCTTGCCAGCCAGGGCCTGGGCCCCGGCAGCACGGTCCTGCTGGTGGTGGACAAATGCGACGAACCTCTGAGCATCCTGGTGAGGAATAACAAGGGCCGCAGCAGCACCTACGAGGTACGGCTGACGCAGACCGTGGCCCACCTGAAGCAGCAAGTGAGCGGGCTGGAGGGTGTGCAGGACGACCTGTTCTGGCTGACCTTCGAGGGGAAGCCCCTGGAGGACCAGCTCCCGCTGGGGGAGTACGGCCTCAAGCCCCTGAGCACCGTGTTCATGAATCTGCGCCTGCGGGGAGGCGGCACAGAGCCTGGCGGGCGGAGCTAAGGGCCTCCACCAGCATCCGAGCAGGATCAAGGGCCGGAAATAAAGGCTGTTGTAAAGAGAAA 77 498 0.7913875147051284 2 +ENST00000379370.7 AGTCCCGTCCCCGGCGCGGCCCGCGCGCTCCTCCGCCGCCTCTCGCCTGCGCCATGGCCGGCCGGTCCCACCCGGGCCCGCTGCGGCCGCTGCTGCCGCTCCTTGTGGTGGCCGCGTGCGTCCTGCCCGGAGCCGGCGGGACATGCCCGGAGCGCGCGCTGGAGCGGCGCGAGGAGGAGGCGAACGTGGTGCTCACCGGGACGGTGGAGGAGATCCTCAACGTGGACCCGGTGCAGCACACGTACTCCTGCAAGGTTCGGGTCTGGCGGTACTTGAAGGGCAAAGACCTGGTGGCCCGGGAGAGCCTGCTGGACGGCGGCAACAAGGTGGTGATCAGCGGCTTTGGAGACCCCCTCATCTGTGACAACCAGGTGTCCACTGGGGACACCAGGATCTTCTTTGTGAACCCTGCACCCCCATACCTGTGGCCAGCCCACAAGAACGAGCTGATGCTCAACTCCAGCCTCATGCGGATCACCCTGCGGAACCTGGAGGAGGTGGAGTTCTGTGTGGAAGATAAACCCGGGACCCACTTCACTCCAGTGCCTCCGACGCCTCCTGATGCGTGCCGGGGAATGCTGTGCGGCTTCGGCGCCGTGTGCGAGCCCAACGCGGAGGGGCCGGGCCGGGCGTCCTGCGTCTGCAAGAAGAGCCCGTGCCCCAGCGTGGTGGCGCCTGTGTGTGGGTCGGACGCCTCCACCTACAGCAACGAATGCGAGCTGCAGCGGGCGCAGTGCAGCCAGCAGCGCCGCATCCGCCTGCTCAGCCGCGGGCCGTGCGGCTCGCGGGACCCCTGCTCCAACGTGACCTGCAGCTTCGGCAGCACCTGTGCGCGCTCGGCCGACGGGCTGACGGCCTCGTGCCTGTGCCCCGCGACCTGCCGTGGCGCCCCCGAGGGGACCGTCTGCGGCAGCGACGGCGCCGACTACCCCGGCGAGTGCCAGCTCCTGCGCCGCGCCTGCGCCCGCCAGGAGAATGTCTTCAAGAAGTTCGACGGCCCTTGTGACCCCTGTCAGGGCGCCCTCCCTGACCCGAGCCGCAGCTGCCGTGTGAACCCGCGCACGCGGCGCCCTGAGATGCTCCTACGGCCCGAGAGCTGCCCTGCCCGGCAGGCGCCAGTGTGTGGGGACGACGGAGTCACCTACGAAAACGACTGTGTCATGGGCCGATCGGGGGCCGCCCGGGGTCTCCTCCTGCAGAAAGTGCGCTCCGGCCAGTGCCAGGGTCGAGACCAGTGCCCGGAGCCCTGCCGGTTCAATGCCGTGTGCCTGTCCCGCCGTGGCCGTCCCCGCTGCTCCTGCGACCGCGTCACCTGTGACGGGGCCTACAGGCCCGTGTGTGCCCAGGACGGGCGCACGTATGACAGTGATTGCTGGCGGCAGCAGGCTGAGTGCCGGCAGCAGCGTGCCATCCCCAGCAAGCACCAGGGCCCGTGTGACCAGGCCCCGTCCCCATGCCTCGGGGTGCAGTGTGCATTTGGGGCGACGTGTGCTGTGAAGAACGGGCAGGCAGCGTGTGAATGCCTGCAGGCGTGCTCGAGCCTCTACGATCCTGTGTGCGGCAGCGACGGCGTCACATACGGCAGCGCGTGCGAGCTGGAGGCCACGGCCTGTACCCTCGGGCGGGAGATCCAGGTGGCGCGCAAAGGACCCTGTGACCGCTGCGGGCAGTGCCGCTTTGGAGCCCTGTGCGAGGCCGAGACCGGGCGCTGCGTGTGCCCCTCTGAATGCGTGGCTTTGGCCCAGCCCGTGTGTGGCTCCGACGGGCACACGTACCCCAGCGAGTGCATGCTGCACGTGCACGCCTGCACACACCAGATCAGCCTGCACGTGGCCTCAGCTGGACCCTGTGAGACCTGTGGAGATGCCGTGTGTGCTTTTGGGGCTGTGTGCTCCGCAGGGCAGTGTGTGTGTCCCCGGTGTGAGCACCCCCCGCCCGGCCCCGTGTGTGGCAGCGACGGTGTCACCTACGGCAGTGCCTGCGAGCTACGGGAAGCCGCCTGCCTCCAGCAGACACAGATCGAGGAGGCCCGGGCAGGGCCGTGCGAGCAGGCCGAGTGCGGTTCCGGAGGCTCTGGCTCTGGGGAGGACGGTGACTGTGAGCAGGAGCTGTGCCGGCAGCGCGGTGGCATCTGGGACGAGGACTCGGAGGACGGGCCGTGTGTCTGTGACTTCAGCTGCCAGAGTGTCCCAGGCAGCCCGGTGTGCGGCTCAGATGGGGTCACCTACAGCACCGAGTGTGAGCTGAAGAAGGCCAGGTGTGAGTCACAGCGAGGGCTCTACGTAGCGGCCCAGGGAGCCTGCCGAGGCCCCACCTTCGCCCCGCTGCCGCCTGTGGCCCCCTTACACTGTGCCCAGACGCCCTACGGCTGCTGCCAGGACAATATCACCGCAGCCCGGGGCGTGGGCCTGGCTGGCTGCCCCAGTGCCTGCCAGTGCAACCCCCATGGCTCTTACGGCGGCACCTGTGACCCAGCCACAGGCCAGTGCTCCTGCCGCCCAGGTGTGGGGGGCCTCAGGTGTGACCGCTGTGAGCCTGGCTTCTGGAACTTTCGAGGCATCGTCACCGATGGCCGGAGTGGCTGTACACCCTGCAGCTGTGATCCCCAAGGCGCCGTGCGGGATGACTGTGAGCAGATGACGGGGCTGTGCTCGTGTAAGCCCGGGGTGGCTGGACCCAAGTGTGGGCAGTGTCCAGACGGCCGTGCCCTGGGCCCCGCGGGCTGTGAAGCTGACGCTTCTGCGCCTGCGACCTGTGCGGAGATGCGCTGTGAGTTCGGTGCGCGGTGCGTGGAGGAGTCTGGCTCAGCCCACTGTGTCTGCCCGATGCTCACCTGTCCAGAGGCCAACGCTACCAAGGTCTGTGGGTCAGATGGAGTCACATACGGCAACGAGTGTCAGCTGAAGACCATCGCCTGCCGCCAGGGCCTGCAAATCTCTATCCAGAGCCTGGGCCCGTGCCAGGAGGCTGTTGCTCCCAGCACTCACCCGACATCTGCCTCCGTGACTGTGACCACCCCAGGGCTCCTCCTGAGCCAGGCACTGCCGGCCCCCCCCGGCGCCCTCCCCCTGGCTCCCAGCAGTACCGCACACAGCCAGACCACCCCTCCGCCCTCATCACGACCTCGGACCACTGCCAGCGTCCCCAGGACCACCGTGTGGCCCGTGCTGACGGTGCCCCCCACGGCACCCTCCCCTGCACCCAGCCTGGTGGCGTCCGCCTTTGGTGAATCTGGCAGCACTGATGGAAGCAGCGATGAGGAACTGAGCGGGGACCAGGAGGCCAGTGGGGGTGGCTCTGGGGGGCTCGAGCCCTTGGAGGGCAGCAGCGTGGCCACCCCTGGGCCACCTGTCGAGAGGGCTTCCTGCTACAACTCCGCGTTGGGCTGCTGCTCTGATGGGAAGACGCCCTCGCTGGACGCAGAGGGCTCCAACTGCCCCGCCACCAAGGTGTTCCAGGGCGTCCTGGAGCTGGAGGGCGTCGAGGGCCAGGAGCTGTTCTACACGCCCGAGATGGCTGACCCCAAGTCAGAACTGTTCGGGGAGACAGCCAGGAGCATTGAGAGCACCCTGGACGACCTCTTCCGGAATTCAGACGTCAAGAAGGATTTTCGGAGTGTCCGCTTGCGGGACCTGGGGCCCGGCAAATCCGTCCGCGCCATTGTGGATGTGCACTTTGACCCCACCACAGCCTTCAGGGCACCCGACGTGGCCCGGGCCCTGCTCCGGCAGATCCAGGTGTCCAGGCGCCGGTCCTTGGGGGTGAGGCGGCCGCTGCAGGAGCACGTGCGATTTATGGACTTTGACTGGTTTCCTGCGTTTATCACGGGGGCCACGTCAGGAGCCATTGCTGCGGGAGCCACGGCCAGAGCCACCACTGCATCGCGCCTGCCGTCCTCTGCTGTGACCCCTCGGGCCCCGCACCCCAGTCACACAAGCCAGCCCGTTGCCAAGACCACGGCAGCCCCCACCACACGTCGGCCCCCCACCACTGCCCCCAGCCGTGTGCCCGGACGTCGGCCCCCGGCCCCCCAGCAGCCTCCAAAGCCCTGTGACTCACAGCCCTGCTTCCACGGGGGGACCTGCCAGGACTGGGCATTGGGCGGGGGCTTCACCTGCAGCTGCCCGGCAGGCAGGGGAGGCGCCGTCTGTGAGAAGGTGCTTGGCGCCCCTGTGCCGGCCTTCGAGGGCCGCTCCTTCCTGGCCTTCCCCACTCTCCGCGCCTACCACACGCTGCGCCTGGCACTGGAATTCCGGGCGCTGGAGCCTCAGGGGCTGCTGCTGTACAATGGCAACGCCCGGGGCAAGGACTTCCTGGCATTGGCGCTGCTAGATGGCCGCGTGCAGCTCAGGTTTGACACAGGTTCGGGGCCGGCGGTGCTGACCAGTGCCGTGCCGGTAGAGCCGGGCCAGTGGCACCGCCTGGAGCTGTCCCGGCACTGGCGCCGGGGCACCCTCTCGGTGGATGGTGAGACCCCTGTTCTGGGCGAGAGTCCCAGTGGCACCGACGGCCTCAACCTGGACACAGACCTCTTTGTGGGCGGCGTACCCGAGGACCAGGCTGCCGTGGCGCTGGAGCGGACCTTCGTGGGCGCCGGCCTGAGGGGGTGCATCCGTTTGCTGGACGTCAACAACCAGCGCCTGGAGCTTGGCATTGGGCCGGGGGCTGCCACCCGAGGCTCTGGCGTGGGCGAGTGCGGGGACCACCCCTGCCTGCCCAACCCCTGCCATGGCGGGGCCCCATGCCAGAACCTGGAGGCTGGAAGGTTCCATTGCCAGTGCCCGCCCGGCCGCGTCGGACCAACCTGTGCCGATGAGAAGAGCCCCTGCCAGCCCAACCCCTGCCATGGGGCGGCGCCCTGCCGTGTGCTGCCCGAGGGTGGTGCTCAGTGCGAGTGCCCCCTGGGGCGTGAGGGCACCTTCTGCCAGACAGCCTCGGGGCAGGACGGCTCTGGGCCCTTCCTGGCTGACTTCAACGGCTTCTCCCACCTGGAGCTGAGAGGCCTGCACACCTTTGCACGGGACCTGGGGGAGAAGATGGCGCTGGAGGTCGTGTTCCTGGCACGAGGCCCCAGCGGCCTCCTGCTCTACAACGGGCAGAAGACGGACGGCAAGGGGGACTTCGTGTCGCTGGCACTGCGGGACCGCCGCCTGGAGTTCCGCTACGACCTGGGCAAGGGGGCAGCGGTCATCAGGAGCAGGGAGCCAGTCACCCTGGGAGCCTGGACCAGGGTCTCACTGGAGCGAAACGGCCGCAAGGGTGCCCTGCGTGTGGGCGACGGCCCCCGTGTGTTGGGGGAGTCCCCGGTTCCGCACACCGTCCTCAACCTGAAGGAGCCGCTCTACGTAGGGGGCGCTCCCGACTTCAGCAAGCTGGCCCGTGCTGCTGCCGTGTCCTCTGGCTTCGACGGTGCCATCCAGCTGGTCTCCCTCGGAGGCCGCCAGCTGCTGACCCCGGAGCACGTGCTGCGGCAGGTGGACGTCACGTCCTTTGCAGGTCACCCCTGCACCCGGGCCTCAGGCCACCCCTGCCTCAATGGGGCCTCCTGCGTCCCGAGGGAGGCTGCCTATGTGTGCCTGTGTCCCGGGGGATTCTCAGGACCGCACTGCGAGAAGGGGCTGGTGGAGAAGTCAGCGGGGGACGTGGATACCTTGGCCTTTGACGGGCGGACCTTTGTCGAGTACCTCAACGCTGTGACCGAGAGCGAGAAGGCACTGCAGAGCAACCACTTTGAACTGAGCCTGCGCACTGAGGCCACGCAGGGGCTGGTGCTCTGGAGTGGCAAGGCCACGGAGCGGGCAGACTATGTGGCACTGGCCATTGTGGACGGGCACCTGCAACTGAGCTACAACCTGGGCTCCCAGCCCGTGGTGCTGCGTTCCACCGTGCCCGTCAACACCAACCGCTGGTTGCGGGTCGTGGCACATAGGGAGCAGAGGGAAGGTTCCCTGCAGGTGGGCAATGAGGCCCCTGTGACCGGCTCCTCCCCGCTGGGCGCCACGCAGCTGGACACTGATGGAGCCCTGTGGCTTGGGGGCCTGCCGGAGCTGCCCGTGGGCCCAGCACTGCCCAAGGCCTACGGCACAGGCTTTGTGGGCTGCTTGCGGGACGTGGTGGTGGGCCGGCACCCGCTGCACCTGCTGGAGGACGCCGTCACCAAGCCAGAGCTGCGGCCCTGCCCCACCCCATGAGCTGGCACCAGAGCCCCGCGCCCGCTGTAATTATTTTCTATTTTTGTAAACTTGTTGCTTTTTGATATGATTTTCTTGCCTGAGTGTTGGCCGGAGGGACTGCTGGCCCGGCCTCCCTTCCGTCCAGGCAGCCGTGCTGCAGACAGACCTAGTGCCGAGGGATGGACAGGCGAGGTGGCAGCGTGGAGGGCTCGGCGTGGATGGCAGCCTCAGGACACACACCCCTGCCTCAAGGTGCTGAGCCCCCGCCTTGCACTGCGCCTGCCCCACGGTGTCCCCGCCGGGAAGCAGCCCCGGCTCCTGAATCACCCTCGCTCCGTCAGGCGGGACTCGTGTCCCAGAGAGGAAGGGGCTGCTGAGGTCTGATGGGGCCCTTCCTCCGGGTGACCCCACAGGGCCTTTCCAAGCCCCCATTTGAGCTGCTCCTTCCTGTGTGTGCTCTGGGCCCTGCCTCGGCCTCCTGCGCCAATACTGTGACTTCCAAACAATGTTACTGCTGGGCACAGCTCTGCGTTGCTCCCGTGCTGCCTGCGCCAGCCCCAGGCTGCTGAGGAGCAGAGGCCAGACCAGGGCCGATCTGGGTGTCCTGACCCTCAGCTGGCCCTGCCCAGCCACCCTGGACGTGACCGTATCCCTCTGCCACACCCCAGGCCCTGCGAGGGGCTATCGAGAGGAGCTCACTGTGGGATGGGGTTGACCTCTGCCGCCTGCCTGGGTATCTGGGCCTGGCCATGGCTGTGTTCTTCATGTGTTGATTTTATTTGACCCCTGGAGTGGTGGGTCTCATCTTTCCCATCTCGCCTGAGAGCGGCTGAGGGCTGCCTCACTGCAAATCCTCCCCACAGCGTCAGTGAAAGTCGTCCTTGTCTCAGAATGACCAGGGGCCAGCCAGTGTCTGACCAAGGTCAAGGGGCAGGTGCAGAGGTGGCAGGGATGGCTCCGAAGCCAGAAATGCCTTAAACTGCAACGTCCCGTCCCTTCCCCACCCCCATCCCATCCCCACCCCCAGCCCCAGCCCAGTCCTCCTAGGAGCAGGACCCGATGAAGCGGGCGGCGGTGGGGCTGGGTGCCGTGTTACTAACTCTAGTATGTTTCTGTGTCAATCGCTGTGAAATAAAGTCTGAAAACTTTAAAA 53 6138 0.09680575867567566 4 +ENST00000421241.6 GGGCGGGGTGTACGAAAGAGAAACCCGGAGGGCGCCGGGGACTGGGCCGGGGTCTGCAGGGCTCAGCTGAGCCCATGAGCTCCCAGAGCTAACCCCTGAACACCCAGGCGGGCAAAGGGCTGATGTCGGTAGTCCCCATCCTGGAGGGGCAGGCTCTGCGCATCTGCTCCTGGCATGGCGCTGCGGCACCTCGCCCTCCTGGCTGGCCTTCTCGTGGGAGTCGCCAGCAAGTCCATGGAGAACACGGCCCAGCTGCCCGAGTGCTGTGTGGATGTGGTGGGCGTCAACGCCAGCTGCCCAGGCGCAAGTCTGTGTGGTCCAGGCTGTTACAGGCGCTGGAACGCGGACGGGAGCGCCAGCTGCGTCCGCTGTGGGAACGGAACCCTCCCAGCCTACAACGGCTCCGAGTGTAGAAGCTTTGCTGGCCCGGGTGCGCCATTCCCCATGAACAGAAGCTCAGGGACCCCCGGGCGGCCACATCCTGGGGCTCCGCGCGTGGCCGCCTCCCTCTTCCTGGGCACGTTCTTCATTAGCTCCGGCCTCATCCTCTCCGTAGCTGGGTTCTTCTACCTCAAGCGCTCCAGTAAACTCCCCAGGGCCTGCTACAGAAGAAACAAAGCTCCGGCCCTGCAGCCTGGCGAAGCCGCTGCAATGATCCCCCCGCCACAGTCCTCAGTACGGAAGCCGCGCTACGTCAGGCGGGAGCGGCCCCTGGACAGGGCCACGGATCCCGCTGCCTTCCCGGGGGAGGCCCGTATCAGCAATGTCTGACCTGGAGGCCGAGACCACGCCACGCACTTGGCGGCAGGGACCCGGAGGCCGACCCCTTGGCGGGAACCAGCACAAAGTGTTGGCATCGCCCGGCGCCCGGGACAGTCCTGGGCACAGCCTCGGCTCTGAGTCCCTCCGCCTCCCAGCGACGGACGCCAAAGGGTCCCGGGCCGCCTGAGGCTCCTCCCCACCACAGCCATCTCGTTTATCGGACCAGGAGCAGGCATCCATGAGACCTCAGAGCTTCAGATCGAGGCCTTGGGGGGTCCGGGCCCCCCCAGGAAACACGGTGAGGCCCCAGCGCCTGCAGCCAAAGCTGGCACGATCTATGGGGCAGGTGCCGCTCTGCCTAGAAAAGCCAGGGGCTCTGCTGCCGTGCCCTCCAGAGCCCACAGCGGGCAGGACTCCTCCAGCACCACCACACCCAGTGGCCCGAGACCCCTCTGAGAACAGTGAGGCTGGTCCTCGTGCCGTTCCAGCCGGTGCCCGGCCAGTGGGGAGGACACAGCCTAGGAACCAGCTGCCTGAGACCAGGGTGCCTCTGGGCTGTCCTCCCGCGTGGCGGAGACCCCAAGCACGCAGCCACCCATTTCCGGAGCTGCAGGATAGAGCTTCCTCTTGATCTCTGTTTTTAAGCAGAAATTCATTGTGCAGAAAAGTCCTCCAGAGCTCTGTGGCCCCGCTCGGATCCGCTGGACCCCCATGCCTGGCTGATCCCTGCCCACGTGGGGCAGGCCCACATCTAACCCCCACAAGTCACTGCCTCACTGCACCTGCCAAGGCTGCCCTGGCGCTGAGTCCTGGGGTCCCTCCCGGAGTTCCTGGGAGAAAGGCGCCGTCGTGGCCGCCTCCCGCACGCCAGGCCCGGGCTCCACCGTGGGTCTCAGACGCCCTGCGGCACCGGCACCGTCTGCTTTAGCATGGGACCCCCCTCTGAGGGGTGGCCTGGCCTTCGGGGTCCCCACGCTCCTTTGCGAAGTCCACTGTGGGTGCCATCATGGTCTCCGGGACCTGGGCCAGCGGGAACGTGGGGGCACTGGGTGTGCTGATATAAAGTCGGCATTACTCAA 174 597 -0.1919400409857143 0 +ENST00000360001.11 GCACCGCCCCCGCCCGCAAGAAAGATGGCAGTGGCCTGATCCGGGCCCGTTGGCGGCGTCACTGACGCTTCGCTCCGGTCCTCGGATCCCGAGCGCGGGGAGGCAGACCGACTGTGAGCTGCTTGTCCCCATCCTGCGGCCGTCCTGGGGACACAGAGCCCTCCGTGGTGCCCGGGGATTGGATTGGAGCCAGGACCTCACTTCCTCCTCTGCCCCTGCCCCTGCCCCTCCCAGCACCTGGCCCACACCCTGCAGCCCGCCCCATGGTCTGGCCCTGGGTGGCGATGGCGTCCAGGTGGGGTCCCCTCATTGGCCTGGCTCCGTGCTGCCTCTGGCTCCTGGGGGCAGTCCTTCTGATGGACGCGTCTGCACGGCCTGCCAACCACTCGTCCACTCGAGAGAGAGTAGCCAACAGGGAGGAGAATGAGATCCTGCCCCCAGACCACCTGAACGGGGTGAAGCTGGAGATGGACGGGCACCTCAATCGCGGCTTCCACCAGGAGGTCTTCCTAGGCAAGGACCTGGGTGGCTTTGATGAGGACGCGGAGCCGCGGCGGAGCCGGAGGAAGCTGATGGTCATCTTTTCCAAGGTGGATGTGAACACTGACCGGAAGATCAGTGCCAAGGAGATGCAGCGCTGGATCATGGAGAAGACGGCCGAGCACTTCCAGGAGGCCATGGAGGAGAGCAAGACACACTTCCGCGCCGTGGACCCTGACGGGGACGGTCACGTGTCTTGGGACGAGTATAAGGTGAAGTTTTTGGCGAGTAAAGGCCATAGCGAGAAGGAGGTTGCCGACGCCATCAGGCTCAACGAGGAACTCAAAGTGGATGAGGAAACACAGGAAGTCCTGGAGAACCTGAAGGACCGCTGGTACCAGGCGGACAGCCCCCCTGCAGACCTGCTGCTGACGGAGGAGGAGTTCCTGTCGTTCCTCCACCCCGAGCACAGCCGGGGAATGCTCAGGTTCATGGTGAAGGAGATCGTCCGGGACCTGGACCAGGACGGTGACAAGCAGCTCTCTGTGCCCGAGTTCATCTCCCTGCCCGTGGGCACCGTGGAGAACCAGCAGGGCCAGGACATTGACGACAACTGGGTGAAAGACAGAAAAAAGGAGTTTGAGGAGCTCATTGACTCCAACCACGACGGCATCGTGACCGCCGAGGAGCTGGAGAGCTACATGGACCCCATGAACGAGTACAACGCGCTGAACGAGGCCAAGCAGATGATCGCCGTCGCCGACGAGAACCAGAACCACCACCTGGAGCCCGAGGAGGTGCTCAAGTACAGCGAGTTCTTCACGGGCAGCAAGCTGGTGGACTACGCGCGCAGCGTGCACGAGGAGTTTTGAGCGCCCGGCCGCGCCCCGCGCCGCCCCCCACGCACCACCGGGGCGGCCTCGCGGGTGACTCCGGGCTCCGTGGCTGTCCCGGACCCCACCTCTTCCCTGCCGCCCGCCACCGGCCGACCGACCGCGGCTGCCCCAGTTGATGAGCGGCGTGTCCCCTCTGCAGCGCGCACCCCGGCGGGGCTTTGGCTGTGACGCGGTCGGGGCGCGGGGCTGGGCTGTGGCCCCGCGGCGCCGCCTCCTCCCTGGTCCCTCGAAATCGTGGCATCTCACTTCTGAGAACGAAATCTCGCTTCAGTCACTCTGCCGAAGGCGCTGACGGCATCGCGGCCGGAACCTCTGGGCCCGGCCCCTCCCAGGGCCGCCGCTCCGTGGGAAAAAACAGCTCCTCCATTTCCTTGAAAACTGAACGATTATTAAAAATAGATTAAACTTCGCTGGAAATGAGTAGCCAGGAAGTTCAGGGGAGGGTGCCGGGTCCTTCCCGGGCCTGGCGTGTCGGAGCCACCCAGGTCCCGCAGCTGCCGCTGAGAAAATGCAAATATTTGTTGTGACAAGAATCACATACATTTACTTTAAATATAGTTGCCTTTTTTGGTCAGCTTCA 284 1068 0.734010135871795 4 +ENST00000379198.5 ACTCGCGAGTCCGGCCTGGGCCGCCGGCCCGGCGCGGGCGCCATGAAGCTGCTGCGGCGGGCGTGGCGGCGGCGGGCGGCGCTAGGCCTGGGCACGCTGGCGCTGTGCGGGGCGGCGCTGCTCTACCTGGCGCGCTGCGCGGCCGAGCCCGGGGACCCCAGGGCGATGTCGGGCCGCAGCCCGCCTCCCCCCGCGCCCGCGCGCGCCGCCGCCTTCCTGGCAGTGCTGGTGGCCAGCGCGCCCCGCGCCGCCGAGCGCCGCAGCGTGATCCGCAGCACGTGGCTTGCGCGGCGCGGGGCCCCGGGCGACGTGTGGGCGCGCTTTGCCGTGGGCACGGCCGGCCTGGGCGCCGAGGAGCGGCGCGCCCTGGAGCGGGAGCAGGCGCGGCACGGGGACCTGCTGCTGCTGCCCGCGCTGCGCGACGCCTACGAAAACCTCACGGCCAAGGTGCTGGCCATGCTGGCCTGGCTGGACGAGCACGTGGCCTTCGAGTTCGTGCTCAAGGCGGACGACGACTCCTTCGCGCGGCTGGACGCGCTGCTGGCCGAGCTGCGCGCCCGCGAGCCCGCGCGCCGCCGCCGCCTCTACTGGGGCTTCTTCTCGGGCCGCGGCCGCGTCAAGCCGGGGGGGCGCTGGCGCGAGGCCGCCTGGCAACTCTGCGACTACTACCTGCCCTACGCGCTGGGCGGCGGCTACGTGCTCTCGGCCGACCTGGTGCACTACCTGCGCCTCAGCCGCGACTACCTGCGCGCCTGGCACAGCGAGGACGTGTCTCTGGGCGCCTGGCTGGCGCCGGTGGACGTCCAGCGGGAGCACGACCCGCGCTTCGACACCGAATACCGGTCCCGCGGCTGCAGCAACCAGTACCTGGTGACGCACAAGCAGAGCCTGGAGGACATGCTGGAGAAGCACGCGACGCTGGCGCGCGAGGGCCGCCTGTGCAAGCGCGAGGTGCAGCTGCGCCTGTCCTACGTGTACGACTGGTCCGCGCCGCCCTCGCAGTGCTGCCAGAGAAGGGAGGGCATCCCCTGAGCCGCCGCGGCCCGGCCCTCCGGGACACCTGCTTCACCCGGCGGCGCCTTGGGGCAGGTGCCGAGCGGGCGCACTACGCCCGGGCCCCAAGGCCCCCGTCCCGCAGCCACGCTTGTGGTCGCTGCGTCCCGGTCTGCGTTTGGGAGACCCCTGGGGGTTGCCGGGGCAGCGCGCCGTGTCCAGGTGGAGGTGCCCGTTCCTGGACCTCAGCGAGCCTGAGCCGGGCCCGGCCGCACGCTGACCCCCGTGCTGTCCCCGACCGGCTCACGGGGCTGGGCTCCGATCTTCCGTGTCTCTTATCAGTGGCGTTTCTCACGTCTGCGTCTCAGATCTAACGTGGTTTCACATCAATCCGCTTTCATGGGATTTTGGTCTCTGTCCAGTGACTTCGTGGTAAATGTAACTCAGTGTTTGCTTGCGACTTATTTATAAATATTGTAAGTTTGTGTCGATGAGTGTAAGTTGGCAGTGCGCACGTCTCGGTTTTTTTACATGATTTAAGGAAAGACTTTTATGTCAGAACTTGGTGCCTGTACCGTCAACCCCGCTGCTGCCCGTGTTTAAACGCAGGAGAACTTTAAAACTGGCCATCTATCTTTTCAGTGTACAAGTCACTGAACCCATTGTTTCTTTCTGAAGAGACTTTCCTTTCAAGGCTTCCCATGGGTCCGCGCCACACAGGGCCGGTGCTGCTTTATTTCAGACTCTGCCCCAGGTTCCAGGAATCCGAACCCCGGAGTGCTGACGCGGTTCCCCAACTTCCGCCTTAAGAAAACAGGACCAGCCGGCACCAGGCCCGTCTCTCACGTACTTTAACACATCCTTGAAAGCCCCTCGTTTAATGAGAAAAGCGAACACTGCGGTCCTTGCCAAAGTAAAATGAAGCTGCCCCAGGACAAGGGGTTACCATGAGCTCCCTGGAGTCCGACGCGGGTTTTCTCTCTGGGGGACCTGGGTGGTCCCCGCTGTGGTCTTTGTTGTCCCACTTTGGGACCGGGTCCAGTCTGGGGTCTAGTCTCGAGCATCAGGGTCAGGCTCGGGGCAGGGCTGGGTTAGGCTCCGGGTCAGTCTTGCCATGGGTTTGGGAGCAGGTTTGGGTTACTTGCGTTTGAAGGCAGCAGTGGTCTCAGGAGGAAGAAACGGGGGCGGGAGAGAGTGGTGATCTGTGGTCAGTGGGTCAGTGACCTGCACGGTGATTCTCCCACCTCCAAAAGGTAGGGGTGGGACTGGAGGCGTCCCTAGGTCAGGCCGTTGAGTTCGAGCTCCGATGGGCCACCTTGAATCCAGGACTGACCGCCCGTGTGTGCACAGTTTGTTCTTGGACGAGGACTCGTGAGGATCGAGGGCTGGGGACCCCGGTGTGAGCAGGATGGGGCCCTGCCCTCCCGTGGGAGTTGTGGACTCGAGCCCAGGGGCTGCCCGTCACAGCGGTGTCCCAGGTCCCTGCCATCCGATTTTACCTGGGATGTCTTCTCTGGAGTTTGGAATTGCTTGAGGAACCCTGCGTGTGCTTGGAGAGGCCAGAGGGCTTGCTGAGAACCCCATGGACAGTGGAGAGCGGGATTCGAACCAAGGGCTGGACTCCCACACCTCTGGCCTGCGTCGCCCAGTTCTTTGTGGCTCTGAAGAATTGGCCGCTGTGGAAAAGAGCAAATGTCCGAGACCCCCAACAGGAAGAGTCTAAAAATCCAGTTTGCAACCACTTCTGACCTACAAAAAAATGGAAATTTAGTGTTTTTCAGCCTAAGACATTAAATTTCATATCAGAACAAA 42 990 0.06170675702777776 4 +ENST00000349431.11 GGTTCCGCCCCGCGAGCGGCCATCTTGGAGGCTGAGGCGGCGGCGGCGGCGCTGCGGCGGGTTCGGTGGGCCCAATCCCGGGGCGGTGCGGCTGTTTCGGGCGCGGGCCCCGCTTTTCCGCACCCTGCTCCGGCCTCGACTACGGCGAGCCTGAGCGCGGCGGCGGCCCACGCGCAGCGACAGGGAGAGATGAGCAGCACCAGCAGTAAGAGGGCTCCGACCACGGCAACCCAGAGGCTGAAGCAGGACTACCTTCGCATTAAGAAAGACCCGGTGCCTTACATCTGTGCCGAGCCCCTCCCTTCGAATATTCTCGAGTGGCACTATGTCGTCCGAGGCCCAGAGATGACCCCTTATGAAGGTGGCTATTATCATGGAAAACTAATTTTTCCCAGAGAATTTCCTTTCAAACCTCCCAGTATCTATATGATCACTCCCAACGGGAGGTTTAAGTGCAACACCAGGCTGTGTCTTTCTATCACGGATTTCCACCCGGACACGTGGAACCCGGCCTGGTCTGTCTCCACCATCCTGACTGGGCTCCTGAGCTTCATGGTGGAGAAGGGCCCCACCCTGGGCAGTATAGAGACGTCGGACTTCACGAAAAGACAACTGGCAGTGCAGAGTTTAGCATTTAATTTGAAAGATAAAGTCTTTTGTGAATTATTTCCTGAAGTCGTGGAGGAGATTAAACAAAAACAGAAAGCACAAGACGAACTCAGTAGCAGACCCCAGACTCTCCCCTTGCCAGACGTGGTTCCAGACGGGGAGACGCACCTCGTCCAGAACGGGATTCAGCTGCTCAACGGGCATGCGCCGGGGGCCGTCCCAAACCTCGCAGGGCTCCAGCAGGCCAACCGGCACCACGGACTCCTGGGTGGCGCCCTGGCGAACTTGTTTGTGATAGTTGGGTTTGCAGCCTTTGCTTACACGGTCAAGTACGTGCTGAGGAGCATCGCGCAGGAGTGAGGCCCAGGCGCCGAGACCCAAGGCGCCACTGAGGGCACCGCGCACCAGAGCGTGACCTCGGCAGGCTGGACACACTGCCCAGCACAGGCAGACCCACCAGGCTCCTAGGTTTAGCTTTTAAAAACCTGAAAGGGGAAGCAAAAACCAAAATGTGTGACTGGGCTTTGGAGGAGACTGGAGCCTCAGCCCTGTCCTGGCCACGGGCCGCTGGGGCTGGTGTGGGTGGGCCTTGTGTGCTGGATTTGTAGCTTATCTTCCGTGTTGTCTTTGGACCTGTTTTAGTAAACCCGTTTTTCATTTTATTAGATGTGGTCACTTAGAAATGCAAACTTGCTGCCGACCGCGGGCTGCTCCTGCGTTCTTGGAGCTCCTGGCGCGTTTCTCGGAGCTCCCGGCTCCTCAGCGGGTGGGAACCTCGGGGCCCAGGGGTGGAGCTGGCGTCCGCGGGTGCTGGTCTGGCCTGGCCGTGTGGTGATGAGGCTTAGCGGGGCCAGTGACGGCCGTGGCTCAGGATCCATAAGTCGGGGTTTGGTCTCAGCATTTACAAATGTGTTTACAGTCAGAATGAAACACATTCCTTCTAGAAAGTGCTTGGGGGTTTTTGCTGCCCTGGAAGCCAGGAGCCTGCTCACTCCAACCACAAGTCGCCCTTGACTGCGGCGGCCGCGAGCGGGGCGGGGGCTGCCGGTGCCCTCCGCAGGCCGGGCCTCCTGGGCGCCCCTCGGTGCTGCAGGCTGGGGGGCCTTGGGTACCTGCAGAGCCTTTTCTCTGAATTCCTTATGTCCGGTGGGCCAGAAGCCCGTCCTCCTATGCTGGTGGAAGGCGGAGGACCGGAGTCCCTGCAGAAGGCCCCGTGCACTCGGGGGCCTCCCTCACATCCCGTGCCCCCTGCGCTGGCCTTCACAGTAGGTAATGGCTCCGGCCCGGGTGTTCGCTGTCCACGGAACATGGCAGAGGGGCACCCCGGCCCGGAAAGACGCCAGAGCCAGCAGGGGCTGTTTCGGGCCGCGTGGCTCCCCGGGTCTCGGCCGTCTCCCCTCTTCTGCGTCTGTTCCGTGACTTCGCCTGGGTGGGATGTACCGCAGGTGCATCGCGTCGAGGTGGGGCACGGCCGCCGGCAAGAAACCCACCCTGTCCGGAGGCGGGCGTGAGACAAGCCCAGCCCGCACGCGCTCATCTTTCTTCGTTTTTTGATCAGTTTATTCAGAATTGCTCTATAATTTACCAATTGTATGTATTTAACCTATTCTTGTGGAAAAAAAAGGTCTTTCATTATATCTTTATTTCTGAA 189 780 0.5229675528311688 6 +ENST00000435064.6 GCAGTGCATCACCGCAGGCGGGCCTCGCGGGTCCGGGAGCGCGGCGGAGACGATGCCTGAGATCAGAGTCACGCCCTTGGGGGCCGGCCAGGACGTGGGCCGAAGCTGCATCCTGGTCTCCATTGCGGGCAAGAATGTCATGCTGGACTGTGGAATGCACATGGGCTTCAATGACGACCGACGCTTCCCTGACTTCTCCTACATCACCCAGAACGGCCGCCTAACAGACTTCCTGGACTGTGTGATCATTAGCCACTTCCACCTGGACCACTGCGGGGCACTCCCCTACTTCAGCGAGATGGTGGGCTACGACGGGCCCATCTACATGACTCACCCCACCCAGGCCATCTGCCCCATCTTGCTGGAGGACTACCGCAAGATCGCCGTAGACAAGAAGGGCGAGGCCAACTTCTTCACCTCCCAGATGATCAAAGACTGCATGAAGAAGGTGGTGGCTGTCCACCTCCACCAGACGGTCCAGGTAGATGATGAGCTGGAGATCAAGGCCTACTATGCAGGCCACGTGCTGGGGGCAGCCATGTTCCAGATTAAAGTGGGCTCAGAGTCTGTGGTCTACACGGGTGATTATAACATGACCCCAGACCGACACTTAGGAGCTGCCTGGATTGACAAGTGCCGCCCCAACCTGCTCATCACAGAGTCCACGTACGCCACGACCATCCGTGACTCCAAGCGCTGCCGGGAGCGAGACTTCCTGAAGAAAGTCCACGAGACCGTGGAGCGTGGTGGGAAGGTGCTGATACCTGTGTTCGCGCTGGGCCGCGCCCAGGAGCTCTGCATCCTCCTGGAGACCTTCTGGGAGCGCATGAACCTGAAGGTGCCCATCTACTTCTCCACGGGGCTGACCGAGAAGGCCAACCACTACTACAAGCTGTTCATCCCCTGGACCAACCAGAAGATCCGCAAGACTTTCGTGCAGAGGAACATGTTTGAGTTCAAGCACATCAAGGCCTTCGACCGGGCTTTTGCTGACAACCCAGGACCGATGGTTGTGTTTGCCACGCCAGGAATGCTGCACGCTGGGCAGTCCCTGCAGATCTTCCGGAAATGGGCCGGAAACGAAAAGAACATGGTCATCATGCCCGGCTACTGCGTGCAGGGCACCGTCGGCCACAAGATCCTCAGCGGGCAGCGGAAGCTCGAGATGGAGGGGCGGCAGGTGCTGGAGGTCAAGATGCAGGTGGAGTACATGTCATTCAGCGCACACGCGGACGCCAAGGGCATCATGCAGCTGGTGGGCCAGGCAGAGCCGGAGAGCGTGCTGCTGGTGCATGGCGAGGCCAAGAAGATGGAGTTCCTGAAGCAGAAGATCGAGCAGGAGCTCCGGGTCAACTGCTACATGCCGGCCAATGGCGAGACGGTGACGCTGCCCACAAGCCCCAGCATCCCCGTAGGCATCTCGCTGGGGCTGCTGAAGCGGGAGATGGCGCAGGGGCTGCTCCCTGAGGCCAAGAAGCCTCGGCTCCTGCACGGCACCCTGATCATGAAGGACAGCAACTTCCGGCTGGTGTCCTCAGAGCAAGCCCTCAAAGAGCTGGGTCTGGCTGAGCACCAGCTGCGCTTCACCTGCCGCGTGCACCTGCATGACACACGCAAGGAGCAGGAGACGGCATTGCGCGTCTACAGCCACCTCAAGAGCGTCCTGAAGGACCACTGTGTGCAGCACCTCCCAGACGGCTCTGTGACTGTGGAGTCCGTCCTCCTCCAGGCCGCCGCCCCTTCTGAGGACCCAGGCACCAAGGTGCTGCTGGTCTCCTGGACCTACCAGGACGAGGAGCTGGGGAGCTTCCTCACATCTCTGCTGAAGAAGGGCCTCCCCCAGGCCCCCAGCTGAGGCCGGCAACTCACCCAGCCGCCACCTCTGCCCTCTCCCAGCTGGACAGACCCTGGGCCTGCACTTCAGGACTGTGGGTGCCCTGGGTGAACAGACCCTGCAGGTCCCATCCCTGGGGACAGAGGCCTTGTGTCACCTGCCTGCCCAGGCAGCTGTTTGCAGCTGAAGAAACAAACTGGTCTCCAGGCTGTCTTGCCTTTATTCCTGGTTAGGGCAGGTGGTCCTAGACAGCAGTTTCCAGTAAAAGCTGAACAAAAGA 52 1803 0.3808095025384615 9 +ENST00000378609.9 AGGGGCCCGCCGCGCCCATCCCGATGGCTGGAGGCGTCTGAGGGGCGGACGGAGGCGGCGGCGGCGGCGGCGGGAGCGGGAGCGGGCGGCGAGTGGGGAGCGGGGCCGGGAGTGGAGCAGCCGCCGCGGCGGGACTGGACCGAGCCTCGCCGGCGCGCACCTGCCCGCAGCGCCCGCGGAGCGCGCAGCGCGGCCCGAGCGCGACGACCTGCCGAGCGGCGGCCGAGGCGGCGGTGTGGGCGCGTCAGGCCGCGACGAGGGCGCTGAGACAAATTTACATGTATTGGAGACCAGACCAGAAGCCCTTCTGAATTAAGATCTCACATTCTTGAAGGTGGCATTGAAGAGCACTAAGATCGGAAGATGAGTGAGCTTGACCAGTTACGGCAGGAGGCCGAGCAACTTAAGAACCAGATTCGAGACGCCAGGAAAGCATGTGCAGATGCAACTCTCTCTCAGATCACAAACAACATCGACCCAGTGGGAAGAATCCAAATGCGCACGAGGAGGACACTGCGGGGGCACCTGGCCAAGATCTACGCCATGCACTGGGGCACAGACTCCAGGCTTCTCGTCAGTGCCTCGCAGGATGGTAAACTTATCATCTGGGACAGCTACACCACCAACAAGGTCCACGCCATCCCTCTGCGCTCCTCCTGGGTCATGACCTGTGCATATGCCCCTTCTGGGAACTATGTGGCCTGCGGTGGCCTGGATAACATTTGCTCCATTTACAATCTGAAAACTCGTGAGGGGAACGTGCGCGTGAGTCGTGAGCTGGCAGGACACACAGGTTACCTGTCCTGCTGCCGATTCCTGGATGACAATCAGATCGTCACCAGCTCTGGAGACACCACGTGTGCCCTGTGGGACATCGAGACCGGCCAGCAGACGACCACGTTTACCGGACACACTGGAGATGTCATGAGCCTTTCTCTTGCTCCTGACACCAGACTGTTCGTCTCTGGTGCTTGTGATGCTTCAGCCAAACTCTGGGATGTGCGAGAAGGCATGTGCCGGCAGACCTTCACTGGCCACGAGTCTGACATCAATGCCATTTGCTTCTTTCCAAATGGCAATGCATTTGCCACTGGCTCAGACGACGCCACCTGCAGGCTGTTTGACCTTCGTGCTGACCAGGAGCTCATGACTTACTCCCATGACAACATCATCTGCGGGATCACCTCTGTCTCCTTCTCCAAGAGCGGGCGCCTCCTCCTTGCTGGGTACGACGACTTCAACTGCAACGTCTGGGATGCACTCAAAGCCGACCGGGCAGGTGTCTTGGCTGGGCATGACAACCGCGTCAGCTGCCTGGGCGTGACTGACGATGGCATGGCTGTGGCGACAGGGTCCTGGGATAGCTTCCTCAAGATCTGGAACTAACGCCAGTAGCATGTGGATGCCATGGAGACTGGAAGACCATTCCAACTTGGACGCGTTACCATGAGAGCATATCCTATCCAACCGTACTAACGTGGACACCCTACACCTCCCCTCAGAACTTCAAAAGGGCAAGATCTTTTTTCCTTCACTTATTGCTGAAACCAAGAGCACAATTCCCATTGAGAGAAAGATCTCTGTGCTGTAAACTAAAACAAATTGTGCATTCCTTCCGGGGCCATCGTCTTTGTTTTCTTTTTTGTCTTGAATGAATTTTAAAAGGAAATATATAATAAAAATGTTAACCAGAAGGTAAACTTGAGTGTAATTGTCAGACAGACACACTTTTCCACCAGTGTATTTGAATTTTAGACCAGTGACCCTGTTTTGTGGCATTCATGCAAAACATGCTGAGGGCTTTGTTCATCTGGTCATCGTGTCCAAATTTCAGTCATGTTTGTAGCAAGATTTTGGAAGCATTCATATTTCCTTTTTAAAATGTATTCCTTTGTGTTCAACAGTTAATCAAAACCAGAGAGTCTAGGGCAGCCTCTCTGATGTTGTCAATGATGTAAATTCAGTCCCTGGTTTTTAATTTTCTGTCTGATGTCACAGATCATTGTTGCACACAAACGTGGCATAGAAAAGAACATGTTCAGAAGCCATGGGGCCAAGCACATGCGGGGACGGTCTCAAATGCGTGATCAGAGAATCCTTCACCTTTGCTGAAAAGTGAGCTCAGATCCAGCACCATGTTCCTCCTGACCCATCCTGTCTATCTTCTCAGTTGAGTTTTTAATCTCACTTTGGGTTTCCTTGTGAAGTTGGAGGGAAGTTTATAATAGCCTAACACTACCCCACCCCCAACTAGGAGGAACCTCTGTTTTCAAGAGAGATGCCTGTCCTGTGCTTGGATAGTCAGTCAATTATTTGTGTATGAAACAATGTACAAATCAATGTTTTGAAAATAATGATCTCAGACTTTCTAAGTTAAATTTTAAAAATTTTGATTGTTTGCCATATTGGGTGGGTTTACTCTTAGAATCGCATGCTGTAGAAATGCTCAAAAGTGCATATGGGACTCAGTCCTTAGGTGTTCTTTTTCTTTTAAGAAATAACCTCTTACAGTTGTAACCATTGCGGCTCTGTCCACTTCTCGTTGCTGCTCTGTGGCACATATCGGAAGCAGTACAGCGCGCGGCTCTACACGCTTGGGTAGCGGGATAAGTCACTGTTTTCTTTATTTCTTTAAAAAAAAAAAAGTTCTGTTGCAAACGACTGCTGTTGGATTCTGAGGGTGGGGAGGGAGAGAGAGGGAGGGAGAGGGAGTGAAGAGCCTGCCCTCCTATATGGATTCTTCAGGGCCCTCCACATCTGAGGTGGCTCATTCCCATCACACACAGATTGTCCTGGTGTTCATTTCAAGGCCAGTGTTCAGCAGCAGCGTTTGGAAAGCAGGTTCTGTGGGACCCCCCGCCCCGCCCCCCGCACTCCTTCATAGCAGCAGTAGTGGCTTCTCCATCCTGTTTTCTGCAACATTCTATACAAAACTGTGCTGTGACCTTGCGGTAGGCCTGGATCTGGCAAAGAGAATACAAATGAAACCCCTTCTTTCTCTTTCCGTCCAACAACTCTGTAGAGCTCTCTGCACCCTTACCCCTTTCCACCTTTTGTATTTAATTTTAAAGTCAGTGTACTGCAAGGAAGCTGGATGCAAGATAGATACTATATTAAACTGTACTGTTATTTAAGATGTAATAAAGCAGTTTGACATGAGGGA 363 1023 1.0992302917051282 8 diff --git a/skills/codonfm-finetune/evals/files/variants_labeled.csv b/skills/codonfm-finetune/evals/files/variants_labeled.csv new file mode 100644 index 0000000..674222a --- /dev/null +++ b/skills/codonfm-finetune/evals/files/variants_labeled.csv @@ -0,0 +1,21 @@ +id,ref_seq,ref_codon,alt_codon,codon_position,label +variant_00,ATGGCTGAATTTCCGTAA,GCT,GCC,1,0 +variant_01,ATGGCTGAATTTCCGTAA,GAA,GAG,2,1 +variant_02,ATGGCTGAATTTCCGTAA,GCT,GTT,1,0 +variant_03,ATGGCTGAATTTCCGTAA,GAA,GAC,2,1 +variant_04,ATGGCTGAATTTCCGTAA,GCT,GCC,1,0 +variant_05,ATGGCTGAATTTCCGTAA,GAA,GAG,2,1 +variant_06,ATGGCTGAATTTCCGTAA,GCT,GTT,1,0 +variant_07,ATGGCTGAATTTCCGTAA,GAA,GAC,2,1 +variant_08,ATGGCTGAATTTCCGTAA,GCT,GCC,1,0 +variant_09,ATGGCTGAATTTCCGTAA,GAA,GAG,2,1 +variant_10,ATGGCTGAATTTCCGTAA,GCT,GTT,1,0 +variant_11,ATGGCTGAATTTCCGTAA,GAA,GAC,2,1 +variant_12,ATGGCTGAATTTCCGTAA,GCT,GCC,1,0 +variant_13,ATGGCTGAATTTCCGTAA,GAA,GAG,2,1 +variant_14,ATGGCTGAATTTCCGTAA,GCT,GTT,1,0 +variant_15,ATGGCTGAATTTCCGTAA,GAA,GAC,2,1 +variant_16,ATGGCTGAATTTCCGTAA,GCT,GCC,1,0 +variant_17,ATGGCTGAATTTCCGTAA,GAA,GAG,2,1 +variant_18,ATGGCTGAATTTCCGTAA,GCT,GTT,1,0 +variant_19,ATGGCTGAATTTCCGTAA,GAA,GAC,2,1 diff --git a/skills/codonfm-finetune/evals/files/variants_labeled.provenance.json b/skills/codonfm-finetune/evals/files/variants_labeled.provenance.json new file mode 100644 index 0000000..57b9e09 --- /dev/null +++ b/skills/codonfm-finetune/evals/files/variants_labeled.provenance.json @@ -0,0 +1,6 @@ +{ + "purpose": "Synthetic schema fixture for input validation only", + "rows": 20, + "labels": "Arbitrary binary labels; not measured variant effects, not a scientific training dataset", + "splitting": "0.8/0.1/0.1 yields 16/2/2 rows; repeated sequence contexts make this unsuitable for performance evaluation" +} diff --git a/skills/codonfm-finetune/scripts/prepare_ribonn.py b/skills/codonfm-finetune/scripts/prepare_ribonn.py new file mode 100644 index 0000000..ed6d76d --- /dev/null +++ b/skills/codonfm-finetune/scripts/prepare_ribonn.py @@ -0,0 +1,125 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Prepare public human RiboNN TE data for CodonBertDataset, without ML deps.""" + +import argparse +import csv +import http.client +import io +import json +import math +import time +from collections import Counter +from contextlib import closing +from pathlib import Path + + +DATA_REVISION = "512fca642b6b7b61ae494ad83cfc2b72831636d2" +DATA_HOST = "raw.githubusercontent.com" +DATA_PATH = ( + f"/CenikLab/TE_classic_ML/{DATA_REVISION}" + "/data/data_with_human_TE_cellline_all_NA_plain.csv" +) +DATA_URL = f"https://{DATA_HOST}{DATA_PATH}" + + +def prepare(handle, max_rows_per_split=8, max_codons=2046, deadline=None): + """Slice CDSs and keep source folds: 0-7 train, 8 val, 9 test. + + The upstream file is TSV despite its .csv suffix. A capped subset is a + preparation/smoke example, not the notebook's full cross-validation study. + """ + if max_rows_per_split < 0 or max_codons < 1: + raise ValueError("Row cap must be non-negative and max_codons must be positive") + reader = csv.DictReader(handle, delimiter="\t") + required = {"transcript_id", "tx_sequence", "utr5_size", "cds_size", "mean_te", "fold"} + if not required <= set(reader.fieldnames or []): + raise ValueError("Expected RiboNN TSV columns: " + ", ".join(sorted(required))) + rows, seen_ids, sequence_splits = [], set(), {} + counts, skipped = Counter(), Counter() + examined = 0 + for raw in reader: + examined += 1 + if deadline is not None and time.monotonic() > deadline: + raise TimeoutError("Dataset preparation exceeded the network time budget; use --input for an offline file") + try: + start, length, fold = int(raw["utr5_size"]), int(raw["cds_size"]), int(raw["fold"]) + value = float(raw["mean_te"]) + transcript = raw["tx_sequence"].upper().replace("U", "T") + row_id = raw["transcript_id"].strip() + except (TypeError, ValueError, AttributeError): + skipped["invalid_metadata"] += 1 + continue + if not row_id or start < 0 or length <= 0 or start + length > len(transcript) or length % 3: + skipped["invalid_cds_bounds_or_id"] += 1 + continue + sequence = transcript[start:start + length] + if set(sequence) - set("ACGT") or not math.isfinite(value) or fold not in range(10): + skipped["invalid_sequence_label_or_fold"] += 1 + continue + if length // 3 > max_codons: + skipped["cds_exceeds_context"] += 1 + continue + split = "val" if fold == 8 else "test" if fold == 9 else "train" + if row_id in seen_ids: + skipped["duplicate_transcript"] += 1 + continue + if sequence in sequence_splits and sequence_splits[sequence] != split: + raise ValueError("Identical CDS appears in different source folds; resolve split leakage before training") + seen_ids.add(row_id) + sequence_splits[sequence] = split + if max_rows_per_split and counts[split] >= max_rows_per_split: + continue + rows.append({"id": row_id, "ref_seq": sequence, "value": value, "split": split}) + counts[split] += 1 + if max_rows_per_split and all(counts[s] >= max_rows_per_split for s in ("train", "val", "test")): + break + if not all(counts[s] for s in ("train", "val", "test")): + raise ValueError("Prepared data must have non-empty train, val, and test splits; check source folds 0-9") + return rows, { + "rows_examined": examined, "rows_written": len(rows), "split_counts": dict(counts), + "skipped": dict(skipped), "label": "mean_te (unchanged)", + "fold_mapping": {"train": list(range(8)), "val": [8], "test": [9]}, + "max_rows_per_split": max_rows_per_split, "max_codons": max_codons, + "purpose": "input preparation; a capped subset is not a scientific benchmark", + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--input", type=Path, help="Existing upstream-format TSV; omit to stream the pinned public data") + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--max-rows-per-split", type=int, default=8, help="Default 8 for a small example; 0 processes all rows") + parser.add_argument("--max-codons", type=int, default=2046) + args = parser.parse_args() + if args.input and args.input.resolve() == args.output.resolve(): + parser.error("Input and output must be different files") + try: + if args.input: + with args.input.open(encoding="utf-8-sig", newline="") as handle: + rows, report = prepare(handle, args.max_rows_per_split, args.max_codons) + else: + # Fixed HTTPS endpoint with default certificate verification; no redirects or retries. + deadline = time.monotonic() + 60 + with closing(http.client.HTTPSConnection(DATA_HOST, timeout=10)) as connection: + connection.request("GET", DATA_PATH) + with connection.getresponse() as response: + if response.status != 200: + raise ValueError(f"Dataset download returned HTTP {response.status}; expected 200 without redirects") + with io.TextIOWrapper(response, encoding="utf-8-sig", newline="") as handle: + rows, report = prepare(handle, args.max_rows_per_split, args.max_codons, deadline) + report.update({"source": str(args.input) if args.input else DATA_URL, "upstream_url": DATA_URL}) + args.output.parent.mkdir(parents=True, exist_ok=True) + with args.output.open("w", newline="", encoding="utf-8") as handle: + writer = csv.DictWriter(handle, fieldnames=["id", "ref_seq", "value", "split"]) + writer.writeheader() + writer.writerows(rows) + args.output.with_suffix(".metadata.json").write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + except (ValueError, OSError, http.client.HTTPException) as exc: + parser.exit(1, f"Preparation failed: {exc}\n") + print(json.dumps(report, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/skills/codonfm-finetune/skill-card.md b/skills/codonfm-finetune/skill-card.md new file mode 100644 index 0000000..1dc1532 --- /dev/null +++ b/skills/codonfm-finetune/skill-card.md @@ -0,0 +1,87 @@ +## Description:
+Fine-tune public CodonFM Encodon checkpoints on labeled coding-sequence or coding-variant data using LoRA, head-only, or full fine-tuning.
+ +This skill is ready for commercial/non-commercial use.
+ +## Owner +NVIDIA
+ +### License/Terms of Use:
+Apache 2.0
+## Use Case:
+Developers and computational biologists use this skill to fine-tune NVIDIA CodonFM Encodon foundation models on labeled coding-sequence or coding-variant datasets for regression or classification tasks.
+ +### Deployment Geography for Use:
+Global
+ +## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Not Specified]
+**Credential Type(s):** [None identified]
+ +Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+ +## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
+ +## Reference(s):
+- [NV-CodonFM-Encodon-80M-v1 (Hugging Face)](https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1)
+- [NV-CodonFM-Encodon-600M-v1 (Hugging Face)](https://huggingface.co/nvidia/NV-CodonFM-Encodon-600M-v1)
+- [NV-CodonFM-Encodon-1B-v1 (Hugging Face)](https://huggingface.co/nvidia/NV-CodonFM-Encodon-1B-v1)
+- [NV-CodonFM-Encodon on NGC](https://catalog.ngc.nvidia.com/orgs/nvidia/teams/clara/models/nv_codonfm_encodon)
+- [CenikLab TE Classic ML Data](https://github.com/CenikLab/TE_classic_ML/tree/main/data)
+- [NVIDIA Deep Bio Research](https://research.nvidia.com/labs/dbr)
+ + +## Skill Output:
+**Output Type(s):** [Shell commands, Configuration instructions, Files]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [None]
+ +## Evaluation Agents Used:
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + + +## Evaluation Tasks:
+Evaluated against 3 positive evaluation tasks in isolated k8s-sandbox pods, covering LoRA fine-tuning preparation, variant classification, and data preparation workflows.
+ +## Evaluation Metrics Used:
+Reported benchmark dimensions:
+- Security: Whether the skill is safe to use: checks for unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Whether the final answer is correct against the reference answer.
+- Discoverability: Whether the right skill was loaded when needed: skill selection, decoy avoidance, and workflow execution.
+- Effectiveness: Whether the skill helped complete the task: goal completion (50%) and expected workflow adherence (50%).
+- Efficiency: Whether wasted tool calls and token usage were avoided: tool-call productivity (50%) and token efficiency (50%).
+ +Underlying evaluation signals used in this run:
+- `security`: Checks for unsafe operations, secret leakage, and unauthorized access.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `skill_execution`: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity (legacy wire id; routing scored under Discoverability).
+- `token_efficiency`: Actual uncached prompt plus completion usage.
+ + + +## Evaluation Results:
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 85.0% | 82.8% | +| Security | 100.0% → 66.7% (-33.3 points) | 33.3% → 66.7% (+33.4 points) | +| Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Discoverability | 91.7% | 76.7% | +| Effectiveness | 86.7% → 86.7% (±0.0 points) | 96.7% → 90.0% (-6.7 points) | +| Efficiency | 80.0% | 80.7% | + +## Skill Version(s):
+be43117 (source: git SHA, committed 2026-10-07)
+ +## Ethical Considerations:
+NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
+ +(For Release on NVIDIA Platforms Only)
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/codonfm-finetune/skill.oms.sig b/skills/codonfm-finetune/skill.oms.sig new file mode 100644 index 0000000..694aee4 --- /dev/null +++ b/skills/codonfm-finetune/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAiY29kb25mbS1maW5ldHVuZSIsCiAgICAgICJkaWdlc3QiOiB7CiAgICAgICAgInNoYTI1NiI6ICIzZjE4NGUwYjk0MzljMmE1NGM4M2YxMzAzZWFlZTkyOGE1YWY5NzAwOWFmZWIwMTBhMmI4NGQ2YmYwMzAwZjY4IgogICAgICB9CiAgICB9CiAgXSwKICAicHJlZGljYXRlVHlwZSI6ICJodHRwczovL21vZGVsX3NpZ25pbmcvc2lnbmF0dXJlL3YxLjAiLAogICJwcmVkaWNhdGUiOiB7CiAgICAic2VyaWFsaXphdGlvbiI6IHsKICAgICAgImhhc2hfdHlwZSI6ICJzaGEyNTYiLAogICAgICAiaWdub3JlX3BhdGhzIjogWwogICAgICAgICIuZ2l0YXR0cmlidXRlcyIsCiAgICAgICAgIi5naXRpZ25vcmUiLAogICAgICAgICIuZ2l0aHViIiwKICAgICAgICAiLmdpdCIKICAgICAgXSwKICAgICAgIm1ldGhvZCI6ICJmaWxlcyIsCiAgICAgICJhbGxvd19zeW1saW5rcyI6IGZhbHNlCiAgICB9LAogICAgInJlc291cmNlcyI6IFsKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJCRU5DSE1BUksubWQiLAogICAgICAgICJkaWdlc3QiOiAiODlkN2NkNTQ1ZjhhODU4Y2EwYjZkNDgxOTlmN2YzNzMyMDMxNjJmYWYzMWYxYTk5N2Y2MzY0YWY2ZmVlZjk2MiIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJTS0lMTC5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICI1YjU1Y2Q5MTY0Y2M1ZGMyN2U5OTY1Yjc4NWJjODI3ZmM3ZTk0NWY4MGQzMWUyYTI2YTZjMjdhYmExZmE0NmEzIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogImFnZW50cy9vcGVuYWkueWFtbCIsCiAgICAgICAgImRpZ2VzdCI6ICI3OWVjZDk5ODIxN2M1YTc1YTlhMWVmNmJlMTcwYmQwYTNiNjY4NDU0YzcxZDdjY2NkZWNhNTFiZDJhYzkyMmRjIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogImV2YWxzL2V2YWxzLmpzb24iLAogICAgICAgICJkaWdlc3QiOiAiNjEwYWEzOWEyNzI4OGViYzhiNTIwNTY2NjVmMGZlZmEwZTk4MmQwOWY1OThlYzUwMjAyNTM4ODE3ZDhlYmIwYiIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJldmFscy9maWxlcy9jb2RvbmZtX3NvdXJjZS56aXAiLAogICAgICAgICJkaWdlc3QiOiAiZjBjMTIyZjE5YTJkMzE2ZmU5NWNjZmFhMGFjNTFiZDYxM2NmYWJkZDA5ZmQ4OWZiYjc2NjU2MmE1ZjBlZWQyNyIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJldmFscy9maWxlcy9lbmNvZG9uX2NoZWNrcG9pbnQuanNvbiIsCiAgICAgICAgImRpZ2VzdCI6ICI2OThmYTM1NDE0MDAzM2U5YWE4NTkxY2QyODYwNzAzYTRmY2RjMmViMjdlZGM4ZjJmMDc2MGM0MDgzZTU2M2VlIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogImV2YWxzL2ZpbGVzL3JpYm9ubl9zbW9rZS5wcm92ZW5hbmNlLmpzb24iLAogICAgICAgICJkaWdlc3QiOiAiNGMyMzUxNzk0MGIwOGYxODk2ZjMyY2FmNmYwNGU2MDMwNGQzNDNlYmY2M2YzZWZjYTY3ZjJiZjhjNTBjZjc2ZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJldmFscy9maWxlcy9yaWJvbm5fc21va2UudHN2IiwKICAgICAgICAiZGlnZXN0IjogImYzODVhMTgxYTM3YzdhZDU4Y2U0N2VmNWY2OTYyMDk2ZGNiODMwNjMxMjkwYmM4NDdhNjdiODY4OTZmN2FmN2UiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZmlsZXMvdmFyaWFudHNfbGFiZWxlZC5jc3YiLAogICAgICAgICJkaWdlc3QiOiAiZjkwNDQwMDBkNjgxYjI1ZTY3YmFiODhiZGQ5NGE1MjRiNDg0NzA2NWIyOGM4ZWQ3MGNhMWE2MjFiNzlhZGViMyIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJldmFscy9maWxlcy92YXJpYW50c19sYWJlbGVkLnByb3ZlbmFuY2UuanNvbiIsCiAgICAgICAgImRpZ2VzdCI6ICJlYjJlYWMyOGU3N2M0ZGVjMDc3YmU4MDY5NjBhZThiYjMyOWQ4Njg2MzNiMjFmMmM0OTFhNjU1NWJiNTY1NzExIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogInNjcmlwdHMvcHJlcGFyZV9yaWJvbm4ucHkiLAogICAgICAgICJkaWdlc3QiOiAiOGYwZTA0NjBlYzc1OGNjMGQzOGIzYWExMGIzZDFjYzZmNWRlNzc1ODI3NjUxMTg3NWNkMjU1MDU0ODM1ZDk1YyIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJza2lsbC1jYXJkLm1kIiwKICAgICAgICAiZGlnZXN0IjogImQ2NGFhOTljNWU3OWQ0ODNhN2Y3OWNlMTYzM2I3ZTcwNmZjYWNlMjA1OTc3NTJhNDhkOWEwZWRjMTMyMzI0YzgiCiAgICAgIH0KICAgIF0KICB9Cn0=","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCMB+jQojEEjtEmNDADmaSp/3H9/y8b7x+PVgJHL+bJ77mBxXphxqesk2ZtSzUTVwRdQIxAMAFmYyJ4eKBmzCmrQk5Ybvqsfzp4Bvd/Mcp+OSgp6s/rzyGATpxUYvMjC0+0hNs0g==","keyid":""}]}} \ No newline at end of file diff --git a/skills/codonfm-score/BENCHMARK.md b/skills/codonfm-score/BENCHMARK.md new file mode 100644 index 0000000..b77891b --- /dev/null +++ b/skills/codonfm-score/BENCHMARK.md @@ -0,0 +1,125 @@ +# Skill Benchmark: codonfm-score + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `codonfm-score` +- Evaluation date: 2026-10-07 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-5`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 3 evaluation tasks (2 positive, 1 negative) +- Dataset digest: `sha256:5473a6b7ef522b019ec4b46023830ce8ac22da3576b0c97bf10acb2ac28e9d79` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 1 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 96.1% — baseline ran, but no comparable score was available; uplift unavailable | 84.8% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 100.0% (±0.0 points) | 33.3% → 66.7% (+33.4 points) | +| Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Discoverability | 95.0% — baseline ran, but no comparable score was available; uplift unavailable | 80.0% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 90.0% → 93.3% (+3.3 points) | 90.0% → 96.7% (+6.7 points) | +| Efficiency | 92.1% — baseline ran, but no comparable score was available; uplift unavailable | 80.7% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +A partial dimension was calculated from only the available configured signals; review the detailed report before relying on it. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 1,308,984 | 3,855,619 | -2,546,635 | -66.05% | skill 3/3; base 3/3 | +| claude-code | codonfm-score-001 | 677,690 | 2,757,266 | -2,079,576 | -75.42% | skill 1/1; base 1/1 | +| claude-code | codonfm-score-002 | 457,608 | 884,998 | -427,390 | -48.29% | skill 1/1; base 1/1 | +| claude-code | codonfm-score-003 | 173,686 | 213,355 | -39,669 | -18.59% | skill 1/1; base 1/1 | +| codex | All cases | 1,491,127 | 1,204,536 | +286,591 | +23.79% | skill 3/3; base 3/3 | +| codex | codonfm-score-001 | 477,556 | 566,528 | -88,972 | -15.70% | skill 1/1; base 1/1 | +| codex | codonfm-score-002 | 1,000,069 | 624,731 | +375,338 | +60.08% | skill 1/1; base 1/1 | +| codex | codonfm-score-003 | 13,502 | 13,277 | +225 | +1.69% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 2,800,111 | 5,060,155 | -2,260,044 | -44.66% | skill 6/6; base 6/6 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 7 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 3 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/codonfm-score/SKILL.md`) +- **LOW** QUALITY/quality_discoverability: Description very long (385 chars, recommend 50-150) (`skills/codonfm-score/SKILL.md`) +- **LOW** QUALITY/quality_discoverability: No '## Purpose' section (`skills/codonfm-score/SKILL.md`) +- **LOW** QUALITY/quality_reliability: No prerequisites/requirements documented (`skills/codonfm-score/SKILL.md`) +- **LOW** QUALITY/quality_reliability: No limitations documented (`skills/codonfm-score/SKILL.md`) +- 2 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/codonfm-score/SKILL.md b/skills/codonfm-score/SKILL.md new file mode 100644 index 0000000..f7ff434 --- /dev/null +++ b/skills/codonfm-score/SKILL.md @@ -0,0 +1,159 @@ +--- +name: codonfm-score +description: Validate, prepare, or run public CodonFM Encodon masked-codon variant scoring and review compatibility of its scoring workflows. Use only when the user explicitly requests CodonFM or Encodon, or that context is already established in the conversation. Do not select this skill for a generic variant-scoring request without that context; ask for the variant and intended analysis first. +metadata: + author: "NVIDIA BioNeMo " +--- + +# Score variants with public Encodon + +Run general masked-codon `mutation_prediction` only. This produces a research +signal, not a clinical diagnosis or an expression-direction prediction. + +## Instructions + +First confirm CodonFM or Encodon context in the user's request or established +conversation. If that context is missing, ask for any missing variant details +and the intended analysis before choosing a model or inspecting model-specific +files. The presence of this skill or source files alone does not establish +the user's intent. + +For source reviews and command preparation, inspect the supplied source and +metadata without installing the ML runtime. Use an available Python 3 +interpreter with standard-library `zipfile`, `json`, and `csv`; do not assume +the `python` alias or `unzip` exists. Read archive members directly with +`ZipFile.namelist()` and `ZipFile.read()` where possible. If extraction is +needed, use a fresh directory from `tempfile.mkdtemp()` or `mktemp -d` and +preserve existing checkouts and scratch directories. Check whether `rg` is +available; use `grep` or Python if it is absent. Read the source sections +needed for the requested command or compatibility question. + +Check whether the request is executable in public v1 before installing or +downloading anything. For synonymous-codon aggregation or Decodon, inspect the +[parser](../../src/runner.py) and [model configuration](../../src/config.py), +explain the missing feature, and finish. Do not implement the missing workflow, +search private code, or keep retrying unsupported commands. + +Resolve the variant CSV, checkpoint, and output directory from the request and +available files. Validate inputs before inference. Execution requires the +project's ML dependencies and a compatible NVIDIA GPU. If a required resource +is unavailable, return the validated inputs where possible and a command with +the missing prerequisite identified. When scoring is requested and resources +are ready, execute and verify the score arrays. A request for preparation ends +with the inputs and command. If variants are missing, report the required +schema; do not invent variants or silently switch to a public dataset. + +Default to the public 80M checkpoint for demonstrations: +`nvidia/NV-CodonFM-Encodon-80M-v1`, revision +`399ca9fe17b57941a7bebc6788033919b417413c`, file +`NV-CodonFM-Encodon-80M-v1.safetensors` and sibling `config.json`. +Reuse an existing checkpoint or download it when needed for the requested work. +Preserve an explicitly requested model size. + +## Preflight + +1. Confirm `src/runner.py`, `src/data/mutation_dataset.py`, and + `src/inference/encodon.py` exist. +2. Accept only `encodon_80m`, `encodon_600m`, or `encodon_1b` as + `--model_name`. The public parser lists larger names, but its model + configuration does not implement them. +3. For model execution, require a `.ckpt` file, or a `.safetensors` file with + sibling `config.json`. Input preparation can use a planned path. +4. Validate the CSV headers before starting a GPU job. + +## Inputs + +Require these CSV columns: + +- `id`: unique row identifier. +- `ref_seq`: reference coding sequence, not genomic DNA with introns, UTR-only + sequence, or protein sequence. +- `ref_codon` and `alt_codon`: three-nucleotide codons. +- `codon_position`: zero-based codon position relative to the CDS. + +With `--extract-seq`, `MutationDataset` extracts an appropriate sequence window +from `ref_seq`; it does not derive or require `alt_seq`. + +Before running, normalize sequences and codons to uppercase DNA (`A/C/G/T`), +require CDS lengths divisible by three, and check every row satisfies: + +```text +0 <= codon_position < len(ref_seq) / 3 +ref_seq[3 * codon_position : 3 * codon_position + 3] == ref_codon +``` + +The public extractor asserts the second condition and otherwise stops the job. + +## Examples + +Set `CODONFM_DATA_PATH` to the variant CSV, `CODONFM_CHECKPOINT_PATH` to the +checkpoint, and `CODONFM_RUN_DIR` to your chosen output directory: + +Use the interpreter from the configured ML environment for inference. The +example uses `python`; substitute that environment's interpreter path if the +alias is unavailable. + +```bash +python -m src.runner eval \ + --exp_name variant_scoring \ + --model_name encodon_80m \ + --checkpoint_path "$CODONFM_CHECKPOINT_PATH" \ + --data_path "$CODONFM_DATA_PATH" \ + --process_item mutation_pred_mlm \ + --dataset_name MutationDataset \ + --task_type mutation_prediction \ + --extract-seq \ + --mask_mutation \ + --num_nodes 1 \ + --num_gpus 1 \ + --num_workers 0 \ + --val_batch_size 2 \ + --out_dir "$CODONFM_RUN_DIR" \ + --predictions_output_dir "$CODONFM_RUN_DIR/predictions" +``` + +Do not remove `--mask_mutation`: without it, the reference codon remains +visible at the scored position and invalidates masked-codon LLR scoring. +For preparation requests, inspect the CSV directly against the input schema +and reference-position checks above, then report the rows checked and provide +the scoring command. Extra columns are allowed; use `--ref_seq_col` if the +reference sequence has a different column name. These checks do not require +the ML runtime. The command above performs inference when resources are ready. + +The existing `--dryrun` optionally builds runtime configuration and skips +execution. It requires the ML dependencies, can create the prediction directory, +and does not read the CSV or load weights. Do not use it as evidence that inputs, +checkpoint compatibility, or prediction quality have been validated. + +## Outputs + +`--predictions_output_dir` receives: + +- `ref_likelihoods_merged.npy` +- `alt_likelihoods_merged.npy` +- `likelihood_ratios_merged.npy` +- `ids_merged.npy` + +Load the arrays with NumPy and align scores by `ids_merged.npy`. The reported +LLR is `log p(ref_codon) - log p(alt_codon)`; a larger positive value means the +alternate codon is less probable in context. It does not say whether +expression goes up or down. + +## Reporting + +Keep the final answer concise and self-contained, with the requested command +or compatibility conclusion near the start. For command preparation, include +each row's validation result, the complete command, all four output filenames, +and the LLR definition and sign interpretation. Cite the inspected source +locations for the command, outputs, and scoring semantics. State whether +inference ran; report numerical scores only when execution produced them. + +## Boundaries + +- General `mutation_prediction` handles both synonymous and missense changes. +- Do not use `missense_prediction`, `missense_inference`, `MissenseDataset`, + `mutation_pred_clm`, `--organism_token`, or `--causal`; those are newer + unavailable public-release features. +- If a user asks specifically for synonymous-codon-aggregated missense + scoring, explain that public v1 only provides the general ref/alt LLR. Do not + silently substitute the two methods. diff --git a/skills/codonfm-score/agents/openai.yaml b/skills/codonfm-score/agents/openai.yaml new file mode 100644 index 0000000..7dba4a4 --- /dev/null +++ b/skills/codonfm-score/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "CodonFM Variant Scoring" + short_description: "Score coding variants with public Encodon models" + default_prompt: "Use $codonfm-score to validate and score coding variants with a public Encodon checkpoint." diff --git a/skills/codonfm-score/evals/config.yml b/skills/codonfm-score/evals/config.yml new file mode 100644 index 0000000..31036d7 --- /dev/null +++ b/skills/codonfm-score/evals/config.yml @@ -0,0 +1,6 @@ +schema_version: 1 + +harbor: + agents: + claude-code: + model: aws/anthropic/bedrock-claude-opus-5 diff --git a/skills/codonfm-score/evals/evals.json b/skills/codonfm-score/evals/evals.json new file mode 100644 index 0000000..5614d84 --- /dev/null +++ b/skills/codonfm-score/evals/evals.json @@ -0,0 +1,52 @@ +{ + "skill_name": "codonfm-score", + "evals": [ + { + "id": "codonfm-score-001", + "prompt": "Review the public CodonFM software using the supplied synthetic test fixtures. Inspect codonfm_source.zip and encodon_checkpoint.json, validate both example rows in variants.csv, and prepare an Encodon masked-codon scoring command that matches the supplied implementation.\n\nExplain the expected output files and how to interpret positive and negative scores, citing the relevant source. Provide the command without running inference or inventing numerical results.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json", + "files/variants.csv" + ], + "expected_output": "Two validated variant rows and a public mutation_prediction command, with the correct output contract and reference-versus-alternate score interpretation.", + "assertions": [ + "The command uses mutation_prediction, mutation_pred_mlm, and MutationDataset", + "The command includes --mask_mutation, --extract-seq, --checkpoint_path, and explicit prediction/output paths", + "The agent validates both variant rows, including reference codon agreement at the zero-based CDS position", + "The response identifies ref_likelihoods_merged.npy, alt_likelihoods_merged.npy, likelihood_ratios_merged.npy, and ids_merged.npy as expected inference outputs", + "The agent defines LLR as log p(ref_codon) minus log p(alt_codon), does not equate it with expression direction or clinical effect, and does not fabricate scores" + ], + "expected_skill": "codonfm-score", + "expected_script": null + }, + { + "id": "codonfm-score-002", + "prompt": "Review the supplied public CodonFM source and checkpoint metadata for a software compatibility question: does this implementation support running synonymous-codon-aggregated missense_prediction?\n\nCite the relevant source locations and explain how that workflow relates to the available variant-scoring method. Base the answer on what the supplied implementation actually supports. This task requires source inspection and explanation; no new implementation or inference run is needed.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json" + ], + "expected_output": "The agent explains that the aggregation workflow is unavailable and distinguishes it from ordinary masked reference/alternate codon LLR scoring.", + "assertions": [ + "The agent identifies missense_prediction and missense_inference as unavailable in the supplied public implementation", + "The agent cites inspected source and distinguishes general mutation_prediction from synonymous-codon aggregation", + "The agent answers the compatibility question without implementing a missing feature or silently substituting the available method" + ], + "expected_skill": "codonfm-score", + "expected_script": null + }, + { + "id": "codonfm-score-003", + "prompt": "Score this variant.", + "files": [], + "expected_output": "The agent requests the missing variant and analysis context without assuming CodonFM.", + "assertions": [ + "The agent does not invoke a codonfm-* skill without CodonFM or Encodon context", + "The agent identifies missing inputs without inventing a variant or choosing a model" + ], + "expected_skill": null, + "expected_script": null + } + ] +} diff --git a/skills/codonfm-score/evals/files/codonfm_source.zip b/skills/codonfm-score/evals/files/codonfm_source.zip new file mode 100644 index 0000000..eb76611 Binary files /dev/null and b/skills/codonfm-score/evals/files/codonfm_source.zip differ diff --git a/skills/codonfm-score/evals/files/encodon_checkpoint.json b/skills/codonfm-score/evals/files/encodon_checkpoint.json new file mode 100644 index 0000000..6c6230d --- /dev/null +++ b/skills/codonfm-score/evals/files/encodon_checkpoint.json @@ -0,0 +1,33 @@ +{ + "repo_id": "nvidia/NV-CodonFM-Encodon-80M-v1", + "revision": "399ca9fe17b57941a7bebc6788033919b417413c", + "model_name": "encodon_80m", + "filename": "NV-CodonFM-Encodon-80M-v1.safetensors", + "size_bytes": 307351588, + "config_filename": "config.json", + "config": { + "vocab_size": 69, + "hidden_size": 1024, + "num_hidden_layers": 6, + "num_attention_heads": 8, + "intermediate_size": 4096, + "hidden_act": "gelu", + "hidden_dropout_prob": 0.1, + "attention_probs_dropout_prob": 0.1, + "initializer_range": 0.02, + "layer_norm_eps": 1e-12, + "pad_token_id": 3, + "position_embedding_type": "rotary", + "classifier_dropout": 0.1, + "rotary_theta": 10000.0, + "ignore_index": -100, + "loss_type": "cross_entropy", + "lora": false, + "lora_alpha": 32.0, + "lora_r": 16, + "lora_dropout": 0.1, + "finetune_strategy": "full" + }, + "source_url": "https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1/tree/399ca9fe17b57941a7bebc6788033919b417413c", + "weights_included": false +} diff --git a/skills/codonfm-score/evals/files/variants.csv b/skills/codonfm-score/evals/files/variants.csv new file mode 100644 index 0000000..18ce7b5 --- /dev/null +++ b/skills/codonfm-score/evals/files/variants.csv @@ -0,0 +1,3 @@ +id,ref_seq,ref_codon,alt_codon,codon_position +variant_00,ATGGCTGAATTTCCGTAA,GCT,GCC,1 +variant_01,ATGGCTGAATTTCCGTAA,GAA,GAG,2 diff --git a/skills/codonfm-score/skill-card.md b/skills/codonfm-score/skill-card.md new file mode 100644 index 0000000..e45d748 --- /dev/null +++ b/skills/codonfm-score/skill-card.md @@ -0,0 +1,86 @@ +## Description:
+Validate, prepare, or run public CodonFM Encodon masked-codon variant scoring and review compatibility of its scoring workflows.
+ +This skill is ready for commercial/non-commercial use.
+ +## Owner +NVIDIA
+ +### License/Terms of Use:
+Apache 2.0
+## Use Case:
+Developers and computational biologists use this skill to validate variant CSV inputs and run or prepare masked-codon scoring commands with public Encodon models.
+ +### Deployment Geography for Use:
+Global
+ +## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Not Specified]
+**Credential Type(s):** [None identified]
+ +Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+ +## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
+ +## Reference(s):
+- [NV-CodonFM-Encodon-80M-v1 (Hugging Face)](https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1)
+- [NV-CodonFM-Encodon-600M-v1 (Hugging Face)](https://huggingface.co/nvidia/NV-CodonFM-Encodon-600M-v1)
+- [NV-CodonFM-Encodon-1B-v1 (Hugging Face)](https://huggingface.co/nvidia/NV-CodonFM-Encodon-1B-v1)
+- [NV CodonFM Encodon (NGC Catalog)](https://catalog.ngc.nvidia.com/orgs/nvidia/teams/clara/models/nv_codonfm_encodon)
+- [NVIDIA Deep Bio Research](https://research.nvidia.com/labs/dbr)
+ + +## Skill Output:
+**Output Type(s):** [Shell commands, Analysis]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [Produces NumPy arrays (ref_likelihoods, alt_likelihoods, likelihood_ratios, ids) when inference executes]
+ +## Evaluation Agents Used:
+- Claude Code (`aws/anthropic/bedrock-claude-opus-5`)
+- Codex (`openai/openai/gpt-5.5`)
+ + + +## Evaluation Tasks:
+3 evaluation tasks (2 positive, 1 negative) in isolated sandbox pods.
+ +## Evaluation Metrics Used:
+Reported benchmark dimensions:
+- Security: Checks for unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Checks final-answer correctness against the reference answer.
+- Discoverability: Checks whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- Effectiveness: Checks whether the skill helped complete the user's goal and followed the expected workflow behavior.
+- Efficiency: Checks tool-call productivity and token efficiency to detect wasted skill and tool usage.
+ +Underlying evaluation signals used in this run:
+- `security`: Unsafe operations, secret leakage, and unauthorized access.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `skill_execution`: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity (legacy wire id; routing is scored under Discoverability).
+- `token_efficiency`: Actual uncached prompt plus completion usage (50% of Efficiency).
+ + + +## Evaluation Results:
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 96.1% | 84.8% | +| Security | 100.0% → 100.0% (±0.0 pts) | 33.3% → 66.7% (+33.4 pts) | +| Correctness | 100.0% → 100.0% (±0.0 pts) | 100.0% → 100.0% (±0.0 pts) | +| Discoverability | 95.0% | 80.0% | +| Effectiveness | 90.0% → 93.3% (+3.3 pts) | 90.0% → 96.7% (+6.7 pts) | +| Efficiency | 92.1% | 80.7% | + +## Skill Version(s):
+be43117 (source: git SHA, committed 2026-10-07)
+ +## Ethical Considerations:
+NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
+ +(For Release on NVIDIA Platforms Only)
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/codonfm-score/skill.oms.sig b/skills/codonfm-score/skill.oms.sig new file mode 100644 index 0000000..35a6a5c --- /dev/null +++ b/skills/codonfm-score/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAiY29kb25mbS1zY29yZSIsCiAgICAgICJkaWdlc3QiOiB7CiAgICAgICAgInNoYTI1NiI6ICI3NDdhY2EwMjJjYzE2Mzk5ODY0N2Y5NjE1MzFjN2JhZGUwMmE0NjQ3Njk4NGFhMTZlZDk3OGExOTBlNjU1ZWFiIgogICAgICB9CiAgICB9CiAgXSwKICAicHJlZGljYXRlVHlwZSI6ICJodHRwczovL21vZGVsX3NpZ25pbmcvc2lnbmF0dXJlL3YxLjAiLAogICJwcmVkaWNhdGUiOiB7CiAgICAic2VyaWFsaXphdGlvbiI6IHsKICAgICAgImFsbG93X3N5bWxpbmtzIjogZmFsc2UsCiAgICAgICJoYXNoX3R5cGUiOiAic2hhMjU2IiwKICAgICAgIm1ldGhvZCI6ICJmaWxlcyIsCiAgICAgICJpZ25vcmVfcGF0aHMiOiBbCiAgICAgICAgIi5naXQiLAogICAgICAgICIuZ2l0aWdub3JlIiwKICAgICAgICAiLmdpdGh1YiIsCiAgICAgICAgIi5naXRhdHRyaWJ1dGVzIgogICAgICBdCiAgICB9LAogICAgInJlc291cmNlcyI6IFsKICAgICAgewogICAgICAgICJkaWdlc3QiOiAiMTgxNDdiMWFiYzNhMDNiZjVhZWQ1YjA3ZTA1ZTAzZGM1ODJmMWQxZDM1MDM2ODg0NmZlYjA2MzQxY2E0MGFmZSIsCiAgICAgICAgIm5hbWUiOiAiQkVOQ0hNQVJLLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJkaWdlc3QiOiAiMThmNzY2MmE1M2E4NmFjNWZjNjk1MjYxZTgxZDE1YjkwNGZlMzJkNTc0MGRjNTdiMmVlZmZkNzFjNDBhMzU2ZiIsCiAgICAgICAgIm5hbWUiOiAiU0tJTEwubWQiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgImRpZ2VzdCI6ICJhYzNiYzVlNWM3YzM4YzI4NDI3M2M2ODE4MzQyNmFhNDlmOWYzM2ZlZDhkM2E1OGUzY2YyMWJhYjhmYmY1MmE2IiwKICAgICAgICAibmFtZSI6ICJhZ2VudHMvb3BlbmFpLnlhbWwiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgImRpZ2VzdCI6ICJkMDBhYWJjMzU0NjM2Yjg5NjcyOTVjN2ZjOGE5OWYyZTBiMzIzMjg1Yjc3ZTg0ZTY1NmI0OTgwYTcwZTg1MjY1IiwKICAgICAgICAibmFtZSI6ICJldmFscy9jb25maWcueW1sIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJkaWdlc3QiOiAiZDllODdhZmZiMjE2ZmY4N2I5ZjRiMDE5MWE3MDllMjAzOTcyZjFmOWU5NDU2ZGJkZGEzZjYxNTFmMjFmYzEzNyIsCiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZXZhbHMuanNvbiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiZGlnZXN0IjogImYwYzEyMmYxOWEyZDMxNmZlOTVjY2ZhYTBhYzUxYmQ2MTNjZmFiZGQwOWZkODlmYmI3NjY1NjJhNWYwZWVkMjciLAogICAgICAgICJuYW1lIjogImV2YWxzL2ZpbGVzL2NvZG9uZm1fc291cmNlLnppcCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiZGlnZXN0IjogIjY5OGZhMzU0MTQwMDMzZTlhYTg1OTFjZDI4NjA3MDNhNGZjZGMyZWIyN2VkYzhmMmYwNzYwYzQwODNlNTYzZWUiLAogICAgICAgICJuYW1lIjogImV2YWxzL2ZpbGVzL2VuY29kb25fY2hlY2twb2ludC5qc29uIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJkaWdlc3QiOiAiZTI4YjYzMzBkYjAwNzU5MWVjZGRlMjE1NWEyZjExOGY0MWUyMzY5OTczYjViODBkNTZhNjk4NDZhMjBkNTYwZCIsCiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZmlsZXMvdmFyaWFudHMuY3N2IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJkaWdlc3QiOiAiMGQzNTA3YjhmZDU2MWVhY2QyNDAwMDRjZmY0NzBlYjIxZDFmOGJlZmM0YjUxY2FhM2RlY2FiZjJmM2ZjYmJjOSIsCiAgICAgICAgIm5hbWUiOiAic2tpbGwtY2FyZC5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0KICAgIF0KICB9Cn0=","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGQCMFRcNJbnOy3MCAqDWGV9yEJESKsSC9nFXZ/u3srx2WxqeqRiFye0LeNECTI2POnXSwIwM4GlYHdPzJ4AUlo7qbgQqqDWd7Ex70Ru5YRxQl9l6J4aVNQlcwvpragN/Vc6pc2f","keyid":""}]}} \ No newline at end of file diff --git a/skills/codonfm-setup/BENCHMARK.md b/skills/codonfm-setup/BENCHMARK.md new file mode 100644 index 0000000..4714257 --- /dev/null +++ b/skills/codonfm-setup/BENCHMARK.md @@ -0,0 +1,123 @@ +# Skill Benchmark: codonfm-setup + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `codonfm-setup` +- Evaluation date: 2026-10-07 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-5`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 3 evaluation tasks (3 positive) +- Dataset digest: `sha256:da7c67cf041d32b7e9920ac29786e8d039bfe948533853dbb0bbf38c553dc8d9` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 1 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 92.8% — baseline ran, but no comparable score was available; uplift unavailable | 94.8% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 100.0% (±0.0 points) | 0.0% → 100.0% (+100.0 points) | +| Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Discoverability | 100.0% — baseline ran, but no comparable score was available; uplift unavailable | 88.3% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 87.5% → 75.0% (-12.5 points) | 85.8% → 91.7% (+5.9 points) | +| Efficiency | 89.2% — baseline ran, but no comparable score was available; uplift unavailable | 94.0% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 1,456,315 | 3,711,392 | -2,255,077 | -60.76% | skill 3/3; base 3/3 | +| claude-code | codonfm-setup-001 | 413,540 | 665,232 | -251,692 | -37.84% | skill 1/1; base 1/1 | +| claude-code | codonfm-setup-002 | 511,909 | 542,322 | -30,413 | -5.61% | skill 1/1; base 1/1 | +| claude-code | codonfm-setup-003 | 530,866 | 2,503,838 | -1,972,972 | -78.80% | skill 1/1; base 1/1 | +| codex | All cases | 601,083 | 1,236,238 | -635,155 | -51.38% | skill 3/3; base 3/3 | +| codex | codonfm-setup-001 | 86,689 | 221,903 | -135,214 | -60.93% | skill 1/1; base 1/1 | +| codex | codonfm-setup-002 | 340,006 | 494,727 | -154,721 | -31.27% | skill 1/1; base 1/1 | +| codex | codonfm-setup-003 | 174,388 | 519,608 | -345,220 | -66.44% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 2,057,398 | 4,947,630 | -2,890,232 | -58.42% | skill 6/6; base 6/6 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 8 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 3 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/codonfm-setup/SKILL.md`) +- **MEDIUM** SCHEMA/body_recommended_section: Missing recommended section: '## Examples' (`skills/codonfm-setup/SKILL.md`) +- **LOW** QUALITY/quality_discoverability: Description very long (426 chars, recommend 50-150) (`skills/codonfm-setup/SKILL.md`) +- **LOW** QUALITY/quality_discoverability: No '## Purpose' section (`skills/codonfm-setup/SKILL.md`) +- **LOW** QUALITY/quality_reliability: No prerequisites/requirements documented (`skills/codonfm-setup/SKILL.md`) +- 3 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/codonfm-setup/SKILL.md b/skills/codonfm-setup/SKILL.md new file mode 100644 index 0000000..9cee52b --- /dev/null +++ b/skills/codonfm-setup/SKILL.md @@ -0,0 +1,256 @@ +--- +name: codonfm-setup +description: Set up the public CodonFM v1 repository and download public Encodon checkpoints. Use for requests to build or launch the CodonFM development container, configure local data/checkpoint mounts, verify GPU access, or download public Encodon 80M, 600M, 1B, or Cdwt-1B weights. Do not use for Decodon, Encodon 5B/10B, missense-aggregation, or codon-optimization setup because those implementations are not in the public repository. +metadata: + author: "NVIDIA BioNeMo " +--- + +# CodonFM public setup + +Operate from the public CodonFM repository root. Support only the checked-in +public v1 code and public Encodon checkpoints. + +## Instructions + +1. Determine whether the user wants instructions, a downloaded checkpoint, a + working model environment, or a combination of these. +2. For setup instructions or runtime work, inspect the supplied files and + configuration directly: the [runner](../../src/runner.py), + [model configuration](../../src/config.py), [Dockerfile](../../Dockerfile), + [launcher](../../run_dev.sh), and [requirements](../../requirements.txt). + Runtime setup requires a checkout; a supplied source archive is sufficient + for preparing instructions. +3. Check the [public-v1 boundaries](#public-v1-boundaries). For an unsupported + request, inspect `MODEL_ARCHITECTURES` in `src/config.py`, the model modules, + and any requested script before explaining the boundary and ending that + path. Runner argument choices alone do not establish implementation support. +4. Reuse available environments and checkpoints, and choose explicit paths + from the user's project. +5. Follow only the requested paths below. Environment setup alone does not + require a checkpoint download; download weights only when the request needs + them and a suitable local checkpoint is unavailable. + +| Requested scope | Action and completion condition | +| --- | --- | +| Instructions only | Inspect the supplied source/configuration, provide the commands described under Reporting setup instructions, then stop. No installation or GPU verification is required. | +| Checkpoint only | Follow Download a checkpoint, check the downloaded files, report their paths, then stop. No Docker, GPU, or model runtime is required. | +| Working model environment | Follow Runtime preflight, choose the container or direct-host path, then Verify the runtime. Report the checks performed and any remaining limitations. | + +For supplied source archives, inspect selected files with the available Python +3 standard library (`zipfile.ZipFile.namelist()` and `read()`) without extracting +the whole archive. If extraction is needed, use a fresh directory from +`mktemp -d` or `tempfile.mkdtemp()`. Preserve existing checkouts and temporary +directories; do not delete or overwrite them to prepare a source inspection. + +The runner's optional `--dryrun` requires the ML dependencies to be installed +already. It constructs runtime configuration, then stops before execution. +It does not install packages, validate CSV data, or load weights. Preparing +setup instructions does not require running it. + +## Reporting setup instructions + +For instruction requests, put complete commands for the requested setup path +early in a compact, self-contained answer, even when also writing a guide file. + +- For downloads, use supplied checkpoint metadata for the exact repository, + revision, weight filename, and `config.json`. Show the destination directory + and keep the weights and configuration together. +- For containers, state Docker/GPU prerequisites, explain existing-container + replacement before the launcher command, and show explicit host data and + checkpoint paths and the checkpoint mount at `/data/checkpoints`. +- For direct-host setup, include `python3.11 -m venv`, + `python -m pip install -r requirements.txt`, a writable `MPLCONFIGDIR`, + `torch.cuda.is_available()` verification, and explicit host checkpoint paths. +- State which checks actually ran and what remains unverified before model + execution. Written instructions alone do not establish a working environment. + +For a compatibility-only question, give the source-backed availability answer +without adding an unrelated installation procedure. + +## Runtime preflight + +For a working environment, check hardware before installing the runtime: use +`nvidia-smi` if available, or check CUDA through an existing PyTorch installation. +Actual model execution requires the ML dependencies and a compatible NVIDIA GPU. +Compare the driver with the CUDA version required by the selected runtime using +[NVIDIA's compatibility guidance](https://docs.nvidia.com/deploy/cuda-compatibility/minor-version-compatibility.html). +For the Dockerfile's `nvcr.io/nvidia/pytorch:24.10-py3` base, also check the +[24.10 driver requirements](https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-24-10.html#driver-requirements). +If a prerequisite is missing, follow Failure handling below. + +### Container preflight + +1. Confirm `Dockerfile`, `run_dev.sh`, and `src/runner.py` exist. +2. Confirm `docker info` succeeds. Docker must have NVIDIA Container Toolkit + configured for `--gpus all`; host GPU visibility alone does not establish + container GPU access. Verify access in the launched container below. +3. Run `bash -n run_dev.sh` before launching it. +4. Resolve existing absolute host paths for data and checkpoints. Always pass + both path flags to the launcher rather than relying on `/data/codonfm` + defaults. Create missing project directories only as needed for the request. +5. Check for an existing container before launch: + +```bash +docker ps -a --filter name='^/codon-fm-dev-container$' +``` + +If an exact-name container is running, `run_dev.sh` stops and removes it; tell +the user before replacement. If it is stopped, the script cannot reuse the +name, so obtain confirmation before removing it with +`docker rm codon-fm-dev-container`. If removal is declined, preserve the +container, skip this launch, and report the name conflict. + +The public script uses host networking/IPC and mounts the user's SSH directory +read-only; disclose this before execution. It has no opt-out flags for these +settings. If they conflict with the user's constraints, use the direct-host +path when feasible; otherwise report that container launch remains blocked. + +## Build and launch + +Set `CODONFM_REPO_DIR`, `CODONFM_DATA_DIR`, and `CODONFM_CHECKPOINT_DIR` to +existing absolute paths chosen for the project. + +```bash +cd "${CODONFM_REPO_DIR:?Set the repository path}" +bash run_dev.sh \ + --data-dir "${CODONFM_DATA_DIR:?Set the host data path}" \ + --checkpoints-dir "${CODONFM_CHECKPOINT_DIR:?Set the host checkpoint path}" +``` + +The host checkpoint directory is mounted at `/data/checkpoints` inside the +container. The image is `codon-fm-dev`; the container is +`codon-fm-dev-container`. + +Use only the checked-in public code and the dependency versions declared in +its `Dockerfile` and `requirements.txt`. Continue to Verify the runtime after +launch; checkpoint downloads are a separate requested action. + +## Run directly without Docker + +Use this path when the user prefers host execution or Docker is unavailable. +It requires a compatible NVIDIA driver, Python 3.11 for the commands below, +and a writable checkout. Confirm `python3.11 --version` succeeds before +installation. Reuse a compatible project environment; otherwise create a +dedicated virtual environment. Set `CODONFM_CACHE_DIR` to a writable cache +directory before running these commands: + +```bash +cd "${CODONFM_REPO_DIR:?Set the repository path}" +python3.11 -m venv .venv +. .venv/bin/activate +python -m pip install --upgrade pip +python -m pip install -r requirements.txt +mkdir -p "${CODONFM_CACHE_DIR:?Set a writable cache path}/matplotlib" +export MPLCONFIGDIR="$CODONFM_CACHE_DIR/matplotlib" +python -c "import sys, torch; available = torch.cuda.is_available(); \ +print(available, torch.cuda.get_device_name(0) if available else 'CUDA unavailable'); \ +sys.exit(0 if available else 1)" +``` + +The last command is the direct-host GPU verification; interpret it as described +under Verify the runtime. The requirements file configures the CUDA 12.4 +PyTorch index for xFormers. Use explicit host paths in subsequent runner +commands; no `/data/checkpoints` mount is created on this path. + +## Download a checkpoint + +Run only for a requested checkpoint. Reuse a suitable local copy first. +Check `hf --help` and `hf download --help` in the environment that will perform +the download. If the CLI is missing, use a separate download virtual environment +and `python -m pip install huggingface_hub`; preserve the model environment's +dependency versions. The [CLI documentation](https://huggingface.co/docs/huggingface_hub/en/guides/cli) +describes installation and supported options. Public ungated downloads do not +require `hf auth login`. + +Set `CODONFM_CHECKPOINT_DIR` to an absolute writable directory in the environment +running `hf`: the chosen host checkpoint root on the host, or `/data/checkpoints` +inside the launched container. Host shell variables are not automatically set +inside the container. Use supplied metadata for exact filenames and revisions; +keep the weights and `config.json` together. + +For the public 1B checkpoint: + +```bash +hf download nvidia/NV-CodonFM-Encodon-1B-v1 \ + NV-CodonFM-Encodon-1B-v1.safetensors config.json \ + --local-dir "${CODONFM_CHECKPOINT_DIR:?Set the checkpoint root}/encodon-1b" +``` + +### Small checkpoint example + +For a small demonstration, prefer the original public Encodon 80M weights: + +```bash +hf download nvidia/NV-CodonFM-Encodon-80M-v1 \ + NV-CodonFM-Encodon-80M-v1.safetensors config.json \ + --revision 399ca9fe17b57941a7bebc6788033919b417413c \ + --local-dir "${CODONFM_CHECKPOINT_DIR:?Set the checkpoint root}/encodon-80m" +``` + +The [checkpoint](https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1/tree/main) +is publicly accessible without a gated-model approval, and the weight file is +307,351,588 bytes. It need not be mirrored to GitHub LFS. The `-TE-` model IDs +use TransformerEngine in `bionemo-recipes`; use the original model IDs with this +public CodonFM codebase. Download only the weights and `config.json`, and reuse +an existing local checkpoint. + +Other supported public model IDs are: + +- `nvidia/NV-CodonFM-Encodon-80M-v1` +- `nvidia/NV-CodonFM-Encodon-600M-v1` +- `nvidia/NV-CodonFM-Encodon-Cdwt-1B-v1` + +Use `--model_name encodon_80m`, `encodon_600m`, or `encodon_1b` according to +architecture size. Cdwt-1B uses `encodon_1b` because Cdwt is a checkpoint +training property, not a separate architecture. + +For `.safetensors`, keep `config.json` in the same directory as the model +file. Never invent a Decodon or undocumented checkpoint path. + +After a successful download, confirm the expected files exist, `config.json` +parses, and any supplied byte size or checksum matches. Report the absolute +file paths and revision. A checkpoint-only request ends here; it does not +continue to GPU verification. For a combined request, continue only the other +requested path. + +## Verify the runtime + +This section applies only to working-environment requests. For direct-host +execution, use the GPU check at the end of Run directly without Docker in the +model's activated environment. For a running container, use a host terminal: + +```bash +docker exec codon-fm-dev-container python -c \ + "import sys, torch; available = torch.cuda.is_available(); \ +print(available, torch.cuda.get_device_name(0) if available else 'CUDA unavailable'); \ +sys.exit(0 if available else 1)" +``` + +Expect `True`, a GPU name, and exit status zero. `False` or an exception means +runtime verification failed; report the missing prerequisite or error. A CUDA +check establishes GPU access, not successful checkpoint loading or model +execution. Finish the environment request by reporting the verified runtime, +available checkpoint paths, and any checks that remain unperformed. + +## Failure handling + +- If a command fails, diagnose the reported cause. Retry an unchanged command + at most once for a transient failure, such as a download timeout. For a + persistent failure, stop that path and report the error and needed fix. +- For failed downloads, preserve the cache and partial files, retry the same + supported command when appropriate, and report which files remain missing + or unverified. Do not invent retry flags or claim an incomplete download + succeeded. +- If runtime prerequisites or container constraints cannot be met, complete + independent work within the request: inspect supplied source/configuration, + prepare setup commands, or download a requested checkpoint when possible. + Report completed work and the unmet prerequisites; do not claim the runtime + is working. + +## Public-v1 boundaries + +- Supported: Encodon 80M, 600M, 1B, and Cdwt-1B. +- Not supported: Decodon, Encodon 5B/10B, sequence generation, specialized + missense aggregation/fine-tuning, and `scripts/codon_optimize.py`. +- CodonFM consumes coding sequences. It is not a variant caller, aligner, GTF + annotator, or general VCF analysis tool. diff --git a/skills/codonfm-setup/agents/openai.yaml b/skills/codonfm-setup/agents/openai.yaml new file mode 100644 index 0000000..4c7925b --- /dev/null +++ b/skills/codonfm-setup/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "CodonFM Setup" + short_description: "Set up public CodonFM and Encodon checkpoints" + default_prompt: "Use $codonfm-setup to configure the public CodonFM environment and download an Encodon checkpoint." diff --git a/skills/codonfm-setup/evals/config.yml b/skills/codonfm-setup/evals/config.yml new file mode 100644 index 0000000..31036d7 --- /dev/null +++ b/skills/codonfm-setup/evals/config.yml @@ -0,0 +1,6 @@ +schema_version: 1 + +harbor: + agents: + claude-code: + model: aws/anthropic/bedrock-claude-opus-5 diff --git a/skills/codonfm-setup/evals/evals.json b/skills/codonfm-setup/evals/evals.json new file mode 100644 index 0000000..e5ac568 --- /dev/null +++ b/skills/codonfm-setup/evals/evals.json @@ -0,0 +1,54 @@ +{ + "skill_name": "codonfm-setup", + "evals": [ + { + "id": "codonfm-setup-001", + "prompt": "Prepare instructions for setting up public CodonFM in its development container and downloading Encodon 80M. Use the supplied source and checkpoint metadata. Explain the prerequisites, host directories, and checkpoint location inside the container.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json" + ], + "expected_output": "Setup and download instructions using the original public Encodon 80M checkpoint, its configuration file, and the checked-in container launcher, with accurate prerequisites and mount paths.", + "assertions": [ + "The download command uses nvidia/NV-CodonFM-Encodon-80M-v1, the supplied revision, NV-CodonFM-Encodon-80M-v1.safetensors, and config.json", + "The launcher command uses run_dev.sh with explicit host data/checkpoint directories and explains the /data/checkpoints mount", + "The agent explains existing-container replacement behavior before any proposed launch", + "The response explains Docker and GPU requirements for model execution and does not claim that setup instructions prove a working environment" + ], + "expected_skill": "codonfm-setup", + "expected_script": null + }, + { + "id": "codonfm-setup-002", + "prompt": "Can I use the public CodonFM repository for Decodon sequence generation? Check the supplied source before recommending a setup procedure.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json" + ], + "expected_output": "The agent identifies Decodon generation as unavailable in the supplied public implementation and does not invent a checkpoint or setup procedure.", + "assertions": [ + "The agent identifies Decodon as unavailable in the provided public model configuration", + "The agent does not recommend an undocumented Decodon download or implement the missing feature" + ], + "expected_skill": "codonfm-setup", + "expected_script": null + }, + { + "id": "codonfm-setup-003", + "prompt": "Prepare instructions for installing public CodonFM directly on a CUDA host where Docker is unavailable. Use the supplied source and checkpoint metadata. Include environment setup, checkpoint paths, and checks to run before model execution.", + "files": [ + "files/codonfm_source.zip", + "files/encodon_checkpoint.json" + ], + "expected_output": "Direct-host setup instructions using the public dependencies, a Python 3.11 virtual environment, writable cache, and CUDA verification, with user-chosen checkpoint paths.", + "assertions": [ + "The instructions use python3.11 -m venv and python -m pip install -r requirements.txt", + "The instructions set MPLCONFIGDIR to a writable location and provide torch.cuda.is_available verification", + "The agent uses the direct-host path without requiring Docker", + "The response clearly identifies verification still needed before claiming the environment is ready for model execution" + ], + "expected_skill": "codonfm-setup", + "expected_script": null + } + ] +} diff --git a/skills/codonfm-setup/evals/files/codonfm_source.zip b/skills/codonfm-setup/evals/files/codonfm_source.zip new file mode 100644 index 0000000..eb76611 Binary files /dev/null and b/skills/codonfm-setup/evals/files/codonfm_source.zip differ diff --git a/skills/codonfm-setup/evals/files/encodon_checkpoint.json b/skills/codonfm-setup/evals/files/encodon_checkpoint.json new file mode 100644 index 0000000..6c6230d --- /dev/null +++ b/skills/codonfm-setup/evals/files/encodon_checkpoint.json @@ -0,0 +1,33 @@ +{ + "repo_id": "nvidia/NV-CodonFM-Encodon-80M-v1", + "revision": "399ca9fe17b57941a7bebc6788033919b417413c", + "model_name": "encodon_80m", + "filename": "NV-CodonFM-Encodon-80M-v1.safetensors", + "size_bytes": 307351588, + "config_filename": "config.json", + "config": { + "vocab_size": 69, + "hidden_size": 1024, + "num_hidden_layers": 6, + "num_attention_heads": 8, + "intermediate_size": 4096, + "hidden_act": "gelu", + "hidden_dropout_prob": 0.1, + "attention_probs_dropout_prob": 0.1, + "initializer_range": 0.02, + "layer_norm_eps": 1e-12, + "pad_token_id": 3, + "position_embedding_type": "rotary", + "classifier_dropout": 0.1, + "rotary_theta": 10000.0, + "ignore_index": -100, + "loss_type": "cross_entropy", + "lora": false, + "lora_alpha": 32.0, + "lora_r": 16, + "lora_dropout": 0.1, + "finetune_strategy": "full" + }, + "source_url": "https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1/tree/399ca9fe17b57941a7bebc6788033919b417413c", + "weights_included": false +} diff --git a/skills/codonfm-setup/skill-card.md b/skills/codonfm-setup/skill-card.md new file mode 100644 index 0000000..8d7021d --- /dev/null +++ b/skills/codonfm-setup/skill-card.md @@ -0,0 +1,90 @@ +## Description:
+Set up the public CodonFM v1 repository and download public Encodon checkpoints.
+ +This skill is ready for commercial/non-commercial use.
+ +## Owner +NVIDIA
+ +### License/Terms of Use:
+Apache 2.0
+## Use Case:
+Developers and computational biologists who need to set up the CodonFM development environment, build or launch the development container, configure local data and checkpoint mounts, verify GPU access, or download public Encodon 80M, 600M, 1B, or Cdwt-1B weights.
+ +### Deployment Geography for Use:
+Global
+ +## Requirements / Dependencies:
+**Requires API Key or External Credential:** [No]
+**Credential Type(s):** [None]
+ +Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+ +## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
+ +## Reference(s):
+- [NV-CodonFM-Encodon-80M-v1](https://huggingface.co/nvidia/NV-CodonFM-Encodon-80M-v1)
+- [NV-CodonFM-Encodon-600M-v1](https://huggingface.co/nvidia/NV-CodonFM-Encodon-600M-v1)
+- [NV-CodonFM-Encodon-1B-v1](https://huggingface.co/nvidia/NV-CodonFM-Encodon-1B-v1)
+- [NV-CodonFM-Encodon-Cdwt-1B-v1](https://huggingface.co/nvidia/NV-CodonFM-Encodon-Cdwt-1B-v1)
+- [NGC CodonFM Encodon Catalog](https://catalog.ngc.nvidia.com/orgs/nvidia/teams/clara/models/nv_codonfm_encodon)
+- [HuggingFace Hub CLI Documentation](https://huggingface.co/docs/huggingface_hub/en/guides/cli)
+- [NVIDIA CUDA Compatibility](https://docs.nvidia.com/deploy/cuda-compatibility/minor-version-compatibility.html)
+- [PyTorch 24.10 Release Notes](https://docs.nvidia.com/deeplearning/frameworks/pytorch-release-notes/rel-24-10.html#driver-requirements)
+- [NVIDIA Deep Biology Research](https://research.nvidia.com/labs/dbr)
+ + +## Skill Output:
+**Output Type(s):** [Shell commands, Configuration instructions]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [None]
+ +## Evaluation Agents Used:
+- Claude Code (`aws/anthropic/bedrock-claude-opus-5`)
+- Codex (`openai/openai/gpt-5.5`)
+ + + +## Evaluation Tasks:
+3 evaluation tasks (3 positive) from a curated dataset, each run in an isolated sandbox pod.
+ +## Evaluation Metrics Used:
+Reported benchmark dimensions:
+- Security: Whether the skill is safe to use, checking for unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Whether the final answer is correct against the reference answer.
+- Discoverability: Whether the right skill was loaded when needed, including skill selection and decoy avoidance.
+- Effectiveness: Whether the skill helped complete the user's goal (50% goal accuracy + 50% expected workflow adherence).
+- Efficiency: Whether the skill avoided wasted tool calls and token usage (50% tool-call productivity + 50% token efficiency).
+ +Underlying evaluation signals used in this run:
+- `security`: Checks for unsafe operations, secret leakage, and unauthorized access.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `skill_execution`: Whether the expected skill was selected and the workflow executed.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity; routing is scored under Discoverability.
+- `token_efficiency`: Actual uncached prompt plus completion usage.
+ + + +## Evaluation Results:
+| Measure | Claude Code | Codex | +|---|---:|---:| +| Overall | 92.8% | 94.8% | +| Security | 100.0% | 100.0% | +| Correctness | 100.0% | 100.0% | +| Discoverability | 100.0% | 88.3% | +| Effectiveness | 75.0% | 91.7% | +| Efficiency | 89.2% | 94.0% | + +## Skill Version(s):
+be43117 (source: git SHA, committed 2026-10-07)
+ +## Ethical Considerations:
+NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
+ +(For Release on NVIDIA Platforms Only)
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/codonfm-setup/skill.oms.sig b/skills/codonfm-setup/skill.oms.sig new file mode 100644 index 0000000..5526194 --- /dev/null +++ b/skills/codonfm-setup/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAiY29kb25mbS1zZXR1cCIsCiAgICAgICJkaWdlc3QiOiB7CiAgICAgICAgInNoYTI1NiI6ICIxNTQwODA5MDY4ZmY3OTBlZWQ2MjMxMDJjZDQ2MTg4NzE0OTk1Y2RmNGZiNTViNmE1YWY4MTEzOTU5YTU3YTJmIgogICAgICB9CiAgICB9CiAgXSwKICAicHJlZGljYXRlVHlwZSI6ICJodHRwczovL21vZGVsX3NpZ25pbmcvc2lnbmF0dXJlL3YxLjAiLAogICJwcmVkaWNhdGUiOiB7CiAgICAicmVzb3VyY2VzIjogWwogICAgICB7CiAgICAgICAgIm5hbWUiOiAiQkVOQ0hNQVJLLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICI4NzMzNmVlZTAxZWM2OTcwNDJkNGExYjg4ODUwYTQ5OWIyZTU2MDA1YzQwYTE1OGIyY2VjZTIzMzk0YTM0NGU4IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiU0tJTEwubWQiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjNmZWZjYjlkZDJjNzg5N2M0M2E5MDI5ZWVhMDNlYjNhYzFlZWViOTBmYzA4YTkzZTQ0NTg2MDllMThmNTk3MmIiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJhZ2VudHMvb3BlbmFpLnlhbWwiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImNiN2ZmZGFiOWI5NjlkMTVhNzllNzg5NTFkZjQxMzNjZTc0MzBiOGUwY2U4MDIzZGU4YTIxNDA2NmM0MTZhNWYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJldmFscy9jb25maWcueW1sIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJkMDBhYWJjMzU0NjM2Yjg5NjcyOTVjN2ZjOGE5OWYyZTBiMzIzMjg1Yjc3ZTg0ZTY1NmI0OTgwYTcwZTg1MjY1IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZXZhbHMuanNvbiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMWEwOWNmN2ViYTJmNmY3MmU4Nzc3YTc5M2JkZGFjZTBjYTViYTgxZmZiYTQ3MWRjNDJhN2YyZDBhNGM4OThmYyIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogImV2YWxzL2ZpbGVzL2NvZG9uZm1fc291cmNlLnppcCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZjBjMTIyZjE5YTJkMzE2ZmU5NWNjZmFhMGFjNTFiZDYxM2NmYWJkZDA5ZmQ4OWZiYjc2NjU2MmE1ZjBlZWQyNyIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogImV2YWxzL2ZpbGVzL2VuY29kb25fY2hlY2twb2ludC5qc29uIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICI2OThmYTM1NDE0MDAzM2U5YWE4NTkxY2QyODYwNzAzYTRmY2RjMmViMjdlZGM4ZjJmMDc2MGM0MDgzZTU2M2VlIgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2tpbGwtY2FyZC5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiYTBmNzE0ZDk4YjZjYTQ5YjBhN2ViNWE1MzRmNDY0OTVkMTZlN2Y4MzQ5Mzg2NTQ3ZjBmOTY3Nzc3N2EzMTBiMiIKICAgICAgfQogICAgXSwKICAgICJzZXJpYWxpemF0aW9uIjogewogICAgICAiYWxsb3dfc3ltbGlua3MiOiBmYWxzZSwKICAgICAgImhhc2hfdHlwZSI6ICJzaGEyNTYiLAogICAgICAiaWdub3JlX3BhdGhzIjogWwogICAgICAgICIuZ2l0YXR0cmlidXRlcyIsCiAgICAgICAgIi5naXQiLAogICAgICAgICIuZ2l0aHViIiwKICAgICAgICAiLmdpdGlnbm9yZSIKICAgICAgXSwKICAgICAgIm1ldGhvZCI6ICJmaWxlcyIKICAgIH0KICB9Cn0=","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCMQDg2AITMO433iKwbnuYdZfe6fGixctjWrS6sYN0Q/kTA+gVNq1Hb3A1klwgJyXzSUsCMFosYIeR1p/ROBYX2JzwPdu0cUYlCACz5S+ABYegL9L5IgFnNdXdG7Py/uS9cySYHg==","keyid":""}]}} \ No newline at end of file diff --git a/skills/stage_eval_context.py b/skills/stage_eval_context.py new file mode 100644 index 0000000..29707bc --- /dev/null +++ b/skills/stage_eval_context.py @@ -0,0 +1,49 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Refresh the identical public source fixtures given to both Harbor conditions. + +Run after changing src/, Dockerfile, run_dev.sh, or requirements.txt. These +small deterministic ZIPs contain code, not skills, answers, weights, or deps. +""" + +import hashlib +import io +import json +import zipfile +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] + + +def build_context(): + paths = sorted((ROOT / "src").rglob("*.py")) + paths += [ROOT / name for name in ("Dockerfile", "run_dev.sh", "requirements.txt")] + contents = {path.relative_to(ROOT).as_posix(): path.read_bytes() for path in paths} + contents["source-manifest.json"] = (json.dumps({ + "repository": "https://github.com/NVIDIA-BioNeMo/CodonFM", + "purpose": "Public source snapshot for input preparation and API inspection; no weights or dependencies included", + "sha256": {name: hashlib.sha256(data).hexdigest() for name, data in sorted(contents.items())}, + }, indent=2) + "\n").encode() + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w", compression=zipfile.ZIP_DEFLATED) as archive: + for name, data in sorted(contents.items()): + info = zipfile.ZipInfo(name, date_time=(2026, 1, 1, 0, 0, 0)) + info.compress_type = zipfile.ZIP_DEFLATED + info.external_attr = 0o644 << 16 + archive.writestr(info, data) + return buffer.getvalue() + + +def main(): + data = build_context() + for skill in sorted((ROOT / "skills").glob("codonfm-*")): + dest = skill / "evals/files/codonfm_source.zip" + dest.parent.mkdir(parents=True, exist_ok=True) + dest.write_bytes(data) + print(f"{dest.relative_to(ROOT)}: {len(data)} bytes") + + +if __name__ == "__main__": + main() diff --git a/src/tasks.py b/src/tasks.py index 0a8d48c..57c403f 100644 --- a/src/tasks.py +++ b/src/tasks.py @@ -157,10 +157,19 @@ def evaluate( model.configure_model() data.setup("test") - if os.path.exists(model_ckpt_path): + # Safetensors files contain model weights only. Loading one again through + # torch.load() raises an unpickling error, so only inspect Lightning + # checkpoints for an optional datamodule state. + if ( + os.path.exists(model_ckpt_path) + and Path(model_ckpt_path).suffix.lower() == ".ckpt" + ): logging.info(f"Loading dataset checkpoint from {model_ckpt_path}") - data.load_state_dict(torch.load(model_ckpt_path)) - model.prediction_counter = data.init_global_step + checkpoint = torch.load(model_ckpt_path, map_location="cpu") + datamodule_state = checkpoint.get(data.__class__.__qualname__) + if datamodule_state is not None: + data.load_state_dict(datamodule_state) + model.prediction_counter = data.init_global_step trainer.logger = logger trainer.callbacks = list(callbacks.values()) @@ -169,4 +178,4 @@ def evaluate( trainer.predict(model, datamodule=data, return_predictions=False) - return \ No newline at end of file + return diff --git a/tests/skills/test_public_skills.py b/tests/skills/test_public_skills.py new file mode 100644 index 0000000..d6585be --- /dev/null +++ b/tests/skills/test_public_skills.py @@ -0,0 +1,397 @@ +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Static contract tests for the public CodonFM skills. + +These tests intentionally use only the Python standard library so command +usage can be checked before installing the GPU runtime. +""" + +import argparse +import ast +import csv +import hashlib +import importlib.util +import io +import json +import re +import shlex +import ssl +import subprocess +import sys +import tempfile +import unittest +import zipfile +from contextlib import redirect_stderr, redirect_stdout +from pathlib import Path +from unittest.mock import patch + + +REPO_ROOT = Path(__file__).resolve().parents[2] +SKILLS_ROOT = REPO_ROOT / "skills" +PUBLIC_SKILLS = { + "codonfm-setup", + "codonfm-score", + "codonfm-embed", + "codonfm-finetune", +} + + +def _frontmatter(skill_text: str) -> dict[str, str]: + match = re.match(r"\A---\n(.*?)\n---\n", skill_text, re.DOTALL) + if match is None: + raise AssertionError("SKILL.md is missing YAML frontmatter") + result = {} + for line in match.group(1).splitlines(): + if line.startswith(" "): + continue + key, separator, value = line.partition(":") + if not separator: + raise AssertionError(f"Invalid frontmatter line: {line}") + result[key.strip()] = value.strip() + return result + + +def _runner_commands(skill_text: str) -> list[list[str]]: + commands = [] + for block in re.findall(r"```bash\n(.*?)```", skill_text, re.DOTALL): + normalized = block.replace("\\\n", " ") + tokens = shlex.split(normalized, comments=True) + for index in range(len(tokens) - 3): + if tokens[index:index + 3] == ["python", "-m", "src.runner"]: + commands.append(tokens[index + 3:]) + break + return commands + + +def _public_runner_parser() -> argparse.ArgumentParser: + """Build the checked-in parser without importing the runner's GPU deps.""" + tree = ast.parse((REPO_ROOT / "src/runner.py").read_text()) + get_parser = next( + node + for node in tree.body + if isinstance(node, ast.FunctionDef) and node.name == "get_parser" + ) + parser_module = ast.Module(body=[get_parser], type_ignores=[]) + namespace = {"argparse": argparse} + exec(compile(parser_module, "src/runner.py", "exec"), namespace) + return namespace["get_parser"]() + + +def _evaluate_function(namespace): + """Load only tasks.evaluate so it can be tested without GPU packages.""" + tree = ast.parse((REPO_ROOT / "src/tasks.py").read_text()) + evaluate = next( + node + for node in tree.body + if isinstance(node, ast.FunctionDef) and node.name == "evaluate" + ) + module = ast.Module(body=[evaluate], type_ignores=[]) + exec(compile(module, "src/tasks.py", "exec"), namespace) + return namespace["evaluate"] + + +def _ribonn_helper(): + path = SKILLS_ROOT / "codonfm-finetune/scripts/prepare_ribonn.py" + spec = importlib.util.spec_from_file_location("prepare_ribonn", path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +class PublicSkillContractTests(unittest.TestCase): + def test_expected_public_skill_set(self): + actual = { + path.name + for path in SKILLS_ROOT.iterdir() + if path.is_dir() and (path / "SKILL.md").exists() + } + self.assertTrue(PUBLIC_SKILLS.issubset(actual)) + self.assertFalse((SKILLS_ROOT / "codonfm-optimize").exists()) + + def test_frontmatter_and_ui_metadata(self): + for skill_name in PUBLIC_SKILLS: + with self.subTest(skill=skill_name): + skill_dir = SKILLS_ROOT / skill_name + text = (skill_dir / "SKILL.md").read_text() + metadata = _frontmatter(text) + self.assertEqual(set(metadata), {"name", "description", "metadata"}) + self.assertEqual(metadata["name"], skill_name) + self.assertNotIn("TODO", text) + + ui = (skill_dir / "agents/openai.yaml").read_text() + self.assertIn("display_name:", ui) + self.assertIn("short_description:", ui) + self.assertIn(f"${skill_name}", ui) + + def test_evals_are_valid_and_named(self): + for skill_name in PUBLIC_SKILLS: + with self.subTest(skill=skill_name): + eval_path = SKILLS_ROOT / skill_name / "evals/evals.json" + payload = json.loads(eval_path.read_text()) + self.assertEqual(payload["skill_name"], skill_name) + self.assertGreater(len(payload["evals"]), 0) + ids = [case["id"] for case in payload["evals"]] + self.assertEqual(len(ids), len(set(ids))) + + def test_eval_inputs_exist_and_source_fixtures_match_repository(self): + expected_archive = None + for skill_name in sorted(PUBLIC_SKILLS): + evals = SKILLS_ROOT / skill_name / "evals" + payload = json.loads((evals / "evals.json").read_text()) + for case in payload["evals"]: + for relative in case.get("files", []): + path = (evals / relative).resolve() + self.assertTrue(path.is_relative_to(evals.resolve())) + self.assertTrue(path.is_file(), str(path)) + archive_path = evals / "files/codonfm_source.zip" + data = archive_path.read_bytes() + if expected_archive is not None: + self.assertEqual(data, expected_archive, "Trials must receive identical public source") + expected_archive = data + with zipfile.ZipFile(archive_path) as archive: + manifest = json.loads(archive.read("source-manifest.json")) + for name, sha in manifest["sha256"].items(): + self.assertEqual(archive.read(name), (REPO_ROOT / name).read_bytes(), + "Refresh source fixtures with python skills/stage_eval_context.py") + self.assertEqual(hashlib.sha256(archive.read(name)).hexdigest(), sha) + self.assertFalse(any(name.startswith("skills/") or name.endswith(".safetensors") + for name in archive.namelist())) + + def test_ribonn_preparation_preserves_data_without_runtime_dependencies(self): + files = SKILLS_ROOT / "codonfm-finetune/evals/files" + helper = SKILLS_ROOT / "codonfm-finetune/scripts/prepare_ribonn.py" + with tempfile.TemporaryDirectory() as tmp: + output = Path(tmp) / "prepared.csv" + result = subprocess.run([sys.executable, "-S", str(helper), "--input", + str(files / "ribonn_smoke.tsv"), "--output", str(output)], + cwd=REPO_ROOT, capture_output=True, text=True, timeout=10) + self.assertEqual(result.returncode, 0, result.stderr) + report = json.loads(result.stdout) + self.assertEqual(report["split_counts"], {"train": 8, "val": 2, "test": 2}) + self.assertEqual(json.loads(output.with_suffix(".metadata.json").read_text()), report) + with (files / "ribonn_smoke.tsv").open() as handle: + raw = {row["transcript_id"]: row for row in csv.DictReader(handle, delimiter="\t")} + with output.open() as handle: + rows = list(csv.DictReader(handle)) + self.assertEqual(len(rows), 12) + for row in rows: + source = raw[row["id"]] + start, length = int(source["utr5_size"]), int(source["cds_size"]) + self.assertEqual(row["ref_seq"], source["tx_sequence"][start:start + length]) + self.assertEqual(float(row["value"]), float(source["mean_te"])) + fold = int(source["fold"]) + self.assertEqual(row["split"], "val" if fold == 8 else "test" if fold == 9 else "train") + + def test_ribonn_download_requires_direct_https_success(self): + helper = _ribonn_helper() + fixture = (SKILLS_ROOT / "codonfm-finetune/evals/files/ribonn_smoke.tsv").read_bytes() + for status in (200, 301, 302, 303, 307, 308, 404, 500): + with self.subTest(status=status), tempfile.TemporaryDirectory() as tmp: + output = Path(tmp) / "prepared.csv" + response = io.BytesIO(fixture) + response.status = status + with ( + patch.object(helper.http.client, "HTTPSConnection") as https, + patch.object(sys, "argv", ["prepare_ribonn.py", "--output", str(output)]), + redirect_stdout(io.StringIO()) as stdout, + redirect_stderr(io.StringIO()) as stderr, + ): + connection = https.return_value + connection.getresponse.return_value = response + if status == 200: + helper.main() + report = json.loads(stdout.getvalue()) + self.assertEqual(report["split_counts"], {"train": 8, "val": 2, "test": 2}) + self.assertEqual(report["source"], helper.DATA_URL) + self.assertTrue(output.is_file()) + else: + with self.assertRaises(SystemExit) as failure: + helper.main() + self.assertEqual(failure.exception.code, 1) + self.assertIn(f"HTTP {status}", stderr.getvalue()) + self.assertEqual(list(Path(tmp).iterdir()), []) + https.assert_called_once_with("raw.githubusercontent.com", timeout=10) + connection.request.assert_called_once_with("GET", helper.DATA_PATH) + connection.close.assert_called_once_with() + self.assertTrue(response.closed) + + def test_ribonn_download_handles_transport_errors_without_output(self): + helper = _ribonn_helper() + errors = (TimeoutError("read timed out"), ssl.SSLCertVerificationError("untrusted certificate"), + helper.http.client.BadStatusLine("invalid HTTP response")) + for error in errors: + with self.subTest(error=type(error).__name__), tempfile.TemporaryDirectory() as tmp: + output = Path(tmp) / "prepared.csv" + with ( + patch.object(helper.http.client, "HTTPSConnection") as https, + patch.object(sys, "argv", ["prepare_ribonn.py", "--output", str(output)]), + redirect_stderr(io.StringIO()) as stderr, + ): + https.return_value.getresponse.side_effect = error + with self.assertRaises(SystemExit) as failure: + helper.main() + self.assertEqual(failure.exception.code, 1) + self.assertIn("Preparation failed:", stderr.getvalue()) + self.assertEqual(list(Path(tmp).iterdir()), []) + https.assert_called_once() + https.return_value.close.assert_called_once_with() + + def test_documented_runner_commands_parse(self): + parser = _public_runner_parser() + commands = [] + for skill_name in PUBLIC_SKILLS: + text = (SKILLS_ROOT / skill_name / "SKILL.md").read_text() + commands.extend((skill_name, command) for command in _runner_commands(text)) + + self.assertEqual({name for name, _ in commands}, { + "codonfm-score", + "codonfm-embed", + "codonfm-finetune", + }) + for skill_name, command in commands: + with self.subTest(skill=skill_name): + parser.parse_args(command) + + def test_fragile_command_requirements(self): + score = _runner_commands( + (SKILLS_ROOT / "codonfm-score/SKILL.md").read_text() + )[0] + self.assertIn("--mask_mutation", score) + self.assertIn("--extract-seq", score) + self.assertEqual(score[score.index("--num_gpus") + 1], "1") + + embed = _runner_commands( + (SKILLS_ROOT / "codonfm-embed/SKILL.md").read_text() + )[0] + self.assertEqual(embed[embed.index("--num_gpus") + 1], "1") + + finetune = _runner_commands( + (SKILLS_ROOT / "codonfm-finetune/SKILL.md").read_text() + )[0] + self.assertIn("--pretrained_ckpt_path", finetune) + self.assertNotIn("--checkpoint_path", finetune) + for flag in ( + "--lr", + "--check_val_every_n_epoch", + "--checkpoints_dir", + "--use_downstream_head", + ): + self.assertIn(flag, finetune) + self.assertEqual( + finetune[finetune.index("--check_val_every_n_epoch") + 1], "1" + ) + + def test_no_unavailable_feature_in_runner_commands(self): + forbidden = { + "MissenseDataset", + "missense_prediction", + "missense_synom_agg", + "missense_inference", + "missense_seq", + "decodon_200m", + "decodon_1b", + "mutation_pred_clm", + } + for skill_name in PUBLIC_SKILLS: + text = (SKILLS_ROOT / skill_name / "SKILL.md").read_text() + for command in _runner_commands(text): + with self.subTest(skill=skill_name, command=command): + self.assertTrue(forbidden.isdisjoint(command)) + + def test_setup_usage_matches_public_script(self): + text = (SKILLS_ROOT / "codonfm-setup/SKILL.md").read_text() + self.assertIn("bash run_dev.sh", text) + self.assertIn("--data-dir", text) + self.assertIn("--checkpoints-dir", text) + self.assertIn("/data/checkpoints", text) + self.assertIn("hf download nvidia/NV-CodonFM-Encodon-1B-v1", text) + self.assertIn("python3.11 -m venv .venv", text) + self.assertIn("python -m pip install -r requirements.txt", text) + self.assertIn("export MPLCONFIGDIR=", text) + self.assertIn("Use only the checked-in public code", text) + + def test_safetensors_eval_is_not_loaded_as_a_lightning_checkpoint(self): + calls = {"torch_load": 0, "predict": 0} + + class FakeTorch: + @staticmethod + def load(*args, **kwargs): + calls["torch_load"] += 1 + raise AssertionError("torch.load must not read safetensors") + + class FakeLogger: + def log_hyperparams(self, config): + self.config = config + + class FakeData: + init_global_step = 0 + + def setup(self, stage): + self.stage = stage + + def load_state_dict(self, state): + self.state = state + + class FakeModel: + prediction_counter = 0 + + def configure_model(self): + self.configured = True + + class FakeTrainer: + def __init__(self, **kwargs): + self.kwargs = kwargs + + def predict(self, *args, **kwargs): + calls["predict"] += 1 + + namespace = { + "Any": object, + "Dict": dict, + "Path": Path, + "Trainer": FakeTrainer, + "logging": type("Logging", (), {"info": staticmethod(lambda message: None)}), + "os": __import__("os"), + "seed_everything": lambda *args, **kwargs: None, + "torch": FakeTorch, + } + evaluate = _evaluate_function(namespace) + + with tempfile.TemporaryDirectory() as temp_dir: + model_path = Path(temp_dir) / "model.safetensors" + model_path.touch() + data = FakeData() + model = FakeModel() + evaluate( + config={ + "log": FakeLogger(), + "data": data, + "trainer": {}, + "model": model, + "callbacks": {}, + }, + config_dict={}, + model_ckpt_path=str(model_path), + out_dir=temp_dir, + ) + + self.assertEqual(calls["torch_load"], 0) + self.assertEqual(calls["predict"], 1) + self.assertEqual(data.stage, "test") + self.assertTrue(model.configured) + + def test_checked_in_references_exist(self): + for relative_path in ( + "notebooks/4-EnCodon-Downstream-Task-riboNN.ipynb", + "notebooks/5-EnCodon-Downstream-Task-mRFP-expression.ipynb", + "notebooks/6-EnCodon-Downstream-Task-mRNA-stability.ipynb", + "src/data/codon_bert_dataset.py", + "src/data/mutation_dataset.py", + "src/inference/encodon.py", + ): + self.assertTrue((REPO_ROOT / relative_path).is_file(), relative_path) + + +if __name__ == "__main__": + unittest.main()