{
  "meta": {
    "title": "Atlas of Input Representation Methods",
    "taxonomy_version": "input-representation-taxonomy-v1",
    "generated_from": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/snapshot_full_55_semantic_correction_2026-08-17",
    "canonical_corpus": [
      "/Users/bogdan.didenko/lpnu/review/data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles",
      "data/living_catalog_updates/update_2026-08-09/11_docling_vlm/profiles",
      "data/living_catalog_updates/update_2026-08-09/11_docling_vlm_manual_recall_xunzi_2026-08-11/profiles"
    ],
    "artifact_roots": [
      "/Users/bogdan.didenko/lpnu/review"
    ],
    "record_count": 55,
    "study_count": 54,
    "model_count": 109,
    "configuration_count": 467,
    "route_count": 585,
    "membership_group_count": 47,
    "source_figure_count": 67,
    "models_with_cropped_figure": 89,
    "models_without_suitable_figure": 20,
    "crop_ledger": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/snapshot_full_55_semantic_correction_2026-08-17/crop_ledger.json",
    "classification_unit": "One grounded route: a single source object carried through its transformation chain to the model-visible carrier consumed by the model.",
    "organizing_principle": "Partition routes by the model-visible carrier mechanism. Biological modality, lifecycle phase, text role, and fusion topology are orthogonal dimensions; they do not decide the family unless they change the carrier that reaches the model."
  },
  "families": [
    {
      "family_id": "text_native_token_stream",
      "name": "Text-native token streams",
      "definition": "Routes whose visible carrier is ordinary text tokens, including free-form prompts, structured task scaffolds, and serialized biological context rendered as text. The biologic content is carried by language tokens rather than dense vectors, images, or geometric states.",
      "structural_criterion": "The generative backbone consumes standard text tokens, even when the text is biologically specialized or template-driven.",
      "leaves": [
        {
          "leaf_id": "F1.L1",
          "name": "Plain language prompts and questions",
          "definition": "Natural-language prompts, questions, instructions, or conversational turns that reach the model as ordinary text without a specialized biological serialization step.",
          "include_when": [
            "The visible carrier is plain prose, a question, or an instruction.",
            "The route does not require a domain-specific serializer before model intake."
          ],
          "exclude_when": [
            "The source is first converted into a gene list, cell sentence, ranked profile, or other biological serialization.",
            "The visible carrier is a dense embedding, an image, or a diffusion state."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000086::route_003",
            "full_2026-07-06__rec_000086::route_006",
            "full_2026-07-06__rec_000827::route_004",
            "full_2026-07-06__rec_000090::route_032",
            "full_2026-07-06__rec_000090::route_015",
            "full_2026-07-06__rec_000950::route_010",
            "full_2026-07-06__rec_000771::route_002"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        },
        {
          "leaf_id": "F1.L2",
          "name": "Structured biological prompts and task scaffolds",
          "definition": "Text prompts that encode explicit biological slots, metadata fields, retrieval directives, benchmark scaffolds, or relation-context prompts.",
          "include_when": [
            "The text contains explicit biological fields, slots, or template-like task structure.",
            "The prompt functions as a structured control frame rather than free-form conversation."
          ],
          "exclude_when": [
            "The input is only plain conversational text with no task scaffolding.",
            "The carrier is a biological sentence serialization or a dense embedding prefix."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000086::route_010",
            "full_2026-07-06__rec_000086::route_012",
            "full_2026-07-06__rec_000086::route_015",
            "full_2026-07-06__rec_000090::route_021",
            "full_2026-07-06__rec_000090::route_022",
            "full_2026-07-06__rec_000090::route_023",
            "full_2026-07-06__rec_000090::route_024",
            "full_2026-07-06__rec_000090::route_025",
            "full_2026-07-06__rec_000090::route_033",
            "full_2026-07-06__rec_000090::route_034",
            "full_2026-07-06__rec_000090::route_035"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000827::route_004",
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_002"
          ]
        },
        {
          "leaf_id": "F1.L3",
          "name": "Serialized biological context and ordered profiles",
          "definition": "Text-token sequences produced by serializing cells, genes, neighborhoods, ranked profiles, or multi-cell contexts into language-like sentences or ordered lists before model intake.",
          "include_when": [
            "The route converts biological objects into cell sentences, spatial sentences, gene lists, ranked expression profiles, or comparable serialized text.",
            "The visible carrier is a textual summary of biological context, often across multiple nearby entities or samples."
          ],
          "exclude_when": [
            "The route is a simple question or instruction with no biological serialization.",
            "The route is a fixed slot template rather than a serialized biological narrative.",
            "The carrier is a dense vector, image, or diffusion state."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000063::route_001",
            "full_2026-07-06__rec_000063::route_002",
            "full_2026-07-06__rec_000063::route_003",
            "full_2026-07-06__rec_000063::route_004",
            "full_2026-07-06__rec_000063::route_005",
            "full_2026-07-06__rec_000063::route_006",
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000090::route_002",
            "full_2026-07-06__rec_000090::route_003",
            "full_2026-07-06__rec_000090::route_004",
            "full_2026-07-06__rec_000090::route_005",
            "full_2026-07-06__rec_000090::route_006",
            "full_2026-07-06__rec_000090::route_007",
            "full_2026-07-06__rec_000090::route_008",
            "full_2026-07-06__rec_000090::route_009",
            "full_2026-07-06__rec_000090::route_010",
            "full_2026-07-06__rec_000090::route_011",
            "full_2026-07-06__rec_000090::route_012",
            "full_2026-07-06__rec_000090::route_013",
            "full_2026-07-06__rec_000090::route_014",
            "full_2026-07-06__rec_000090::route_016",
            "full_2026-07-06__rec_000090::route_017",
            "full_2026-07-06__rec_000090::route_018",
            "full_2026-07-06__rec_000090::route_019",
            "full_2026-07-06__rec_000090::route_020",
            "full_2026-07-06__rec_000090::route_028",
            "full_2026-07-06__rec_000090::route_029",
            "full_2026-07-06__rec_000090::route_030",
            "full_2026-07-06__rec_000090::route_031",
            "full_2026-07-06__rec_000827::route_005",
            "full_2026-07-06__rec_000827::route_006",
            "full_2026-07-06__rec_000827::route_007"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000827::route_004",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        }
      ],
      "code": "F1",
      "label": "Text-native token streams",
      "short": "Lexical",
      "color": "#bd3150",
      "route_count": 325,
      "model_count": 84,
      "subtypes": [
        {
          "leaf_id": "F1.L1",
          "name": "Plain language prompts and questions",
          "definition": "Natural-language prompts, questions, instructions, or conversational turns that reach the model as ordinary text without a specialized biological serialization step.",
          "include_when": [
            "The visible carrier is plain prose, a question, or an instruction.",
            "The route does not require a domain-specific serializer before model intake."
          ],
          "exclude_when": [
            "The source is first converted into a gene list, cell sentence, ranked profile, or other biological serialization.",
            "The visible carrier is a dense embedding, an image, or a diffusion state."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000086::route_003",
            "full_2026-07-06__rec_000086::route_006",
            "full_2026-07-06__rec_000827::route_004",
            "full_2026-07-06__rec_000090::route_032",
            "full_2026-07-06__rec_000090::route_015",
            "full_2026-07-06__rec_000950::route_010",
            "full_2026-07-06__rec_000771::route_002"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "plain_language_prompt_or_question",
          "route_count": 61,
          "model_count": 37,
          "example": {
            "input": "What phenotype does this cell exhibit?",
            "carrier": "[What] [phenotype] [does] [this] [cell] ...",
            "model": "ordinary tokenizer → LLM",
            "note": "Natural-language instructions or questions remain ordinary lexical tokens."
          }
        },
        {
          "leaf_id": "F1.L2",
          "name": "Structured biological prompts and task scaffolds",
          "definition": "Text prompts that encode explicit biological slots, metadata fields, retrieval directives, benchmark scaffolds, or relation-context prompts.",
          "include_when": [
            "The text contains explicit biological fields, slots, or template-like task structure.",
            "The prompt functions as a structured control frame rather than free-form conversation."
          ],
          "exclude_when": [
            "The input is only plain conversational text with no task scaffolding.",
            "The carrier is a biological sentence serialization or a dense embedding prefix."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000086::route_010",
            "full_2026-07-06__rec_000086::route_012",
            "full_2026-07-06__rec_000086::route_015",
            "full_2026-07-06__rec_000090::route_021",
            "full_2026-07-06__rec_000090::route_022",
            "full_2026-07-06__rec_000090::route_023",
            "full_2026-07-06__rec_000090::route_024",
            "full_2026-07-06__rec_000090::route_025",
            "full_2026-07-06__rec_000090::route_033",
            "full_2026-07-06__rec_000090::route_034",
            "full_2026-07-06__rec_000090::route_035"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000827::route_004",
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_002"
          ],
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "route_count": 184,
          "model_count": 61,
          "example": {
            "input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
            "carrier": "task tag + labeled biological fields",
            "model": "template tokenizer → LLM",
            "note": "A controlled scaffold tells the model what biological operation to perform."
          }
        },
        {
          "leaf_id": "F1.L3",
          "name": "Serialized biological context and ordered profiles",
          "definition": "Text-token sequences produced by serializing cells, genes, neighborhoods, ranked profiles, or multi-cell contexts into language-like sentences or ordered lists before model intake.",
          "include_when": [
            "The route converts biological objects into cell sentences, spatial sentences, gene lists, ranked expression profiles, or comparable serialized text.",
            "The visible carrier is a textual summary of biological context, often across multiple nearby entities or samples."
          ],
          "exclude_when": [
            "The route is a simple question or instruction with no biological serialization.",
            "The route is a fixed slot template rather than a serialized biological narrative.",
            "The carrier is a dense vector, image, or diffusion state."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000063::route_001",
            "full_2026-07-06__rec_000063::route_002",
            "full_2026-07-06__rec_000063::route_003",
            "full_2026-07-06__rec_000063::route_004",
            "full_2026-07-06__rec_000063::route_005",
            "full_2026-07-06__rec_000063::route_006",
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000090::route_002",
            "full_2026-07-06__rec_000090::route_003",
            "full_2026-07-06__rec_000090::route_004",
            "full_2026-07-06__rec_000090::route_005",
            "full_2026-07-06__rec_000090::route_006",
            "full_2026-07-06__rec_000090::route_007",
            "full_2026-07-06__rec_000090::route_008",
            "full_2026-07-06__rec_000090::route_009",
            "full_2026-07-06__rec_000090::route_010",
            "full_2026-07-06__rec_000090::route_011",
            "full_2026-07-06__rec_000090::route_012",
            "full_2026-07-06__rec_000090::route_013",
            "full_2026-07-06__rec_000090::route_014",
            "full_2026-07-06__rec_000090::route_016",
            "full_2026-07-06__rec_000090::route_017",
            "full_2026-07-06__rec_000090::route_018",
            "full_2026-07-06__rec_000090::route_019",
            "full_2026-07-06__rec_000090::route_020",
            "full_2026-07-06__rec_000090::route_028",
            "full_2026-07-06__rec_000090::route_029",
            "full_2026-07-06__rec_000090::route_030",
            "full_2026-07-06__rec_000090::route_031",
            "full_2026-07-06__rec_000827::route_005",
            "full_2026-07-06__rec_000827::route_006",
            "full_2026-07-06__rec_000827::route_007"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000827::route_004",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "route_count": 80,
          "model_count": 29,
          "example": {
            "input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
            "carrier": "ordered textual cell/profile sentence",
            "model": "serializer → tokenizer → LLM",
            "note": "A biological vector or set is deterministically rendered as ordered text."
          }
        }
      ]
    },
    {
      "family_id": "discrete_biological_symbol_stream",
      "name": "Discrete biological symbol streams",
      "definition": "Routes whose biological source is represented as a discrete symbol stream native to or added to the tokenizer, rather than as prose text or dense embeddings. This includes biological BPE streams, multi-track structural alphabets, and learned quantized codebook IDs.",
      "structural_criterion": "The model-visible carrier is a tokenized biological sequence or structured biological symbol stream.",
      "leaves": [
        {
          "leaf_id": "F2.L1",
          "name": "Native biological token streams",
          "definition": "DNA, RNA, or protein corpora rendered as discrete biological tokens through an extended tokenizer or chunking protocol.",
          "include_when": [
            "The route uses tokenized DNA, RNA, protein, or comparable biological sequence corpora.",
            "The visible form is a chunked biological token stream rather than a prose prompt or embedding."
          ],
          "exclude_when": [
            "The route is an ordinary text prompt or instruction.",
            "The route is a dense embedding prefix or an image patch input."
          ],
          "positive_route_refs": [
            "june_update_2026-06-10__rec_000350::route_001",
            "june_update_2026-06-10__rec_000350::route_002",
            "june_update_2026-06-10__rec_000350::route_003"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        },
        {
          "leaf_id": "F2.L2",
          "name": "Multi-track structural symbol streams",
          "definition": "A token stream that explicitly preserves multiple structural tracks or aligned alphabets, such as sequence plus secondary-structure or tertiary-structure tracks.",
          "include_when": [
            "The route preserves several aligned structural tracks in one discrete token stream.",
            "The visible carrier is a structured symbolic alphabet, not an embedding."
          ],
          "exclude_when": [
            "The route is a single-track tokenized corpus.",
            "The route is a continuous prefix, image patch, or diffusion state."
          ],
          "positive_route_refs": [
            "june_update_2026-06-10__rec_000350::route_005",
            "june_update_2026-06-10__rec_000350::route_006"
          ],
          "counterexample_route_refs": [
            "june_update_2026-06-10__rec_000350::route_001",
            "full_2026-07-06__rec_000950::route_002",
            "full_2026-07-06__rec_000086::route_002",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        },
        {
          "leaf_id": "F2.L3",
          "name": "Learned quantized IDs and codebook tokens",
          "definition": "Routes whose biological content is rendered as learned discrete IDs from VQ, RVQ, or codebook-style tokenizers. These are discrete symbols, not continuous embeddings and not native tokenizer words.",
          "include_when": [
            "The route uses VQ, RVQ, or codebook IDs as model-visible inputs.",
            "The discrete IDs are embedded into the LLM vocabulary or otherwise treated as symbol tokens.",
            "The route is not just a continuous latent prefix."
          ],
          "exclude_when": [
            "The route is a continuous embedding or resampler output.",
            "The route is ordinary prose text or a native biological token corpus."
          ],
          "positive_route_refs": [
            "june_update_2026-06-10__rec_000121::route_003",
            "june_update_2026-06-10__rec_000121::route_004",
            "june_update_2026-06-10__rec_000121::route_006"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000350::route_001"
          ]
        }
      ],
      "code": "F2",
      "label": "Discrete biological symbol streams",
      "short": "Discrete bio",
      "color": "#13856f",
      "route_count": 56,
      "model_count": 17,
      "subtypes": [
        {
          "leaf_id": "F2.L1",
          "name": "Native biological token streams",
          "definition": "DNA, RNA, or protein corpora rendered as discrete biological tokens through an extended tokenizer or chunking protocol.",
          "include_when": [
            "The route uses tokenized DNA, RNA, protein, or comparable biological sequence corpora.",
            "The visible form is a chunked biological token stream rather than a prose prompt or embedding."
          ],
          "exclude_when": [
            "The route is an ordinary text prompt or instruction.",
            "The route is a dense embedding prefix or an image patch input."
          ],
          "positive_route_refs": [
            "june_update_2026-06-10__rec_000350::route_001",
            "june_update_2026-06-10__rec_000350::route_002",
            "june_update_2026-06-10__rec_000350::route_003"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "native_biological_token_stream",
          "route_count": 50,
          "model_count": 15,
          "example": {
            "input": "A C G T G C A ...",
            "carrier": "native nucleotide/amino-acid token IDs",
            "model": "biological tokenizer → generator",
            "note": "The biological alphabet itself forms the discrete sequence consumed by the model."
          }
        },
        {
          "leaf_id": "F2.L2",
          "name": "Multi-track structural symbol streams",
          "definition": "A token stream that explicitly preserves multiple structural tracks or aligned alphabets, such as sequence plus secondary-structure or tertiary-structure tracks.",
          "include_when": [
            "The route preserves several aligned structural tracks in one discrete token stream.",
            "The visible carrier is a structured symbolic alphabet, not an embedding."
          ],
          "exclude_when": [
            "The route is a single-track tokenized corpus.",
            "The route is a continuous prefix, image patch, or diffusion state."
          ],
          "positive_route_refs": [
            "june_update_2026-06-10__rec_000350::route_005",
            "june_update_2026-06-10__rec_000350::route_006"
          ],
          "counterexample_route_refs": [
            "june_update_2026-06-10__rec_000350::route_001",
            "full_2026-07-06__rec_000950::route_002",
            "full_2026-07-06__rec_000086::route_002",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "multi_track_structural_symbol_stream",
          "route_count": 4,
          "model_count": 3,
          "example": {
            "input": "AA: M K T ...   SS: H H C ...",
            "carrier": "aligned sequence + structure tracks",
            "model": "multi-track tokenizer → generator",
            "note": "Several synchronized symbolic tracks encode sequence and structure together."
          }
        },
        {
          "leaf_id": "F2.L3",
          "name": "Learned quantized IDs and codebook tokens",
          "definition": "Routes whose biological content is rendered as learned discrete IDs from VQ, RVQ, or codebook-style tokenizers. These are discrete symbols, not continuous embeddings and not native tokenizer words.",
          "include_when": [
            "The route uses VQ, RVQ, or codebook IDs as model-visible inputs.",
            "The discrete IDs are embedded into the LLM vocabulary or otherwise treated as symbol tokens.",
            "The route is not just a continuous latent prefix."
          ],
          "exclude_when": [
            "The route is a continuous embedding or resampler output.",
            "The route is ordinary prose text or a native biological token corpus."
          ],
          "positive_route_refs": [
            "june_update_2026-06-10__rec_000121::route_003",
            "june_update_2026-06-10__rec_000121::route_004",
            "june_update_2026-06-10__rec_000121::route_006"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000350::route_001"
          ],
          "subtype_id": "learned_quantized_id_or_codebook_token",
          "route_count": 2,
          "model_count": 1,
          "example": {
            "input": "continuous biology → quantizer",
            "carrier": "[BIO_187] [BIO_042] [BIO_913]",
            "model": "VQ/RVQ codebook IDs → generator",
            "note": "A learned codebook turns continuous biological states into discrete IDs."
          }
        }
      ]
    },
    {
      "family_id": "dense_continuous_carrier",
      "name": "Dense continuous carriers",
      "definition": "Routes whose model-visible form is a learned dense vector, soft-token block, prefix embedding, resampled latent, or similar continuous carrier inserted into a generative backbone. The biological source is encoded rather than discretely tokenized.",
      "structural_criterion": "The source is converted into continuous embeddings, soft tokens, or learned latent blocks before reaching the generator.",
      "leaves": [
        {
          "leaf_id": "F3.L1",
          "name": "Direct projected embeddings",
          "definition": "A continuous embedding projected from a biological encoder into the language or multimodal backbone, typically via a linear layer, projector, or latent-space mapping.",
          "include_when": [
            "The route explicitly projects biological inputs into an embedding space or unified latent space.",
            "The visible carrier is a dense vector block rather than text tokens or image patches."
          ],
          "exclude_when": [
            "The carrier is a fixed prefix block of learned virtual tokens.",
            "The carrier is an image patch sequence or a discrete biological token stream.",
            "The carrier is a diffusion state or geometric constraint."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000086::route_004",
            "full_2026-07-06__rec_000086::route_005",
            "full_2026-07-06__rec_000086::route_007",
            "full_2026-07-06__rec_000086::route_008",
            "full_2026-07-06__rec_000086::route_009",
            "full_2026-07-06__rec_000086::route_013",
            "full_2026-07-06__rec_000827::route_010",
            "full_2026-07-06__rec_000827::route_011",
            "full_2026-07-06__rec_000827::route_012",
            "full_2026-07-06__rec_000827::route_013",
            "june_update_2026-06-10__rec_000248::route_003",
            "june_update_2026-06-10__rec_000248::route_005",
            "june_update_2026-06-10__rec_000248::route_009"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000950::route_001",
            "full_2026-07-06__rec_000827::route_004",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        },
        {
          "leaf_id": "F3.L2",
          "name": "Virtual-token prefixes",
          "definition": "A fixed-length block of learned virtual tokens or multimodal prefix tokens that stands in for the biological source before being concatenated or inserted into the backbone.",
          "include_when": [
            "The route uses a fixed prefix, placeholder-bearing virtual-token block, or multimodal prefix.",
            "The biological source is not passed as ordinary text but as a learned prefix-like embedding bundle."
          ],
          "exclude_when": [
            "The route uses direct projection into a latent space without a dedicated prefix block.",
            "The carrier is a pooled embedding, an image patch set, or a diffusion state."
          ],
          "positive_route_refs": [
            "june_update_2026-06-10__rec_000148::route_001",
            "june_update_2026-06-10__rec_000148::route_003",
            "june_update_2026-06-10__rec_000148::route_005",
            "june_update_2026-06-10__rec_000148::route_006",
            "june_update_2026-06-10__rec_000148::route_008",
            "june_update_2026-06-10__rec_000148::route_010",
            "june_update_2026-06-10__rec_000148::route_017",
            "june_update_2026-06-10__rec_000148::route_018",
            "june_update_2026-06-10__rec_000152::route_001",
            "june_update_2026-06-10__rec_000152::route_002",
            "june_update_2026-06-10__rec_000152::route_004",
            "june_update_2026-06-10__rec_000152::route_005",
            "june_update_2026-06-10__rec_000152::route_006",
            "june_update_2026-06-10__rec_000152::route_008",
            "june_update_2026-06-10__rec_000152::route_009",
            "june_update_2026-06-10__rec_000152::route_010"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000950::route_002",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        },
        {
          "leaf_id": "F3.L3",
          "name": "Connector-mediated embeddings",
          "definition": "A dense carrier that reaches the backbone through a learned connector such as cross-attention, a resampler, a query-former, an adapter layer, or a query connector.",
          "include_when": [
            "The route explicitly uses cross-attention, a resampler, a query-former, a query connector, or an adapter.",
            "The biological content is converted to dense tokens before fusion with the backbone."
          ],
          "exclude_when": [
            "The route is a fixed virtual-token prefix.",
            "The route is a direct embedding projection without a separate mediator.",
            "The route is a raw image or a discrete biological token stream."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000771::route_001",
            "full_2026-07-06__rec_000771::route_003",
            "full_2026-07-06__rec_000827::route_001",
            "full_2026-07-06__rec_000827::route_002",
            "full_2026-07-06__rec_000827::route_003",
            "full_2026-07-06__rec_000827::route_004",
            "full_2026-07-06__rec_000827::route_008",
            "full_2026-07-06__rec_000827::route_009",
            "june_update_2026-06-10__rec_000248::route_001",
            "june_update_2026-06-10__rec_000248::route_002",
            "june_update_2026-06-10__rec_000248::route_004",
            "june_update_2026-06-10__rec_000248::route_006"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        },
        {
          "leaf_id": "F3.L4",
          "name": "Pooled or aggregated embeddings",
          "definition": "A continuous carrier formed by aggregating multiple inputs into one embedding before model intake, such as averaging selected-cell embeddings or similar pooling operations.",
          "include_when": [
            "The route compresses multiple biological items into one embedding before downstream use.",
            "The visible form is a pooled representation rather than a single-source projection."
          ],
          "exclude_when": [
            "The route keeps separate tokens for each instance in the prompt.",
            "The route uses raw images, discrete tokens, or a diffusion state."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000827::route_003"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000827::route_001",
            "full_2026-07-06__rec_000090::route_014",
            "full_2026-07-06__rec_000950::route_002",
            "june_update_2026-06-10__rec_000121::route_004"
          ]
        }
      ],
      "code": "F3",
      "label": "Dense continuous carriers",
      "short": "Continuous",
      "color": "#2d63a7",
      "route_count": 129,
      "model_count": 39,
      "subtypes": [
        {
          "leaf_id": "F3.L1",
          "name": "Direct projected embeddings",
          "definition": "A continuous embedding projected from a biological encoder into the language or multimodal backbone, typically via a linear layer, projector, or latent-space mapping.",
          "include_when": [
            "The route explicitly projects biological inputs into an embedding space or unified latent space.",
            "The visible carrier is a dense vector block rather than text tokens or image patches."
          ],
          "exclude_when": [
            "The carrier is a fixed prefix block of learned virtual tokens.",
            "The carrier is an image patch sequence or a discrete biological token stream.",
            "The carrier is a diffusion state or geometric constraint."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000086::route_004",
            "full_2026-07-06__rec_000086::route_005",
            "full_2026-07-06__rec_000086::route_007",
            "full_2026-07-06__rec_000086::route_008",
            "full_2026-07-06__rec_000086::route_009",
            "full_2026-07-06__rec_000086::route_013",
            "full_2026-07-06__rec_000827::route_010",
            "full_2026-07-06__rec_000827::route_011",
            "full_2026-07-06__rec_000827::route_012",
            "full_2026-07-06__rec_000827::route_013",
            "june_update_2026-06-10__rec_000248::route_003",
            "june_update_2026-06-10__rec_000248::route_005",
            "june_update_2026-06-10__rec_000248::route_009"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000950::route_001",
            "full_2026-07-06__rec_000827::route_004",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "direct_projected_embedding",
          "route_count": 76,
          "model_count": 27,
          "example": {
            "input": "RNA vector x ∈ ℝᵍ",
            "carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
            "model": "projected vectors → generative backbone",
            "note": "An external biological representation is linearly or nonlinearly projected into model space."
          }
        },
        {
          "leaf_id": "F3.L2",
          "name": "Virtual-token prefixes",
          "definition": "A fixed-length block of learned virtual tokens or multimodal prefix tokens that stands in for the biological source before being concatenated or inserted into the backbone.",
          "include_when": [
            "The route uses a fixed prefix, placeholder-bearing virtual-token block, or multimodal prefix.",
            "The biological source is not passed as ordinary text but as a learned prefix-like embedding bundle."
          ],
          "exclude_when": [
            "The route uses direct projection into a latent space without a dedicated prefix block.",
            "The carrier is a pooled embedding, an image patch set, or a diffusion state."
          ],
          "positive_route_refs": [
            "june_update_2026-06-10__rec_000148::route_001",
            "june_update_2026-06-10__rec_000148::route_003",
            "june_update_2026-06-10__rec_000148::route_005",
            "june_update_2026-06-10__rec_000148::route_006",
            "june_update_2026-06-10__rec_000148::route_008",
            "june_update_2026-06-10__rec_000148::route_010",
            "june_update_2026-06-10__rec_000148::route_017",
            "june_update_2026-06-10__rec_000148::route_018",
            "june_update_2026-06-10__rec_000152::route_001",
            "june_update_2026-06-10__rec_000152::route_002",
            "june_update_2026-06-10__rec_000152::route_004",
            "june_update_2026-06-10__rec_000152::route_005",
            "june_update_2026-06-10__rec_000152::route_006",
            "june_update_2026-06-10__rec_000152::route_008",
            "june_update_2026-06-10__rec_000152::route_009",
            "june_update_2026-06-10__rec_000152::route_010"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000950::route_002",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "virtual_token_prefix",
          "route_count": 9,
          "model_count": 3,
          "example": {
            "input": "cell embedding + prompt",
            "carrier": "<bio₁> <bio₂> ... <bioₖ> [prompt tokens]",
            "model": "soft prefix → LLM stream",
            "note": "Continuous vectors occupy token-like prefix positions without ordinary token IDs."
          }
        },
        {
          "leaf_id": "F3.L3",
          "name": "Connector-mediated embeddings",
          "definition": "A dense carrier that reaches the backbone through a learned connector such as cross-attention, a resampler, a query-former, an adapter layer, or a query connector.",
          "include_when": [
            "The route explicitly uses cross-attention, a resampler, a query-former, a query connector, or an adapter.",
            "The biological content is converted to dense tokens before fusion with the backbone."
          ],
          "exclude_when": [
            "The route is a fixed virtual-token prefix.",
            "The route is a direct embedding projection without a separate mediator.",
            "The route is a raw image or a discrete biological token stream."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000771::route_001",
            "full_2026-07-06__rec_000771::route_003",
            "full_2026-07-06__rec_000827::route_001",
            "full_2026-07-06__rec_000827::route_002",
            "full_2026-07-06__rec_000827::route_003",
            "full_2026-07-06__rec_000827::route_004",
            "full_2026-07-06__rec_000827::route_008",
            "full_2026-07-06__rec_000827::route_009",
            "june_update_2026-06-10__rec_000248::route_001",
            "june_update_2026-06-10__rec_000248::route_002",
            "june_update_2026-06-10__rec_000248::route_004",
            "june_update_2026-06-10__rec_000248::route_006"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000090::route_001",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "connector_mediated_embedding",
          "route_count": 24,
          "model_count": 10,
          "example": {
            "input": "image / omics encoder states",
            "carrier": "Q-Former or adapter query vectors",
            "model": "connector → LLM cross-modal interface",
            "note": "A learned connector selects or transforms encoder states before language generation."
          }
        },
        {
          "leaf_id": "F3.L4",
          "name": "Pooled or aggregated embeddings",
          "definition": "A continuous carrier formed by aggregating multiple inputs into one embedding before model intake, such as averaging selected-cell embeddings or similar pooling operations.",
          "include_when": [
            "The route compresses multiple biological items into one embedding before downstream use.",
            "The visible form is a pooled representation rather than a single-source projection."
          ],
          "exclude_when": [
            "The route keeps separate tokens for each instance in the prompt.",
            "The route uses raw images, discrete tokens, or a diffusion state."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000827::route_003"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000827::route_001",
            "full_2026-07-06__rec_000090::route_014",
            "full_2026-07-06__rec_000950::route_002",
            "june_update_2026-06-10__rec_000121::route_004"
          ],
          "subtype_id": "pooled_or_aggregated_embedding",
          "route_count": 20,
          "model_count": 10,
          "example": {
            "input": "{gene/cell/patch embeddings}",
            "carrier": "mean/attention pool = one compact vector",
            "model": "aggregator → generator",
            "note": "Many local states are summarized before they reach the generative component."
          }
        }
      ]
    },
    {
      "family_id": "visual_raster_carrier",
      "name": "Visual raster carriers",
      "definition": "Routes whose visible carrier is an image, patch grid, slide tile, or other rasterized visual carrier processed by a vision encoder or equivalent visual front end.",
      "structural_criterion": "The source is presented as pixels, patches, or visual tokens rather than language tokens or dense embeddings alone.",
      "leaves": [
        {
          "leaf_id": "F4.L1",
          "name": "Raw slide or patch input",
          "definition": "A histology or similar biomedical image presented directly as a raster image, patch, or patch set to a vision encoder.",
          "include_when": [
            "The route starts from a raw image patch or whole-slide image.",
            "The visible carrier is an image tensor or patch grid."
          ],
          "exclude_when": [
            "The route is a projected embedding or text prompt.",
            "The route is a diffusion state or discrete biological token stream."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000060::route_002",
            "full_2026-07-06__rec_000060::route_003",
            "june_update_2026-06-10__rec_000148::route_001",
            "june_update_2026-06-10__rec_000148::route_003",
            "june_update_2026-06-10__rec_000148::route_005",
            "june_update_2026-06-10__rec_000148::route_006",
            "june_update_2026-06-10__rec_000148::route_007",
            "june_update_2026-06-10__rec_000148::route_008",
            "june_update_2026-06-10__rec_000148::route_009",
            "june_update_2026-06-10__rec_000148::route_010",
            "june_update_2026-06-10__rec_000148::route_011",
            "june_update_2026-06-10__rec_000148::route_013",
            "june_update_2026-06-10__rec_000148::route_016",
            "june_update_2026-06-10__rec_000148::route_018",
            "june_update_2026-06-10__rec_000152::route_001",
            "june_update_2026-06-10__rec_000152::route_002"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000827::route_008",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        },
        {
          "leaf_id": "F4.L2",
          "name": "Patch-context or case-level visual reasoning",
          "definition": "A visual route where multiple patches or registered sections are combined into a case-level image context before prediction or reasoning.",
          "include_when": [
            "The route uses neighboring patches, serial sections, or case-level visual context.",
            "The visible carrier is still visual, but the unit is a patch set or contextualized slide view."
          ],
          "exclude_when": [
            "The route uses only a single raw patch without contextual aggregation.",
            "The route is text-only or embedding-only."
          ],
          "positive_route_refs": [
            "june_update_2026-06-10__rec_000148::route_015",
            "june_update_2026-06-10__rec_000148::route_017",
            "june_update_2026-06-10__rec_000152::route_004",
            "june_update_2026-06-10__rec_000152::route_005",
            "june_update_2026-06-10__rec_000152::route_006",
            "june_update_2026-06-10__rec_000152::route_007",
            "june_update_2026-06-10__rec_000152::route_008"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000060::route_002",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_002",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        }
      ],
      "code": "F4",
      "label": "Visual raster carriers",
      "short": "Visual",
      "color": "#c66a12",
      "route_count": 56,
      "model_count": 11,
      "subtypes": [
        {
          "leaf_id": "F4.L1",
          "name": "Raw slide or patch input",
          "definition": "A histology or similar biomedical image presented directly as a raster image, patch, or patch set to a vision encoder.",
          "include_when": [
            "The route starts from a raw image patch or whole-slide image.",
            "The visible carrier is an image tensor or patch grid."
          ],
          "exclude_when": [
            "The route is a projected embedding or text prompt.",
            "The route is a diffusion state or discrete biological token stream."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000060::route_002",
            "full_2026-07-06__rec_000060::route_003",
            "june_update_2026-06-10__rec_000148::route_001",
            "june_update_2026-06-10__rec_000148::route_003",
            "june_update_2026-06-10__rec_000148::route_005",
            "june_update_2026-06-10__rec_000148::route_006",
            "june_update_2026-06-10__rec_000148::route_007",
            "june_update_2026-06-10__rec_000148::route_008",
            "june_update_2026-06-10__rec_000148::route_009",
            "june_update_2026-06-10__rec_000148::route_010",
            "june_update_2026-06-10__rec_000148::route_011",
            "june_update_2026-06-10__rec_000148::route_013",
            "june_update_2026-06-10__rec_000148::route_016",
            "june_update_2026-06-10__rec_000148::route_018",
            "june_update_2026-06-10__rec_000152::route_001",
            "june_update_2026-06-10__rec_000152::route_002"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000827::route_008",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "raw_slide_or_patch_input",
          "route_count": 50,
          "model_count": 11,
          "example": {
            "input": "whole-slide image",
            "carrier": "224×224 RGB tissue patches",
            "model": "patch encoder → multimodal generator",
            "note": "Pixels or image patches are the model-facing carrier."
          }
        },
        {
          "leaf_id": "F4.L2",
          "name": "Patch-context or case-level visual reasoning",
          "definition": "A visual route where multiple patches or registered sections are combined into a case-level image context before prediction or reasoning.",
          "include_when": [
            "The route uses neighboring patches, serial sections, or case-level visual context.",
            "The visible carrier is still visual, but the unit is a patch set or contextualized slide view."
          ],
          "exclude_when": [
            "The route uses only a single raw patch without contextual aggregation.",
            "The route is text-only or embedding-only."
          ],
          "positive_route_refs": [
            "june_update_2026-06-10__rec_000148::route_015",
            "june_update_2026-06-10__rec_000148::route_017",
            "june_update_2026-06-10__rec_000152::route_004",
            "june_update_2026-06-10__rec_000152::route_005",
            "june_update_2026-06-10__rec_000152::route_006",
            "june_update_2026-06-10__rec_000152::route_007",
            "june_update_2026-06-10__rec_000152::route_008"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000060::route_002",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000950::route_002",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "patch_context_or_case_level_visual_reasoning",
          "route_count": 6,
          "model_count": 3,
          "example": {
            "input": "ROI + neighboring patches + case context",
            "carrier": "ordered visual token bank",
            "model": "context aggregator → multimodal LLM",
            "note": "Visual evidence is organized across regions or slides before reasoning."
          }
        }
      ]
    },
    {
      "family_id": "geometric_or_diffusion_state_carrier",
      "name": "Geometric and diffusion-state carriers",
      "definition": "Routes whose model-visible form is a geometric constraint, coordinate state, or noisy latent state used by a diffusion or structure-generation model. These routes are structurally distinct from ordinary embeddings because the carrier is organized by geometry or diffusion time.",
      "structural_criterion": "The model-visible carrier is a coordinate, constraint object, or time-indexed noisy latent state rather than a token sequence or patch grid.",
      "leaves": [
        {
          "leaf_id": "F5.L1",
          "name": "Noisy diffusion state",
          "definition": "A time-indexed noisy latent state or training state used directly in a diffusion process.",
          "include_when": [
            "The route exposes noisy coordinates or another diffusion-time latent state to the model.",
            "The carrier is part of forward or reverse diffusion dynamics."
          ],
          "exclude_when": [
            "The route is a static coordinate constraint or a text prompt.",
            "The route is an image, token stream, or dense embedding prefix."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000950::route_001",
            "full_2026-07-06__rec_000950::route_011"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000950::route_002",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000090::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        },
        {
          "leaf_id": "F5.L2",
          "name": "Coordinate, backbone, or shape conditioning",
          "definition": "A geometric conditioning route in which the model receives coordinates, backbones, substructures, or point-cloud-derived geometry as the visible carrier.",
          "include_when": [
            "The route conditions on backbone coordinates, substructures, or point clouds.",
            "The model-visible form is geometric rather than textual."
          ],
          "exclude_when": [
            "The route is an abstract symbolic label or a free-form text prompt.",
            "The route is a noisy diffusion state rather than a stable geometric condition."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000950::route_002",
            "full_2026-07-06__rec_000950::route_004",
            "full_2026-07-06__rec_000950::route_005",
            "full_2026-07-06__rec_000950::route_006",
            "full_2026-07-06__rec_000950::route_007"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000950::route_001",
            "full_2026-07-06__rec_000950::route_011",
            "full_2026-07-06__rec_000086::route_003",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        },
        {
          "leaf_id": "F5.L3",
          "name": "Symbolic structural constraints",
          "definition": "A non-textual or lightly symbolic structural condition such as symmetry, class labels, or other geometry-level control signals supplied to the generator.",
          "include_when": [
            "The route conditions generation with symbolic structural labels or abstract geometry constraints.",
            "The carrier is a constraint signal rather than a language prompt or a dense embedding."
          ],
          "exclude_when": [
            "The route is a free-form natural-language caption or question.",
            "The route is a coordinate/backbone carrier or noisy diffusion state."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000950::route_003",
            "full_2026-07-06__rec_000950::route_008"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000950::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "full_2026-07-06__rec_000086::route_010",
            "june_update_2026-06-10__rec_000121::route_003"
          ]
        }
      ],
      "code": "F5",
      "label": "Geometric and diffusion-state carriers",
      "short": "Geometric",
      "color": "#7450a8",
      "route_count": 19,
      "model_count": 5,
      "subtypes": [
        {
          "leaf_id": "F5.L1",
          "name": "Noisy diffusion state",
          "definition": "A time-indexed noisy latent state or training state used directly in a diffusion process.",
          "include_when": [
            "The route exposes noisy coordinates or another diffusion-time latent state to the model.",
            "The carrier is part of forward or reverse diffusion dynamics."
          ],
          "exclude_when": [
            "The route is a static coordinate constraint or a text prompt.",
            "The route is an image, token stream, or dense embedding prefix."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000950::route_001",
            "full_2026-07-06__rec_000950::route_011"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000950::route_002",
            "full_2026-07-06__rec_000086::route_002",
            "full_2026-07-06__rec_000090::route_001",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "noisy_diffusion_state",
          "route_count": 3,
          "model_count": 3,
          "example": {
            "input": "biological state x₀ + noise ε",
            "carrier": "xₜ = √αₜx₀ + √(1−αₜ)ε",
            "model": "conditioned denoiser / flow model",
            "note": "The generator consumes an evolving noisy state rather than a token stream."
          }
        },
        {
          "leaf_id": "F5.L2",
          "name": "Coordinate, backbone, or shape conditioning",
          "definition": "A geometric conditioning route in which the model receives coordinates, backbones, substructures, or point-cloud-derived geometry as the visible carrier.",
          "include_when": [
            "The route conditions on backbone coordinates, substructures, or point clouds.",
            "The model-visible form is geometric rather than textual."
          ],
          "exclude_when": [
            "The route is an abstract symbolic label or a free-form text prompt.",
            "The route is a noisy diffusion state rather than a stable geometric condition."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000950::route_002",
            "full_2026-07-06__rec_000950::route_004",
            "full_2026-07-06__rec_000950::route_005",
            "full_2026-07-06__rec_000950::route_006",
            "full_2026-07-06__rec_000950::route_007"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000950::route_001",
            "full_2026-07-06__rec_000950::route_011",
            "full_2026-07-06__rec_000086::route_003",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "coordinate_backbone_or_shape_conditioning",
          "route_count": 5,
          "model_count": 3,
          "example": {
            "input": "residue/atom coordinates (xᵢ,yᵢ,zᵢ)",
            "carrier": "equivariant geometric state",
            "model": "geometry-aware generator",
            "note": "Coordinates, backbones, or explicit shapes condition generation directly."
          }
        },
        {
          "leaf_id": "F5.L3",
          "name": "Symbolic structural constraints",
          "definition": "A non-textual or lightly symbolic structural condition such as symmetry, class labels, or other geometry-level control signals supplied to the generator.",
          "include_when": [
            "The route conditions generation with symbolic structural labels or abstract geometry constraints.",
            "The carrier is a constraint signal rather than a language prompt or a dense embedding."
          ],
          "exclude_when": [
            "The route is a free-form natural-language caption or question.",
            "The route is a coordinate/backbone carrier or noisy diffusion state."
          ],
          "positive_route_refs": [
            "full_2026-07-06__rec_000950::route_003",
            "full_2026-07-06__rec_000950::route_008"
          ],
          "counterexample_route_refs": [
            "full_2026-07-06__rec_000950::route_002",
            "full_2026-07-06__rec_000950::route_001",
            "full_2026-07-06__rec_000086::route_010",
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "subtype_id": "symbolic_structural_constraint",
          "route_count": 11,
          "model_count": 2,
          "example": {
            "input": "motif anchors + distance constraints",
            "carrier": "symbolic geometry/structure constraints",
            "model": "constraint-conditioned generator",
            "note": "Explicit structural rules constrain what the model may generate."
          }
        }
      ]
    }
  ],
  "architectures": [
    {
      "model_id": "model_a45de7f446e9",
      "model_name": "ALTER",
      "record_id": "full_2026-07-06__rec_001218",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_ea83dc2e6084",
      "paper_title": "Any-to-Any Learning in Computational Pathology via Triplet Multimodal Pretraining",
      "doi": "10.48550/arXiv.2505.12711",
      "paper_url": "https://doi.org/10.48550/arXiv.2505.12711",
      "route_count": 8,
      "configuration_count": 5,
      "family_counts": {
        "visual_raster_carrier": 5,
        "dense_continuous_carrier": 2,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "raw_slide_or_patch_input": 5,
        "pooled_or_aggregated_embedding": 2,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier",
        "visual_raster_carrier"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "pooled_or_aggregated_embedding",
        "raw_slide_or_patch_input"
      ],
      "primary_subtype": "raw_slide_or_patch_input",
      "modalities": [
        "histology/slide image",
        "omics",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation"
      ],
      "text_roles": [
        "biological_payload",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001218_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001218_da83e96a8915/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of our pretraining framework, ALTER . (a) ALTER processes each modality using modality-specific encoders, followed by a universal sequence Transformer and task-specific projection heads for downstream prediction. (b) The three-tiered constraints of ALTER, which can enable model to align multimodal inputs without requiring full modality pairing. (c) ALTER can be applied to any downstream task by integrating task-specific projection heads for fine-tuning.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel schematic of a multimodal biomedical AI architecture.\n\nVisible panels and labels:\n- Panel `(a) Overall Architecture`: shows three biological/clinical input modalities: `Whole Slide Images`, `Omics`, and `Pathology Reports`.\n- Each modality is passed through a corresponding tokenizer:\n  - `Slide Encoder`\n  - `Omics Tokenizer`\n  - `Report Tokenizer`\n- Tokenized outputs go into modality-specific components labeled:\n  - `Slide Sequence Projector`\n  - `Omics Sequence Projector`\n  - `Report Sequence Projector`\n- These feed into a central `Universal Sequence Transformer`, followed by `Task-specific Projection Heads`.\n\nPanel `(b) Stage 1: Pre-training Universal Sequence Transformer`:\n- Shows pretraining of the `Universal Sequence Transformer`.\n- Includes contrastive learning objectives labeled:\n  - `Inter-modal Contrast`\n  - `Inter-sample Contrast`\n  - `Intra-modal Contrast with Masked Language Modeling`\n- Inputs are labeled `WSI Features`, `Omics Features`, and `Text Features`.\n- A small example text sequence includes biomedical terms such as `adrenal`, `carcinoma`, `diagnosis`, `MASK`, and `cortical`.\n\nPanel `(c) Stage 2: Training Task Projection Heads with Downstream Tasks`:\n- Shows frozen or reused `Universal Sequence Transformer` feeding into `Task Specific Projection Heads`.\n- Downstream task examples are labeled:\n  - `Cancer Subtyping`\n  - `Survival Analysis`\n  - `Report Generation`\n\nBiological/clinical source objects:\n- Histopathology whole-slide images.\n- Omics data.\n- Pathology reports.\n\nModel interfaces and transformations:\n- Raw modality-specific biomedical inputs are tokenized or encoded.\n- Tokens/features are projected into sequence representations.\n- A universal sequence transformer integrates multimodal sequences.\n- Task-specific heads adapt the shared representation to downstream clinical tasks.\n\nVisible findings/claims:\n- The figure proposes a two-stage multimodal learning framework: pretraining a universal sequence transformer with contrastive and masked language modeling objectives, then training task-specific heads for cancer subtyping, survival analysis, and report generation.",
        "page_no": 4,
        "sha256": "1082c05fd353b8d9f16bb5a81f5d19d7729869a16478cd9438a685c9dc40dd10",
        "pixel_width": 795,
        "pixel_height": 380,
        "crop_box": {
          "x": 0.02,
          "y": 0.02,
          "width": 0.44,
          "height": 0.42
        },
        "panel_label": "(a) WSI pretraining route",
        "visible_input_object": "whole slide images",
        "visible_model_interface": "slide encoder -> slide sequence projector -> universal sequence transformer",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the top-left WSI pathway in panel (a), keeping the source slide, modality-specific encoding/projection, and the universal sequence transformer while excluding omics, reports, and downstream heads.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_79351a652029",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "diagnostic reports",
          "actual_model_visible_form": "report token sequence T with a [CLS] embedding"
        },
        {
          "subtype_id": "pooled_or_aggregated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_a525acddbe12",
          "example_input": "{gene/cell/patch embeddings}",
          "example_carrier": "mean/attention pool = one compact vector",
          "example_interface": "aggregator → generator",
          "actual_source": "genomic profiles",
          "actual_model_visible_form": "aggregated gene pathway features with a [CLS] token"
        },
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_814eb4ef1898",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "whole slide images",
          "actual_model_visible_form": "WSI feature bag with a [CLS] token"
        }
      ],
      "routes": [
        {
          "route_id": "route_814eb4ef1898",
          "configuration_id": "config_2d9f3edb813e",
          "route_label": "WSI pretraining route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "pretraining with any subset of three modalities-whole slide images, genomic profiles, and diagnostic reports",
          "source_object_verbatim": "whole slide images",
          "source_object_normalized": "whole-slide images",
          "source_modality_normalized": "histology/slide image",
          "transformation_chain_verbatim": [
            "decompose W into non-overlapping patches",
            "extract patch features with UNI",
            "2-layer TransMIL slide encoder",
            "region-wise aggregation",
            "concatenate with the [CLS] token",
            "modality-shared fusion",
            "modality-specific decoupling",
            "universal sequence Transformer"
          ],
          "model_visible_form_verbatim": "WSI feature bag with a [CLS] token",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "modality-specific encoders followed by a universal sequence Transformer",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "To address these challenges, we propose ALTER , the A ny-to-any L earning in compu T ational pathology via tripl E t multimodal p R etraining, a pretraining paradigm designed to develop FMs that flexibly integrate multimodal data across diverse scenarios. ALTER enables pretraining with any subset of three modalities-whole slide images, genomic profiles, and diagnostic reports-allowing the model to both accept arbitrary modality combinations and learn mutual cross-modal mappings for any downstream tasks. The key idea is to leverage weak supervision via contrastive learning and masked language modeling (MLM) to build robust intra-modal, inter-modal, and intra-sample constraints. These three-tiered constraints comprehensively encompass the interaction scope in multimodal fusion, supporting flexible modality combinations at both training and inference. To reduce computational overhead, ALTER incorporates modality-specific aggregation modules for WSIs and omics, enabling efficient cross-modal interaction beyond simple concatenation. Notably, ALTER is modular and generalizable: it can be extended to any multimodal scenario and is compatible with different backbone FMs. In this paper, we demonstrate the multimodal adaptability of ALTER by training an any-to-any model with 6,850 tri-modal pairs on 29 cancer sites from The Cancer Genome Atlas (TCGA) and conduct diverse downstream task validations on 10 public datasets.",
          "section_heading": "3.1 Problem Formulation",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/11",
            "#/texts/13",
            "#/texts/14",
            "#/texts/8",
            "#/texts/9"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001218::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001218::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a525acddbe12",
          "configuration_id": "config_2d9f3edb813e",
          "route_label": "Genomic pretraining route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "pretraining with any subset of three modalities-whole slide images, genomic profiles, and diagnostic reports",
          "source_object_verbatim": "genomic profiles",
          "source_object_normalized": "gene-expression profiles",
          "source_modality_normalized": "omics",
          "transformation_chain_verbatim": [
            "utilize term-frequency-analysis to discretize gene expression values",
            "pretrain a Performer as the gene encoder",
            "group genes into pathways using biological pathway databases",
            "pool each pathway to obtain aggregated gene features",
            "modality-shared fusion",
            "modality-specific decoupling",
            "universal sequence Transformer"
          ],
          "model_visible_form_verbatim": "aggregated gene pathway features with a [CLS] token",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "modality-specific encoders followed by a universal sequence Transformer",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "To address these challenges, we propose ALTER , the A ny-to-any L earning in compu T ational pathology via tripl E t multimodal p R etraining, a pretraining paradigm designed to develop FMs that flexibly integrate multimodal data across diverse scenarios. ALTER enables pretraining with any subset of three modalities-whole slide images, genomic profiles, and diagnostic reports-allowing the model to both accept arbitrary modality combinations and learn mutual cross-modal mappings for any downstream tasks. The key idea is to leverage weak supervision via contrastive learning and masked language modeling (MLM) to build robust intra-modal, inter-modal, and intra-sample constraints. These three-tiered constraints comprehensively encompass the interaction scope in multimodal fusion, supporting flexible modality combinations at both training and inference. To reduce computational overhead, ALTER incorporates modality-specific aggregation modules for WSIs and omics, enabling efficient cross-modal interaction beyond simple concatenation. Notably, ALTER is modular and generalizable: it can be extended to any multimodal scenario and is compatible with different backbone FMs. In this paper, we demonstrate the multimodal adaptability of ALTER by training an any-to-any model with 6,850 tri-modal pairs on 29 cancer sites from The Cancer Genome Atlas (TCGA) and conduct diverse downstream task validations on 10 public datasets.",
          "section_heading": "3.2 Modality Encoder",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper discretizes gene-expression values and then pools pathway features; the frozen taxonomy does not provide a dedicated omics-token subtype, so a pooled dense carrier is the closest fit.",
          "pages": [
            1,
            2
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/11",
            "#/texts/13",
            "#/texts/14",
            "#/texts/8",
            "#/texts/9"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001218::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001218::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_79351a652029",
          "configuration_id": "config_2d9f3edb813e",
          "route_label": "Report pretraining route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "pretraining with any subset of three modalities-whole slide images, genomic profiles, and diagnostic reports",
          "source_object_verbatim": "diagnostic reports",
          "source_object_normalized": "diagnostic reports",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenize R into a sequence T",
            "use BioBERT as text encoder",
            "modality-shared fusion",
            "modality-specific decoupling",
            "universal sequence Transformer"
          ],
          "model_visible_form_verbatim": "report token sequence T with a [CLS] embedding",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "modality-specific encoders followed by a universal sequence Transformer",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "To address these challenges, we propose ALTER , the A ny-to-any L earning in compu T ational pathology via tripl E t multimodal p R etraining, a pretraining paradigm designed to develop FMs that flexibly integrate multimodal data across diverse scenarios. ALTER enables pretraining with any subset of three modalities-whole slide images, genomic profiles, and diagnostic reports-allowing the model to both accept arbitrary modality combinations and learn mutual cross-modal mappings for any downstream tasks. The key idea is to leverage weak supervision via contrastive learning and masked language modeling (MLM) to build robust intra-modal, inter-modal, and intra-sample constraints. These three-tiered constraints comprehensively encompass the interaction scope in multimodal fusion, supporting flexible modality combinations at both training and inference. To reduce computational overhead, ALTER incorporates modality-specific aggregation modules for WSIs and omics, enabling efficient cross-modal interaction beyond simple concatenation. Notably, ALTER is modular and generalizable: it can be extended to any multimodal scenario and is compatible with different backbone FMs. In this paper, we demonstrate the multimodal adaptability of ALTER by training an any-to-any model with 6,850 tri-modal pairs on 29 cancer sites from The Cancer Genome Atlas (TCGA) and conduct diverse downstream task validations on 10 public datasets.",
          "section_heading": "B Data Processing",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The reports are ordinary clinical prose rather than a literal prompt or question, so the frozen text subtype is only an approximate match.",
          "pages": [
            1,
            2
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/11",
            "#/texts/13",
            "#/texts/14",
            "#/texts/8",
            "#/texts/9"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001218::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001218::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2997abc848de",
          "configuration_id": "config_7cb973758416",
          "route_label": "WSI survival-prediction route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "survival prediction (multimodality)",
          "source_object_verbatim": "whole slide images",
          "source_object_normalized": "whole-slide images",
          "source_modality_normalized": "histology/slide image",
          "transformation_chain_verbatim": [
            "decompose W into non-overlapping patches",
            "extract patch features with UNI",
            "2-layer TransMIL slide encoder",
            "region-wise aggregation",
            "concatenate with the [CLS] token",
            "task-specific heads appended to ALTER"
          ],
          "model_visible_form_verbatim": "patient-level embedding from WSI [CLS] tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "modality-specific encoders with task-specific heads appended to ALTER",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "each containing WSIs and gene expressions with corresponding survival outcome data",
          "section_heading": "4.1 Datasets and Tasks",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15
          ],
          "doc_item_refs": [
            "#/texts/200",
            "#/texts/201",
            "#/texts/202",
            "#/texts/203",
            "#/texts/204",
            "#/texts/205"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001218::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001218::0002",
            "dense::full_2026-07-06__rec_001218::0003",
            "dense::full_2026-07-06__rec_001218::0004",
            "dense::full_2026-07-06__rec_001218::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f1f43f8f60ae",
          "configuration_id": "config_7cb973758416",
          "route_label": "Genomic survival-prediction route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "survival prediction (multimodality)",
          "source_object_verbatim": "gene expressions",
          "source_object_normalized": "gene-expression profiles",
          "source_modality_normalized": "omics",
          "transformation_chain_verbatim": [
            "discretize gene expression values",
            "pretrain a Performer as the gene encoder",
            "group genes into pathways using biological pathway databases",
            "pool each pathway to obtain aggregated gene features",
            "task-specific heads appended to ALTER"
          ],
          "model_visible_form_verbatim": "patient-level embedding from gene pathway [CLS] tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "modality-specific encoders with task-specific heads appended to ALTER",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "each containing WSIs and gene expressions with corresponding survival outcome data",
          "section_heading": "4.1 Datasets and Tasks",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper discretizes gene-expression values and then pools pathways; the final visible carrier is best treated as a pooled dense representation.",
          "pages": [
            15
          ],
          "doc_item_refs": [
            "#/texts/200",
            "#/texts/201",
            "#/texts/202",
            "#/texts/203",
            "#/texts/204",
            "#/texts/205"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001218::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001218::0002",
            "dense::full_2026-07-06__rec_001218::0003",
            "dense::full_2026-07-06__rec_001218::0004",
            "dense::full_2026-07-06__rec_001218::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_06651029e746",
          "configuration_id": "config_1d924205fa17",
          "route_label": "WSI cancer-subtyping route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "cancer subtyping ( h. → h. )",
          "source_object_verbatim": "whole slide images",
          "source_object_normalized": "whole-slide images",
          "source_modality_normalized": "histology/slide image",
          "transformation_chain_verbatim": [
            "decompose W into non-overlapping patches",
            "extract patch features with UNI",
            "2-layer TransMIL slide encoder",
            "region-wise aggregation",
            "concatenate with the [CLS] token",
            "frozen fusion layers",
            "task-specific heads"
          ],
          "model_visible_form_verbatim": "slide-level embedding from WSI [CLS] tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "frozen fusion layers with task-specific heads",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Cancer subtyping ( h. → h. ).",
          "section_heading": "4.1 Datasets and Tasks",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            9,
            15,
            16
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/tables/1",
            "#/tables/2",
            "#/texts/100",
            "#/texts/102",
            "#/texts/207",
            "#/texts/209",
            "#/texts/210",
            "#/texts/211",
            "#/texts/213",
            "#/texts/217",
            "#/texts/79",
            "#/texts/80",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84",
            "#/texts/98",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001218::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001218::0006",
            "dense::full_2026-07-06__rec_001218::0007",
            "dense::full_2026-07-06__rec_001218::0008",
            "dense::full_2026-07-06__rec_001218::0009"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3255c1091034",
          "configuration_id": "config_58effcb8cce2",
          "route_label": "WSI gene-mutation route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "gene mutation prediction ( h. → g. )",
          "source_object_verbatim": "whole slide images",
          "source_object_normalized": "whole-slide images",
          "source_modality_normalized": "histology/slide image",
          "transformation_chain_verbatim": [
            "decompose W into non-overlapping patches",
            "extract patch features with UNI",
            "2-layer TransMIL slide encoder",
            "region-wise aggregation",
            "concatenate with the [CLS] token",
            "frozen fusion layers",
            "task-specific heads"
          ],
          "model_visible_form_verbatim": "slide-level embedding from WSI [CLS] tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "frozen fusion layers with task-specific heads",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "gene mutation prediction ( h. → g. ).",
          "section_heading": "4.1 Datasets and Tasks",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            16
          ],
          "doc_item_refs": [
            "#/tables/2",
            "#/texts/219",
            "#/texts/79",
            "#/texts/80",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001218::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001218::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_37eae69f2672",
          "configuration_id": "config_53ea83ea3c04",
          "route_label": "WSI report-generation route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "report generation ( h. → t. )",
          "source_object_verbatim": "whole slide images",
          "source_object_normalized": "whole-slide images",
          "source_modality_normalized": "histology/slide image",
          "transformation_chain_verbatim": [
            "decompose W into non-overlapping patches",
            "extract patch features with UNI",
            "2-layer TransMIL slide encoder",
            "region-wise aggregation",
            "concatenate with the [CLS] token",
            "frozen fusion layers",
            "task-specific heads"
          ],
          "model_visible_form_verbatim": "slide-level embedding from WSI [CLS] tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "frozen fusion layers with task-specific heads",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Report generation ( h. → t. ).",
          "section_heading": "4.1 Datasets and Tasks",
          "supporting_figure_or_table": "Table 4",
          "evidence_status": "explicit_text",
          "uncertainty": "The main text does not spell out the exact decoder head, but the WSI input route for report generation is explicit.",
          "pages": [
            7,
            8,
            16
          ],
          "doc_item_refs": [
            "#/tables/2",
            "#/texts/79",
            "#/texts/80",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84",
            "#/texts/95"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001218::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001218::0011"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_22149166d2ab"
    },
    {
      "model_id": "model_699b7d92d927",
      "model_name": "Bio-BLIP",
      "record_id": "june_update_2026-06-10__rec_000194",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_44ce36825624",
      "paper_title": "Bio-BLIP: A Multimodal Architecture for Transferable Reasoning in Genomic Variant Interpretation",
      "doi": "10.64898/2026.05.12.724740",
      "paper_url": "https://doi.org/10.64898/2026.05.12.724740",
      "route_count": 6,
      "configuration_count": 2,
      "family_counts": {
        "dense_continuous_carrier": 6
      },
      "subtype_counts": {
        "virtual_token_prefix": 3,
        "connector_mediated_embedding": 3
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "connector_mediated_embedding",
        "virtual_token_prefix"
      ],
      "primary_subtype": "virtual_token_prefix",
      "modalities": [
        "DNA sequence",
        "gene neighborhood",
        "protein sequence"
      ],
      "lifecycle_phases": [
        "pretraining"
      ],
      "fusion_topologies": [
        "cross_attention",
        "prefix"
      ],
      "text_roles": [
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/june_update_2026_06_10_rec_000194_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/june_update_2026_06_10_rec_000194_b24922d872f2/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: (a) Embeddings (E) from Biological Foundation Models (BioFMs) are input to Bio-BLIP to provide multiple modalities of information, here from AlphaGenome , GenePT , and ESM-2 . (b) Modality-specific Q-formers align diverse biological embeddings with captions of genomic variants, which contain functional and locational information. The Q-formers output query tokens encoding relevant information from each modality. (c) The Master Qformer integrates modality-specific tokens through cross-attention, producing a fixed-length visual prefix (d) The LLM, its weights frozen , takes as input the multimodal prefix and a prompt. During pretraining, the LLM loss backpropagates only to train the Q-formers. Pretraining and evaluation is conducted on variant annotation; all weights are frozen and the model is then applied to downstream evaluation tasks.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic architecture diagram titled “BioBLIP Architecture.”\n\nVisible structure and labels:\n- Panel/region labels: A, B, C, D in red.\n- Left side shows multiple biological embedding sources:\n  - “AlphaGenome Ref/Alt Embeddings” feeding into E1.\n  - “GenePT Embeddings” feeding into E2.\n  - “ESM Protein Embeddings” feeding into En.\n- Each embedding source is shown as a blue triangular input block labeled E1, E2, or En.\n\nBiological source objects:\n- Chromosome or DNA sequence context from AlphaGenome reference/alternate embeddings.\n- Gene representations from GenePT embeddings.\n- Protein representations from ESM protein embeddings.\n\nTransformations:\n- Each embedding source feeds into a black “Qformer” module:\n  - “DNA Qformer”\n  - “Gene Qformer”\n  - “Protein Qformer”\n- These produce modality-specific query tokens:\n  - Orange “DNA Specific Query Tokens”\n  - Red “Gene Specific Query Tokens”\n  - Green “Protein Specific Query Tokens”\n- The query tokens are combined into a pink “Master Qformer.”\n- The Master Qformer outputs a “Fixed Length Variant Prefix,” shown as a sequence of colored tokens.\n\nModel interfaces:\n- The fixed-length variant prefix is passed into a blue “Frozen LLM.”\n- The Frozen LLM receives an accompanying “Text Prompt.”\n- The prompt instructs the model to describe a variant as a JSON object with fields such as chromosome, user chromosome location, gene, consequence, nearest genes, and output only JSON.\n- The output is a “Generated Variant Annotation” shown as JSON-like text.\n\nTasks/findings represented:\n- The bottom evaluation block contains:\n  - “Pretraining and Eval” with “Variant Annotation.”\n  - “Downstream Evaluation Tasks” including “Target Gene Identification” and “Regulatory Variant Prioritization.”\n- The figure presents a multimodal biological embedding-to-language-model architecture for variant annotation and downstream genomic variant interpretation tasks.",
        "page_no": 2,
        "sha256": "001b6901e0e8d2244eab92f22f9f429c912407f2def511bc173547ca501aa099",
        "pixel_width": 786,
        "pixel_height": 382,
        "crop_box": {
          "x": 0.0,
          "y": 0.08,
          "width": 0.735,
          "height": 0.56
        },
        "panel_label": "A/C",
        "visible_input_object": "AlphaGenome ref/alt embeddings for a genomic variant",
        "visible_model_interface": "DNA Qformer -> DNA-specific query tokens -> Master Qformer -> fixed-length variant prefix",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the left-to-center pretraining path with the source embeddings, DNA Qformer, query-token output, Master Qformer, and fixed-length prefix. It excludes the frozen LLM, generated annotation, and downstream-task panels while preserving the arrows and labels needed to read one grounded input route.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "connector_mediated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_58874dee1f0f",
          "example_input": "image / omics encoder states",
          "example_carrier": "Q-Former or adapter query vectors",
          "example_interface": "connector → LLM cross-modal interface",
          "actual_source": "Pre-computed AlphaGenome embeddings for the reference and alternate alleles of a genomic variant",
          "actual_model_visible_form": "modality-specific query tokens"
        },
        {
          "subtype_id": "virtual_token_prefix",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_2828f4dfe3f3",
          "example_input": "cell embedding + prompt",
          "example_carrier": "<bio₁> <bio₂> ... <bioₖ> [prompt tokens]",
          "example_interface": "soft prefix → LLM stream",
          "actual_source": "Pre-computed AlphaGenome embeddings for the reference and alternate alleles of a genomic variant",
          "actual_model_visible_form": "fixed-length multimodal prefix"
        }
      ],
      "routes": [
        {
          "route_id": "route_2828f4dfe3f3",
          "configuration_id": "config_0817cabcec36",
          "route_label": "AlphaGenome embeddings to Bio-BLIP",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "generative variant annotation",
          "source_object_verbatim": "Pre-computed AlphaGenome embeddings for the reference and alternate alleles of a genomic variant",
          "source_object_normalized": "AlphaGenome reference and alternate allele embeddings",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "pre-computed and stored embeddings",
            "DNA Q-former",
            "Master Q-former",
            "fixed-length multimodal prefix",
            "frozen LLM"
          ],
          "model_visible_form_verbatim": "fixed-length multimodal prefix",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "virtual_token_prefix",
          "insertion_or_fusion_verbatim": "Master Q-former attached to a frozen LLM",
          "fusion_topology": "prefix",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Pre-computed AlphaGenome [4] embeddings for the reference and alternate alleles of a genomic variant",
          "section_heading": "2.2 Pretraining Regime",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/121",
            "#/texts/122",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/127",
            "#/texts/128"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000194::route_001"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000194::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e09961882576",
          "configuration_id": "config_0817cabcec36",
          "route_label": "GenePT embeddings to Bio-BLIP",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "generative variant annotation",
          "source_object_verbatim": "GenePT embeddings for the up to 15 protein-coding genes nearest to the variant",
          "source_object_normalized": "nearest protein-coding genes",
          "source_modality_normalized": "gene neighborhood",
          "transformation_chain_verbatim": [
            "pre-computed and stored embeddings",
            "Gene Q-former",
            "Master Q-former",
            "fixed-length multimodal prefix",
            "frozen LLM"
          ],
          "model_visible_form_verbatim": "fixed-length multimodal prefix",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "virtual_token_prefix",
          "insertion_or_fusion_verbatim": "Master Q-former attached to a frozen LLM",
          "fusion_topology": "prefix",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "GenePT [20] embeddings for the up to 15 protein-coding genes nearest to the variant.",
          "section_heading": "2.2 Pretraining Regime",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/121",
            "#/texts/122",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/127",
            "#/texts/128"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000194::route_002"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000194::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fc3883ca9463",
          "configuration_id": "config_0817cabcec36",
          "route_label": "ESM-2 embeddings to Bio-BLIP",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "generative variant annotation",
          "source_object_verbatim": "ESM-2 embeddings for the reference and alternate aminoacid sequences of the nearest protein-coding gene to the genomic variant",
          "source_object_normalized": "nearest protein-coding gene aminoacid sequences",
          "source_modality_normalized": "protein sequence",
          "transformation_chain_verbatim": [
            "pre-computed and stored embeddings",
            "Protein Q-former",
            "Master Q-former",
            "fixed-length multimodal prefix",
            "frozen LLM"
          ],
          "model_visible_form_verbatim": "fixed-length multimodal prefix",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "virtual_token_prefix",
          "insertion_or_fusion_verbatim": "Master Q-former attached to a frozen LLM",
          "fusion_topology": "prefix",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "ESM-2 [21] embeddings for the reference and alternate aminoacid sequences of the nearest protein-coding gene to the genomic variant.",
          "section_heading": "2.2 Pretraining Regime",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/121",
            "#/texts/122",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/127",
            "#/texts/128"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000194::route_003"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000194::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_58874dee1f0f",
          "configuration_id": "config_4e7c6fdf224a",
          "route_label": "AlphaGenome embeddings to Bio-BLIP Stage 1 DNA Q-former",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "modality-specific text alignment and information extraction",
          "source_object_verbatim": "Pre-computed AlphaGenome embeddings for the reference and alternate alleles of a genomic variant",
          "source_object_normalized": "AlphaGenome reference and alternate allele embeddings",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "pre-computed and stored embeddings",
            "DNA Q-former",
            "modality-specific query tokens"
          ],
          "model_visible_form_verbatim": "modality-specific query tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "DNA Q-former",
          "fusion_topology": "cross_attention",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Each embedding type receives its own modality-specific Q-former.",
          "section_heading": "2.2 Pretraining Regime",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3
          ],
          "doc_item_refs": [
            "#/texts/104"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000194::route_009"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fdca8e3fdb99",
          "configuration_id": "config_4e7c6fdf224a",
          "route_label": "GenePT embeddings to Bio-BLIP Stage 1 Gene Q-former",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "modality-specific text alignment and information extraction",
          "source_object_verbatim": "GenePT embeddings for the up to 15 protein-coding genes nearest to the variant",
          "source_object_normalized": "nearest protein-coding genes",
          "source_modality_normalized": "gene neighborhood",
          "transformation_chain_verbatim": [
            "pre-computed and stored embeddings",
            "Gene Q-former",
            "modality-specific query tokens"
          ],
          "model_visible_form_verbatim": "modality-specific query tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "Gene Q-former",
          "fusion_topology": "cross_attention",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "The GenePT Q-former and ESM Q-former are trained with 8 query vectors each.",
          "section_heading": "2.2 Pretraining Regime",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3,
            4,
            5,
            6,
            11
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/104",
            "#/texts/108",
            "#/texts/109",
            "#/texts/110",
            "#/texts/111",
            "#/texts/112",
            "#/texts/114",
            "#/texts/17"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000194::route_010"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000194::0050"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5ab21f81834c",
          "configuration_id": "config_4e7c6fdf224a",
          "route_label": "ESM-2 embeddings to Bio-BLIP Stage 1 Protein Q-former",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "modality-specific text alignment and information extraction",
          "source_object_verbatim": "ESM-2 embeddings for the reference and alternate aminoacid sequences of the nearest protein-coding gene",
          "source_object_normalized": "nearest protein-coding gene aminoacid sequences",
          "source_modality_normalized": "protein sequence",
          "transformation_chain_verbatim": [
            "pre-computed and stored embeddings",
            "Protein Q-former",
            "modality-specific query tokens"
          ],
          "model_visible_form_verbatim": "modality-specific query tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "Protein Q-former",
          "fusion_topology": "cross_attention",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Each embedding type receives its own modality-specific Q-former.",
          "section_heading": "2.2 Pretraining Regime",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3
          ],
          "doc_item_refs": [
            "#/texts/104"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000194::route_011"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_ee721887da1b"
    },
    {
      "model_id": "model_5421186d4cbe",
      "model_name": "BioGPT",
      "record_id": "full_2026-07-06__rec_001519",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_ff0e5e9d25b9",
      "paper_title": "BioGPT: A Generative Transformer-Based Framework for Personalized Genomic Medicine and Rare Disease Diagnosis",
      "doi": "10.58496/mjaih/2025/015",
      "paper_url": "https://doi.org/10.58496/mjaih/2025/015",
      "route_count": 2,
      "configuration_count": 2,
      "family_counts": {
        "discrete_biological_symbol_stream": 1,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "native_biological_token_stream": 1,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream",
        "plain_language_prompt_or_question"
      ],
      "primary_subtype": "plain_language_prompt_or_question",
      "modalities": [
        "DNA",
        "text"
      ],
      "lifecycle_phases": [
        "unclear"
      ],
      "fusion_topologies": [
        "cross_attention"
      ],
      "text_roles": [
        "biological_payload"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001519_figure_007.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001519_57bf8abbd7af/figure_007.png",
        "figure_index": 7,
        "caption": "Fig. 1. BioGPT System Architecture for Multi-Modal Rare Disease Diagnosis.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow diagram for a genomics/deep learning model pipeline.\n\nVisible elements:\n- Left side input labeled “Genomic Sequences,” showing example nucleotide tokens: `A C C G A G A G G T`.\n- A “Tokenization and Embedding” section showing sequence fragments converted into colored token blocks.\n- An “Embedding” section with labels `k` and `mer`, indicating k-mer style embeddings.\n- An “Embediction BPE” label near the bottom, suggesting BPE-based embedding/tokenization.\n- Central model block:\n  - Top labels “Pretraining” and “Fine-tuning” feeding into “BioGPT.”\n  - A large “MULTI-MODAL FUSION” module containing repeated “Transformer Encoder” blocks and one “Transformer Decoder” block.\n  - Vertical side labels “REMBEDDING” on the left and “GENERATIVE” on the right.\n  - A “Generative Decoding” block below the fusion module.\n  - Output token blocks leading to a bottom label “Generative decoding.”\n- Right side outputs:\n  - “Attention Visualization” shown as a blue heatmap grid.\n  - “Rare Disease Diagnosis” with symptom boxes labeled “Symptom 1,” “Symptom 2,” and “Symptom 3,” feeding into a final box labeled “Rare Disease.”\n\nBiological source objects:\n- Genomic nucleotide sequences.\n\nTransformations/model interfaces:\n- Genomic sequences are tokenized and embedded, including k-mer/BPE-like representations.\n- Embeddings enter a multimodal transformer-based fusion system connected to BioGPT.\n- The model supports pretraining, fine-tuning, generative decoding, and produces interpretability/diagnostic outputs.\n\nFindings/output represented:\n- Attention visualization heatmap.\n- Rare disease diagnosis inferred from symptoms.",
        "page_no": 3,
        "sha256": "e8e264781a469663cb2e0a8962be8316dc705962055223a71f2185cec5ffdfff",
        "pixel_width": 592,
        "pixel_height": 351,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.66,
          "height": 0.75
        },
        "panel_label": "genomic input to multimodal fusion",
        "visible_input_object": "Genomic sequences tokenized into overlapping k-mers",
        "visible_model_interface": "BioGPT and MULTI-MODAL FUSION with REMBEDDING and the incoming arrow into the fusion block",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the source sequence label, tokenization/embedding steps, the k-mer/BPE carrier labels on the left, and the arrow into the multimodal fusion module, while excluding the right-side output-only diagnosis/attention panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_88e8d2744e24",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "Genomic sequences",
          "actual_model_visible_form": "overlapping k-mer token stream"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_424e87dd1e00",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "Clinical narratives",
          "actual_model_visible_form": "BPE token stream"
        }
      ],
      "routes": [
        {
          "route_id": "route_88e8d2744e24",
          "configuration_id": "config_68986844968e",
          "route_label": "genomic sequence stream",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "Genomic sequences, represented as overlapping k-mers",
          "source_object_verbatim": "Genomic sequences",
          "source_object_normalized": "genomic sequences",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "tokenized into overlapping k-mers",
            "independently embedded",
            "fused within a shared latent space via cross-attention"
          ],
          "model_visible_form_verbatim": "overlapping k-mer token stream",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "cross-attention within the multi-modal transformer fusion module",
          "fusion_topology": "cross_attention",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Genomic sequences, represented as overlapping k-mers",
          "section_heading": "3.1 System Overview",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes the same input stream in both pretraining and fine-tuning contexts, so a single lifecycle phase is not uniquely separable.",
          "pages": [
            1,
            3
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/pictures/2",
            "#/pictures/3",
            "#/pictures/6",
            "#/texts/1",
            "#/texts/2",
            "#/texts/53",
            "#/texts/54",
            "#/texts/55",
            "#/texts/56",
            "#/texts/57",
            "#/texts/58",
            "#/texts/6",
            "#/texts/7",
            "#/texts/8"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001519::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001519::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_424e87dd1e00",
          "configuration_id": "config_b346192e560b",
          "route_label": "clinical narrative stream",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "Clinical narratives, comprising unstructured patient data such as symptoms, lab results, and case histories",
          "source_object_verbatim": "Clinical narratives",
          "source_object_normalized": "clinical narratives",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "normalized",
            "encoded using Byte Pair Encoding (BPE)",
            "independently embedded",
            "fused within a shared latent space via cross-attention"
          ],
          "model_visible_form_verbatim": "BPE token stream",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "cross-attention within the multi-modal transformer fusion module",
          "fusion_topology": "cross_attention",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Clinical narratives, comprising unstructured patient data such as symptoms, lab results, and case histories.",
          "section_heading": "3.1 System Overview",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes the same input stream in both pretraining and fine-tuning contexts, so a single lifecycle phase is not uniquely separable.",
          "pages": [
            1,
            3
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/pictures/2",
            "#/pictures/3",
            "#/pictures/6",
            "#/texts/1",
            "#/texts/2",
            "#/texts/53",
            "#/texts/54",
            "#/texts/55",
            "#/texts/56",
            "#/texts/57",
            "#/texts/58",
            "#/texts/6",
            "#/texts/7",
            "#/texts/8"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001519::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001519::0001"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_fdbd3ba264a9"
    },
    {
      "model_id": "model_77f5da416a80",
      "model_name": "BioMedGPT-10B",
      "record_id": "full_2026-07-06__rec_003214",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_89ea5bd2179f",
      "paper_title": "Biomedgpt: Open multimodal generative pre-trained transformer for biomedicine",
      "doi": "",
      "paper_url": "",
      "route_count": 7,
      "configuration_count": 7,
      "family_counts": {
        "dense_continuous_carrier": 4,
        "text_native_token_stream": 3
      },
      "subtype_counts": {
        "direct_projected_embedding": 4,
        "structured_biological_prompt_or_task_scaffold": 3
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "protein/peptide",
        "small molecule",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning"
      ],
      "fusion_topologies": [
        "placeholder_replacement",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003214_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003214_ccaac5236ad7/figure_002.png",
        "figure_index": 2,
        "caption": "Figure 2: The overview of BioMedGPT-10B. BioMedGPT-LM is the large language model of BioMedGPT, which serves as a cognitive core to jointly comprehend various biological modalities through natural language. In BioMedGPT-10B, the parameter size of the large language model is 7B. BioMedGPT-10B adopts GraphMVP [Liu et al., 2022] as the 2D molecular graph encoder, ESM2-3B [Lin et al., 2022] as the protein sequence encoder, and conducts feature space alignment via a neural network adaptor. BioMedGPT can be applied to many multimodal downstream tasks such as biomedical QA, molecule QA, and protein QA.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel schematic describing BioMedGPT model training and downstream biomedical QA.\n\nVisible panels and labels:\n- Top panel: “BioMedGPT-LM: Incremental Training on Biomedical Corpus”\n  - Shows “Meta AI LLaMA 2” on the left.\n  - Arrow points to a “Biomedical Corpus” box containing source logos/text for arXiv, PubMed Central, and WIPO with document thumbnails.\n  - Arrow points to the output model labeled “BioMedGPT-LM”.\n\n- Bottom-left panel: “BioMedGPT-10B: Multi-Modal Alignment between Biomedical and Natural Language”\n  - Shows BioMedGPT-LM as the central language model interface.\n  - Two input pipelines align biomedical modalities with text:\n    - Molecule pathway: molecule structure image enters a “Molecule Encoder”, then a neural/network block, then token-like fields labeled “Instruct”, “Mol”, and “Question”.\n    - Protein pathway: protein structure image enters a “Protein Encoder”, then a neural/network block, then fields labeled “Instruct”, “Prot”, and “Question”.\n  - Both pathways feed upward into BioMedGPT-LM.\n  - Side labels indicate “Instruct & Question”.\n\n- Right panel: “Downstream Tasks”\n  - “BioMedical QA”: instruction-style prompt about a biomedical judgment question and an answer box showing “Yes.”\n  - “Molecule QA”: molecule structure image plus prompt containing molecule markup tags such as `<molecule>...</molecule>`, with an answer box describing the molecule as a dicarboxylic acid monoamide.\n  - “Protein QA”: protein structure image plus prompt containing protein markup tags such as `<protein>...</protein>`, with an answer box describing a bifunctional serine/threonine kinase and phosphorylase-related function.\n\nBiological/source objects:\n- Biomedical text corpora from arXiv, PubMed Central, and WIPO.\n- Molecular structure diagrams.\n- Protein structure renderings.\n\nTransformations/model interfaces:\n- LLaMA 2 is incrementally trained on biomedical corpora to create BioMedGPT-LM.\n- Molecule and protein encoders transform structural biomedical inputs into aligned representations for BioMedGPT-LM.\n- Text prompts combine instructions, modality tokens, and questions to support downstream QA.\n\nFindings/claims shown:\n- The figure presents BioMedGPT-LM and BioMedGPT-10B as models for biomedical language modeling, multimodal molecule/protein alignment, and biomedical, molecule, and protein question answering.",
        "page_no": 4,
        "sha256": "8fac451c1ee9a846f85c420a7aa96d139a3554d6a6c9be9b4575f664eae510cf",
        "pixel_width": 915,
        "pixel_height": 673,
        "crop_box": {
          "x": 0.015,
          "y": 0.3,
          "width": 0.455,
          "height": 0.65
        },
        "panel_label": "BioMedGPT-10B: Multi-Modal Alignment between Biomedical and Natural Language",
        "visible_input_object": "Molecule structure diagram feeding the Molecule Encoder and the Instruct / Mol / Question scaffold into BioMedGPT-LM",
        "visible_model_interface": "BioMedGPT-LM with the upward molecule alignment path and the token-style prompt interface labeled Instruct, Mol, and Question",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop isolates the left multimodal-alignment subpanel and keeps the molecule source object, transformation path through the Molecule Encoder, and the immediate insertion interface into BioMedGPT-LM. It excludes the output-only downstream QA examples and the protein branch.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_907b394108c6",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "molecules from PubChemQA",
          "actual_model_visible_form": "<molecule><moleculeHere></molecule> {text_input}"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_8ab7d83662c7",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "PubMedQA training set of biomedical multiple-choice QA pairs",
          "actual_model_visible_form": "text question-answer pairs"
        }
      ],
      "routes": [
        {
          "route_id": "route_907b394108c6",
          "configuration_id": "config_805ccede2f3b",
          "route_label": "molecule-text multimodal fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "answering questions with regard to a given molecule",
          "source_object_verbatim": "molecules from PubChemQA",
          "source_object_normalized": "molecule",
          "source_modality_normalized": "small molecule",
          "transformation_chain_verbatim": [
            "remove molecules that cannot be processed by RDKit",
            "generate 2D molecular graphs",
            "encode with GraphMVP",
            "project with an independent modality adaptor"
          ],
          "model_visible_form_verbatim": "<molecule><moleculeHere></molecule> {text_input}",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "### Human: <molecule><moleculeHere></molecule> {text_input}. ### Assistant: {text_output}",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "To build the connections between molecular and protein structures with natural language, we perform multimodal fine-tuning, which involves answering questions with regard to a given molecule or protein. As shown in Table 1, we design prompt templates to help BioMedGPT-LM understand the context more accurately in a role-play manner. The <moleculeHere> and <proteinHere> symbolics represent the aligned molecular and protein features, where each atom of a molecule and each residue of a protein is considered as a token. {text_input} is populated by questions in the training set, and {text_output} is populated by answers to calculate the auto-regressive loss.",
          "section_heading": "3.2 Multimodal alignment between molecules, proteins, and natural language",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5,
            6
          ],
          "doc_item_refs": [
            "#/texts/136",
            "#/texts/137",
            "#/texts/138",
            "#/texts/140",
            "#/texts/141",
            "#/texts/142",
            "#/texts/143",
            "#/texts/144"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003214::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8dba88613ac4",
          "configuration_id": "config_9f390f3de96c",
          "route_label": "protein-text multimodal fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "answering questions with regard to a given protein",
          "source_object_verbatim": "proteins from UniProtQA",
          "source_object_normalized": "protein",
          "source_modality_normalized": "protein/peptide",
          "transformation_chain_verbatim": [
            "encode protein sequences with ESM-2-3B",
            "project with an independent modality adaptor",
            "align to the feature space of BioMedGPT-LM-7B"
          ],
          "model_visible_form_verbatim": "<protein><proteinHere></protein> {text_input}",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "### Human: <protein><proteinHere></protein> {text_input}. ### Assistant: {text_output}",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "To build the connections between molecular and protein structures with natural language, we perform multimodal fine-tuning, which involves answering questions with regard to a given molecule or protein. As shown in Table 1, we design prompt templates to help BioMedGPT-LM understand the context more accurately in a role-play manner. The <moleculeHere> and <proteinHere> symbolics represent the aligned molecular and protein features, where each atom of a molecule and each residue of a protein is considered as a token. {text_input} is populated by questions in the training set, and {text_output} is populated by answers to calculate the auto-regressive loss.",
          "section_heading": "3.2 Multimodal alignment between molecules, proteins, and natural language",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5,
            6
          ],
          "doc_item_refs": [
            "#/texts/136",
            "#/texts/137",
            "#/texts/138",
            "#/texts/140",
            "#/texts/141",
            "#/texts/142",
            "#/texts/143"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003214::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8ab7d83662c7",
          "configuration_id": "config_d235dfa27fba",
          "route_label": "PubMedQA biomedical QA fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "training set of PubMedQA",
          "source_object_verbatim": "PubMedQA training set of biomedical multiple-choice QA pairs",
          "source_object_normalized": "PubMedQA training questions and answers",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "fine-tune the language model of BioMedGPT-10B on the training sets of PubMedQA and MedMCQA"
          ],
          "model_visible_form_verbatim": "text question-answer pairs",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "language-model fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "We fine-tune the language model of BioMedGPT-10B on the training sets of PubMedQA and MedMCQA",
          "section_heading": "4.1 Biomedical QA",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper states PubMedQA and MedMCQA together; this route is split to satisfy the one-source-object-per-route constraint.",
          "pages": [
            6,
            7
          ],
          "doc_item_refs": [
            "#/texts/149",
            "#/texts/150",
            "#/texts/151",
            "#/texts/152",
            "#/texts/154",
            "#/texts/155",
            "#/texts/156",
            "#/texts/157"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003214::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1266d867c22a",
          "configuration_id": "config_7c579f0ae115",
          "route_label": "MedMCQA biomedical QA fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "training set of MedMCQA",
          "source_object_verbatim": "MedMCQA training set of biomedical multiple-choice QA pairs",
          "source_object_normalized": "MedMCQA training questions and answers",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "fine-tune the language model of BioMedGPT-10B on the training sets of PubMedQA and MedMCQA"
          ],
          "model_visible_form_verbatim": "text question-answer pairs",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "language-model fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "We fine-tune the language model of BioMedGPT-10B on the training sets of PubMedQA and MedMCQA",
          "section_heading": "4.1 Biomedical QA",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper states PubMedQA and MedMCQA together; this route is split to satisfy the one-source-object-per-route constraint.",
          "pages": [
            6,
            7
          ],
          "doc_item_refs": [
            "#/texts/149",
            "#/texts/150",
            "#/texts/151",
            "#/texts/152",
            "#/texts/154",
            "#/texts/155",
            "#/texts/156",
            "#/texts/157"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003214::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_296252896033",
          "configuration_id": "config_93d6111bd0d4",
          "route_label": "biomedical QA out-of-domain evaluation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "out-of-domain (OOD) evaluation on USMLE without additional fine-tuning on its training set",
          "source_object_verbatim": "USMLE multiple-choice questions",
          "source_object_normalized": "USMLE multiple-choice questions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "evaluate on USMLE without additional fine-tuning"
          ],
          "model_visible_form_verbatim": "text multiple-choice questions",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "language-model evaluation",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "out-of-domain (OOD) evaluation on USMLE without additional fine-tuning on its training set.",
          "section_heading": "4.1 Biomedical QA",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            6,
            7
          ],
          "doc_item_refs": [
            "#/texts/149",
            "#/texts/150",
            "#/texts/151",
            "#/texts/152",
            "#/texts/154",
            "#/texts/155",
            "#/texts/156",
            "#/texts/157"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003214::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8e681fc01a87",
          "configuration_id": "config_4ee24c8d744b",
          "route_label": "molecule QA inference",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "generating a text response given a specific molecule and a text query over its properties",
          "source_object_verbatim": "a specific molecule",
          "source_object_normalized": "molecule",
          "source_modality_normalized": "small molecule",
          "transformation_chain_verbatim": [
            "encode molecule structure",
            "align to natural language",
            "generate text response"
          ],
          "model_visible_form_verbatim": "<molecule><moleculeHere></molecule> plus text query",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "prompt-based multimodal alignment",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Molecule QA involves generating a text response given a specific molecule and a text query over its properties.",
          "section_heading": "4.2 Molecule QA",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            7,
            8
          ],
          "doc_item_refs": [
            "#/tables/2",
            "#/texts/161",
            "#/texts/162",
            "#/texts/164",
            "#/texts/165",
            "#/texts/166",
            "#/texts/167"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003214::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_59d034a752b8",
          "configuration_id": "config_6011f840729d",
          "route_label": "protein QA inference",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "generating a text response to a query about a given protein",
          "source_object_verbatim": "a given protein",
          "source_object_normalized": "protein",
          "source_modality_normalized": "protein/peptide",
          "transformation_chain_verbatim": [
            "encode protein sequence",
            "align to natural language",
            "generate text response"
          ],
          "model_visible_form_verbatim": "<protein><proteinHere></protein> plus text query",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "prompt-based multimodal alignment",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Protein QA involves generating a text response to a query about a given protein",
          "section_heading": "4.3 Protein QA",
          "supporting_figure_or_table": "Table 4",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            8
          ],
          "doc_item_refs": [
            "#/tables/3",
            "#/texts/169",
            "#/texts/170",
            "#/texts/171",
            "#/texts/172",
            "#/texts/173"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003214::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_c17be9263882"
    },
    {
      "model_id": "model_fba98a35871c",
      "model_name": "BioMedGPT-LM-7B",
      "record_id": "full_2026-07-06__rec_003214",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_89ea5bd2179f",
      "paper_title": "Biomedgpt: Open multimodal generative pre-trained transformer for biomedicine",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "pretraining"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003214_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003214_ccaac5236ad7/figure_002.png",
        "figure_index": 2,
        "caption": "Figure 2: The overview of BioMedGPT-10B. BioMedGPT-LM is the large language model of BioMedGPT, which serves as a cognitive core to jointly comprehend various biological modalities through natural language. In BioMedGPT-10B, the parameter size of the large language model is 7B. BioMedGPT-10B adopts GraphMVP [Liu et al., 2022] as the 2D molecular graph encoder, ESM2-3B [Lin et al., 2022] as the protein sequence encoder, and conducts feature space alignment via a neural network adaptor. BioMedGPT can be applied to many multimodal downstream tasks such as biomedical QA, molecule QA, and protein QA.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel schematic describing BioMedGPT model training and downstream biomedical QA.\n\nVisible panels and labels:\n- Top panel: “BioMedGPT-LM: Incremental Training on Biomedical Corpus”\n  - Shows “Meta AI LLaMA 2” on the left.\n  - Arrow points to a “Biomedical Corpus” box containing source logos/text for arXiv, PubMed Central, and WIPO with document thumbnails.\n  - Arrow points to the output model labeled “BioMedGPT-LM”.\n\n- Bottom-left panel: “BioMedGPT-10B: Multi-Modal Alignment between Biomedical and Natural Language”\n  - Shows BioMedGPT-LM as the central language model interface.\n  - Two input pipelines align biomedical modalities with text:\n    - Molecule pathway: molecule structure image enters a “Molecule Encoder”, then a neural/network block, then token-like fields labeled “Instruct”, “Mol”, and “Question”.\n    - Protein pathway: protein structure image enters a “Protein Encoder”, then a neural/network block, then fields labeled “Instruct”, “Prot”, and “Question”.\n  - Both pathways feed upward into BioMedGPT-LM.\n  - Side labels indicate “Instruct & Question”.\n\n- Right panel: “Downstream Tasks”\n  - “BioMedical QA”: instruction-style prompt about a biomedical judgment question and an answer box showing “Yes.”\n  - “Molecule QA”: molecule structure image plus prompt containing molecule markup tags such as `<molecule>...</molecule>`, with an answer box describing the molecule as a dicarboxylic acid monoamide.\n  - “Protein QA”: protein structure image plus prompt containing protein markup tags such as `<protein>...</protein>`, with an answer box describing a bifunctional serine/threonine kinase and phosphorylase-related function.\n\nBiological/source objects:\n- Biomedical text corpora from arXiv, PubMed Central, and WIPO.\n- Molecular structure diagrams.\n- Protein structure renderings.\n\nTransformations/model interfaces:\n- LLaMA 2 is incrementally trained on biomedical corpora to create BioMedGPT-LM.\n- Molecule and protein encoders transform structural biomedical inputs into aligned representations for BioMedGPT-LM.\n- Text prompts combine instructions, modality tokens, and questions to support downstream QA.\n\nFindings/claims shown:\n- The figure presents BioMedGPT-LM and BioMedGPT-10B as models for biomedical language modeling, multimodal molecule/protein alignment, and biomedical, molecule, and protein question answering.",
        "page_no": 4,
        "sha256": "8fac451c1ee9a846f85c420a7aa96d139a3554d6a6c9be9b4575f664eae510cf",
        "pixel_width": 915,
        "pixel_height": 673,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 1,
          "height": 0.293
        },
        "panel_label": "BioMedGPT-LM: Incremental Training on Biomedical Corpus",
        "visible_input_object": "Biomedical corpus from arXiv, PubMed Central, and WIPO document sources",
        "visible_model_interface": "BioMedGPT-LM output after incremental training from LLaMA 2 on the biomedical corpus",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This top panel is the smallest coherent crop that still shows the grounded input route: source corpus logos/documents, the transformation arrow, and the BioMedGPT-LM target model. It excludes the downstream task panels and keeps the labels and arrows readable.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_8d63adfae7ac",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "biomedical documents from S2ORC",
          "actual_model_visible_form": "sentence-based chunks tokenized by the Llama2 tokenizer"
        }
      ],
      "routes": [
        {
          "route_id": "route_8d63adfae7ac",
          "configuration_id": "config_da25008dc492",
          "route_label": "incremental biomedical text training",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "large-scale biomedical literature on top of Llama2-Chat-7B",
          "source_object_verbatim": "biomedical documents from S2ORC",
          "source_object_normalized": "biomedical documents",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "remove author information",
            "remove reference citations",
            "remove chart data",
            "partition into sentence-based chunks",
            "tokenize with the Llama2 tokenizer"
          ],
          "model_visible_form_verbatim": "sentence-based chunks tokenized by the Llama2 tokenizer",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "incremental training on top of Llama2-Chat-7B",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "We perform incremental training on Llama2-Chat-7B with extensive biomedical documents from S2ORC.",
          "section_heading": "3.1 BioMedGPT-LM-7B: Incremental training on large-scale biomedical literature",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/131",
            "#/texts/132",
            "#/texts/133",
            "#/texts/134"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003214::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_0cc72bd82901"
    },
    {
      "model_id": "model_0e0c28b60ee7",
      "model_name": "BIOREASON",
      "record_id": "full_2026-07-06__rec_001074",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_08046f30aead",
      "paper_title": "BIOREASON: Incentivizing Multimodal Biological Reasoning within a DNA-LLM Model",
      "doi": "10.48550/arXiv.2505.23579",
      "paper_url": "https://doi.org/10.48550/arXiv.2505.23579",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question"
      ],
      "primary_subtype": "plain_language_prompt_or_question",
      "modalities": [
        "DNA"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "other_explicit"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001074_figure_003.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001074_c72d500e989d/figure_003.png",
        "figure_index": 3,
        "caption": "Figure 3: Case Study of BIOREASON's Output",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel schematic/comparison figure with three labeled columns:\n\n- **Left panel: “Question”**\n  - A prompt-style box showing a KEGG data point question.\n  - Includes labels such as **“KEGG Data Point”**, **“Question”**, **“Network Definition of the Pathway”**, and **“Genes in the Pathway.”**\n  - Biological source objects mentioned: **chromosome 17**, **Actin(monomenic) // PFN1 // Actin(filamentous)**, **ACTB**, **ACTG1**, and **PFN1 / profilin 1**.\n  - The question asks for the **biological effect of the PFN1 allele**, specifically what disease it contributes to.\n\n- **Middle panel: “Ground Truth (KEGG)”**\n  - A green box describing reasoning steps from KEGG ground truth.\n  - States that the variant **KEGG_800** represents a **C>G substitution at position 4945969 on chromosome 17**, occurring in the **PFN1 gene**.\n  - Describes the nucleotide change as potentially affecting a **functional protein domain**.\n  - Links PFN1 mutation to disruption of the **actin cytoskeleton**, **axonal transport defects**, **motor neuron degeneration**, and ultimately **Amyotrophic Lateral Sclerosis (ALS)**.\n\n- **Right panel: “BioReason’s Output”**\n  - A model-output style box containing generated reasoning with `<think>` tags and an answer.\n  - It states that the **C>G mutation in PFN1** likely disrupts **profilin-1 function**, impairing actin dynamics.\n  - It connects this to **neuronal cytoskeletal dysfunction**, **motor neuron degeneration**, and **ALS**.\n  - Final answer shown: **amyotrophic lateral sclerosis (ALS)**.\n\nOverall finding: the figure compares a KEGG-derived ground-truth reasoning chain with BioReason’s generated reasoning for a PFN1 mutation, both concluding that the mutation contributes to **amyotrophic lateral sclerosis (ALS)**.",
        "page_no": 9,
        "sha256": "90d544b99e4eadf4440cc465687622ac1249e0938261d06ac09f815ac55f3550",
        "pixel_width": 795,
        "pixel_height": 207,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 0.345,
          "height": 1
        },
        "panel_label": "Question",
        "visible_input_object": "KEGG data point prompt about the PFN1 allele on chromosome 17 in the Actin(monomeric) // PFN1* // Actin(filamentous) pathway",
        "visible_model_interface": "Text-native prompt/question shown in the left panel, with the KEGG data point label and gene/pathway context",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the grounded input route: the question panel with the PFN1 allele, chromosome 17, and pathway context. It excludes the model output and ground-truth explanation panels, which are not needed to understand the input carrier.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_0bb6ee8832aa",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "PFN1 allele on chromosome 17",
          "actual_model_visible_form": "a PFN1 allele on chromosome 17 within the pathway Actin(monomeric) // PFN1* // Actin(filamentous)"
        }
      ],
      "routes": [
        {
          "route_id": "route_0bb6ee8832aa",
          "configuration_id": "config_bee4b50ef6b8",
          "route_label": "BIOREASON PFN1 case study question route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Case Study of BIOREASON's Output",
          "source_object_verbatim": "PFN1 allele on chromosome 17",
          "source_object_normalized": "PFN1 allele",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "a PFN1 allele on chromosome 17 within the pathway Actin(monomeric) // PFN1* // Actin(filamentous)",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "analysis of a PFN1 allele on chromosome 17 within the pathway Actin(monomeric) // PFN1* // Actin(filamentous)",
          "fusion_topology": "other_explicit",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "consider its analysis of a PFN1 allele on chromosome 17 within the pathway Actin(monomeric) // PFN1* // Actin(filamentous).",
          "section_heading": "5.5 Case Study",
          "supporting_figure_or_table": "Figure 3",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            9
          ],
          "doc_item_refs": [
            "#/texts/111"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001074::0028"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_357f1bf88b0b"
    },
    {
      "model_id": "model_9be63ad52f90",
      "model_name": "BIOVERSE",
      "record_id": "full_2026-07-06__rec_002049",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_f456cf1340ce",
      "paper_title": "BIOVERSE: Representation Alignment of Biomedical Modalities to LLMs for Multi-Modal Reasoning",
      "doi": "10.48550/arXiv.2510.01428",
      "paper_url": "https://doi.org/10.48550/arXiv.2510.01428",
      "route_count": 9,
      "configuration_count": 7,
      "family_counts": {
        "dense_continuous_carrier": 9
      },
      "subtype_counts": {
        "direct_projected_embedding": 9
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "RNA",
        "protein/peptide",
        "small molecule"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning"
      ],
      "fusion_topologies": [
        "placeholder_replacement",
        "shared_latent_alignment"
      ],
      "text_roles": [
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_002049_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_002049_b1c397ddeaa1/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: BIOVERSE base architecture: a modality-specific BioFM encodes a biological entity, and its output embeddings are mapped by a projection layer into the LLM's embedding space via special tokens (e.g. [BIO] ). In the alignment stage, only the projection layer P θ is trainable, while the encoder f b and the LLM g remain frozen. In the subsequent instruction-tuning stage, we allow both P θ and the low-rank adapter (LORA) within the LLM to be trainable. Stage 1 (S1) can be trained using autoregressive (AR) or contrastive (CT) loss, while stage 2 (S2) is always AR.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow/model architecture for integrating scRNA-seq data with a language model.\n\nVisible content:\n- Left side input blocks:\n  - `scRNA-seq Data`, shown as a gene expression heatmap/table.\n  - Labels indicate `Row: Cell`, `Column: Gene`, `Color: Expression value`.\n  - Annotation: `1+ token per cell`.\n  - `Text (Prompt)` example: “What is the most likely cell type?” with `7 tokens`.\n  - `Text (Label)` example: “CD8 T Cell” with `3 tokens`.\n\nBiological source object:\n- Single-cell RNA-seq expression data representing cells and genes.\n- Example target biological label: `CD8 T Cell`.\n\nTransformations and model components:\n- scRNA-seq data passes through a `BioEncoder`.\n- A pooling step labeled `CLS (or X-pooling)` produces `last_hidden_state`, annotated as `1x768` and `1024x768`.\n- This feeds a `2-layer MLP`, producing a `[BIO]` token embedding annotated as `1x2048`.\n- Text prompt and label pass through an `LLM Tokenizer` and `LLM Embedding Layer`.\n- Prompt token embeddings are annotated `7x2048`.\n- Label token embeddings are annotated `3x2048`.\n- A projection component labeled `Bio -> Language Projection` maps biological representations into the LLM embedding space.\n- The `[BIO]` token is injected into the prompt as soft tokens.\n- The combined inputs go through an `LLM Forward Pass`.\n- Output is `Predict label tokens`.\n\nTraining/interface labels:\n- `LoRA` is shown connected to the LLM forward pass.\n- Blue path labeled `S1 (with AR): update projection layer only`.\n- Another label: `S2: update LoRA & projection (optional)`.\n- Yellow box: `AR: Auto-regressive training using cross-entropy loss`.\n- Yellow box near bottom: `CT: training using contrastive loss`.\n- Dashed orange paths indicate `[BIO]` injection and contrastive training connections.\n\nOverall finding/purpose:\n- The figure describes a multimodal training architecture where scRNA-seq-derived cell embeddings are projected into a language-model token space, injected into prompts as biological soft tokens, and used to predict cell-type label tokens, with optional LoRA adaptation and either autoregressive or contrastive training objectives.",
        "page_no": 3,
        "sha256": "67805aaf59d42f2868da3d9ae3581360a1f36e2d1ffef246f7355f81ada6ad24",
        "pixel_width": 711,
        "pixel_height": 310,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.75,
          "height": 0.71
        },
        "panel_label": "scRNA-seq to [BIO] prompt injection",
        "visible_input_object": "scRNA-seq expression profile with text prompt",
        "visible_model_interface": "BioEncoder -> 2-layer MLP projection -> [BIO] soft-token injection into prompt embeddings",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the source scRNA-seq panel, the BioEncoder, the projection stage, and the [BIO] insertion into the prompt, which are enough to read one grounded input route. It excludes the output-only prediction side and most of the lower label/output panel.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_d58aa4c7d1dd",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "scRNA-seq profile",
          "actual_model_visible_form": "projected bio embeddings as [BIO] soft tokens in the LLM input space"
        }
      ],
      "routes": [
        {
          "route_id": "route_d58aa4c7d1dd",
          "configuration_id": "config_a1e1d340bbb5",
          "route_label": "scRNA-seq profile to projected BIO soft-token prompt",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "AR",
          "source_object_verbatim": "scRNA-seq profile",
          "source_object_normalized": "single-cell RNA-seq profile",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "BioEncoder",
            "2-layer MLP projection",
            "[BIO] soft-token injection"
          ],
          "model_visible_form_verbatim": "projected bio embeddings as [BIO] soft tokens in the LLM input space",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "injected at a placeholder (e.g., [BIO]) within the query q",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "High-throughput assays such as scRNA-seq, proteomics, and small-molecules profiling generate rich, high-dimensional data that are critical for biomedical discovery. Biomedical foundation models (BioFMs; also referred to as BMFMs) trained on those inputs, e.g., scGPT (Cui et al., 2024) for single-cell RNA sequencing (scRNA-seq), ESM-2 (Lin et al., 2023) for proteins, Molformer (Ross et al., 2022) for small molecules, capture expressive representations but lack instruction-following and open-ended reasoning. In contrast, general-purpose large language models (LLMs) excel at language interaction and can nominally ingest sequences like proteins or Simplified Molecular Input Line Entry System (SMILES) strings, but tokenization yields short, uninformative fragments and they cannot parse modalities such as scRNA-seq, where a cell's gene expression vector cannot be represented meaningfully as a token sequence. Bridging these strengths requires a framework that preserves modality-specific encoders, aligns their embeddings with the LLM token space, and enables reasoning across them.",
          "section_heading": "3.3 MODULAR ARCHITECTURE",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            4
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/11",
            "#/texts/12",
            "#/texts/15",
            "#/texts/16",
            "#/texts/17",
            "#/texts/36",
            "#/texts/37",
            "#/texts/38",
            "#/texts/39",
            "#/texts/40",
            "#/texts/41",
            "#/texts/42",
            "#/texts/43",
            "#/texts/9"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002049::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002049::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4224ba560f81",
          "configuration_id": "config_d625f1596463",
          "route_label": "scRNA-seq profile to contrastive text alignment",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "CT",
          "source_object_verbatim": "scRNA-seq profile",
          "source_object_normalized": "single-cell RNA-seq profile",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "BioEncoder",
            "projection layer",
            "contrastive alignment with paired text embeddings"
          ],
          "model_visible_form_verbatim": "projected bio embeddings aligned with frozen text embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "bidirectional InfoNCE alignment",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "In the base training mode illustrated in Figure 1, all data is processed through LLM's forward pass, and a standard autoregressive cross-entropy loss (with teacher forcing) is used to guide learning. Alternatively, to avoid the costly forward pass required by a large LLM, and to enable alignment with efficient encoder models that are often co-trained with the decoder for retrieval, we evaluate an alternative alignment mode. In this variant, we use contrastive learning to directly align the bio embeddings with their paired text embeddings. From here onward, we denote the first alignment strategy as AR (autoregressive) and the second as CT (contrastive).",
          "section_heading": "3.4 ALIGNMENT OBJECTIVES",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3
          ],
          "doc_item_refs": [
            "#/texts/30",
            "#/texts/31",
            "#/texts/32"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002049::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b7fbc161d721",
          "configuration_id": "config_4a75fa6fd0c0",
          "route_label": "PBMC10K scRNA-seq profile to generative cell-type annotation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "zero-shot generative cell type annotation",
          "source_object_verbatim": "PBMC10K scRNA-seq profile",
          "source_object_normalized": "single-cell RNA-seq profile",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "BioEncoder",
            "projection layer",
            "[BIO] token",
            "gene-list evidence in prompt",
            "prompt-level constraints"
          ],
          "model_visible_form_verbatim": "projected bio embeddings as [BIO] soft tokens in the LLM input space plus gene-list evidence in the prompt",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "soft token injection into the prompt",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "As shown in Table 1, although majority voting achieves relatively high accuracy, it fails on minority classes, leading to poor macroF 1 . Prior BioFMs such as scGPT (Cui et al., 2024) and alignmentbased models like LangCell (Zhao et al., 2024) and scMMGPT (Shi et al., 2025), when performing cell type annotation under zero-shot setting, fundamentally operate in a candidate-space matching paradigm. These models project cells and a predefined set of candidate labels and their descriptions into a shared embedding space and assign the nearest match. LangCell achieves the highest scores, reflecting the relative ease of candidate-space matching. By contrast, generative models operates in a generative regime: the LLM must produce a natural language label rather than selecting the nearest candidate. In our setup, we apply prompt-level constraints, instructing the model to select only from a predefined option set without decoding-level enforcement. The model nevertheless engages in True Label: CD14+ Monocytes Predicted Label: Based on the sorted expressed genes, the most likely immune cell subtype is CD14+ Monocytes. The presence of genes such as TYROBP (DAP12), FCER1G (FcgRI), ITGB2 (CD29), and ITGAM (CD11b) suggests a monocytic lineage. [...skip] The absence of B cellspecific genes and T cell receptor genes (TR genes) further supports this conclusion.",
          "section_heading": "5.2.1 ZERO-SHOT GENERATIVE CELL TYPE ANNOTATION",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            7,
            8
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/11",
            "#/texts/12",
            "#/texts/15",
            "#/texts/16",
            "#/texts/17",
            "#/texts/84",
            "#/texts/85",
            "#/texts/88",
            "#/texts/9"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002049::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002049::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c61f8f6a5c1e",
          "configuration_id": "config_e0a1d4043f6b",
          "route_label": "protein sequence to projected BIO soft-token prompt",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "protein-text pairs from UniProtKB",
          "source_object_verbatim": "protein sequence",
          "source_object_normalized": "protein sequence",
          "source_modality_normalized": "protein/peptide",
          "transformation_chain_verbatim": [
            "BioFM encoder",
            "pooling",
            "projection layer",
            "[BIO] token injected into the prompt as soft tokens"
          ],
          "model_visible_form_verbatim": "projected bio embeddings as [BIO] soft tokens in the LLM input space",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "injected at a placeholder (e.g., [BIO]) within the query q",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "For alignment, we construct paired biological entities ( x b ) and textual descriptions ( t b ) across three modalities. Protein: We obtain protein-text pairs from UniProtKB, where each amino acid sequence is linked to curated Gene Ontology (GO) terms representing its functional annotations across the three GO namespaces: Biological Process, Molecular Function, and Cellular Component. GO term metadata is derived from the official GO ontology, and annotations are obtained from UniProt crossreferences. To ensure reliability, we retain only experimentally supported GO annotations, yielding high-quality supervision for aligning BioFM protein embeddings with language representations. Small Molecule: For small molecules, we leverage LLASmol (Yu et al., 2024), which provides SMILES-text pairs with chemically grounded descriptions. Specifically, we select two datasets from the LLASmol collection for BioFM alignment: SMILES-to-IUPAC conversion and molecule captioning. In both cases, each molecule is represented as a SMILES string paired with naturallanguage annotations of structure, properties, or activities. scRNA-seq: For single-cell data, we adopt CellWhisperer (Schaefer et al., 2024), which aligns scRNA-seq profiles with cell-type and tissue-level textual metadata. Following the dataset protocol, we use the CellxGene subset (Perkel, 2024), where pseudo-bulk RNA samples are generated by averaging single-cell profiles, and natural-language descriptions are produced from cell and tissue metadata using large language models. This enables alignment between transcriptomic embeddings and ontological descriptions.",
          "section_heading": "4.2.1 ALIGNMENT",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            6
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/11",
            "#/texts/12",
            "#/texts/15",
            "#/texts/16",
            "#/texts/17",
            "#/texts/70",
            "#/texts/9"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002049::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002049::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8b4b2de14039",
          "configuration_id": "config_e0a1d4043f6b",
          "route_label": "protein sequence to contrastive text alignment",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "protein-text pairs from UniProtKB",
          "source_object_verbatim": "protein sequence",
          "source_object_normalized": "protein sequence",
          "source_modality_normalized": "protein/peptide",
          "transformation_chain_verbatim": [
            "BioFM encoder",
            "projection layer",
            "contrastive alignment with paired text embeddings"
          ],
          "model_visible_form_verbatim": "projected bio embeddings aligned with frozen text embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "bidirectional InfoNCE alignment",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "In the base training mode illustrated in Figure 1, all data is processed through LLM's forward pass, and a standard autoregressive cross-entropy loss (with teacher forcing) is used to guide learning. Alternatively, to avoid the costly forward pass required by a large LLM, and to enable alignment with efficient encoder models that are often co-trained with the decoder for retrieval, we evaluate an alternative alignment mode. In this variant, we use contrastive learning to directly align the bio embeddings with their paired text embeddings. From here onward, we denote the first alignment strategy as AR (autoregressive) and the second as CT (contrastive).",
          "section_heading": "3.4 ALIGNMENT OBJECTIVES",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3
          ],
          "doc_item_refs": [
            "#/texts/30",
            "#/texts/31",
            "#/texts/32"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002049::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9d7c941595fc",
          "configuration_id": "config_eca65a3e71b5",
          "route_label": "protein sequence to free-text protein description",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "protein text generation tasks",
          "source_object_verbatim": "protein sequence",
          "source_object_normalized": "protein sequence",
          "source_modality_normalized": "protein/peptide",
          "transformation_chain_verbatim": [
            "BioFM encoder",
            "projection layer",
            "[BIO] token",
            "prompt"
          ],
          "model_visible_form_verbatim": "projected bio embeddings injected into the LLM input space as soft tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "soft token injection into the prompt",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We evaluate all four protein-oriented text generation benchmarks from Mol-Instructions (Fang et al., 2023): (1) catalytic activity prediction, (2) domain/motif prediction, (3) functional description generation, and (4) protein function prediction. Each task provides a protein sequence as input, and the model must generate free-text outputs describing a specific property of that sequence. Together, these tasks probe both factual grounding (e.g., motif recognition) and open-ended description ability, testing whether the model can jointly reason over the protein sequence and the accompanying prompt.",
          "section_heading": "5.2.3 PROTEIN-ORIENTED TEXT GENERATION",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            9
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/11",
            "#/texts/12",
            "#/texts/15",
            "#/texts/16",
            "#/texts/17",
            "#/texts/9",
            "#/texts/98",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002049::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002049::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c4cd401ae46b",
          "configuration_id": "config_71ede05b6f9f",
          "route_label": "SMILES string to projected BIO soft-token prompt",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "SMILES-text pairs from LLASmol",
          "source_object_verbatim": "SMILES string",
          "source_object_normalized": "SMILES string",
          "source_modality_normalized": "small molecule",
          "transformation_chain_verbatim": [
            "BioFM encoder",
            "pooling",
            "projection layer",
            "[BIO] token injected into the prompt as soft tokens"
          ],
          "model_visible_form_verbatim": "projected bio embeddings as [BIO] soft tokens in the LLM input space",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "injected at a placeholder (e.g., [BIO]) within the query q",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "For alignment, we construct paired biological entities ( x b ) and textual descriptions ( t b ) across three modalities. Protein: We obtain protein-text pairs from UniProtKB, where each amino acid sequence is linked to curated Gene Ontology (GO) terms representing its functional annotations across the three GO namespaces: Biological Process, Molecular Function, and Cellular Component. GO term metadata is derived from the official GO ontology, and annotations are obtained from UniProt crossreferences. To ensure reliability, we retain only experimentally supported GO annotations, yielding high-quality supervision for aligning BioFM protein embeddings with language representations. Small Molecule: For small molecules, we leverage LLASmol (Yu et al., 2024), which provides SMILES-text pairs with chemically grounded descriptions. Specifically, we select two datasets from the LLASmol collection for BioFM alignment: SMILES-to-IUPAC conversion and molecule captioning. In both cases, each molecule is represented as a SMILES string paired with naturallanguage annotations of structure, properties, or activities. scRNA-seq: For single-cell data, we adopt CellWhisperer (Schaefer et al., 2024), which aligns scRNA-seq profiles with cell-type and tissue-level textual metadata. Following the dataset protocol, we use the CellxGene subset (Perkel, 2024), where pseudo-bulk RNA samples are generated by averaging single-cell profiles, and natural-language descriptions are produced from cell and tissue metadata using large language models. This enables alignment between transcriptomic embeddings and ontological descriptions.",
          "section_heading": "4.2.1 ALIGNMENT",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            6
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/11",
            "#/texts/12",
            "#/texts/15",
            "#/texts/16",
            "#/texts/17",
            "#/texts/70",
            "#/texts/9"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002049::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002049::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_595dcf1b86f3",
          "configuration_id": "config_71ede05b6f9f",
          "route_label": "SMILES string to contrastive text alignment",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "SMILES-text pairs from LLASmol",
          "source_object_verbatim": "SMILES string",
          "source_object_normalized": "SMILES string",
          "source_modality_normalized": "small molecule",
          "transformation_chain_verbatim": [
            "BioFM encoder",
            "projection layer",
            "contrastive alignment with paired text embeddings"
          ],
          "model_visible_form_verbatim": "projected bio embeddings aligned with frozen text embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "bidirectional InfoNCE alignment",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "In the base training mode illustrated in Figure 1, all data is processed through LLM's forward pass, and a standard autoregressive cross-entropy loss (with teacher forcing) is used to guide learning. Alternatively, to avoid the costly forward pass required by a large LLM, and to enable alignment with efficient encoder models that are often co-trained with the decoder for retrieval, we evaluate an alternative alignment mode. In this variant, we use contrastive learning to directly align the bio embeddings with their paired text embeddings. From here onward, we denote the first alignment strategy as AR (autoregressive) and the second as CT (contrastive).",
          "section_heading": "3.4 ALIGNMENT OBJECTIVES",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3
          ],
          "doc_item_refs": [
            "#/texts/30",
            "#/texts/31",
            "#/texts/32"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002049::route_008"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fad8a3478784",
          "configuration_id": "config_178dc3728fb8",
          "route_label": "SMILES string to molecular description",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "molecular description generation",
          "source_object_verbatim": "SMILES representation",
          "source_object_normalized": "SMILES string",
          "source_modality_normalized": "small molecule",
          "transformation_chain_verbatim": [
            "BioFM encoder",
            "projection layer",
            "[BIO] token",
            "prompt"
          ],
          "model_visible_form_verbatim": "projected bio embeddings injected into the LLM input space as soft tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "soft token injection into the prompt",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "SMILES representation",
          "section_heading": "5.2.2 MOLECULE DESCRIPTION GENERATION",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            8,
            9
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/11",
            "#/texts/12",
            "#/texts/15",
            "#/texts/16",
            "#/texts/17",
            "#/texts/9",
            "#/texts/93",
            "#/texts/96"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002049::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002049::0001"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_000c4dd8661a"
    },
    {
      "model_id": "model_79e042109d33",
      "model_name": "C2S-Scale",
      "record_id": "full_2026-07-06__rec_000090",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_c2885aa81441",
      "paper_title": "SCALING LARGE LANGUAGE MODELS FOR NEXT-GENERATION SINGLE-CELL ANALYSIS",
      "doi": "10.1101/2025.04.14.648850",
      "paper_url": "https://doi.org/10.1101/2025.04.14.648850",
      "route_count": 14,
      "configuration_count": 14,
      "family_counts": {
        "text_native_token_stream": 14
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 8,
        "structured_biological_prompt_or_task_scaffold": 6
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "graph/network",
        "spatial transcriptomics",
        "text",
        "transcriptomics"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "shared_latent_alignment",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_000090_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_000090_0d9defcc25c1/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: C2S-Scale unifies transcriptomics and natural language for scalable single-cell analysis. A , The C2S-Scale corpus integrates over 50 million single-cell transcriptomes curated from the CELLxGENE and Human Cell Atlas repositories, linking transcriptomic profiles with rich cell, tissue, and donor-level metadata as well as scientific abstracts and papers. B , C2S-Scale expands upon the Cell2Sentence framework by increasing model capacity up to 27 billion parameters, building a multi-task training corpus of over 1 trillion tokens, and expanding task diversity to include multimodal and multicellular inputs. C , Transcriptomic profiles are transformed into 'cell sentences', ordered sequences of gene names ranked by expression level. This enables the direct application of pretrained LLMs without requiring architectural modifications. D , C2S-Scale is initialized from pre-trained LLM checkpoints. An initial training phase on cell sentences is followed by instruction fine-tuning for a variety of downstream tasks. E , This unified training regime enables broad downstream capabilities, ranging from standard cell annotation to complex tasks like perturbation prediction and biological question answering.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific schematic describing a cell/language foundation model pipeline.\n\nPanel A, “DATA SOURCES AND ANNOTATIONS,” shows five input/source categories with icons and scale labels:\n- Single-cell: “800+ datasets”\n- Cell clusters: “170M+ single-cell & bulk samples”\n- Tissue sections: “100M+ annotations”\n- Paper abstracts: “21M+ perturbations”\n- Donor metadata: “1B+ tokens”\n\nPanel B, “SCALING DIMENSIONS,” shows four boxed scaling factors:\n- Model Capacity: line plot with “Scaling law” and network icons\n- Dataset Size: “170M+ cells from over 800 datasets from CellxGene and Human Cell Atlas”\n- Multimodality: transcriptomics, perturbations, biological papers, donor metadata\n- Context Diversity: unified text interface for single-cell, multi-cell, and natural language input\n\nPanel C, “CELL SENTENCE REPRESENTATION,” shows a gene expression matrix with rows Cell 1-4 and columns genes A-F, transformed into textual “Cell Sentences,” e.g. gene-token sequences like “GeneE GeneB GeneD ...”.\n\nPanel D, “MODEL TRAINING PROCEDURE,” shows a three-stage training workflow:\n- Stage 1: Natural language pre-training using paper/document inputs into a Base LLM labeled “Pythia, Gemma-2”\n- Stage 2: Continued training on cell sentences, producing “C2S-Scale”\n- Stage 3: Instruction fine-tuning for downstream tasks, producing “Fine-tuned C2S-Scale”\nIt shows checkpoints between the base language model and cell-to-sentence model.\n\nPanel E, “DOWNSTREAM TASKS,” lists applications with example prompts/outputs:\n- Cell Type Annotation\n- Cell Generation\n- Perturbation Response Prediction\n- Cluster Captioning\n- Dataset Interpretation\n- Question Answering\n\nBiological source objects visible include single cells, cell clusters, tissue sections, gene expression matrices, perturbation/drug icons, transcriptomics/DNA icon, donor metadata, and biological paper abstracts. The central transformation is converting gene expression profiles into ordered gene-token “cell sentences” for language-model training and downstream biological prediction or interpretation tasks.",
        "page_no": 29,
        "sha256": "dbee1b0a516ede93b6c40755aa370db7565b72588132c8a55fbce99d12196816",
        "pixel_width": 917,
        "pixel_height": 986,
        "crop_box": {
          "x": 0,
          "y": 0.47,
          "width": 0.59,
          "height": 0.23
        },
        "panel_label": "C. CELL SENTENCE REPRESENTATION",
        "visible_input_object": "Gene expression matrix / single-cell transcriptomic profiles",
        "visible_model_interface": "Cell sentences made of ordered gene-name token sequences",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop isolates panel C, which shows the grounded source object (gene expression matrix), the transformation arrow, and the model-visible carrier (cell sentences). It excludes downstream output-only panels while keeping the labels needed to interpret the input route.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_7cb2dbf173a7",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "single-cell transcriptomic profiles",
          "actual_model_visible_form": "cell sentences made of ordered gene-name token sequences"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_26c7508ccc65",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene interaction metadata from CellPhoneDB and BioGRID",
          "actual_model_visible_form": "natural language interaction prompts"
        }
      ],
      "routes": [
        {
          "route_id": "route_7cb2dbf173a7",
          "configuration_id": "config_7de8c496a60a",
          "route_label": "single-cell language modeling and conditional cell generation",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Single cell language modeling; Cell type annotation; Conditional cell generation",
          "source_object_verbatim": "single-cell transcriptomic profiles",
          "source_object_normalized": "single-cell transcriptomic profiles",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "remove genes expressed in fewer than three cells and cells that express fewer than 100 genes",
            "normalize counts to 1 × 10 4 and log-transform",
            "convert expression profiles into cell sentences"
          ],
          "model_visible_form_verbatim": "cell sentences made of ordered gene-name token sequences",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "prompt formatting with the pretrained tokenizer; no new special tokens",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Following established conventions [61], we preprocess raw scRNA-seq data by removing genes that are expressed in fewer than three cells and cells that express fewer than 100 genes, followed by normalizing counts to 1 × 10 4 and log-transforming the data. Notably, we do not do any quality control for mitochondrial genes, as we found that proportions varied significantly across datasets and a single threshold was either too permissive or too restrictive across the corpus. For each dataset, the transcriptomic profiles were converted into cell sentences, and the accompanying annotations were preserved to construct natural language prompts. This resulted in a multimodal corpus linking expression profiles with textual descriptors of biological context.",
          "section_heading": "4.1 Data Collection",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3,
            8,
            38
          ],
          "doc_item_refs": [
            "#/texts/1618",
            "#/texts/1619",
            "#/texts/416",
            "#/texts/417",
            "#/texts/418",
            "#/texts/79",
            "#/texts/80",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/85",
            "#/texts/86",
            "#/texts/87"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000090::0001",
            "dense::full_2026-07-06__rec_000090::0025"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_92e25ae511ab",
          "configuration_id": "config_b4dd8cadb16d",
          "route_label": "multi-cell language modeling and conditional sample generation",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Multiple cell language modeling; Tissue sample annotation; Sample cell type(s) annotation; Conditional sample generation (tissue); Conditional sample generation (cell type); Conditional sample generation (abstract); Natural language interpretation",
          "source_object_verbatim": "multiple single-cell transcriptomic profiles from the same donor sample, tissue, or dataset",
          "source_object_normalized": "multiple single-cell transcriptomic profiles from the same donor sample, tissue, or dataset",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "sample multiple cells from the same donor sample",
            "convert each cell into a cell sentence",
            "combine cell sentences with task-specific natural language instructions and metadata"
          ],
          "model_visible_form_verbatim": "multiple cell sentences in a shared prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "combine the cell sentence representation of one or more cells with task-specific natural language instructions",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Prompts were constructed by combining the cell sentence representation of one or more cells with task-specific natural language instructions.",
          "section_heading": "4.3 Multi-Task Dataset Creation",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9,
            10,
            37,
            38,
            39,
            40
          ],
          "doc_item_refs": [
            "#/texts/1601",
            "#/texts/1602",
            "#/texts/1603",
            "#/texts/1604",
            "#/texts/1605",
            "#/texts/1606",
            "#/texts/1607",
            "#/texts/1608",
            "#/texts/1621",
            "#/texts/1622",
            "#/texts/1623",
            "#/texts/1624",
            "#/texts/1625",
            "#/texts/1626",
            "#/texts/1627",
            "#/texts/1628",
            "#/texts/1634",
            "#/texts/1635",
            "#/texts/1636",
            "#/texts/1650",
            "#/texts/1651",
            "#/texts/1652",
            "#/texts/1653",
            "#/texts/1654",
            "#/texts/1655",
            "#/texts/1658",
            "#/texts/1659",
            "#/texts/479",
            "#/texts/480",
            "#/texts/481",
            "#/texts/482",
            "#/texts/483",
            "#/texts/484",
            "#/texts/485",
            "#/texts/486"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000090::0024",
            "dense::full_2026-07-06__rec_000090::0026",
            "dense::full_2026-07-06__rec_000090::0027",
            "dense::full_2026-07-06__rec_000090::0029"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5000a33edeb7",
          "configuration_id": "config_93a6487f7864",
          "route_label": "gene set enumeration and naming",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Gene set enumeration; Gene set naming",
          "source_object_verbatim": "gene set name or list of genes in a gene set",
          "source_object_normalized": "gene set name or gene list",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "construct prompts from gene set names and gene lists"
          ],
          "model_visible_form_verbatim": "gene set name or list of genes",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "prompt formatting",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Training infrastructure varied by model scale to accommodate computational requirements. The C2S-Scale 410M 466 and 1B models were trained at Yale University using the Huggingface Transformers library (version 4.46.3) [64] and 467 PyTorch (version 2.4.1) [65] on a High Performance Computing (HPC) cluster running Red Hat Enterprise Linux 468 release 8.10. Each of these models was trained on 1-2 Nvidia H100 GPUs for several days (exact compute duration 469",
          "section_heading": "Table 1: Pretraining tasks for C2S-Scale multi-task training.",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            11
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/587",
            "#/texts/588",
            "#/texts/590",
            "#/texts/591",
            "#/texts/592"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_38607301656a",
          "configuration_id": "config_328922eb081f",
          "route_label": "cluster captioning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Cluster captioning",
          "source_object_verbatim": "cells from the same scRNA-seq cluster",
          "source_object_normalized": "cells from the same scRNA-seq cluster",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "curate 30 scRNA-seq datasets",
            "perform preprocessing, clustering, and differential expression analysis",
            "generate captions with GPT-4o",
            "randomly sample two cells from a cluster to form a multi-cell context prompt"
          ],
          "model_visible_form_verbatim": "two cell sentences from a cluster plus an instruction prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "multi-cell context prompt",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "For each training sample, we randomly sampled two cells from a cluster to form a multi-cell context prompt",
          "section_heading": "4.7.5 Cluster captioning",
          "supporting_figure_or_table": "Figure 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            40
          ],
          "doc_item_refs": [
            "#/pictures/9",
            "#/texts/1659",
            "#/texts/810",
            "#/texts/811",
            "#/texts/812",
            "#/texts/813"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000090::0030"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_cba015cfd76f",
          "configuration_id": "config_f70f0e040253",
          "route_label": "dataset interpretation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Dataset interpretation",
          "source_object_verbatim": "multiple cell sentences from the same tissue and donor plus the associated study abstract",
          "source_object_normalized": "multiple cell sentences from the same tissue and donor plus the associated study abstract",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "sample 5-20 cells from the same tissue and donor",
            "generate 500 variations of each ground-truth abstract using GPT-3.5-Turbo"
          ],
          "model_visible_form_verbatim": "multi-cell context prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "concatenate 5-20 cell sentences into the prompt",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Task Setup and Test Set Construction To assess the model's ability to produce high-level biological insights from raw transcriptomic data, we designed a dataset-level natural language interpretation task. Models were presented with a multi-cell context (5-20 cells sampled from the same tissue and donor) and prompted to generate a biological abstract summarizing the experiment. To prevent memorization of the original study abstracts, we generated 500 variations of each ground-truth abstract using GPT-3.5-Turbo, increasing diversity in language and phrasing while preserving factual content (Supplementary Figure 13). Since dataset interpretation was included as a main stage training task (Table 1), we evaluated base C2S-Scale checkpoints immediately following the main training phase without additional fine-tuning. We constructed two separate evaluation sets to measure performance. The In-Distribution (ID) Test Set comprises 3,065 samples derived from 613 scRNA-seq datasets within the C2S-Scale pretraining corpus; while these specific samples were strictly held out from training, the model had been exposed to other cells and abstract samples from these same studies. To rigorously benchmark generalization to unseen biological contexts, we constructed an Out-of-Distribution (OOD) Test Set comprising 400 samples from two studies completely unseen during pretraining: a pancreas dataset [23] and a human retina study [68].",
          "section_heading": "4.7.6 Dataset interpretation",
          "supporting_figure_or_table": "Figure 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            16,
            39,
            40
          ],
          "doc_item_refs": [
            "#/texts/1650",
            "#/texts/1651",
            "#/texts/1652",
            "#/texts/1653",
            "#/texts/1654",
            "#/texts/1655",
            "#/texts/1658",
            "#/texts/815",
            "#/texts/816",
            "#/texts/817",
            "#/texts/818",
            "#/texts/819",
            "#/texts/820",
            "#/texts/821",
            "#/texts/822"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000090::0028"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8b224797b192",
          "configuration_id": "config_9c070c42984b",
          "route_label": "spatial neighborhood prediction and related spatial tasks",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Spatial neighborhood prediction; Conditional neighbor generation; Niche label prediction; Same niche prediction",
          "source_object_verbatim": "multiple cell sentences from human liver tissue neighborhoods",
          "source_object_normalized": "multiple cell sentences from human liver tissue neighborhoods",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "define neighborhoods by a radius of 0.02 pixels",
            "sample positive or negative multi-cell contexts from the same or different neighborhood",
            "construct prompts for neighborhood, neighbor generation, niche label, and same-niche tasks"
          ],
          "model_visible_form_verbatim": "multiple cell sentences",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "multi-cell context prompt",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "To train C2S-Scale on spatial and multi-cellular relationships, we designed the following tasks. 1. Spatial neighborhood prediction: Given multiple cell sentences, predict whether these cells come from the same neighborhood. 2. Conditional neighbor generation: Given multiple cell sentences from a neighborhood, generate a novel cell sentence that would belong to the same neighborhood. 3. Niche label prediction: Given a cell sentence for a single cell, predict the niche label annotation for that cell. 4. Same niche prediction: Given multiple cell sentences, predict whether all of these cells have the same niche label or different niches.",
          "section_heading": "4.7.7 Spatial niche prediction",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16,
            33,
            42,
            43,
            44
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/texts/1551",
            "#/texts/1744",
            "#/texts/1745",
            "#/texts/1746",
            "#/texts/1747",
            "#/texts/1748",
            "#/texts/1749",
            "#/texts/1750",
            "#/texts/1751",
            "#/texts/1756",
            "#/texts/1757",
            "#/texts/1758",
            "#/texts/1759",
            "#/texts/1760",
            "#/texts/1761",
            "#/texts/1762",
            "#/texts/1763",
            "#/texts/873",
            "#/texts/874",
            "#/texts/875",
            "#/texts/876",
            "#/texts/877",
            "#/texts/878"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000090::0016",
            "dense::full_2026-07-06__rec_000090::0034",
            "dense::full_2026-07-06__rec_000090::0035",
            "dense::full_2026-07-06__rec_000090::0036",
            "dense::full_2026-07-06__rec_000090::0037"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_26c7508ccc65",
          "configuration_id": "config_2c9d4bb1b06f",
          "route_label": "gene interaction-augmented spatial reasoning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "CellPhoneDB and BioGRID interaction prompts",
          "source_object_verbatim": "gene interaction metadata from CellPhoneDB and BioGRID",
          "source_object_normalized": "gene interaction metadata from CellPhoneDB and BioGRID",
          "source_modality_normalized": "graph/network",
          "transformation_chain_verbatim": [
            "retain interactions involving the 1,000 genes in the CosMx data",
            "restrict to genes coding for extracellular proteins",
            "format interactions as natural language prompts"
          ],
          "model_visible_form_verbatim": "natural language interaction prompts",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "included in the training data mixture as prompts",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "we included gene interaction metadata from CellPhoneDB [41] and BioGRID [42] in the training data mixture.",
          "section_heading": "4.7.7 Spatial niche prediction",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16
          ],
          "doc_item_refs": [
            "#/texts/873",
            "#/texts/874",
            "#/texts/875",
            "#/texts/876",
            "#/texts/877",
            "#/texts/878"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6c6756ff33b1",
          "configuration_id": "config_e101a68fa73e",
          "route_label": "single-cell question answering from raw cell sentences",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Question answering",
          "source_object_verbatim": "raw cell sentences sampled from the corresponding dataset",
          "source_object_normalized": "raw cell sentences sampled from the corresponding dataset",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "sample cells from scRNA-seq datasets",
            "pair them with associated biological manuscripts",
            "use GPT-4.5 to generate question-answer pairs"
          ],
          "model_visible_form_verbatim": "raw cell sentences in a hybrid prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "hybrid context prompt",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "GPT-4.5 was provided with a hybrid context consisting of: (1) key text sections (Abstract, Results, Discussion) from a published scRNA-seq study, and (2) raw cell sentences sampled from the corresponding dataset.",
          "section_heading": "4.7.8 Question answering",
          "supporting_figure_or_table": "Figure 14",
          "evidence_status": "explicit_text",
          "uncertainty": "The study text is auxiliary context for QA-pair construction; the C2S-Scale route is the raw-cell-sentence side of the input.",
          "pages": [
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/880",
            "#/texts/881",
            "#/texts/882",
            "#/texts/883",
            "#/texts/884",
            "#/texts/885",
            "#/texts/886",
            "#/texts/887"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000090::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ade35c66eb40",
          "configuration_id": "config_e995b9f3ef6c",
          "route_label": "cytokine-stimulation perturbation prediction",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Cytokine Stimulation; Perturbation response prediction",
          "source_object_verbatim": "immune cells exposed to individual and combinatorial cytokines in the Dong et al. dataset",
          "source_object_normalized": "immune cells exposed to individual and combinatorial cytokines",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "retain 5000 most highly variable genes",
            "pair untreated and treated samples under each condition",
            "fine-tune on cell sentence generation and natural language label prediction",
            "align with GRPO using scGPT embedding-space rewards"
          ],
          "model_visible_form_verbatim": "cell sentence plus cell type, perturbation, and exposure labels",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt formatting with condition labels and response-cell generation",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The training of C2S models for the Dong et al. dataset followed a structured two-stage process",
          "section_heading": "4.7.9 Perturbation prediction",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            17,
            41
          ],
          "doc_item_refs": [
            "#/texts/1697",
            "#/texts/1698",
            "#/texts/1699",
            "#/texts/943",
            "#/texts/944",
            "#/texts/945"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_010",
            "full_2026-07-06__rec_000090::route_018"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000090::0031"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6054e5a31b07",
          "configuration_id": "config_02f830343904",
          "route_label": "L1000 bulk perturbation prediction",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Small Molecule Perturbations in Bulk RNAseq",
          "source_object_verbatim": "LINCS L1000 bulk RNA-seq profiles of treated and untreated cell lines",
          "source_object_normalized": "LINCS L1000 bulk RNA-seq profiles of treated and untreated cell lines",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "train on the 978 landmark genes",
            "pair untreated and treated samples by cell line",
            "fine-tune on cell-sentence generation",
            "refine with GRPO using Kendall's tau reward"
          ],
          "model_visible_form_verbatim": "cell sentence plus compound and cell-line metadata",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt formatting with perturbation metadata",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "For the L1000 dataset [52], we trained on the 978 genes measured in the original study",
          "section_heading": "4.7.9 Perturbation prediction",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            18,
            41
          ],
          "doc_item_refs": [
            "#/texts/1004",
            "#/texts/1005",
            "#/texts/1701",
            "#/texts/1702",
            "#/texts/1703",
            "#/texts/1704",
            "#/texts/1705",
            "#/texts/1706",
            "#/texts/1707",
            "#/texts/1708"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_011",
            "full_2026-07-06__rec_000090::route_019"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000090::0032"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d1fbf31866b4",
          "configuration_id": "config_76990fdf8f04",
          "route_label": "SciPlex3 drug perturbation prediction",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Small Molecule Perturbations in single-cell RNAseq",
          "source_object_verbatim": "single-cell RNA-seq profiles from SciPlex3",
          "source_object_normalized": "single-cell RNA-seq profiles from SciPlex3",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "use the ChemCPA 2,000-gene feature set",
            "train on the out-of-distribution split with unseen drugs",
            "use pathway information rather than RDKit embeddings",
            "evaluate generated response profiles"
          ],
          "model_visible_form_verbatim": "cell sentence plus drug, dose, cell-line, and pathway metadata",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt formatting with perturbation and pathway annotations",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We utilized the SciPlex3 dataset [55], which consists of 187 small molecule perturbations applied across three cell lines",
          "section_heading": "4.7.9 Perturbation prediction",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            18,
            41
          ],
          "doc_item_refs": [
            "#/texts/1006",
            "#/texts/1007",
            "#/texts/1008",
            "#/texts/1009",
            "#/texts/1010",
            "#/texts/1011",
            "#/texts/1012",
            "#/texts/1013",
            "#/texts/1711",
            "#/texts/1712",
            "#/texts/1713",
            "#/texts/1714",
            "#/texts/1715",
            "#/texts/1716",
            "#/texts/1717",
            "#/texts/1718"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_012",
            "full_2026-07-06__rec_000090::route_020"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000090::0033"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4890ce7d64b4",
          "configuration_id": "config_aa6fd5dcdc8b",
          "route_label": "cell type annotation (downstream fine-tuning)",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Cell type annotation",
          "source_object_verbatim": "single-cell transcriptomic profiles from immune, pancreas, and lung datasets",
          "source_object_normalized": "single-cell transcriptomic profiles from immune, pancreas, and lung datasets",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "filter genes expressed in fewer than 3 cells",
            "normalize library size to 10,000 counts per cell",
            "apply a log1p transformation",
            "convert expression profiles into cell sentences",
            "append a natural language instruction prompt"
          ],
          "model_visible_form_verbatim": "cell sentence plus natural language instruction",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt formatting with instruction appended to the cell sentence",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Input prompts combined the cell sentence with natural language instructions",
          "section_heading": "4.7.1 Cell type annotation",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13,
            14
          ],
          "doc_item_refs": [
            "#/texts/737",
            "#/texts/739",
            "#/texts/740"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_014"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d09976e7764b",
          "configuration_id": "config_83d6ff91fa18",
          "route_label": "cell generation (evaluation)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cell generation",
          "source_object_verbatim": "metadata prompts such as cell type or tissue conditions",
          "source_object_normalized": "metadata prompts such as cell type or tissue conditions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "format a natural language prompt with relevant metadata",
            "generate a valid cell sentence or multiple cell sentences"
          ],
          "model_visible_form_verbatim": "metadata-conditioned prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt formatting",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "the model was then tasked with generating a valid cell sentence (or multiple sentences) based on the metadata",
          "section_heading": "4.7.2 Cell generation",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            14
          ],
          "doc_item_refs": [
            "#/texts/742",
            "#/texts/743",
            "#/texts/744",
            "#/texts/745"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_015"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1862ed4360b8",
          "configuration_id": "config_6a4092dc936d",
          "route_label": "single-cell and bulk integration",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Single-cell bulk integration",
          "source_object_verbatim": "single-cell lung tissue profiles and pseudo-bulk RNA-seq samples",
          "source_object_normalized": "single-cell lung tissue profiles and pseudo-bulk RNA-seq samples",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "aggregate donor, cell type, and batch into pseudo-bulk samples",
            "randomly sample ten single-cell profiles from matching conditions to construct pairs",
            "embed single-cell and pseudo-bulk samples separately",
            "compute cosine similarity and FOSCTTM"
          ],
          "model_visible_form_verbatim": "paired single-cell and pseudo-bulk samples",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "separate embedding of each modality followed by similarity comparison",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "biological_payload",
          "input_status": "paired_alignment_input",
          "evidence_quote": "we designed a simple single-cell and bulk RNA seq integration task",
          "section_heading": "4.7.4 Single-cell bulk integration",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15
          ],
          "doc_item_refs": [
            "#/texts/807",
            "#/texts/808"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_017"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_2bfb9536cf22"
    },
    {
      "model_id": "model_9ead93555f87",
      "model_name": "C2S-Scale perturbation model",
      "record_id": "full_2026-07-06__rec_000090",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_c2885aa81441",
      "paper_title": "SCALING LARGE LANGUAGE MODELS FOR NEXT-GENERATION SINGLE-CELL ANALYSIS",
      "doi": "10.1101/2025.04.14.648850",
      "paper_url": "https://doi.org/10.1101/2025.04.14.648850",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "mixed transcriptomics"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_000090_figure_007.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_000090_0d9defcc25c1/figure_007.png",
        "figure_index": 7,
        "caption": "Figure 7: Virtual screening with C2S-Scale nominates a context-dependent modulator of antigen presentation. A, Schematic of the dual-context virtual screening workflow. The model predicts perturbation outcomes on MHC-I antigen presentation gene sets using initial states derived from primary tumor profiles (immune context-positive) and cell line profiles (immune context-neutral). B, Volcano plot displaying predicted drug effects in the primary tumor screen. Silmitasertib (red star) is identified as a high-scoring novel candidate. Inset details the top-ranking compounds, stratified by prior evidence: known MHC-I modulators (dark blue), plausible candidates with pathway links (light blue), and novel hits (grey). C, Experimental validation of the positive control, trametinib. Dose-response assessment of HLA-A,B,C surface levels (Mean Fluorescence Intensity; MFI) in WAGA cells treated with indicated concentrations of trametinib in the absence (gray) or presence (blue) of IFNβ (2 U/ml). D, Predicted context specificity of silmitasertib, showing a significantly higher antigen presentation score in the primary tumor context compared to the cell line context. E, Experimental validation of the novel hit, silmitasertib. Dose-response assessment of MHC-I MFI (normalized to vehicle control) in WAGA cells treated with silmitasertib without (gray) or with (blue) low-dose IFNβ (2 U/ml). Error bars represent mean ± s.d. Significance calculated via two-way t-test with Benjamini-Hochberg correction for multiple comparisons; ∗ P < 0 . 05 , ∗ ∗ P < 0 . 01 , ∗ ∗ ∗ P < 0 . 001 , ∗ ∗ ∗ ∗ P < 0 . 0001 .",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel biomedical research figure with panels labeled A-E.\n\nPanel A: Schematic “Virtual Screening Workflow.” Biological source objects are primary tumors marked “immune context-positive” and cell lines marked “immune context-neutral.” These feed into a C2S model/interface and a compound library of 4000+ drugs. Output categories indicate increased MHC-I antigen presentation gene set score as “hits” versus “no effect.”\n\nPanel B: Scatter plot titled “Primary Tumor Screen Results.” Axes show predicted antigen presentation score versus `-log10(p-value)`. Most points are gray/novel, with highlighted reported or plausible drugs in blue and silmitasertib marked with a red star. Labeled compounds include silmitasertib, crizotinib, nutlin-3, trametinib, GSK-126, afatinib, and paclitaxel. A zoomed inset states “16 / 23 Reported or Plausible.”\n\nPanel C: Bar plot for “Trametinib + IFNβ (Known Hit - Positive Control).” Biological measurement is HLA ABC MFI across trametinib concentrations in nM. Gray bars are DMSO only; blue bars are DMSO + 5 U/ml IFNβ. Significance brackets indicate increased HLA ABC MFI, especially with IFNβ.\n\nPanel D: Bar plot titled “Silmitasertib Context Specificity.” It compares predicted antigen presentation score in “Primary Tumor Context” versus “WAGA Cell Line Context,” with higher score in primary tumor context and significance marked.\n\nPanel E: Bar plot for “Silmitasertib + IFNβ (Novel Hit - C2S-predicted).” HLA ABC MFI is plotted across silmitasertib concentrations in nM. Gray bars are DMSO only; blue bars are DMSO + 2 U/ml IFNβ. Results show increased HLA ABC MFI with silmitasertib plus IFNβ, with significance brackets at multiple doses.\n\nOverall finding: the figure presents a C2S-based virtual screening workflow predicting compounds that increase MHC-I antigen presentation, highlights silmitasertib as a context-specific novel hit, and validates increased HLA ABC surface expression with IFNβ combination treatment.",
        "page_no": 35,
        "sha256": "693f314fb2b7d33dc7ba9a1d1b6d88730fba4aa74a580fa1a7d94b98debf4eab",
        "pixel_width": 930,
        "pixel_height": 507,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.47,
          "height": 0.49
        },
        "panel_label": "A",
        "visible_input_object": "Primary tumors (immune context-positive) and cell lines (immune context-neutral) feeding into C2S with a compound library",
        "visible_model_interface": "C2S virtual screening workflow with arrows from the source states through the model to predicted hits/no effect",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop isolates panel A only and keeps the source objects, C2S interface, compound-library input, and output arrows/labels needed to understand the virtual screening route while excluding the downstream result panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_b13affe14b84",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "a perturbation corpus combining LINCS L1000 bulk RNA-seq and Tahoe-100M single-cell perturbation data",
          "actual_model_visible_form": "delta sentence prompt with perturbed-control context"
        }
      ],
      "routes": [
        {
          "route_id": "route_b13affe14b84",
          "configuration_id": "config_a4393294c22e",
          "route_label": "virtual screening perturbation corpus",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Virtual screen setup",
          "source_object_verbatim": "a perturbation corpus combining LINCS L1000 bulk RNA-seq and Tahoe-100M single-cell perturbation data",
          "source_object_normalized": "combined perturbation corpus of L1000 bulk RNA-seq and Tahoe-100M single-cell perturbation data",
          "source_modality_normalized": "mixed transcriptomics",
          "transformation_chain_verbatim": [
            "initialize from a Gemma-3 4B checkpoint",
            "train on combined perturbation corpora",
            "predict top 200 upregulated and top 200 downregulated DEGs as delta sentences",
            "map predicted delta sentences back to log-fold changes"
          ],
          "model_visible_form_verbatim": "delta sentence prompt with perturbed-control context",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "prompt formatting for delta-sentence prediction",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "A delta sentence captures the vector difference between a perturbed and control state",
          "section_heading": "4.9 Virtual Screen Setup",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            19,
            20
          ],
          "doc_item_refs": [
            "#/texts/1079",
            "#/texts/1080",
            "#/texts/1081",
            "#/texts/1082",
            "#/texts/1083",
            "#/texts/1084",
            "#/texts/1085",
            "#/texts/1086"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000090::route_021"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_0cc72bd82901"
    },
    {
      "model_id": "model_fcda05dbbcd5",
      "model_name": "CaLMFlow",
      "record_id": "full_2026-07-06__rec_001687",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_d75a5cb4cc6a",
      "paper_title": "CaLMFlow: Volterra Flow Matching using Causal Language Models",
      "doi": "10.48550/arXiv.2410.05292",
      "paper_url": "https://doi.org/10.48550/arXiv.2410.05292",
      "route_count": 6,
      "configuration_count": 5,
      "family_counts": {
        "geometric_or_diffusion_state_carrier": 2,
        "text_native_token_stream": 2,
        "dense_continuous_carrier": 2
      },
      "subtype_counts": {
        "coordinate_backbone_or_shape_conditioning": 1,
        "noisy_diffusion_state": 1,
        "structured_biological_prompt_or_task_scaffold": 1,
        "direct_projected_embedding": 2,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier",
        "geometric_or_diffusion_state_carrier"
      ],
      "subtypes": [
        "coordinate_backbone_or_shape_conditioning",
        "direct_projected_embedding",
        "noisy_diffusion_state",
        "plain_language_prompt_or_question",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "continuous numerical vectors",
        "text"
      ],
      "lifecycle_phases": [
        "inference",
        "unclear"
      ],
      "fusion_topologies": [
        "concatenation",
        "prefix",
        "side_or_generative_conditioning",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query",
        "modality_or_task_selector",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001687_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001687_a48c004d6e93/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of the CaLMFlow framework. CaLMFlow takes as input textual conditions and flows and generates the next time point for the flows. The textual condition is tokenized and embedded using the LLM tokenizer and embedding layer while the conditional flows are transformed into spatial-temporal tokens using a learned projection. If multiple conditional flows are input simultaneously, the tokens are ordered by flow, space, and then time. The LLM applies causal language modeling and generates the next time point for each flow.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow diagram for generating conditional flows with a tokenized representation and an LLM-based next-token prediction model.\n\nVisible panels and labels:\n- **Input Conditional Flows**: Shows “Source distribution (t=0)” and “Target distribution (t=1)” with several 2D point-cloud distributions. Conditions are labeled as **Textual condition**, **Cell type**, and **Image label**. Example image-label panels include small image patches resembling digits or microscopy-like images.\n- **Tokenized Flow Tensor**: Shows a 3D tensor/grid labeled with axes including **Time (flow)** and trajectory-related dimensions. A highlighted token is labeled with components resembling `(t0, x0, s0)`, and the tensor is described as “Sequentialize” into token sequences.\n- **Next Token Prediction**: Shows an autoregressive sequence modeling setup with the label **Volterra IE** and an equation form for predicting `z_t` from prior state and an integral term. A central block labeled **LLM** receives condition tokens and flow tokens at times `t=0 ... t=T`. A lower block labeled **VAE Decoder** outputs parameters labeled `(μ, σ)`.\n- **Generated Flows**: Shows generated 2D point-cloud distributions at multiple time points labeled **t=0**, **t=0.1**, **t=0.2**, ellipsis, and **t=1**, illustrating evolution from initial to final distributions.\n\nBiological source objects:\n- No explicit biological specimen, tissue, cell image, or molecular object is shown. The “Cell type” label appears as a conditioning category, but the plotted objects are abstract colored point clouds.\n\nTransformations and model interfaces:\n- Conditional input distributions are converted into a **tokenized flow tensor**.\n- The tensor is sequentialized into token sequences.\n- Condition tokens and flow tokens are passed into an **LLM** for next-token prediction.\n- A **VAE Decoder** appears to decode predicted latent outputs into distribution parameters.\n- The final output is a sequence of generated flow states over time.\n\nFindings shown:\n- The figure communicates a method pipeline, not experimental results. It visually indicates that conditional source distributions can be tokenized, modeled autoregressively, and decoded to generate time-evolving flows toward target distributions.",
        "page_no": 2,
        "sha256": "39a5d9428f7b67b48168fadd51531712dcb3935c42fc2cd33810b549b5999176",
        "pixel_width": 783,
        "pixel_height": 244,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.54,
          "height": 1.0
        },
        "panel_label": "Input Conditional Flows + Tokenized Flow Tensor",
        "visible_input_object": "Source distribution (t=0) point clouds with conditioning labels",
        "visible_model_interface": "Tokenized flow tensor with sequentialization arrow toward model input",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop preserves the grounded input path from source distributions through tokenization into the flow tensor, with the readable labels and arrow needed to interpret the transformation. It excludes the next-token/decoder and generated-output panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "coordinate_backbone_or_shape_conditioning",
          "family_id": "geometric_or_diffusion_state_carrier",
          "route_id": "route_f8ebcb685ef1",
          "example_input": "residue/atom coordinates (xᵢ,yᵢ,zᵢ)",
          "example_carrier": "equivariant geometric state",
          "example_interface": "geometry-aware generator",
          "actual_source": "the initial source distribution (e.g., a Gaussian)",
          "actual_model_visible_form": "a sequence ( z t 0 , z t 1 , . . . , z t N )"
        },
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_1f5299997eba",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "conditional flow-matching trajectories",
          "actual_model_visible_form": "embedded flow-matching conditional trajectories"
        },
        {
          "subtype_id": "noisy_diffusion_state",
          "family_id": "geometric_or_diffusion_state_carrier",
          "route_id": "route_a05532f42152",
          "example_input": "biological state x₀ + noise ε",
          "example_carrier": "xₜ = √αₜx₀ + √(1−αₜ)ε",
          "example_interface": "conditioned denoiser / flow model",
          "actual_source": "Gaussian noise",
          "actual_model_visible_form": "initial conditions"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_cd801d11023c",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "image label",
          "actual_model_visible_form": "image-label condition"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_45715fc3d13b",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "perturbation conditions",
          "actual_model_visible_form": "embedded text prompts"
        }
      ],
      "routes": [
        {
          "route_id": "route_f8ebcb685ef1",
          "configuration_id": "config_299257729549",
          "route_label": "Synthetic conditional trajectories",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "synthetic datasets",
          "source_object_verbatim": "the initial source distribution (e.g., a Gaussian)",
          "source_object_normalized": "initial source distribution",
          "source_modality_normalized": "continuous numerical vectors",
          "transformation_chain_verbatim": [
            "discretized into N time steps",
            "modeled as a sequence of states"
          ],
          "model_visible_form_verbatim": "a sequence ( z t 0 , z t 1 , . . . , z t N )",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "coordinate_backbone_or_shape_conditioning",
          "insertion_or_fusion_verbatim": "passed into the CLM",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Solving a nonlinear integral equation generally requires some iterative procedure, where an initial guess is refined iteratively. Since the evaluation of z at time t requires evaluation of z at all time points between 0 and t due to the integral ∫ t 0 G ( z s , t, s ) ds . Observe that once an approximation (guess) z j ( t ) has been obtained, all evaluations happen in parallel for each iteration, since we can integrate using z j ( t ) , see Zappala et al. (2024) for details. We present the model with sequences starting from the initial state z 0 and extending to various lengths (e.g., predicting from z 0 to z 1 , z 0 to z 2 , up to z 0 to z N ). This trains the model on multiple sub-trajectories, which can be used in inference in an iterative manner to output the complete trajectory.",
          "section_heading": "3.2 SOLVING VOLTERRA INTEGRAL EQUATIONS WITH CAUSAL LANGUAGE MODELS",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper frames the input as a discretized continuous trajectory; the exact carrier subtype is inferred.",
          "pages": [
            3,
            4,
            18,
            19,
            20,
            21
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/pictures/6",
            "#/texts/252",
            "#/texts/253",
            "#/texts/255",
            "#/texts/258",
            "#/texts/268",
            "#/texts/270",
            "#/texts/43",
            "#/texts/44",
            "#/texts/46",
            "#/texts/47",
            "#/texts/48",
            "#/texts/49",
            "#/texts/50",
            "#/texts/51"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001687::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001687::0009",
            "dense::full_2026-07-06__rec_001687::0038"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a05532f42152",
          "configuration_id": "config_6ff25415292f",
          "route_label": "Unconditional single-cell noise",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "unconditional generation experiment",
          "source_object_verbatim": "Gaussian noise",
          "source_object_normalized": "Gaussian noise",
          "source_modality_normalized": "continuous numerical vectors",
          "transformation_chain_verbatim": [
            "used as initial conditions"
          ],
          "model_visible_form_verbatim": "initial conditions",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "noisy_diffusion_state",
          "insertion_or_fusion_verbatim": "from Gaussian noise as initial conditions",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Gaussian noise",
          "section_heading": "5.2.1 UNCONDITIONAL GENERATION OF SINGLE-CELL DATA",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            7
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/93",
            "#/texts/95",
            "#/texts/96"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001687::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001687::0019",
            "dense::full_2026-07-06__rec_001687::0020"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_45715fc3d13b",
          "configuration_id": "config_a014627e87ff",
          "route_label": "Single-cell perturbation prompt",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "single-cell perturbation response prediction",
          "source_object_verbatim": "perturbation conditions",
          "source_object_normalized": "perturbation condition prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "represented as simple text prompts",
            "tokenized using AutoTokenizer.from_pretrained(\"EleutherAI/pythia-160m\")",
            "embedded using the embedding layers of a customized PyThia-160M model"
          ],
          "model_visible_form_verbatim": "embedded text prompts",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prepended to the embedded flow-matching conditional trajectories",
          "fusion_topology": "prefix",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We leverage CLMs' inherent capabilities to encode and comprehend natural language by representing perturbation conditions as simple text prompts (see A.1.3 for details). These prompts are prepended to the embedded flow-matching conditional trajectories and processed through the CLM's tokenizer and embedding layers. For details on conditional encodings for other models, see A.1.3.",
          "section_heading": "A.1.3 SINGLE CELL CONDITIONAL ENCODING",
          "supporting_figure_or_table": "Table 4",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper uses this route in both training and inference, so the lifecycle phase is collapsed to unclear.",
          "pages": [
            1,
            2,
            8,
            17,
            18
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/106",
            "#/texts/107",
            "#/texts/108",
            "#/texts/109",
            "#/texts/16",
            "#/texts/17",
            "#/texts/18",
            "#/texts/19",
            "#/texts/21",
            "#/texts/22",
            "#/texts/229",
            "#/texts/23",
            "#/texts/230",
            "#/texts/231",
            "#/texts/232",
            "#/texts/233",
            "#/texts/234",
            "#/texts/235",
            "#/texts/236",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001687::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001687::0001",
            "dense::full_2026-07-06__rec_001687::0002",
            "dense::full_2026-07-06__rec_001687::0005",
            "dense::full_2026-07-06__rec_001687::0036"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1f5299997eba",
          "configuration_id": "config_a014627e87ff",
          "route_label": "Single-cell conditional trajectories",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "single-cell perturbation response prediction",
          "source_object_verbatim": "conditional flow-matching trajectories",
          "source_object_normalized": "conditional trajectory embeddings",
          "source_modality_normalized": "continuous numerical vectors",
          "transformation_chain_verbatim": [
            "embedded",
            "processed through the CLM's tokenizer and embedding layers"
          ],
          "model_visible_form_verbatim": "embedded flow-matching conditional trajectories",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "prepended with the text prompts",
          "fusion_topology": "prefix",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "We leverage CLMs' inherent capabilities to encode and comprehend natural language by representing perturbation conditions as simple text prompts (see A.1.3 for details). These prompts are prepended to the embedded flow-matching conditional trajectories and processed through the CLM's tokenizer and embedding layers. For details on conditional encodings for other models, see A.1.3.",
          "section_heading": "5.2.2 SINGLE-CELL PERTURBATION RESPONSE PREDICTION",
          "supporting_figure_or_table": "Table 4",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper treats prompts and trajectories as a composite conditioning scheme; this route isolates the trajectory side.",
          "pages": [
            3,
            4,
            8,
            17,
            18,
            21,
            22
          ],
          "doc_item_refs": [
            "#/texts/106",
            "#/texts/107",
            "#/texts/108",
            "#/texts/109",
            "#/texts/229",
            "#/texts/230",
            "#/texts/231",
            "#/texts/232",
            "#/texts/233",
            "#/texts/234",
            "#/texts/235",
            "#/texts/236",
            "#/texts/43",
            "#/texts/44",
            "#/texts/46",
            "#/texts/47",
            "#/texts/48",
            "#/texts/49",
            "#/texts/50",
            "#/texts/51"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001687::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001687::0036",
            "dense::full_2026-07-06__rec_001687::0045"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1562a2c1b3ef",
          "configuration_id": "config_016f94e5bc27",
          "route_label": "Multi-trajectory spatiotemporal tokens",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "multi-trajectory tokenization",
          "source_object_verbatim": "more than one conditional trajectory",
          "source_object_normalized": "conditional trajectory batch",
          "source_modality_normalized": "continuous numerical vectors",
          "transformation_chain_verbatim": [
            "sampled as M spatiotemporal sequences of tokens",
            "ordered by flow, space, and then time"
          ],
          "model_visible_form_verbatim": "M spatiotemporal sequences of tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "input to the CLM",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "more than one conditional trajectory",
          "section_heading": "4.2 MULTI-TRAJECTORY TOKENIZATION",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes a batched multi-trajectory representation rather than a single trajectory.",
          "pages": [
            2,
            5,
            9,
            10
          ],
          "doc_item_refs": [
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/64",
            "#/texts/66",
            "#/texts/67",
            "#/texts/68",
            "#/texts/69",
            "#/texts/70",
            "#/texts/71",
            "#/texts/72",
            "#/texts/74",
            "#/texts/75",
            "#/texts/76"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001687::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001687::0007",
            "dense::full_2026-07-06__rec_001687::0010",
            "dense::full_2026-07-06__rec_001687::0011"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_cd801d11023c",
          "configuration_id": "config_200d022b888d",
          "route_label": "MNIST image-label condition",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "conditional image generation on MNIST",
          "source_object_verbatim": "image label",
          "source_object_normalized": "image class label",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenized",
            "embedded using the LLM tokenizer and embedding layer"
          ],
          "model_visible_form_verbatim": "image-label condition",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "condition tokens and flow tokens are passed into an LLM",
          "fusion_topology": "concatenation",
          "text_role": "modality_or_task_selector",
          "input_status": "actual_model_input",
          "evidence_quote": "Image label",
          "section_heading": null,
          "supporting_figure_or_table": "Figure 1; Figure 9",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The exact label-encoding pipeline is not spelled out in text.",
          "pages": [
            2,
            5,
            9
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/124",
            "#/texts/21",
            "#/texts/22",
            "#/texts/74",
            "#/texts/75",
            "#/texts/76"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001687::route_012"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001687::0003"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_8ec5e76d5e6a"
    },
    {
      "model_id": "model_0dc740f11cc3",
      "model_name": "CatBoost",
      "record_id": "full_2026-07-06__rec_001187",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_9645a7bdccf5",
      "paper_title": "Precious3GPT: Multimodal Multi-Species Multi-Omics Multi-Tissue Transformer for Aging Research and Drug Discovery",
      "doi": "10.1101/2024.07.25.605062",
      "paper_url": "https://doi.org/10.1101/2024.07.25.605062",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 1
      },
      "subtype_counts": {
        "pooled_or_aggregated_embedding": 1
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "pooled_or_aggregated_embedding"
      ],
      "primary_subtype": "pooled_or_aggregated_embedding",
      "modalities": [
        "DNA methylation and dense embeddings"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "concatenation"
      ],
      "text_roles": [
        "no_text_on_this_route"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "Neither blind selection is supported by the contact sheet. The visible figures for this route are downstream evaluations only: Figure 3E/F and Figure 4A show age-prediction performance or correlations, not the actual CatBoost input route. Figure 1/6 are generic architecture schematics, but they do not visibly show the specific fusion of averaged P3GPT embeddings with promoter beta-values that feeds the CatBoost predictor.",
      "illustrative_examples": [
        {
          "subtype_id": "pooled_or_aggregated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_acbc9f71b368",
          "example_input": "{gene/cell/patch embeddings}",
          "example_carrier": "mean/attention pool = one compact vector",
          "example_interface": "aggregator → generator",
          "actual_source": "P3GPT embeddings and ß-values",
          "actual_model_visible_form": "P3GPT embeddings of the 50 most methylated genes"
        }
      ],
      "routes": [
        {
          "route_id": "route_acbc9f71b368",
          "configuration_id": "config_8700f9c1c72d",
          "route_label": "P3GPT-derived aging clock feature extraction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "P3GPT-derived aging clocks",
          "source_object_verbatim": "P3GPT embeddings and ß-values",
          "source_object_normalized": "P3GPT-derived embeddings plus promoter beta-values",
          "source_modality_normalized": "DNA methylation and dense embeddings",
          "transformation_chain_verbatim": [
            "averaged P3GPT embeddings of the 50 most methylated genes for each sample",
            "stacked the embeddings with a matrix of average ß-values on the promoters of 360 genes",
            "train a CatBoost predictor of chronological age"
          ],
          "model_visible_form_verbatim": "P3GPT embeddings of the 50 most methylated genes",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "we first averaged P3GPT embeddings of the 50 most methylated genes for each sample... Then we stacked the embeddings with a matrix of average ß-values on the promoters of 360 genes",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "The stacked P3GPT embeddings and ß-values were then used to train a CatBoost predictor of chronological age",
          "section_heading": "P3GPT-derived aging clocks",
          "supporting_figure_or_table": "Figure 3E; Figure 3F; Figure 4A",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13,
            26
          ],
          "doc_item_refs": [
            "#/texts/171",
            "#/texts/172",
            "#/texts/173",
            "#/texts/174",
            "#/texts/67",
            "#/texts/68",
            "#/texts/69",
            "#/texts/70"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001187::0004"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_998d40c02fca"
    },
    {
      "model_id": "model_a0f67b1880c7",
      "model_name": "Cell-o1",
      "record_id": "full_2026-07-06__rec_001617",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4f442dcf4ed4",
      "paper_title": "Cell-o1: Training LLMs to Solve Single-Cell Reasoning Puzzles with Reinforcement Learning",
      "doi": "10.48550/arXiv.2506.02911",
      "paper_url": "https://doi.org/10.48550/arXiv.2506.02911",
      "route_count": 2,
      "configuration_count": 2,
      "family_counts": {
        "text_native_token_stream": 2
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 2
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell RNA sequencing batch with donor metadata",
        "single-cell RNA sequencing cell with donor metadata"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001617_figure_007.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001617_76b0056d59c3/figure_007.png",
        "figure_index": 7,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow for single-cell gene expression data and annotation.\n\nVisible elements:\n- Left panel labeled “A Batch of Cells from a Donor,” showing a syringe/sample collection icon and multiple colored circular cells.\n- Middle panel labeled “Donor Information,” with a small clinical metadata table:\n  - Tissue: Lung\n  - Disease: Lung Adenocarcinoma\n  - Age: 68\n  - Sex: Male\n  - Smoking Status: Non-smoker\n- Middle-right panel labeled “Gene Expression Profiles,” showing a heatmap-like matrix with rows for Cell 1, Cell 2, Cell 3, … Cell N and columns Gene 1 through Gene N.\n- Right panel labeled “Contextual Metadata,” containing a natural-language prompt: “The cell is from a male at the 68-year-old stage, originating from the lung. The patient has been diagnosed with Lung adenocarcinoma.”\n- Far-right panel labeled “Top Expressed Genes,” listing top genes per cell:\n  - Cell 1: S100A9, TMSB10, RPL37, RPL18A…\n  - Cell 2: MALAT1, FTL, B2M, CD74…\n  - Cell N: IGLC3, IGLC2, IGHM, MT-CO3,…\n\nBiological source objects:\n- Donor-derived cells, apparently single cells from lung tissue.\n- Clinical context indicates lung adenocarcinoma from a 68-year-old male non-smoker.\n\nTransformations/model interface:\n- Raw donor cell batch plus donor metadata are paired with gene expression profiles.\n- Metadata is converted into contextual natural-language text.\n- Expression profiles are summarized as top expressed genes per cell.\n\nFinding/message:\n- The figure illustrates integration of single-cell gene expression data with donor clinical metadata to produce cell-level contextual descriptions and top expressed gene lists.",
        "page_no": 17,
        "sha256": "d6de6dfd9786dfe9e0ede5e846dcb2af9314dccdea07df4a3902ae9a94d7e657",
        "pixel_width": 792,
        "pixel_height": 129,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 0.8,
          "height": 1
        },
        "panel_label": "source-to-context route",
        "visible_input_object": "Batch of cells from a donor with donor clinical metadata and gene expression profiles",
        "visible_model_interface": "Natural-language contextual metadata prompt built from donor and cell-expression inputs",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "The left and middle panels show the grounded inputs (cell batch, donor metadata, gene-expression profiles) and the right-side contextual metadata sentence that serves as the model-visible text carrier. The top-expressed-genes panel is output-like and excluded.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_6c9a758213b0",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "the gene expression profile of a single cell from a specific donor",
          "actual_model_visible_form": "a single-cell prompt with donor context, candidate labels, and structured <think>/<answer> instructions"
        }
      ],
      "routes": [
        {
          "route_id": "route_6c9a758213b0",
          "configuration_id": "config_81231981abe8",
          "route_label": "cell-level reasoning prompt (Cell-o1)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cell-level Reasoning Setup",
          "source_object_verbatim": "the gene expression profile of a single cell from a specific donor",
          "source_object_normalized": "single cell from a specific donor with top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing cell with donor metadata",
          "transformation_chain_verbatim": [
            "use the single cell's top expressed genes",
            "combine them with donor context",
            "provide a fixed candidate label set",
            "format the response with reasoning tags"
          ],
          "model_visible_form_verbatim": "a single-cell prompt with donor context, candidate labels, and structured <think>/<answer> instructions",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt template",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are an expert assistant specialized in cell type annotation. You will be given the gene expression profile of a single cell from a specific donor. The top expressed genes are listed in descending order. Use both gene expression and donor context to determine the correct cell type. You will also receive a list of candidate cell types-choose the one that best fits this cell . Include your detailed reasoning within <think> and </think> tags, and provide your final answer within <answer> and </answer> tags. The final answer should be a single string with exactly one cell type.",
          "section_heading": "C Cell-level vs. Batch-level Reasoning",
          "supporting_figure_or_table": "Table 10",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            21
          ],
          "doc_item_refs": [
            "#/texts/1015",
            "#/texts/1016",
            "#/texts/1017",
            "#/texts/1018",
            "#/texts/1019",
            "#/texts/1020"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1abad50b2c88",
          "configuration_id": "config_0ee079663382",
          "route_label": "open-ended QA prompt (Cell-o1)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Open-ended QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "provide gene expression profile and metadata",
            "remove the constrained label set",
            "ask for free-form cell type generation"
          ],
          "model_visible_form_verbatim": "free-form textual generation of a cell type name for each cell",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompting without constrained labels",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_285e25ae2bfb",
      "model_name": "Cell2Text",
      "record_id": "june_update_2026-06-10__rec_000248",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_4250fe0b5a77",
      "paper_title": "Cell2Text: Multimodal LLM for generating textual descriptions from single-cell RNA-Seq profiles",
      "doi": "",
      "paper_url": "",
      "route_count": 2,
      "configuration_count": 2,
      "family_counts": {
        "dense_continuous_carrier": 1,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "direct_projected_embedding": 1,
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "RNA-seq",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "side_or_generative_conditioning",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/june_update_2026_06_10_rec_000248_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/june_update_2026_06_10_rec_000248_de5b21014e76/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of the Cell2Text framework. The model takes single-cell RNA-seq profiles as input and processes them through a pretrained Geneformer encoder to generate contextualized gene-level embeddings. These embeddings are projected into the semantic space of the language model via a lightweight adapter module, aligning biological signals with linguistic representations. A pretrained, instruction-tuned LLM decoder then generates structured natural language descriptions that capture cellular identity, tissue of origin, disease associations, and pathway activity.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow for a single-cell/gene-expression-to-language-model system.\n\nVisible elements:\n- Left side shows an input cell represented as an ordered gene sequence: “cell 1: Gene Z Gene E ... Gene C”.\n- A downward branch lists “Cells Metadata” including fields such as cell id, cell type, development stage, disease, sex, tissue, tissue_general, and pathways.\n- A text target labeled “cell description (Cp)” gives an example natural-language description of the cell, mentioning an endothelial cell and vein context.\n\nBiological source objects:\n- Genes from a single cell.\n- Cell-level metadata including cell type, tissue, developmental stage, disease state, sex, and pathway annotations.\n- Pathway labels visible include examples such as “HALLMARK_COAGULATION” and “HALLMARK_MYC_TARGETS_V2”.\n\nTransformations/model components:\n- Gene sequence is passed through a “cell encoder”.\n- The encoder produces “gene embeddings”.\n- Gene embeddings are passed into a “modality adapter”.\n- The modality adapter maps embeddings into token-like dimensions shown with labels such as 1152, 2048, and 3072.\n- The adapted embeddings are inserted into a language-model input stream alongside “instruction embeddings” and “prompt embeddings”.\n- A “language model decoder” consumes these embeddings.\n\nModel interface/output:\n- The decoder predicts “next token probability p”.\n- Training objective shown as cross-entropy loss: `L_sft = L_CrossEntropy(Cp, p)`, comparing the generated probability distribution with the target cell description `Cp`.\n\nFinding/claim visible:\n- The figure illustrates supervised fine-tuning where biological cell/gene embeddings are adapted into a language model decoder to generate textual cell descriptions from single-cell gene and metadata inputs.",
        "page_no": 4,
        "sha256": "56abf86ae3158af55d3ac567068e0e83f0def5b306fd4a1a9599608086aabe6c",
        "pixel_width": 856,
        "pixel_height": 460,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.79,
          "height": 0.36
        },
        "panel_label": "top-left to top-right input-to-adapter route",
        "visible_input_object": "cell 1 gene sequence and the cell encoder -> gene embeddings -> modality adapter chain",
        "visible_model_interface": "projected gene embeddings entering the modality adapter, with the adapter box label visible",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the source cell gene sequence, the arrow into the cell encoder, the gene embeddings bars, and the modality adapter box. It is the smallest coherent view of the actual scRNA-seq-to-LLM embedding route while excluding the decoder/output panel.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_40860db1335d",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "single-cell RNA-seq profiles",
          "actual_model_visible_form": "projected gene embeddings"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_53995d978c3b",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "instruction-following prompt structure",
          "actual_model_visible_form": "instruction-following prompt text"
        }
      ],
      "routes": [
        {
          "route_id": "route_40860db1335d",
          "configuration_id": "config_f14ba9f518fd",
          "route_label": "scRNA-seq profiles to projected gene embeddings",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "generate comprehensive and accurate natural language descriptions of single cells",
          "source_object_verbatim": "single-cell RNA-seq profiles",
          "source_object_normalized": "single-cell RNA-seq profiles",
          "source_modality_normalized": "RNA-seq",
          "transformation_chain_verbatim": [
            "single-cell RNA-seq profiles",
            "Geneformer encoder",
            "lightweight adapter module",
            "projected into the LLM's input embedding space"
          ],
          "model_visible_form_verbatim": "projected gene embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "projected into the LLM's input embedding space",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Figure 1: Overview of the Cell2Text framework. The model takes single-cell RNA-seq profiles as input and processes them through a pretrained Geneformer encoder to generate contextualized gene-level embeddings. These embeddings are projected into the semantic space of the language model via a lightweight adapter module, aligning biological signals with linguistic representations. A pretrained, instruction-tuned LLM decoder then generates structured natural language descriptions that capture cellular identity, tissue of origin, disease associations, and pathway activity.",
          "section_heading": "3 METHODOLOGY",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            3,
            4
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/10",
            "#/texts/29",
            "#/texts/31"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000248::route_001"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000248::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_53995d978c3b",
          "configuration_id": "config_a236036d3438",
          "route_label": "instruction-following prompt to Cell2Text decoder",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "specific instruction-following prompt structure",
          "source_object_verbatim": "instruction-following prompt structure",
          "source_object_normalized": "instruction-following prompt structure",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "instruction-following prompt structure"
          ],
          "model_visible_form_verbatim": "instruction-following prompt text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "integrates a system message and the contextualized gene embeddings",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "we adopted a specific instruction-following prompt structure",
          "section_heading": "3.2.1 FULL FINE-TUNING",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/47",
            "#/texts/48",
            "#/texts/49"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000248::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_c17be9263882"
    },
    {
      "model_id": "model_35105bbde921",
      "model_name": "Cell2Text-Gemma-4B",
      "record_id": "full_2026-07-06__rec_001838",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4250fe0b5a77",
      "paper_title": "Cell2Text: Multimodal LLM for Generating Single-Cell Descriptions from RNA-Seq Data",
      "doi": "10.48550/arXiv.2509.24840",
      "paper_url": "https://doi.org/10.48550/arXiv.2509.24840",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 1
      },
      "subtype_counts": {
        "direct_projected_embedding": 1
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "single-cell RNA sequencing"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "encoder_decoder"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001838_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001838_bb6ae63f17ad/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of the Cell2Text framework. The model takes single-cell RNA-seq profiles as input and processes them through a pretrained Geneformer encoder to generate contextualized gene-level embeddings. These embeddings are projected into the semantic space of the language model via a lightweight adapter module, aligning biological signals with linguistic representations. A pretrained, instruction-tuned LLM decoder then generates structured natural language descriptions that capture cellular identity, tissue of origin, disease associations, and pathway activity.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow for a cell-description generation or multimodal language-model training setup.\n\nVisible elements:\n- Input biological source object: a single-cell gene expression vector labeled “cell 1: Gene Z Gene E ... Gene C”.\n- Cell metadata block lists attributes such as cell id, cell type “vein endothelial cell”, development stage “prime adult stage”, disease “normal”, sex “male”, tissue “caudate lobe of liver”, tissue general “liver”, and pathway labels including “HALLMARK_COAGULATION” and “HALLMARK_MYC_TARGETS_V2”.\n- A textual target/output caption labeled “cell description (Cp)” states that the sample consists of a vein endothelial cell and an endothelial cell that is part of a vein.\n- Transformations:\n  - Gene expression is passed through a “cell encoder”.\n  - Output becomes “gene embeddings”.\n  - Gene embeddings are passed through a “modality adapter”.\n  - The adapter maps embeddings into dimensions labeled 1152, 2048, and 3072.\n- Model interface:\n  - A sequence of embeddings is fed into a “language model decoder”.\n  - Embedding types are color-coded and labeled:\n    - green: instruction embeddings\n    - orange: gene embeddings\n    - green: prompt embeddings\n    - gray: generated/next-token-related embeddings\n- Training objective:\n  - The decoder predicts “next token probability p”.\n  - Loss is shown as `L_sft = L_CrossEntropy(Cp, p)`.\n- Finding/claim shown by the figure:\n  - The diagram illustrates supervised fine-tuning of a language model decoder to generate cell descriptions from gene-expression-derived embeddings plus instruction and prompt embeddings.",
        "page_no": 4,
        "sha256": "56abf86ae3158af55d3ac567068e0e83f0def5b306fd4a1a9599608086aabe6c",
        "pixel_width": 856,
        "pixel_height": 460,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.84,
          "height": 0.42
        },
        "panel_label": "input-to-adapter route",
        "visible_input_object": "Cell 1 gene-expression source text and the cell-encoder pathway",
        "visible_model_interface": "cell encoder -> gene embeddings -> modality adapter",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the source gene-expression input, the cell encoder, the gene-embedding carrier, and the modality adapter box with arrows, while excluding the downstream decoder/output area.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_89003c6e8325",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "single-cell RNA-seq profiles",
          "actual_model_visible_form": "sequence embeddings projected into the LLM's input embedding space"
        }
      ],
      "routes": [
        {
          "route_id": "route_89003c6e8325",
          "configuration_id": "config_e37c16992a01",
          "route_label": "Cell2Text-Gemma-4B full fine-tuning route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "full fine-tuning for cell description generation",
          "source_object_verbatim": "single-cell RNA-seq profiles",
          "source_object_normalized": "single-cell RNA-seq expression profiles",
          "source_modality_normalized": "single-cell RNA sequencing",
          "transformation_chain_verbatim": [
            "single-cell RNA-seq profiles",
            "Geneformer encoder",
            "gene-level embeddings",
            "lightweight adapter module",
            "instruction-following prompt structure",
            "Gemma3-4B-it decoder"
          ],
          "model_visible_form_verbatim": "sequence embeddings projected into the LLM's input embedding space",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "a lightweight adapter module projects Geneformer outputs into the language model's semantic space",
          "fusion_topology": "encoder_decoder",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Single-cell RNA sequencing has transformed biology by enabling the measurement of gene expression at cellular resolution, providing information for cell types, states, and disease contexts. Recently, single-cell foundation models have emerged as powerful tools for learning transferable representations directly from expression profiles, improving performance on classification and clustering tasks. However, these models are limited to discrete prediction heads, which collapse cellular complexity into predefined labels that fail to capture the richer, contextual explanations biologists need. We introduce Cell2Text, a multimodal generative framework that translates scRNA-seq profiles into structured natural language descriptions. By integrating gene-level embeddings from single-cell foundation models with pretrained large language models, Cell2Text generates coherent summaries that capture cellular identity, tissue origin, disease associations, and pathway activity, generalizing to unseen cells. Empirically, Cell2Text outperforms baselines on classification accuracy, demonstrates strong ontological consistency using PageRank-based similarity metrics, and achieves high semantic fidelity in text generation. These results demonstrate that coupling expression data with natural language offers both stronger predictive performance and inherently interpretable outputs, pointing to a scalable path for label-efficient characterization of unseen cells.",
          "section_heading": "3.2.1 FULL FINE-TUNING",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            5
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/41",
            "#/texts/42",
            "#/texts/43"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001838::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001838::0001"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_000c4dd8661a"
    },
    {
      "model_id": "model_99366b543213",
      "model_name": "Cell2Text-Llama-1B",
      "record_id": "full_2026-07-06__rec_001838",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4250fe0b5a77",
      "paper_title": "Cell2Text: Multimodal LLM for Generating Single-Cell Descriptions from RNA-Seq Data",
      "doi": "10.48550/arXiv.2509.24840",
      "paper_url": "https://doi.org/10.48550/arXiv.2509.24840",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 1
      },
      "subtype_counts": {
        "direct_projected_embedding": 1
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "single-cell RNA sequencing"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "encoder_decoder"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001838_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001838_bb6ae63f17ad/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of the Cell2Text framework. The model takes single-cell RNA-seq profiles as input and processes them through a pretrained Geneformer encoder to generate contextualized gene-level embeddings. These embeddings are projected into the semantic space of the language model via a lightweight adapter module, aligning biological signals with linguistic representations. A pretrained, instruction-tuned LLM decoder then generates structured natural language descriptions that capture cellular identity, tissue of origin, disease associations, and pathway activity.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow for a cell-description generation or multimodal language-model training setup.\n\nVisible elements:\n- Input biological source object: a single-cell gene expression vector labeled “cell 1: Gene Z Gene E ... Gene C”.\n- Cell metadata block lists attributes such as cell id, cell type “vein endothelial cell”, development stage “prime adult stage”, disease “normal”, sex “male”, tissue “caudate lobe of liver”, tissue general “liver”, and pathway labels including “HALLMARK_COAGULATION” and “HALLMARK_MYC_TARGETS_V2”.\n- A textual target/output caption labeled “cell description (Cp)” states that the sample consists of a vein endothelial cell and an endothelial cell that is part of a vein.\n- Transformations:\n  - Gene expression is passed through a “cell encoder”.\n  - Output becomes “gene embeddings”.\n  - Gene embeddings are passed through a “modality adapter”.\n  - The adapter maps embeddings into dimensions labeled 1152, 2048, and 3072.\n- Model interface:\n  - A sequence of embeddings is fed into a “language model decoder”.\n  - Embedding types are color-coded and labeled:\n    - green: instruction embeddings\n    - orange: gene embeddings\n    - green: prompt embeddings\n    - gray: generated/next-token-related embeddings\n- Training objective:\n  - The decoder predicts “next token probability p”.\n  - Loss is shown as `L_sft = L_CrossEntropy(Cp, p)`.\n- Finding/claim shown by the figure:\n  - The diagram illustrates supervised fine-tuning of a language model decoder to generate cell descriptions from gene-expression-derived embeddings plus instruction and prompt embeddings.",
        "page_no": 4,
        "sha256": "56abf86ae3158af55d3ac567068e0e83f0def5b306fd4a1a9599608086aabe6c",
        "pixel_width": 856,
        "pixel_height": 460,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 1,
          "height": 0.77
        },
        "panel_label": "top workflow band",
        "visible_input_object": "single-cell RNA-seq profile / cell gene-expression input labeled “cell 1: Gene Z Gene E ... Gene C”",
        "visible_model_interface": "cell encoder -> gene embeddings -> modality adapter -> instruction/prompt embeddings feeding the language model decoder",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the main end-to-end workflow from the input cell profile through encoding, projected gene embeddings, and the decoder interface, while cutting off the lower output-only probability/loss region.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_22e9c9cfdc9b",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "single-cell RNA-seq profiles",
          "actual_model_visible_form": "sequence embeddings projected into the LLM's input embedding space"
        }
      ],
      "routes": [
        {
          "route_id": "route_22e9c9cfdc9b",
          "configuration_id": "config_50b9e7a80491",
          "route_label": "Cell2Text-Llama-1B full fine-tuning route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "full fine-tuning for cell description generation",
          "source_object_verbatim": "single-cell RNA-seq profiles",
          "source_object_normalized": "single-cell RNA-seq expression profiles",
          "source_modality_normalized": "single-cell RNA sequencing",
          "transformation_chain_verbatim": [
            "single-cell RNA-seq profiles",
            "Geneformer encoder",
            "gene-level embeddings",
            "lightweight adapter module",
            "instruction-following prompt structure",
            "Meta-Llama-3.2-1B-Instruct decoder"
          ],
          "model_visible_form_verbatim": "sequence embeddings projected into the LLM's input embedding space",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "a lightweight adapter module projects Geneformer outputs into the language model's semantic space",
          "fusion_topology": "encoder_decoder",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We performed full fine-tuning on both Meta-Llama-3.2-1B-Instruct and Gemma3-4B-it models",
          "section_heading": "3.2.1 FULL FINE-TUNING",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            5
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/47",
            "#/texts/48",
            "#/texts/49"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001838::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001838::0001"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_000c4dd8661a"
    },
    {
      "model_id": "model_9c283f71363e",
      "model_name": "Cell2Text-Llama-1B-LoRA",
      "record_id": "full_2026-07-06__rec_001838",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4250fe0b5a77",
      "paper_title": "Cell2Text: Multimodal LLM for Generating Single-Cell Descriptions from RNA-Seq Data",
      "doi": "10.48550/arXiv.2509.24840",
      "paper_url": "https://doi.org/10.48550/arXiv.2509.24840",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 1
      },
      "subtype_counts": {
        "direct_projected_embedding": 1
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "single-cell RNA sequencing"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "encoder_decoder"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001838_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001838_bb6ae63f17ad/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of the Cell2Text framework. The model takes single-cell RNA-seq profiles as input and processes them through a pretrained Geneformer encoder to generate contextualized gene-level embeddings. These embeddings are projected into the semantic space of the language model via a lightweight adapter module, aligning biological signals with linguistic representations. A pretrained, instruction-tuned LLM decoder then generates structured natural language descriptions that capture cellular identity, tissue of origin, disease associations, and pathway activity.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow for a cell-description generation or multimodal language-model training setup.\n\nVisible elements:\n- Input biological source object: a single-cell gene expression vector labeled “cell 1: Gene Z Gene E ... Gene C”.\n- Cell metadata block lists attributes such as cell id, cell type “vein endothelial cell”, development stage “prime adult stage”, disease “normal”, sex “male”, tissue “caudate lobe of liver”, tissue general “liver”, and pathway labels including “HALLMARK_COAGULATION” and “HALLMARK_MYC_TARGETS_V2”.\n- A textual target/output caption labeled “cell description (Cp)” states that the sample consists of a vein endothelial cell and an endothelial cell that is part of a vein.\n- Transformations:\n  - Gene expression is passed through a “cell encoder”.\n  - Output becomes “gene embeddings”.\n  - Gene embeddings are passed through a “modality adapter”.\n  - The adapter maps embeddings into dimensions labeled 1152, 2048, and 3072.\n- Model interface:\n  - A sequence of embeddings is fed into a “language model decoder”.\n  - Embedding types are color-coded and labeled:\n    - green: instruction embeddings\n    - orange: gene embeddings\n    - green: prompt embeddings\n    - gray: generated/next-token-related embeddings\n- Training objective:\n  - The decoder predicts “next token probability p”.\n  - Loss is shown as `L_sft = L_CrossEntropy(Cp, p)`.\n- Finding/claim shown by the figure:\n  - The diagram illustrates supervised fine-tuning of a language model decoder to generate cell descriptions from gene-expression-derived embeddings plus instruction and prompt embeddings.",
        "page_no": 4,
        "sha256": "56abf86ae3158af55d3ac567068e0e83f0def5b306fd4a1a9599608086aabe6c",
        "pixel_width": 856,
        "pixel_height": 460,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.79,
          "height": 0.55
        },
        "panel_label": "Source RNA-seq to adapter/interface",
        "visible_input_object": "Single-cell RNA-seq profile labeled 'cell 1: Gene Z Gene E ... Gene C'",
        "visible_model_interface": "Cell encoder -> gene embeddings -> modality adapter, with the downward arrow toward the decoder input interface",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the grounded input route readable: the source gene-expression vector, the cell encoder, the gene-embedding carrier, and the modality adapter that projects into the model interface. It excludes the lower output-generation area.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_d52d8c7a5bcf",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "single-cell RNA-seq profiles",
          "actual_model_visible_form": "sequence embeddings projected into the LLM's input embedding space"
        }
      ],
      "routes": [
        {
          "route_id": "route_d52d8c7a5bcf",
          "configuration_id": "config_7b64886571d4",
          "route_label": "Cell2Text-Llama-1B-LoRA parameter-efficient fine-tuning route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "parameter-efficient fine-tuning using Low-Rank Adaptation (LoRA) for cell description generation",
          "source_object_verbatim": "single-cell RNA-seq profiles",
          "source_object_normalized": "single-cell RNA-seq expression profiles",
          "source_modality_normalized": "single-cell RNA sequencing",
          "transformation_chain_verbatim": [
            "single-cell RNA-seq profiles",
            "Geneformer encoder",
            "gene-level embeddings",
            "lightweight adapter module",
            "instruction-following prompt structure",
            "Meta-Llama-3.2-1B-Instruct decoder with LoRA on self-attention modules"
          ],
          "model_visible_form_verbatim": "sequence embeddings projected into the LLM's input embedding space",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "a lightweight adapter module projects Geneformer outputs into the language model's semantic space and LoRA selectively injects trainable low-rank matrices into the transformer architecture's attention mechanism",
          "fusion_topology": "encoder_decoder",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Given the substantial computational requirements of full fine-tuning and the specificity of our task, we additionally explored parameter-efficient fine-tuning using Low-Rank Adaptation (LoRA) (Hu et al., 2022) specifically on the Meta-Llama-3.2-1B-Instruct model. LoRA selectively injects trainable low-rank matrices into the transformer architecture's attention mechanism, significantly reducing the number of trainable parameters while maintaining a performance close to full fine-tuning. This approach allowed us to efficiently adapt the pre-trained LLM to our domain-specific task with limited computational resources, while investigating its effect on generation quality compared to the full fine-tuning approach.",
          "section_heading": "3.2.2 PARAMETER-EFFICIENT FINE-TUNING (PEFT)",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            6
          ],
          "doc_item_refs": [
            "#/texts/10",
            "#/texts/52"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001838::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001838::0001"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_000c4dd8661a"
    },
    {
      "model_id": "model_c0a097206f2c",
      "model_name": "CellHermes",
      "record_id": "full_2026-07-06__rec_001126",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_9444a1483055",
      "paper_title": "Language may be all omics needs: Harmonizing multimodal data for omics understanding with CellHermes",
      "doi": "10.1101/2025.11.07.687322",
      "paper_url": "https://doi.org/10.1101/2025.11.07.687322",
      "route_count": 14,
      "configuration_count": 14,
      "family_counts": {
        "text_native_token_stream": 13,
        "dense_continuous_carrier": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 10,
        "serialized_biological_context_or_ordered_profile": 2,
        "plain_language_prompt_or_question": 1,
        "pooled_or_aggregated_embedding": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "pooled_or_aggregated_embedding",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "graph/network",
        "tabular",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "other_explicit",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "no_text_on_this_route",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001126_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001126_3bfaf5b3be46/figure_001.png",
        "figure_index": 1,
        "caption": "Fig. 1  |  CellHermes  construction  by  integrating  different  self-supervised  learning techniques. a.  Comparative diagram of different foundation model construction between traditional  methods  and  our  method; In traditional methods, we usually take one of the",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel schematic figure labeled **a-d** describing a CellHermes-style workflow for applying/fine-tuning NLP large language models to omics and biological network data.\n\nPanel **a** compares two training strategies:\n- **Training foundation model from scratch**: single omics data are tokenized, passed into an LLM architecture, pretrained, and produce a foundation model.\n- **Fine-tuning existing NLP-LLMs (CellHermes)**: multiple omics datasets are textualized, passed into existing LLMs, and fine-tuned to produce a foundation model.\n\nPanel **b** shows emulation of self-supervised learning with question-answer pairs:\n- Source object: **gene expression data**, represented as a cell-by-gene matrix with genes labeled G1-G4.\n- Transformation: genes in cell sequences are masked using **[MASK]**.\n- Interfaces/tasks:\n  - **MLM in sequence**: prompt asks to recover masked genes from a sequence sorted by expression levels.\n  - **Autoregressive in sequence**: prompt asks to predict next gene names among highly expressed genes ordered by descent.\n- Source object: **protein-protein network**, represented as a small graph with gene/protein nodes G1-G4.\n- Transformations:\n  - **Masking node** for graph node prediction.\n  - **Masking edge** for graph link prediction.\n- Interfaces/tasks:\n  - **Mask node prediction in graph**: prompt uses masked node plus 3-hop and 2-hop neighbor lists to recover the gene.\n  - **Mask link prediction in graph**: prompt asks whether a candidate gene is connected within two hops, returning Yes/No.\n- Outputs are shown as reconstructed cells or reconstructed graphs.\n\nPanel **c** shows prompt and training design:\n- **Human-AI collaboration prompts design**: human-written prompts are expanded by AI assistance into diverse prompts.\n- **CellHermes training**: an instruction dataset containing MLM, autoregressive, mask node prediction, and mask link prediction tasks is used to instruction-tune open-source LLMs, including Llama, Qwen, and Deepseek, with LoRA, producing CellHermes.\n\nPanel **d** shows downstream model roles:\n- **LLM as an encoder**: text inputs such as “Gene BRCA1”, “Gene BRCA1 in a breast cancer cell”, and ranked gene-expression cell descriptions are mapped to gene embeddings, contextualized gene embeddings, or cell embeddings.\n- **LLM as a predictor**: CellHermes interfaces with biological databases including Perturbase, DepMap, BioGRID, CellMarker, CTD, Orphanet, and DisGeNET to predict perturbation response, cell fitness, and gene interaction.\n- **LLM as an explainer**: user provides a cell gene-expression description; CellHermes returns a natural-language explanation predicting cell state/reactivation and referencing genes/pathways visible in the schematic.\n\nVisible biological source objects include omics datasets, gene expression matrices, gene sequences, protein-protein interaction graphs, genes/proteins labeled G1-G4, BRCA1 examples, biological databases, cell states, perturbation response, cell fitness, and gene interactions.",
        "page_no": 6,
        "sha256": "6a7a715c101c144d379ccfea21fd46087cc7cf1b5a4bcae5481def0aa8fe47c8",
        "pixel_width": 846,
        "pixel_height": 1215,
        "crop_box": {
          "x": 0.02,
          "y": 0.14,
          "width": 0.8,
          "height": 0.19
        },
        "panel_label": "b",
        "visible_input_object": "gene expression data",
        "visible_model_interface": "natural-language question-answer prompt over ranked gene names",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the upper sequence-emulation section of panel b, keeping the source gene-expression matrix, the [MASK] transformation arrow, the masked cell-sequence labels, and the MLM/autoregressive instruction boxes, while excluding reconstructed outputs and the lower graph panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_c2c95046be4c",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "gene name",
          "actual_model_visible_form": "the complete gene name (e.g. BRCA1)"
        },
        {
          "subtype_id": "pooled_or_aggregated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_25b99cf078be",
          "example_input": "{gene/cell/patch embeddings}",
          "example_carrier": "mean/attention pool = one compact vector",
          "example_interface": "aggregator → generator",
          "actual_source": "a cell's gene expression profile",
          "actual_model_visible_form": "expression-weighted gene-embedding vector"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_ff6373f5dec8",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "single-cell transcriptomic profile",
          "actual_model_visible_form": "natural language prefix-to-next-token instruction-response pair over ranked gene names"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_4f41a9d21678",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "single-cell transcriptomic profile",
          "actual_model_visible_form": "natural language instruction-response pair over ranked gene names"
        }
      ],
      "routes": [
        {
          "route_id": "route_4f41a9d21678",
          "configuration_id": "config_1b70a9ab19ec",
          "route_label": "single-cell transcriptome MLM pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "masked language modeling (MLM)",
          "source_object_verbatim": "single-cell transcriptomic profile",
          "source_object_normalized": "single-cell transcriptomic profile",
          "source_modality_normalized": "tabular",
          "transformation_chain_verbatim": [
            "row-normalized (summing to 10,000) and log-normalized",
            "ranked gene expression values",
            "truncated the top 100 genes",
            "serialized into a gene-expression 'sentence'",
            "converted into question-answer pairs"
          ],
          "model_visible_form_verbatim": "natural language instruction-response pair over ranked gene names",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templates on a pretrained LLaMA-3.1-8B-Instruct backbone with LoRA",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "In MLM, the instruction specifies which tokens are masked",
          "section_heading": "Self-supervised learning (SSL) emulation instruction dataset construction",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3,
            4,
            5,
            6,
            7,
            20,
            21,
            28
          ],
          "doc_item_refs": [
            "#/texts/19",
            "#/texts/20",
            "#/texts/21",
            "#/texts/22",
            "#/texts/23",
            "#/texts/27",
            "#/texts/28",
            "#/texts/30",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84",
            "#/texts/85",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0001",
            "dense::full_2026-07-06__rec_001126::0003",
            "dense::full_2026-07-06__rec_001126::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ff6373f5dec8",
          "configuration_id": "config_e59ac3481d1d",
          "route_label": "single-cell transcriptome autoregressive pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "autoregressive prediction",
          "source_object_verbatim": "single-cell transcriptomic profile",
          "source_object_normalized": "single-cell transcriptomic profile",
          "source_modality_normalized": "tabular",
          "transformation_chain_verbatim": [
            "row-normalized (summing to 10,000) and log-normalized",
            "ranked gene expression values",
            "truncated the top 100 genes",
            "serialized into a gene-expression 'sentence'",
            "converted into question-answer pairs"
          ],
          "model_visible_form_verbatim": "natural language prefix-to-next-token instruction-response pair over ranked gene names",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "instruction templates on a pretrained LLaMA-3.1-8B-Instruct backbone with LoRA",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "In AR, the instruction is the prefix sequence",
          "section_heading": "Self-supervised learning (SSL) emulation instruction dataset construction",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3,
            4,
            5,
            6,
            7,
            20,
            21,
            28
          ],
          "doc_item_refs": [
            "#/texts/19",
            "#/texts/20",
            "#/texts/21",
            "#/texts/22",
            "#/texts/23",
            "#/texts/27",
            "#/texts/28",
            "#/texts/30",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84",
            "#/texts/85",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0001",
            "dense::full_2026-07-06__rec_001126::0003",
            "dense::full_2026-07-06__rec_001126::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_392326cd1a03",
          "configuration_id": "config_5b163fc754a4",
          "route_label": "PPI masked node pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "masked node prediction",
          "source_object_verbatim": "protein-protein interaction (PPI) network",
          "source_object_normalized": "protein-protein interaction network",
          "source_modality_normalized": "graph/network",
          "transformation_chain_verbatim": [
            "downloaded from BioGRID and filtered for Homo sapiens physical interactions",
            "represented as a graph G=(V,A)",
            "designed five template types",
            "masked a node and used its neighborhood context",
            "converted into question-answer pairs"
          ],
          "model_visible_form_verbatim": "natural language graph-context instruction with a masked node",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templates on a pretrained LLaMA-3.1-8B-Instruct backbone with LoRA",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "the instruction encodes the graph context with a masked node",
          "section_heading": "Self-supervised learning (SSL) emulation instruction dataset construction",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            20,
            21
          ],
          "doc_item_refs": [
            "#/texts/22",
            "#/texts/23",
            "#/texts/27",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84",
            "#/texts/85",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0002",
            "dense::full_2026-07-06__rec_001126::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_93a958aa7b2a",
          "configuration_id": "config_9375a57b2eac",
          "route_label": "PPI masked link pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "masked link prediction",
          "source_object_verbatim": "protein-protein interaction (PPI) network",
          "source_object_normalized": "protein-protein interaction network",
          "source_modality_normalized": "graph/network",
          "transformation_chain_verbatim": [
            "downloaded from BioGRID and filtered for Homo sapiens physical interactions",
            "represented as a graph G=(V,A)",
            "designed ten template types",
            "masked an edge and used its neighborhood context",
            "converted into question-answer pairs"
          ],
          "model_visible_form_verbatim": "natural language graph-context instruction with a missing edge",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templates on a pretrained LLaMA-3.1-8B-Instruct backbone with LoRA",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "the instruction specifies the graph neighborhood with a missing edge",
          "section_heading": "Self-supervised learning (SSL) emulation instruction dataset construction",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            20,
            21
          ],
          "doc_item_refs": [
            "#/texts/22",
            "#/texts/23",
            "#/texts/27",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84",
            "#/texts/85",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0002",
            "dense::full_2026-07-06__rec_001126::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c2c95046be4c",
          "configuration_id": "config_cecac222d356",
          "route_label": "gene embedding from gene symbol",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "gene embeddings",
          "source_object_verbatim": "gene name",
          "source_object_normalized": "gene",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "complete gene name provided as input",
            "tokenized as text",
            "final sub-token embedding extracted from the last transformer layer"
          ],
          "model_visible_form_verbatim": "the complete gene name (e.g. BRCA1)",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "extracting the final sub-token embedding from the last transformer layer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "gene embeddings are extracted by taking the final sub-token embedding from the last transformer layer",
          "section_heading": "Gene embeddings",
          "supporting_figure_or_table": "Fig. 2a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/texts/104",
            "#/texts/106",
            "#/texts/109",
            "#/texts/110",
            "#/texts/111"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0006",
            "dense::full_2026-07-06__rec_001126::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6c8f21176e6c",
          "configuration_id": "config_d01cbb37c601",
          "route_label": "cell embedding from ranked gene sentence",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "sentence-based embedding (CellHermes-s)",
          "source_object_verbatim": "a 'cell sentence' consisting of ranked gene names",
          "source_object_normalized": "ranked gene-name cell sentence",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "ranked gene names",
            "cell sentence",
            "last sub-token embedding extracted from the final transformer layer"
          ],
          "model_visible_form_verbatim": "ranked gene-name sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "extracting the last sub-token embedding from the final transformer layer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "using as input a 'cell sentence' consisting of ranked gene names (CellHermes-s)",
          "section_heading": "Embedding cells while preserving biological signals and mitigating batch effects",
          "supporting_figure_or_table": "Fig. 3a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            7,
            11,
            12,
            13,
            22,
            23,
            27
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/2",
            "#/texts/104",
            "#/texts/106",
            "#/texts/109",
            "#/texts/110",
            "#/texts/111",
            "#/texts/167",
            "#/texts/36",
            "#/texts/47",
            "#/texts/49",
            "#/texts/51"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0006",
            "dense::full_2026-07-06__rec_001126::0008"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_25b99cf078be",
          "configuration_id": "config_1ab1af327eb6",
          "route_label": "weighted gene-embedding average",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "weighted gene-embedding average (CellHermes-w)",
          "source_object_verbatim": "a cell's gene expression profile",
          "source_object_normalized": "cell gene expression profile",
          "source_modality_normalized": "tabular",
          "transformation_chain_verbatim": [
            "convert a cell into gene embeddings",
            "compute the weighted average of gene embeddings based on expression levels"
          ],
          "model_visible_form_verbatim": "expression-weighted gene-embedding vector",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "weighted average of gene embeddings based on expression levels",
          "fusion_topology": "other_explicit",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "weighted average of gene embeddings based on expression levels",
          "section_heading": "Embedding cells while preserving biological signals and mitigating batch effects",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            7,
            11,
            12,
            13,
            22,
            23,
            27
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/2",
            "#/texts/104",
            "#/texts/106",
            "#/texts/109",
            "#/texts/110",
            "#/texts/111",
            "#/texts/167",
            "#/texts/36",
            "#/texts/47",
            "#/texts/49",
            "#/texts/51"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_013"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0006",
            "dense::full_2026-07-06__rec_001126::0008"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0edf23432715",
          "configuration_id": "config_3402b100d775",
          "route_label": "cell-specific gene embeddings with cell context",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "cell-type-specific gene embeddings",
          "source_object_verbatim": "one cell's transcriptomic information",
          "source_object_normalized": "cell transcriptomic context",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "incorporated the transcriptomic information of one cell into the prompt",
            "generated gene embeddings tailored to that cell"
          ],
          "model_visible_form_verbatim": "prompt containing a cell transcriptomic context and a gene query",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "incorporating the transcriptomic information of one cell into the prompt",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "we incorporated the transcriptomic information of one cell into the prompt",
          "section_heading": "Embedding genes and constructs into cell-specific gene networks",
          "supporting_figure_or_table": "Fig. 2e",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            9,
            10,
            11,
            22,
            23
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/104",
            "#/texts/106",
            "#/texts/109",
            "#/texts/110",
            "#/texts/111",
            "#/texts/38",
            "#/texts/39",
            "#/texts/40",
            "#/texts/41",
            "#/texts/43",
            "#/texts/45"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0006",
            "dense::full_2026-07-06__rec_001126::0013"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ccfc70eb3ae8",
          "configuration_id": "config_8ef7057c2b5c",
          "route_label": "perturbation prediction instruction tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "perturbation prediction (DEG prediction and gene expression change direction prediction)",
          "source_object_verbatim": "perturbation data from Perturbase",
          "source_object_normalized": "Perturbase perturbation data",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "adopted the STAMP framework",
            "structured prediction of differentially expressed gene (DEG) outcomes",
            "gene expression change direction prediction",
            "converted into question-answer pairs"
          ],
          "model_visible_form_verbatim": "natural language question-answer pair for perturbation response",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction fine-tuning on the existing natural language-LLM with LoRA",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "For the perturbation prediction task, we adopted the STAMP framework",
          "section_heading": "CellHermes as a predictor to enable multi-task biological prediction",
          "supporting_figure_or_table": "Fig. 4c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13,
            23
          ],
          "doc_item_refs": [
            "#/texts/113",
            "#/texts/114",
            "#/texts/115",
            "#/texts/116",
            "#/texts/117",
            "#/texts/118",
            "#/texts/119",
            "#/texts/120",
            "#/texts/53",
            "#/texts/54",
            "#/texts/55"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_95763cec2e30",
          "configuration_id": "config_738a40ebc45e",
          "route_label": "cell fitness prediction instruction tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "cell fitness prediction",
          "source_object_verbatim": "cell fitness data from DepMap",
          "source_object_normalized": "DepMap-derived dataset",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "constructed from the DepMap-derived dataset",
            "evaluated unseen gene and unseen cell line scenarios",
            "converted into question-answer pairs"
          ],
          "model_visible_form_verbatim": "natural language question-answer pair for cell fitness",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction fine-tuning on the existing natural language-LLM with LoRA",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "For cell fitness prediction, we used the DepMap-derived dataset",
          "section_heading": "CellHermes as a predictor to enable multi-task biological prediction",
          "supporting_figure_or_table": "Fig. 4e",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23
          ],
          "doc_item_refs": [
            "#/texts/113",
            "#/texts/114",
            "#/texts/115",
            "#/texts/116",
            "#/texts/117",
            "#/texts/118",
            "#/texts/119",
            "#/texts/120"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_038e564e42f0",
          "configuration_id": "config_26aa01d602cb",
          "route_label": "genetic interaction classification instruction tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "genetic interaction classification",
          "source_object_verbatim": "genetic interaction examples from BioGRID",
          "source_object_normalized": "genetic interaction examples",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "evaluated performance across four different genetic interaction (GI) types",
            "converted into question-answer pairs"
          ],
          "model_visible_form_verbatim": "natural language question-answer pair for genetic interaction type",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction fine-tuning on the existing natural language-LLM with LoRA",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "For genetic interaction classification, we evaluated performance across four different genetic interaction (GI) types",
          "section_heading": "CellHermes as a predictor to enable multi-task biological prediction",
          "supporting_figure_or_table": "Fig. 4f",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            14,
            15,
            23
          ],
          "doc_item_refs": [
            "#/texts/113",
            "#/texts/114",
            "#/texts/115",
            "#/texts/116",
            "#/texts/117",
            "#/texts/118",
            "#/texts/119",
            "#/texts/120"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_010"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8f43e17eb0ce",
          "configuration_id": "config_9b3859ba3b80",
          "route_label": "gene-set classification instruction tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "gene set-based tasks (cell type, disease, phenotype, and rare disease)",
          "source_object_verbatim": "gene sets from CellMarker, CTD, Orphanet, and DisGeNET",
          "source_object_normalized": "gene sets",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "built from CellMarker, CTD, Orphanet, and DisGeNET",
            "classified gene sets according to their associated category",
            "converted into unified instruction-response format"
          ],
          "model_visible_form_verbatim": "natural language instruction-response pair for gene-set category classification",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction fine-tuning on the existing natural language-LLM with LoRA",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "For gene set-based tasks (cell type, disease, phenotype, and rare disease), the objective was to classify gene sets according to their associated category",
          "section_heading": "CellHermes as a predictor to enable multi-task biological prediction",
          "supporting_figure_or_table": "Fig. 4f",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23
          ],
          "doc_item_refs": [
            "#/texts/113",
            "#/texts/114",
            "#/texts/115",
            "#/texts/116",
            "#/texts/117",
            "#/texts/118",
            "#/texts/119",
            "#/texts/120"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_011"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001126::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_499c67ed1380",
          "configuration_id": "config_058714fbf0fa",
          "route_label": "tumor-reactivity classification instruction tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "tumor-reactivity classification",
          "source_object_verbatim": "T cell transcriptomic profiles from melanoma patients",
          "source_object_normalized": "T cell transcriptomic profiles",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "use transcriptomic profiles of more than 1,000 T cells",
            "construct instruction-label pairs",
            "fine-tune CellHermes for tumor-reactivity classification"
          ],
          "model_visible_form_verbatim": "instruction-label pair for tumor-reactivity classification",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction fine-tuning on the existing natural language-LLM with LoRA",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "we applied CellHermes to predict the tumor reactivity of T cells based on their transcriptomic profiles",
          "section_heading": "CellHermes as an explainer to provide interpretable explanations for biological discovery",
          "supporting_figure_or_table": "Fig. 5a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16,
            18,
            19
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/64",
            "#/texts/65",
            "#/texts/66",
            "#/texts/69",
            "#/texts/70",
            "#/texts/71"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_014"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_26c3e6a61748",
          "configuration_id": "config_2c546774b81a",
          "route_label": "tumor-reactivity explanation prompt",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "text-based reasoning explanation for tumor-reactivity predictions",
          "source_object_verbatim": "the original transcriptome and the predicted phenotype",
          "source_object_normalized": "original transcriptome plus predicted phenotype",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "provide the original transcriptome and predicted phenotype",
            "prompt CellHermes to generate a reasoning-based explanation"
          ],
          "model_visible_form_verbatim": "prompt built from the original transcriptome and predicted phenotype",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "a prompt is constructed using both the original transcriptome and the predicted phenotype",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "a prompt is constructed using both the original transcriptome and the predicted phenotype",
          "section_heading": "CellHermes as an explainer to provide interpretable explanations for biological discovery",
          "supporting_figure_or_table": "Fig. 5b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            18,
            19
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/70",
            "#/texts/71"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001126::route_015"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_9337e69362f7"
    },
    {
      "model_id": "model_8b27781d1156",
      "model_name": "CellTosg2Sequence",
      "record_id": "july_update_2026-07-06__rec_000060",
      "collection_batch_id": "july_update_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_c16cbcac015c",
      "paper_title": "CellTosg2Sequence: A Unified Text-Omics-Signaling-Graph Large Language Model for Single-Cell Analysis",
      "doi": "10.64898/2026.06.16.732397",
      "paper_url": "https://doi.org/10.64898/2026.06.16.732397",
      "route_count": 15,
      "configuration_count": 12,
      "family_counts": {
        "text_native_token_stream": 13,
        "dense_continuous_carrier": 2
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 7,
        "virtual_token_prefix": 2,
        "structured_biological_prompt_or_task_scaffold": 4,
        "plain_language_prompt_or_question": 2
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold",
        "virtual_token_prefix"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "graph/network",
        "mixed",
        "single-cell RNA-seq",
        "single-cell gene expression",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "prefix",
        "shared_latent_alignment",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "metadata_or_context",
        "no_text_on_this_route",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/july_update_2026_07_06_rec_000060_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/july_update_2026_07_06_rec_000060_d8977b6d9a26/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: CellTosg2Sequence overall workflow. Raw scRNA-seq expression profiles are ranked into a Cell Sentence (topK =200 HGNC gene symbols in descending expression order). A heterogeneous biomedical KG assembled from 11 public databases is encoded by a 2-layer R-GCN with RotatE scoring and projected into 5120-dim virtual tokens via the KGProjector MLP. The virtual tokens are prepended to the textual prompt (cell sentence + metadata + task instruction) and fed into the Qwen2.5-32B-Instruct backbone. The same checkpoint handles cell-type annotation (CTA), perturbation prediction (CPA), and gene-perturbation significance (PSA) through a unified text-in / text-out interface.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a workflow diagram titled “CellTosg2Sequence Overall WorkFlow.” It contains multiple labeled panels connected by arrows.\n\nVisible components:\n- “Raw RNAseq Data”: a cartoon RNA/database icon and an expression profile heatmap with rows labeled Cell 1, Cell 2, Cell 3, ..., Cell N.\n- “LLM Agent Quality Control & Filter”: database/server icons and an OpenAI-style logo, indicating LLM-based filtering or quality control.\n- “Cell Sentence Arrangement in Prompt”: a prompt-like box showing gene tokens such as “Gene A: 3, Gene B:0.9, Gene C:5...” and “TOP k Ranking In Descending Order,” followed by ordered genes like “Gene C, Gene A, Gene B...”.\n- “Biological Prior Knowledge in 11 entities from Public Databases”: lists sources including Protein, UniProt, HGNC, STRING v12, Pathway, Reactome, TRUST, Drug, DrugBank, ChEMBL, Gene, HGNC, Disease, DisGeNET + Open Targets, TF, DoRothEA + TRRUST.\n- Protein/embedding panel: mentions “mean-pooling the residue embeddings of the frozen ESM-2 150M to obtain Per-Gene 640-dimensional priors.”\n- Graph model training panel: labeled “Graph training on R-GCN encoder + RotatE decoder,” with “GraphMessage Passing:R-GCN” and “RotatE:decoder + scorer.”\n- Knowledge graph panel: “Rich Knowledge Graph with Neighborhood Attention and Informative prior,” shown as a colored node-link graph.\n- “Knowledge Graph Project into LLM as Virtual Token”: neural-network style diagram with input nodes Vi1–Vi4 and output tokens Ti1–Ti3, with “Feed in Nodes.”\n- “Example of Original Embedding Style”: a prompt box with tasks including CTA and PSA, using gene-sequence-like colored gene tokens.\n- “Virtual Embedding Injection”: arrow from the virtual token projection into the final prompt.\n- “Final Prompt”: large prompt box containing task examples:\n  - CPA: post-perturbation cell type question.\n  - PSA: asks whether perturbing Gene A in K562 cells causes significant changes in Gene B expression.\n  - CTA: asks about presence of IGHM as a marker for B cells.\n  - Answers shown include gene sequences, “Up regulated,” and “B cells.”\n\nBiological source objects include RNA-seq expression profiles, cells, genes, proteins, pathways, drugs, diseases, transcription factors, and public biological knowledge databases. The figure describes transforming raw RNA-seq data into ranked gene-token sequences, integrating biological priors from databases and protein embeddings, training a relational graph model, projecting graph information into LLM virtual tokens, and injecting those embeddings into prompts for cell and perturbation-related prediction tasks.",
        "page_no": 5,
        "sha256": "cef5f4e670e4c7134da69a4020f641c7f5c300f93fb39f12d26cf27d4e2f182b",
        "pixel_width": 860,
        "pixel_height": 559,
        "crop_box": {
          "x": 0,
          "y": 0.07,
          "width": 0.7,
          "height": 0.26
        },
        "panel_label": "Raw scRNA-seq to Cell Sentence",
        "visible_input_object": "Raw RNAseq Data / normalized scRNA-seq expression profile",
        "visible_model_interface": "Cell Sentence Arrangement in Prompt with ranked gene tokens",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the source profile heatmap, the QC/filter step, the arrow into the prompt, and the ranked gene-list box, which together show the input route from scRNA-seq to the model-visible cell sentence. It excludes the output/example panels and unrelated lower-half modules.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_4812cf2c6dd3",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "task instruction string",
          "actual_model_visible_form": "task instruction text"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_237ac9359686",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "normalised single-cell expression profile",
          "actual_model_visible_form": "ranked top-200 HGNC gene symbols"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_59d07de37bae",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "biological metadata (tissue, compartment, disease)",
          "actual_model_visible_form": "natural-language metadata fields"
        },
        {
          "subtype_id": "virtual_token_prefix",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_32f7f6418b4f",
          "example_input": "cell embedding + prompt",
          "example_carrier": "<bio₁> <bio₂> ... <bioₖ> [prompt tokens]",
          "example_interface": "soft prefix → LLM stream",
          "actual_source": "curated heterogeneous biomedical KG",
          "actual_model_visible_form": "compact virtual tokens in the LM hidden space"
        }
      ],
      "routes": [
        {
          "route_id": "route_237ac9359686",
          "configuration_id": "config_f866c60e00af",
          "route_label": "Raw scRNA-seq to Cell Sentence",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Cell Sentence Representation and Prompt Format",
          "source_object_verbatim": "normalised single-cell expression profile",
          "source_object_normalized": "normalised single-cell expression profile",
          "source_modality_normalized": "single-cell RNA-seq",
          "transformation_chain_verbatim": [
            "quality control",
            "log-normalisation to library size 10^4",
            "rank genes by expression magnitude",
            "keep the topK (K=200) HGNC symbols"
          ],
          "model_visible_form_verbatim": "ranked top-200 HGNC gene symbols",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "[CellSentence] block in the prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Prompt format. Each prompt has four blocks: (1) KG virtual tokens [KG_VT] : one virtual token per input gene (block length K =200 ), aligned one-to-one with the Cell Sentence; (2) Biological metadata [Meta] : short natural-language fields (tissue, compartment, disease, optional KG triples); (3) Cell Sentence [CellSentence] : the ranked gene list as HGNC symbols; (4) Task instruction [Task] : one short line specifying the output format. The same skeleton serves classification, free-generation annotation, and drug-response tasks through the same interface.",
          "section_heading": "3.1 Cell Sentence Representation and Prompt Format",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6
          ],
          "doc_item_refs": [
            "#/texts/34",
            "#/texts/35",
            "#/texts/36",
            "#/texts/37"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000060::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_32f7f6418b4f",
          "configuration_id": "config_71003106a075",
          "route_label": "Biomedical KG to virtual tokens",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Knowledge Graph Construction and Virtual-Token Injection",
          "source_object_verbatim": "curated heterogeneous biomedical KG",
          "source_object_normalized": "curated heterogeneous biomedical knowledge graph",
          "source_modality_normalized": "graph/network",
          "transformation_chain_verbatim": [
            "merge public databases",
            "2-layer R-GCN encoding",
            "RotatE triple scoring",
            "ESM-2 prior warm-start",
            "KGProjector MLP"
          ],
          "model_visible_form_verbatim": "compact virtual tokens in the LM hidden space",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "virtual_token_prefix",
          "insertion_or_fusion_verbatim": "prepended to each cell sentence as [KG_VT] prefix tokens",
          "fusion_topology": "prefix",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "compact virtual tokens",
          "section_heading": "3.3 Knowledge Graph Construction and Virtual-Token Injection",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9,
            10
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/160",
            "#/texts/163",
            "#/texts/166",
            "#/texts/167"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000060::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_59d07de37bae",
          "configuration_id": "config_b98b227a684d",
          "route_label": "Cell-state metadata in prompt",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Cell Sentence Representation and Prompt Format",
          "source_object_verbatim": "biological metadata (tissue, compartment, disease)",
          "source_object_normalized": "biological metadata",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "formatted as short natural-language fields"
          ],
          "model_visible_form_verbatim": "natural-language metadata fields",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "[Meta] block",
          "fusion_topology": "concatenation",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Biological metadata",
          "section_heading": "3.1 Cell Sentence Representation and Prompt Format",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6
          ],
          "doc_item_refs": [
            "#/texts/34",
            "#/texts/35",
            "#/texts/36",
            "#/texts/37"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000060::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4812cf2c6dd3",
          "configuration_id": "config_b98b227a684d",
          "route_label": "Task instruction prompt",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Cell Sentence Representation and Prompt Format",
          "source_object_verbatim": "task instruction string",
          "source_object_normalized": "task instruction string",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "rendered as a short output-format instruction"
          ],
          "model_visible_form_verbatim": "task instruction text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "[Task] block",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Task instruction",
          "section_heading": "3.1 Cell Sentence Representation and Prompt Format",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6
          ],
          "doc_item_refs": [
            "#/texts/34",
            "#/texts/35",
            "#/texts/36",
            "#/texts/37"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000060::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0687e285a7ee",
          "configuration_id": "config_5a979383a857",
          "route_label": "Masked Cell Sentence input",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Masked Gene Prediction (T0b)",
          "source_object_verbatim": "Cell Sentence",
          "source_object_normalized": "Cell Sentence",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "replace about 15% of gene symbols with <mask> tokens"
          ],
          "model_visible_form_verbatim": "masked Cell Sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "sequence-level masked-gene input",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "T0b is a sequence-prior task: given a Cell Sentence with ∼ 15% of gene symbols replaced by <mask> tokens, the model generates the masked gene identities from the surrounding ranked-gene context. This task tests co-occurrence statistics and sequence-level gene co-expression priors rather than phenotype-level reasoning-the complementary domain in which the KG channel provides the most gains. Performance is measured by positional accuracy, set precision/recall/F1, and perplexity. Results confirm that the three-stage training preserves strong gene-sequence repre-",
          "section_heading": "5.2 Masked Gene Prediction (T0b)",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            20,
            21
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/texts/560",
            "#/texts/564",
            "#/texts/654"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000060::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9c4534f3c5a3",
          "configuration_id": "config_a51477288467",
          "route_label": "Perturbation metadata prompt",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Task D1: Drug-Response Prediction (Tahoe-100M)",
          "source_object_verbatim": "drug condition metadata",
          "source_object_normalized": "drug condition metadata",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "formatted into [Drug] fields",
            "includes Name, Dose, and Target"
          ],
          "model_visible_form_verbatim": "drug name, dose, and target text fields",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "[Drug] block in the prompt",
          "fusion_topology": "concatenation",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "[Drug] Name: Nutlin -3a",
          "section_heading": "I Example Prompts",
          "supporting_figure_or_table": "Figure 9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            42
          ],
          "doc_item_refs": [
            "#/texts/929"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000060::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_649acb378557",
          "configuration_id": "config_a51477288467",
          "route_label": "KG triples in prompt",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Task D1: Drug-Response Prediction (Tahoe-100M)",
          "source_object_verbatim": "KG triples",
          "source_object_normalized": "KG triples",
          "source_modality_normalized": "graph/network",
          "transformation_chain_verbatim": [
            "textualized as relationship triples in the prompt"
          ],
          "model_visible_form_verbatim": "text triples",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "[KG Triples] block",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "[KG Triples]",
          "section_heading": "I Example Prompts",
          "supporting_figure_or_table": "Figure 9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            42
          ],
          "doc_item_refs": [
            "#/texts/929"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000060::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_72245eef6feb",
          "configuration_id": "config_a51477288467",
          "route_label": "Drug-target virtual token",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Task D1: Drug-Response Prediction (Tahoe-100M)",
          "source_object_verbatim": "drug target MDM2",
          "source_object_normalized": "drug target MDM2",
          "source_modality_normalized": "graph/network",
          "transformation_chain_verbatim": [
            "KG embedding",
            "projected into a drug-target virtual token"
          ],
          "model_visible_form_verbatim": "<vt_MDM2>",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "virtual_token_prefix",
          "insertion_or_fusion_verbatim": "[Drug_Target_VT] token",
          "fusion_topology": "prefix",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "[Drug_Target_VT] <vt_MDM2>",
          "section_heading": "I Example Prompts",
          "supporting_figure_or_table": "Figure 9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            42
          ],
          "doc_item_refs": [
            "#/texts/929"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000060::route_008"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4a446a857f1e",
          "configuration_id": "config_f3ec4678f01d",
          "route_label": "Perturbation meta-labels prompt",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Example of single cell Perturbation LLM Query",
          "source_object_verbatim": "meta-labels (drug, CRISPR target, cell line, tissue)",
          "source_object_normalized": "meta-labels (drug, CRISPR target, cell line, tissue)",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "preprocessed into metadata fields",
            "merged with virtual KG tokens into a structured prompt"
          ],
          "model_visible_form_verbatim": "natural-language perturbation metadata fields",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "[Meta] block in the prompt",
          "fusion_topology": "concatenation",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "meta-labels (drug, CRISPR target, cell line, tissue)",
          "section_heading": "I Example Prompts",
          "supporting_figure_or_table": "Figure 9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            41
          ],
          "doc_item_refs": [
            "#/pictures/8",
            "#/texts/870"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000060::route_010"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_065c8bd53fce",
          "configuration_id": "config_6efefda4295d",
          "route_label": "Label-string alignment input",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Stage-II KG-anchor training",
          "source_object_verbatim": "label string",
          "source_object_normalized": "label string",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "encoded by the same model pathway used for KG-anchor alignment",
            "mean-pooled last-layer hidden state is projected toward frozen label-string embeddings"
          ],
          "model_visible_form_verbatim": "label string text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "parallel label-string pathway in Stage-II training",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Label String 'T cell'",
          "section_heading": "Training Process for Enhancing Attention of Label token toward KG graph token",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            14,
            15
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/texts/261",
            "#/texts/299",
            "#/texts/300",
            "#/texts/301",
            "#/texts/302",
            "#/texts/305",
            "#/texts/306"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000060::route_011"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b0f2d87800d6",
          "configuration_id": "config_62ac337f6f02",
          "route_label": "HCA train split cell sentence",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Cells with h ≤ 7 go to train",
          "source_object_verbatim": "hca_train",
          "source_object_normalized": "HCA train split",
          "source_modality_normalized": "single-cell RNA-seq",
          "transformation_chain_verbatim": [
            "quality control (min. 500 genes per cell, < 20% mitochondrial reads)",
            "log-normalisation to library size 10^4",
            "conversion to ranked Cell Sentences of topK =200 HGNC gene symbols",
            "deterministic hash-based partition",
            "h ≤ 7 go to train"
          ],
          "model_visible_form_verbatim": "ranked Cell Sentences of topK =200 HGNC gene symbols",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conversion to ranked Cell Sentences of topK =200 HGNC gene symbols",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Cells with h ≤ 7 go to train",
          "section_heading": "3.2.1 Single-cell expression corpus",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7
          ],
          "doc_item_refs": [
            "#/texts/44"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000060::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_55a22722eb6a",
          "configuration_id": "config_84d290b9d158",
          "route_label": "HCA validation split cell sentence",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "h =8 to validation",
          "source_object_verbatim": "hca_val",
          "source_object_normalized": "HCA validation split",
          "source_modality_normalized": "single-cell RNA-seq",
          "transformation_chain_verbatim": [
            "quality control (min. 500 genes per cell, < 20% mitochondrial reads)",
            "log-normalisation to library size 10^4",
            "conversion to ranked Cell Sentences of topK =200 HGNC gene symbols",
            "deterministic hash-based partition",
            "h =8 to validation"
          ],
          "model_visible_form_verbatim": "ranked Cell Sentences of topK =200 HGNC gene symbols",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conversion to ranked Cell Sentences of topK =200 HGNC gene symbols",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "h =8 to validation",
          "section_heading": "3.2.1 Single-cell expression corpus",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7
          ],
          "doc_item_refs": [
            "#/texts/44"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000060::0008"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1640cf949003",
          "configuration_id": "config_16d90abf83e6",
          "route_label": "HCA test split cell sentence",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cells with h ≤ 7 go to train, h =8 to validation, and h =9 to test.",
          "source_object_verbatim": "hca_test",
          "source_object_normalized": "HCA test split",
          "source_modality_normalized": "single-cell RNA-seq",
          "transformation_chain_verbatim": [
            "min. 500 genes per cell",
            "< 20% mitochondrial reads",
            "log-normalisation to library size 10^4",
            "conversion to ranked Cell Sentences of topK =200 HGNC gene symbols",
            "deterministic hash-based partition"
          ],
          "model_visible_form_verbatim": "ranked Cell Sentences of topK =200 HGNC gene symbols",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conversion to ranked Cell Sentences of topK =200 HGNC gene symbols",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "h =9 to test",
          "section_heading": "3.2.1 Single-cell expression corpus",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7
          ],
          "doc_item_refs": [
            "#/texts/44"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000060::0009"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f8d016f6aacd",
          "configuration_id": "config_524a6d1d5baa",
          "route_label": "Task B2 tissue prediction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Given a cell's gene expression profile, predict which tissue it came from (70 possible tissues, e.g., brain, blood, lung).",
          "source_object_verbatim": "a cell's gene expression profile",
          "source_object_normalized": "cell gene expression profile",
          "source_modality_normalized": "single-cell gene expression",
          "transformation_chain_verbatim": [
            "the top-200 ranked genes of a cell",
            "predict which tissue it came from"
          ],
          "model_visible_form_verbatim": "the top-200 ranked genes of a cell",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "The model reads the top-200 ranked genes of a cell",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Given a cell's gene expression profile, predict which tissue it came from (70 possible tissues, e.g., brain, blood, lung).",
          "section_heading": "4.2 Task B2: Tissue Prediction",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            17
          ],
          "doc_item_refs": [
            "#/texts/448"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000060::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1089dd088ca1",
          "configuration_id": "config_e1ed7e5ce494",
          "route_label": "Task B3 disease prediction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Given a cell's gene profile and tissue context (but without the disease label in the prompt), predict the disease state from 254 conditions-healthy cells are labelled 'normal.'",
          "source_object_verbatim": "a cell's gene profile and tissue context",
          "source_object_normalized": "cell gene profile plus tissue context",
          "source_modality_normalized": "mixed",
          "transformation_chain_verbatim": [
            "top-200 ranked genes of a cell",
            "tissue context in the prompt",
            "predict the disease state from 254 conditions"
          ],
          "model_visible_form_verbatim": "a cell's gene profile and tissue context",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "Given a cell's gene profile and tissue context (but without the disease label in the prompt)",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Given a cell's gene profile and tissue context (but without the disease label in the prompt), predict the disease state from 254 conditions",
          "section_heading": "4.3 Task B3: Disease Prediction",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            17
          ],
          "doc_item_refs": [
            "#/texts/450"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000060::0011"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_acce00dbb33d"
    },
    {
      "model_id": "model_68562cc69016",
      "model_name": "CHATCELL",
      "record_id": "full_2026-07-06__rec_002517",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_cb14749fff1d",
      "paper_title": "ChatCell: Facilitating Single-Cell Analysis with Natural Language",
      "doi": "10.48550/arXiv.2402.08303",
      "paper_url": "https://doi.org/10.48550/arXiv.2402.08303",
      "route_count": 6,
      "configuration_count": 4,
      "family_counts": {
        "text_native_token_stream": 6
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 1,
        "structured_biological_prompt_or_task_scaffold": 4,
        "serialized_biological_context_or_ordered_profile": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_002517_figure_003.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_002517_46d135b6aad4/figure_003.png",
        "figure_index": 3,
        "caption": "Figure 1: CHATCELL facilitates single-cell analysis through conversational interactions. Black and red texts denote human and single-cell language, respectively.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic scientific/biomedical figure showing a conversational model interface applied to single-cell gene expression data.\n\nVisible elements:\n- A chat-style exchange between a user and an AI assistant.\n- The user asks for the top 100 primary genes in a randomly selected cell, sorted by expression level.\n- The assistant returns a red gene list beginning with examples such as `GM42418`, `IL1RAPL1`, `GPHN`, `CDK8`, `LARS2`, `JARID2`, `ZC3H7A`, `FGFR2`, `CAMK1D`, `HEXB`, `MALAT1`, `MACF1`, `TANC1`, `AHNAK`.\n- The user asks what the most likely cell type is based on the gene profile.\n- The assistant answers: “Endothelial.”\n- The user then asks for the 100 genes with the highest expression levels in another endothelial cell, ordered highest to lowest.\n- The assistant returns another red gene list with similar gene names.\n\nBiological source objects:\n- Stylized red blood/cell icons on the left, suggesting single-cell samples.\n- A large illustrated cell on the right, representing an endothelial cell or cell-type schematic.\n- Gene-expression profiles are shown as text lists rather than plots.\n\nModel/interface elements:\n- Human user icons and AI/chatbot icons.\n- Chat bubbles connected visually to a highlighted cell illustration.\n- The figure represents a workflow where a language-model-like interface queries, classifies, and retrieves genes from single-cell expression data.\n\nTransformations/findings:\n- A single-cell gene-expression vector or ranked gene list is used to infer cell type.\n- The inferred cell type shown is endothelial.\n- A second query retrieves highly expressed genes from another endothelial cell.\n- No quantitative axes, legends, panel letters, or statistical values are visible.",
        "page_no": 1,
        "sha256": "aa05c8f6b5e606739ece432e8e4361b1404957c4c3d388bedf561b1fad469526",
        "pixel_width": 436,
        "pixel_height": 348,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 1,
          "height": 0.78
        },
        "panel_label": "Figure 1",
        "visible_input_object": "Chat prompt plus ranked gene-expression list for a single cell, with the linked endothelial cell schematic on the right",
        "visible_model_interface": "Conversational chat bubbles showing the model receiving gene-expression text and returning a cell-type answer",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the first full question-answer exchange, the readable gene list, the endothelial classification reply, and the connecting schematic/arrow structure. It excludes the lower repeated example while preserving one grounded cell-type-annotation route.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_e06d358227d9",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "a prompt X",
          "actual_model_visible_form": "instruction text"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_b4eb31a00390",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "a gene sequence",
          "actual_model_visible_form": "instruction text"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_218bfcdb85c8",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "a prompt X requesting a cell sentence for a given cell type",
          "actual_model_visible_form": "instruction text"
        }
      ],
      "routes": [
        {
          "route_id": "route_e06d358227d9",
          "configuration_id": "config_7b7b33de4f88",
          "route_label": "random cell sentence generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Random Cell Sentence Generation",
          "source_object_verbatim": "a prompt X",
          "source_object_normalized": "a prompt requesting random cell sentence generation",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "instruction text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "fed directly as instructions",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Given a prompt X , the model is instructed to generate a cell sentence at random.",
          "section_heading": "A Single-cell Analysis Tasks",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            5,
            12
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/12",
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/152",
            "#/texts/153",
            "#/texts/154",
            "#/texts/155",
            "#/texts/156",
            "#/texts/16",
            "#/texts/166",
            "#/texts/57",
            "#/texts/58"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002517::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002517::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_218bfcdb85c8",
          "configuration_id": "config_d712584c4503",
          "route_label": "pseudo-cell generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Pseudo-cell Generation",
          "source_object_verbatim": "a prompt X requesting a cell sentence for a given cell type",
          "source_object_normalized": "a prompt requesting a cell sentence for a given cell type",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "instruction text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed directly as instructions",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "the prompt X requests the model to construct a cell sentence for a given cell type",
          "section_heading": "A Single-cell Analysis Tasks",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            12
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/12",
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/152",
            "#/texts/153",
            "#/texts/154",
            "#/texts/155",
            "#/texts/156",
            "#/texts/16"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002517::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002517::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b4eb31a00390",
          "configuration_id": "config_3ddc919c3776",
          "route_label": "cell type annotation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cell Type Annotation",
          "source_object_verbatim": "a gene sequence",
          "source_object_normalized": "a gene sequence",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "instruction text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "fed directly as instructions into the model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Cell Type Annotation. For cell type annotation, the model is tasked with precisely classifying cells into their respective types based on gene expression patterns encapsulated in cell sentences. Here, the prompt X involves providing a gene sequence for the model to determine the cell type, with the target Y being the accurate identification and classification of that cell type. This task is fundamental for understanding cellular functions and interactions within tissues and organs, playing a crucial role in developmental biology and regenerative medicine.",
          "section_heading": "A Single-cell Analysis Tasks",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            12
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/12",
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/152",
            "#/texts/153",
            "#/texts/154",
            "#/texts/155",
            "#/texts/156",
            "#/texts/16"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002517::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002517::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ba946f8587c7",
          "configuration_id": "config_0e811051c918",
          "route_label": "drug sensitivity prediction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Drug Sensitivity Prediction",
          "source_object_verbatim": "a cell sentence along with a specific drug",
          "source_object_normalized": "a cell sentence along with a specific drug",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "instruction text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed directly as instructions into the model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The prompt X presents a cell sentence along with a specific drug",
          "section_heading": "A Single-cell Analysis Tasks",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            12
          ],
          "doc_item_refs": [
            "#/texts/152",
            "#/texts/153",
            "#/texts/154",
            "#/texts/155",
            "#/texts/156"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002517::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b34f6289051f",
          "configuration_id": "config_0e811051c918",
          "route_label": "drug sensitivity prediction with seen task descriptions",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Drug Sensitivity Prediction",
          "source_object_verbatim": "a task description seen in training, a cell sentence, and a specific drug",
          "source_object_normalized": "a seen task description, a cell sentence, and a specific drug",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "instruction text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed directly as instructions into the model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "CHATCELL demonstrates robustness to novel phrasing. Given the diversity of human expression, maintaining robust performance across variously phrased instructions is essential for a chat model. Hence, we replace all task descriptions in the test set with variations that the model has not previously encountered during training or validation, to assess its proficiency with unseen instructions. Figure 5 illustrates the results for the drug sensitivity prediction. Despite the novel phrasing of instructions, the model's performance remains almost unchanged. This highlights the model's robustness and generalization ability, effectively handling tasks with varied descriptions and enhancing conversational single-cell analysis.",
          "section_heading": "5 Further Analysis",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            7
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/77",
            "#/texts/78",
            "#/texts/79",
            "#/texts/80"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002517::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e96a31addae7",
          "configuration_id": "config_0e811051c918",
          "route_label": "drug sensitivity prediction with unseen task descriptions",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Drug Sensitivity Prediction",
          "source_object_verbatim": "a paraphrased task description unseen during training, a cell sentence, and a specific drug",
          "source_object_normalized": "an unseen paraphrased task description, a cell sentence, and a specific drug",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "instruction text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed directly as instructions into the model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "CHATCELL demonstrates robustness to novel phrasing. Given the diversity of human expression, maintaining robust performance across variously phrased instructions is essential for a chat model. Hence, we replace all task descriptions in the test set with variations that the model has not previously encountered during training or validation, to assess its proficiency with unseen instructions. Figure 5 illustrates the results for the drug sensitivity prediction. Despite the novel phrasing of instructions, the model's performance remains almost unchanged. This highlights the model's robustness and generalization ability, effectively handling tasks with varied descriptions and enhancing conversational single-cell analysis.",
          "section_heading": "5 Further Analysis",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            7
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/77",
            "#/texts/78",
            "#/texts/79",
            "#/texts/80"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002517::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_67d200027193"
    },
    {
      "model_id": "model_6f7f8cee95d1",
      "model_name": "ChatNT",
      "record_id": "full_2026-07-06__rec_000771",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_35ae37a7f86c",
      "paper_title": "A multimodal conversational agent for DNA, RNA and protein tasks",
      "doi": "10.1038/s42256-025-01047-1",
      "paper_url": "https://doi.org/10.1038/s42256-025-01047-1",
      "route_count": 4,
      "configuration_count": 4,
      "family_counts": {
        "dense_continuous_carrier": 4
      },
      "subtype_counts": {
        "connector_mediated_embedding": 4
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "connector_mediated_embedding"
      ],
      "primary_subtype": "connector_mediated_embedding",
      "modalities": [
        "DNA"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "placeholder_replacement"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_000771_figure_003.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_000771_82b0d7e83c3d/figure_003.png",
        "figure_index": 3,
        "caption": "Fig. 1 | ChatNT, a conversational agent that can be prompted to solve a variety of biological tasks. a , An illustration of the different categories of downstream tasks included during training. UTR, untranslated region. b , Statistics on the number of English and DNA tokens available for each task in our genomics instructions dataset. English question-answer instructions are tokenized with the LLaMA tokenizer 30 , while DNA sequences are tokenized using the Nucleotide Transformer tokenizer 15 . c , The ChatNT approach to build a multimodal and multitask genomics AI system. The ChatNT conversational agent can be",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific figure about ChatNT, an AI assistant/model for biological sequences.\n\nPanel a:\n- Schematic overview connecting genomic/sequence regulatory elements and downstream biological prediction tasks.\n- Visible biological source objects include chromatin with histones, an enhancer, lncRNA, DNA accessibility, promoter, UTRs, exons, introns, splice sites, polyadenylation/PolyA, RNA-like sequence, protein-like folded structure, and DNA double helix.\n- Task labels shown around the sequence include protein fitness, protein melting, RNA degradation, RNA stability, and DNA methylation.\n- A chatbot icon is shown with the text: “Hello, I am ChatNT, an AI assistant that can handle biological sequences. How can I help you?”\n\nPanel b:\n- Horizontal bar chart titled “Number of tokens per task category.”\n- Compares English tokens and DNA tokens across task categories.\n- Categories visible: Promoter activity, Expression variants, Splice sites, lncRNA, Enhancer activity, Polyadenylation, Promoters, Protein fitness, Protein melting, Enhancers, Histones, Accessibility, RNA degradation, DNA methylation.\n- X-axis is “Number of tokens (million).”\n- Notable large DNA-token counts include enhancer activity, polyadenylation, enhancers, and histones.\n\nPanel c:\n- Diagram titled “ChatNT architecture.”\n- Shows nucleotide input tokens such as ACT, TCA, CGT, TAC, GAG entering a DNA encoder.\n- Output flows through a DNA resampler into a compact token/interface representation, then into an English language model.\n- A prompt box asks: “Determine the degradation rate of the human RNA sequence @myseq.fna on a scale from -5 to 5.”\n- The model output box states: “The degradation rate for this sequence is 1.83.”\n- The transformation/interface depicted is DNA sequence encoding/resampling into an English-language-model-compatible representation for biological sequence reasoning.",
        "page_no": 3,
        "sha256": "3f493252148b2d96d35d9f09a55aa5e024b8258c0cc959999cb24a8d41ffa94b",
        "pixel_width": 1043,
        "pixel_height": 849,
        "crop_box": {
          "x": 0.379,
          "y": 0.668,
          "width": 0.463,
          "height": 0.33
        },
        "panel_label": "c",
        "visible_input_object": "DNA sequence tokens (ACT/TCA/CGT/TAC/GAG)",
        "visible_model_interface": "DNA encoder -> DNA resampler -> compact token strip -> English language model",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops panel c’s grounded input path only: DNA token inputs, the encoder/resampler transformation, the insertion/interface strip, and the English language model. It excludes the output-only answer box and unrelated panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "connector_mediated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_12e86a771664",
          "example_input": "image / omics encoder states",
          "example_carrier": "Q-Former or adapter query vectors",
          "example_interface": "connector → LLM cross-modal interface",
          "actual_source": "DNA sequence",
          "actual_model_visible_form": "K resampled DNA embedding vectors"
        }
      ],
      "routes": [
        {
          "route_id": "route_12e86a771664",
          "configuration_id": "config_3b68c196f02e",
          "route_label": "DNA sequence genomics route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "genomics tasks framed in English",
          "source_object_verbatim": "DNA sequence",
          "source_object_normalized": "DNA sequence",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "DNA sequence tokenization into 6-mers",
            "DNA encoder produces DNA token embeddings",
            "dense neural network projection into the English word dimension",
            "Perceiver resampler with question-aware cross-attention resamples the DNA tokens",
            "resampled DNA embeddings are inserted in place of the DNA sequence placeholder tokens in the English input sequence"
          ],
          "model_visible_form_verbatim": "K resampled DNA embedding vectors",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "inserted in place of the DNA sequence placeholder tokens in the English input sequence",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Meanwhile, the English prompt is tokenized and English tokens embeddings are produced for each tokens. The K resampled DNA embedding vectors are then inserted in place of the DNA sequence placeholder tokens in the English input sequence. In the case of multiple input DNA sequences, these operations are applied consecutively and independently for each DNA sequence. We experimented with several values of K in practice and observed that low values such as 1 or 4 are not enough for the DNA encoder to impact the behaviour of the frozen English decoder. We found K = 64 to provide a good trade-off between the input length of the English decoder and the performance in practice.",
          "section_heading": "ChatNT model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/721",
            "#/texts/726",
            "#/texts/727",
            "#/texts/728"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000771::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9fe977bd9211",
          "configuration_id": "config_addd7d464405",
          "route_label": "RNA task route via corresponding DNA",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "RNA polyadenylation and degradation tasks",
          "source_object_verbatim": "corresponding DNA sequence",
          "source_object_normalized": "corresponding DNA sequence",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "DNA sequence tokenization into 6-mers",
            "DNA encoder produces DNA token embeddings",
            "dense neural network projection into the English word dimension",
            "Perceiver resampler with question-aware cross-attention resamples the DNA tokens",
            "resampled DNA embeddings are inserted in place of the DNA sequence placeholder tokens in the English input sequence"
          ],
          "model_visible_form_verbatim": "K resampled DNA embedding vectors",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "inserted in place of the DNA sequence placeholder tokens in the English input sequence",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Meanwhile, the English prompt is tokenized and English tokens embeddings are produced for each tokens. The K resampled DNA embedding vectors are then inserted in place of the DNA sequence placeholder tokens in the English input sequence. In the case of multiple input DNA sequences, these operations are applied consecutively and independently for each DNA sequence. We experimented with several values of K in practice and observed that low values such as 1 or 4 are not enough for the DNA encoder to impact the behaviour of the frozen English decoder. We found K = 64 to provide a good trade-off between the input length of the English decoder and the performance in practice.",
          "section_heading": "ChatNT solves transcriptomics and proteomics tasks",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/721",
            "#/texts/726",
            "#/texts/727",
            "#/texts/728"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000771::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3c3a5f5b5f89",
          "configuration_id": "config_c9d5acdde76d",
          "route_label": "Protein task route via CDS",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "protein tasks",
          "source_object_verbatim": "coding DNA sequence (CDS)",
          "source_object_normalized": "coding DNA sequence (CDS)",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "DNA sequence tokenization into 6-mers",
            "DNA encoder produces DNA token embeddings",
            "dense neural network projection into the English word dimension",
            "Perceiver resampler with question-aware cross-attention resamples the DNA tokens",
            "resampled DNA embeddings are inserted in place of the DNA sequence placeholder tokens in the English input sequence"
          ],
          "model_visible_form_verbatim": "K resampled DNA embedding vectors",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "inserted in place of the DNA sequence placeholder tokens in the English input sequence",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Meanwhile, the English prompt is tokenized and English tokens embeddings are produced for each tokens. The K resampled DNA embedding vectors are then inserted in place of the DNA sequence placeholder tokens in the English input sequence. In the case of multiple input DNA sequences, these operations are applied consecutively and independently for each DNA sequence. We experimented with several values of K in practice and observed that low values such as 1 or 4 are not enough for the DNA encoder to impact the behaviour of the frozen English decoder. We found K = 64 to provide a good trade-off between the input length of the English decoder and the performance in practice.",
          "section_heading": "ChatNT solves transcriptomics and proteomics tasks",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/721",
            "#/texts/726",
            "#/texts/727",
            "#/texts/728"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000771::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0e6d44b44412",
          "configuration_id": "config_63fbe9cc1710",
          "route_label": "Multi-turn multi-sequence DNA route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "multiple turns with consecutive questions; exchanges where the question refers to multiple sequences",
          "source_object_verbatim": "one or multiple DNA sequences",
          "source_object_normalized": "one or multiple DNA sequences",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "DNA sequence tokenization into 6-mers",
            "DNA encoder produces DNA token embeddings",
            "dense neural network projection into the English word dimension",
            "Perceiver resampler with question-aware cross-attention resamples the DNA tokens",
            "resampled DNA embeddings are inserted in place of the DNA sequence placeholder tokens in the English input sequence"
          ],
          "model_visible_form_verbatim": "K resampled DNA embedding vectors",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "inserted in place of the DNA sequence placeholder tokens in the English input sequence",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "we also added more complex examples with multiple turns with consecutive questions that can be related or not, and exchanges where the question refers to multiple sequences",
          "section_heading": "New curated genomics instructions dataset of biologically relevant tasks",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            11
          ],
          "doc_item_refs": [
            "#/texts/736",
            "#/texts/737",
            "#/texts/738",
            "#/texts/739",
            "#/texts/740"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000771::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_e367b465c0b9"
    },
    {
      "model_id": "model_efc702564a41",
      "model_name": "Chroma",
      "record_id": "full_2026-07-06__rec_000950",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_6a2c72fbcaf1",
      "paper_title": "Illuminating protein space with a programmable generative model",
      "doi": "10.1038/s41586-023-06728-8",
      "paper_url": "https://doi.org/10.1038/s41586-023-06728-8",
      "route_count": 12,
      "configuration_count": 11,
      "family_counts": {
        "geometric_or_diffusion_state_carrier": 11,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "noisy_diffusion_state": 1,
        "coordinate_backbone_or_shape_conditioning": 3,
        "symbolic_structural_constraint": 7,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "geometric_or_diffusion_state_carrier"
      ],
      "subtypes": [
        "coordinate_backbone_or_shape_conditioning",
        "noisy_diffusion_state",
        "plain_language_prompt_or_question",
        "symbolic_structural_constraint"
      ],
      "primary_subtype": "symbolic_structural_constraint",
      "modalities": [
        "geometric protein structure",
        "geometric shape specification",
        "geometric symmetry specification",
        "protein structure",
        "protein structure constraints",
        "protein structure semantics",
        "text"
      ],
      "lifecycle_phases": [
        "inference"
      ],
      "fusion_topologies": [
        "encoder_decoder",
        "side_or_generative_conditioning",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query",
        "no_text_on_this_route",
        "semantic_annotation"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_000950_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_000950_3d82859ac01c/figure_002.png",
        "figure_index": 2,
        "caption": "Fig. 1 | Chroma is a generative model for proteins and protein complexes that combines structured diffusion for protein backbones with scalable molecular neural networks for backbone synthesis and all-atom design. a , A correlated diffusion process with chain and radius-of-gyration constraints gradually transforms protein structures into random collapsed polymers (right to left). The reverse process (left to right) can be expressed in terms of a time-dependent optimal denoiser ̂ t ( , ) θ t x x that maps noisy coordinates x t at time t to predicted denoised coordinates 0 x . b , We parameterize this in terms",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel scientific schematic labeled **a**, **b**, and **c** describing a protein/polymer diffusion-based design workflow.\n\nPanel **a** shows a progression from a **collapsed polymer system** through multiple increasingly organized colored 3D polymer/protein-like structures. Arrows indicate **Training: forwards polymer diffusion** from structured complex toward collapsed polymer, and **Generation: reverse polymer diffusion** toward a **protein complex backbone**. A final arrow labeled **Design network** leads to an **all-atom complex**, with a zoomed inset of atomic protein structure. Visible state labels include **x₁**, **xₜ**, and **x₀**.\n\nPanel **b** shows a **noisy structure xₜ** processed through a **random graph neural network** using **O(N)** or **O(N log[N]) edges**, producing **confidence-weighted predicted inter-residue geometries**. These are passed to an **equivariant geometry solver**, yielding a **predicted denoised structure x̂θ(xₜ, t)**. The biological objects are multichain protein-like complexes rendered as colored chains or residue clouds.\n\nPanel **c** presents a probabilistic model interface: **time-dependent prior log[pₜ(x)]** plus **time-dependent conditioner(s) log[pθ(y|x)]** leading by downward arrow to **time-dependent posterior log[pθ(x|y)]**. Conditioner icons and labels shown are **Symmetry**, **Substructure**, **Shape**, and **Semantics**.\n\nOverall, the figure depicts a generative protein-complex backbone/all-atom design pipeline using diffusion, graph neural networks, predicted inter-residue geometries, and conditioning objectives.",
        "page_no": 2,
        "sha256": "61414301703f6e8f681586c4d50d999d7eeb591e0234ff7f745948df1fa1f4a9",
        "pixel_width": 1050,
        "pixel_height": 454,
        "crop_box": {
          "x": 0.0,
          "y": 0.42,
          "width": 0.74,
          "height": 0.56
        },
        "panel_label": "b",
        "visible_input_object": "Noisy structure x_t",
        "visible_model_interface": "Random graph neural network \u000b confidence-weighted predicted inter-residue geometries \u000b equivariant geometry solver",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the full inference route in panel b: source noisy structure x_t, intermediate graph/GNN and geometry-solver interface, and the predicted denoised structure. It excludes panel a and panel c while keeping the arrows and labels readable.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "coordinate_backbone_or_shape_conditioning",
          "family_id": "geometric_or_diffusion_state_carrier",
          "route_id": "route_a46c8a9da4a4",
          "example_input": "residue/atom coordinates (xᵢ,yᵢ,zᵢ)",
          "example_carrier": "equivariant geometric state",
          "example_interface": "geometry-aware generator",
          "actual_source": "sampled backbone",
          "actual_model_visible_form": "sampled backbone"
        },
        {
          "subtype_id": "noisy_diffusion_state",
          "family_id": "geometric_or_diffusion_state_carrier",
          "route_id": "route_b48c4af698c6",
          "example_input": "biological state x₀ + noise ε",
          "example_carrier": "xₜ = √αₜx₀ + √(1−αₜ)ε",
          "example_interface": "conditioned denoiser / flow model",
          "actual_source": "protein structures",
          "actual_model_visible_form": "noisy coordinates x_t"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_159cdd15d900",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "natural language captions",
          "actual_model_visible_form": "caption prompts"
        },
        {
          "subtype_id": "symbolic_structural_constraint",
          "family_id": "geometric_or_diffusion_state_carrier",
          "route_id": "route_7776f0a5da8d",
          "example_input": "motif anchors + distance constraints",
          "example_carrier": "symbolic geometry/structure constraints",
          "example_interface": "constraint-conditioned generator",
          "actual_source": "asymmetric subunit",
          "actual_model_visible_form": "symmetry constraints"
        }
      ],
      "routes": [
        {
          "route_id": "route_b48c4af698c6",
          "configuration_id": "config_e2bcf5a460d6",
          "route_label": "correlated protein backbone diffusion",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "unconditional samples of proteins and protein complexes",
          "source_object_verbatim": "protein structures",
          "source_object_normalized": "protein structures",
          "source_modality_normalized": "protein structure",
          "transformation_chain_verbatim": [
            "correlated noise process",
            "reverse polymer diffusion"
          ],
          "model_visible_form_verbatim": "noisy coordinates x_t",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "noisy_diffusion_state",
          "insertion_or_fusion_verbatim": "time-dependent optimal denoiser",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "The reverse process (left to right) can be expressed in terms of a time-dependent optimal denoiser",
          "section_heading": "A scalable generative model for protein systems",
          "supporting_figure_or_table": "Fig. 1a",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            2
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/14"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a46c8a9da4a4",
          "configuration_id": "config_ec3c84fcc42d",
          "route_label": "backbone-conditioned sequence and side-chain design",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Given backbones from this diffusion process",
          "source_object_verbatim": "sampled backbone",
          "source_object_normalized": "sampled backbone",
          "source_modality_normalized": "protein structure",
          "transformation_chain_verbatim": [
            "conditional sequence and side-chain decoding layers",
            "joint generative model for the sequences and structure of a protein complex"
          ],
          "model_visible_form_verbatim": "sampled backbone",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "coordinate_backbone_or_shape_conditioning",
          "insertion_or_fusion_verbatim": "the Chroma design network then generates sequence and side-chain conformations that are conditioned on the sampled backbone",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Given backbones from this diffusion process, the Chroma design network then generates sequence and side-chain conformations that are conditioned on the sampled backbone",
          "section_heading": "A scalable generative model for protein systems",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            2
          ],
          "doc_item_refs": [
            "#/texts/93",
            "#/texts/94"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_7776f0a5da8d",
          "configuration_id": "config_aebdbb7ff9dc",
          "route_label": "symmetry-conditioned complex generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "conditioning on the symmetry of protein complexes",
          "source_object_verbatim": "asymmetric subunit",
          "source_object_normalized": "asymmetric protein subunit",
          "source_modality_normalized": "geometric symmetry specification",
          "transformation_chain_verbatim": [
            "diffusion-conditioner framework",
            "tessellates an asymmetric subunit"
          ],
          "model_visible_form_verbatim": "symmetry constraints",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "tessellates an asymmetric subunit in the energy function",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "conditioning on the symmetry of protein complexes can readily generate samples under arbitrary symmetry groups",
          "section_heading": "Programmability",
          "supporting_figure_or_table": "Fig. 3a",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/145",
            "#/texts/154",
            "#/texts/155",
            "#/texts/156"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000950::0028",
            "dense::full_2026-07-06__rec_000950::0030",
            "dense::full_2026-07-06__rec_000950::0034",
            "dense::full_2026-07-06__rec_000950::0048"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_dde3a4ff2302",
          "configuration_id": "config_30285285de77",
          "route_label": "substructure and motif infilling",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "protein infilling or outfilling",
          "source_object_verbatim": "fixed substructures or motifs",
          "source_object_normalized": "protein substructure or motif",
          "source_modality_normalized": "geometric protein structure",
          "transformation_chain_verbatim": [
            "geometrical constraints",
            "outfill proteins from fixed substructures",
            "graft motifs into larger structures"
          ],
          "model_visible_form_verbatim": "partial substructure constraints",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "coordinate_backbone_or_shape_conditioning",
          "insertion_or_fusion_verbatim": "remove one of the halves and regenerate the missing half",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "We next explored substructure conditioning",
          "section_heading": "Programmability",
          "supporting_figure_or_table": "Fig. 3b",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/145",
            "#/texts/154",
            "#/texts/155",
            "#/texts/156"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000950::0025",
            "dense::full_2026-07-06__rec_000950::0031",
            "dense::full_2026-07-06__rec_000950::0032",
            "dense::full_2026-07-06__rec_000950::0033",
            "dense::full_2026-07-06__rec_000950::0045",
            "dense::full_2026-07-06__rec_000950::0046",
            "dense::full_2026-07-06__rec_000950::0047"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f1dd746f351d",
          "configuration_id": "config_820acd84f1ef",
          "route_label": "inter-atomic distance conditioning",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "enforce particular distances between atoms",
          "source_object_verbatim": "atoms in the structures",
          "source_object_normalized": "inter-atomic distance constraints",
          "source_modality_normalized": "geometric protein structure",
          "transformation_chain_verbatim": [
            "geometrical constraints",
            "distance-based conditioning"
          ],
          "model_visible_form_verbatim": "inter-atomic distance constraints",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "distance-based conditioning",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "enforce particular distances between atoms",
          "section_heading": "Programmability",
          "supporting_figure_or_table": "Fig. 3b",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper names the conditioning objective but does not spell out the exact tensor-level input encoding.",
          "pages": [
            3
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/145"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a9ab693f3f0f",
          "configuration_id": "config_346ccf310ac1",
          "route_label": "shape-conditioned generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "conditioning the generation of single protein chains on the shapes of the Latin alphabet and Arabic numerals",
          "source_object_verbatim": "user-provided point clouds",
          "source_object_normalized": "point clouds",
          "source_modality_normalized": "geometric shape specification",
          "transformation_chain_verbatim": [
            "heuristic classifier gradients",
            "optimal transport distances between atoms and point clouds"
          ],
          "model_visible_form_verbatim": "user-provided point clouds",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "coordinate_backbone_or_shape_conditioning",
          "insertion_or_fusion_verbatim": "adding heuristic classifier gradients based on optimal transport distances between atoms in the structures and user-provided point clouds",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "user-provided point clouds",
          "section_heading": "Programmability",
          "supporting_figure_or_table": "Fig. 3c",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper describes the shape conditioner at a high level rather than giving the full input encoding.",
          "pages": [
            3,
            4,
            5
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/145",
            "#/texts/157",
            "#/texts/189",
            "#/texts/190",
            "#/texts/191"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000950::0029",
            "dense::full_2026-07-06__rec_000950::0043",
            "dense::full_2026-07-06__rec_000950::0049",
            "dense::full_2026-07-06__rec_000950::0057"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ab0d8afb5c61",
          "configuration_id": "config_f51a8a28c295",
          "route_label": "secondary-structure class conditioning",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "conditioning on secondary-structure composition",
          "source_object_verbatim": "secondary-structure class labels",
          "source_object_normalized": "secondary-structure class labels",
          "source_modality_normalized": "protein structure semantics",
          "transformation_chain_verbatim": [
            "neural networks trained to predict protein properties",
            "bias unconditional samples"
          ],
          "model_visible_form_verbatim": "property-classifier scores",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "bias unconditional samples",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "semantic_annotation",
          "input_status": "actual_model_input",
          "evidence_quote": "Neural networks trained to predict protein properties can bias unconditional samples",
          "section_heading": "Programmability",
          "supporting_figure_or_table": "Fig. 4a",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The excerpt describes guidance through property predictions rather than a raw tensor encoding.",
          "pages": [
            5,
            6
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/189",
            "#/texts/190",
            "#/texts/191",
            "#/texts/195"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000950::0053"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ccd3934eda3c",
          "configuration_id": "config_9362de33af21",
          "route_label": "fold class conditioning",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "conditioning on protein semantics, such as secondary structure, fold class",
          "source_object_verbatim": "fold class",
          "source_object_normalized": "protein fold class",
          "source_modality_normalized": "protein structure semantics",
          "transformation_chain_verbatim": [
            "protein semantics",
            "sampling conditioners",
            "bias the diffusion process towards these properties"
          ],
          "model_visible_form_verbatim": "fold-class classifier scores",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "compiled into a set of sampling conditioners",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "semantic_annotation",
          "input_status": "actual_model_input",
          "evidence_quote": "compiled into a set of sampling conditioners",
          "section_heading": "Programmability",
          "supporting_figure_or_table": "Fig. 4a",
          "evidence_status": "text_plus_figure",
          "uncertainty": "This route is absent from the fixed discovery inventory and is recovered only from dense coverage evidence.",
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/154"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000950::0037",
            "dense::full_2026-07-06__rec_000950::0054"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1ea651fde5b6",
          "configuration_id": "config_55a3906cbcaa",
          "route_label": "CATH topology conditioning",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Condition on CATH topology",
          "source_object_verbatim": "CATH topology annotations",
          "source_object_normalized": "CATH topology annotations",
          "source_modality_normalized": "protein structure semantics",
          "transformation_chain_verbatim": [
            "neural network trained to predict CATH topology annotations",
            "drive generation towards samples with high predicted probabilities"
          ],
          "model_visible_form_verbatim": "topology-classifier scores",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "drive generation towards samples with high predicted probabilities",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "semantic_annotation",
          "input_status": "actual_model_input",
          "evidence_quote": "A neural network trained to predict CATH topology annotations can routinely drive generation towards samples with high predicted probabilities",
          "section_heading": "Programmability",
          "supporting_figure_or_table": "Fig. 4b",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper describes predicted-probability guidance rather than a concrete low-level encoding.",
          "pages": [
            5,
            6
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/189",
            "#/texts/190",
            "#/texts/191",
            "#/texts/195"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000950::0055"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_159cdd15d900",
          "configuration_id": "config_fe83bf7f4ba9",
          "route_label": "natural-language caption conditioning",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "natural language captions",
          "source_object_verbatim": "natural language captions",
          "source_object_normalized": "natural language captions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "fine-tuning a multi-label predictor",
            "bias a pretrained large language model into a structure caption predictor"
          ],
          "model_visible_form_verbatim": "caption prompts",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "semantic conditioning on natural language captions",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "semantic conditioning on natural language captions",
          "section_heading": "Programmability",
          "supporting_figure_or_table": "Fig. 4b",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper does not specify a token-level prompt interface beyond caption conditioning.",
          "pages": [
            3,
            5
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/145",
            "#/texts/189",
            "#/texts/190",
            "#/texts/191"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000950::0038",
            "dense::full_2026-07-06__rec_000950::0039",
            "dense::full_2026-07-06__rec_000950::0040",
            "dense::full_2026-07-06__rec_000950::0051",
            "dense::full_2026-07-06__rec_000950::0056"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_db75cb3b4b73",
          "configuration_id": "config_d351e036e129",
          "route_label": "inter-residue contact conditioning",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "programmable generation of protein systems",
          "source_object_verbatim": "inter-residue contact specifications",
          "source_object_normalized": "inter-residue contact specifications",
          "source_modality_normalized": "protein structure constraints",
          "transformation_chain_verbatim": [
            "hard constraints and soft penalties",
            "composable primitives"
          ],
          "model_visible_form_verbatim": "contact constraints",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "composable primitives",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "inter-residue distance and contact",
          "section_heading": "Discussion",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The discussion names the supported property but does not show a dedicated worked example or exact representation.",
          "pages": [
            2
          ],
          "doc_item_refs": [
            "#/texts/93",
            "#/texts/94"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_010"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5a806a5a73a1",
          "configuration_id": "config_d351e036e129",
          "route_label": "domain conditioning",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "programmable generation of protein systems",
          "source_object_verbatim": "protein domain specifications",
          "source_object_normalized": "protein domain specifications",
          "source_modality_normalized": "protein structure semantics",
          "transformation_chain_verbatim": [
            "hard constraints and soft penalties",
            "composable primitives"
          ],
          "model_visible_form_verbatim": "domain constraints",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "composable primitives",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "domain, sub-structure and semantic specification from classifiers",
          "section_heading": "Discussion",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The main text mentions domain as a programmable property but does not spell out a concrete domain-conditioning example or representation.",
          "pages": [
            7,
            8
          ],
          "doc_item_refs": [
            "#/texts/454",
            "#/texts/455",
            "#/texts/457",
            "#/texts/458"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000950::route_011"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_04e6eca108d0"
    },
    {
      "model_id": "model_422e09e03da5",
      "model_name": "Clinical-LongFormer",
      "record_id": "full_2026-07-06__rec_001277",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_55720b9d7702",
      "paper_title": "WITHDRAWN: OKR-Cell: Open World Knowledge Aided Single-Cell Foundation Model with Robust Cross-Modal Cell-Language Pre-training",
      "doi": "10.64898/2026.01.09.698573",
      "paper_url": "https://doi.org/10.64898/2026.01.09.698573",
      "route_count": 2,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 2
      },
      "subtype_counts": {
        "direct_projected_embedding": 2
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "unclear"
      ],
      "fusion_topologies": [
        "other_explicit"
      ],
      "text_roles": [
        "metadata_or_context"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "The crop shows an OKR-Cell multimodal pre-training schematic, not the Clinical-LongFormer reliability-screening path with original/augmented text dense vectors. The figure does not responsibly evidence the requested route.",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_834ff7bbf34f",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "original text T_ori_i",
          "actual_model_visible_form": "dense vectors"
        }
      ],
      "routes": [
        {
          "route_id": "route_834ff7bbf34f",
          "configuration_id": "config_00c5c600c476",
          "route_label": "original text to Clinical-LongFormer reliability screening",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "Reliability Screening (RS)",
          "source_object_verbatim": "original text T_ori_i",
          "source_object_normalized": "original text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "text encoding"
          ],
          "model_visible_form_verbatim": "dense vectors",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "textual feature extractor f_RS(·)",
          "fusion_topology": "other_explicit",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Encode original text T ori i and augmented text T aug i into dense vectors via Clinical-Longformer",
          "section_heading": "4.2.1 LLM-enriched Textual Corpus Curation",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": "Upstream screening stage; the schema has no dedicated data-curation lifecycle label.",
          "pages": [
            19
          ],
          "doc_item_refs": [
            "#/texts/414",
            "#/texts/415",
            "#/texts/416",
            "#/texts/417",
            "#/texts/418"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001277::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2ba5313f5a9d",
          "configuration_id": "config_00c5c600c476",
          "route_label": "augmented text to Clinical-LongFormer reliability screening",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "Reliability Screening (RS)",
          "source_object_verbatim": "augmented text T_aug_i",
          "source_object_normalized": "augmented text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "text encoding"
          ],
          "model_visible_form_verbatim": "dense vectors",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "textual feature extractor f_RS(·)",
          "fusion_topology": "other_explicit",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Encode original text T ori i and augmented text T aug i into dense vectors via Clinical-Longformer",
          "section_heading": "4.2.1 LLM-enriched Textual Corpus Curation",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": "Upstream screening stage; the schema has no dedicated data-curation lifecycle label.",
          "pages": [
            19
          ],
          "doc_item_refs": [
            "#/texts/414",
            "#/texts/415",
            "#/texts/416",
            "#/texts/417",
            "#/texts/418"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001277::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_000c4dd8661a"
    },
    {
      "model_id": "model_14bc733e1633",
      "model_name": "compound-phenotype scoring model",
      "record_id": "full_2026-07-06__rec_003629",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_6fd67e763424",
      "paper_title": "Phenotype-Guided In Silico Molecular Generation Using Large Language Models",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "transcriptomics and small-molecule compound"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "Figure 3 is a scorer benchmark: PR/ROC curves, positive-vs-negative score separation, and threshold analyses. It does not visibly show the actual text-based DEG input route or an immediate model interface for scoring compound-phenotype pairs, so it is output/evaluation evidence rather than route evidence.",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_221414b2ba08",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "compound-phenotype pair",
          "actual_model_visible_form": "text-based ranked lists of differentially expressed genes"
        }
      ],
      "routes": [
        {
          "route_id": "route_221414b2ba08",
          "configuration_id": "config_89156b119ca3",
          "route_label": "Scoring-based prioritization of GEMGen-generated compounds",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "prioritize GEMGen-generated molecules according to their likelihood of reproducing a given transcriptional signature",
          "source_object_verbatim": "compound-phenotype pair",
          "source_object_normalized": "compound and transcriptomic signature",
          "source_modality_normalized": "transcriptomics and small-molecule compound",
          "transformation_chain_verbatim": [
            "compound-phenotype pair scoring",
            "score thresholding",
            "precision/recall and ROC evaluation"
          ],
          "model_visible_form_verbatim": "text-based ranked lists of differentially expressed genes",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "scoring the correspondence between chemical perturbations and transcriptomic changes",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "compound-phenotype scoring model",
          "section_heading": "Scoring-based prioritization of GEMGen-generated compounds",
          "supporting_figure_or_table": "Fig. 3a-b",
          "evidence_status": "text_plus_figure",
          "uncertainty": "This evaluation route is absent from the fixed discovery inventory and is recovered from the dense Graph pass.",
          "pages": [
            6,
            22
          ],
          "doc_item_refs": [
            "#/texts/286",
            "#/texts/566"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003629::0061",
            "dense::full_2026-07-06__rec_003629::0013",
            "dense::full_2026-07-06__rec_003629::0040"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_0cc72bd82901"
    },
    {
      "model_id": "model_7ed6448bcdeb",
      "model_name": "DeepSeek",
      "record_id": "full_2026-07-06__rec_003008",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_b6246bd13ef4",
      "paper_title": "Aligning LLMs with Biomedical Knowledge using Balanced Fine-Tuning",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question"
      ],
      "primary_subtype": "plain_language_prompt_or_question",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "inference"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003008_figure_005.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003008_6a0122710a41/figure_005.png",
        "figure_index": 5,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nPanel: single visible panel labeled **a**.\n\nDescription: The figure shows a workflow for converting gene information into embeddings. A **User prompt** box contains “Tell me about gene.” This input is passed through a **DeepSeek** model labeled **LLM-BFT**, producing an **LLM response** box with text such as “Gene name, function, interaction…”. The response is transformed into separate text entries labeled **Text of gene x** through **Text of gene y**, indicating multiple gene-specific textual descriptions. These texts are then passed into **Youtu-Embedding**, labeled **Text to embedding**, producing a **Gene embedding (N,2048)** matrix shown as a pink heatmap-like grid.\n\nBiological source objects: genes represented as gene-specific text descriptions.\n\nTransformations/model interfaces: user prompt → DeepSeek LLM response → gene text descriptions → Youtu-Embedding text-to-embedding model → numerical gene embedding matrix.\n\nFindings: The panel illustrates a pipeline for deriving 2048-dimensional gene embeddings from LLM-generated gene descriptions; no experimental results or quantitative findings are shown.",
        "page_no": 16,
        "sha256": "a46b5537dc98bb9f5100065a56ac2cba80b5fe12ee5f4ae14eff4158d0938750",
        "pixel_width": 787,
        "pixel_height": 137,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 0.889,
          "height": 1
        },
        "panel_label": "a",
        "visible_input_object": "User prompt: \"Tell me about gene.\"",
        "visible_model_interface": "Direct text prompting into DeepSeek LLM-BFT, then text-to-embedding via Youtu-Embedding; includes the LLM response and gene text carriers before the embedding output.",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "Crops the full left-to-middle workflow needed to ground the route: source prompt, DeepSeek transformation, gene text carriers, and the Youtu-Embedding interface. It excludes the output-only embedding heatmap on the right.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_b9151c53252e",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "User prompt: \"Tell me about gene.\"",
          "actual_model_visible_form": "user prompt text"
        }
      ],
      "routes": [
        {
          "route_id": "route_b9151c53252e",
          "configuration_id": "config_e6a8e72e657d",
          "route_label": "Gene prompt workflow for embedding extraction",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "gene information workflow for embedding extraction",
          "source_object_verbatim": "User prompt: \"Tell me about gene.\"",
          "source_object_normalized": "User prompt Tell me about gene",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "DeepSeek LLM-BFT generates a response",
            "gene descriptions are extracted"
          ],
          "model_visible_form_verbatim": "user prompt text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "direct text prompting",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Extended Data Figure 6: Workflow for extracting biological embeddings from LLM-BFT. a : LLMBFT generates responses based on entities of interest (e.g., a specific gene). The textual description of the gene is input into Tencent Youtu-Embedding to obtain gene embeddings. b : For a single-cell dataset, gene embeddings are weighted by gene expression values to generate cell embeddings.",
          "section_heading": null,
          "supporting_figure_or_table": "Extended Data Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": "The panel labels the model as DeepSeek/LLM-BFT without restating the exact size.",
          "pages": [
            5,
            6,
            7,
            16
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/pictures/5",
            "#/texts/155",
            "#/texts/42",
            "#/texts/44",
            "#/texts/48",
            "#/texts/49",
            "#/texts/50"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_357f1bf88b0b"
    },
    {
      "model_id": "model_c7150172725e",
      "model_name": "DeepSeek-R1-Distill (1.5B)",
      "record_id": "full_2026-07-06__rec_003008",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_b6246bd13ef4",
      "paper_title": "Aligning LLMs with Biomedical Knowledge using Balanced Fine-Tuning",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question"
      ],
      "primary_subtype": "plain_language_prompt_or_question",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "paired_alignment_supervision"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "None of the visible figures shows the NuminaMath source route or an immediate math fine-tuning interface. Figure 3 is only benchmark accuracy on math datasets, which is evaluation output rather than grounded input-route evidence. The other figures are biomedical or unrelated schematic/embedding panels, so they do not support this route.",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_469c5890ae77",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "NuminaMath dataset examples",
          "actual_model_visible_form": "mathematical reasoning instruction-response pairs"
        }
      ],
      "routes": [
        {
          "route_id": "route_469c5890ae77",
          "configuration_id": "config_a1920ad5a8ee",
          "route_label": "NuminaMath mathematical reasoning fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "on the NuminaMath dataset",
          "source_object_verbatim": "NuminaMath dataset examples",
          "source_object_normalized": "NuminaMath dataset examples",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "fine-tuned on the NuminaMath dataset"
          ],
          "model_visible_form_verbatim": "mathematical reasoning instruction-response pairs",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "standard supervised fine-tuning / BFT on the model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "To conveniently verify the effectiveness of BFT's components, we fine-tuned a relatively small-scale model, DeepSeek-R1-Distill (1.5B), on the NuminaMath dataset [8]. The fine-tuned model was then evaluated on multiple widely used mathematical reasoning benchmarks, including math_oai [9], minerva_math [10], and olympiadbench [11].",
          "section_heading": "2.2 Ablation study",
          "supporting_figure_or_table": "Extended Data Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/texts/18",
            "#/texts/20",
            "#/texts/21",
            "#/texts/22"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_357f1bf88b0b"
    },
    {
      "model_id": "model_0b491a7a5d54",
      "model_name": "DeepSeek-R1-Distill 70B",
      "record_id": "full_2026-07-06__rec_003008",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_b6246bd13ef4",
      "paper_title": "Aligning LLMs with Biomedical Knowledge using Balanced Fine-Tuning",
      "doi": "",
      "paper_url": "",
      "route_count": 2,
      "configuration_count": 2,
      "family_counts": {
        "text_native_token_stream": 2
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 1,
        "serialized_biological_context_or_ordered_profile": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "inference"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003008_figure_005.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003008_6a0122710a41/figure_005.png",
        "figure_index": 5,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nPanel: single visible panel labeled **a**.\n\nDescription: The figure shows a workflow for converting gene information into embeddings. A **User prompt** box contains “Tell me about gene.” This input is passed through a **DeepSeek** model labeled **LLM-BFT**, producing an **LLM response** box with text such as “Gene name, function, interaction…”. The response is transformed into separate text entries labeled **Text of gene x** through **Text of gene y**, indicating multiple gene-specific textual descriptions. These texts are then passed into **Youtu-Embedding**, labeled **Text to embedding**, producing a **Gene embedding (N,2048)** matrix shown as a pink heatmap-like grid.\n\nBiological source objects: genes represented as gene-specific text descriptions.\n\nTransformations/model interfaces: user prompt → DeepSeek LLM response → gene text descriptions → Youtu-Embedding text-to-embedding model → numerical gene embedding matrix.\n\nFindings: The panel illustrates a pipeline for deriving 2048-dimensional gene embeddings from LLM-generated gene descriptions; no experimental results or quantitative findings are shown.",
        "page_no": 16,
        "sha256": "a46b5537dc98bb9f5100065a56ac2cba80b5fe12ee5f4ae14eff4158d0938750",
        "pixel_width": 787,
        "pixel_height": 137,
        "crop_box": {
          "x": 0.0,
          "y": 0.03,
          "width": 0.85,
          "height": 0.82
        },
        "panel_label": "a",
        "visible_input_object": "User prompt text: \"Tell me about gene.\"",
        "visible_model_interface": "DeepSeek LLM-BFT -> LLM response -> text-of-gene carrier -> Youtu-Embedding text-to-embedding interface",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the left and middle workflow only: prompt, model response, gene-text carrier, and the insertion interface into Youtu-Embedding. Excludes the rightmost output heatmap as output-only.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_aff9dbfc0d11",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "the prompt \"Tell me about gene CTSL\"",
          "actual_model_visible_form": "user prompt text"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_9db368b35e63",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "gene set: ZMPSTE24, BANF1, WRN, LMNA",
          "actual_model_visible_form": "gene set prompt text"
        }
      ],
      "routes": [
        {
          "route_id": "route_aff9dbfc0d11",
          "configuration_id": "config_7c90b43f117b",
          "route_label": "Gene knowledge prompt response generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Tell me about gene CTSL",
          "source_object_verbatim": "the prompt \"Tell me about gene CTSL\"",
          "source_object_normalized": "Tell me about gene CTSL prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "prompted to produce a gene description"
          ],
          "model_visible_form_verbatim": "user prompt text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "direct text prompting",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Extended Data Table 1: Comparison of SFT and BFT Responses to the \"Tell me about gene CTSL\" Prompt.",
          "section_heading": null,
          "supporting_figure_or_table": "Extended Data Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            19
          ],
          "doc_item_refs": [
            "#/texts/163"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9db368b35e63",
          "configuration_id": "config_f43a2f899c54",
          "route_label": "Biological process reasoning from gene sets",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "biological process reasoning form gene set",
          "source_object_verbatim": "gene set: ZMPSTE24, BANF1, WRN, LMNA",
          "source_object_normalized": "gene set ZMPSTE24 BANF1 WRN LMNA",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "prompted with a gene set",
            "infer a biological process or pathway name"
          ],
          "model_visible_form_verbatim": "gene set prompt text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "direct text prompting",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Extended Data Table 2: Comparison between BFT, SFT, and Real Research on the biological process reasoning form gene set: ZMPSTE24, BANF1, WRN, LMNA.",
          "section_heading": null,
          "supporting_figure_or_table": "Extended Data Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            20
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/165"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_d8f17c473542"
    },
    {
      "model_id": "model_3d02d9393c92",
      "model_name": "DeepSeek-R1-Distill series (14B, 32B, and 70B)",
      "record_id": "full_2026-07-06__rec_003008",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_b6246bd13ef4",
      "paper_title": "Aligning LLMs with Biomedical Knowledge using Balanced Fine-Tuning",
      "doi": "",
      "paper_url": "",
      "route_count": 7,
      "configuration_count": 5,
      "family_counts": {
        "text_native_token_stream": 7
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 4,
        "serialized_biological_context_or_ordered_profile": 3
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "plain_language_prompt_or_question",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "This figure is a benchmark-results panel with comparative scores, not an input or carrier path for the named model. It does not visibly support a responsible model-input route, so the crop should be removed.",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_a11168101d2d",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "OpenAI Health Bench Consensus subset tasks",
          "actual_model_visible_form": "clinical and biomedical reasoning tasks"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_ce3413c5632f",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "MsigDB benchmark gene sets",
          "actual_model_visible_form": "gene-set prompt text"
        }
      ],
      "routes": [
        {
          "route_id": "route_a11168101d2d",
          "configuration_id": "config_cbbb988cc8ae",
          "route_label": "OpenAI Health Bench Consensus fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Consensus subset",
          "source_object_verbatim": "OpenAI Health Bench Consensus subset tasks",
          "source_object_normalized": "OpenAI Health Bench Consensus subset tasks",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "fine-tuned on the Consensus subset"
          ],
          "model_visible_form_verbatim": "clinical and biomedical reasoning tasks",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "supervised fine-tuning / BFT",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "Consensus subset",
          "section_heading": "2.3.1 Medicine: BFT performs well on the OpenAI Health Bench",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            3
          ],
          "doc_item_refs": [
            "#/texts/25",
            "#/texts/26",
            "#/texts/27",
            "#/texts/6"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003008::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ccfe017e70f5",
          "configuration_id": "config_fe4b33c2af55",
          "route_label": "OpenAI Health Bench Hard evaluation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Hard subset",
          "source_object_verbatim": "OpenAI Health Bench Hard subset tasks",
          "source_object_normalized": "OpenAI Health Bench Hard subset tasks",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "evaluated on the Hard subset"
          ],
          "model_visible_form_verbatim": "real-world clinical and biomedical reasoning questions",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "fed directly to the fine-tuned model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Hard subset",
          "section_heading": "2.3.1 Medicine: BFT performs well on the OpenAI Health Bench",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3
          ],
          "doc_item_refs": [
            "#/texts/25",
            "#/texts/26",
            "#/texts/27"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1efccd4707e6",
          "configuration_id": "config_1eacd39a4f7b",
          "route_label": "MMLU forgetting evaluation after Health Bench fine-tuning",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "MMLU benchmark",
          "source_object_verbatim": "MMLU benchmark questions",
          "source_object_normalized": "MMLU benchmark questions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "evaluated on the MMLU benchmark"
          ],
          "model_visible_form_verbatim": "general-domain benchmark questions",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "fed directly to the fine-tuned model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "MMLU benchmark",
          "section_heading": "2.3.2 General area: reducing forgetfulness",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/29",
            "#/texts/31",
            "#/texts/33",
            "#/texts/34"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_008"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1701512ab87c",
          "configuration_id": "config_223be17f91fd",
          "route_label": "CMMLU forgetting evaluation after Health Bench fine-tuning",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "CMMLU benchmark",
          "source_object_verbatim": "CMMLU benchmark questions",
          "source_object_normalized": "CMMLU benchmark questions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "evaluated on the CMMLU benchmark"
          ],
          "model_visible_form_verbatim": "general-domain benchmark questions",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "fed directly to the fine-tuned model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "CMMLU benchmark",
          "section_heading": "2.3.2 General area: reducing forgetfulness",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/29",
            "#/texts/31",
            "#/texts/33",
            "#/texts/34"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_009"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ce3413c5632f",
          "configuration_id": "config_002034fa0e26",
          "route_label": "MSigDB biological process reasoning evaluation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "biological process reasoning benchmark",
          "source_object_verbatim": "MsigDB benchmark gene sets",
          "source_object_normalized": "MsigDB benchmark gene sets",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "evaluated on the biological process reasoning benchmark"
          ],
          "model_visible_form_verbatim": "gene-set prompt text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "fed directly to the fine-tuned model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Among the three models, we chose DeepSeek-R1-Distill 70B, which has the most accurate reasoning results. We compared the BFT-based 70B LLM with two latest baselines. As shown in Figure 1f, the BFT-based LLM outperforms GeneAgent in biological process reasoning tasks, demonstrating stronger reasoning ability in gene interactions and related processes. Unlike GeneAgent, the BFTbased LLM does not rely on external API calls and database access (such as OpenAI and NCBI), nor does it require the design of an agent scheduling process. This indicates that BFT has enabled LLM to learn biological knowledge.",
          "section_heading": "2.3.3 Biology: BFT improves reasoning about biological processes",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/36",
            "#/texts/37",
            "#/texts/38",
            "#/texts/39"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_010"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4f0a7233781e",
          "configuration_id": "config_002034fa0e26",
          "route_label": "NeST biological process reasoning evaluation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "biological process reasoning benchmark",
          "source_object_verbatim": "NeST benchmark gene sets",
          "source_object_normalized": "NeST benchmark gene sets",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "evaluated on the biological process reasoning benchmark"
          ],
          "model_visible_form_verbatim": "gene-set prompt text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "fed directly to the fine-tuned model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Among the three models, we chose DeepSeek-R1-Distill 70B, which has the most accurate reasoning results. We compared the BFT-based 70B LLM with two latest baselines. As shown in Figure 1f, the BFT-based LLM outperforms GeneAgent in biological process reasoning tasks, demonstrating stronger reasoning ability in gene interactions and related processes. Unlike GeneAgent, the BFTbased LLM does not rely on external API calls and database access (such as OpenAI and NCBI), nor does it require the design of an agent scheduling process. This indicates that BFT has enabled LLM to learn biological knowledge.",
          "section_heading": "2.3.3 Biology: BFT improves reasoning about biological processes",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/36",
            "#/texts/37",
            "#/texts/38",
            "#/texts/39"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_011"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b4746361feed",
          "configuration_id": "config_002034fa0e26",
          "route_label": "Gene Ontology biological process reasoning evaluation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "biological process reasoning benchmark",
          "source_object_verbatim": "Gene Ontology benchmark gene sets",
          "source_object_normalized": "Gene Ontology benchmark gene sets",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "evaluated on the biological process reasoning benchmark"
          ],
          "model_visible_form_verbatim": "gene-set prompt text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "fed directly to the fine-tuned model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Among the three models, we chose DeepSeek-R1-Distill 70B, which has the most accurate reasoning results. We compared the BFT-based 70B LLM with two latest baselines. As shown in Figure 1f, the BFT-based LLM outperforms GeneAgent in biological process reasoning tasks, demonstrating stronger reasoning ability in gene interactions and related processes. Unlike GeneAgent, the BFTbased LLM does not rely on external API calls and database access (such as OpenAI and NCBI), nor does it require the design of an agent scheduling process. This indicates that BFT has enabled LLM to learn biological knowledge.",
          "section_heading": "2.3.3 Biology: BFT improves reasoning about biological processes",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/36",
            "#/texts/37",
            "#/texts/38",
            "#/texts/39"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_012"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_d8f17c473542"
    },
    {
      "model_id": "model_cc36a8d0a233",
      "model_name": "Deepseek-V3",
      "record_id": "full_2026-07-06__rec_001277",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_55720b9d7702",
      "paper_title": "WITHDRAWN: OKR-Cell: Open World Knowledge Aided Single-Cell Foundation Model with Robust Cross-Modal Cell-Language Pre-training",
      "doi": "10.64898/2026.01.09.698573",
      "paper_url": "https://doi.org/10.64898/2026.01.09.698573",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "unclear"
      ],
      "fusion_topologies": [
        "retrieval_or_tool_context"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001277_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001277_8e87e5eae10d/figure_001.png",
        "figure_index": 1,
        "caption": "Fig. 1 (A) The schematic overview of the OKR-CELL method. (B) The illustration of several downstream tasks implemented via OKR-CELL, including cell clustering, batch affect correlation, cell-type annotation and cross-modal retrieval.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific schematic with two labeled panels: **A** and **B**.\n\n**Panel A:** Shows a workflow for constructing and pretraining a cross-modal cell-text model. Biological source objects include a human/body lung region illustration, dissociated cells, and **scRNA-seq** data represented as a gene-by-cell expression matrix. A text source labeled **Cell Description** is split into **Chunks**, sampled/retrieved, stored in a **Database**, processed by an **LLM**, and passed through a **Reliability Screen**. The cell modality is transformed from gene expression values into **Gene Tokens**, then passed into a **Cell Encoder**. The text modality is transformed into **Text Tokens**, then passed into a **Text Encoder**. The two encoders interface through a **Cross-modal Similarity** module. The pretraining block is labeled **Intra-modal Cellular Generative Pre-training** and includes masked gene expression values prediction. A second block is labeled **Cross-modal Cell-text Pre-training**, showing contrastive-style alignment between cells and text, with legend items including **Cell**, **Push**, **Real Positive Text**, **Fake Positive Text**, **Hard Negative Text**, and **Easy Negative Text**.\n\n**Panel B:** Shows downstream use of a **Pretrained Cell Encoder**. Biological inputs include organ/tissue icons and microscopy-like cellular imagery, followed by **scRNA-seq** gene expression matrices. Gene expression values are converted into **Gene Tokens** and passed into the pretrained cell encoder. The output is connected to multiple application panels: **Clustering**, **Batch-effect Correction**, **Cell type Annotation**, **Zero-shot Annotation**, **Few-shot Annotation**, and **Cross-modal Retrieval**.\n\nOverall, the figure depicts a computational biology / single-cell omics model pipeline that integrates scRNA-seq gene expression data with textual cell descriptions for cross-modal pretraining and downstream cell analysis tasks.",
        "page_no": 5,
        "sha256": "cdf370825d3d0bfcd7e4241b5d498050e78ed3dde00fe987b44c69c89b629577",
        "pixel_width": 745,
        "pixel_height": 520,
        "crop_box": {
          "x": 0.03,
          "y": 0.19,
          "width": 0.88,
          "height": 0.41
        },
        "panel_label": "A",
        "visible_input_object": "Cell Description text box and the text-augmentation pipeline (Chunk -> Retriever -> Database -> LLM)",
        "visible_model_interface": "Text-native prompt sequence entering the LLM, with the retrieval/database context and reliability screen visible",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the panel A text route tightly enough to keep the source text, transformation steps, and LLM insertion point readable while excluding panel B and the cell-encoder side that is not needed for this route.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_050dc5bde59a",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "standardized original text",
          "actual_model_visible_form": "combined prompt sequence"
        }
      ],
      "routes": [
        {
          "route_id": "route_050dc5bde59a",
          "configuration_id": "config_cfd69d086451",
          "route_label": "RAG prompt to Deepseek-V3 for text augmentation",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "RAG enhanced Textual Descriptions Generation",
          "source_object_verbatim": "standardized original text",
          "source_object_normalized": "standardized original text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "Data Preprocessing and Structuralization",
            "Constructing a Biomedical Knowledge Base",
            "RAG enhanced Textual Descriptions Generation"
          ],
          "model_visible_form_verbatim": "combined prompt sequence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "feeding the combined prompt sequence into the LLM",
          "fusion_topology": "retrieval_or_tool_context",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "feeding the combined prompt sequence into the LLM",
          "section_heading": "4.2.1 LLM-enriched Textual Corpus Curation",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": "Upstream data-curation step rather than a named model-training phase in the target schema.",
          "pages": [
            19
          ],
          "doc_item_refs": [
            "#/texts/414",
            "#/texts/415",
            "#/texts/416",
            "#/texts/417",
            "#/texts/418"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001277::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_a2ae3a8f5414",
      "model_name": "GEMGen",
      "record_id": "full_2026-07-06__rec_003629",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_6fd67e763424",
      "paper_title": "Phenotype-Guided In Silico Molecular Generation Using Large Language Models",
      "doi": "",
      "paper_url": "",
      "route_count": 9,
      "configuration_count": 7,
      "family_counts": {
        "text_native_token_stream": 9
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 7,
        "structured_biological_prompt_or_task_scaffold": 2
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "chemical perturbation and transcriptomics",
        "transcriptomic signature",
        "transcriptomics"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning"
      ],
      "fusion_topologies": [
        "side_or_generative_conditioning",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003629_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003629_cb28e7e12c02/figure_001.png",
        "figure_index": 1,
        "caption": "Figure  1.  Overview  of  the  GEMGen  framework.  a, Conceptual  illustration  of GEMGen. Given transcriptomic profiles describing transitions  between  healthy  and diseased  cellular  states,  GEMGen  generates  candidate  compounds  predicted  to induce the desired  phenotypic  change. b, End-to-end  workflow  of  GEMGen. Transcriptomic  changes  are  provided  as  text-based, rank-ordered  lists  of  up-  and down-regulated genes within a specific cellular context. GEMGen generates candidate compounds in SMILES format, which are optionally filtered by additional modules and ranked using the GEMGen scorer to prioritize candidates for experimental validation and downstream development. c , Model architecture and training strategy. GEMGen builds  on  a  scientific  foundation  model  (e.g.,  NatureLM)  adapted  from  a  generalpurpose  large  language  model  (e.g.,  GPT-5).  On  this  basis,  GEMGen  performs bioentity representation generation, followed by phenotype-guided compound generation  and  compound-phenotype  matching. d ,  Transformation  of  quantitative transcriptomic measurements into standardized, text-based features. Paired",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel scientific schematic and results figure about **GEMGen**, a compound-generation framework for phenotype-guided biological intervention.\n\nPanel descriptions:\n\n- **Panel a:** Conceptual illustration of GEMGen generating **compounds** to shift a biological/cellular state from **diseased** toward **healthy**. Shows orange molecular structures, a GEMGen icon, cell-like circular icons, and a colored 3D landscape labeled healthy and diseased.\n\n- **Panel b:** Workflow showing an **instruction-based compound generation interface**. Input instruction asks to design a compound that, when applied to a specified **cell type**, will enhance expression of one **gene list** and reduce expression of another. GEMGen produces multiple responses as **compound SMILES strings**. Outputs are passed through additional filtering modules, a GEMGen matching score from **0-1**, then **experimental validation and clinical development**.\n\n- **Panel c:** Model architecture/training concept. A **general-purpose large language model** is adapted using scientific-domain data such as literature and scientific entities into a **scientific foundation model**. Within GEMGen, modules include:\n  - **Bioentity representation generation**, including cell type and gene-name description generation.\n  - **Phenotype-guided compound generation**.\n  - **Compound-phenotype matching**.\n\n- **Panel d:** Feature representation schematic comparing transcriptomic input levels. Shows paired cellular transcriptomes, **DGE analysis**, extraction of **differentially expressed genes**, and conversion into **ranked DEGs**. Example upregulated genes include **ACTB, MYC, MYOF**; example downregulated genes include **CHD2, RPL9, ZNF292**. The panel contrasts quantitative feature levels with robustness in feature extraction.\n\n- **Panel e:** Two density plots comparing duplicate pairs versus random pairs:\n  - **DEG fold changes (log)** cosine similarity, with reported **W-distance: 0.115**.\n  - **DEG embeddings** cosine similarity, with reported **W-distance: 0.198**.\n  The plots suggest stronger separation for DEG embeddings than raw fold-change features.\n\n- **Panel f:** Training and evaluation sources/tasks. Training uses approximately **180k L1000 compound-phenotype pairs**. Evaluation includes **held-out compounds from L1000**, **single-cell chemical perturbations**, **genetic perturbation**, and **cell state transition** tasks.\n\nBiological/source objects visible: cells/cell states, transcriptomes, differentially expressed genes, gene lists, compounds/molecular structures, cell-type and gene-name representations, chemical and genetic perturbation contexts.",
        "page_no": 29,
        "sha256": "c7f59d9ca6eff0b9a0cc5c39797b5a54b7b9b7ff0f4f869a1ea48fec12f76b60",
        "pixel_width": 788,
        "pixel_height": 927,
        "crop_box": {
          "x": 0.03,
          "y": 0.57,
          "width": 0.72,
          "height": 0.24
        },
        "panel_label": "d",
        "visible_input_object": "Cell transcriptomes from paired cellular states, converted by DGE analysis into ranked differentially expressed gene lists",
        "visible_model_interface": "Text-based ranked DEGs with upregulated and downregulated gene lists",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop isolates panel d, which contains the source object, transformation arrow, and model-visible ranked DEG representation needed to ground phenotype-guided input preparation. It excludes the output-only and unrelated architecture/validation panels while keeping the labels readable.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_9ace8021aa2c",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "paired compound and perturbed-transcriptome information from the L1000 chemical perturbation dataset",
          "actual_model_visible_form": "text-based ranked lists of up-regulated and down-regulated genes in an instruction-response prompt"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_f1d270cc5d39",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "compound-phenotype pairs represented by cell-compound perturbation data",
          "actual_model_visible_form": "natural language instructions and responses with a binary yes/no output"
        }
      ],
      "routes": [
        {
          "route_id": "route_9ace8021aa2c",
          "configuration_id": "config_966e748fa315",
          "route_label": "Phenotype-guided compound generation, Phase I",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "phenotype-guided compound generation task",
          "source_object_verbatim": "paired compound and perturbed-transcriptome information from the L1000 chemical perturbation dataset",
          "source_object_normalized": "L1000 compound perturbation transcriptomic profiles",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "retrieved Level 3 transcriptomic profiles",
            "computed differential gene expression signatures with the Characteristic Direction method",
            "transformed transcriptomic signatures into text-based ranked lists of differentially expressed genes"
          ],
          "model_visible_form_verbatim": "text-based ranked lists of up-regulated and down-regulated genes in an instruction-response prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conditioning de novo molecular generation on textual representations",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Phase I was conducted on compound perturbations curated from the original L1000 dataset.",
          "section_heading": "GEMGen Training",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16,
            17,
            21
          ],
          "doc_item_refs": [
            "#/texts/540",
            "#/texts/541",
            "#/texts/543",
            "#/texts/561"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003629::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003629::0063",
            "dense::full_2026-07-06__rec_003629::0066",
            "dense::full_2026-07-06__rec_003629::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_26eb3d665aaa",
          "configuration_id": "config_966e748fa315",
          "route_label": "Phenotype-guided compound generation, Phase II",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "phenotype-guided compound generation task",
          "source_object_verbatim": "paired compound and perturbed-transcriptome information from the L1000CDS2 dataset",
          "source_object_normalized": "L1000CDS2 transcriptomic signatures",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "used CD-derived differential gene expression signatures",
            "provided broader gene coverage",
            "continued instruction fine-tuning"
          ],
          "model_visible_form_verbatim": "text-based ranked lists of up-regulated and down-regulated genes in an instruction-response prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "continued instruction fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Phase II involved continued instruction fine-tuning with the L1000CDS² dataset",
          "section_heading": "GEMGen Training",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16,
            17,
            21
          ],
          "doc_item_refs": [
            "#/texts/540",
            "#/texts/541",
            "#/texts/543",
            "#/texts/561"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003629::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003629::0063",
            "dense::full_2026-07-06__rec_003629::0066",
            "dense::full_2026-07-06__rec_003629::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f1d270cc5d39",
          "configuration_id": "config_ae19acdd8098",
          "route_label": "Compound-phenotype matching, Phase I",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "compound-phenotype matching task",
          "source_object_verbatim": "compound-phenotype pairs represented by cell-compound perturbation data",
          "source_object_normalized": "compound-phenotype pairs",
          "source_modality_normalized": "chemical perturbation and transcriptomics",
          "transformation_chain_verbatim": [
            "transformed quantitative transcriptomic measurements into text-based ranked lists of differentially expressed genes",
            "paired the ranked gene lists with compounds",
            "constructed negative samples by pairing compounds with unrelated expression profiles"
          ],
          "model_visible_form_verbatim": "natural language instructions and responses with a binary yes/no output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction-response format with classification head",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "The model was trained to produce a binary response ('yes' or 'no') for each compound-phenotype pair",
          "section_heading": "GEMGen Training",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            16,
            17,
            21,
            22,
            23
          ],
          "doc_item_refs": [
            "#/texts/141",
            "#/texts/142",
            "#/texts/143",
            "#/texts/144",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/149",
            "#/texts/540",
            "#/texts/541",
            "#/texts/543",
            "#/texts/561",
            "#/texts/566",
            "#/texts/567",
            "#/texts/569"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003629::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003629::0063",
            "dense::full_2026-07-06__rec_003629::0067",
            "dense::full_2026-07-06__rec_003629::0010",
            "dense::full_2026-07-06__rec_003629::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_7c652f603a0e",
          "configuration_id": "config_ae19acdd8098",
          "route_label": "Compound-phenotype matching, Phase II",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "compound-phenotype matching task",
          "source_object_verbatim": "compound-phenotype pairs represented by cell-compound perturbation data",
          "source_object_normalized": "compound-phenotype pairs",
          "source_modality_normalized": "chemical perturbation and transcriptomics",
          "transformation_chain_verbatim": [
            "continued instruction fine-tuning using the same data corpus",
            "used the same binary yes/no response scheme",
            "optimized the scorer with a reduced learning rate and batch size"
          ],
          "model_visible_form_verbatim": "natural language instructions and responses with a binary yes/no output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "classification head on the final hidden state of the response token",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "Phase II continued for 3,008 steps on 8 GPUs using a reduced learning rate of 8e-6 and a batch size of 128.",
          "section_heading": "GEMGen Training",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            16,
            17,
            21,
            22,
            23
          ],
          "doc_item_refs": [
            "#/texts/141",
            "#/texts/142",
            "#/texts/143",
            "#/texts/144",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/149",
            "#/texts/540",
            "#/texts/541",
            "#/texts/543",
            "#/texts/561",
            "#/texts/566",
            "#/texts/567",
            "#/texts/569"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003629::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003629::0063",
            "dense::full_2026-07-06__rec_003629::0067",
            "dense::full_2026-07-06__rec_003629::0010",
            "dense::full_2026-07-06__rec_003629::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_500d73af64a6",
          "configuration_id": "config_6e4c1b523c8d",
          "route_label": "Phenotype-guided compound generation, L1000 novel compound evaluation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "novel compound test set",
          "source_object_verbatim": "held-out L1000 compound perturbation profiles",
          "source_object_normalized": "held-out L1000 perturbation profiles",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "computed differential gene expression signatures with the Characteristic Direction method",
            "converted transcriptomic signatures into text-based ranked lists of differentially expressed genes"
          ],
          "model_visible_form_verbatim": "text-based ranked lists of up-regulated and down-regulated genes in an instruction-response prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conditioning de novo molecular generation on textual representations",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "we held out 404 L1000 samples as a 'novel compound' test set",
          "section_heading": "Benchmarking GEMGen for phenotype-guided molecular generation",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            13,
            29,
            30,
            31
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/texts/196",
            "#/texts/198",
            "#/texts/199",
            "#/texts/200",
            "#/texts/201",
            "#/texts/202",
            "#/texts/203",
            "#/texts/204",
            "#/texts/511",
            "#/texts/512",
            "#/texts/513",
            "#/texts/514",
            "#/texts/515",
            "#/texts/516",
            "#/texts/517",
            "#/texts/518",
            "#/texts/592",
            "#/texts/594"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003629::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003629::0025",
            "dense::full_2026-07-06__rec_003629::0028"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_be93bab9a08c",
          "configuration_id": "config_fca9663da8a4",
          "route_label": "Phenotype-guided compound generation, Tahoe-100M novel condition evaluation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "novel condition test set",
          "source_object_verbatim": "high-quality compound-phenotype pairs from the Tahoe-100M single-cell chemical perturbation dataset",
          "source_object_normalized": "Tahoe-100M compound-phenotype pairs",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "filtered cells with quality-control criteria",
            "retained cells treated with a 5 μM drug dose",
            "sampled 100 cells before and 100 cells after perturbation",
            "computed E-distance and retained combinations with E-distance > 500",
            "calculated CD-derived differential gene expression signatures"
          ],
          "model_visible_form_verbatim": "text-based ranked lists of up-regulated and down-regulated genes in an instruction-response prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conditioning de novo molecular generation on textual representations",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "we curated 640 high-quality compound-phenotype pairs from the Tahoe-100M dataset",
          "section_heading": "Benchmarking GEMGen for phenotype-guided molecular generation",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": "Discovery inventory marked this candidate grounding_valid=false, but the canonical Methods and Figure 2 explicitly describe the Tahoe-100M evaluation route.",
          "pages": [
            4,
            5,
            6,
            13,
            29,
            30,
            31,
            32,
            33
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/texts/196",
            "#/texts/198",
            "#/texts/199",
            "#/texts/200",
            "#/texts/201",
            "#/texts/202",
            "#/texts/203",
            "#/texts/204",
            "#/texts/511",
            "#/texts/512",
            "#/texts/513",
            "#/texts/514",
            "#/texts/515",
            "#/texts/516",
            "#/texts/517",
            "#/texts/518",
            "#/texts/592",
            "#/texts/594",
            "#/texts/596",
            "#/texts/598"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003629::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003629::0025",
            "dense::full_2026-07-06__rec_003629::0028",
            "dense::full_2026-07-06__rec_003629::0037",
            "dense::full_2026-07-06__rec_003629::0038",
            "dense::full_2026-07-06__rec_003629::0039"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_cf88b829ce2c",
          "configuration_id": "config_b70ac59a3848",
          "route_label": "Genetic perturbation compound generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "genetic perturbation test set",
          "source_object_verbatim": "CRISPR interference Perturb-seq single-cell transcriptomic profiles for genome-scale genetic perturbations",
          "source_object_normalized": "genetic perturbation transcriptomic signatures",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "selected 21 genes with robust and distinctive transcriptional responses",
            "identified corresponding upregulated and downregulated gene sets",
            "reformatted the signatures into instruction-response prompts"
          ],
          "model_visible_form_verbatim": "text-based ranked lists of up-regulated and down-regulated genes in an instruction-response prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conditioning de novo molecular generation on textual representations",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "we leveraged a CRISPR interference Perturb-seq dataset from Replogle et al.",
          "section_heading": "GEMGen generates compounds that phenocopy genetic perturbations",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            13
          ],
          "doc_item_refs": [
            "#/texts/331",
            "#/texts/332",
            "#/texts/333",
            "#/texts/335",
            "#/texts/336",
            "#/texts/337",
            "#/texts/338",
            "#/texts/339",
            "#/texts/511",
            "#/texts/512",
            "#/texts/513",
            "#/texts/514",
            "#/texts/515",
            "#/texts/516",
            "#/texts/517",
            "#/texts/518"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003629::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003629::0062",
            "dense::full_2026-07-06__rec_003629::0025"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2c6e23e7d686",
          "configuration_id": "config_00294700e76f",
          "route_label": "Fibrosis phenotype compound generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "fibrosis-associated transcriptional signature",
          "source_object_verbatim": "fibrosis-associated transcriptional signature marked by ACTA2, COL1A1, FN1, and TGFB1",
          "source_object_normalized": "fibrosis transcriptional signature",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "identified the fibrosis-associated transcriptional program",
            "used the signature as input to GEMGen",
            "generated candidate compounds predicted to suppress fibroblast activation"
          ],
          "model_visible_form_verbatim": "text-based ranked lists of up-regulated and down-regulated genes in an instruction-response prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conditioning de novo molecular generation on textual representations",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "we used this fibrosis-associated transcriptional signature as input to GEMGen",
          "section_heading": "GEMGen generates compounds reverting fibrosis phenotype",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9,
            10,
            13
          ],
          "doc_item_refs": [
            "#/texts/422",
            "#/texts/423",
            "#/texts/425",
            "#/texts/426",
            "#/texts/427",
            "#/texts/428",
            "#/texts/429",
            "#/texts/430",
            "#/texts/511",
            "#/texts/512",
            "#/texts/513",
            "#/texts/514",
            "#/texts/515",
            "#/texts/516",
            "#/texts/517",
            "#/texts/518"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003629::route_010"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003629::0002",
            "dense::full_2026-07-06__rec_003629::0025"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1306d62b8271",
          "configuration_id": "config_0f9d4a1ed530",
          "route_label": "KEAP1 CRISPR perturbation signature generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "KEAP1 CRISPR perturbation signature",
          "source_object_verbatim": "KEAP1 CRISPR perturbation signature",
          "source_object_normalized": "KEAP1 genetic perturbation signature",
          "source_modality_normalized": "transcriptomic signature",
          "transformation_chain_verbatim": [
            "KEAP1 CRISPR perturbation signature",
            "generated compounds",
            "experimental testing"
          ],
          "model_visible_form_verbatim": "text-based, rank-ordered lists of up- and down-regulated genes",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "used as input to GEMGen to generate candidate compounds",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Genetic perturbations induce characteristic transcriptional changes that are used as input to GEMGen for de novo generation of candidate small molecules.",
          "section_heading": "GEMGen identifies new KEAP1 inhibitors",
          "supporting_figure_or_table": "Fig. 4a",
          "evidence_status": "text_plus_figure",
          "uncertainty": "This is a dense-only split of the broader genetic-perturbation evaluation family.",
          "pages": [
            34
          ],
          "doc_item_refs": [
            "#/texts/600"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003629::0001"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_2bfb9536cf22"
    },
    {
      "model_id": "model_6cd07bd7db7c",
      "model_name": "Gemma-7B",
      "record_id": "full_2026-07-06__rec_003043",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_3bbf77870504",
      "paper_title": "Geneverse: A collection of Open-source Multimodal Large Language Models for Genomic and Proteomic Research",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "Figure 4 is a scorer/evaluator schematic, not the model’s input route or immediate inference interface. Figure 6 is an actual instruction example, but it is for protein naming / APOC1 marker questions rather than the gene-function prompt route. The contact sheet does not visibly show a gene-function prompt being fed to Gemma-7B.",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_95a5e9d8fe90",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene name in the prompt",
          "actual_model_visible_form": "text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_95a5e9d8fe90",
          "configuration_id": "config_c4e1ebbb6016",
          "route_label": "gene-function prompt to Gemma-7B",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "gene function description",
          "source_object_verbatim": "gene name in the prompt",
          "source_object_normalized": "GLI1",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "instruction example"
          ],
          "model_visible_form_verbatim": "text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "text-only prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Our task is to generate the summary of gene functions based on the prompt only containing the task description.",
          "section_heading": "4.1 Benchmarking LLMs for summarizing of gene functions",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            16
          ],
          "doc_item_refs": [
            "#/texts/438",
            "#/texts/70",
            "#/texts/71",
            "#/texts/72",
            "#/texts/73"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003043::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003043::0035"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_3841ec0733e7",
      "model_name": "gene_eng_gpt2_para_seg",
      "record_id": "full_2026-07-06__rec_002243",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_2db24aaf3a41",
      "paper_title": "Human Genome Book:Words,Sentences and Paragraphs",
      "doi": "10.1101/2025.01.23.634629",
      "paper_url": "https://doi.org/10.1101/2025.01.23.634629",
      "route_count": 2,
      "configuration_count": 2,
      "family_counts": {
        "text_native_token_stream": 1,
        "discrete_biological_symbol_stream": 1
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 1,
        "native_biological_token_stream": 1
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream",
        "plain_language_prompt_or_question"
      ],
      "primary_subtype": "plain_language_prompt_or_question",
      "modalities": [
        "DNA sequence",
        "text"
      ],
      "lifecycle_phases": [
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "metadata_or_context"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "No panel visibly shows the exact route input or immediate model interface. Figure 1 is only a generic training/fine-tuning workflow and does not visibly show Wikipedia text with <p_end> markers or the chromosome-1 segmentation input. Figure 4 is a conceptual genome hierarchy, and Figure 5 is a content screenshot rather than a model route or interface.",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_e3205dec51da",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "GRCh38.p14 Chromosome 1 segments",
          "actual_model_visible_form": "Chromosome 1 DNA sequence segments"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_6ba807e0d8ae",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "Wikipedia text",
          "actual_model_visible_form": "Wikipedia text with <p_end> markers"
        }
      ],
      "routes": [
        {
          "route_id": "route_6ba807e0d8ae",
          "configuration_id": "config_dd78ee229e66",
          "route_label": "Wikipedia paragraph segmentation training",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "paragraph segmentation",
          "source_object_verbatim": "Wikipedia text",
          "source_object_normalized": "Wikipedia text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "newline characters between paragraphs replaced with <p_end>",
            "continuous pre-training"
          ],
          "model_visible_form_verbatim": "Wikipedia text with <p_end> markers",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "GPT2LMHeadModel with a causal language modeling head",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "We  used  the  GPT2LMHeadModel  with  a  causal  language  modeling  head  as  the  base  model  for paragraph  prediction.  The  training  process  was  based  on  a  language  modeling  task,  employing continuous  pre-training  to  enable  the  model  to  learn  the  patterns  of  paragraph  boundaries.  Postpreprocessing, texts marked with <p_end> were used as training inputs, with the model's objective being  to  predict  the  next  token,  including  both  regular  vocabulary  and  the  paragraph  end  marker <p_end>. Through this approach, the task of predicting paragraph boundaries was integrated into the language modeling task without the need to design additional task heads.",
          "section_heading": "2.4. Text Segmentation Model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            10,
            11,
            12
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/20",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/27",
            "#/texts/67",
            "#/texts/79",
            "#/texts/80",
            "#/texts/82",
            "#/texts/90"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0017",
            "dense::full_2026-07-06__rec_002243::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e3205dec51da",
          "configuration_id": "config_1fb52b6c832d",
          "route_label": "Chromosome 1 paragraph segmentation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Genome Segmentation - Chapter Division - Sentence Splitting",
          "source_object_verbatim": "GRCh38.p14 Chromosome 1 segments",
          "source_object_normalized": "Chromosome 1 DNA sequence segments",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "divided Chromosome 1 into 25 segments",
            "applied a segmentation model to further divide the sequences into basic paragraphs"
          ],
          "model_visible_form_verbatim": "Chromosome 1 DNA sequence segments",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "segmentation model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "We first divided Chromosome 1 into 25 segments",
          "section_heading": "2.7. Genome Segmentation - Chapter Division - Sentence Splitting",
          "supporting_figure_or_table": "Figure 4. Structure of the Genome Catalog.",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/texts/101",
            "#/texts/103",
            "#/texts/105",
            "#/texts/107",
            "#/texts/108",
            "#/texts/97",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0022",
            "dense::full_2026-07-06__rec_002243::0023"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_fdbd3ba264a9"
    },
    {
      "model_id": "model_58435d2084b0",
      "model_name": "gene_eng_gpt2_summary",
      "record_id": "full_2026-07-06__rec_002243",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_2db24aaf3a41",
      "paper_title": "Human Genome Book:Words,Sentences and Paragraphs",
      "doi": "10.1101/2025.01.23.634629",
      "paper_url": "https://doi.org/10.1101/2025.01.23.634629",
      "route_count": 4,
      "configuration_count": 4,
      "family_counts": {
        "text_native_token_stream": 4
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 2,
        "serialized_biological_context_or_ordered_profile": 2
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "DNA sequence",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "inference"
      ],
      "fusion_topologies": [
        "concatenation",
        "side_or_generative_conditioning",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query",
        "metadata_or_context",
        "paired_alignment_supervision"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "The crop contains the summarization training pipeline and resulting model labels, but not the actual text input format or any DNA prompt/interface for the claimed routes.",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_d2a0d71b1a7b",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "summaries from all DNA paragraphs within the section",
          "actual_model_visible_form": "concatenated summary sequence"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_61ca2ca26bf3",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "Amazon English Review Dataset review content",
          "actual_model_visible_form": "'[Original Text] TL;DR: [Summary]' text sequence"
        }
      ],
      "routes": [
        {
          "route_id": "route_61ca2ca26bf3",
          "configuration_id": "config_164b71abdcfa",
          "route_label": "Amazon review summarization training",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "summarization",
          "source_object_verbatim": "Amazon English Review Dataset review content",
          "source_object_normalized": "Amazon English Review Dataset review content",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "review content used as input text",
            "formatted as '[Original Text] TL;DR: [Summary]'"
          ],
          "model_visible_form_verbatim": "'[Original Text] TL;DR: [Summary]' text sequence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "causal language modeling head",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "each training sample was organized into the format '[Original Text] TL;DR: [Summary]'",
          "section_heading": "2.6. Summary model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            12
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/20",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/27",
            "#/texts/88",
            "#/texts/90",
            "#/texts/92"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0019",
            "dense::full_2026-07-06__rec_002243::0020",
            "dense::full_2026-07-06__rec_002243::0004"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6aa93231ea9d",
          "configuration_id": "config_d90d572bfa0e",
          "route_label": "DNA summarization inference",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "summary generation",
          "source_object_verbatim": "DNA sequences",
          "source_object_normalized": "DNA sequences",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "formatted as '[Original Text] TL;DR:'",
            "dynamic masking"
          ],
          "model_visible_form_verbatim": "'[Original Text] TL;DR:' prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "masked causal language modeling head",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "User inputs, such as DNA sequences, were formatted as '[Original Text] TL;DR:'.",
          "section_heading": "2.6. Summary model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            12
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/20",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/27",
            "#/texts/92"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0021"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d2a0d71b1a7b",
          "configuration_id": "config_4dd437867bf9",
          "route_label": "DNA section title generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Title Generation for 'DNA Sections'",
          "source_object_verbatim": "summaries from all DNA paragraphs within the section",
          "source_object_normalized": "DNA paragraph summaries",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "extract summaries from all DNA paragraphs within the section",
            "concatenate these summaries in order to form a long sequence",
            "perform summary extraction on the concatenated sequence to obtain the title"
          ],
          "model_visible_form_verbatim": "concatenated summary sequence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "concatenate these summaries in order to form a long sequence",
          "fusion_topology": "concatenation",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Extract summaries from all DNA paragraphs within the section.",
          "section_heading": "Title Generation for 'DNA Sections'",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13,
            14,
            15
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/texts/101",
            "#/texts/103",
            "#/texts/105",
            "#/texts/107",
            "#/texts/108",
            "#/texts/114",
            "#/texts/115",
            "#/texts/97",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_013"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0028"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_da112bfb76ee",
          "configuration_id": "config_475bf78d5ab3",
          "route_label": "DNA volume title generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Title Generation for 'DNA Volumes'",
          "source_object_verbatim": "titles of all DNA Sections within the volume",
          "source_object_normalized": "DNA section titles",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "concatenate the titles of all DNA Sections within the volume to form a long sequence",
            "perform summary extraction on the concatenated sequence to obtain the title"
          ],
          "model_visible_form_verbatim": "concatenated title sequence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "concatenate the titles of all DNA Sections within the volume to form a long sequence",
          "fusion_topology": "concatenation",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "concatenate the titles of all 'DNA Sections' within the volume to form a long sequence.",
          "section_heading": "Title Generation for 'DNA Volumes'",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15
          ],
          "doc_item_refs": [
            "#/texts/117",
            "#/texts/118",
            "#/texts/119"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_014"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0029"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_2bfb9536cf22"
    },
    {
      "model_id": "model_fa94e391fe47",
      "model_name": "Geneverse",
      "record_id": "full_2026-07-06__rec_003043",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_3bbf77870504",
      "paper_title": "Geneverse: A collection of Open-source Multimodal Large Language Models for Genomic and Proteomic Research",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "other_explicit"
      ],
      "text_roles": [
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003043_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003043_460a3d629653/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: The landscape of Geneverse. To generate LLMs for genomic and proteomic analysis, we incorporate the training datasets from rephrased descriptions for gene functions as well as synthetic descriptions from GPT 3.5. We then adjust the base model with different strategies and select the best candidate. To generate MLLMs for genomic and proteomic analysis, we incorporate the training datasets from known databases, including both descriptions and corresponding images. We then finetune the base model with different strategies and select the best candidate. The logo of Geneverse is generated by DALLE (OpenAI, 2024).",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow figure for a biomedical AI model named **Geneverse**.\n\nVisible content:\n- Left panel labeled **“Preprocessing steps”**:\n  - Input stack labeled **“Descriptions from databases”** containing repeated cards with text like **“MERTK Ensemble id... HGNC id ...”**\n  - Arrow labeled **“GPT 3.5 rephrased”** pointing toward augmented text descriptions.\n  - A **GPT-3.5** icon combined with a box labeled **“Biomedical instructions”**, feeding into generated descriptions.\n\n- Middle panel labeled **“Training datasets”**:\n  - Top subsection: **“Augmented Descriptions”**, showing blue stacked cards labeled **“MERTK Ensemble id... HGNC id ...”**\n  - Middle subsection: **“Descriptions from GPT 3.5”**, showing another set of blue stacked description cards.\n  - Bottom subsection: **“Descriptions from databases”**, showing red stacked cards with text like **“This protein/gene is known as...”**\n  - Bottom also includes **“Biomedical images”**, illustrated by a protein structure image and a microscopy/histology-like image.\n  - Text and biomedical images are visually combined with a plus symbol, indicating multimodal training inputs.\n\n- Right model panel:\n  - Dashed red box containing model/source icons, including a stylized **M**, a circular animal/cartoon-like icon, a **Google “G”**, and another icon, with ellipsis between them.\n  - This panel is labeled externally by the title/logo **“Geneverse”**.\n\n- Output/application area:\n  - Arrow from the model panel to **“Downstream applications”**.\n  - Icons suggest biological or biomedical tasks involving DNA/gene symbols, a cloud/thought bubble, molecular or cellular graphics, histology/microscopy images, and green check/red cross classification-like outputs.\n\nBiological source objects shown:\n- Gene/protein descriptions from databases.\n- Specific example text references **MERTK**, **Ensembl ID**, and **HGNC ID**.\n- Biomedical instructions.\n- Biomedical images, including a protein structure and microscopy/histology-like imagery.\n\nTransformations/model interfaces:\n- Database descriptions are rephrased by **GPT-3.5**.\n- GPT-3.5 plus biomedical instructions generates additional descriptions.\n- Text descriptions and biomedical images are combined into training datasets.\n- The trained/assembled Geneverse system is used for downstream biomedical applications.\n\nNo explicit quantitative findings, experimental results, or performance metrics are visible in the image.",
        "page_no": 3,
        "sha256": "0818b9a6512a994aa99711be5f9d73c3ee63c9ad5ea74f885855bb299389d7c7",
        "pixel_width": 761,
        "pixel_height": 349,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.58,
          "height": 0.65
        },
        "panel_label": "Preprocessing and description augmentation panels",
        "visible_input_object": "Descriptions from databases / gene functional description data",
        "visible_model_interface": "GPT-3.5 rephrasing path with the GPT-3.5 icon and arrow into augmented descriptions",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the source text cards, the GPT-3.5 transformation arrow, and the augmented-description output panel, which is enough to ground the gene function description training route. It excludes the downstream model and application panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_c450584a8acc",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene functional description data",
          "actual_model_visible_form": "training dataset with both real data and synthetic data"
        }
      ],
      "routes": [
        {
          "route_id": "route_c450584a8acc",
          "configuration_id": "config_dbaf35a033bb",
          "route_label": "gene function description training",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "generation of descriptions for gene functions",
          "source_object_verbatim": "gene functional description data",
          "source_object_normalized": "gene function descriptions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "gene functional description data",
            "training dataset with both real data and synthetic data",
            "finetuned LLMs",
            "generation of descriptions for gene functions"
          ],
          "model_visible_form_verbatim": "training dataset with both real data and synthetic data",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "designed instructions",
          "fusion_topology": "other_explicit",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "We prepare the training dataset with both real data and synthetic data with a data augmentation policy.",
          "section_heading": "3.3 Supervised finetuning process of LLM",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4
          ],
          "doc_item_refs": [
            "#/texts/62"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003043::0001",
            "dense::full_2026-07-06__rec_003043::0025",
            "dense::full_2026-07-06__rec_003043::0029",
            "dense::full_2026-07-06__rec_003043::0030",
            "dense::full_2026-07-06__rec_003043::0031"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_69a439438d24",
      "model_name": "GenNA",
      "record_id": "june_update_2026-06-10__rec_000133",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_476f8441c752",
      "paper_title": "GenNA: Conditional generation of nucleotide sequences guided by natural-language annotations",
      "doi": "10.64898/2026.04.22.720063",
      "paper_url": "https://doi.org/10.64898/2026.04.22.720063",
      "route_count": 11,
      "configuration_count": 10,
      "family_counts": {
        "text_native_token_stream": 10,
        "discrete_biological_symbol_stream": 1
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 3,
        "structured_biological_prompt_or_task_scaffold": 7,
        "native_biological_token_stream": 1
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "RNA",
        "genomic DNA",
        "nucleotide sequence",
        "text",
        "text-serialized nucleotide sequence plus natural-language annotation",
        "text-serialized nucleotide sequence plus species label"
      ],
      "lifecycle_phases": [
        "evaluation",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "placeholder_replacement",
        "prefix",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "metadata_or_context",
        "modality_or_task_selector",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/june_update_2026_06_10_rec_000133_figure_005.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/june_update_2026_06_10_rec_000133_3abbdd0df15f/figure_005.png",
        "figure_index": 5,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific figure labeled **a–h** describing a multimodal nucleotide sequence foundation model and tokenizer analysis.\n\nPanel **a** shows a workflow for **multi-modal nucleotide sequences**. Inputs include **NCBI RefSeq GenBank files** with plasmids/genomes/chromosomes, plus hybrid nucleotide language materials: **genomic DNA and RNA**, **natural language annotations**, and **XML structural tags**. These are processed by a **cross-modal BPE tokenizer** with **6,000 vocabulary size** and used to train a **GenNA Foundation Model**, described as a **Qwen3 decoder-only transformer with 3.6B parameters**. The model interface is shown for causal language modeling / next-token prediction over mixed biological and annotation tokens. Output capabilities include sequence-function alignment, understanding biological rules, zero-shot variant effect prediction, species-specific genomic fingerprints, latent-space clustering, phylogenetic tree reconstruction, unconditional self-guided generation, targeted ncRNA generation, and protein-coding sequence generation.\n\nPanel **b** is a pie chart showing source composition: **RNA 73.1%** and **genomic DNA 26.9%**.\n\nPanel **c** is a length distribution histogram comparing **genomic DNA** and **RNA** sequence lengths in characters. Both distributions are concentrated at shorter lengths with a long tail extending toward 20,000 characters.\n\nPanel **d** contains two taxonomic composition pie charts. Visible groups include **plant**, **invertebrate**, **vertebrate mammalian**, **vertebrate other**, **fungi**, and **protozoa**, with vertebrate other and plant comprising large portions.\n\nPanel **e** illustrates cross-modal BPE tokenization of a multimodal input sequence containing a tRNA annotation and nucleotide sequence, e.g. `...tRNA-Glu<seq>ATAC<gene><tRNA>TATTT...`. The sequence is transformed into a **BPE token sequence**, then mapped into **token IDs**.\n\nPanel **f** is a donut chart of tokenizer vocabulary composition for **vocabulary 6000**, broken down by token type and count. It includes nucleotide k-mers from **1-mer** through **9-mer and above**, plus **N-containing kmers** and **non-nucleotide** tokens. The largest visible segment is **7-mer**, followed by **8-mer** and **6-mer** categories.\n\nPanel **g** is a bar chart of **coverage by k-mer length**, showing decreasing coverage as k-mer length increases. Bars are labeled for k=1 through k=8 with percentages, including **100.0%** for 1-mers and lower coverage for longer k-mers.\n\nPanel **h** is a scatter plot comparing **character count** and **token count**, with a fitted line. The relationship is highly linear, with visible annotation approximately **Fit: y = 0.199x** and **r = 0.999**.",
        "page_no": 32,
        "sha256": "1b2b03a63676c1c927164e85299bdad3dfdfd36cb5503ce899dd75a1beb0bcd6",
        "pixel_width": 972,
        "pixel_height": 562,
        "crop_box": {
          "x": 0.496,
          "y": 0.0,
          "width": 0.504,
          "height": 0.358
        },
        "panel_label": "e",
        "visible_input_object": "Multimodal Input Sequence with tRNA-Glu<seq> / nucleotide string",
        "visible_model_interface": "Cross-Modal BPE Tokenization mapping to BPE Token Sequence and Token IDs",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops only panel e, keeping the source sequence box, the tokenization arrow, the BPE token sequence, and the token-ID mapping needed to understand the actual input route. It excludes output-only evaluation panels and unrelated plots.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_a732f19c65b6",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "wild-type protein-coding gene sequences",
          "actual_model_visible_form": "mutant and wild-type sequence inputs"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_1f96c335f3ca",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "GenBank-formatted genomic DNA annotation files",
          "actual_model_visible_form": "unified training samples"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_3f2cd944ba2e",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "the modality label (Genomic DNA or RNA)",
          "actual_model_visible_form": "the modality label"
        }
      ],
      "routes": [
        {
          "route_id": "route_1f96c335f3ca",
          "configuration_id": "config_026e86537e4b",
          "route_label": "Gene-associated genomic DNA pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "multimodal nucleotide pretraining corpus",
          "source_object_verbatim": "GenBank-formatted genomic DNA annotation files",
          "source_object_normalized": "genomic DNA annotation files",
          "source_modality_normalized": "genomic DNA",
          "transformation_chain_verbatim": [
            "collect all available GenBank-formatted genomic DNA and RNA annotation files from the NCBI RefSeq database",
            "restrict the corpus to gene-associated regions with explicit annotations",
            "parse GenBank records",
            "extract molecule type, source species, gene name, natural-language functional descriptions, and structured annotations",
            "write these structured annotations together with the corresponding underlying nucleotide sequences into unified training samples"
          ],
          "model_visible_form_verbatim": "unified training samples",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "written together with the corresponding underlying nucleotide sequences into unified training samples",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "To construct the multimodal nucleotide pretraining corpus, we collected all available 338 GenBank-formatted 26 genomic  DNA  and  RNA  annotation  files  for  eukaryotic  species  from  the  NCBI 339 RefSeq database 25 .  Because intergenic and other non-gene regions often contain abundant repetitive 340 and  low-complexity  sequence,  and  a  genomic  language-modeling  study  reported  better  downstream 341 performance when  training was concentrated on annotated gene regions rather than mixed 342 whole-genome sequence 17 ,  we primarily restricted the corpus to gene-associated regions with explicit 343 annotations. 344",
          "section_heading": "Pretraining corpus construction",
          "supporting_figure_or_table": "Fig. 1a",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes the DNA and RNA corpus jointly; this route isolates the genomic DNA half for accounting.",
          "pages": [
            11,
            12,
            13
          ],
          "doc_item_refs": [
            "#/texts/384",
            "#/texts/385",
            "#/texts/386",
            "#/texts/387",
            "#/texts/388",
            "#/texts/389",
            "#/texts/390",
            "#/texts/391",
            "#/texts/448",
            "#/texts/449",
            "#/texts/450",
            "#/texts/451",
            "#/texts/452",
            "#/texts/453",
            "#/texts/454",
            "#/texts/455"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_001"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000133::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_775271f6aa86",
          "configuration_id": "config_026e86537e4b",
          "route_label": "Gene-associated RNA pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "multimodal nucleotide pretraining corpus",
          "source_object_verbatim": "GenBank-formatted RNA annotation files",
          "source_object_normalized": "RNA annotation files",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "collect all available GenBank-formatted genomic DNA and RNA annotation files from the NCBI RefSeq database",
            "restrict the corpus to gene-associated regions with explicit annotations",
            "parse GenBank records",
            "extract molecule type, source species, gene name, natural-language functional descriptions, and structured annotations",
            "write these structured annotations together with the corresponding underlying nucleotide sequences into unified training samples"
          ],
          "model_visible_form_verbatim": "unified training samples",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "written together with the corresponding underlying nucleotide sequences into unified training samples",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "To construct the multimodal nucleotide pretraining corpus, we collected all available 338 GenBank-formatted 26 genomic  DNA  and  RNA  annotation  files  for  eukaryotic  species  from  the  NCBI 339 RefSeq database 25 .  Because intergenic and other non-gene regions often contain abundant repetitive 340 and  low-complexity  sequence,  and  a  genomic  language-modeling  study  reported  better  downstream 341 performance when  training was concentrated on annotated gene regions rather than mixed 342 whole-genome sequence 17 ,  we primarily restricted the corpus to gene-associated regions with explicit 343 annotations. 344",
          "section_heading": "Pretraining corpus construction",
          "supporting_figure_or_table": "Fig. 1a",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes the DNA and RNA corpus jointly; this route isolates the RNA half for accounting.",
          "pages": [
            11,
            12,
            13
          ],
          "doc_item_refs": [
            "#/texts/384",
            "#/texts/385",
            "#/texts/386",
            "#/texts/387",
            "#/texts/388",
            "#/texts/389",
            "#/texts/390",
            "#/texts/391",
            "#/texts/448",
            "#/texts/449",
            "#/texts/450",
            "#/texts/451",
            "#/texts/452",
            "#/texts/453",
            "#/texts/454",
            "#/texts/455"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_001"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000133::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3f2cd944ba2e",
          "configuration_id": "config_63689ed67983",
          "route_label": "Unconditional self-guided generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "unconditional self-guided generation",
          "source_object_verbatim": "the modality label (Genomic DNA or RNA)",
          "source_object_normalized": "modality label",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "provide only the modality label as the initial input",
            "autoregressively generate the complete metadata fields",
            "generate species, gene name, functional annotation, and the corresponding nucleotide sequence"
          ],
          "model_visible_form_verbatim": "the modality label",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "as the initial input",
          "fusion_topology": "prefix",
          "text_role": "modality_or_task_selector",
          "input_status": "actual_model_input",
          "evidence_quote": "In the unconditional self-guided generation task, the model received only the modality label (Genomic 496",
          "section_heading": "Unconditional self-generation and semantic fidelity analysis",
          "supporting_figure_or_table": "Fig. 3a-b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16,
            32
          ],
          "doc_item_refs": [
            "#/texts/577",
            "#/texts/578",
            "#/texts/579",
            "#/texts/580",
            "#/texts/581",
            "#/texts/582",
            "#/texts/583",
            "#/texts/584"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_36bbccf780e7",
          "configuration_id": "config_b79f84373b6a",
          "route_label": "Species-specific generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "species-specific generation",
          "source_object_verbatim": "species name",
          "source_object_normalized": "conditioning species name",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "use the unified prompt template",
            "prompt the model with a species name",
            "generate nucleotide sequences from six representative model organisms"
          ],
          "model_visible_form_verbatim": "unified prompt template: [Molecule Type], [Species Name], [Gene Symbol], [Functional Annotation]<seq>",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "used a unified prompt template as the input prefix",
          "fusion_topology": "prefix",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "a ,  Schematic  of  the  three  different  generation paradigms:  unconditional  self-guided  generation, 759 species-specific  generation,  and  downstream  task-directed  generation. b , c ,  Cross-modal  semantic 760 consistency of generated sequences from unconditional and species-specific generation tasks. Scatter 761 plots  show  cosine  similarity  between  generated  text  metadata  and  BLASTP-derived  functional 762 annotations, with horizontal dashed lines indicating the high-confidence threshold (0.8). Marginal plots 763 show data density. d ,  Probability density distributions of CDS GC content (top) and CDS GC3 content 764 (bottom)  for  sequences  generated  by  GenNA  versus  real  gene  sequences  of  fruit  fly (Drosophila 765 melanogaster) . e , Comparison  of CDS  GC  (top)  and  CDS  GC3  (bottom)  content  across  six 766 representative eukaryotic model organisms. f , Radar charts comparing synonymous codon usage bias 767 between real reference sequences (dashed black lines) and generated sequences (solid colored lines). g , 768 Global amino acid frequencies of real and generated sequences across 6 model organisms. 769",
          "section_heading": "Species-specific genome feature evaluation",
          "supporting_figure_or_table": "Fig. 3a, 3c-g",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            16,
            17,
            28,
            29,
            30,
            34
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/540",
            "#/texts/541",
            "#/texts/542",
            "#/texts/543",
            "#/texts/544",
            "#/texts/545",
            "#/texts/546",
            "#/texts/547",
            "#/texts/592",
            "#/texts/593",
            "#/texts/594",
            "#/texts/595",
            "#/texts/596",
            "#/texts/597",
            "#/texts/598",
            "#/texts/599",
            "#/texts/818",
            "#/texts/819",
            "#/texts/821",
            "#/texts/822",
            "#/texts/823",
            "#/texts/824",
            "#/texts/825"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_003"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000133::0007",
            "dense::june_update_2026-06-10__rec_000133::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_cf7bc194d0b4",
          "configuration_id": "config_b3ee6cd40045",
          "route_label": "tRNA-targeted generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "61 tRNA prompt categories with different anti-codon",
          "source_object_verbatim": "tRNA Name",
          "source_object_normalized": "tRNA name / anticodon class",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "construct inputs using the format 'Genomic DNA, [tRNA Name]<seq>'",
            "generate candidate tRNA sequences",
            "evaluate generated sequences with tRNAscan-SE"
          ],
          "model_visible_form_verbatim": "Genomic DNA, [tRNA Name]<seq>",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "construct inputs using the format 'Genomic DNA, [tRNA Name]<seq>'",
          "fusion_topology": "prefix",
          "text_role": "modality_or_task_selector",
          "input_status": "actual_model_input",
          "evidence_quote": "constructed with the correct  and substituted species  labels, respectively.  The  resulting 10 g3400 10 PPL 469",
          "section_heading": "tRNA generation",
          "supporting_figure_or_table": "Fig. 4a-c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            16
          ],
          "doc_item_refs": [
            "#/texts/540",
            "#/texts/541",
            "#/texts/542",
            "#/texts/543",
            "#/texts/544",
            "#/texts/545",
            "#/texts/546",
            "#/texts/547"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_72b2cf8c9eac",
          "configuration_id": "config_4b97e93d59bf",
          "route_label": "rRNA-targeted generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "different classes of ribosomal RNAs (rRNAs), including 5S, 5.8S, 18S, and 28S rRNAs",
          "source_object_verbatim": "rRNA Name",
          "source_object_normalized": "rRNA class name",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "construct inputs using the template 'RNA, [rRNA Name]<seq>'",
            "generate candidate rRNA sequences",
            "align generated sequences to Rfam covariance models using cmsearch"
          ],
          "model_visible_form_verbatim": "RNA, [rRNA Name]<seq>",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "construct inputs using the template 'RNA, [rRNA Name]<seq>'",
          "fusion_topology": "prefix",
          "text_role": "modality_or_task_selector",
          "input_status": "actual_model_input",
          "evidence_quote": "matrix  was  subjected  to  hierarchical  clustering  using  Ward's  minimum-variance  method 56 ,  and  the 470",
          "section_heading": "rRNA generation",
          "supporting_figure_or_table": "Fig. 4d",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            16
          ],
          "doc_item_refs": [
            "#/texts/540",
            "#/texts/541",
            "#/texts/542",
            "#/texts/543",
            "#/texts/544",
            "#/texts/545",
            "#/texts/546",
            "#/texts/547"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f9ce1bf63b0b",
          "configuration_id": "config_f6c8b0dac2f5",
          "route_label": "Histone-targeted generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "five major histone classes, H1, H2A, H2B, H3, and H4",
          "source_object_verbatim": "histone family",
          "source_object_normalized": "histone family label",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "construct inputs using the template 'RNA, [histone]<seq>'",
            "generate RNA sequences encoding histone classes",
            "translate generated sequences and compare physicochemical properties"
          ],
          "model_visible_form_verbatim": "RNA, [histone]<seq>",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "construct inputs using the template 'RNA, [histone]<seq>'",
          "fusion_topology": "prefix",
          "text_role": "modality_or_task_selector",
          "input_status": "actual_model_input",
          "evidence_quote": "For protein-coding genes, we further generated sequences corresponding to five major histone classes, H1, H2A, H2B, H3, and H4,  and analyzed the distributions of their translated products",
          "section_heading": "Histone generation",
          "supporting_figure_or_table": "Fig. 4e-h",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9
          ],
          "doc_item_refs": [
            "#/texts/302",
            "#/texts/303",
            "#/texts/304",
            "#/texts/305",
            "#/texts/306",
            "#/texts/307",
            "#/texts/308",
            "#/texts/309"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0c90be4af21f",
          "configuration_id": "config_e1de6b1ebaa0",
          "route_label": "Sequence-function consistency evaluation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "matched and mismatched sequence-function pairings",
          "source_object_verbatim": "real gene sample nucleotide sequence",
          "source_object_normalized": "nucleotide sequence paired with native functional annotation",
          "source_modality_normalized": "text-serialized nucleotide sequence plus natural-language annotation",
          "transformation_chain_verbatim": [
            "hold the underlying nucleotide sequence fixed",
            "systematically replace the functional-annotation component of the prompt",
            "compute conditional perplexity for matched and mismatched pairings"
          ],
          "model_visible_form_verbatim": "nucleotide sequence and functional-annotation prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "the functional-annotation component of the prompt was systematically replaced",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "the functional-annotation component of the prompt was systematically replaced",
          "section_heading": "Sequence-function semantic consistency evaluation",
          "supporting_figure_or_table": "Fig. 2a-b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13,
            14,
            15,
            16
          ],
          "doc_item_refs": [
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/505",
            "#/texts/506",
            "#/texts/507",
            "#/texts/508",
            "#/texts/509",
            "#/texts/510",
            "#/texts/511",
            "#/texts/512"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_007"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000133::0005",
            "dense::june_update_2026-06-10__rec_000133::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_75b32f3d4dbc",
          "configuration_id": "config_b60819476986",
          "route_label": "Species-label perturbation evaluation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "incorrect species prompts",
          "source_object_verbatim": "RNA sequence of the ppap2d gene",
          "source_object_normalized": "fixed RNA sequence with perturbed species prompt",
          "source_modality_normalized": "text-serialized nucleotide sequence plus species label",
          "transformation_chain_verbatim": [
            "keep the sequence unchanged",
            "replace the original species prompt with labels from highly divergent species",
            "perform all pairwise species-label substitutions and measure the induced perplexity change"
          ],
          "model_visible_form_verbatim": "species prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt prefix",
          "fusion_topology": "prefix",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "g1850 g3404 g4666g1876 g2869 , g1876 g2870 , … , g1876 g3021 g4667 formed by the prompt and the sequence under evaluation, the average perplexity was 417",
          "section_heading": "Capturing evolutionary constraints and phylogenetic representations",
          "supporting_figure_or_table": "Fig. 2h-j",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            13,
            14
          ],
          "doc_item_refs": [
            "#/texts/190",
            "#/texts/191",
            "#/texts/192",
            "#/texts/193",
            "#/texts/194",
            "#/texts/195",
            "#/texts/196",
            "#/texts/197",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_008"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000133::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a732f19c65b6",
          "configuration_id": "config_6d80a87b5c59",
          "route_label": "In silico mutation scanning",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "in-silico mutation pipeline",
          "source_object_verbatim": "wild-type protein-coding gene sequences",
          "source_object_normalized": "wild-type protein-coding gene sequences",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "select protein-coding genes from the validation set",
            "introduce single-nucleotide substitutions and single-base deletions at each position in untranslated regions and coding sequence",
            "classify CDS mutations as synonymous, missense, nonsense, stop-loss, or frameshift"
          ],
          "model_visible_form_verbatim": "mutant and wild-type sequence inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence evaluation input",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "substitutions and single-base deletions at each position in the untranslated regions (5 UTR and 3 UTR) 429",
          "section_heading": "In silico mutation scanning",
          "supporting_figure_or_table": "Fig. 2e-f",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13,
            14
          ],
          "doc_item_refs": [
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/487",
            "#/texts/488",
            "#/texts/489",
            "#/texts/490",
            "#/texts/491",
            "#/texts/492",
            "#/texts/493",
            "#/texts/494"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_009"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000133::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_29d75af53bc0",
          "configuration_id": "config_06abbcccfa72",
          "route_label": "Dynamic information-masking pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "dynamic information-masking strategy during the later stage of training",
          "source_object_verbatim": "training inputs with species identifier or gene identifier in the prompt",
          "source_object_normalized": "training prompt with optional species or gene identifier",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "randomly removed the species identifier or gene identifier in the prompt",
            "applied with probability 0.5 during the later stage of training"
          ],
          "model_visible_form_verbatim": "prompt with the species/gene identifier removed",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt masking",
          "fusion_topology": "placeholder_replacement",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "randomly removed with probability 0.5",
          "section_heading": "Pretraining protocol",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            12,
            13
          ],
          "doc_item_refs": [
            "#/texts/448",
            "#/texts/449",
            "#/texts/450",
            "#/texts/451",
            "#/texts/452",
            "#/texts/453",
            "#/texts/454",
            "#/texts/455"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000133::route_010"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_aaacf24f224c"
    },
    {
      "model_id": "model_0e34c1fd11f6",
      "model_name": "Genolator",
      "record_id": "full_2026-07-06__rec_002304",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_6a167f2added",
      "paper_title": "Genolator: A Multimodal Large Language Model Fusing Natural Language, Genomic, and Structural Tokens for Protein Function Interpretation",
      "doi": "10.1101/2025.11.14.688396",
      "paper_url": "https://doi.org/10.1101/2025.11.14.688396",
      "route_count": 4,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 3,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "pooled_or_aggregated_embedding": 3,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "pooled_or_aggregated_embedding"
      ],
      "primary_subtype": "pooled_or_aggregated_embedding",
      "modalities": [
        "DNA",
        "natural language",
        "protein sequence",
        "protein structure"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "prefix"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_002304_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_002304_0f15beb21c49/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Exemplary illustration of a user assessing the molecular function of a DNA sequence with a generic query. Left: The conversation between Genolator and the user. Right: A schematic overview of the backend process handling the user's query and parsing of the provided DNA sequence.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a workflow schematic for a biological sequence/model interface system labeled “Genolator.” It is split by a vertical dashed divider into two main regions:\n\n- Left panel: “User Input via Interface”\n  - Shows a chat-style interaction between a user and Genolator.\n  - Visible user-provided DNA sequence: `ATGCGTACCTTACGCAGTACCGTTAGCATGCCGATTAACG`.\n  - The interface asks: “What types of molecular activities may be associated with the given protein?”\n  - A response bubble describes possible regulatory protein functions, including interaction with regulatory proteins, cellular signaling pathways, gene expression effects, and transcriptional regulation.\n  - Biological visuals include DNA double-helix graphics and a blue protein-like structure.\n  - User icon is shown as a scientist/doctor holding DNA.\n\n- Right panel: “Backend”\n  - Shows a model pipeline converting sequence input into embeddings and concatenated tokens.\n  - A translated amino-acid/protein sequence is shown: `MRTCERVPYSMPINT`.\n  - Components labeled:\n    - `Evo2`\n    - `ESMFold`\n    - `DNA Embedding`\n    - `AA Embedding`\n    - `Structure Embedding`\n    - `DNA Projector`\n    - `AA Projector`\n    - `Structure Projector`\n    - `Token Concatenation`\n  - Arrows indicate DNA input flows into Evo2 and sequence/structure processing; ESMFold produces a protein structure representation used for structure embedding.\n  - The final concatenated representation appears to feed back toward the interface response.\n\nBiological source objects shown include DNA sequences, amino-acid/protein sequence, DNA helices, and predicted protein structure imagery. The figure depicts transformations from DNA sequence to DNA embedding, amino-acid embedding, and protein structure embedding, followed by projection and token concatenation for a backend model that supports biological question answering.",
        "page_no": 6,
        "sha256": "7bbf665a8a1ee700b481cd1f6897108b58c035ed57d90bef04035b5fce833f43",
        "pixel_width": 903,
        "pixel_height": 557,
        "crop_box": {
          "x": 0.03,
          "y": 0.28,
          "width": 0.68,
          "height": 0.38
        },
        "panel_label": "User input to backend DNA route",
        "visible_input_object": "DNA sequence input",
        "visible_model_interface": "Evo2 -> DNA Embedding",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This is the smallest coherent crop that still shows the DNA sequence bar, the arrow into Evo2, and the resulting DNA Embedding path. It excludes the output-only response bubble and most unrelated backend branches while preserving enough labels and arrows to ground the DNA-sequence input route.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_a2c45897af77",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "user's natural language query",
          "actual_model_visible_form": "textual input tokens"
        },
        {
          "subtype_id": "pooled_or_aggregated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_958373e19abf",
          "example_input": "{gene/cell/patch embeddings}",
          "example_carrier": "mean/attention pool = one compact vector",
          "example_interface": "aggregator → generator",
          "actual_source": "DNA sequence",
          "actual_model_visible_form": "4096-dimensional embedding vector"
        }
      ],
      "routes": [
        {
          "route_id": "route_958373e19abf",
          "configuration_id": "config_dab2ca7192c5",
          "route_label": "Genolator DNA-sequence input",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "questions regarding genome functionality",
          "source_object_verbatim": "DNA sequence",
          "source_object_normalized": "DNA sequence",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "Evo2",
            "mean pooling"
          ],
          "model_visible_form_verbatim": "4096-dimensional embedding vector",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "virtual token projectors followed by token concatenation with the natural-language question",
          "fusion_topology": "prefix",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Genolator is a finetuned multimodal Llama 33-35 model, designed to fuse natural language with genomic  language.  It  receives  a  question,  formulated  in  natural  language  (English),  and information  from  DNA-sequences,  amino  acid  sequences  and  protein  structures  in  tandem. Each modality is presented to Genolator in the form of an embedding, a latent representation of the encoded information, from a machine learning model trained on the specific modality. DNA sequences are handled by the genomic foundation model Evo2 22 ,  amino acids by the protein language  model  ESM-2 36 and  protein  structure  embeddings  created  by  a  graph  autoencoder trained on protein structures 30 .    Genolator utilizes token projectors to fuse the input from the modality encoders with the natural language question and presents the output to a Llama model which  processes  the  input  and  formulates  a  response  which  will  be  returned  in  English. Genolator was designed to answer questions regarding genome functionality. To achieve this, genome  functionality  descriptions  were  derived  from  the  three  Gene  Ontology  (GO)-Term 29 aspects:",
          "section_heading": "Genolator - a multimodal large language model regarding genome functionality:",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            7,
            8,
            29,
            30
          ],
          "doc_item_refs": [
            "#/texts/113",
            "#/texts/114",
            "#/texts/116",
            "#/texts/19",
            "#/texts/20",
            "#/texts/21",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/31",
            "#/texts/33",
            "#/texts/34",
            "#/texts/35",
            "#/texts/36"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002304::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002304::0001",
            "dense::full_2026-07-06__rec_002304::0005",
            "dense::full_2026-07-06__rec_002304::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b11ae6a91236",
          "configuration_id": "config_dab2ca7192c5",
          "route_label": "Genolator amino-acid-sequence input",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "questions regarding genome functionality",
          "source_object_verbatim": "amino acid sequence",
          "source_object_normalized": "amino acid sequence",
          "source_modality_normalized": "protein sequence",
          "transformation_chain_verbatim": [
            "ESM-2",
            "mean pooling"
          ],
          "model_visible_form_verbatim": "2,560-dimensional embedding vector",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "virtual token projectors followed by token concatenation with the natural-language question",
          "fusion_topology": "prefix",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Genolator is a finetuned multimodal Llama 33-35 model, designed to fuse natural language with genomic  language.  It  receives  a  question,  formulated  in  natural  language  (English),  and information  from  DNA-sequences,  amino  acid  sequences  and  protein  structures  in  tandem. Each modality is presented to Genolator in the form of an embedding, a latent representation of the encoded information, from a machine learning model trained on the specific modality. DNA sequences are handled by the genomic foundation model Evo2 22 ,  amino acids by the protein language  model  ESM-2 36 and  protein  structure  embeddings  created  by  a  graph  autoencoder trained on protein structures 30 .    Genolator utilizes token projectors to fuse the input from the modality encoders with the natural language question and presents the output to a Llama model which  processes  the  input  and  formulates  a  response  which  will  be  returned  in  English. Genolator was designed to answer questions regarding genome functionality. To achieve this, genome  functionality  descriptions  were  derived  from  the  three  Gene  Ontology  (GO)-Term 29 aspects:",
          "section_heading": "Genolator - a multimodal large language model regarding genome functionality:",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            8
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/13",
            "#/texts/14",
            "#/texts/16",
            "#/texts/19",
            "#/texts/20",
            "#/texts/21",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/27",
            "#/texts/28",
            "#/texts/29",
            "#/texts/30"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002304::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002304::0002",
            "dense::full_2026-07-06__rec_002304::0005",
            "dense::full_2026-07-06__rec_002304::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_73685d611122",
          "configuration_id": "config_dab2ca7192c5",
          "route_label": "Genolator protein-structure input",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "questions regarding genome functionality",
          "source_object_verbatim": "protein structure",
          "source_object_normalized": "protein structure",
          "source_modality_normalized": "protein structure",
          "transformation_chain_verbatim": [
            "AlphaFold Protein Structure Database",
            "Graphein graph representation",
            "Graph Convolutional Autoencoder",
            "mean pooling"
          ],
          "model_visible_form_verbatim": "128-dimensional embedding vector",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "virtual token projectors followed by token concatenation with the natural-language question",
          "fusion_topology": "prefix",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Genolator is a finetuned multimodal Llama 33-35 model, designed to fuse natural language with genomic  language.  It  receives  a  question,  formulated  in  natural  language  (English),  and information  from  DNA-sequences,  amino  acid  sequences  and  protein  structures  in  tandem. Each modality is presented to Genolator in the form of an embedding, a latent representation of the encoded information, from a machine learning model trained on the specific modality. DNA sequences are handled by the genomic foundation model Evo2 22 ,  amino acids by the protein language  model  ESM-2 36 and  protein  structure  embeddings  created  by  a  graph  autoencoder trained on protein structures 30 .    Genolator utilizes token projectors to fuse the input from the modality encoders with the natural language question and presents the output to a Llama model which  processes  the  input  and  formulates  a  response  which  will  be  returned  in  English. Genolator was designed to answer questions regarding genome functionality. To achieve this, genome  functionality  descriptions  were  derived  from  the  three  Gene  Ontology  (GO)-Term 29 aspects:",
          "section_heading": "Genolator - a multimodal large language model regarding genome functionality:",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/13",
            "#/texts/14",
            "#/texts/16",
            "#/texts/19",
            "#/texts/20",
            "#/texts/21",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002304::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002304::0003",
            "dense::full_2026-07-06__rec_002304::0005",
            "dense::full_2026-07-06__rec_002304::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a2c45897af77",
          "configuration_id": "config_dab2ca7192c5",
          "route_label": "Genolator natural-language-question input",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "questions regarding genome functionality",
          "source_object_verbatim": "user's natural language query",
          "source_object_normalized": "natural language query",
          "source_modality_normalized": "natural language",
          "transformation_chain_verbatim": [
            "tokenization"
          ],
          "model_visible_form_verbatim": "textual input tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "prepended to the multimodal virtual-token sequence",
          "fusion_topology": "prefix",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Genolator is a finetuned multimodal Llama 33-35 model, designed to fuse natural language with genomic  language.  It  receives  a  question,  formulated  in  natural  language  (English),  and information  from  DNA-sequences,  amino  acid  sequences  and  protein  structures  in  tandem. Each modality is presented to Genolator in the form of an embedding, a latent representation of the encoded information, from a machine learning model trained on the specific modality. DNA sequences are handled by the genomic foundation model Evo2 22 ,  amino acids by the protein language  model  ESM-2 36 and  protein  structure  embeddings  created  by  a  graph  autoencoder trained on protein structures 30 .    Genolator utilizes token projectors to fuse the input from the modality encoders with the natural language question and presents the output to a Llama model which  processes  the  input  and  formulates  a  response  which  will  be  returned  in  English. Genolator was designed to answer questions regarding genome functionality. To achieve this, genome  functionality  descriptions  were  derived  from  the  three  Gene  Ontology  (GO)-Term 29 aspects:",
          "section_heading": "Genolator - a multimodal large language model regarding genome functionality:",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/13",
            "#/texts/14",
            "#/texts/16",
            "#/texts/19",
            "#/texts/20",
            "#/texts/21",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002304::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002304::0004",
            "dense::full_2026-07-06__rec_002304::0005",
            "dense::full_2026-07-06__rec_002304::0006"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_087f460aac0a"
    },
    {
      "model_id": "model_6b93a9600bf1",
      "model_name": "GP-GPT",
      "record_id": "full_2026-07-06__rec_003131",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_d3496a7d304f",
      "paper_title": "GP-GPT: Large Language Model for Gene-Phenotype Mapping",
      "doi": "",
      "paper_url": "",
      "route_count": 9,
      "configuration_count": 7,
      "family_counts": {
        "text_native_token_stream": 9
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 7,
        "plain_language_prompt_or_question": 2
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning"
      ],
      "fusion_topologies": [
        "concatenation",
        "placeholder_replacement"
      ],
      "text_roles": [
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003131_figure_003.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003131_25f480721b1e/figure_003.png",
        "figure_index": 3,
        "caption": "Figure 3: Tuning model at the first stage using instruction mask training data. The bio-text has been fitted into the input format provided by the Llama model. The signs: '### Instruction:, '### Input:', and '### Output:', stand for the indicators inside the model input. The red words indicate the replaceable gene entities and phenotype entities.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image shows a single schematic/text panel labeled **“First stage: Mask training examples”**. It presents multiple JSON-like training examples for a masked language/model training task.\n\nVisible elements:\n- A blue title banner at the top: **“First stage: Mask training examples”**.\n- A dashed rectangular border containing several separated text examples.\n- Each example contains fields resembling:\n  - `### Instruction`\n  - `### Input`\n  - `### Output`\n- The task instruction says the model should act as a bioinformatic expert and fill masked parts based on text.\n- Mask tokens are shown as orange `<mask>` placeholders.\n- Predicted/target filled tokens are highlighted in red.\n\nBiological/source objects mentioned:\n- Genes: **TPT1P13**, **SSX2IP**, **RPL3P12**, **MAGEC2**\n- Phenotype/trait: **Body Weight**\n- Gene ID: **117178**\n- Numeric identifier: **643421**\n\nModel/training interface:\n- The figure illustrates masked text completion examples for bioinformatics relation/entity reconstruction.\n- Inputs contain masked spans, and outputs show the completed phrase.\n\nVisible findings/examples:\n- “gene TPT1P13 have variant in intergenic related with **Body Weight**”\n- “ID for gene SSX2IP is **117178**”\n- “gene name for 643421 is **RPL3P12**”\n- “evidence show **Body Weight** is phenotype relate with **gene TPT1P13**”\n- “gene **MAGEC2** have variant in intergenic related with **Body Weight**”",
        "page_no": 10,
        "sha256": "b6c7d3ad60e9a883cc825bdaddf18f55970b97e6dc13d4a401f49c6d5291651b",
        "pixel_width": 931,
        "pixel_height": 1123,
        "crop_box": {
          "x": 0.02,
          "y": 0.059,
          "width": 0.962,
          "height": 0.205
        },
        "panel_label": "First stage: Mask training examples",
        "visible_input_object": "Masked genomics text example with gene/phenotype entities and <mask> tokens",
        "visible_model_interface": "Instruction / Input / Output scaffold with ### Instruction, ### Input, and ### Output labels",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the first masked-text training example and its prompt scaffold, which grounds the stage-1 masked genomics text route while excluding later examples that are redundant for understanding the interface.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_504a9017e136",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "disease name",
          "actual_model_visible_form": "question-answer instruction"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_84fcd76e9fd3",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene entities data and phenotype/disease entities data from the dbGaP",
          "actual_model_visible_form": "masked text data with <mask> tokens"
        }
      ],
      "routes": [
        {
          "route_id": "route_84fcd76e9fd3",
          "configuration_id": "config_d20ec823eb71",
          "route_label": "Stage 1 masked genomics text",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "instruction mask prediction fine-tuning",
          "source_object_verbatim": "gene entities data and phenotype/disease entities data from the dbGaP",
          "source_object_normalized": "dbGaP gene and phenotype/disease entity data",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "organized into masked text data"
          ],
          "model_visible_form_verbatim": "masked text data with <mask> tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction mask prediction fine-tuning",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "gene entities data and phenotype/disease entities data from the dbGaP are organized into masked text data",
          "section_heading": "Construction of the Multi-task and Multi-level Genomics Training Corpus",
          "supporting_figure_or_table": "Figure 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/60",
            "#/texts/62",
            "#/texts/63",
            "#/texts/64",
            "#/texts/78",
            "#/texts/79",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003131::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003131::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e0db3750e2c7",
          "configuration_id": "config_0e04c1ed51e4",
          "route_label": "Protein-function inferring",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "protein-function inferring",
          "source_object_verbatim": "protein entities and their functions",
          "source_object_normalized": "protein entity and function context",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "protein entities and their functions"
          ],
          "model_visible_form_verbatim": "Task Prompt, Input, and Output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "question-answer format",
          "fusion_topology": "concatenation",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Protein-function inferring formulated both protein entities and their functions.",
          "section_heading": "Construction of the Multi-task and Multi-level Genomics Training Corpus",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/60",
            "#/texts/62",
            "#/texts/63",
            "#/texts/64",
            "#/texts/78",
            "#/texts/79",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003131::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003131::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9e462c7819b0",
          "configuration_id": "config_4897976b13b9",
          "route_label": "Gene-function inferring",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "gene-function inferring",
          "source_object_verbatim": "gene entity and its functions",
          "source_object_normalized": "gene entity and function context",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "gene entity and its functions"
          ],
          "model_visible_form_verbatim": "Task Prompt, Input, and Output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "question-answer format",
          "fusion_topology": "concatenation",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Gene-function inferring formulated the gene entity and its functions.",
          "section_heading": "Construction of the Multi-task and Multi-level Genomics Training Corpus",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/60",
            "#/texts/62",
            "#/texts/63",
            "#/texts/64",
            "#/texts/78",
            "#/texts/79",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003131::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003131::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e37e280eb801",
          "configuration_id": "config_2e42258e0096",
          "route_label": "Protein molecular features",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "gene-protein-phenotype/disease contexts",
          "source_object_verbatim": "protein molecular features",
          "source_object_normalized": "protein molecular feature context",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "protein entity, the phenotype, and the molecular relations"
          ],
          "model_visible_form_verbatim": "Task Prompt, Input, and Output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "question-answer format",
          "fusion_topology": "concatenation",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Protein-molecular features formulated the protein entity, the phenotype, and the molecular relations",
          "section_heading": "Construction of the Multi-task and Multi-level Genomics Training Corpus",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9,
            10,
            11,
            12,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/pictures/3",
            "#/pictures/4",
            "#/pictures/5",
            "#/texts/60",
            "#/texts/62",
            "#/texts/63",
            "#/texts/64",
            "#/texts/66",
            "#/texts/68",
            "#/texts/70",
            "#/texts/78",
            "#/texts/79",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003131::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003131::0004",
            "dense::full_2026-07-06__rec_003131::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_03df9dd1f40a",
          "configuration_id": "config_2e42258e0096",
          "route_label": "Protein pathogenesis features",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "gene-protein-phenotype/disease contexts",
          "source_object_verbatim": "protein pathogenesis features",
          "source_object_normalized": "protein pathogenesis feature context",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "protein entity, the phenotype, and the pathogenesis relations"
          ],
          "model_visible_form_verbatim": "Task Prompt, Input, and Output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "question-answer format",
          "fusion_topology": "concatenation",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Protein-pathogenesis features formulated the protein entity, the phenotype, and the pathogenesis relations",
          "section_heading": "Construction of the Multi-task and Multi-level Genomics Training Corpus",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9,
            10,
            11,
            12,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/pictures/3",
            "#/pictures/4",
            "#/pictures/5",
            "#/texts/60",
            "#/texts/62",
            "#/texts/63",
            "#/texts/64",
            "#/texts/66",
            "#/texts/68",
            "#/texts/70",
            "#/texts/78",
            "#/texts/79",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003131::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003131::0004",
            "dense::full_2026-07-06__rec_003131::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_eb37ae7bc053",
          "configuration_id": "config_2e42258e0096",
          "route_label": "Gene inheritance features",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "gene-protein-phenotype/disease contexts",
          "source_object_verbatim": "gene inheritance features",
          "source_object_normalized": "gene inheritance feature context",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "gene entity and inheritance facts"
          ],
          "model_visible_form_verbatim": "Task Prompt, Input, and Output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "question-answer format",
          "fusion_topology": "concatenation",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Gene-inheritance features formulated the gene entity and inheritance facts.",
          "section_heading": "Construction of the Multi-task and Multi-level Genomics Training Corpus",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9,
            10,
            11,
            12,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/pictures/3",
            "#/pictures/4",
            "#/pictures/5",
            "#/texts/60",
            "#/texts/62",
            "#/texts/63",
            "#/texts/64",
            "#/texts/66",
            "#/texts/68",
            "#/texts/70",
            "#/texts/78",
            "#/texts/79",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003131::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003131::0004",
            "dense::full_2026-07-06__rec_003131::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_504a9017e136",
          "configuration_id": "config_5d23996acb2c",
          "route_label": "Gene-disease QA",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "question-answer tasks",
          "source_object_verbatim": "disease name",
          "source_object_normalized": "disease name",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "Gene disease association questions from the QA database GeneTuring",
            "question created by pasting 'What are genes related to ' before the disease name and a question mark after the disease name"
          ],
          "model_visible_form_verbatim": "question-answer instruction",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "question-answer paradigm",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "A question was created by pasting 'What are genes related to ' before the disease name and a question mark after the disease name.",
          "section_heading": "3.3 Evaluation",
          "supporting_figure_or_table": "Figure 7",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            16,
            17,
            18
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/tables/0",
            "#/texts/104",
            "#/texts/106",
            "#/texts/107",
            "#/texts/109",
            "#/texts/87",
            "#/texts/89",
            "#/texts/90",
            "#/texts/91",
            "#/texts/92"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003131::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003131::0010",
            "dense::full_2026-07-06__rec_003131::0011",
            "dense::full_2026-07-06__rec_003131::0015"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5c3299a14655",
          "configuration_id": "config_b915bb133de4",
          "route_label": "Relation determination",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "relation determination task",
          "source_object_verbatim": "gene-disease pairs from TBGA data",
          "source_object_normalized": "TBGA gene-disease pairs",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "instruction and text",
            "generate answers in yes/no based on the text"
          ],
          "model_visible_form_verbatim": "instruction and text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "yes/no text generation",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The relation determination prompt is composed of instruction and text, models are asked to generate answers in yes/no based on the text.",
          "section_heading": "3.3 Evaluation",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3,
            15,
            16,
            17,
            18,
            19
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/tables/0",
            "#/tables/1",
            "#/texts/104",
            "#/texts/106",
            "#/texts/107",
            "#/texts/109",
            "#/texts/111",
            "#/texts/116",
            "#/texts/18",
            "#/texts/19",
            "#/texts/21",
            "#/texts/87",
            "#/texts/94",
            "#/texts/96"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003131::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003131::0010",
            "dense::full_2026-07-06__rec_003131::0012",
            "dense::full_2026-07-06__rec_003131::0015"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4449868be025",
          "configuration_id": "config_46e064a3c0d1",
          "route_label": "Gene-disease sentence completion",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "sentence completion tasks",
          "source_object_verbatim": "disease name",
          "source_object_normalized": "disease name",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "question created by pasting 'The name of the gene related to ' before the disease name and ' is' after the disease name"
          ],
          "model_visible_form_verbatim": "sentence completion instruction",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "question-answer paradigm",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "To create sentence completion tasks, a question was created by pasting 'The name of the gene related to ' before the disease name and ' is' after the disease name.",
          "section_heading": "3.3 Evaluation",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            16,
            17,
            18
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/tables/0",
            "#/texts/104",
            "#/texts/106",
            "#/texts/107",
            "#/texts/109",
            "#/texts/87",
            "#/texts/89",
            "#/texts/90",
            "#/texts/91",
            "#/texts/92"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003131::route_010"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003131::0010",
            "dense::full_2026-07-06__rec_003131::0015"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_85e0c6746826"
    },
    {
      "model_id": "model_72e70c356b7a",
      "model_name": "GPT-2 Medium",
      "record_id": "full_2026-07-06__rec_003206",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_dd3851edd677",
      "paper_title": "Cell2Sentence: teaching large language models the language of biology",
      "doi": "",
      "paper_url": "",
      "route_count": 6,
      "configuration_count": 6,
      "family_counts": {
        "text_native_token_stream": 6
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 1,
        "structured_biological_prompt_or_task_scaffold": 2,
        "serialized_biological_context_or_ordered_profile": 3
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "single-cell transcriptomics",
        "text"
      ],
      "lifecycle_phases": [
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003206_figure_006.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003206_211e7d00b696/figure_006.png",
        "figure_index": 6,
        "caption": "Figure 6: Depiction of three types of prompts used during training and generation. From left to right: unconditional cell generation, conditional cell generation (e.g.: with cell type), and autoregressive cell type prediction. For our setup, we generate and prompt with 100 genes.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image contains three side-by-side rounded panels describing model tasks for single-cell gene expression analysis.\n\nPanel 1: “Unconditional cell generation”\n- Shows a model interface with:\n  - Model input: “Rank the first 100 genes by expression level found in a cell.”\n  - Model output: a ranked gene list including IGKV2-14, IGHV4-4, ENSG00000275063.1, MT-CO1, MTRNR2L12, JUN, HSPA1A, IGHG1, HSPD1, IGHG3, DNAJB1, NFKBIA, TPT1, MT-CO2, HSP90AA1, TXNDC5, MT-ATP6, FTL, HSPA8, FKBP11, ZFP36, HSPH1, RPS14, RPS12, RPS27, and others.\n\nPanel 2: “Conditional cell generation”\n- Shows a model interface with:\n  - Model input: “Provide a ranking of the top 100 expressed genes in a CD4 T-cell in PBMC by decreasing levels.”\n  - Model output: a ranked gene list including IGHV3-7, MALAT1, MT-CO2, JCHAIN, IGLV-70, MT-CYB, HSPA6, MT-CO1, TMSB4X, MT-ATP6, MT-ND3, B2M, MT-CO3, MT-ND4L, DNAJB1, UBC, MTRNR2L12, ACTB, RPL41, SSR4, HSP90AA1, HSPA1B, HSPA1A, and others.\n\nPanel 3: “Cell type prediction”\n- Shows a model interface with:\n  - Model input: asks to pinpoint the cell type associated with 100 genes with highest expression, listing genes such as IGHV4-59, IGHG1, HSPA5, HSPA1A, MT-CO1, MT-CO2, MTRNR2L12, HSP90B1, IGKV3-20, RPLP1, MT-CYB, B2M.\n  - Model output: “macrophage.”\n\nBiological source objects: gene expression-ranked cells, PBMC, CD4 T-cell, and predicted macrophage cell type.\n\nTransformations/model interfaces: text-prompt inputs are mapped to text outputs for unconditional gene ranking generation, conditional gene ranking generation, and cell type prediction from marker/high-expression genes.",
        "page_no": 11,
        "sha256": "3ac6412c3ba2e32cc2c4da4a1c866b22cb245dd1542282de1967aa3d5d47096d",
        "pixel_width": 796,
        "pixel_height": 180,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 0.333,
          "height": 1
        },
        "panel_label": "Unconditional cell generation",
        "visible_input_object": "natural language prompt: 'Rank the first 100 genes by expression level found in a cell.'",
        "visible_model_interface": "text prompt to ranked gene-list output",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "The left panel is self-contained and grounds the unconditional text-prompt route: it shows the model input prompt and the generated ranked gene list, while excluding the other two panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_21eb7f55a5cb",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "natural language prompt",
          "actual_model_visible_form": "text prompt"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_f2aef6e12c3a",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "sequence of 100 genes",
          "actual_model_visible_form": "prompt containing a sequence of 100 genes"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_ee1f4dbde15d",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "cell type label",
          "actual_model_visible_form": "text prompt containing a cell type label"
        }
      ],
      "routes": [
        {
          "route_id": "route_21eb7f55a5cb",
          "configuration_id": "config_ee0bdb70b269",
          "route_label": "GPT-2 Medium unconditional cell generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "produce a sequence of 100 genes without any prescribed cell type label",
          "source_object_verbatim": "natural language prompt",
          "source_object_normalized": "natural language instruction prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "prompt template selection"
          ],
          "model_visible_form_verbatim": "text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "formatted as prompts",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "produce a sequence of 100 genes without any prescribed cell type label",
          "section_heading": "3.2.1 Tasks",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9
          ],
          "doc_item_refs": [
            "#/texts/100",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ee1f4dbde15d",
          "configuration_id": "config_4a11406ae46f",
          "route_label": "GPT-2 Medium conditional cell generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "generate a sequence of 100 genes given a specific cell type label",
          "source_object_verbatim": "cell type label",
          "source_object_normalized": "cell type label",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "prompt template selection",
            "merge prompt with the specified cell type"
          ],
          "model_visible_form_verbatim": "text prompt containing a cell type label",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "merge prompt with the specified cell type",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "generate a sequence of 100 genes given a specific cell type label",
          "section_heading": "3.2.1 Tasks",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9
          ],
          "doc_item_refs": [
            "#/texts/100",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f2aef6e12c3a",
          "configuration_id": "config_344f483c8b5d",
          "route_label": "GPT-2 Medium cell type prediction",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "generate a cell type label following a provided sequence of 100 genes",
          "source_object_verbatim": "sequence of 100 genes",
          "source_object_normalized": "ranked gene sequence",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "prompt template selection",
            "combine the prompt with the cell sentence"
          ],
          "model_visible_form_verbatim": "prompt containing a sequence of 100 genes",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "combine the prompt with the cell sentence",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "generate a cell type label following a provided sequence of 100 genes",
          "section_heading": "3.2.1 Tasks",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9
          ],
          "doc_item_refs": [
            "#/texts/100",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_aaa39b0c5aa3",
          "configuration_id": "config_a38519ecf168",
          "route_label": "GPT-2 Medium Cell2Sentence encoding",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Cell2Sentence transformation",
          "source_object_verbatim": "single-cell gene expression count matrix",
          "source_object_normalized": "single-cell gene expression profile",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "filter cells and genes",
            "quality control",
            "row-normalization to 10,000 transcript counts",
            "log-normalization",
            "rank-order transformation",
            "truncate to top 100 genes"
          ],
          "model_visible_form_verbatim": "cell sentences",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "converted into cell sentences",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Figure 1: Overview of the Cell2Sentence framework. Input single-cell data, including metadata, are converted into cell sentences for LLM fine-tuning. Inference, via prompting, generates new cell sentences that can be converted back to gene expression space.",
          "section_heading": "1 Introduction",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            3
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/40",
            "#/texts/42",
            "#/texts/43"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003206::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d18659a37757",
          "configuration_id": "config_0575720b9f98",
          "route_label": "GPT-2 Medium metadata-conditioned cell generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "generate cells from text prompts",
          "source_object_verbatim": "natural language prompt containing cell type, tissue, or disease metadata",
          "source_object_normalized": "biologically conditioned natural language prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "prompt template selection",
            "merge prompt with the specified biological metadata"
          ],
          "model_visible_form_verbatim": "text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "formatted as prompts",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Model Input: “T-cell in multiple sclerosis”",
          "section_heading": null,
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": "Shown only as an overview-figure prompting example rather than a separately benchmarked task form.",
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/40",
            "#/texts/42",
            "#/texts/43"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_010"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5ff04e63ef39",
          "configuration_id": "config_66d3fd06527c",
          "route_label": "GPT-2 Medium metadata-annotated Cell2Sentence encoding",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Cell2Sentence transformation with biological metadata annotations",
          "source_object_verbatim": "single-cell data including metadata annotations",
          "source_object_normalized": "single-cell gene expression profile with metadata annotations",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "filter cells and genes",
            "quality control",
            "row-normalization to 10,000 transcript counts",
            "log-normalization",
            "rank-order transformation",
            "attach biological metadata such as cell type, tissue, or disease",
            "truncate to top 100 genes"
          ],
          "model_visible_form_verbatim": "cell sentences with biological metadata",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "converted into cell sentences",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Cell sentences may be annotated with biological metadata, such as cell type, tissue, or disease.",
          "section_heading": "2 Results",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes metadata annotation generically; the exact serialization of the metadata into the prompt is not fully specified.",
          "pages": [
            3,
            4
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/51",
            "#/texts/54"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_012"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_67d200027193"
    },
    {
      "model_id": "model_6694aa9f2590",
      "model_name": "GPT-2 Small",
      "record_id": "full_2026-07-06__rec_003206",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_dd3851edd677",
      "paper_title": "Cell2Sentence: teaching large language models the language of biology",
      "doi": "",
      "paper_url": "",
      "route_count": 6,
      "configuration_count": 6,
      "family_counts": {
        "text_native_token_stream": 6
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 1,
        "structured_biological_prompt_or_task_scaffold": 2,
        "serialized_biological_context_or_ordered_profile": 3
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "single-cell transcriptomics",
        "text"
      ],
      "lifecycle_phases": [
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003206_figure_006.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003206_211e7d00b696/figure_006.png",
        "figure_index": 6,
        "caption": "Figure 6: Depiction of three types of prompts used during training and generation. From left to right: unconditional cell generation, conditional cell generation (e.g.: with cell type), and autoregressive cell type prediction. For our setup, we generate and prompt with 100 genes.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image contains three side-by-side rounded panels describing model tasks for single-cell gene expression analysis.\n\nPanel 1: “Unconditional cell generation”\n- Shows a model interface with:\n  - Model input: “Rank the first 100 genes by expression level found in a cell.”\n  - Model output: a ranked gene list including IGKV2-14, IGHV4-4, ENSG00000275063.1, MT-CO1, MTRNR2L12, JUN, HSPA1A, IGHG1, HSPD1, IGHG3, DNAJB1, NFKBIA, TPT1, MT-CO2, HSP90AA1, TXNDC5, MT-ATP6, FTL, HSPA8, FKBP11, ZFP36, HSPH1, RPS14, RPS12, RPS27, and others.\n\nPanel 2: “Conditional cell generation”\n- Shows a model interface with:\n  - Model input: “Provide a ranking of the top 100 expressed genes in a CD4 T-cell in PBMC by decreasing levels.”\n  - Model output: a ranked gene list including IGHV3-7, MALAT1, MT-CO2, JCHAIN, IGLV-70, MT-CYB, HSPA6, MT-CO1, TMSB4X, MT-ATP6, MT-ND3, B2M, MT-CO3, MT-ND4L, DNAJB1, UBC, MTRNR2L12, ACTB, RPL41, SSR4, HSP90AA1, HSPA1B, HSPA1A, and others.\n\nPanel 3: “Cell type prediction”\n- Shows a model interface with:\n  - Model input: asks to pinpoint the cell type associated with 100 genes with highest expression, listing genes such as IGHV4-59, IGHG1, HSPA5, HSPA1A, MT-CO1, MT-CO2, MTRNR2L12, HSP90B1, IGKV3-20, RPLP1, MT-CYB, B2M.\n  - Model output: “macrophage.”\n\nBiological source objects: gene expression-ranked cells, PBMC, CD4 T-cell, and predicted macrophage cell type.\n\nTransformations/model interfaces: text-prompt inputs are mapped to text outputs for unconditional gene ranking generation, conditional gene ranking generation, and cell type prediction from marker/high-expression genes.",
        "page_no": 11,
        "sha256": "3ac6412c3ba2e32cc2c4da4a1c866b22cb245dd1542282de1967aa3d5d47096d",
        "pixel_width": 796,
        "pixel_height": 180,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.333,
          "height": 0.47
        },
        "panel_label": "Unconditional cell generation",
        "visible_input_object": "Natural language prompt: “Rank the first 100 genes by expression level found in a cell.”",
        "visible_model_interface": "Text prompt",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This top-left crop preserves the panel title and the model input label/prompt for the unconditional cell-generation route, which is enough to identify the source object and model-visible carrier while excluding the other panels and most output text.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_be99d57da557",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "natural language prompt",
          "actual_model_visible_form": "text prompt"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_818f0e3128ff",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "sequence of 100 genes",
          "actual_model_visible_form": "prompt containing a sequence of 100 genes"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_8ec148f22a65",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "cell type label",
          "actual_model_visible_form": "text prompt containing a cell type label"
        }
      ],
      "routes": [
        {
          "route_id": "route_be99d57da557",
          "configuration_id": "config_f40fb484227c",
          "route_label": "GPT-2 Small unconditional cell generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "produce a sequence of 100 genes without any prescribed cell type label",
          "source_object_verbatim": "natural language prompt",
          "source_object_normalized": "natural language instruction prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "prompt template selection"
          ],
          "model_visible_form_verbatim": "text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "formatted as prompts",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "produce a sequence of 100 genes without any prescribed cell type label",
          "section_heading": "3.2.1 Tasks",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9
          ],
          "doc_item_refs": [
            "#/texts/100",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8ec148f22a65",
          "configuration_id": "config_9434f2689e74",
          "route_label": "GPT-2 Small conditional cell generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "generate a sequence of 100 genes given a specific cell type label",
          "source_object_verbatim": "cell type label",
          "source_object_normalized": "cell type label",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "prompt template selection",
            "merge prompt with the specified cell type"
          ],
          "model_visible_form_verbatim": "text prompt containing a cell type label",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "merge prompt with the specified cell type",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "generate a sequence of 100 genes given a specific cell type label",
          "section_heading": "3.2.1 Tasks",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9
          ],
          "doc_item_refs": [
            "#/texts/100",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_818f0e3128ff",
          "configuration_id": "config_e5cd43ffa358",
          "route_label": "GPT-2 Small cell type prediction",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "generate a cell type label following a provided sequence of 100 genes",
          "source_object_verbatim": "sequence of 100 genes",
          "source_object_normalized": "ranked gene sequence",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "prompt template selection",
            "combine the prompt with the cell sentence"
          ],
          "model_visible_form_verbatim": "prompt containing a sequence of 100 genes",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "combine the prompt with the cell sentence",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "generate a cell type label following a provided sequence of 100 genes",
          "section_heading": "3.2.1 Tasks",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9
          ],
          "doc_item_refs": [
            "#/texts/100",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6e6e12bf204d",
          "configuration_id": "config_9ebf23b25393",
          "route_label": "GPT-2 Small Cell2Sentence encoding",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Cell2Sentence transformation",
          "source_object_verbatim": "single-cell gene expression count matrix",
          "source_object_normalized": "single-cell gene expression profile",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "filter cells and genes",
            "quality control",
            "row-normalization to 10,000 transcript counts",
            "log-normalization",
            "rank-order transformation",
            "truncate to top 100 genes"
          ],
          "model_visible_form_verbatim": "cell sentences",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "converted into cell sentences",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Figure 1: Overview of the Cell2Sentence framework. Input single-cell data, including metadata, are converted into cell sentences for LLM fine-tuning. Inference, via prompting, generates new cell sentences that can be converted back to gene expression space.",
          "section_heading": "1 Introduction",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            3
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/40",
            "#/texts/42",
            "#/texts/43"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003206::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_76a243c3cd9d",
          "configuration_id": "config_b4e2ce033c22",
          "route_label": "GPT-2 Small metadata-conditioned cell generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "generate cells from text prompts",
          "source_object_verbatim": "natural language prompt containing cell type, tissue, or disease metadata",
          "source_object_normalized": "biologically conditioned natural language prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "prompt template selection",
            "merge prompt with the specified biological metadata"
          ],
          "model_visible_form_verbatim": "text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "formatted as prompts",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Model Input: “T-cell in multiple sclerosis”",
          "section_heading": null,
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": "Shown only as an overview-figure prompting example rather than a separately benchmarked task form.",
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/40",
            "#/texts/42",
            "#/texts/43"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_009"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_95b1852fc5f4",
          "configuration_id": "config_edc07f5d6e29",
          "route_label": "GPT-2 Small metadata-annotated Cell2Sentence encoding",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Cell2Sentence transformation with biological metadata annotations",
          "source_object_verbatim": "single-cell data including metadata annotations",
          "source_object_normalized": "single-cell gene expression profile with metadata annotations",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "filter cells and genes",
            "quality control",
            "row-normalization to 10,000 transcript counts",
            "log-normalization",
            "rank-order transformation",
            "attach biological metadata such as cell type, tissue, or disease",
            "truncate to top 100 genes"
          ],
          "model_visible_form_verbatim": "cell sentences with biological metadata",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "converted into cell sentences",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Cell sentences may be annotated with biological metadata, such as cell type, tissue, or disease.",
          "section_heading": "2 Results",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes metadata annotation generically; the exact serialization of the metadata into the prompt is not fully specified.",
          "pages": [
            3,
            4
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/51",
            "#/texts/54"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003206::route_011"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_67d200027193"
    },
    {
      "model_id": "model_1fb19463c978",
      "model_name": "GPT-4",
      "record_id": "full_2026-07-06__rec_002105",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_266ca9b669f4",
      "paper_title": "Multimodal learning of transcriptomes and text enables interactive single-cell RNA-seq data exploration with natural-language chats",
      "doi": "10.1101/2024.10.15.618501",
      "paper_url": "https://doi.org/10.1101/2024.10.15.618501",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "inference"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "metadata_or_context"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "No panel visibly shows the GPT-4 disease-validation metadata annotation input path. Figure 1 is only a broad workflow overview, and the other figures are downstream evaluation or interaction panels, so none responsibly supports this exact route.",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_50dfc1a52d64",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "metadata for 14,112 primary tissue samples annotated with a disease state",
          "actual_model_visible_form": "metadata prompt with disease-state context"
        }
      ],
      "routes": [
        {
          "route_id": "route_50dfc1a52d64",
          "configuration_id": "config_52b7f20c931a",
          "route_label": "Disease validation metadata annotation with GPT-4",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "zero-shot prompting for the validation dataset",
          "source_object_verbatim": "metadata for 14,112 primary tissue samples annotated with a disease state",
          "source_object_normalized": "disease-annotated sample metadata",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "obtained 14,112 primary tissue samples that were annotated with a disease state",
            "manually curated metadata obtained from SRA, GEO, and PubMed",
            "prepared textual annotations from metadata downloaded via the Entrez API"
          ],
          "model_visible_form_verbatim": "metadata prompt with disease-state context",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "OpenAI API with zero-shot prompting",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "pipelines (Ewels et al. 2020; Patel, Ewels, et al. 2024; Patel, Garcia, et al. 2024) for preparing the transcriptome 415 data, whereas textual annotations were prepared from metadata downloaded via the Entrez API (biopython) and GPT-4 using the OpenAI API with zero-shot prompting. Because this dataset contains multiple samples with highly overlapping textual annotations, we additionally derived a deduplicated disease validation dataset for use in retrieval scoring. To that end, we processed all 14,112 textual annotations with BioBERT, performed",
          "section_heading": "Validation dataset. To guide model development, we derived a thematically coherent disease validation dataset from GEO.",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10
          ],
          "doc_item_refs": [
            "#/texts/159",
            "#/texts/160",
            "#/texts/161",
            "#/texts/162",
            "#/texts/163",
            "#/texts/164",
            "#/texts/165",
            "#/texts/166"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002105::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_2b474a3814ef",
      "model_name": "GPT-4",
      "record_id": "full_2026-07-06__rec_000827",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_f500772cfe83",
      "paper_title": "Multimodal learning enables chat-based exploration of single-cell data",
      "doi": "10.1038/s41587-025-02857-9",
      "paper_url": "https://doi.org/10.1038/s41587-025-02857-9",
      "route_count": 4,
      "configuration_count": 2,
      "family_counts": {
        "text_native_token_stream": 4
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1,
        "serialized_biological_context_or_ordered_profile": 3
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "RNA",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "metadata_or_context"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "The crop shows a dataset-to-LLM fine-tuning diagram and a Mistral 7B block, not GPT-4 or the transcriptome textual-annotation input required by the claimed route.",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_74816136a511",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "the transcriptome's top 50 most highly expressed genes",
          "actual_model_visible_form": "gene-list text prompt"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_aeb5e2abdcbf",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "metadata downloaded through the Entrez API",
          "actual_model_visible_form": "metadata text"
        }
      ],
      "routes": [
        {
          "route_id": "route_aeb5e2abdcbf",
          "configuration_id": "config_b7484e645621",
          "route_label": "Human Diseases metadata annotation preparation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "prepare textual annotations from metadata downloaded through the Entrez API",
          "source_object_verbatim": "metadata downloaded through the Entrez API",
          "source_object_normalized": "Entrez metadata",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "Entrez API metadata download",
            "zero-shot GPT-4 prompting",
            "textual annotation preparation"
          ],
          "model_visible_form_verbatim": "metadata text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "using GPT-4 through the OpenAI API with zero-shot prompting",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "using GPT-4 through the OpenAI API with zero-shot prompting",
          "section_heading": "Collection and curation of evaluation and demonstration data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            12,
            13
          ],
          "doc_item_refs": [
            "#/texts/1100",
            "#/texts/1101",
            "#/texts/1105",
            "#/texts/1106",
            "#/texts/1107",
            "#/texts/1108",
            "#/texts/1109",
            "#/texts/1110"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_74816136a511",
          "configuration_id": "config_7538ed7a00b0",
          "route_label": "top-gene conversation generation",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "generate chat-like conversations about transcriptomes and cell states",
          "source_object_verbatim": "the transcriptome's top 50 most highly expressed genes",
          "source_object_normalized": "top 50 most highly expressed genes",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "top 50 most highly expressed genes",
            "prompt construction for conversation generation"
          ],
          "model_visible_form_verbatim": "gene-list text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "on the basis of the transcriptome's top 50 most highly expressed genes",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "the transcriptome's top 50 most highly expressed genes",
          "section_heading": "Multimodal training dataset of chat conversations",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The paper states GPT-4 or Mixtral 8x7b could be used; this route captures the GPT-4 branch named in the candidate.",
          "pages": [
            14
          ],
          "doc_item_refs": [
            "#/texts/1128",
            "#/texts/1129",
            "#/texts/1130",
            "#/texts/1131",
            "#/texts/1132"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000827::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e0daee322d2c",
          "configuration_id": "config_7538ed7a00b0",
          "route_label": "GSVA-derived gene-set conversation generation",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "generate chat-like conversations about transcriptomes and cell states",
          "source_object_verbatim": "the top 50 GSVA-derived gene sets",
          "source_object_normalized": "top 50 GSVA-derived gene sets",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "top 50 GSVA-derived gene sets",
            "prompt construction for conversation generation"
          ],
          "model_visible_form_verbatim": "gene-set label text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "on the basis of the top 50 GSVA-derived gene sets",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "the top 50 GSVA-derived gene sets",
          "section_heading": "Multimodal training dataset of chat conversations",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The paper states GPT-4 or Mixtral 8x7b could be used; this route captures the GPT-4 branch named in the candidate.",
          "pages": [
            14
          ],
          "doc_item_refs": [
            "#/texts/1128",
            "#/texts/1129",
            "#/texts/1130",
            "#/texts/1131",
            "#/texts/1132"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_010"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000827::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_7961776cdf5d",
          "configuration_id": "config_7538ed7a00b0",
          "route_label": "textual annotation conversation generation",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "generate chat-like conversations about transcriptomes and cell states",
          "source_object_verbatim": "the transcriptome's textual annotation",
          "source_object_normalized": "textual annotation",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "transcriptome textual annotation",
            "prompt construction for conversation generation"
          ],
          "model_visible_form_verbatim": "textual annotation text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "on the basis of the transcriptome's textual annotation",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "the transcriptome's textual annotation",
          "section_heading": "Multimodal training dataset of chat conversations",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The paper states GPT-4 or Mixtral 8x7b could be used; this route captures the GPT-4 branch named in the candidate.",
          "pages": [
            14
          ],
          "doc_item_refs": [
            "#/texts/1128",
            "#/texts/1129",
            "#/texts/1130",
            "#/texts/1131",
            "#/texts/1132"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_011"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000827::0003"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_2bfb9536cf22"
    },
    {
      "model_id": "model_36c6451d1b55",
      "model_name": "GPT-4o",
      "record_id": "full_2026-07-06__rec_000827",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_f500772cfe83",
      "paper_title": "Multimodal learning enables chat-based exploration of single-cell data",
      "doi": "10.1038/s41587-025-02857-9",
      "paper_url": "https://doi.org/10.1038/s41587-025-02857-9",
      "route_count": 2,
      "configuration_count": 2,
      "family_counts": {
        "text_native_token_stream": 2
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 1,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query",
        "metadata_or_context"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_000827_figure_019.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_000827_745a8d0e1225/figure_019.png",
        "figure_index": 19,
        "caption": "Extended Data Fig. 2 | See next page for caption.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific workflow/analysis figure about text-embedding variants and CellWhisperer scoring.\n\nVisible panels and labels:\n- Left panel: A workflow box titled “Annotation queries” containing four colored query examples:\n  - “The embryonic stage begins with a fertilized oocyte [...]”\n  - “The blastula stage begins with the loss of the zona [...]”\n  - “During the gastrula stage the extraembryonic mesoderm [...]”\n  - “During organogenesis somites progressively form neural [...]”\n- An arrow labeled “LLM” transforms these annotation queries into a lower box titled “Query variants”, showing multiple colored dot variants and example generated paraphrases such as:\n  - “Upon fertilization, an oocyte that contains [...]”\n  - “The onset of the embryonic stage is marked by [...]”\n- A rightward arrow labeled “CellWhisperer embedding model” leads to a scatter plot titled “densMAP of variant text embeddings”.\n- Middle panel: densMAP scatter plot with axes “UMAP 1” and “UMAP 2”. Points are colored by query group; visible clusters separate red/yellow points in the upper-right, cyan points lower-middle, and purple points lower-left.\n- Another rightward arrow labeled “CellWhisperer scoring of Human Development dataset” leads to a heatmap.\n- Right panel: Heatmap titled “Pairwise correlation across all cells in the Human Development dataset”. It shows a clustered correlation matrix with dendrograms on the left and bottom, colored side annotations matching query groups, and a colorbar labeled “Correlation between CellWhisperer scores” ranging approximately from -1 to 1.\n\nBiological/source objects:\n- Human Development dataset.\n- Developmental-stage annotation text mentioning embryonic stage, blastula, gastrula, organogenesis, fertilized oocyte, zona, extraembryonic mesoderm, somites, and neural development.\n- The figure does not show biological images or cells directly; it analyzes text-derived scores across cells.\n\nTransformations/model interfaces:\n- Annotation queries are expanded into LLM-generated query variants.\n- Query variants are embedded with the CellWhisperer embedding model.\n- Variant text embeddings are visualized using densMAP/UMAP.\n- CellWhisperer scores are computed on a Human Development dataset.\n- Pairwise correlations of CellWhisperer scores across all cells are clustered and shown as a heatmap.\n\nVisible finding:\n- Query variants form distinct embedding clusters corresponding to the colored annotation groups.\n- The correlation heatmap shows block structure, indicating that variants within the same or related query groups have similar CellWhisperer score patterns, while some groups are negatively or weakly correlated with others.",
        "page_no": 18,
        "sha256": "23652830e882b36a6fb2f7d6761d5467f750e562d1164a5a3f86b8fe29517d6a",
        "pixel_width": 1011,
        "pixel_height": 260,
        "crop_box": {
          "x": 0.0,
          "y": 0.04,
          "width": 0.38,
          "height": 0.82
        },
        "panel_label": "left workflow panel",
        "visible_input_object": "Annotation queries and query variants text boxes with the LLM arrow between them",
        "visible_model_interface": "LLM rewrite/condensation step producing query variants from the annotation queries",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the source annotation text, the LLM transformation label, and the resulting query-variant carrier with both arrows intact. It excludes the embedding plot and heatmap, which are downstream outputs and not needed to understand the input route.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_5d1c44891b0e",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "developmental-stage query text",
          "actual_model_visible_form": "query text"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_7fa24c55923f",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "Carnegie stage annotations",
          "actual_model_visible_form": "Carnegie stage annotation text"
        }
      ],
      "routes": [
        {
          "route_id": "route_7fa24c55923f",
          "configuration_id": "config_c6cff05cc790",
          "route_label": "Carnegie stage annotation condensation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "condensed their annotations using GPT-4o",
          "source_object_verbatim": "Carnegie stage annotations",
          "source_object_normalized": "Carnegie stage annotations",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "23 Carnegie stages grouped into four developmental bins",
            "annotations condensed into queries"
          ],
          "model_visible_form_verbatim": "Carnegie stage annotation text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "with the following prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "We condensed their annotations using GPT-4o",
          "section_heading": "Evaluation of the CellWhisperer embedding model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13
          ],
          "doc_item_refs": [
            "#/texts/1112",
            "#/texts/1113",
            "#/texts/1114",
            "#/texts/1115"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5d1c44891b0e",
          "configuration_id": "config_b184b153cb13",
          "route_label": "developmental query variant generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Rewrite the provided text in five different variants",
          "source_object_verbatim": "developmental-stage query text",
          "source_object_normalized": "developmental-stage query text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "LLM rewrite prompt",
            "five different variants"
          ],
          "model_visible_form_verbatim": "query text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "Rewrite the provided text in five different variants",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We generated five variants per query using GPT-4o",
          "section_heading": "Evaluation of the CellWhisperer embedding model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13
          ],
          "doc_item_refs": [
            "#/texts/1115"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_d8f17c473542"
    },
    {
      "model_id": "model_cabb39a8d967",
      "model_name": "GPT-4o",
      "record_id": "full_2026-07-06__rec_001617",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4f442dcf4ed4",
      "paper_title": "Cell-o1: Training LLMs to Solve Single-Cell Reasoning Puzzles with Reinforcement Learning",
      "doi": "10.48550/arXiv.2506.02911",
      "paper_url": "https://doi.org/10.48550/arXiv.2506.02911",
      "route_count": 4,
      "configuration_count": 4,
      "family_counts": {
        "text_native_token_stream": 4
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 4
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell RNA sequencing batch with donor metadata",
        "single-cell RNA sequencing cell with donor metadata"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001617_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001617_76b0056d59c3/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of this work. Cell-o1 achieves state-of-the-art on the CellPuzzles task.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific schematic and results figure about the “CellPuzzles” task and a Cell-o1 training strategy.\n\nVisible panels and labels:\n- Left panel: “The CellPuzzles Task”\n  - Subpanel “(1) Previous LLM-based Annotation (Cell by Cell; No reasoning)”\n  - Input example lists marker genes: “MALAT1, RPS27, RPL10, RPL13, RPL41, EEF1A1, …”\n  - Output: “This is a CD8-positive, alpha-beta T cell.”\n  - Subpanel “(2) CellPuzzles (Batch-level; Reasoning required)”\n  - Input asks to jointly assign one unique cell type to each cell in a batch using expressed genes and donor context, and provide reasoning plus final answer.\n  - Output shows reasoning text and a small cell-to-cell assignment diagram with labels including:\n    - CD4-positive, alpha-beta T Cell\n    - CD8-positive, alpha-beta T Cell\n    - Capillary Endothelial Cell\n    - Plasma Cell\n    - Respiratory Basal Cell\n\n- Middle panel: “Cell-o1 Training Strategy”\n  - Subpanel “(1) Reasoning Distillation”\n  - Shows example reasoning boxes labeled “Reasoning 1” and “Reasoning 2” with corresponding “Answer 1” and “Answer 2” diagrams.\n  - Reasoning 1 states cell 9 shows strong immunoglobulin expression and is marked “Reject.”\n  - Reasoning 2 states there are 10 cells and 10 candidate types, so each type must match exactly one cell, marked “Accept.”\n  - Subpanel “(2) Reinforcement Learning”\n  - Flow diagram includes:\n    - Input Question\n    - Policy Model\n    - Reference Model\n    - Reward Function\n    - Advantage\n    - Group Computation\n    - Reward\n    - Update\n    - KL\n    - Cold Start\n\n- Right panel: “Results”\n  - Subpanel “(1) Cell-level Accuracy” shown as a radial/polar bar chart.\n  - Values visible include 0.42, 0.58, 0.65, 0.68, 0.26, 0.26, 0.14, 0.09.\n  - Cell-o1 is highlighted with “(+5.65%).”\n  - Subpanel “(2) Batch-level Accuracy” shown as another radial/polar bar chart.\n  - Values visible include 0.03, 0.14, 0.19, 0.33, and multiple 0.00/0.01 labels.\n  - Cell-o1 is highlighted with “(+73.05%).”\n\nBiological source objects:\n- Single-cell gene-expression examples and cell type annotations.\n- Cell types shown include T cells, endothelial cells, plasma cells, and respiratory basal cells.\n- Marker genes shown include MALAT1, RPS27, RPL10, RPL13, RPL41, EEF1A1, VIM, ADAMDEC1, CCL, and IGHA1/JCHAIN.\n\nModel interfaces and transformations:\n- The figure contrasts prior cell-by-cell LLM annotation with batch-level reasoning over multiple cells.\n- It depicts reasoning distillation where candidate reasoning-answer pairs are accepted or rejected.\n- It depicts reinforcement learning using a policy model, reference model, reward function, KL regularization, group computation, advantages, and rewards.\n\nFindings:\n- Cell-o1 appears to outperform comparison models in both cell-level and batch-level accuracy.\n- Reported improvement labels are +5.65% for cell-level accuracy and +73.05% for batch-level accuracy.\n- Legend compares Llama3.1-8B-Instruct, Qwen2.5-7B-Instruct, Llama3.3-70B-Instruct, GPT-4o-mini, GPT-4o, o3-mini, o1, and Cell-o1.",
        "page_no": 1,
        "sha256": "54b9bdf02b4a0ee126b6139a699bdb14ba90c40a9bcc4c6870cf609d32e78134",
        "pixel_width": 759,
        "pixel_height": 408,
        "crop_box": {
          "x": 0,
          "y": 0.37,
          "width": 0.34,
          "height": 0.63
        },
        "panel_label": "left-batch-level-panel",
        "visible_input_object": "A batch of cells from the same donor, with per-cell expressed genes and donor context",
        "visible_model_interface": "Structured batch-level prompt with candidate cell types and ordered answer formatting, plus the batch-level reasoning/assignment diagram",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the left CellPuzzles batch-level section only, keeping the prompt, cell/label assignment diagram, and readable arrows needed to ground the batch-level input route while excluding the middle/results panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_3346bafbaa10",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "the gene expression profile of a single cell from a specific donor",
          "actual_model_visible_form": "a single-cell prompt with donor context, candidate labels, and structured <think>/<answer> instructions"
        }
      ],
      "routes": [
        {
          "route_id": "route_3346bafbaa10",
          "configuration_id": "config_83f676f523b2",
          "route_label": "cell-level reasoning prompt (GPT-4o)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cell-level Reasoning Setup",
          "source_object_verbatim": "the gene expression profile of a single cell from a specific donor",
          "source_object_normalized": "single cell from a specific donor with top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing cell with donor metadata",
          "transformation_chain_verbatim": [
            "use the single cell's top expressed genes",
            "combine them with donor context",
            "provide a fixed candidate label set",
            "format the response with reasoning tags"
          ],
          "model_visible_form_verbatim": "a single-cell prompt with donor context, candidate labels, and structured <think>/<answer> instructions",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt template",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are an expert assistant specialized in cell type annotation. You will be given the gene expression profile of a single cell from a specific donor. The top expressed genes are listed in descending order. Use both gene expression and donor context to determine the correct cell type. You will also receive a list of candidate cell types-choose the one that best fits this cell . Include your detailed reasoning within <think> and </think> tags, and provide your final answer within <answer> and </answer> tags. The final answer should be a single string with exactly one cell type.",
          "section_heading": "C Cell-level vs. Batch-level Reasoning",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            21
          ],
          "doc_item_refs": [
            "#/texts/1015",
            "#/texts/1016",
            "#/texts/1017",
            "#/texts/1018",
            "#/texts/1019",
            "#/texts/1020"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_008"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d52f38caf9ac",
          "configuration_id": "config_d56038733eb8",
          "route_label": "batch-level reasoning prompt (GPT-4o)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Batch-level Reasoning Setup",
          "source_object_verbatim": "a batch of N cells from the same donor",
          "source_object_normalized": "batch of N cells from the same donor with ranked top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "combine with donor context",
            "present the candidate label set",
            "directly predict answers without reasoning traces"
          ],
          "model_visible_form_verbatim": "batch-level structured input without reasoning traces",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "direct prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are an expert assistant specialized in cell type annotation. You will be given a batch of N cells from the same donor, where each cell represents a unique cell type. For each cell, the top-expressed genes are provided in descending order of expression. Using both the gene expression data and donor information, determine the correct cell type for each cell. You will also receive a list of N candidate cell types, and each candidate must be assigned to exactly one cell. Ensure that you consider all cells and candidate types together, rather than annotating each cell individually. Include your detailed reasoning within <think> and </think> tags, and provide your final answer within <answer> and </answer> tags. The final answer should be a single string listing the assigned cell types in order, separated by ' | '.",
          "section_heading": "C Cell-level vs. Batch-level Reasoning",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/603",
            "#/texts/604",
            "#/texts/605"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_012"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c7edfd5df171",
          "configuration_id": "config_d31cbaed0251",
          "route_label": "open-ended QA prompt (GPT-4o)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Open-ended QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "provide gene expression profile and metadata",
            "remove the constrained label set",
            "ask for free-form cell type generation"
          ],
          "model_visible_form_verbatim": "free-form textual generation of a cell type name for each cell",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompting without constrained labels",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_018"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a03dbfe863f0",
          "configuration_id": "config_626166b17e6a",
          "route_label": "constrained QA prompt (GPT-4o)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Constrained QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "convert donor metadata into natural language context",
            "attach a predefined candidate label set",
            "require a single ordered answer string"
          ],
          "model_visible_form_verbatim": "a structured batch-level text prompt with candidate labels and ordered answer output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "predefined candidate label set with structured prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_024"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_a62125696693",
      "model_name": "GPT-4o-mini",
      "record_id": "full_2026-07-06__rec_001617",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4f442dcf4ed4",
      "paper_title": "Cell-o1: Training LLMs to Solve Single-Cell Reasoning Puzzles with Reinforcement Learning",
      "doi": "10.48550/arXiv.2506.02911",
      "paper_url": "https://doi.org/10.48550/arXiv.2506.02911",
      "route_count": 4,
      "configuration_count": 4,
      "family_counts": {
        "text_native_token_stream": 4
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 4
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell RNA sequencing batch with donor metadata",
        "single-cell RNA sequencing cell with donor metadata"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001617_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001617_76b0056d59c3/figure_002.png",
        "figure_index": 2,
        "caption": "Figure 2: CellPuzzles formulates cell type annotation as a batch-level reasoning task that integrates gene expression and contextual metadata, inspired by how experts annotate cells in practice.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic comparison of two cell type annotation workflows with four main panel regions arranged in two rows.\n\nTop row: A human expert task workflow. The left “Input” panel shows clustered single-cell data with colored clusters labeled Cluster 2, Cluster 3, Cluster 4, and Cluster N, each associated with marker gene lists such as “SFTPC, SFTPA1, SFTPB, NAPSA, ...”, “C1QA, APOE, MARCO, LYZ, ...”, and “IGHM, MZB1, JCHAIN, XBP1, ...”. The center panel labeled “Human Expert Task” shows an expert assigning a cell type label to each cluster using representative marker genes, contextual metadata, biological knowledge, and reference sources. The right “Output” panel lists assigned cell types including Pulmonary Alveolar Type 2 (AT2) Cells, CD8+ Cytotoxic T Cells, Lung Pericyte, Alveolar Macrophages, and Plasma Cells.\n\nBottom row: A “CellPuzzles Task” workflow. The left “Input” panel shows individual cells from clusters, with representative sampled cells labeled [Cell 1], [Cell 2], [Cell 3], [Cell 4], and [Cell N], each paired with gene lists such as “S100A9, TMSB10, RPL37, ...”, “MALAT1, FTL, B2M, ...”, “MALAT1, FTL, AKR1B10, ...”, “B2M, MT2A, TMSB4X, ...”, and “IGLC3, IGLC2, IGHM, ...”. Arrows indicate selected representative cells from clusters. The center panel labeled “CellPuzzles Task” shows a model/interface labeled “Cell-o1” receiving N cells, contextual metadata, and N candidate cell types, then reasoning to determine the optimal label assignment and provide reasoning traces. The right “Output” panel shows colored cell-to-label assignment lines connecting cells to candidate labels including CD4-positive, alpha-beta T Cell; CD8-positive, alpha-beta T Cell; Smooth Muscle Cell; Lung Macrophage; Non-classical Monocyte; Capillary Endothelial Cell; Plasma Cell; and Respiratory Basal Cell.\n\nBiological source objects include single-cell clusters, individual cells, marker genes, and immune/lung-related cell type labels. The figure depicts a transformation from marker gene or cell-level input data to annotated biological cell type outputs, comparing expert manual annotation with an automated CellPuzzles/Cell-o1 reasoning task.",
        "page_no": 3,
        "sha256": "e74192b02a6c544c4d4ae6c01c1b71c75747f0270a5e4e6bc77aba0c8a2259ee",
        "pixel_width": 791,
        "pixel_height": 318,
        "crop_box": {
          "x": 0.0,
          "y": 0.49,
          "width": 0.81,
          "height": 0.51
        },
        "panel_label": "Bottom-row input + CellPuzzles task",
        "visible_input_object": "Representative single-cell inputs [Cell 1]-[Cell N] with top genes and cluster-to-cell arrows",
        "visible_model_interface": "CellPuzzles Task panel showing N cells, contextual metadata, N candidate cell types, and explicit reasoning/answer instruction",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the grounded input route from sampled cells and marker genes into the CellPuzzles task interface, while excluding the output-only panel on the right. It preserves the readable arrows, cell labels, and the model-visible prompt needed to understand the transformation.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_795c27e38c2b",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "the gene expression profile of a single cell from a specific donor",
          "actual_model_visible_form": "a single-cell prompt with donor context, candidate labels, and structured <think>/<answer> instructions"
        }
      ],
      "routes": [
        {
          "route_id": "route_795c27e38c2b",
          "configuration_id": "config_009f05c2ad30",
          "route_label": "cell-level reasoning prompt (GPT-4o-mini)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cell-level Reasoning Setup",
          "source_object_verbatim": "the gene expression profile of a single cell from a specific donor",
          "source_object_normalized": "single cell from a specific donor with top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing cell with donor metadata",
          "transformation_chain_verbatim": [
            "use the single cell's top expressed genes",
            "combine them with donor context",
            "provide a fixed candidate label set",
            "format the response with reasoning tags"
          ],
          "model_visible_form_verbatim": "a single-cell prompt with donor context, candidate labels, and structured <think>/<answer> instructions",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt template",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are an expert assistant specialized in cell type annotation. You will be given the gene expression profile of a single cell from a specific donor. The top expressed genes are listed in descending order. Use both gene expression and donor context to determine the correct cell type. You will also receive a list of candidate cell types-choose the one that best fits this cell . Include your detailed reasoning within <think> and </think> tags, and provide your final answer within <answer> and </answer> tags. The final answer should be a single string with exactly one cell type.",
          "section_heading": "C Cell-level vs. Batch-level Reasoning",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            21
          ],
          "doc_item_refs": [
            "#/texts/1015",
            "#/texts/1016",
            "#/texts/1017",
            "#/texts/1018",
            "#/texts/1019",
            "#/texts/1020"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_dcf75e6ddb0d",
          "configuration_id": "config_f5cb8b4e9918",
          "route_label": "batch-level reasoning prompt (GPT-4o-mini)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Batch-level Reasoning Setup",
          "source_object_verbatim": "a batch of N cells from the same donor",
          "source_object_normalized": "batch of N cells from the same donor with ranked top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "combine with donor context",
            "present the candidate label set",
            "directly predict answers without reasoning traces"
          ],
          "model_visible_form_verbatim": "batch-level structured input without reasoning traces",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "direct prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are an expert assistant specialized in cell type annotation. You will be given a batch of N cells from the same donor, where each cell represents a unique cell type. For each cell, the top-expressed genes are provided in descending order of expression. Using both the gene expression data and donor information, determine the correct cell type for each cell. You will also receive a list of N candidate cell types, and each candidate must be assigned to exactly one cell. Ensure that you consider all cells and candidate types together, rather than annotating each cell individually. Include your detailed reasoning within <think> and </think> tags, and provide your final answer within <answer> and </answer> tags. The final answer should be a single string listing the assigned cell types in order, separated by ' | '.",
          "section_heading": "C Cell-level vs. Batch-level Reasoning",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/603",
            "#/texts/604",
            "#/texts/605"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_011"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_69221bb4cbfb",
          "configuration_id": "config_f1e45f8b416b",
          "route_label": "open-ended QA prompt (GPT-4o-mini)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Open-ended QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "provide gene expression profile and metadata",
            "remove the constrained label set",
            "ask for free-form cell type generation"
          ],
          "model_visible_form_verbatim": "free-form textual generation of a cell type name for each cell",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompting without constrained labels",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_017"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_22c5b93e6465",
          "configuration_id": "config_2c89d5fc1190",
          "route_label": "constrained QA prompt (GPT-4o-mini)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Constrained QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "convert donor metadata into natural language context",
            "attach a predefined candidate label set",
            "require a single ordered answer string"
          ],
          "model_visible_form_verbatim": "a structured batch-level text prompt with candidate labels and ordered answer output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "predefined candidate label set with structured prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_023"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_8553f59d67fa",
      "model_name": "GPT-OSS-120B",
      "record_id": "full_2026-07-06__rec_003008",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_b6246bd13ef4",
      "paper_title": "Aligning LLMs with Biomedical Knowledge using Balanced Fine-Tuning",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "inference"
      ],
      "fusion_topologies": [
        "concatenation"
      ],
      "text_roles": [
        "biological_payload"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "No figure responsibly supports the biology data synthesis route for GPT-OSS-120B. The visible panels are about gene embeddings or downstream evaluation, not NCBI text being used to generate share-GPT-formatted samples.",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_e7ac8502029f",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "NCBI text provided by GenePT",
          "actual_model_visible_form": "prompt template plus input text"
        }
      ],
      "routes": [
        {
          "route_id": "route_e7ac8502029f",
          "configuration_id": "config_6ed6c84d88e3",
          "route_label": "Biology training data synthesis from NCBI text",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "generate a dataset in the share-gpt format",
          "source_object_verbatim": "NCBI text provided by GenePT",
          "source_object_normalized": "NCBI text provided by GenePT",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "used as the knowledge base",
            "GPT-OSS-120B generates share-gpt-formatted samples"
          ],
          "model_visible_form_verbatim": "prompt template plus input text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompted generation",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "In the biology domain, we fine-tuned LLMs using both SFT and BFT. The training data construction process is as follows: NCBI text provided by GenePT [15] is used as the knowledge base, and GPT-OSS-120B [16] is employed to generate a dataset in the share-gpt format. Extended Data Figure 3 illustrates examples of the constructed samples.",
          "section_heading": "2.3.3 Biology: BFT improves reasoning about biological processes",
          "supporting_figure_or_table": "Extended Data Figure 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/36",
            "#/texts/37",
            "#/texts/38",
            "#/texts/39"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003008::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_d2833fcd27aa",
      "model_name": "gpt2-gene-eng",
      "record_id": "full_2026-07-06__rec_002243",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_2db24aaf3a41",
      "paper_title": "Human Genome Book:Words,Sentences and Paragraphs",
      "doi": "10.1101/2025.01.23.634629",
      "paper_url": "https://doi.org/10.1101/2025.01.23.634629",
      "route_count": 3,
      "configuration_count": 1,
      "family_counts": {
        "discrete_biological_symbol_stream": 2,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "native_biological_token_stream": 2,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream",
        "plain_language_prompt_or_question"
      ],
      "primary_subtype": "native_biological_token_stream",
      "modalities": [
        "DNA sequence",
        "protein/peptide",
        "text"
      ],
      "lifecycle_phases": [
        "pretraining"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "metadata_or_context"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_002243_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_002243_2c739b46e03e/figure_001.png",
        "figure_index": 1,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a workflow diagram for training and fine-tuning GPT-2 style models across biological and natural-language sequence data.\n\nVisible elements:\n\n- Top input panels:\n  - “DNA Sequence” with a DNA/sequence image.\n  - “Protein Sequence” with a protein structure image.\n  - “Natural Language” with a document/text icon.\n- These three sources feed into “Unified BPE Tokenization/Encoder.”\n- A downward arrow labeled “Pre-training” leads to “GPT-2 Pre-trained Model (GPT2-gene-eng).”\n- A “PAWS English Datasets” box and a “Sequence Similarity Classification Dataset” arrow feed into a “Finetune” block.\n- Text annotation beside this stage: “Language Capability Transfer Eng-->DNA” and “Verifiable.”\n- The fine-tuned output is labeled “GPT2-gene-eng-ft.”\n- Lower fine-tuning branches:\n  - “English Summary Datasets” feeds into a “Finetune” block producing “gene_eng_gpt2_para_seg.”\n  - “English Paragraph Segmentation Datasets” feeds into another “Finetune” block producing “gene_eng_gpt2_summary.”\n- Both outputs point toward “Human Genome,” which then points to a “Genome Book” image.\n- A side annotation states “Language Capability Transfer Eng-->DNA,” “Unverified?”, and a curved dashed arrow labeled “Indirect Verification?”\n\nBiological source objects shown or referenced include DNA sequences, protein sequences/structures, and the human genome. The main transformations are unified BPE tokenization, GPT-2 pre-training, dataset-specific fine-tuning, and application to genome text/book generation. The figure presents a model-training pipeline rather than quantitative findings.",
        "page_no": 5,
        "sha256": "f317a0eb287574a9980d3b339c7f368ea20156081aede388e2636d6aa9f9916b",
        "pixel_width": 859,
        "pixel_height": 703,
        "crop_box": {
          "x": 0.03,
          "y": 0.0,
          "width": 0.66,
          "height": 0.39
        },
        "panel_label": "DNA pretraining route",
        "visible_input_object": "DNA Sequence",
        "visible_model_interface": "DNA Sequence -> Unified BPE Tokenization/Encoder -> Pre-training -> GPT-2 Pre-trained Model (GPT2-gene-eng)",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps one complete grounded route from the DNA source object through the shared tokenizer/interface and pre-training into the visible GPT-2 pre-trained model, while excluding downstream fine-tuning and output-only panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_bf01b711c418",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "DNA sequence data",
          "actual_model_visible_form": "tokenized DNA sequence fragments"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_a70369fbdf43",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "OpenWebText dataset combined with Wikipedia data",
          "actual_model_visible_form": "tokenized English text"
        }
      ],
      "routes": [
        {
          "route_id": "route_bf01b711c418",
          "configuration_id": "config_b6eaa1e1287d",
          "route_label": "DNA pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "A pre-trained model was built using English data, DNA sequences, and protein sequences, with a unified BPE tokenizer.",
          "source_object_verbatim": "DNA sequence data",
          "source_object_normalized": "DNA sequence fragments",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "fragment extraction ranging from 300 to 1000 base pairs (bp)",
            "unified BPE tokenization"
          ],
          "model_visible_form_verbatim": "tokenized DNA sequence fragments",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "unified BPE tokenizer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "A pre-trained model was built using English data, DNA sequences, and protein sequences, with a unified BPE tokenizer.",
          "section_heading": "2.2. Pretrained model",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            8
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/texts/15",
            "#/texts/20",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/27",
            "#/texts/38",
            "#/texts/42",
            "#/texts/46",
            "#/texts/47",
            "#/texts/49",
            "#/texts/50"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0001",
            "dense::full_2026-07-06__rec_002243::0007",
            "dense::full_2026-07-06__rec_002243::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9fcc57f4af69",
          "configuration_id": "config_b6eaa1e1287d",
          "route_label": "Protein pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "A pre-trained model was built using English data, DNA sequences, and protein sequences, with a unified BPE tokenizer.",
          "source_object_verbatim": "Protein sequence data",
          "source_object_normalized": "protein sequence fragments",
          "source_modality_normalized": "protein/peptide",
          "transformation_chain_verbatim": [
            "protein sequence extraction from UniProt",
            "unified BPE tokenization"
          ],
          "model_visible_form_verbatim": "tokenized protein sequences",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "unified BPE tokenizer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "From the UniProt database, we extracted 10 GB of protein sequence data",
          "section_heading": "2.2. Pretrained model",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            8
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/texts/20",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/27",
            "#/texts/40",
            "#/texts/42",
            "#/texts/46",
            "#/texts/47",
            "#/texts/49",
            "#/texts/50"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0001",
            "dense::full_2026-07-06__rec_002243::0008",
            "dense::full_2026-07-06__rec_002243::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a70369fbdf43",
          "configuration_id": "config_b6eaa1e1287d",
          "route_label": "English text pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "A pre-trained model was built using English data, DNA sequences, and protein sequences, with a unified BPE tokenizer.",
          "source_object_verbatim": "OpenWebText dataset combined with Wikipedia data",
          "source_object_normalized": "English text corpus",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "English text extraction",
            "unified BPE tokenization"
          ],
          "model_visible_form_verbatim": "tokenized English text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "unified BPE tokenizer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "An intriguing direction within these studies is the exploration of multilingual transfer capabilities in large  language  models.  Research  on  multilingual  transfer  has  demonstrated  how  large  pre-trained models  can  effectively  apply  knowledge  learned  in  one  source  language  to  other  languages.  These studies  have  validated  the  effectiveness  of  cross-linguistic  knowledge  transfer,  including  between different natural languages, as well as between programming languages and natural language [25][26] [27][28][29] .  Moreover, some recent experimental research has shown that the transfer phenomenon from natural language capabilities to DNA sequences also exists, making it possible to directly apply natural language processing techniques to DNA sequence analysis [30] . In this paper, we leverage the transfer of natural language capabilities to DNA language to construct a structured human genomic 'book.' Specifically, we pre-trained a GPT-2 model, gpt2-gene-eng ,  on English, DNA, and protein sequences using a unified BPE tokenizer. We then fine-tuned this model using the English semantic similarity dataset from PAWSX, resulting in a model, gpt2-gene-eng-ft , capable of transferring natural language abilities to DNA sequences. Based on this fine-tuned model, we  further  trained  three  new  models  using  English  datasets  for  sentence  splitting,  paragraph segmentation, and summarization tasks, respectively. These three models were subsequently applied to  process  human genome data, producing a genomic 'book' that includes DNA words, sentences, and paragraphs. Additionally, by utilizing the pre-trained gpt2-gene-eng model, we established a mapping between the  DNA  vocabulary  and  the  English  vocabulary,  enabling  the  creation  of  an  English-translated version of the human genomic book.",
          "section_heading": "2.2. Pretrained model",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            8
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/texts/15",
            "#/texts/20",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/27",
            "#/texts/42",
            "#/texts/46",
            "#/texts/47",
            "#/texts/49",
            "#/texts/50"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0001",
            "dense::full_2026-07-06__rec_002243::0012"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_fdbd3ba264a9"
    },
    {
      "model_id": "model_5e9bb9e443dc",
      "model_name": "gpt2-gene-eng-ft",
      "record_id": "full_2026-07-06__rec_002243",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_2db24aaf3a41",
      "paper_title": "Human Genome Book:Words,Sentences and Paragraphs",
      "doi": "10.1101/2025.01.23.634629",
      "paper_url": "https://doi.org/10.1101/2025.01.23.634629",
      "route_count": 4,
      "configuration_count": 3,
      "family_counts": {
        "text_native_token_stream": 2,
        "discrete_biological_symbol_stream": 2
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1,
        "plain_language_prompt_or_question": 1,
        "native_biological_token_stream": 2
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream",
        "plain_language_prompt_or_question",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "native_biological_token_stream",
      "modalities": [
        "DNA sequence",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "inference"
      ],
      "fusion_topologies": [
        "other_explicit",
        "retrieval_or_tool_context",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "metadata_or_context",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_002243_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_002243_2c739b46e03e/figure_001.png",
        "figure_index": 1,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a workflow diagram for training and fine-tuning GPT-2 style models across biological and natural-language sequence data.\n\nVisible elements:\n\n- Top input panels:\n  - “DNA Sequence” with a DNA/sequence image.\n  - “Protein Sequence” with a protein structure image.\n  - “Natural Language” with a document/text icon.\n- These three sources feed into “Unified BPE Tokenization/Encoder.”\n- A downward arrow labeled “Pre-training” leads to “GPT-2 Pre-trained Model (GPT2-gene-eng).”\n- A “PAWS English Datasets” box and a “Sequence Similarity Classification Dataset” arrow feed into a “Finetune” block.\n- Text annotation beside this stage: “Language Capability Transfer Eng-->DNA” and “Verifiable.”\n- The fine-tuned output is labeled “GPT2-gene-eng-ft.”\n- Lower fine-tuning branches:\n  - “English Summary Datasets” feeds into a “Finetune” block producing “gene_eng_gpt2_para_seg.”\n  - “English Paragraph Segmentation Datasets” feeds into another “Finetune” block producing “gene_eng_gpt2_summary.”\n- Both outputs point toward “Human Genome,” which then points to a “Genome Book” image.\n- A side annotation states “Language Capability Transfer Eng-->DNA,” “Unverified?”, and a curved dashed arrow labeled “Indirect Verification?”\n\nBiological source objects shown or referenced include DNA sequences, protein sequences/structures, and the human genome. The main transformations are unified BPE tokenization, GPT-2 pre-training, dataset-specific fine-tuning, and application to genome text/book generation. The figure presents a model-training pipeline rather than quantitative findings.",
        "page_no": 5,
        "sha256": "f317a0eb287574a9980d3b339c7f368ea20156081aede388e2636d6aa9f9916b",
        "pixel_width": 859,
        "pixel_height": 703,
        "crop_box": {
          "x": 0.096,
          "y": 0.38,
          "width": 0.516,
          "height": 0.23
        },
        "panel_label": "PAWSX fine-tuning branch",
        "visible_input_object": "PAWS English Datasets",
        "visible_model_interface": "Sequence Similarity Classification Dataset -> Finetune -> GPT2-gene-eng-ft",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop tightly preserves the PAWS English datasets box, the sequence-similarity arrow into Finetune, and the resulting GPT2-gene-eng-ft label. It omits the unrelated top pretraining sources and lower genome-generation panels while keeping one complete grounded route readable.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_e8cc2811eb6b",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "DNA paragraphs",
          "actual_model_visible_form": "DNA paragraph text"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_ee89475f41a7",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "natural language text",
          "actual_model_visible_form": "tokenized natural language text"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_37a7cca91a29",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "English semantic similarity dataset from the PAWSX dataset",
          "actual_model_visible_form": "tokenized English sentence pairs"
        }
      ],
      "routes": [
        {
          "route_id": "route_37a7cca91a29",
          "configuration_id": "config_e5935fc62197",
          "route_label": "PAWSX semantic similarity fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "English semantic similarity task from the PAWSX dataset",
          "source_object_verbatim": "English semantic similarity dataset from the PAWSX dataset",
          "source_object_normalized": "PAWS-X English semantic similarity dataset",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "classification fine-tuning on the English sequence similarity dataset"
          ],
          "model_visible_form_verbatim": "tokenized English sentence pairs",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "classification head",
          "fusion_topology": "other_explicit",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "we fine-tuned the pre-trained model using the English semantic similarity task from the PAWSX dataset",
          "section_heading": "2.3. Finetune model with multilingual transfer ability",
          "supporting_figure_or_table": "Figure 1. Construction Process of the Human Genome Book.",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            9,
            10,
            15,
            16
          ],
          "doc_item_refs": [
            "#/texts/15",
            "#/texts/20",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/27",
            "#/texts/55",
            "#/texts/56",
            "#/texts/57",
            "#/texts/58",
            "#/texts/59",
            "#/texts/60",
            "#/texts/61",
            "#/texts/65"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ee89475f41a7",
          "configuration_id": "config_5e2704f905bd",
          "route_label": "Sentence boundary prediction on natural language text",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "sentence segmentation",
          "source_object_verbatim": "natural language text",
          "source_object_normalized": "natural language text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "predict the '.' token"
          ],
          "model_visible_form_verbatim": "tokenized natural language text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "causal language modeling head",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "sentence splitting can be accomplished by using gpt2-gene-eng-ft to predict the '.' token",
          "section_heading": "2.5. Sentence Boundary Detection Model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The paper does not give a separate sentence-splitting model name; it attributes the task to gpt2-gene-eng-ft.",
          "pages": [
            3,
            4,
            5,
            11
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/20",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/27",
            "#/texts/82"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0018"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e8cc2811eb6b",
          "configuration_id": "config_5e2704f905bd",
          "route_label": "DNA paragraph sentence splitting",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "sentence segmentation",
          "source_object_verbatim": "DNA paragraphs",
          "source_object_normalized": "DNA paragraphs",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "processed each paragraph into sentences"
          ],
          "model_visible_form_verbatim": "DNA paragraph text",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sentence segmentation model",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "we used a sentence segmentation model to process each paragraph into sentences",
          "section_heading": "2.7. Genome Segmentation - Chapter Division - Sentence Splitting",
          "supporting_figure_or_table": "Figure 4. Structure of the Genome Catalog.",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            13,
            14,
            15
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/texts/101",
            "#/texts/105",
            "#/texts/107",
            "#/texts/108",
            "#/texts/109",
            "#/texts/114",
            "#/texts/115"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_010"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002243::0024"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_66ca1cdc7b10",
          "configuration_id": "config_ec89ce1818db",
          "route_label": "DNA paragraph translation to English",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Translation of DNA to English",
          "source_object_verbatim": "DNA paragraph sequences",
          "source_object_normalized": "DNA paragraph sequences",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "construct translation dictionary from DNA to English",
            "tokenized and translated each DNA paragraph sequence"
          ],
          "model_visible_form_verbatim": "tokenized DNA paragraph sequences",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "translation dictionary built from vector similarity mapping",
          "fusion_topology": "retrieval_or_tool_context",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "After constructing the translation dictionary from DNA to English, we tokenized and translated each DNA paragraph sequence.",
          "section_heading": "2.8. Translation of DNA to English",
          "supporting_figure_or_table": "Figure 5. Screenshot of the Genomic Book Content.",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The translation step is dictionary-based rather than an end-to-end neural generation route.",
          "pages": [
            15,
            16
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/121",
            "#/texts/122",
            "#/texts/123",
            "#/texts/124",
            "#/texts/128"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002243::route_012"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_8d5343810816"
    },
    {
      "model_id": "model_3827a28edefc",
      "model_name": "H2O",
      "record_id": "june_update_2026-06-10__rec_000152",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_1e6901520f8e",
      "paper_title": "H2O: A Foundation Model Bridging Histopathology to Spatial Multi-Omics Profiling",
      "doi": "10.64898/2026.04.21.717342",
      "paper_url": "https://doi.org/10.64898/2026.04.21.717342",
      "route_count": 5,
      "configuration_count": 5,
      "family_counts": {
        "visual_raster_carrier": 5
      },
      "subtype_counts": {
        "raw_slide_or_patch_input": 2,
        "patch_context_or_case_level_visual_reasoning": 3
      },
      "families": [
        "visual_raster_carrier"
      ],
      "subtypes": [
        "patch_context_or_case_level_visual_reasoning",
        "raw_slide_or_patch_input"
      ],
      "primary_subtype": "patch_context_or_case_level_visual_reasoning",
      "modalities": [
        "histopathology image"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "inference"
      ],
      "fusion_topologies": [
        "encoder_decoder",
        "shared_latent_alignment"
      ],
      "text_roles": [
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/june_update_2026_06_10_rec_000152_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/june_update_2026_06_10_rec_000152_5827375351f8/figure_002.png",
        "figure_index": 2,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a scientific schematic figure, panel labeled **“b.”**, titled **“Model Architecture and Training Procedure.”**\n\nIt depicts a computational biology pipeline combining **H&E imaging** and **spatial sequencing** data for **spatial transcriptomics prediction**.\n\nVisible elements include:\n\n- **Input biological data sources**\n  - H&E imaging, shown with a microscope icon and tissue image tiles.\n  - Spatial sequencing, shown with an instrument icon and a grid-like spatial transcriptomics matrix.\n\n- **Feature extraction / model components**\n  - An **Image FM** branch processes histology image patches.\n  - An **ST FM** branch processes spatial transcriptomics features.\n  - A diameter-based feature extraction inset shows circular regions at approximate scales labeled **2 µm**, **55 µm**, and **150 µm**.\n  - Extracted image and ST feature vectors are fed into a **Contrastive Learning** module.\n\n- **Contrastive learning**\n  - A similarity matrix is shown, with diagonal matching entries highlighted.\n  - Feature vectors from image and spatial transcriptomics modalities are aligned.\n  - The output connects to a convolutional block labeled **“8 Neighbors”**, followed by **“Conv & Concat.”**\n\n- **Fusion and prediction**\n  - A large yellow module labeled **FiLM** receives concatenated convolutional features and diameter/context features.\n  - The FiLM-modulated representation is passed to **ST Prediction**, producing a predicted spatial transcriptomics grid.\n\n- **Legend / training status**\n  - Gray arrows indicate **Input**.\n  - Green arrows indicate **Output**.\n  - Flame icon indicates **Trainable**.\n  - Snowflake icon indicates **Untrainable**.\n\n- **Lower training procedure panels**\n  - **Image FM** panel: shows **DINO V2** with **Student Image Transformer** and **Teacher Image Transformer**, fine-tuned using datasets labeled **TCGA**, **GTEx**, and **In-house**, producing an Image FM.\n  - **ST FM** panel: shows a **Single Cell FM** fine-tuned using **HEST1K** to produce a **Spatial Transcriptomics FM**.\n\nNo quantitative results or empirical findings are shown; the figure presents the architecture and training workflow.",
        "page_no": 4,
        "sha256": "370c807d40891c12e7df381d0f5e155b32c3952421ff2c119b3fdfbc026ffb51",
        "pixel_width": 1162,
        "pixel_height": 648,
        "crop_box": {
          "x": 0.032,
          "y": 0.065,
          "width": 0.584,
          "height": 0.427
        },
        "panel_label": "b.",
        "visible_input_object": "H&E imaging / H&E image patches",
        "visible_model_interface": "Image FM branch feeding contrastive learning and similarity-matrix alignment",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the H&E source icon, the patch-level histology input, the Image FM label, and the arrows into the contrastive-learning block with its similarity matrix, which is enough to ground the H&E image branch for alignment while excluding the output-only prediction side and lower panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "patch_context_or_case_level_visual_reasoning",
          "family_id": "visual_raster_carrier",
          "route_id": "route_45bef80d938d",
          "example_input": "ROI + neighboring patches + case context",
          "example_carrier": "ordered visual token bank",
          "example_interface": "context aggregator → multimodal LLM",
          "actual_source": "a central H&E image patch with neighboring patches and spot diameter",
          "actual_model_visible_form": "a central H&E image patch with neighboring patches and spot diameter"
        },
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_2f7983ac4cee",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "H&E image patches",
          "actual_model_visible_form": "H&E image patches"
        }
      ],
      "routes": [
        {
          "route_id": "route_2f7983ac4cee",
          "configuration_id": "config_d24ed576ac3c",
          "route_label": "H&E image branch for contrastive alignment",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "align a trainable histopathology foundation model and a parameter-frozen ST foundation model using paired image and gene expression data",
          "source_object_verbatim": "H&E image patches",
          "source_object_normalized": "histopathology H&E image patches",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [
            "DINOv2-trained ViT encoding",
            "contrastive learning alignment"
          ],
          "model_visible_form_verbatim": "H&E image patches",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "contrastive learning aligns a trainable histopathology foundation model and a parameter-frozen ST foundation model using paired image and gene expression data",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "no_text_on_this_route",
          "input_status": "paired_alignment_input",
          "evidence_quote": "contrastive learning aligns a trainable histopathology foundation model and a parameter-frozen ST foundation model using paired image and gene expression data",
          "section_heading": "Model Training",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            20
          ],
          "doc_item_refs": [
            "#/texts/858",
            "#/texts/859"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000152::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_45bef80d938d",
          "configuration_id": "config_acf5a1faab6f",
          "route_label": "H&E to ST prediction with spatial context",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "predict the omics expression for each spot using the spatial context information",
          "source_object_verbatim": "a central H&E image patch with neighboring patches and spot diameter",
          "source_object_normalized": "central H&E patch with neighboring patches and spot diameter",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [
            "DINOv2-trained ViT image encoding",
            "1D convolution across adjacent patches",
            "FiLM diameter conditioning",
            "H2O decoder"
          ],
          "model_visible_form_verbatim": "a central H&E image patch with neighboring patches and spot diameter",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "patch_context_or_case_level_visual_reasoning",
          "insertion_or_fusion_verbatim": "grounded on the cross-modal aligned embeddings, the H2O decoder predicts the omics expression for each spot using the spatial context information",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "H2O is a multi-omics-guided histopathology foundation model designed to align whole-slide imaging features with  spatially  resolved  transcriptomics  and  proteomics  profiles  through  contrastive  learning.  To  enable comprehensive coverage of various tissues, organs, and diseases (Extended Data Fig. 2) across both H&E and ST modalities, we curated a diverse dataset, which contains three major parts for H2O training and evaluation (Fig. 1a). Samples from The Cancer Genome Atlas (TCGA) [48], the Genotype-Tissue Expression (GTEx) [49] project, and additional in-house breast cancer collections (Methods) were used to train the histopathology FM. Then we employed a gene expression FM based on scGPT [35], initialized from whole-human pretrained checkpoints and further fine-tuned on ST data from the HEST-1k [47] dataset to encode spatial transcriptomics knowledge. Subsequently, we trained H2O using matched H&E and ST samples from the HEST-1k dataset. To infer transcriptomics profiles from a central image patch, H2O leverages morphological context from its local neighborhood and incorporates a Feature-wise Linear Modulation (FiLM) layer for resolution-aware feature calibration. See Extended Data Fig. 3 for ablation study of these modules. We further collected three additional datasets, HTSA, OpenST, and HTAPP, to investigate the ability of H2O in imparting molecular features to H&E features.",
          "section_heading": "Image embeddings",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            4,
            5,
            6,
            7,
            8,
            10,
            11,
            19,
            20
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/177",
            "#/texts/178",
            "#/texts/179",
            "#/texts/180",
            "#/texts/181",
            "#/texts/182",
            "#/texts/228",
            "#/texts/229",
            "#/texts/230",
            "#/texts/231",
            "#/texts/233",
            "#/texts/234",
            "#/texts/235",
            "#/texts/236",
            "#/texts/354",
            "#/texts/355",
            "#/texts/356",
            "#/texts/500",
            "#/texts/501",
            "#/texts/73",
            "#/texts/799",
            "#/texts/800",
            "#/texts/801",
            "#/texts/802",
            "#/texts/804",
            "#/texts/805"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000152::route_003"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000152::0001",
            "dense::june_update_2026-06-10__rec_000152::0002",
            "dense::june_update_2026-06-10__rec_000152::0003",
            "dense::june_update_2026-06-10__rec_000152::0004",
            "dense::june_update_2026-06-10__rec_000152::0005",
            "dense::june_update_2026-06-10__rec_000152::0006",
            "dense::june_update_2026-06-10__rec_000152::0007",
            "dense::june_update_2026-06-10__rec_000152::0014"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3d88753d6312",
          "configuration_id": "config_f2771616bab5",
          "route_label": "H&E to SP prediction with dual decoder",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "simultaneously predict both ST and SP distributions from the same histopathological image input",
          "source_object_verbatim": "the same histopathological image input",
          "source_object_normalized": "histopathological image input",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [
            "DINOv2-trained ViT image encoding",
            "1D convolution across neighboring patches",
            "FiLM diameter conditioning",
            "parallel SP decoder"
          ],
          "model_visible_form_verbatim": "the same histopathological image input",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "patch_context_or_case_level_visual_reasoning",
          "insertion_or_fusion_verbatim": "parallel SP decoder sharing an identical architecture with the original ST decoder",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "This dual-decoder framework enabled the model to simultaneously predict both ST and SP distributions from the same histopathological image input.",
          "section_heading": "H2O produces complementary spatial transcriptomics and spatial proteomics",
          "supporting_figure_or_table": "Fig. 6a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            16,
            17,
            18
          ],
          "doc_item_refs": [
            "#/texts/618",
            "#/texts/619",
            "#/texts/620",
            "#/texts/621",
            "#/texts/622",
            "#/texts/623",
            "#/texts/624",
            "#/texts/625"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000152::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_49aa3575b33d",
          "configuration_id": "config_b0977a74f5e5",
          "route_label": "Developmental thymus H&E for temporal ST prediction",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "predict ST across time",
          "source_object_verbatim": "histopathology images of developmental stages from fetal to paediatric thymus",
          "source_object_normalized": "developmental thymus histopathology images",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [
            "H&E image encoding",
            "H2O prediction across developmental stages"
          ],
          "model_visible_form_verbatim": "histopathology images of developmental stages from fetal to paediatric thymus",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "H2O decoder",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Histopathology images of developmental stages from fetal to paediatric thymus were processed by H2O to predict ST across time.",
          "section_heading": "Fig. 4 | H2O generalizes to developmental systems in the human thymus.",
          "supporting_figure_or_table": "Fig. 4a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            11,
            12,
            13
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/502",
            "#/texts/503",
            "#/texts/504",
            "#/texts/547",
            "#/texts/548",
            "#/texts/549"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000152::route_006"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000152::0008"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2856dcc135f2",
          "configuration_id": "config_9e3419b50d4a",
          "route_label": "Serial metastatic lymph node H&E for 3D ST reconstruction",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "predict spatial gene expression across sections, allowing a three-dimensional construction",
          "source_object_verbatim": "serial H&E slides from a metastatic lymph node",
          "source_object_normalized": "serial H&E slide set from a metastatic lymph node",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [
            "image registration",
            "H2O prediction across sections",
            "3D stacking"
          ],
          "model_visible_form_verbatim": "registered serial H&E sections",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "patch_context_or_case_level_visual_reasoning",
          "insertion_or_fusion_verbatim": "H2O prediction pipeline",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "a, Overview of the workflow for applying H2O to the OpenST dataset. Serial H&E slides from a metastatic lymph node were aligned and processed with H2O to predict spatial gene expression across sections, allowing a  three-dimensional  construction. b, H2O-predicted  expression  of  selected  marker  genes  across  sections, including  TIMP1  for  cancer-associated  fibroblasts,  LYZ  for  macrophages,  and  S100A8  for  lymph  nodeassociated myeloid cells. PCC between prediction and experiments measured profiles are shown at the left bottom corner of each sample. Predictions captured both the compartmentalization of lymph node tissue and tumor-associated infiltration. c , schematic of 3D sections. d, H2O-predicted cross-sectional views of marker genes, including LYZ, ACTA2 and SPP1, in the 3D stack. e, Three-dimensional renderings of  the marker expression of H2O predictions. The visualization revealed continuous tumor-stroma structures and invasive patterns of tumor penetration into the lymph node, which cannot be resolved from individual two-dimensional sections.",
          "section_heading": "Fig. 5 | H2O constructs 3D tumor architecture with clinical translational potential.",
          "supporting_figure_or_table": "Fig. 5a",
          "evidence_status": "explicit_text",
          "uncertainty": "The discovery record was flagged grounding_valid=false, but the canonical paper explicitly confirms the OpenST serial-section 3D reconstruction route.",
          "pages": [
            15
          ],
          "doc_item_refs": [
            "#/texts/613",
            "#/texts/614",
            "#/texts/615",
            "#/texts/616"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000152::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_10855a0f50f0"
    },
    {
      "model_id": "model_5588f179e503",
      "model_name": "in-house DINO-V2-based histopathology FM",
      "record_id": "june_update_2026-06-10__rec_000152",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_1e6901520f8e",
      "paper_title": "H2O: A Foundation Model Bridging Histopathology to Spatial Multi-Omics Profiling",
      "doi": "10.64898/2026.04.21.717342",
      "paper_url": "https://doi.org/10.64898/2026.04.21.717342",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "visual_raster_carrier": 1
      },
      "subtype_counts": {
        "raw_slide_or_patch_input": 1
      },
      "families": [
        "visual_raster_carrier"
      ],
      "subtypes": [
        "raw_slide_or_patch_input"
      ],
      "primary_subtype": "raw_slide_or_patch_input",
      "modalities": [
        "histopathology image"
      ],
      "lifecycle_phases": [
        "pretraining"
      ],
      "fusion_topologies": [
        "unclear"
      ],
      "text_roles": [
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/june_update_2026_06_10_rec_000152_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/june_update_2026_06_10_rec_000152_5827375351f8/figure_002.png",
        "figure_index": 2,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a scientific schematic figure, panel labeled **“b.”**, titled **“Model Architecture and Training Procedure.”**\n\nIt depicts a computational biology pipeline combining **H&E imaging** and **spatial sequencing** data for **spatial transcriptomics prediction**.\n\nVisible elements include:\n\n- **Input biological data sources**\n  - H&E imaging, shown with a microscope icon and tissue image tiles.\n  - Spatial sequencing, shown with an instrument icon and a grid-like spatial transcriptomics matrix.\n\n- **Feature extraction / model components**\n  - An **Image FM** branch processes histology image patches.\n  - An **ST FM** branch processes spatial transcriptomics features.\n  - A diameter-based feature extraction inset shows circular regions at approximate scales labeled **2 µm**, **55 µm**, and **150 µm**.\n  - Extracted image and ST feature vectors are fed into a **Contrastive Learning** module.\n\n- **Contrastive learning**\n  - A similarity matrix is shown, with diagonal matching entries highlighted.\n  - Feature vectors from image and spatial transcriptomics modalities are aligned.\n  - The output connects to a convolutional block labeled **“8 Neighbors”**, followed by **“Conv & Concat.”**\n\n- **Fusion and prediction**\n  - A large yellow module labeled **FiLM** receives concatenated convolutional features and diameter/context features.\n  - The FiLM-modulated representation is passed to **ST Prediction**, producing a predicted spatial transcriptomics grid.\n\n- **Legend / training status**\n  - Gray arrows indicate **Input**.\n  - Green arrows indicate **Output**.\n  - Flame icon indicates **Trainable**.\n  - Snowflake icon indicates **Untrainable**.\n\n- **Lower training procedure panels**\n  - **Image FM** panel: shows **DINO V2** with **Student Image Transformer** and **Teacher Image Transformer**, fine-tuned using datasets labeled **TCGA**, **GTEx**, and **In-house**, producing an Image FM.\n  - **ST FM** panel: shows a **Single Cell FM** fine-tuned using **HEST1K** to produce a **Spatial Transcriptomics FM**.\n\nNo quantitative results or empirical findings are shown; the figure presents the architecture and training workflow.",
        "page_no": 4,
        "sha256": "370c807d40891c12e7df381d0f5e155b32c3952421ff2c119b3fdfbc026ffb51",
        "pixel_width": 1162,
        "pixel_height": 648,
        "crop_box": {
          "x": 0.025,
          "y": 0.69,
          "width": 0.475,
          "height": 0.31
        },
        "panel_label": "Image FM",
        "visible_input_object": "TCGA / GTEx / In-house image cohorts",
        "visible_model_interface": "DINO V2 finetuning panel with Student Image Transformer and Teacher Image Transformer",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop isolates the lower-left Image FM training route, keeping the source datasets, finetuning arrow, and model block readable while excluding the unrelated ST and prediction panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_61505310f0e3",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "H&E image patches from TCGA, GTEx, and in-house breast cancer cohorts",
          "actual_model_visible_form": "H&E image patches"
        }
      ],
      "routes": [
        {
          "route_id": "route_61505310f0e3",
          "configuration_id": "config_55e2545e3917",
          "route_label": "H&E patches for histopathology image FM training",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "train the histopathology image foundation model",
          "source_object_verbatim": "H&E image patches from TCGA, GTEx, and in-house breast cancer cohorts",
          "source_object_normalized": "H&E image patches from TCGA, GTEx, and in-house cohorts",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [
            "patch extraction",
            "checkerboard sampling",
            "DINO-V2-based histopathology FM training"
          ],
          "model_visible_form_verbatim": "H&E image patches",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "we first trained an in-house DINO-V2-based histopathology FM using FFPE and fresh frozen tissue samples",
          "fusion_topology": "unclear",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "we first trained an in-house DINO-V2-based histopathology FM using FFPE and fresh frozen tissue samples",
          "section_heading": "An image FM aligned to ST latent space with contrastive learning.",
          "supporting_figure_or_table": "Fig. 1a",
          "evidence_status": "explicit_text",
          "uncertainty": "No separate fusion module is named for this pretraining route; the paper describes it as FM training rather than an explicit multimodal fusion step.",
          "pages": [
            6
          ],
          "doc_item_refs": [
            "#/texts/274",
            "#/texts/275",
            "#/texts/276",
            "#/texts/277",
            "#/texts/278",
            "#/texts/279",
            "#/texts/280",
            "#/texts/281"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000152::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_cb06ed43c31a"
    },
    {
      "model_id": "model_867ffa020512",
      "model_name": "InstructCell",
      "record_id": "full_2026-07-06__rec_001319",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_40ede81c16fd",
      "paper_title": "A Multi-Modal AI Copilot for Single-Cell Analysis with Instruction Following",
      "doi": "10.48550/arXiv.2501.08187",
      "paper_url": "https://doi.org/10.48550/arXiv.2501.08187",
      "route_count": 5,
      "configuration_count": 3,
      "family_counts": {
        "text_native_token_stream": 3,
        "dense_continuous_carrier": 2
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 3,
        "connector_mediated_embedding": 2
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "connector_mediated_embedding",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell gene expression",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "interleaving",
        "side_or_generative_conditioning",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001319_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001319_afbdc8a8325b/figure_001.png",
        "figure_index": 1,
        "caption": "Fig. 1: Overview of InstructCell. a, Summary of incorporated single-cell data. InstructCell incorporates 299,155 scRNA-seq samples from human and mouse origins, spanning multiple organs. CPCG denotes Conditional Pseudo-cell Generation , CTA denotes Cell Type Annotation , and DSP denotes Drug Sensitivity Prediction . b, Architecture of the multi-modal cell language model. The model processes both text and single-cell data via three primary components: a Q-Former to capture single-cell gene expression knowledge, a pre-trained LM as the backbone, and a cell reconstruction module for generating single-cell gene expression profiles. c, Construction of multi-modal single-cell instruction data. Complete instruction-response pairs are formed by combining required and optional attributes from text and single-cell modalities. d, Simulation of diverse communication styles. LLMs generate chat templates with varying traits (personality, motivation, and proficiency) to produce instructions that convey task-related information in different communication styles.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific figure about a single-cell foundation/model interface for gene expression and language tasks.\n\nPanel **a**: Circular sunburst-style dataset overview with labeled sectors for model/data sources including **CPCG**, **DSF**, **CTA**, **CTM**, and related datasets such as **Mouse-Atlas**, **GSE108097**, **GSE130984**, **GSE118127**, **He-2020**, **Sestan**, **Xie-2018**, **Ma-2020**, **Bachudas/Ponce-2019**, **BANC-seq**, and **Tabula-Sapiens**. Outer labels show biological source tissues/organs including lung, spleen, blood, vascular, bladder/thymus, muscle, brain, liver, pancreas, skin, and bone marrow, with small organ icons and counts.\n\nPanel **b**: Architecture schematic. A **Q-Former module** converts a **gene expression profile** through MLP, self-attention, cross-attention, feed-forward layers, and learnable queries into output features. These features interface with a **pre-trained language model** using text tokens and special cell tokens. A **cell reconstruction module** shows encoder-decoder reconstruction from latent space back to gene expression targets, with hidden states and signal terms.\n\nPanel **c**: Three task/interface examples:\n- **Conditional pseudo-cell generation**: required attributes are tissue, species, cell type, and sequencing protocol; output is a generated single-cell gene expression profile.\n- **Cell type annotation**: required input is single-cell gene expression; optional attributes include sequencing protocol, tissue, species, and options; output is an annotated cell type.\n- **Drug sensitivity prediction**: required inputs are single-cell gene expression and drug; optional attributes include sequencing protocol, tissue, species, and options; output is a drug response label.\n\nPanel **d**: Prompt/conversation generation schematic for conditional pseudo-cell generation. It defines questioner traits such as **personality**, **motivation**, and **proficiency**, plus task attributes including **cell type**, **species**, **tissue**, and **sequencing protocol**. Example generated prompts compare different questioner personas, such as skeptical or adventurous users, asking for a single-cell gene expression profile under specified biological conditions.\n\nVisible finding/claim: the figure presents a framework linking single-cell gene expression profiles with text through a Q-Former and pretrained language model to support pseudo-cell generation, cell type annotation, drug sensitivity prediction, and persona-conditioned instruction data generation.",
        "page_no": 3,
        "sha256": "6f60f9f509156496ac4c8e043a761f32b87bf6e9a1c20285da03bd7bc592b5f8",
        "pixel_width": 1039,
        "pixel_height": 1173,
        "crop_box": {
          "x": 0.428,
          "y": 0.031,
          "width": 0.329,
          "height": 0.31
        },
        "panel_label": "b",
        "visible_input_object": "single-cell gene expression profile",
        "visible_model_interface": "Q-Former bridge into the pre-trained LM with <CELL> insertion and text-token context",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "Crop the panel-b architecture region that shows the source gene expression profile, the Q-Former transformation, and the insertion into the pre-trained LM. It keeps the readable labels and arrows needed for the gene-expression input route while excluding the output-only cell reconstruction area and unrelated panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "connector_mediated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_35cf8fd29c94",
          "example_input": "image / omics encoder states",
          "example_carrier": "Q-Former or adapter query vectors",
          "example_interface": "connector → LLM cross-modal interface",
          "actual_source": "single-cell gene expression profile",
          "actual_model_visible_form": "cell embeddings"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_14152f1e2537",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "natural language generation conditions",
          "actual_model_visible_form": "text embeddings"
        }
      ],
      "routes": [
        {
          "route_id": "route_14152f1e2537",
          "configuration_id": "config_ea30fd3ce47b",
          "route_label": "CPCG text-conditions input",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Conditional Pseudo-cell Generation (CPCG)",
          "source_object_verbatim": "natural language generation conditions",
          "source_object_normalized": "generation conditions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenization",
            "LM input embedding layer",
            "pre-trained LM backbone"
          ],
          "model_visible_form_verbatim": "text embeddings",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed into the pre-trained LM as the textual input",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "we directly provide the generation conditions to InstructCell in textual form",
          "section_heading": "InstructCell enables conditional pseudo-cell generation",
          "supporting_figure_or_table": "Fig. 2(a)",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            6,
            23,
            30
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/195",
            "#/texts/196",
            "#/texts/247",
            "#/texts/248",
            "#/texts/249",
            "#/texts/42",
            "#/texts/43",
            "#/texts/44",
            "#/texts/45",
            "#/texts/47",
            "#/texts/48"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001319::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001319::0037",
            "dense::full_2026-07-06__rec_001319::0042",
            "dense::full_2026-07-06__rec_001319::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_be3424510352",
          "configuration_id": "config_38a38b1932fb",
          "route_label": "CTA instruction text input",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Cell Type Annotation (CTA)",
          "source_object_verbatim": "natural language instructions",
          "source_object_normalized": "natural-language instructions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenization",
            "LM input embedding layer",
            "pre-trained LM backbone"
          ],
          "model_visible_form_verbatim": "text embeddings",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed into the pre-trained LM as the instruction/question text",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Given an interleaved sequence of texts and gene expression profile, InstructCell generates the textual response that conveys its prediction regarding the cell type",
          "section_heading": "InstructCell boosts the performance of cell type annotation",
          "supporting_figure_or_table": "Fig. 3(a)",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            16,
            17,
            18,
            19,
            23
          ],
          "doc_item_refs": [
            "#/texts/113",
            "#/texts/114",
            "#/texts/115",
            "#/texts/116",
            "#/texts/117",
            "#/texts/118",
            "#/texts/119",
            "#/texts/121",
            "#/texts/147",
            "#/texts/195",
            "#/texts/196",
            "#/texts/51",
            "#/texts/52",
            "#/texts/53"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001319::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001319::0043",
            "dense::full_2026-07-06__rec_001319::0049",
            "dense::full_2026-07-06__rec_001319::0050",
            "dense::full_2026-07-06__rec_001319::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_35cf8fd29c94",
          "configuration_id": "config_38a38b1932fb",
          "route_label": "CTA gene-expression input",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Cell Type Annotation (CTA)",
          "source_object_verbatim": "single-cell gene expression profile",
          "source_object_normalized": "single-cell gene expression profile",
          "source_modality_normalized": "single-cell gene expression",
          "transformation_chain_verbatim": [
            "special tokens <CELL> and </CELL>",
            "Q-Former module",
            "pre-trained LM backbone"
          ],
          "model_visible_form_verbatim": "cell embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "encoded by the Q-Former and inserted into the interleaved text sequence via <CELL> and </CELL>",
          "fusion_topology": "interleaving",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Here, W is a learnable matrix with dimensions ( | V | +3) × d , where V is the original vocabulary of the pre-trained LM. Note that three special tokens, < CELL > , < /CELL > , and < SIGNAL > , are added to the vocabulary. These tokens are initialized randomly at the beginning of training. On the other hand, if x is a single cell's gene expression profile in the input sequence, it is transformed into k ( k ≥ 1) cell embeddings of dimension d by the cell encoder qformer( · ; ϕ ):",
          "section_heading": "Input embeddings",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            16,
            17,
            18,
            19,
            23
          ],
          "doc_item_refs": [
            "#/texts/113",
            "#/texts/114",
            "#/texts/115",
            "#/texts/116",
            "#/texts/117",
            "#/texts/118",
            "#/texts/119",
            "#/texts/121",
            "#/texts/147",
            "#/texts/195",
            "#/texts/196",
            "#/texts/51",
            "#/texts/52",
            "#/texts/53"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001319::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001319::0043",
            "dense::full_2026-07-06__rec_001319::0049",
            "dense::full_2026-07-06__rec_001319::0050",
            "dense::full_2026-07-06__rec_001319::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_61f880bd73b6",
          "configuration_id": "config_f76ed6c108bb",
          "route_label": "DSP drug text input",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Drug Sensitivity Prediction (DSP)",
          "source_object_verbatim": "natural language drug information",
          "source_object_normalized": "drug information",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenization",
            "LM input embedding layer",
            "pre-trained LM backbone"
          ],
          "model_visible_form_verbatim": "text embeddings",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed into the pre-trained LM as the drug prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "DSP involves providing the model with drug information and single-cell gene expression data",
          "section_heading": "InstructCell enhances precision of drug sensitivity prediction",
          "supporting_figure_or_table": "Fig. 4(a)",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            9,
            10,
            16,
            17,
            18,
            19,
            23
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/pictures/3",
            "#/texts/113",
            "#/texts/114",
            "#/texts/115",
            "#/texts/116",
            "#/texts/117",
            "#/texts/118",
            "#/texts/119",
            "#/texts/121",
            "#/texts/147",
            "#/texts/195",
            "#/texts/196",
            "#/texts/55",
            "#/texts/56",
            "#/texts/57",
            "#/texts/58",
            "#/texts/60",
            "#/texts/62"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001319::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001319::0044",
            "dense::full_2026-07-06__rec_001319::0045",
            "dense::full_2026-07-06__rec_001319::0049",
            "dense::full_2026-07-06__rec_001319::0050",
            "dense::full_2026-07-06__rec_001319::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4734953682d7",
          "configuration_id": "config_f76ed6c108bb",
          "route_label": "DSP gene-expression input",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Drug Sensitivity Prediction (DSP)",
          "source_object_verbatim": "single-cell gene expression data",
          "source_object_normalized": "single-cell gene expression data",
          "source_modality_normalized": "single-cell gene expression",
          "transformation_chain_verbatim": [
            "special tokens <CELL> and </CELL>",
            "Q-Former module",
            "pre-trained LM backbone"
          ],
          "model_visible_form_verbatim": "cell embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "encoded by the Q-Former and inserted into the interleaved text sequence via <CELL> and </CELL>",
          "fusion_topology": "interleaving",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "As illustrated in Fig. 1(c), DSP involves providing the model with drug information and single-cell gene expression data, allowing it to predict whether a cell is resistant or sensitive to a given drug. For our experiments, we gathered scRNA-seq data from three organs, including datasets from humans (GSE149383 and GSE117872) and mice (GSE110894), paired with drug sensitivity information. Notably, the GSE117872 dataset includes an additional category, labeled 'holiday', which refers to observations made during off-treatment periods.",
          "section_heading": "Input embeddings",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            9,
            10,
            16,
            17,
            18,
            19,
            23
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/pictures/3",
            "#/texts/113",
            "#/texts/114",
            "#/texts/115",
            "#/texts/116",
            "#/texts/117",
            "#/texts/118",
            "#/texts/119",
            "#/texts/121",
            "#/texts/147",
            "#/texts/195",
            "#/texts/196",
            "#/texts/55",
            "#/texts/56",
            "#/texts/57",
            "#/texts/58",
            "#/texts/60",
            "#/texts/62"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001319::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001319::0044",
            "dense::full_2026-07-06__rec_001319::0045",
            "dense::full_2026-07-06__rec_001319::0049",
            "dense::full_2026-07-06__rec_001319::0050",
            "dense::full_2026-07-06__rec_001319::0017"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_dbca3a16da8f"
    },
    {
      "model_id": "model_0f8f776fb068",
      "model_name": "LEONINE",
      "record_id": "full_2026-07-06__rec_003188",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_36da55ad5ac4",
      "paper_title": "Multimodal Cell Context Instruction Tuning for Conditional DNA Regulatory Sequence Generation with Large Language Models",
      "doi": "",
      "paper_url": "",
      "route_count": 3,
      "configuration_count": 2,
      "family_counts": {
        "discrete_biological_symbol_stream": 1,
        "dense_continuous_carrier": 2
      },
      "subtype_counts": {
        "native_biological_token_stream": 1,
        "direct_projected_embedding": 2
      },
      "families": [
        "discrete_biological_symbol_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "native_biological_token_stream"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "DNA sequence",
        "RNA expression profile",
        "categorical cellular metadata"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "pretraining"
      ],
      "fusion_topologies": [
        "shared_latent_alignment",
        "side_or_generative_conditioning"
      ],
      "text_roles": [
        "biological_payload",
        "metadata_or_context"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003188_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003188_07429b15f6f8/figure_002.png",
        "figure_index": 2,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic scientific figure describing a DNA large language model workflow for generated enhancer sequence design.\n\nVisible elements:\n- Top output label: “Generated Enhancer Sequence y” with example nucleotide tokens “AGAT-CG” and blue block-like sequence representations.\n- Main model block: a green horizontal module labeled “DNA Large Language Model fψ,” with example “HyenaDNA.”\n- Inputs/context modules at the bottom:\n  - Left dotted panel labeled “Multimodal Cell Context,” containing:\n    - “Cell Type Encoder”\n    - “Gene Profile Encoder”\n    - “Alignment Wc”\n    - “Alignment Wg”\n    - Cell-related icons and variables `xc`, `xg`\n    - Context representation `Hctx = [Hc, Hg]`\n  - Right dotted panel labeled “Promoter Sequence,” containing:\n    - Variable `xp`\n    - Representation `Hp`\n    - Example nucleotide sequence “CGTA-AG”\n- Arrows indicate information flow from encoded cell context and promoter sequence into the DNA language model, which produces the generated enhancer sequence.\n\nBiological source objects shown:\n- DNA/enhancer sequence\n- Promoter sequence\n- Cell type information\n- Gene profile information\n\nModel interfaces/transformations:\n- Cell type and gene profile inputs are encoded, aligned through `Wc` and `Wg`, and combined into multimodal cell context `Hctx`.\n- Promoter sequence `xp` is represented as `Hp`.\n- These representations condition or interface with a DNA large language model `fψ`.\n- The model outputs/generated sequence is labeled as enhancer sequence `y`.\n\nNo experimental results, quantitative findings, or performance claims are visible in this cropped figure.",
        "page_no": 10,
        "sha256": "3ccbd37efdb2c59e8fc2c384cda3bc7e7c27e61714232288d10edee942e7fa2e",
        "pixel_width": 869,
        "pixel_height": 273,
        "crop_box": {
          "x": 0,
          "y": 0.15,
          "width": 0.69,
          "height": 0.85
        },
        "panel_label": "Multimodal cell context to DNA LLM interface",
        "visible_input_object": "cell type index x_c and gene expression profile x_g",
        "visible_model_interface": "Alignment W_c/W_g into H_ctx = [H_c, H_g], feeding DNA Large Language Model f_psi",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the left multimodal cell-context panel, the alignment blocks, H_ctx, and the arrow into the DNA LLM, which is enough to ground the cell-type/gene-profile input route while excluding the output-only enhancer sequence panel.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_46816780e725",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "discrete cell type index x_c",
          "actual_model_visible_form": "language embedding tokens H_c"
        },
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_6a62e21c93db",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "promoter sequence x_p",
          "actual_model_visible_form": "promoter sequence embedding H_p"
        }
      ],
      "routes": [
        {
          "route_id": "route_6a62e21c93db",
          "configuration_id": "config_18304eae1e88",
          "route_label": "LEONINE promoter conditioning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "conditional enhancer generation given promoter sequence and multimodal cellular context",
          "source_object_verbatim": "promoter sequence x_p",
          "source_object_normalized": "promoter sequence",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "word embedding of f_psi",
            "promoter embedding H_p"
          ],
          "model_visible_form_verbatim": "promoter sequence embedding H_p",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "conditioned on H_ctx",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Consequently, given the cell context tokens Hctx and a pair of promoter-enhancer sequence xp y , the loss function that maximizes the likelihood of conditional probability p y x c xg x p is constructed as:",
          "section_heading": "2.2. Multimodal Cell Context Instruction Tuning",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            10,
            11,
            12,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/pictures/2",
            "#/pictures/3",
            "#/tables/0",
            "#/texts/156",
            "#/texts/164",
            "#/texts/172",
            "#/texts/180",
            "#/texts/52",
            "#/texts/53",
            "#/texts/54",
            "#/texts/55",
            "#/texts/56",
            "#/texts/57",
            "#/texts/58"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003188::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003188::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_46816780e725",
          "configuration_id": "config_ac510554df83",
          "route_label": "LEONINE cell type conditioning during feature alignment pre-training",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "aligning the multimodal cell context with the DNA token embedding",
          "source_object_verbatim": "discrete cell type index x_c",
          "source_object_normalized": "cell type index",
          "source_modality_normalized": "categorical cellular metadata",
          "transformation_chain_verbatim": [
            "trainable cell type embedding module e_phi",
            "modality alignment tensor W_e",
            "language embedding tokens H_c",
            "cell context tokens H_ctx"
          ],
          "model_visible_form_verbatim": "language embedding tokens H_c",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "combined into H_ctx and aligned with the pre-trained LLM token embedding",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Specifically, for a gene expression profile xg , we consider the pre-trained Geneformer single cell transcriptomes encoder [17], which provides the gene expression feature zg = g x g ∈ ℝ d g . The gene embedding features before the last Transformer layer are used in our experiments. For a discrete cell type index xc , we utilize a trainable cell type embedding module e ϕ ⋅ that maps the discrete cell type index into an embedding vector zc = e ϕ xc ∈ ℝ d e . Since both gene expression features and cell type embeddings are in different modalities compared to DNA tokens, we apply trainable modality alignment tensors Wg and We to convert zg and zc into language embedding tokens Hg ∈ ℝ l g × d ℎ and Hc ∈ ℝ l c × d ℎ , which have the same dimensionality as the nucleotide token embedding space dℎ in the DNA LLM:",
          "section_heading": "2.3. Training and Generation",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            10,
            11,
            12,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/pictures/2",
            "#/pictures/3",
            "#/tables/0",
            "#/texts/156",
            "#/texts/164",
            "#/texts/172",
            "#/texts/180",
            "#/texts/52",
            "#/texts/53",
            "#/texts/54",
            "#/texts/55",
            "#/texts/56",
            "#/texts/60",
            "#/texts/68",
            "#/texts/69"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003188::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003188::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_dd8e38c8f10d",
          "configuration_id": "config_ac510554df83",
          "route_label": "LEONINE gene expression conditioning during feature alignment pre-training",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "aligning the multimodal cell context with the DNA token embedding",
          "source_object_verbatim": "gene expression profile x_g",
          "source_object_normalized": "gene expression profile",
          "source_modality_normalized": "RNA expression profile",
          "transformation_chain_verbatim": [
            "pre-trained Geneformer single cell transcriptomes encoder",
            "modality alignment tensor W_g",
            "language embedding tokens H_g",
            "cell context tokens H_ctx"
          ],
          "model_visible_form_verbatim": "language embedding tokens H_g",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "combined into H_ctx and aligned with the pre-trained LLM token embedding",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "gene expression profile xg",
          "section_heading": "2.3. Training and Generation",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            10,
            11,
            12,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/pictures/2",
            "#/pictures/3",
            "#/tables/0",
            "#/texts/156",
            "#/texts/164",
            "#/texts/172",
            "#/texts/180",
            "#/texts/52",
            "#/texts/53",
            "#/texts/54",
            "#/texts/55",
            "#/texts/56",
            "#/texts/60",
            "#/texts/68",
            "#/texts/69"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003188::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003188::0002"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_22936f4d27e9"
    },
    {
      "model_id": "model_9b1025af9fe7",
      "model_name": "LEONINE-GPT-2",
      "record_id": "full_2026-07-06__rec_003188",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_36da55ad5ac4",
      "paper_title": "Multimodal Cell Context Instruction Tuning for Conditional DNA Regulatory Sequence Generation with Large Language Models",
      "doi": "",
      "paper_url": "",
      "route_count": 3,
      "configuration_count": 2,
      "family_counts": {
        "discrete_biological_symbol_stream": 1,
        "dense_continuous_carrier": 2
      },
      "subtype_counts": {
        "native_biological_token_stream": 1,
        "direct_projected_embedding": 2
      },
      "families": [
        "discrete_biological_symbol_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "native_biological_token_stream"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "DNA sequence",
        "RNA expression profile",
        "categorical cellular metadata"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "pretraining"
      ],
      "fusion_topologies": [
        "shared_latent_alignment",
        "side_or_generative_conditioning"
      ],
      "text_roles": [
        "biological_payload",
        "metadata_or_context"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003188_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003188_07429b15f6f8/figure_002.png",
        "figure_index": 2,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic scientific figure describing a DNA large language model workflow for generated enhancer sequence design.\n\nVisible elements:\n- Top output label: “Generated Enhancer Sequence y” with example nucleotide tokens “AGAT-CG” and blue block-like sequence representations.\n- Main model block: a green horizontal module labeled “DNA Large Language Model fψ,” with example “HyenaDNA.”\n- Inputs/context modules at the bottom:\n  - Left dotted panel labeled “Multimodal Cell Context,” containing:\n    - “Cell Type Encoder”\n    - “Gene Profile Encoder”\n    - “Alignment Wc”\n    - “Alignment Wg”\n    - Cell-related icons and variables `xc`, `xg`\n    - Context representation `Hctx = [Hc, Hg]`\n  - Right dotted panel labeled “Promoter Sequence,” containing:\n    - Variable `xp`\n    - Representation `Hp`\n    - Example nucleotide sequence “CGTA-AG”\n- Arrows indicate information flow from encoded cell context and promoter sequence into the DNA language model, which produces the generated enhancer sequence.\n\nBiological source objects shown:\n- DNA/enhancer sequence\n- Promoter sequence\n- Cell type information\n- Gene profile information\n\nModel interfaces/transformations:\n- Cell type and gene profile inputs are encoded, aligned through `Wc` and `Wg`, and combined into multimodal cell context `Hctx`.\n- Promoter sequence `xp` is represented as `Hp`.\n- These representations condition or interface with a DNA large language model `fψ`.\n- The model outputs/generated sequence is labeled as enhancer sequence `y`.\n\nNo experimental results, quantitative findings, or performance claims are visible in this cropped figure.",
        "page_no": 10,
        "sha256": "3ccbd37efdb2c59e8fc2c384cda3bc7e7c27e61714232288d10edee942e7fa2e",
        "pixel_width": 869,
        "pixel_height": 273,
        "crop_box": {
          "x": 0,
          "y": 0.17,
          "width": 0.73,
          "height": 0.83
        },
        "panel_label": "left multimodal cell context + model interface",
        "visible_input_object": "cell type index x_c and gene expression profile x_g feeding Multimodal Cell Context H_ctx",
        "visible_model_interface": "Alignment W_c and W_g map x_c/x_g into H_c/H_g, combined as H_ctx = [H_c, H_g] and passed into DNA Large Language Model f_ψ",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop removes the output-only enhancer panel while keeping the cell/gene inputs, alignment transforms, H_ctx fusion, and the LLM entry interface needed to ground a real input route.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_ebddf67fefd4",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "discrete cell type index x_c",
          "actual_model_visible_form": "language embedding tokens H_c"
        },
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_df28d4f6fd67",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "promoter sequence x_p",
          "actual_model_visible_form": "promoter sequence embedding H_p"
        }
      ],
      "routes": [
        {
          "route_id": "route_df28d4f6fd67",
          "configuration_id": "config_8efd4f36f07e",
          "route_label": "LEONINE-GPT-2 promoter conditioning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "conditional enhancer generation given promoter sequence and multimodal cellular context",
          "source_object_verbatim": "promoter sequence x_p",
          "source_object_normalized": "promoter sequence",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "word embedding of f_psi",
            "promoter embedding H_p"
          ],
          "model_visible_form_verbatim": "promoter sequence embedding H_p",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "conditioned on H_ctx",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Consequently, given the cell context tokens Hctx and a pair of promoter-enhancer sequence xp y , the loss function that maximizes the likelihood of conditional probability p y x c xg x p is constructed as:",
          "section_heading": "2.2. Multimodal Cell Context Instruction Tuning",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            10,
            11,
            12,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/pictures/2",
            "#/pictures/3",
            "#/tables/0",
            "#/texts/156",
            "#/texts/164",
            "#/texts/172",
            "#/texts/180",
            "#/texts/52",
            "#/texts/53",
            "#/texts/54",
            "#/texts/55",
            "#/texts/56",
            "#/texts/57",
            "#/texts/58"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003188::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003188::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ebddf67fefd4",
          "configuration_id": "config_a1b1aead09de",
          "route_label": "LEONINE-GPT-2 cell type conditioning during feature alignment pre-training",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "aligning the multimodal cell context with the DNA token embedding",
          "source_object_verbatim": "discrete cell type index x_c",
          "source_object_normalized": "cell type index",
          "source_modality_normalized": "categorical cellular metadata",
          "transformation_chain_verbatim": [
            "trainable cell type embedding module e_phi",
            "modality alignment tensor W_e",
            "language embedding tokens H_c",
            "cell context tokens H_ctx"
          ],
          "model_visible_form_verbatim": "language embedding tokens H_c",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "combined into H_ctx and aligned with the pre-trained LLM token embedding",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Specifically, for a gene expression profile xg , we consider the pre-trained Geneformer single cell transcriptomes encoder [17], which provides the gene expression feature zg = g x g ∈ ℝ d g . The gene embedding features before the last Transformer layer are used in our experiments. For a discrete cell type index xc , we utilize a trainable cell type embedding module e ϕ ⋅ that maps the discrete cell type index into an embedding vector zc = e ϕ xc ∈ ℝ d e . Since both gene expression features and cell type embeddings are in different modalities compared to DNA tokens, we apply trainable modality alignment tensors Wg and We to convert zg and zc into language embedding tokens Hg ∈ ℝ l g × d ℎ and Hc ∈ ℝ l c × d ℎ , which have the same dimensionality as the nucleotide token embedding space dℎ in the DNA LLM:",
          "section_heading": "2.3. Training and Generation",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            10,
            11,
            12,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/pictures/2",
            "#/pictures/3",
            "#/tables/0",
            "#/texts/156",
            "#/texts/164",
            "#/texts/172",
            "#/texts/180",
            "#/texts/52",
            "#/texts/53",
            "#/texts/54",
            "#/texts/55",
            "#/texts/56",
            "#/texts/60",
            "#/texts/68",
            "#/texts/69"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003188::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003188::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_507b0019ae79",
          "configuration_id": "config_a1b1aead09de",
          "route_label": "LEONINE-GPT-2 gene expression conditioning during feature alignment pre-training",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "aligning the multimodal cell context with the DNA token embedding",
          "source_object_verbatim": "gene expression profile x_g",
          "source_object_normalized": "gene expression profile",
          "source_modality_normalized": "RNA expression profile",
          "transformation_chain_verbatim": [
            "pre-trained Geneformer single cell transcriptomes encoder",
            "modality alignment tensor W_g",
            "language embedding tokens H_g",
            "cell context tokens H_ctx"
          ],
          "model_visible_form_verbatim": "language embedding tokens H_g",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "combined into H_ctx and aligned with the pre-trained LLM token embedding",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "gene expression profile xg",
          "section_heading": "2.3. Training and Generation",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            10,
            11,
            12,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/pictures/2",
            "#/pictures/3",
            "#/tables/0",
            "#/texts/156",
            "#/texts/164",
            "#/texts/172",
            "#/texts/180",
            "#/texts/52",
            "#/texts/53",
            "#/texts/54",
            "#/texts/55",
            "#/texts/56",
            "#/texts/60",
            "#/texts/68",
            "#/texts/69"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003188::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003188::0002"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_22936f4d27e9"
    },
    {
      "model_id": "model_0ea0b64c5788",
      "model_name": "LLaMA-2 7B",
      "record_id": "full_2026-07-06__rec_001381",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_d0e026b8e9c9",
      "paper_title": "Language-Enhanced Representation Learning for Single-Cell Transcriptomics",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "pretraining"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001381_figure_007.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001381_9e666a0934af/figure_007.png",
        "figure_index": 7,
        "caption": "Figure 3: Overview of scMMGPT. (1) Cross-modal Discriminative Objective: Given paired cell and text inputs, the model learns to identify the correct textual description of a cell by aligning the outputs of the scLLM and text LLM. (2, 3) Cross-modal Generative Objectives: scMMGPT strengthens multimodal alignment through a unified generative pre-training strategy, jointly optimizing cell-to-text and text-to-cell translation tasks to facilitate bidirectional knowledge transfer.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow for multimodal cell information modeling using single-cell RNA-seq data and text.\n\nVisible components:\n\n- Left panel: “Multimodal Cell Information”\n  - Shows a human body/lung illustration with arrows labeled “scRNA-seq” and “Biological Databases.”\n  - Shows an “Experimental Evidence in scRNA-seq Matrices” block: a heatmap-like matrix with rows labeled Cell 1, Cell 2, …, Cell N and columns labeled Gene 1, Gene 2, …, Gene M.\n  - Shows “Domain Knowledge in Textual Descriptions”: an example text box describing a “classical monocyte,” sourced from human lung female donor samples.\n\n- Middle panel: “scRNA-seq Data”\n  - Gene expression values are transformed into “Gene Tokens.”\n  - Example gene-expression bar/sequence includes gene names such as SEC23B, MT-CO2, RPL37, MT-CYB and values like 127, 3, 18, 69.\n  - Text descriptions are transformed into a “Text Token Sequence,” with tokens such as “[BOS], This, cell, is, a, …, [EOS].”\n\n- Main model panel:\n  - A blue “scLLM” block processes gene tokens.\n  - An orange “Text LLM” block processes text tokens.\n  - Cross-modal projectors connect the two modalities:\n    - “Q-former Projector”\n    - “Cross-Attn Projection”\n  - Intermediate feature boxes are labeled “Cell Features” and “Text Features.”\n  - Arrows indicate cell-to-text and text-to-cell projection paths.\n  - A central label reads “Cross-modal Projectors.”\n\n- Right panel: output tasks/findings\n  - “1. Cell-Text Discrimination”: produces a “Relevance Score.”\n  - “2. Text-to-Cell Generation”: decodes text-derived information into a gene-expression/cell representation.\n  - “3. Cell-To-Text Generation”: decodes cell representation into a textual description, shown as the classical monocyte example.\n\nOverall, the figure depicts a multimodal architecture aligning scRNA-seq gene-token representations with textual biological descriptions using an scLLM, a text LLM, and cross-modal projection modules for discrimination and bidirectional generation tasks.",
        "page_no": 4,
        "sha256": "c1acaeb6141c4bb543733f074f0bcdef463979db71492c67f4ffa275d4491c4d",
        "pixel_width": 786,
        "pixel_height": 253,
        "crop_box": {
          "x": 0.0,
          "y": 0.42,
          "width": 0.78,
          "height": 0.58
        },
        "panel_label": "Text route / tokenization interface",
        "visible_input_object": "Textual description of a classical monocyte in the lower-left source box",
        "visible_model_interface": "Text Token Sequence feeding the Text LLM via the tokenizer and arrow into the orange Text LLM block",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the grounded text-input route: the source cell description, the tokenized text sequence, and the immediate insertion interface into Text LLM. It excludes the output-only right panel and the unrelated scRNA-seq/top pathway.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_ba81ad8a8e8d",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "the textual description of cells",
          "actual_model_visible_form": "a sequence of tokens t ( i ) = [ t ( i ) 1 , t ( i ) 2 , . . . , t ( i ) T ]"
        }
      ],
      "routes": [
        {
          "route_id": "route_ba81ad8a8e8d",
          "configuration_id": "config_2994db049e48",
          "route_label": "Text LLM tokenization of cell descriptions",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "text generation",
          "source_object_verbatim": "the textual description of cells",
          "source_object_normalized": "cell textual descriptions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenize the textual description of cells into a sequence of tokens"
          ],
          "model_visible_form_verbatim": "a sequence of tokens t ( i ) = [ t ( i ) 1 , t ( i ) 2 , . . . , t ( i ) T ]",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "using the tokenizer in text LLM",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Text LLM for Text Generation. For the text LLM, we utilize LLaMA-2 7B [47], a decoder-only transformer that excels at text generation. Its extensive pre-training and generative architecture make it well-suited for biomedical text understanding and generation, such as describing cellular states. We tokenize the textual description of cells into a sequence of tokens t ( i ) = [ t ( i ) 1 , t ( i ) 2 , . . . , t ( i ) T ] using the tokenizer in text LLM.",
          "section_heading": "3.2 Cell Representation Learning & Language Generation with Pre-Trained Models",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            17
          ],
          "doc_item_refs": [
            "#/texts/203",
            "#/texts/468",
            "#/texts/470",
            "#/texts/471"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001381::0067",
            "dense::full_2026-07-06__rec_001381::0084",
            "dense::full_2026-07-06__rec_001381::0085"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_0cc72bd82901"
    },
    {
      "model_id": "model_4b363c370943",
      "model_name": "LLaMA2-13B",
      "record_id": "full_2026-07-06__rec_003043",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_3bbf77870504",
      "paper_title": "Geneverse: A collection of Open-source Multimodal Large Language Models for Genomic and Proteomic Research",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003043_figure_006.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003043_460a3d629653/figure_006.png",
        "figure_index": 6,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a two-panel schematic labeled **a** and **b**, each titled **“Instruction example”** and showing an LLM-style question-answer interface.\n\nPanel **a** shows an example involving a protein structure. The prompt asks: “What is the name of this protein? Please summarize your answer in one sentence.” A small ribbon-like protein structure image appears beside the prompt. The LLM response box answers: “It is AF-O43280-F1.”\n\nPanel **b** shows an example involving biological microscopy/histology imagery. The prompt asks whether **gene APOC1** is a marker gene of **cell type Macrophage**, followed by “Please summarize your answer in one sentence.” A purple-stained tissue microscopy image appears beside the prompt. The LLM response box answers: “Yes.”\n\nBoth panels depict instruction-following model interfaces with biological source objects as inputs: a protein structure in panel **a** and a histological tissue image plus gene/cell-type query in panel **b**. The figure illustrates LLM question answering over biological or biomedical visual/contextual inputs.",
        "page_no": 16,
        "sha256": "9e63b9a5663cc3acc3d474d94d9d01570d390f25288c9abcc3bc702f6efb3458",
        "pixel_width": 901,
        "pixel_height": 373,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.49,
          "height": 1.0
        },
        "panel_label": "a",
        "visible_input_object": "Protein-structure instruction example with text prompt and protein image",
        "visible_model_interface": "Text-only LMM prompt-to-answer scaffold with the instruction box and response box",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "Panel a is the smallest coherent crop that preserves the grounded input route: the instruction prompt, the biological source object, and the LMM response interface. It excludes panel b and keeps the labels readable.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_9640f876d90e",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene name in the prompt",
          "actual_model_visible_form": "text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_9640f876d90e",
          "configuration_id": "config_dab2b272adfe",
          "route_label": "gene-function prompt to LLaMA2-13B",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "gene function description",
          "source_object_verbatim": "gene name in the prompt",
          "source_object_normalized": "GLI1",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "instruction example"
          ],
          "model_visible_form_verbatim": "text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "text-only prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Our task is to generate the summary of gene functions based on the prompt only containing the task description.",
          "section_heading": "4.1 Benchmarking LLMs for summarizing of gene functions",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            16
          ],
          "doc_item_refs": [
            "#/texts/438",
            "#/texts/70",
            "#/texts/71",
            "#/texts/72",
            "#/texts/73"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003043::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003043::0035"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_c443713ee155",
      "model_name": "LLaMA2-7B",
      "record_id": "full_2026-07-06__rec_003043",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_3bbf77870504",
      "paper_title": "Geneverse: A collection of Open-source Multimodal Large Language Models for Genomic and Proteomic Research",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003043_figure_006.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003043_460a3d629653/figure_006.png",
        "figure_index": 6,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a two-panel schematic labeled **a** and **b**, each titled **“Instruction example”** and showing an LLM-style question-answer interface.\n\nPanel **a** shows an example involving a protein structure. The prompt asks: “What is the name of this protein? Please summarize your answer in one sentence.” A small ribbon-like protein structure image appears beside the prompt. The LLM response box answers: “It is AF-O43280-F1.”\n\nPanel **b** shows an example involving biological microscopy/histology imagery. The prompt asks whether **gene APOC1** is a marker gene of **cell type Macrophage**, followed by “Please summarize your answer in one sentence.” A purple-stained tissue microscopy image appears beside the prompt. The LLM response box answers: “Yes.”\n\nBoth panels depict instruction-following model interfaces with biological source objects as inputs: a protein structure in panel **a** and a histological tissue image plus gene/cell-type query in panel **b**. The figure illustrates LLM question answering over biological or biomedical visual/contextual inputs.",
        "page_no": 16,
        "sha256": "9e63b9a5663cc3acc3d474d94d9d01570d390f25288c9abcc3bc702f6efb3458",
        "pixel_width": 901,
        "pixel_height": 373,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.49,
          "height": 0.46
        },
        "panel_label": "a",
        "visible_input_object": "Protein structure image paired with the instruction-example text prompt.",
        "visible_model_interface": "Text-only instruction prompt box for the LMM input route.",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps panel a’s source object, the instruction-example prompt, and the text-only input interface while excluding the output-only response box and panel b.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_22d13b63dfa8",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene name in the prompt",
          "actual_model_visible_form": "text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_22d13b63dfa8",
          "configuration_id": "config_293dd526689e",
          "route_label": "gene-function prompt to LLaMA2-7B",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "gene function description",
          "source_object_verbatim": "gene name in the prompt",
          "source_object_normalized": "GLI1",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "instruction example"
          ],
          "model_visible_form_verbatim": "text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "text-only prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Our task is to generate the summary of gene functions based on the prompt only containing the task description.",
          "section_heading": "4.1 Benchmarking LLMs for summarizing of gene functions",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            16
          ],
          "doc_item_refs": [
            "#/texts/438",
            "#/texts/70",
            "#/texts/71",
            "#/texts/72",
            "#/texts/73"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003043::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003043::0035"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_78760ba29751",
      "model_name": "LLaMA3-8B",
      "record_id": "full_2026-07-06__rec_003043",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_3bbf77870504",
      "paper_title": "Geneverse: A collection of Open-source Multimodal Large Language Models for Genomic and Proteomic Research",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003043_figure_006.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003043_460a3d629653/figure_006.png",
        "figure_index": 6,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a two-panel schematic labeled **a** and **b**, each titled **“Instruction example”** and showing an LLM-style question-answer interface.\n\nPanel **a** shows an example involving a protein structure. The prompt asks: “What is the name of this protein? Please summarize your answer in one sentence.” A small ribbon-like protein structure image appears beside the prompt. The LLM response box answers: “It is AF-O43280-F1.”\n\nPanel **b** shows an example involving biological microscopy/histology imagery. The prompt asks whether **gene APOC1** is a marker gene of **cell type Macrophage**, followed by “Please summarize your answer in one sentence.” A purple-stained tissue microscopy image appears beside the prompt. The LLM response box answers: “Yes.”\n\nBoth panels depict instruction-following model interfaces with biological source objects as inputs: a protein structure in panel **a** and a histological tissue image plus gene/cell-type query in panel **b**. The figure illustrates LLM question answering over biological or biomedical visual/contextual inputs.",
        "page_no": 16,
        "sha256": "9e63b9a5663cc3acc3d474d94d9d01570d390f25288c9abcc3bc702f6efb3458",
        "pixel_width": 901,
        "pixel_height": 373,
        "crop_box": {
          "x": 0.45,
          "y": 0.0,
          "width": 0.55,
          "height": 0.43
        },
        "panel_label": "b",
        "visible_input_object": "Text prompt asking whether gene APOC1 is a marker for macrophage cell type",
        "visible_model_interface": "Panel b instruction-example prompt bubble with the text-question interface",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the top half of panel b so the gene query, instruction-example label, and surrounding prompt layout remain readable while excluding the answer box/output-only region.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_e3e67790db59",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene name in the prompt",
          "actual_model_visible_form": "text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_e3e67790db59",
          "configuration_id": "config_77a4478e9b26",
          "route_label": "gene-function prompt to LLaMA3-8B",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "gene function description",
          "source_object_verbatim": "gene name in the prompt",
          "source_object_normalized": "GLI1",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "instruction example"
          ],
          "model_visible_form_verbatim": "text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "text-only prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Our task is to generate the summary of gene functions based on the prompt only containing the task description.",
          "section_heading": "4.1 Benchmarking LLMs for summarizing of gene functions",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            16
          ],
          "doc_item_refs": [
            "#/texts/438",
            "#/texts/70",
            "#/texts/71",
            "#/texts/72",
            "#/texts/73"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003043::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003043::0035"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_61aa6fc79089",
      "model_name": "Llama3.1-8B-Instruct",
      "record_id": "full_2026-07-06__rec_001617",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4f442dcf4ed4",
      "paper_title": "Cell-o1: Training LLMs to Solve Single-Cell Reasoning Puzzles with Reinforcement Learning",
      "doi": "10.48550/arXiv.2506.02911",
      "paper_url": "https://doi.org/10.48550/arXiv.2506.02911",
      "route_count": 4,
      "configuration_count": 4,
      "family_counts": {
        "text_native_token_stream": 4
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 4
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell RNA sequencing batch with donor metadata",
        "single-cell RNA sequencing cell with donor metadata"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001617_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001617_76b0056d59c3/figure_002.png",
        "figure_index": 2,
        "caption": "Figure 2: CellPuzzles formulates cell type annotation as a batch-level reasoning task that integrates gene expression and contextual metadata, inspired by how experts annotate cells in practice.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic comparison of two cell type annotation workflows with four main panel regions arranged in two rows.\n\nTop row: A human expert task workflow. The left “Input” panel shows clustered single-cell data with colored clusters labeled Cluster 2, Cluster 3, Cluster 4, and Cluster N, each associated with marker gene lists such as “SFTPC, SFTPA1, SFTPB, NAPSA, ...”, “C1QA, APOE, MARCO, LYZ, ...”, and “IGHM, MZB1, JCHAIN, XBP1, ...”. The center panel labeled “Human Expert Task” shows an expert assigning a cell type label to each cluster using representative marker genes, contextual metadata, biological knowledge, and reference sources. The right “Output” panel lists assigned cell types including Pulmonary Alveolar Type 2 (AT2) Cells, CD8+ Cytotoxic T Cells, Lung Pericyte, Alveolar Macrophages, and Plasma Cells.\n\nBottom row: A “CellPuzzles Task” workflow. The left “Input” panel shows individual cells from clusters, with representative sampled cells labeled [Cell 1], [Cell 2], [Cell 3], [Cell 4], and [Cell N], each paired with gene lists such as “S100A9, TMSB10, RPL37, ...”, “MALAT1, FTL, B2M, ...”, “MALAT1, FTL, AKR1B10, ...”, “B2M, MT2A, TMSB4X, ...”, and “IGLC3, IGLC2, IGHM, ...”. Arrows indicate selected representative cells from clusters. The center panel labeled “CellPuzzles Task” shows a model/interface labeled “Cell-o1” receiving N cells, contextual metadata, and N candidate cell types, then reasoning to determine the optimal label assignment and provide reasoning traces. The right “Output” panel shows colored cell-to-label assignment lines connecting cells to candidate labels including CD4-positive, alpha-beta T Cell; CD8-positive, alpha-beta T Cell; Smooth Muscle Cell; Lung Macrophage; Non-classical Monocyte; Capillary Endothelial Cell; Plasma Cell; and Respiratory Basal Cell.\n\nBiological source objects include single-cell clusters, individual cells, marker genes, and immune/lung-related cell type labels. The figure depicts a transformation from marker gene or cell-level input data to annotated biological cell type outputs, comparing expert manual annotation with an automated CellPuzzles/Cell-o1 reasoning task.",
        "page_no": 3,
        "sha256": "e74192b02a6c544c4d4ae6c01c1b71c75747f0270a5e4e6bc77aba0c8a2259ee",
        "pixel_width": 791,
        "pixel_height": 318,
        "crop_box": {
          "x": 0,
          "y": 0.49,
          "width": 0.78,
          "height": 0.51
        },
        "panel_label": "bottom row: CellPuzzles input -> Cell-o1 task",
        "visible_input_object": "Representative sampled cells from clusters with top genes and cluster-to-cell arrows",
        "visible_model_interface": "Cell-o1 batch-level task box showing N cells, contextual metadata, N candidate cell types, and reasoning-based label assignment",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop preserves the grounded input route from clustered cells to representative cells and into the CellPuzzles interface, while excluding the output-only panel on the right and the unrelated expert-workflow row above.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_ae1471c9185d",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "the gene expression profile of a single cell from a specific donor",
          "actual_model_visible_form": "a single-cell prompt with donor context and candidate labels, without reasoning traces"
        }
      ],
      "routes": [
        {
          "route_id": "route_ae1471c9185d",
          "configuration_id": "config_ec21e5b052f3",
          "route_label": "cell-level prediction prompt (Llama3.1-8B-Instruct)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cell-level Prediction Setup",
          "source_object_verbatim": "the gene expression profile of a single cell from a specific donor",
          "source_object_normalized": "single cell from a specific donor with top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing cell with donor metadata",
          "transformation_chain_verbatim": [
            "use the single cell's top expressed genes",
            "combine them with donor context",
            "provide the fixed candidate label set",
            "directly classify without reasoning traces"
          ],
          "model_visible_form_verbatim": "a single-cell prompt with donor context and candidate labels, without reasoning traces",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "direct prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "directly predict answers from the input without generating reasoning traces",
          "section_heading": "B.2 Baselines",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            21,
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/8",
            "#/tables/9",
            "#/texts/1015",
            "#/texts/1016",
            "#/texts/1017",
            "#/texts/1018",
            "#/texts/1019",
            "#/texts/1020"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_28b3534a8da1",
          "configuration_id": "config_2fe58c8bb499",
          "route_label": "batch-level prediction prompt (Llama3.1-8B-Instruct)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Batch-level Prediction Setup",
          "source_object_verbatim": "a batch of N cells from the same donor",
          "source_object_normalized": "batch of N cells from the same donor with ranked top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "combine with donor context",
            "present the candidate label set",
            "directly predict answers without reasoning traces"
          ],
          "model_visible_form_verbatim": "batch-level structured input without reasoning traces",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "direct prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are an expert assistant specialized in cell type annotation. You will be given a batch of N cells from the same donor, where each cell represents a unique cell type. For each cell, the top-expressed genes are provided in descending order of expression. Using both the gene expression data and donor information, determine the correct cell type for each cell. You will also receive a list of N candidate cell types, and each candidate must be assigned to exactly one cell. Ensure that you consider all cells and candidate types together, rather than annotating each cell individually. Include your detailed reasoning within <think> and </think> tags, and provide your final answer within <answer> and </answer> tags. The final answer should be a single string listing the assigned cell types in order, separated by ' | '.",
          "section_heading": "B.2 Baselines",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/603",
            "#/texts/604",
            "#/texts/605"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1f9a26bc7035",
          "configuration_id": "config_661abd234cc0",
          "route_label": "open-ended QA prompt (Llama3.1-8B-Instruct)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Open-ended QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "provide gene expression profile and metadata",
            "remove the constrained label set",
            "ask for free-form cell type generation"
          ],
          "model_visible_form_verbatim": "free-form textual generation of a cell type name for each cell",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompting without constrained labels",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_015"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b001b0667306",
          "configuration_id": "config_b560295f29f6",
          "route_label": "constrained QA prompt (Llama3.1-8B-Instruct)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Constrained QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "convert donor metadata into natural language context",
            "attach a predefined candidate label set",
            "require a single ordered answer string"
          ],
          "model_visible_form_verbatim": "a structured batch-level text prompt with candidate labels and ordered answer output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "predefined candidate label set with structured prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_021"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_af471cc4e529",
      "model_name": "LLaMAPro-8B",
      "record_id": "full_2026-07-06__rec_003043",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_3bbf77870504",
      "paper_title": "Geneverse: A collection of Open-source Multimodal Large Language Models for Genomic and Proteomic Research",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "The contact sheet does show an immediate model interface in Figure 6, but it is a multimodal instruction example with an image input, not the specific text-only gene-function prompt route for LLaMAPro-8B. The visible figures that are closer are either evaluation/benchmark plots or generic workflow diagrams, so they do not ground the requested input route.",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_3af6eeab6bde",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene name in the prompt",
          "actual_model_visible_form": "text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_3af6eeab6bde",
          "configuration_id": "config_60ee7fa24e26",
          "route_label": "gene-function prompt to LLaMAPro-8B",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "gene function description",
          "source_object_verbatim": "gene name in the prompt",
          "source_object_normalized": "GLI1",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "instruction example"
          ],
          "model_visible_form_verbatim": "text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "text-only prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Our task is to generate the summary of gene functions based on the prompt only containing the task description.",
          "section_heading": "4.1 Benchmarking LLMs for summarizing of gene functions",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            16
          ],
          "doc_item_refs": [
            "#/texts/438",
            "#/texts/70",
            "#/texts/71",
            "#/texts/72",
            "#/texts/73"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003043::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003043::0035"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_f35a32a6436b",
      "model_name": "LLaVA-7B (LoRA)",
      "record_id": "full_2026-07-06__rec_003043",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_3bbf77870504",
      "paper_title": "Geneverse: A collection of Open-source Multimodal Large Language Models for Genomic and Proteomic Research",
      "doi": "",
      "paper_url": "",
      "route_count": 2,
      "configuration_count": 2,
      "family_counts": {
        "visual_raster_carrier": 2
      },
      "subtype_counts": {
        "raw_slide_or_patch_input": 2
      },
      "families": [
        "visual_raster_carrier"
      ],
      "subtypes": [
        "raw_slide_or_patch_input"
      ],
      "primary_subtype": "raw_slide_or_patch_input",
      "modalities": [
        "visual raster"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "unclear"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003043_figure_006.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003043_460a3d629653/figure_006.png",
        "figure_index": 6,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a two-panel schematic labeled **a** and **b**, each titled **“Instruction example”** and showing an LLM-style question-answer interface.\n\nPanel **a** shows an example involving a protein structure. The prompt asks: “What is the name of this protein? Please summarize your answer in one sentence.” A small ribbon-like protein structure image appears beside the prompt. The LLM response box answers: “It is AF-O43280-F1.”\n\nPanel **b** shows an example involving biological microscopy/histology imagery. The prompt asks whether **gene APOC1** is a marker gene of **cell type Macrophage**, followed by “Please summarize your answer in one sentence.” A purple-stained tissue microscopy image appears beside the prompt. The LLM response box answers: “Yes.”\n\nBoth panels depict instruction-following model interfaces with biological source objects as inputs: a protein structure in panel **a** and a histological tissue image plus gene/cell-type query in panel **b**. The figure illustrates LLM question answering over biological or biomedical visual/contextual inputs.",
        "page_no": 16,
        "sha256": "9e63b9a5663cc3acc3d474d94d9d01570d390f25288c9abcc3bc702f6efb3458",
        "pixel_width": 901,
        "pixel_height": 373,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.49,
          "height": 0.53
        },
        "panel_label": "a",
        "visible_input_object": "Protein structure image next to the instruction prompt in panel a",
        "visible_model_interface": "Instruction-example prompt for LMM, including the source question and the LMM label/robot icon",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the left panel only and keeps the grounded input route: the protein structure source object, the instruction prompt, and the LMM interface label. It excludes the answer/output box and the unrelated right panel.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_f37dc335797b",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "protein structure images from the databases of AlphaFold2",
          "actual_model_visible_form": "image plus text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_f37dc335797b",
          "configuration_id": "config_dcbabd4d7ffe",
          "route_label": "protein structure image to LLaVA-7B (LoRA)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "protein classification",
          "source_object_verbatim": "protein structure images from the databases of AlphaFold2",
          "source_object_normalized": "protein structure images from AlphaFold2",
          "source_modality_normalized": "visual raster",
          "transformation_chain_verbatim": [
            "downloaded from the DeepMind Alphafold2 website",
            "load the image information using Pymol",
            "construct the instruction tuning dataset"
          ],
          "model_visible_form_verbatim": "image plus text prompt",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "LLaVA with LoRA",
          "fusion_topology": "unclear",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The images of protein structure for training and testing come from the databases of AlphaFold2.",
          "section_heading": "4.3 Finetuning an MLLM for genomic and proteomic application",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper does not specify the exact image-text fusion block used inside the LLaVA+LoRA setup.",
          "pages": [
            3,
            4,
            6,
            7
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/59",
            "#/texts/60",
            "#/texts/65",
            "#/texts/86",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89",
            "#/texts/90"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003043::route_019"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003043::0005",
            "dense::full_2026-07-06__rec_003043::0032",
            "dense::full_2026-07-06__rec_003043::0033"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5496539f3e31",
          "configuration_id": "config_1be2942b6999",
          "route_label": "spatial transcriptomic image to LLaVA-7B (LoRA)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "marker gene identification",
          "source_object_verbatim": "spatial transcriptomic data from human breast tissue",
          "source_object_normalized": "spatial transcriptomic data from human breast tissue",
          "source_modality_normalized": "visual raster",
          "transformation_chain_verbatim": [
            "images from Lin et al. (2020) collected from human breast tissue",
            "construct the instruction tuning dataset"
          ],
          "model_visible_form_verbatim": "image plus text prompt",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "LLaVA with LoRA",
          "fusion_topology": "unclear",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The spatial transcriptomic data for training and testing come from (Lin et al., 2020) collected from human breast tissue.",
          "section_heading": "4.3 Finetuning an MLLM for genomic and proteomic application",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper does not specify the exact image-text fusion block used inside the LLaVA+LoRA setup.",
          "pages": [
            3,
            4,
            6,
            7
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/59",
            "#/texts/60",
            "#/texts/65",
            "#/texts/86",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89",
            "#/texts/90",
            "#/texts/92"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003043::route_025"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003043::0006",
            "dense::full_2026-07-06__rec_003043::0032"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_cb06ed43c31a"
    },
    {
      "model_id": "model_4ae9868de1c9",
      "model_name": "Longevity-LLM v0.1",
      "record_id": "full_2026-07-06__rec_003434",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_74691182f5b8",
      "paper_title": "The End of Aging Clocks: Training Foundation Models to Reason in Aging and Longevity",
      "doi": "10.64898/2026.03.28.714980",
      "paper_url": "https://doi.org/10.64898/2026.03.28.714980",
      "route_count": 9,
      "configuration_count": 9,
      "family_counts": {
        "text_native_token_stream": 9
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 4,
        "structured_biological_prompt_or_task_scaffold": 5
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "DNA methylation",
        "DNA methylation-derived coefficients",
        "RNA",
        "clinical biomarker tabular data",
        "genetic/phenotypic tabular data",
        "oncology survival tabular data",
        "protein/peptide"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "The contact sheet only shows downstream evaluation figures: accuracy/MAE bar charts, predicted-vs-true scatter plots, and a Jaccard ranking. None visibly presents an input route or immediate model interface such as CpG beta-value prompts, annotated CpG panels, gene-expression profiles, Olink plasma profiles, or prompt scaffolds. Because the visible content is benchmarking output rather than source-to-model input, no figure is supportable.",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_18bdd6a8183d",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "DNA methylation data",
          "actual_model_visible_form": "CpG beta values"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_9ab0d72dd00d",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "published clock coefficients",
          "actual_model_visible_form": "comparison prompts"
        }
      ],
      "routes": [
        {
          "route_id": "route_18bdd6a8183d",
          "configuration_id": "config_4ccb1980ae00",
          "route_label": "DNAm beta-value prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "age regression",
          "source_object_verbatim": "DNA methylation data",
          "source_object_normalized": "DNA methylation profiles",
          "source_modality_normalized": "DNA methylation",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "CpG beta values",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "full-parameter fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "The training corpus comprised 766,640 examples (1.09 billion tokens) spanning 38 tasks across four  primary  data  modalities,  with  an  additional  191,804  examples  reserved  for  holdout evaluation  ( Supplementary  File  1 ).  DNAm  data  constituted  the  largest  modality  (416,110 training  prompts,  875  million  tokens),  including  CpG  beta  value  profiles  for  age  regression, pairwise  feature  importance  comparisons  derived  from  published  clock  coefficients,  and annotated CpG panels with genomic and pathway-level reasoning traces. Clinical biomarker data from  NHANES 24 contributed  228,381  training  examples  across  age  prediction,  mortality classification,  and  time-to-event  tasks  in  multiple  prompt  formats.  Transcriptomic  data  from GTEx  Portal 25 comprised  98,510  training  prompts  for  age  prediction  from  gene  expression profiles and pairwise tissue age comparisons. Proteomic data from Olink Explore 3072 plasma panels  represented  the  smallest  modality  (7,807  training  examples) with  age  regression, classification,  pairwise  comparison,  and  protein  profile  generation  tasks.  Data  for  proteomic prompt collections  was  obtained  from 26,27 and  the  Immunobiology  of Aging cohort  described originally in 28 and accessed through the Allen Institute of Immunology web portal. Additional tasks in the corpus also included TCGA cancer survival comparisons (obtained through TCGA Research Network 29 ), multi-mutant lifespan prediction, and general aging biology tasks.",
          "section_heading": "Training data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            9,
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/20",
            "#/texts/21",
            "#/texts/26",
            "#/texts/27",
            "#/texts/42",
            "#/texts/43",
            "#/texts/44",
            "#/texts/45",
            "#/texts/54",
            "#/texts/56",
            "#/texts/57",
            "#/texts/58",
            "#/texts/78",
            "#/texts/80",
            "#/texts/81",
            "#/texts/83"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003434::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003434::0001",
            "dense::full_2026-07-06__rec_003434::0003",
            "dense::full_2026-07-06__rec_003434::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9ab0d72dd00d",
          "configuration_id": "config_02463c1a6545",
          "route_label": "Clock-coefficient comparison prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "pairwise feature importance comparisons derived from published clock coefficients",
          "source_object_verbatim": "published clock coefficients",
          "source_object_normalized": "published clock coefficients",
          "source_modality_normalized": "DNA methylation-derived coefficients",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "comparison prompts",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "full-parameter fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "pairwise feature importance comparisons derived from published clock coefficients",
          "section_heading": "Training data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            6,
            7,
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/20",
            "#/texts/21",
            "#/texts/26",
            "#/texts/27",
            "#/texts/54",
            "#/texts/56",
            "#/texts/57",
            "#/texts/78",
            "#/texts/80",
            "#/texts/81",
            "#/texts/83"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003434::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003434::0001",
            "dense::full_2026-07-06__rec_003434::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_324a8991212a",
          "configuration_id": "config_e516832d189e",
          "route_label": "Annotated CpG reasoning prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "annotated CpG panels with genomic and pathway-level reasoning traces",
          "source_object_verbatim": "annotated CpG panels",
          "source_object_normalized": "annotated CpG panels",
          "source_modality_normalized": "DNA methylation",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "CpG panels",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "full-parameter fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "annotated CpG panels with genomic and pathway-level reasoning traces",
          "section_heading": "Training data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            6,
            7,
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/20",
            "#/texts/21",
            "#/texts/26",
            "#/texts/27",
            "#/texts/54",
            "#/texts/56",
            "#/texts/57",
            "#/texts/78",
            "#/texts/80",
            "#/texts/81"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003434::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003434::0001",
            "dense::full_2026-07-06__rec_003434::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_40fd38c19186",
          "configuration_id": "config_96d55f31f26e",
          "route_label": "NHANES clinical biomarker prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "age prediction, mortality classification, and time-to-event tasks in multiple prompt formats",
          "source_object_verbatim": "clinical biomarker data from NHANES",
          "source_object_normalized": "clinical biomarker data",
          "source_modality_normalized": "clinical biomarker tabular data",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "multiple prompt formats",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "full-parameter fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The training corpus comprised 766,640 examples (1.09 billion tokens) spanning 38 tasks across four  primary  data  modalities,  with  an  additional  191,804  examples  reserved  for  holdout evaluation  ( Supplementary  File  1 ).  DNAm  data  constituted  the  largest  modality  (416,110 training  prompts,  875  million  tokens),  including  CpG  beta  value  profiles  for  age  regression, pairwise  feature  importance  comparisons  derived  from  published  clock  coefficients,  and annotated CpG panels with genomic and pathway-level reasoning traces. Clinical biomarker data from  NHANES 24 contributed  228,381  training  examples  across  age  prediction,  mortality classification,  and  time-to-event  tasks  in  multiple  prompt  formats.  Transcriptomic  data  from GTEx  Portal 25 comprised  98,510  training  prompts  for  age  prediction  from  gene  expression profiles and pairwise tissue age comparisons. Proteomic data from Olink Explore 3072 plasma panels  represented  the  smallest  modality  (7,807  training  examples) with  age  regression, classification,  pairwise  comparison,  and  protein  profile  generation  tasks.  Data  for  proteomic prompt collections  was  obtained  from 26,27 and  the  Immunobiology  of Aging cohort  described originally in 28 and accessed through the Allen Institute of Immunology web portal. Additional tasks in the corpus also included TCGA cancer survival comparisons (obtained through TCGA Research Network 29 ), multi-mutant lifespan prediction, and general aging biology tasks.",
          "section_heading": "Training data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            6,
            7,
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/20",
            "#/texts/21",
            "#/texts/26",
            "#/texts/27",
            "#/texts/54",
            "#/texts/56",
            "#/texts/57",
            "#/texts/78",
            "#/texts/80",
            "#/texts/81",
            "#/texts/83"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003434::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003434::0001",
            "dense::full_2026-07-06__rec_003434::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d7decdb9719d",
          "configuration_id": "config_ca725f9d2b5b",
          "route_label": "GTEx transcriptomic prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "age prediction from gene expression profiles and pairwise tissue age comparisons",
          "source_object_verbatim": "transcriptomic data from GTEx Portal",
          "source_object_normalized": "transcriptomic data",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "gene expression profiles",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "full-parameter fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "The training corpus comprised 766,640 examples (1.09 billion tokens) spanning 38 tasks across four  primary  data  modalities,  with  an  additional  191,804  examples  reserved  for  holdout evaluation  ( Supplementary  File  1 ).  DNAm  data  constituted  the  largest  modality  (416,110 training  prompts,  875  million  tokens),  including  CpG  beta  value  profiles  for  age  regression, pairwise  feature  importance  comparisons  derived  from  published  clock  coefficients,  and annotated CpG panels with genomic and pathway-level reasoning traces. Clinical biomarker data from  NHANES 24 contributed  228,381  training  examples  across  age  prediction,  mortality classification,  and  time-to-event  tasks  in  multiple  prompt  formats.  Transcriptomic  data  from GTEx  Portal 25 comprised  98,510  training  prompts  for  age  prediction  from  gene  expression profiles and pairwise tissue age comparisons. Proteomic data from Olink Explore 3072 plasma panels  represented  the  smallest  modality  (7,807  training  examples) with  age  regression, classification,  pairwise  comparison,  and  protein  profile  generation  tasks.  Data  for  proteomic prompt collections  was  obtained  from 26,27 and  the  Immunobiology  of Aging cohort  described originally in 28 and accessed through the Allen Institute of Immunology web portal. Additional tasks in the corpus also included TCGA cancer survival comparisons (obtained through TCGA Research Network 29 ), multi-mutant lifespan prediction, and general aging biology tasks.",
          "section_heading": "Training data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            6,
            7,
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/20",
            "#/texts/21",
            "#/texts/26",
            "#/texts/27",
            "#/texts/54",
            "#/texts/56",
            "#/texts/57",
            "#/texts/78",
            "#/texts/80",
            "#/texts/81",
            "#/texts/83"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003434::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003434::0001",
            "dense::full_2026-07-06__rec_003434::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_973a54854497",
          "configuration_id": "config_930eb89cb323",
          "route_label": "Olink plasma benchmark prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "age regression, classification, and pairwise comparison tasks",
          "source_object_verbatim": "proteomic data from Olink Explore 3072 plasma panels",
          "source_object_normalized": "proteomic plasma profiles",
          "source_modality_normalized": "protein/peptide",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Olink Explore 3072 plasma panels",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "full-parameter fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "The training corpus comprised 766,640 examples (1.09 billion tokens) spanning 38 tasks across four  primary  data  modalities,  with  an  additional  191,804  examples  reserved  for  holdout evaluation  ( Supplementary  File  1 ).  DNAm  data  constituted  the  largest  modality  (416,110 training  prompts,  875  million  tokens),  including  CpG  beta  value  profiles  for  age  regression, pairwise  feature  importance  comparisons  derived  from  published  clock  coefficients,  and annotated CpG panels with genomic and pathway-level reasoning traces. Clinical biomarker data from  NHANES 24 contributed  228,381  training  examples  across  age  prediction,  mortality classification,  and  time-to-event  tasks  in  multiple  prompt  formats.  Transcriptomic  data  from GTEx  Portal 25 comprised  98,510  training  prompts  for  age  prediction  from  gene  expression profiles and pairwise tissue age comparisons. Proteomic data from Olink Explore 3072 plasma panels  represented  the  smallest  modality  (7,807  training  examples) with  age  regression, classification,  pairwise  comparison,  and  protein  profile  generation  tasks.  Data  for  proteomic prompt collections  was  obtained  from 26,27 and  the  Immunobiology  of Aging cohort  described originally in 28 and accessed through the Allen Institute of Immunology web portal. Additional tasks in the corpus also included TCGA cancer survival comparisons (obtained through TCGA Research Network 29 ), multi-mutant lifespan prediction, and general aging biology tasks.",
          "section_heading": "Training data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            6,
            7,
            8,
            10,
            11
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/20",
            "#/texts/21",
            "#/texts/26",
            "#/texts/27",
            "#/texts/54",
            "#/texts/56",
            "#/texts/57",
            "#/texts/62",
            "#/texts/64",
            "#/texts/65",
            "#/texts/66",
            "#/texts/67",
            "#/texts/68",
            "#/texts/69",
            "#/texts/78",
            "#/texts/80",
            "#/texts/81"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003434::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003434::0001",
            "dense::full_2026-07-06__rec_003434::0003",
            "dense::full_2026-07-06__rec_003434::0008"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b9de6ef00cdb",
          "configuration_id": "config_5d9aa28db19f",
          "route_label": "Olink protein-generation prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "protein profile generation conditioned on age and sex",
          "source_object_verbatim": "partial Olink profile conditioned on age and sex",
          "source_object_normalized": "partial Olink proteomic profile conditioned on age and sex",
          "source_modality_normalized": "protein/peptide",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "partial Olink profile conditioned on age and sex",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "full-parameter fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Beyond  regression,  we  assessed  Longevity-LLM's  capacity  to  generate  biologically  plausible proteomic  profiles.  When  instructed  to  predict  the  25  most  abundant  plasma  proteins  from  a partial  Olink  profile  conditioned  on  age  and  sex,  the  SFT  model  attained  a  Jaccard  index  of 0.072, which is significantly greater than any other tested frontier model ( Figure 3C ). This result indicates  that  fine-tuning  yielded  biologically  meaningful  representations  of  the  aging  plasma proteome, enabling the model to carry out tasks beyond numeric age mapping.",
          "section_heading": "Training data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            6,
            7,
            8,
            10,
            11
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/20",
            "#/texts/21",
            "#/texts/26",
            "#/texts/27",
            "#/texts/54",
            "#/texts/56",
            "#/texts/57",
            "#/texts/62",
            "#/texts/64",
            "#/texts/65",
            "#/texts/66",
            "#/texts/67",
            "#/texts/68",
            "#/texts/69",
            "#/texts/78",
            "#/texts/80",
            "#/texts/81",
            "#/texts/83"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003434::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003434::0001",
            "dense::full_2026-07-06__rec_003434::0003",
            "dense::full_2026-07-06__rec_003434::0008"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2bf38e178ec3",
          "configuration_id": "config_657f6e97f266",
          "route_label": "TCGA survival comparison prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "cancer survival comparisons",
          "source_object_verbatim": "TCGA cancer survival data",
          "source_object_normalized": "TCGA cancer survival data",
          "source_modality_normalized": "oncology survival tabular data",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "comparison prompts",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "full-parameter fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The training corpus comprised 766,640 examples (1.09 billion tokens) spanning 38 tasks across four  primary  data  modalities,  with  an  additional  191,804  examples  reserved  for  holdout evaluation  ( Supplementary  File  1 ).  DNAm  data  constituted  the  largest  modality  (416,110 training  prompts,  875  million  tokens),  including  CpG  beta  value  profiles  for  age  regression, pairwise  feature  importance  comparisons  derived  from  published  clock  coefficients,  and annotated CpG panels with genomic and pathway-level reasoning traces. Clinical biomarker data from  NHANES 24 contributed  228,381  training  examples  across  age  prediction,  mortality classification,  and  time-to-event  tasks  in  multiple  prompt  formats.  Transcriptomic  data  from GTEx  Portal 25 comprised  98,510  training  prompts  for  age  prediction  from  gene  expression profiles and pairwise tissue age comparisons. Proteomic data from Olink Explore 3072 plasma panels  represented  the  smallest  modality  (7,807  training  examples) with  age  regression, classification,  pairwise  comparison,  and  protein  profile  generation  tasks.  Data  for  proteomic prompt collections  was  obtained  from 26,27 and  the  Immunobiology  of Aging cohort  described originally in 28 and accessed through the Allen Institute of Immunology web portal. Additional tasks in the corpus also included TCGA cancer survival comparisons (obtained through TCGA Research Network 29 ), multi-mutant lifespan prediction, and general aging biology tasks.",
          "section_heading": "Training data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            6,
            7,
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/20",
            "#/texts/21",
            "#/texts/26",
            "#/texts/27",
            "#/texts/54",
            "#/texts/56",
            "#/texts/57",
            "#/texts/78",
            "#/texts/80",
            "#/texts/81"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003434::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003434::0001",
            "dense::full_2026-07-06__rec_003434::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_56c528001309",
          "configuration_id": "config_25be667e6a30",
          "route_label": "Multi-mutant lifespan prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "multi-mutant lifespan prediction",
          "source_object_verbatim": "multi-mutant lifespan data",
          "source_object_normalized": "multi-mutant lifespan data",
          "source_modality_normalized": "genetic/phenotypic tabular data",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "prompt collection",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "full-parameter fine-tuning",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "multi-mutant lifespan prediction",
          "section_heading": "Training data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            6,
            7,
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/20",
            "#/texts/21",
            "#/texts/26",
            "#/texts/27",
            "#/texts/54",
            "#/texts/56",
            "#/texts/57",
            "#/texts/78",
            "#/texts/80",
            "#/texts/81"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003434::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003434::0001",
            "dense::full_2026-07-06__rec_003434::0003"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_2bfb9536cf22"
    },
    {
      "model_id": "model_23ffab302d6b",
      "model_name": "Med-PaLM M",
      "record_id": "full_2026-07-06__rec_001352",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_c985a0111a3c",
      "paper_title": "Towards Generalist Biomedical AI",
      "doi": "10.48550/arXiv.2307.14334",
      "paper_url": "https://doi.org/10.48550/arXiv.2307.14334",
      "route_count": 29,
      "configuration_count": 17,
      "family_counts": {
        "text_native_token_stream": 15,
        "visual_raster_carrier": 14
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 12,
        "serialized_biological_context_or_ordered_profile": 1,
        "raw_slide_or_patch_input": 12,
        "plain_language_prompt_or_question": 2,
        "patch_context_or_case_level_visual_reasoning": 2
      },
      "families": [
        "text_native_token_stream",
        "visual_raster_carrier"
      ],
      "subtypes": [
        "patch_context_or_case_level_visual_reasoning",
        "plain_language_prompt_or_question",
        "raw_slide_or_patch_input",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "chest X-ray image",
        "genomics sequencing data",
        "mammography image",
        "pathology image",
        "radiology image",
        "skin lesion image",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning"
      ],
      "fusion_topologies": [
        "concatenation",
        "encoder_decoder",
        "interleaving",
        "placeholder_replacement",
        "side_or_generative_conditioning",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query",
        "metadata_or_context",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001352_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001352_bac85e136100/figure_002.png",
        "figure_index": 2,
        "caption": "Figure 2 | Illustration of instruction task prompting with one-shot exemplar. (top) shows the task prompt for the chest X-ray report generation task. It consists of task-specific instructions, a text-only 'one-shot exemplar' (omitting the corresponding image but preserving the target answer), and the actual question. The X-ray image is embedded and interleaved with textual context including view orientation and reason for the study in addition to the question. (bottom) shows the task prompt for the dermatology classification task. We formulate the skin lesion classification task as a multiple choice question answering task with all the class labels provided as individual answer options. Similar to the chest X-ray report generation task, skin lesion image tokens are interleaved with the patient clinical history as additional context to the question. The blue <img> denotes the position in the prompt where the image tokens are embedded.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a composite screenshot with two visible example rows from a medical AI/vision-language dataset or interface.\n\nTop panel:\n- Left: frontal chest X-ray image in a rounded frame.\n- Right: text block labeled “Instructions,” describing a radiology assistant task.\n- The prompt asks for chest X-ray findings including lines/tubes/devices, pleural effusion, lung opacity, pulmonary edema, cardiac silhouette, mediastinum, hilar enlargement, fractures, and skeletal abnormalities.\n- Visible question/answer content references a lateral chest X-ray and amiodarone routine surveillance.\n- The answer describes no relevant change, normal lung volumes, mild bilateral apical scarring, normal cardiac silhouette, tortuous thoracic aorta, no pathologic findings in lung parenchyma, pleura, bones, or soft tissues, and a stable/nonspecific opacity projected over right ribs.\n\nBottom panel:\n- Left: clinical dermatology photograph of a raised dark brown skin lesion on surrounding skin.\n- Right: text block labeled “Instructions,” describing a dermatology assistant task for classifying skin lesions.\n- The visible patient-history fields include age, gender, smoking/drinking status, family cancer history, lesion region, itching, growth, bleeding, elevation, and Fitzpatrick scale.\n- One multiple-choice question lists possible diagnoses: nevus, basal cell carcinoma, squamous cell carcinoma, actinic keratosis, seborrheic keratosis, melanoma.\n- The first visible answer states “Basal Cell Carcinoma.”\n- A second dermatology example begins below with another patient history and the same diagnostic options, but its answer is not visible.\n\nNo model architecture, quantitative plots, microscopy panels, segmentation masks, or experimental transformations are visible.",
        "page_no": 7,
        "sha256": "633f30c1293c78710cdef5144a260b6c7b31772ca89018fad559614b45ed7913",
        "pixel_width": 923,
        "pixel_height": 512,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 1.0,
          "height": 0.5
        },
        "panel_label": "top chest X-ray prompt panel",
        "visible_input_object": "Chest X-ray image with radiology instruction text and <img> placeholder context",
        "visible_model_interface": "tokenized text prompt with embedded image tokens interleaved into the chest X-ray report-generation scaffold",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the full top-panel source image and its readable instruction/prompt text, including the actual multimodal insertion point and the indication/reason-for-study context. It excludes the lower dermatology panel, which is unrelated to the chest X-ray route.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "patch_context_or_case_level_visual_reasoning",
          "family_id": "visual_raster_carrier",
          "route_id": "route_24c930aa65f9",
          "example_input": "ROI + neighboring patches + case context",
          "example_carrier": "ordered visual token bank",
          "example_interface": "context aggregator → multimodal LLM",
          "actual_source": "frontal chest X-ray image",
          "actual_model_visible_form": "image tokens interleaved with text tokens"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_66ceed3a7c7c",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "indication section",
          "actual_model_visible_form": "tokenized text prompt"
        },
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_25afc922cc13",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "VQA-RAD radiology images",
          "actual_model_visible_form": "image tokens interleaved with text tokens"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_c80d1f7e2b18",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "MIMIC-III findings section",
          "actual_model_visible_form": "tokenized text prompt"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_f69f5645b8b6",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "MedQA questions",
          "actual_model_visible_form": "tokenized text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_f69f5645b8b6",
          "configuration_id": "config_5ddf5593b4c7",
          "route_label": "MedQA question answering",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Question Answering | Text | MedQA",
          "source_object_verbatim": "MedQA questions",
          "source_object_normalized": "medical exam questions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "task-specific instruction",
            "multiple-choice question prompt",
            "2-shot exemplar"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "MultiMedQA We used three of the multiple-choice medical question-answering datasets from MultiMedQA [9]: the MedQA [79], MedMCQA [80], and PubMedQA [81] datasets for training and evaluation of Med-PaLM M. These question answering tasks are language-only and do not require the interpretation of additional modalities. The training set consists of 10,178 questions from MedQA and 182,822 questions from MedMCQA. The test set comprises 1,273 questions from MedQA, 4,183 questions from MedMCQA, and 500 questions from PubMedQA. Note that PubMedQA was not included in the training data mixture and only used for evaluation.",
          "section_heading": "A.1.1 Language-only datasets",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            27,
            28
          ],
          "doc_item_refs": [
            "#/tables/6",
            "#/tables/7",
            "#/tables/8",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/485",
            "#/texts/486",
            "#/texts/586",
            "#/texts/588",
            "#/texts/589",
            "#/texts/590",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0058",
            "dense::full_2026-07-06__rec_001352::0083",
            "dense::full_2026-07-06__rec_001352::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f448f8567994",
          "configuration_id": "config_6d46568fec9e",
          "route_label": "MedMCQA question answering",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Question Answering | Text | MedMCQA",
          "source_object_verbatim": "MedMCQA questions",
          "source_object_normalized": "medical exam questions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "task-specific instruction",
            "multiple-choice question prompt",
            "2-shot exemplar"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "MultiMedQA We used three of the multiple-choice medical question-answering datasets from MultiMedQA [9]: the MedQA [79], MedMCQA [80], and PubMedQA [81] datasets for training and evaluation of Med-PaLM M. These question answering tasks are language-only and do not require the interpretation of additional modalities. The training set consists of 10,178 questions from MedQA and 182,822 questions from MedMCQA. The test set comprises 1,273 questions from MedQA, 4,183 questions from MedMCQA, and 500 questions from PubMedQA. Note that PubMedQA was not included in the training data mixture and only used for evaluation.",
          "section_heading": "A.1.1 Language-only datasets",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            27,
            28
          ],
          "doc_item_refs": [
            "#/tables/6",
            "#/tables/7",
            "#/tables/8",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/485",
            "#/texts/486",
            "#/texts/586",
            "#/texts/588",
            "#/texts/589",
            "#/texts/590",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0058",
            "dense::full_2026-07-06__rec_001352::0083",
            "dense::full_2026-07-06__rec_001352::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a6559f47554c",
          "configuration_id": "config_e586a9eed414",
          "route_label": "PubMedQA question answering",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Question Answering | Text | PubMedQA",
          "source_object_verbatim": "PubMedQA questions",
          "source_object_normalized": "biomedical research questions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "task-specific instruction",
            "multiple-choice question prompt"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "PubMedQA was not included in the training data mixture and only used for evaluation",
          "section_heading": "A.1.1 Language-only datasets",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            8,
            23,
            27,
            28
          ],
          "doc_item_refs": [
            "#/tables/6",
            "#/tables/7",
            "#/tables/8",
            "#/texts/123",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/485",
            "#/texts/486",
            "#/texts/586",
            "#/texts/588",
            "#/texts/589",
            "#/texts/590",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0058",
            "dense::full_2026-07-06__rec_001352::0083",
            "dense::full_2026-07-06__rec_001352::0012",
            "dense::full_2026-07-06__rec_001352::0015"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c80d1f7e2b18",
          "configuration_id": "config_1c6cc9c14ad4",
          "route_label": "MIMIC-III radiology report summarization",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Report Summarization | Radiology | MIMIC-III",
          "source_object_verbatim": "MIMIC-III findings section",
          "source_object_normalized": "radiology findings text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "extract findings and impression sections",
            "predict the impression section given the findings section as input"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "predicting the impression section given the findings section as input",
          "fusion_topology": "encoder_decoder",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "predicting the impression section given the findings section as input",
          "section_heading": "A.1.1 Language-only datasets",
          "supporting_figure_or_table": "Table A.4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            28,
            29
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/485",
            "#/texts/486",
            "#/texts/591",
            "#/texts/593",
            "#/texts/594"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0060",
            "dense::full_2026-07-06__rec_001352::0066",
            "dense::full_2026-07-06__rec_001352::0084"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_25afc922cc13",
          "configuration_id": "config_85568fbdea80",
          "route_label": "VQA-RAD radiology image",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Visual Question Answering | Radiology | VQA-RAD",
          "source_object_verbatim": "VQA-RAD radiology images",
          "source_object_normalized": "radiology image",
          "source_modality_normalized": "radiology image",
          "transformation_chain_verbatim": [
            "resize to 224 × 224 × 3",
            "image tokens interleaved with text tokens",
            "text-only 1-shot exemplar with <img> placeholder"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "multimodal context input",
          "fusion_topology": "interleaving",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "VQA-RAD is a radiology visual question answering (VQA) dataset which consists of 315 radiology images and 3,515 question-answer pairs",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            25,
            29,
            30,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/11",
            "#/tables/14",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/531",
            "#/texts/532",
            "#/texts/533",
            "#/texts/534",
            "#/texts/535",
            "#/texts/536",
            "#/texts/537",
            "#/texts/538",
            "#/texts/597",
            "#/texts/599",
            "#/texts/601",
            "#/texts/622",
            "#/texts/623",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0061",
            "dense::full_2026-07-06__rec_001352::0072",
            "dense::full_2026-07-06__rec_001352::0086",
            "dense::full_2026-07-06__rec_001352::0087",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0011"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d7dcfe8ab908",
          "configuration_id": "config_85568fbdea80",
          "route_label": "VQA-RAD question text",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Visual Question Answering | Radiology | VQA-RAD",
          "source_object_verbatim": "VQA-RAD question text",
          "source_object_normalized": "question text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "task-specific instruction",
            "question prompt",
            "text-only 1-shot exemplar"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "VQA-RAD is a radiology visual question answering (VQA) dataset which consists of 315 radiology images and 3,515 question-answer pairs",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            25,
            29,
            30,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/11",
            "#/tables/14",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/531",
            "#/texts/532",
            "#/texts/533",
            "#/texts/534",
            "#/texts/535",
            "#/texts/536",
            "#/texts/537",
            "#/texts/538",
            "#/texts/597",
            "#/texts/599",
            "#/texts/601",
            "#/texts/622",
            "#/texts/623",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0061",
            "dense::full_2026-07-06__rec_001352::0086",
            "dense::full_2026-07-06__rec_001352::0087",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fce69a574de1",
          "configuration_id": "config_9af8e886f341",
          "route_label": "Slake-VQA radiology image",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Visual Question Answering | Radiology | Slake-VQA",
          "source_object_verbatim": "Slake-VQA radiology images",
          "source_object_normalized": "radiology image",
          "source_modality_normalized": "radiology image",
          "transformation_chain_verbatim": [
            "resize to 224 × 224 × 3",
            "image tokens interleaved with text tokens",
            "text-only 1-shot exemplar with <img> placeholder"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "multimodal context input",
          "fusion_topology": "interleaving",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Slake-VQA is a semantically annotated and knowledge-enhanced bilingual (English and Chinese) VQA dataset on radiology images",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            26,
            29,
            30,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/11",
            "#/tables/14",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/542",
            "#/texts/543",
            "#/texts/544",
            "#/texts/597",
            "#/texts/599",
            "#/texts/601",
            "#/texts/622",
            "#/texts/623",
            "#/texts/625",
            "#/texts/626",
            "#/texts/627",
            "#/texts/628",
            "#/texts/630",
            "#/texts/631",
            "#/texts/632",
            "#/texts/633",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0061",
            "dense::full_2026-07-06__rec_001352::0074",
            "dense::full_2026-07-06__rec_001352::0086",
            "dense::full_2026-07-06__rec_001352::0087",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0011"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fbd79b00ab68",
          "configuration_id": "config_9af8e886f341",
          "route_label": "Slake-VQA question text",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Visual Question Answering | Radiology | Slake-VQA",
          "source_object_verbatim": "Slake-VQA question text",
          "source_object_normalized": "question text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "task-specific instruction",
            "question prompt",
            "text-only 1-shot exemplar"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Slake-VQA is a semantically annotated and knowledge-enhanced bilingual (English and Chinese) VQA dataset on radiology images",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            26,
            29,
            30,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/11",
            "#/tables/14",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/542",
            "#/texts/543",
            "#/texts/544",
            "#/texts/597",
            "#/texts/599",
            "#/texts/601",
            "#/texts/622",
            "#/texts/623",
            "#/texts/625",
            "#/texts/626",
            "#/texts/627",
            "#/texts/628",
            "#/texts/630",
            "#/texts/631",
            "#/texts/632",
            "#/texts/633",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0061",
            "dense::full_2026-07-06__rec_001352::0074",
            "dense::full_2026-07-06__rec_001352::0086",
            "dense::full_2026-07-06__rec_001352::0087",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_bca4ff4bec3d",
          "configuration_id": "config_11313b3b47cd",
          "route_label": "Path-VQA pathology image",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Visual Question Answering | Pathology | Path-VQA",
          "source_object_verbatim": "Path-VQA pathology images",
          "source_object_normalized": "pathology image",
          "source_modality_normalized": "pathology image",
          "transformation_chain_verbatim": [
            "resize to 224 × 224 × 3",
            "image tokens interleaved with text tokens",
            "text-only 1-shot exemplar with <img> placeholder"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "multimodal context input",
          "fusion_topology": "interleaving",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Path-VQA is a pathology VQA dataset, containing a total of 4,998 pathology images with 32,799 questionanswer pairs",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            26,
            29,
            30,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/11",
            "#/tables/14",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/542",
            "#/texts/543",
            "#/texts/544",
            "#/texts/597",
            "#/texts/599",
            "#/texts/601",
            "#/texts/622",
            "#/texts/623",
            "#/texts/625",
            "#/texts/626",
            "#/texts/627",
            "#/texts/628",
            "#/texts/630",
            "#/texts/631",
            "#/texts/632",
            "#/texts/633",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0061",
            "dense::full_2026-07-06__rec_001352::0073",
            "dense::full_2026-07-06__rec_001352::0086",
            "dense::full_2026-07-06__rec_001352::0087",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0011"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a296352c6985",
          "configuration_id": "config_11313b3b47cd",
          "route_label": "Path-VQA question text",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Visual Question Answering | Pathology | Path-VQA",
          "source_object_verbatim": "Path-VQA question text",
          "source_object_normalized": "question text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "task-specific instruction",
            "question prompt",
            "text-only 1-shot exemplar"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Path-VQA is a pathology VQA dataset, containing a total of 4,998 pathology images with 32,799 questionanswer pairs",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            26,
            29,
            30,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/11",
            "#/tables/14",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/542",
            "#/texts/543",
            "#/texts/544",
            "#/texts/597",
            "#/texts/599",
            "#/texts/601",
            "#/texts/622",
            "#/texts/623",
            "#/texts/625",
            "#/texts/626",
            "#/texts/627",
            "#/texts/628",
            "#/texts/630",
            "#/texts/631",
            "#/texts/632",
            "#/texts/633",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_010"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0061",
            "dense::full_2026-07-06__rec_001352::0073",
            "dense::full_2026-07-06__rec_001352::0086",
            "dense::full_2026-07-06__rec_001352::0087",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_316a17c531e7",
          "configuration_id": "config_b680eac8f201",
          "route_label": "MIMIC-CXR chest X-ray image",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Report Generation | Chest X-ray | MIMIC-CXR",
          "source_object_verbatim": "MIMIC-CXR chest X-ray image",
          "source_object_normalized": "chest X-ray image",
          "source_modality_normalized": "chest X-ray image",
          "transformation_chain_verbatim": [
            "resize images to 224 × 224 × 3",
            "image tokens interleaved with text tokens",
            "text-only 1-shot exemplar with <img> placeholder"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "multimodal context input",
          "fusion_topology": "interleaving",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "we combined the chest X-ray image with the contextual information from the indication section (reason for the study) to predict the findings section of the target report",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.7",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            8,
            9,
            23,
            26,
            30,
            31,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/12",
            "#/tables/14",
            "#/texts/131",
            "#/texts/132",
            "#/texts/133",
            "#/texts/135",
            "#/texts/136",
            "#/texts/137",
            "#/texts/138",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/545",
            "#/texts/546",
            "#/texts/547",
            "#/texts/602",
            "#/texts/604",
            "#/texts/605",
            "#/texts/622",
            "#/texts/623",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_011"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0062",
            "dense::full_2026-07-06__rec_001352::0075",
            "dense::full_2026-07-06__rec_001352::0092",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_66ceed3a7c7c",
          "configuration_id": "config_b680eac8f201",
          "route_label": "MIMIC-CXR indication text",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Report Generation | Chest X-ray | MIMIC-CXR",
          "source_object_verbatim": "indication section",
          "source_object_normalized": "indication section",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "reason for the study context",
            "predict findings section"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "MIMIC-CXR is a large dataset of chest radiographs with free-text radiology reports [97]. A total of 377,110 images are available in the dataset from 227,835 image studies collected for 65,379 patients. Each patient may have multiple studies and each study may contain one or more images associated with the same free-text report. Images in MIMIC-CXR are collected from multiple view positions: e.g., anterior-posterior (AP), posterioranterior, and lateral (LA). Protected health information (PHI) in radiology reports and images is removed, which results in missing information in some sentences of the reports. Since this dataset contains sequential imaging studies of an individual patient, a large number of reports refer to information in prior studies of the same patient. Each report is annotated with structured labels of 14 common radiological observations using CheXpert labeler [98]. We performed two tasks using this dataset: chest X-ray report generation and binary classification of clinically-relevant pathology observations. We preprocessed the radiology reports by extracting the indication, findings, and impression sections, removing redundant white-spaces in the reports, following previous work [99]. We used the official train/validation/test splits. We discarded images without reports and reports where the findings section can not be extracted across train and test. We also filtered out the reports where the length of findings section exceeds 800 characters. However, unlike most previous work using focusing only on the frontal view, we treated images of different orientation that are associated with the same report as independent samples (retaining the patient-level train/test splits to avoid contamination of the test data). The goal is to improve the image understanding capability of the model to process images of different view positions. In a separate evaluation, we also studied a subset of samples where reports are accompanied by both a front and lateral view (two-view report generation).",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.7",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            8,
            9,
            23,
            26,
            30,
            31,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/12",
            "#/tables/14",
            "#/texts/131",
            "#/texts/132",
            "#/texts/133",
            "#/texts/135",
            "#/texts/136",
            "#/texts/137",
            "#/texts/138",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/545",
            "#/texts/546",
            "#/texts/547",
            "#/texts/602",
            "#/texts/604",
            "#/texts/605",
            "#/texts/622",
            "#/texts/623",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_012"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0062",
            "dense::full_2026-07-06__rec_001352::0075",
            "dense::full_2026-07-06__rec_001352::0092",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ab73c9e4c443",
          "configuration_id": "config_8a01c4273ca1",
          "route_label": "PAD-UFES-20 skin lesion image",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Dermatology | PAD-UFES-20",
          "source_object_verbatim": "PAD-UFES-20 clinical skin lesion images",
          "source_object_normalized": "skin lesion images",
          "source_modality_normalized": "skin lesion image",
          "transformation_chain_verbatim": [
            "resize to 224 × 224 × 3",
            "class-balanced augmentation",
            "text-only 1-shot exemplar"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "multimodal context input",
          "fusion_topology": "interleaving",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "PAD-UFES-20 consists of 2,298 clinical images of skin lesions",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            28,
            29,
            30,
            34
          ],
          "doc_item_refs": [
            "#/tables/10",
            "#/tables/14",
            "#/tables/9",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/488",
            "#/texts/591",
            "#/texts/593",
            "#/texts/594",
            "#/texts/595",
            "#/texts/596",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_013"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0067",
            "dense::full_2026-07-06__rec_001352::0085",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0059",
            "dense::full_2026-07-06__rec_001352::0088"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_dd6b4b0a8833",
          "configuration_id": "config_8a01c4273ca1",
          "route_label": "PAD-UFES-20 patient clinical features",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Dermatology | PAD-UFES-20",
          "source_object_verbatim": "patient clinical features",
          "source_object_normalized": "patient clinical features",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "patient demographics and lesion attributes",
            "multiple-choice question prompt"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "PAD-UFES-20 consists of 2,298 clinical images of skin lesions collected from different smartphone devices with varying resolutions, sizes, and lighting conditions [85]. The data was collected through the Dermatological and Surgical Assistance Program at the Federal University of Espírito Santo (UFES-Brazil), a nonprofit program that provides free skin lesion treatment. The dataset contains six different types of skin lesions including: Basal Cell Carcinoma (BCC), Malignant Melanoma (MEL), Squamous Cell Carcinoma (SCC), Actinic Keratosis (ACK), Melanocytic Nevus (NEV), and Seborrheic Keratosis (SEK). Each image is associated with up to 21 patient clinical features such as patient demographics, family cancer history lesion location, lesion size. We set up a 6-class classification task in a generative framework through a language decoder using skin lesion images and the associated clinical textual features as the multimodal input. Specifically, we selected 14 clinical attributes in the metadata for each lesion including: age , gender , smoke , drink , skin cancer history , cancer history , region , fitspatrick , horizontal and vertical diameters , itch , grew , bleed , and elevation . The class ratio is approximately 16:1:4:14:5:4 over three skin cancers (BCC, MEL, and SCC) and three skin disease (ACK, NEV, and SEK). Since there are no published official train/test splits, we randomly split the dataset into a training set (80%) and a test test (20%) using a stratified sampling to the preserve original class ratio. We applied a series of image augmentation operations using RandAugment [86] to the training set including: autoContrast , equalize , invert , rotate , posterize , solarize , color , and contrast .",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            29,
            34
          ],
          "doc_item_refs": [
            "#/tables/14",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/488",
            "#/texts/595",
            "#/texts/596"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_014"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0067",
            "dense::full_2026-07-06__rec_001352::0088",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0059"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d0d20292a638",
          "configuration_id": "config_38a3f3baa7b4",
          "route_label": "VinDr-Mammo mammography image",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Mammography | VinDr-Mammo",
          "source_object_verbatim": "VinDr-Mammo mammography studies",
          "source_object_normalized": "mammography studies",
          "source_modality_normalized": "mammography image",
          "transformation_chain_verbatim": [
            "image augmentation with contrast, equalize, rotate, shearX, shearY, translateX, and translateY",
            "text-only 1-shot exemplar"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "multimodal context input",
          "fusion_topology": "interleaving",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "VinDr-Mammo is a full-field digital mammography dataset which consists of 5000 breast X-ray imaging studies and a total of 20,000 gray-scale images",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            24,
            25,
            28,
            29,
            30,
            34
          ],
          "doc_item_refs": [
            "#/tables/10",
            "#/tables/14",
            "#/tables/9",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/528",
            "#/texts/530",
            "#/texts/591",
            "#/texts/593",
            "#/texts/594",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_015"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0069",
            "dense::full_2026-07-06__rec_001352::0085",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0059"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_96155896761d",
          "configuration_id": "config_38a3f3baa7b4",
          "route_label": "VinDr-Mammo contextual features",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Mammography | VinDr-Mammo",
          "source_object_verbatim": "laterality and view position",
          "source_object_normalized": "laterality and view position",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "additional contextual features",
            "breast-level BI-RADS classification prompt"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "the laterality and view position of the image was provided as additional contextual features",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            24,
            25,
            29,
            34
          ],
          "doc_item_refs": [
            "#/tables/14",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/528",
            "#/texts/530",
            "#/texts/595",
            "#/texts/596"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_016"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0069",
            "dense::full_2026-07-06__rec_001352::0088",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0059"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4c177264121e",
          "configuration_id": "config_7bacbcbb3c60",
          "route_label": "CBIS-DDSM mass image patch",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Mammography | CBIS-DDSM mass",
          "source_object_verbatim": "CBIS-DDSM mass image patch",
          "source_object_normalized": "mammography patch",
          "source_modality_normalized": "mammography image",
          "transformation_chain_verbatim": [
            "crop by the bounding box of the region-of-interest",
            "0-shot setup"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "multimodal context input",
          "fusion_topology": "interleaving",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Abnormality image patch is cropped by the bounding box of the region-of-interest (ROI) from the full mammogram",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            24,
            25,
            28,
            29,
            30,
            34
          ],
          "doc_item_refs": [
            "#/tables/10",
            "#/tables/14",
            "#/tables/9",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/528",
            "#/texts/530",
            "#/texts/591",
            "#/texts/593",
            "#/texts/594",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_017"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0070",
            "dense::full_2026-07-06__rec_001352::0085",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0059"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4b59d8f2b620",
          "configuration_id": "config_7bacbcbb3c60",
          "route_label": "CBIS-DDSM mass view position",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Mammography | CBIS-DDSM mass",
          "source_object_verbatim": "view position information",
          "source_object_normalized": "view position information",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "CC or MLO view position context",
            "0-shot setup"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "used as the model input along with its view position (CC or MLO) information",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            24,
            25,
            29,
            34
          ],
          "doc_item_refs": [
            "#/tables/14",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/528",
            "#/texts/530",
            "#/texts/595",
            "#/texts/596"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_018"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0070",
            "dense::full_2026-07-06__rec_001352::0088",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0059"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5b512157360b",
          "configuration_id": "config_359cff626769",
          "route_label": "CBIS-DDSM calcification image patch",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Mammography | CBIS-DDSM calcification",
          "source_object_verbatim": "CBIS-DDSM calcification image patch",
          "source_object_normalized": "mammography patch",
          "source_modality_normalized": "mammography image",
          "transformation_chain_verbatim": [
            "crop by the bounding box of the region-of-interest",
            "0-shot setup"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "multimodal context input",
          "fusion_topology": "interleaving",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Abnormality image patch is cropped by the bounding box of the region-of-interest (ROI) from the full mammogram",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            24,
            25,
            28,
            29,
            30,
            34
          ],
          "doc_item_refs": [
            "#/tables/10",
            "#/tables/14",
            "#/tables/9",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/528",
            "#/texts/530",
            "#/texts/591",
            "#/texts/593",
            "#/texts/594",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_019"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0070",
            "dense::full_2026-07-06__rec_001352::0085",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0059"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_12d7d966e290",
          "configuration_id": "config_359cff626769",
          "route_label": "CBIS-DDSM calcification view position",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Mammography | CBIS-DDSM calcification",
          "source_object_verbatim": "CBIS-DDSM calcification view position information",
          "source_object_normalized": "view position information",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "CC or MLO view position context",
            "0-shot setup"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "used as the model input along with its view position (CC or MLO) information",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            24,
            25,
            29,
            34
          ],
          "doc_item_refs": [
            "#/tables/14",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/528",
            "#/texts/530",
            "#/texts/595",
            "#/texts/596"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_020"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0070",
            "dense::full_2026-07-06__rec_001352::0088",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0059"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e41c2b760139",
          "configuration_id": "config_bfbf81110497",
          "route_label": "MIMIC-CXR pathology classification image",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Chest X-ray | MIMIC-CXR",
          "source_object_verbatim": "MIMIC-CXR chest radiograph",
          "source_object_normalized": "chest radiograph",
          "source_modality_normalized": "chest X-ray image",
          "transformation_chain_verbatim": [
            "binary classification of clinically-relevant pathology observations",
            "text-only 1-shot exemplar"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "multimodal context input",
          "fusion_topology": "interleaving",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We performed two tasks using this dataset: chest X-ray report generation and binary classification of clinically-relevant pathology observations",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            23,
            26,
            28,
            29,
            30,
            34
          ],
          "doc_item_refs": [
            "#/tables/10",
            "#/tables/14",
            "#/tables/9",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/545",
            "#/texts/546",
            "#/texts/547",
            "#/texts/591",
            "#/texts/593",
            "#/texts/594",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_021"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0059",
            "dense::full_2026-07-06__rec_001352::0085",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0011"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f27cdb31c4ca",
          "configuration_id": "config_aaf68bad874e",
          "route_label": "PrecisionFDA variant calling image-like example",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Genomics | PrecisionFDA Truth Challenge V2",
          "source_object_verbatim": "sequencing data from PrecisionFDA Truth Challenge V2",
          "source_object_normalized": "genomic sequencing data",
          "source_modality_normalized": "genomics sequencing data",
          "transformation_chain_verbatim": [
            "DeepVariant v1.3.0 example generation",
            "stack channels 1, 2, 3 with channels 4, 5, 6",
            "pad to 224 × 224 × 3"
          ],
          "model_visible_form_verbatim": "RGB image of shape (224, 224, 3)",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "image-like example fed through the vision encoder",
          "fusion_topology": "interleaving",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "PrecisionFDA Truth Challenge V2 was developed for benchmarking the state-of-the-art of variant calling in challenging genomics regions [89]. Genomic variant calling is a task aiming at identifying genetic variants from sequencing data [90], which can identify disease-causing mutations [91]. For variant calling, sequencing data is mapped to the coordinates of a reference genome [92]. The mappings can be represented as an image-like format that computational methods such as DeepVariant [71] use to call variants, or in a human-friendly image format which experts use to inspect and quality control variants of interest [93]. For this task, we used an extensively characterized groundtruth set from the National Institute of Standards and Technology (NIST) [94] for the HG002 sample. We generated examples from sequencing from the PrecisionFDA Truth Challenge V2. For training, we use 4% of the examples from the whole genome (except for chromosome 20, 21, and 22). For evaluation, we used chromosome20, bases 3000001-9444417. This generated 197,038 candidate variants for training and 13,030 candidate variants for evaluation. For each example, the model predicts three possible genotypes, corresponding to how many copies (0, 1, or 2) of the given alternate allele are present. The training set consists of 45,011, 93,246, and 58,781 samples for classes 0, 1, 2, respectively. The evaluation set contains 3,016, 6,169, and 3,845 for classes 0, 1, 2, respectively.",
          "section_heading": "A.1.2 Multimodal datasets",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            6,
            21,
            23,
            25,
            27,
            28,
            29,
            30,
            34
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/tables/10",
            "#/tables/14",
            "#/tables/9",
            "#/texts/436",
            "#/texts/437",
            "#/texts/470",
            "#/texts/471",
            "#/texts/472",
            "#/texts/473",
            "#/texts/474",
            "#/texts/475",
            "#/texts/476",
            "#/texts/477",
            "#/texts/531",
            "#/texts/532",
            "#/texts/533",
            "#/texts/534",
            "#/texts/535",
            "#/texts/536",
            "#/texts/537",
            "#/texts/538",
            "#/texts/591",
            "#/texts/593",
            "#/texts/594",
            "#/texts/77",
            "#/texts/78",
            "#/texts/79",
            "#/texts/80",
            "#/texts/82",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_022"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0071",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0059",
            "dense::full_2026-07-06__rec_001352::0085"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_347b0e24a453",
          "configuration_id": "config_b1111b097343",
          "route_label": "Montgomery County TB chest X-ray image",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "zero-shot classification of tuberculosis (TB) from chest X-ray images",
          "source_object_verbatim": "Montgomery County chest X-ray images",
          "source_object_normalized": "Montgomery County chest X-ray images",
          "source_modality_normalized": "chest X-ray image",
          "transformation_chain_verbatim": [
            "text-only one-shot exemplar",
            "yes/no question answering task"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "prompted with a text-only one-shot exemplar",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "evaluating its ability to detect tuberculosis (TB) abnormality from chest X-ray images",
          "section_heading": "6.2.1 Evidence of generalization to novel medical concepts",
          "supporting_figure_or_table": "Table 4",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            6,
            8,
            12,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/14",
            "#/tables/3",
            "#/texts/123",
            "#/texts/125",
            "#/texts/126",
            "#/texts/127",
            "#/texts/128",
            "#/texts/129",
            "#/texts/160",
            "#/texts/161",
            "#/texts/622",
            "#/texts/623",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_023"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0016",
            "dense::full_2026-07-06__rec_001352::0027",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0015"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_23711b8a6d22",
          "configuration_id": "config_b1111b097343",
          "route_label": "Montgomery County TB text-only exemplar",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "zero-shot classification of tuberculosis (TB) from chest X-ray images",
          "source_object_verbatim": "text-only one-shot exemplar",
          "source_object_normalized": "text-only exemplar",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "dummy text placeholder instead",
            "zero-shot task prompt"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "dummy text placeholder <img>",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "simple task prompt consisting of a single text-only exemplar (without task-specific image and hence zero-shot)",
          "section_heading": "6.2.1 Evidence of generalization to novel medical concepts",
          "supporting_figure_or_table": "Table 4",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            8,
            12,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/14",
            "#/tables/3",
            "#/texts/123",
            "#/texts/125",
            "#/texts/126",
            "#/texts/127",
            "#/texts/128",
            "#/texts/129",
            "#/texts/160",
            "#/texts/161",
            "#/texts/622",
            "#/texts/623"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_024"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0016",
            "dense::full_2026-07-06__rec_001352::0027",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0015"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_dedea019c8c6",
          "configuration_id": "config_ceb43afe2cc2",
          "route_label": "Zero-shot CoT TB reasoning chest X-ray image",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "zero-shot chain-of-thought multimodal medical reasoning",
          "source_object_verbatim": "MC chest X-ray image",
          "source_object_normalized": "chest X-ray image",
          "source_modality_normalized": "chest X-ray image",
          "transformation_chain_verbatim": [
            "text-only exemplar",
            "generate the class prediction and an accompanying report describing the image findings"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "prompted the model with a text-only exemplar",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "we prompted the model with a text-only exemplar to generate a report describing the findings in a given image",
          "section_heading": "6.2.2 Evidence of emergent zero-shot multimodal medical reasoning",
          "supporting_figure_or_table": "Figure 3",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            6,
            8,
            12,
            13,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/2",
            "#/pictures/20",
            "#/tables/14",
            "#/texts/123",
            "#/texts/125",
            "#/texts/126",
            "#/texts/127",
            "#/texts/128",
            "#/texts/129",
            "#/texts/163",
            "#/texts/169",
            "#/texts/622",
            "#/texts/623",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_025"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0030",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0015",
            "dense::full_2026-07-06__rec_001352::0016"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_24c930aa65f9",
          "configuration_id": "config_46c4b9bbddaa",
          "route_label": "Two-view chest X-ray frontal image",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "two-view chest X-ray report generation",
          "source_object_verbatim": "frontal chest X-ray image",
          "source_object_normalized": "frontal chest X-ray image",
          "source_modality_normalized": "chest X-ray image",
          "transformation_chain_verbatim": [
            "paired with a lateral chest X-ray view",
            "zero-shot evaluation"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "patch_context_or_case_level_visual_reasoning",
          "insertion_or_fusion_verbatim": "paired with a lateral chest X-ray view",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "a novel task setup with multi-view visual inputs",
          "section_heading": "6.2.3 Evidence of generalization to novel tasks",
          "supporting_figure_or_table": "Table 5",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper describes the two-view input jointly; this per-view route is split from the combined multi-view setup.",
          "pages": [
            6,
            8,
            12,
            14,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/14",
            "#/tables/4",
            "#/texts/123",
            "#/texts/125",
            "#/texts/126",
            "#/texts/127",
            "#/texts/128",
            "#/texts/129",
            "#/texts/165",
            "#/texts/204",
            "#/texts/622",
            "#/texts/623",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_026"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0028",
            "dense::full_2026-07-06__rec_001352::0033",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0015",
            "dense::full_2026-07-06__rec_001352::0016"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e5b3f0ddc7a8",
          "configuration_id": "config_46c4b9bbddaa",
          "route_label": "Two-view chest X-ray lateral image",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "two-view chest X-ray report generation",
          "source_object_verbatim": "lateral chest X-ray image",
          "source_object_normalized": "lateral chest X-ray image",
          "source_modality_normalized": "chest X-ray image",
          "transformation_chain_verbatim": [
            "paired with a frontal chest X-ray view",
            "zero-shot evaluation"
          ],
          "model_visible_form_verbatim": "image tokens interleaved with text tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "patch_context_or_case_level_visual_reasoning",
          "insertion_or_fusion_verbatim": "paired with a frontal chest X-ray view",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "a novel task setup with multi-view visual inputs",
          "section_heading": "6.2.3 Evidence of generalization to novel tasks",
          "supporting_figure_or_table": "Table 5",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper describes the two-view input jointly; this per-view route is split from the combined multi-view setup.",
          "pages": [
            6,
            8,
            12,
            14,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/14",
            "#/tables/4",
            "#/texts/123",
            "#/texts/125",
            "#/texts/126",
            "#/texts/127",
            "#/texts/128",
            "#/texts/129",
            "#/texts/165",
            "#/texts/204",
            "#/texts/622",
            "#/texts/623",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_027"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0028",
            "dense::full_2026-07-06__rec_001352::0033",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0011",
            "dense::full_2026-07-06__rec_001352::0015",
            "dense::full_2026-07-06__rec_001352::0016"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_74341e8e7404",
          "configuration_id": "config_bfbf81110497",
          "route_label": "MIMIC-CXR abnormality question prompt",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Chest X-ray | MIMIC-CXR",
          "source_object_verbatim": "MIMIC-CXR abnormality question prompt",
          "source_object_normalized": "abnormality question prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "task-specific instruction",
            "multiple-choice question prompt",
            "text-only 1-shot exemplar"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Given the AP view X-ray image <img>. Q: Is cardiomegaly indicated by the image? (A) No (B) Yes",
          "section_heading": "Table A.9 | Examples of the classification tasks in MultiMedBench.",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/14",
            "#/texts/622",
            "#/texts/623",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_029"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b1a0860c865c",
          "configuration_id": "config_aaf68bad874e",
          "route_label": "PrecisionFDA variant-count question prompt",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Medical Image Classification | Genomics | PrecisionFDA Truth Challenge V2",
          "source_object_verbatim": "PrecisionFDA variant-count question prompt",
          "source_object_normalized": "variant-count question prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "task-specific instruction",
            "multiple-choice question prompt",
            "0-shot setup"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction tuning in a unified generative framework",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Given <img> . Q: How many copies of this putative variant are shown in the middle of the image? (A) 0 (B) 1 (C) 2 A:",
          "section_heading": "Table A.9 | Examples of the classification tasks in MultiMedBench.",
          "supporting_figure_or_table": "Table A.9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            6,
            21,
            25,
            27,
            28,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/17",
            "#/pictures/18",
            "#/pictures/19",
            "#/pictures/20",
            "#/tables/0",
            "#/tables/14",
            "#/texts/436",
            "#/texts/437",
            "#/texts/622",
            "#/texts/623",
            "#/texts/77",
            "#/texts/78",
            "#/texts/79",
            "#/texts/80",
            "#/texts/82",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96",
            "#/texts/97",
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001352::route_039"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001352::0071",
            "dense::full_2026-07-06__rec_001352::0099",
            "dense::full_2026-07-06__rec_001352::0100",
            "dense::full_2026-07-06__rec_001352::0012"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_cbdfdd5c56c0"
    },
    {
      "model_id": "model_16a912645955",
      "model_name": "Mistral 7B",
      "record_id": "full_2026-07-06__rec_002105",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_266ca9b669f4",
      "paper_title": "Multimodal learning of transcriptomes and text enables interactive single-cell RNA-seq data exploration with natural-language chats",
      "doi": "10.1101/2024.10.15.618501",
      "paper_url": "https://doi.org/10.1101/2024.10.15.618501",
      "route_count": 4,
      "configuration_count": 4,
      "family_counts": {
        "dense_continuous_carrier": 4
      },
      "subtype_counts": {
        "connector_mediated_embedding": 4
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "connector_mediated_embedding"
      ],
      "primary_subtype": "connector_mediated_embedding",
      "modalities": [
        "RNA",
        "mixed"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "inference"
      ],
      "fusion_topologies": [
        "prefix"
      ],
      "text_roles": [
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_002105_figure_007.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_002105_966eaaaee764/figure_007.png",
        "figure_index": 7,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image shows a workflow and result summary for evaluating transcriptome embeddings with an LLM-based prediction/perplexity setup.\n\nVisible elements:\n- Left panel: stacked “Question-answer pairs (test set, n=200)” documents.\n- Inputs include a purple label “transcriptome embedding,” a question bubble `Q: “Describe this sample”`, and an answer bubble `A: “These are bronchial epithelial …”`.\n- These inputs feed into a vertical green block labeled “LLM adapter,” connected to “Prediction” and “Perplexity.”\n- A “Ground truth” answer is used for perplexity evaluation.\n\nMiddle panel:\n- Scatter/dot plot comparing perplexity values.\n- X-axis labeled “Perplexity, lower is better,” spanning approximately 1.51 to 14.25.\n- Y-axis appears to show quantile/bin levels labeled 1 through 5.\n- Legend distinguishes gray “Mismatched embeddings (n=30)” from purple “Matched embeddings.”\n- A small table at right lists “Quantile for matched embedding (inverted)” with values such as 0.87, 0.70, and 1.00.\n\nRight panel:\n- Horizontal quantile summary labeled “Perplexity quantiles for all 200 test cases.”\n- Axis labeled “Perplexity quantiles (inverted)” from 0.0 to 1.0.\n- A marker appears near the high end of the scale, suggesting matched embeddings tend to achieve favorable inverted perplexity quantiles.\n\nBiological source objects:\n- Transcriptome embeddings from biological samples.\n- The visible example answer refers to “bronchial epithelial” cells/samples.\n\nModel/interface details:\n- A transcriptome embedding and natural-language question are passed through an LLM adapter.\n- The system generates a textual prediction and computes perplexity against the ground-truth answer.\n\nFinding shown:\n- Matched transcriptome embeddings are evaluated against mismatched embeddings using perplexity; matched embeddings appear associated with better, high inverted perplexity quantiles.",
        "page_no": 21,
        "sha256": "29faae94a37913029648190487e5b040b334d7c87c8b0f47db43953a045a03c6",
        "pixel_width": 969,
        "pixel_height": 140,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.38,
          "height": 1.0
        },
        "panel_label": "left workflow panel",
        "visible_input_object": "transcriptome embedding plus question-answer pair (Q: \"Describe this sample\", A: \"These are bronchial epithelial ...\")",
        "visible_model_interface": "LLM adapter feeding Mistral-compatible token embeddings",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "Crops to the left workflow only, preserving the source embedding, question/answer text, arrows, and the LLM adapter insertion interface. This keeps one grounded input route readable while excluding the output-only perplexity plot and quantile panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "connector_mediated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_5e86ef1efcaa",
          "example_input": "image / omics encoder states",
          "example_carrier": "Q-Former or adapter query vectors",
          "example_interface": "connector → LLM cross-modal interface",
          "actual_source": "transcriptome profiles",
          "actual_model_visible_form": "transcriptome-derived token embeddings alongside their corresponding questions"
        }
      ],
      "routes": [
        {
          "route_id": "route_5e86ef1efcaa",
          "configuration_id": "config_3aa1e59baeda",
          "route_label": "Transcriptome embeddings plus questions to Mistral 7B for CellWhisperer fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "question-answer-style conversations",
          "source_object_verbatim": "transcriptome profiles",
          "source_object_normalized": "transcriptome profiles",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "CellWhisperer-derived transcriptome embeddings",
            "adapter layer that converts the transcriptome embeddings into Mistral-compatible token-level embeddings"
          ],
          "model_visible_form_verbatim": "transcriptome-derived token embeddings alongside their corresponding questions",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "a two-layer adapter module that transforms the 2048-dimensional CellWhisperer transcriptome embedding into eight 4096-dimensional embeddings",
          "fusion_topology": "prefix",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "To enable natural-language chats with CellWhisperer's multimodal embedding model, we fine-tuned the Mistral 7B LLM based on data-centric conversations about individual transcriptomes and cell states. We generated 106,610 such conversations for transcriptomes from our training dataset of 1,082,413 GEO and CELLxGENE Census data points. To mitigate historical bias in these community repositories, we took a weighted subsample, inversely proportional to the local point density calculated with densMAP (Narayan et al. 2021), such that transcriptome-text pairs in lowly covered regions were preferentially picked. For each transcriptome, we generated a chat-like conversation using one of two LLMs (GPT-4 or Mixtral) on the basis of the transcriptome's top 50 most highly expressed genes, the top 50 GSVA-derived gene sets, and the transcriptome's textual annotation. We prepared conversations in four ways, resulting in simple , detailed, complex, and conversational chats . The LLM prompts and examples of the generated conversations are shown in Supplementary Note 1 .",
          "section_heading": "Development and training of CellWhisperer's multimodal LLM for natural-language chats",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            11,
            12
          ],
          "doc_item_refs": [
            "#/texts/181",
            "#/texts/182",
            "#/texts/183",
            "#/texts/184",
            "#/texts/185",
            "#/texts/186",
            "#/texts/187",
            "#/texts/188"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002105::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_bb452f4c33e5",
          "configuration_id": "config_c0c5efca9f9e",
          "route_label": "Transcriptome embeddings plus free-text questions to Mistral 7B for CellWhisperer inference",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "natural-language chats and free-text queries",
          "source_object_verbatim": "transcriptome profiles",
          "source_object_normalized": "transcriptome profiles",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "Geneformer-based transcriptome embeddings",
            "adapter layer that converts the transcriptome embeddings into Mistral-compatible token-level embeddings"
          ],
          "model_visible_form_verbatim": "transcriptome-derived token embeddings plus free-text query tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "a two-layer adapter module",
          "fusion_topology": "prefix",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Third, to enable natural-language chats with CellWhisperer, we customized and fine-tuned the Mistral 7B open-source LLM (Jiang et al. 2023) to incorporate transcriptome information in addition to textual input. Our approach is inspired by the ability of multimodal LLMs - such as GPT-4, Gemini, and the open-source LLaVA model (H. Liu et al. 2023) - to interpret and converse about images. We generated a training dataset of 106,610 question-answer-style conversations, including simple rule-based conversations (e.g., 'What does the sample represent?' , with the sample's textual annotation as the designated answer) as well as more complex LLMgenerated conversations that all take transcriptome profiles into account (see Methods section for technical details and Supplementary Note 1 for examples). Based on this training dataset, we used the CellWhispererderived transcriptome embeddings together with the prepared questions as input to the Mistral 7B LLM (with an adapter layer that converts the transcriptome embeddings into Mistral -compatible token-level embeddings), and we fine-tuned this LLM to produce the matched answers. The resulting fine-tuned LLM responds to freetext questions and engages in natural-language chats about cells and their biological functions, gene-regulatory mechanisms, and other biological processes that can be linked to transcriptional cell states.",
          "section_heading": "CellWhisperer links transcriptomes and text through a multimodal embedding model and LLM",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            3
          ],
          "doc_item_refs": [
            "#/texts/24",
            "#/texts/386",
            "#/texts/53",
            "#/texts/54",
            "#/texts/55",
            "#/texts/56",
            "#/texts/57"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002105::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002105::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2951164b5686",
          "configuration_id": "config_fed1b10ee176",
          "route_label": "Conversation evaluation perplexity with matched or mismatched transcriptome embeddings",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "perplexity-based evaluation across 200 question-answer pairs",
          "source_object_verbatim": "200 question-answer pairs from the conversation evaluation dataset",
          "source_object_normalized": "conversation evaluation question-answer pairs",
          "source_modality_normalized": "mixed",
          "transformation_chain_verbatim": [
            "trimmed the conversations to the first question-answer pair",
            "removed passages that did not refer to biological cell states",
            "paired each question with the matched transcriptome embedding and 30 randomly sampled mismatched transcriptome embeddings"
          ],
          "model_visible_form_verbatim": "question text with matched or unmatched transcriptome embeddings and candidate answer tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "transcriptome-derived token embeddings were passed through the multimodal LLM adapter together with the question",
          "fusion_topology": "prefix",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "We first evaluated whether, for a given conversation, CellWhisperer's multimodal LLM preferred the correctly matched transcriptome over randomly mismatched transcriptomes. We quantified this preference using the perplexity metric, which is defined as the exponentiated average negative log-likelihood of the sequence of tokens that correspond to the 'ground truth' answer for a given prompt (i.e., a question and a matched or unmatched transcriptome embedding). In other words, perplexity measures the degree of surprise for the model when confronted with an answer to a given pair of a question and a matched or unmatched transcriptome.",
          "section_heading": "Evaluation of CellWhisperer's multimodal LLM for natural-language chats",
          "supporting_figure_or_table": "Extended Data Figure 3",
          "evidence_status": "explicit_text",
          "uncertainty": "The scored answer tokens are evaluation supervision rather than a standalone generation target, so I classified the route by the transcriptome-conditioned question/embedding input.",
          "pages": [
            12
          ],
          "doc_item_refs": [
            "#/texts/206",
            "#/texts/207",
            "#/texts/208",
            "#/texts/209"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002105::route_008"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_110ecfea952d",
          "configuration_id": "config_92cbb588891d",
          "route_label": "Tabula Sapiens cell-type answer perplexity with Mistral 7B",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "cell type prediction by perplexity",
          "source_object_verbatim": "Tabula Sapiens pseudo-bulk transcriptomes",
          "source_object_normalized": "pseudo-bulk transcriptome profiles",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "asked the prompt 'Which cell type is this cell?'",
            "converted each annotated cell type label into a natural-language answer by prefixing it with 'This cell is a '",
            "computed perplexity over all possible cell-type answer texts and ranked the correct answer against mismatched answers"
          ],
          "model_visible_form_verbatim": "transcriptome embeddings plus a candidate cell-type answer text",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "transcriptome-derived token embeddings were passed through the multimodal LLM adapter with the candidate answer text",
          "fusion_topology": "prefix",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "As a separate line of validation, we assessed whether CellWhisperer's multimodal LLM could correctly predict cell types in the Tabula Sapiens (20 common cell types) dataset. For each of the 20 cell types, we sampled 20 pseudo-bulk transcriptomes, resulting in a total of 400 individual transcriptomes. We asked a simple question ( 'Which cell type is this cell?' ) and converted the annotated cell type label to a natural-language answer by prefixing them with the following text: 'This cell is a ' . To evaluate the preference of a given transcriptome for its annotated cell type, we calculated the perplexity for all possible cell-type answer texts and determined the quantile of the correct cell-type perplexity against the background of all unmatched-answer perplexities.",
          "section_heading": "Evaluation of CellWhisperer's multimodal LLM for natural-language chats",
          "supporting_figure_or_table": "Extended Data Figure 3",
          "evidence_status": "explicit_text",
          "uncertainty": "The candidate answer text is scored during evaluation rather than serving as an ordinary user input, so I kept this as paired alignment input.",
          "pages": [
            12
          ],
          "doc_item_refs": [
            "#/texts/206",
            "#/texts/207",
            "#/texts/208",
            "#/texts/209"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002105::route_009"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_e367b465c0b9"
    },
    {
      "model_id": "model_f1bf6497dcf0",
      "model_name": "Mistral 7B",
      "record_id": "full_2026-07-06__rec_000827",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_f500772cfe83",
      "paper_title": "Multimodal learning enables chat-based exploration of single-cell data",
      "doi": "10.1038/s41587-025-02857-9",
      "paper_url": "https://doi.org/10.1038/s41587-025-02857-9",
      "route_count": 6,
      "configuration_count": 3,
      "family_counts": {
        "text_native_token_stream": 3,
        "dense_continuous_carrier": 3
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 2,
        "connector_mediated_embedding": 3,
        "serialized_biological_context_or_ordered_profile": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "connector_mediated_embedding",
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "connector_mediated_embedding",
      "modalities": [
        "RNA",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "inference"
      ],
      "fusion_topologies": [
        "concatenation",
        "prefix",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query",
        "metadata_or_context",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_000827_figure_011.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_000827_745a8d0e1225/figure_011.png",
        "figure_index": 11,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow/model architecture figure for integrating transcriptomic data with language-model instruction tuning.\n\nVisible elements:\n- Left input block: “Paired dataset: Transcriptomes & annotations,” listing datasets `GEO` with `705,430` data points and `CELLxGENE Census` with `376,983`.\n- Biological source objects: single-cell transcriptomes represented as “2048 top-expressed genes (tokenized)” and natural-language annotations/biomedical text.\n- Left model: `BioBERT`, labeled as pre-trained on biomedical texts, unfrozen, with `12 layers; 110M parameters`.\n- BioBERT output transformation: natural-language annotations are tokenized and passed through BioBERT to produce `768-dimensional` representations, then a `BatchNorm+ReLU` block maps `768→2048` and `2048→2048`, producing normalized `2048-dim.` embeddings.\n- Top-center model: `Geneformer`, labeled pre-trained on `30M single cells`, frozen, with `12 layers, 40M parameters`.\n- Geneformer output transformation: transcriptome gene tokens produce a `512-dimensional` embedding, then a `BatchNorm+ReLU` block maps `512→2048` and `2048→2048`, yielding normalized `2048-dim.` embeddings.\n- Center loss panel: matrix-style contrastive alignment using `InfoNCE loss`, with orange diagonal positive matches, gray negatives, labels such as “Dot product,” “Maximize,” “Minimize,” and batch size `512`.\n- Right input block: “Paired dataset: Transcriptomes & conversations,” listing conversation types and counts: detailed `81,610`, complex `10,000`, simple `5,000`.\n- Right model interface: transcriptome embeddings and question-answer conversations are tokenized and concatenated into an `8*4096-dim.` token embedding sequence.\n- Connector block: `2048→8*4096` and `8*4096→8*4096`, with `GELU`.\n- Right model: `Mistral 7B LLM (Instruct v0.2)`, described as pre-trained on proprietary text data and instruction-tuned on public data.\n- Training schedule labels: first epoch frozen, second epoch unfrozen.\n- Objective: multimodal instruction tuning with cross-entropy loss on answer tokens while question tokens are masked.\n\nNo experimental results or quantitative findings are shown beyond dataset sizes, model dimensions, parameters, and training/interface design.",
        "page_no": 17,
        "sha256": "2d079d60204f55a76fee754a1c4626d32b692e7944a6560e6c61a9c59aaa26f8",
        "pixel_width": 909,
        "pixel_height": 507,
        "crop_box": {
          "x": 0.64,
          "y": 0,
          "width": 0.36,
          "height": 0.84
        },
        "panel_label": "right-hand Mistral input/fusion branch",
        "visible_input_object": "Paired transcriptome/conversation inputs feeding the Mistral instruction-tuning branch",
        "visible_model_interface": "GELU adapter and token-concat interface into Mistral 7B LLM",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the right-side branch with the paired dataset, tokenized question-answer arrow, adapter block, and Mistral 7B input. It excludes the lower loss/output-only panel while keeping the immediate insertion/fusion path readable.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "connector_mediated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_9c9472f8e7e2",
          "example_input": "image / omics encoder states",
          "example_carrier": "Q-Former or adapter query vectors",
          "example_interface": "connector → LLM cross-modal interface",
          "actual_source": "bulk or single-cell transcriptome",
          "actual_model_visible_form": "eight 4,096-dimensional embeddings"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_f379828f3729",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "training-set questions",
          "actual_model_visible_form": "question tokens"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_98786d109e18",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "list of highly expressed genes",
          "actual_model_visible_form": "gene-list prompt text"
        }
      ],
      "routes": [
        {
          "route_id": "route_f379828f3729",
          "configuration_id": "config_8786848df489",
          "route_label": "training-set questions for Mistral 7B",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "fine-tuned this LLM to produce the matched answers",
          "source_object_verbatim": "training-set questions",
          "source_object_normalized": "training-set questions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "question tokenization",
            "paired with transcriptome-derived token embeddings"
          ],
          "model_visible_form_verbatim": "question tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "training-set questions as input to the Mistral 7B LLM",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Third, to enable natural-language chats that take the transcriptome information into account, we customized and fine-tuned the Mistral 7B open-weights LLM 13 to incorporate CellWhisperer transcriptome embeddings in addition to text queries. Our approach is inspired by multimodal LLMs that can interpret and converse about images, such as GPT-4, Gemini and LLaVA 14 . We generated a training dataset of 106,610 conversations including simple rule-based question-answer pairs (for example, 'What does the sample represent?', with the sample's textual annotation as the designated answer) and more complex LLM-generated conversations about transcriptomes and cells (technical details are provided in the Methods, examples in Supplementary Note 2). We used the embeddings together with the training-set questions as input to the Mistral 7B LLM (with an adapter layer that converts the embeddings into Mistral-compatible token-level embeddings) and fine-tuned this LLM to produce the matched answers. The resulting fine-tuned LLM responds to free-text questions and engages in natural-language chats about cells and their biological functions, gene-regulatory mechanisms and other biological processes that can be linked to transcriptional cell states.",
          "section_heading": "Multimodal architecture and training of the CellWhisperer chat model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/30"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9c9472f8e7e2",
          "configuration_id": "config_8786848df489",
          "route_label": "transcriptome embeddings for Mistral 7B",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "fine-tuned this LLM to produce the matched answers",
          "source_object_verbatim": "bulk or single-cell transcriptome",
          "source_object_normalized": "transcriptome",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "CellWhisperer multimodal embedding",
            "two-layer adapter module",
            "Mistral-compatible token-level embeddings"
          ],
          "model_visible_form_verbatim": "eight 4,096-dimensional embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "a two-layer adapter module that transforms the 2,048-dimensional CellWhisperer multimodal embedding",
          "fusion_topology": "prefix",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Third, to enable natural-language chats that take the transcriptome information into account, we customized and fine-tuned the Mistral 7B open-weights LLM 13 to incorporate CellWhisperer transcriptome embeddings in addition to text queries. Our approach is inspired by multimodal LLMs that can interpret and converse about images, such as GPT-4, Gemini and LLaVA 14 . We generated a training dataset of 106,610 conversations including simple rule-based question-answer pairs (for example, 'What does the sample represent?', with the sample's textual annotation as the designated answer) and more complex LLM-generated conversations about transcriptomes and cells (technical details are provided in the Methods, examples in Supplementary Note 2). We used the embeddings together with the training-set questions as input to the Mistral 7B LLM (with an adapter layer that converts the embeddings into Mistral-compatible token-level embeddings) and fine-tuned this LLM to produce the matched answers. The resulting fine-tuned LLM responds to free-text questions and engages in natural-language chats about cells and their biological functions, gene-regulatory mechanisms and other biological processes that can be linked to transcriptional cell states.",
          "section_heading": "Multimodal architecture and training of the CellWhisperer chat model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/30"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_18f89518c544",
          "configuration_id": "config_26dea27b9f22",
          "route_label": "free-text questions for Mistral 7B chat",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "responds to free-text questions and engages in natural-language chats about cells",
          "source_object_verbatim": "free-text questions",
          "source_object_normalized": "free-text questions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenization",
            "combined with transcriptome-derived token embeddings"
          ],
          "model_visible_form_verbatim": "question tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "answer free-text questions about cell states",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Third, to enable natural-language chats that take the transcriptome information into account, we customized and fine-tuned the Mistral 7B open-weights LLM 13 to incorporate CellWhisperer transcriptome embeddings in addition to text queries. Our approach is inspired by multimodal LLMs that can interpret and converse about images, such as GPT-4, Gemini and LLaVA 14 . We generated a training dataset of 106,610 conversations including simple rule-based question-answer pairs (for example, 'What does the sample represent?', with the sample's textual annotation as the designated answer) and more complex LLM-generated conversations about transcriptomes and cells (technical details are provided in the Methods, examples in Supplementary Note 2). We used the embeddings together with the training-set questions as input to the Mistral 7B LLM (with an adapter layer that converts the embeddings into Mistral-compatible token-level embeddings) and fine-tuned this LLM to produce the matched answers. The resulting fine-tuned LLM responds to free-text questions and engages in natural-language chats about cells and their biological functions, gene-regulatory mechanisms and other biological processes that can be linked to transcriptional cell states.",
          "section_heading": "Results",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/30"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_bb26511b84b2",
          "configuration_id": "config_26dea27b9f22",
          "route_label": "transcriptome embeddings for free-text chat",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "responds to free-text questions and engages in natural-language chats about cells",
          "source_object_verbatim": "user-provided transcriptome profiles",
          "source_object_normalized": "transcriptome",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "CellWhisperer transcriptome embeddings",
            "adapter-mediated chat-model conditioning"
          ],
          "model_visible_form_verbatim": "user-provided transcriptome profiles as multimodal input",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "while considering user-provided transcriptome profiles as multimodal input",
          "fusion_topology": "prefix",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Third, to enable natural-language chats that take the transcriptome information into account, we customized and fine-tuned the Mistral 7B open-weights LLM 13 to incorporate CellWhisperer transcriptome embeddings in addition to text queries. Our approach is inspired by multimodal LLMs that can interpret and converse about images, such as GPT-4, Gemini and LLaVA 14 . We generated a training dataset of 106,610 conversations including simple rule-based question-answer pairs (for example, 'What does the sample represent?', with the sample's textual annotation as the designated answer) and more complex LLM-generated conversations about transcriptomes and cells (technical details are provided in the Methods, examples in Supplementary Note 2). We used the embeddings together with the training-set questions as input to the Mistral 7B LLM (with an adapter layer that converts the embeddings into Mistral-compatible token-level embeddings) and fine-tuned this LLM to produce the matched answers. The resulting fine-tuned LLM responds to free-text questions and engages in natural-language chats about cells and their biological functions, gene-regulatory mechanisms and other biological processes that can be linked to transcriptional cell states.",
          "section_heading": "Results",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/30"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_98786d109e18",
          "configuration_id": "config_f655c388e39b",
          "route_label": "highly expressed genes prompt for chat",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "explicitly providing a list of highly expressed genes as part of the prompt, in addition to the transcriptome embedding",
          "source_object_verbatim": "list of highly expressed genes",
          "source_object_normalized": "top 50 highly expressed genes",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "top 50 highly expressed genes",
            "prompt construction for chat"
          ],
          "model_visible_form_verbatim": "gene-list prompt text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "as part of the prompt",
          "fusion_topology": "concatenation",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "We further assessed the perplexity for responses obtained with the Mistral 7B LLM (which the CellWhisperer chat model builds upon) and for the much larger Llama 3.3 70B LLM (Extended Data Fig. 4c). CellWhisperer achieved best results (lowest perplexity values), even on the out-of-distribution Cell Type Conversations dataset, further supporting that our chat model effectively incorporates the CellWhisperer transcriptome embeddings. We also assessed whether the CellWhisperer chat model may benefit from explicitly providing a list of highly expressed genes as part of the prompt (as commonly done when analyzing transcriptomes with text-only LLMs 39,40 ), in addition to the transcriptome embedding. We observed a mild beneficial effect (Extended Data Fig. 4c) and implemented this hybrid approach in the CellWhisperer web tool.",
          "section_heading": "Evaluation of the CellWhisperer chat model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8
          ],
          "doc_item_refs": [
            "#/texts/773",
            "#/texts/774",
            "#/texts/775",
            "#/texts/779",
            "#/texts/780"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d9f23aaa6ac4",
          "configuration_id": "config_f655c388e39b",
          "route_label": "transcriptome embeddings for gene-list-augmented chat",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "explicitly providing a list of highly expressed genes as part of the prompt, in addition to the transcriptome embedding",
          "source_object_verbatim": "transcriptome embedding",
          "source_object_normalized": "transcriptome",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "transcriptome embedding",
            "chat-model conditioning with gene-list prompt"
          ],
          "model_visible_form_verbatim": "transcriptome embedding",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "in addition to the transcriptome embedding",
          "fusion_topology": "prefix",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "We further assessed the perplexity for responses obtained with the Mistral 7B LLM (which the CellWhisperer chat model builds upon) and for the much larger Llama 3.3 70B LLM (Extended Data Fig. 4c). CellWhisperer achieved best results (lowest perplexity values), even on the out-of-distribution Cell Type Conversations dataset, further supporting that our chat model effectively incorporates the CellWhisperer transcriptome embeddings. We also assessed whether the CellWhisperer chat model may benefit from explicitly providing a list of highly expressed genes as part of the prompt (as commonly done when analyzing transcriptomes with text-only LLMs 39,40 ), in addition to the transcriptome embedding. We observed a mild beneficial effect (Extended Data Fig. 4c) and implemented this hybrid approach in the CellWhisperer web tool.",
          "section_heading": "Evaluation of the CellWhisperer chat model",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8
          ],
          "doc_item_refs": [
            "#/texts/773",
            "#/texts/774",
            "#/texts/775",
            "#/texts/779",
            "#/texts/780"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_47e8053eae22"
    },
    {
      "model_id": "model_364e087c07ba",
      "model_name": "Mistral-7B",
      "record_id": "full_2026-07-06__rec_003043",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_3bbf77870504",
      "paper_title": "Geneverse: A collection of Open-source Multimodal Large Language Models for Genomic and Proteomic Research",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003043_figure_006.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003043_460a3d629653/figure_006.png",
        "figure_index": 6,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a two-panel schematic labeled **a** and **b**, each titled **“Instruction example”** and showing an LLM-style question-answer interface.\n\nPanel **a** shows an example involving a protein structure. The prompt asks: “What is the name of this protein? Please summarize your answer in one sentence.” A small ribbon-like protein structure image appears beside the prompt. The LLM response box answers: “It is AF-O43280-F1.”\n\nPanel **b** shows an example involving biological microscopy/histology imagery. The prompt asks whether **gene APOC1** is a marker gene of **cell type Macrophage**, followed by “Please summarize your answer in one sentence.” A purple-stained tissue microscopy image appears beside the prompt. The LLM response box answers: “Yes.”\n\nBoth panels depict instruction-following model interfaces with biological source objects as inputs: a protein structure in panel **a** and a histological tissue image plus gene/cell-type query in panel **b**. The figure illustrates LLM question answering over biological or biomedical visual/contextual inputs.",
        "page_no": 16,
        "sha256": "9e63b9a5663cc3acc3d474d94d9d01570d390f25288c9abcc3bc702f6efb3458",
        "pixel_width": 901,
        "pixel_height": 373,
        "crop_box": {
          "x": 0.49,
          "y": 0.03,
          "width": 0.37,
          "height": 0.53
        },
        "panel_label": "b",
        "visible_input_object": "Instruction example with the gene APOC1 query prompt asking whether it is a marker gene of cell type Macrophage",
        "visible_model_interface": "Text-only prompt in an LMM instruction-example box",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the panel label, the instruction-example header, and the full text prompt that defines the text-only input route. It excludes the answer box and the unrelated left panel while preserving the grounded source object and model-visible carrier.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_a8fb3283b641",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene name in the prompt",
          "actual_model_visible_form": "text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_a8fb3283b641",
          "configuration_id": "config_0f827720913c",
          "route_label": "gene-function prompt to Mistral-7B",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "gene function description",
          "source_object_verbatim": "gene name in the prompt",
          "source_object_normalized": "GLI1",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "instruction example"
          ],
          "model_visible_form_verbatim": "text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "text-only prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Our task is to generate the summary of gene functions based on the prompt only containing the task description.",
          "section_heading": "4.1 Benchmarking LLMs for summarizing of gene functions",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            16
          ],
          "doc_item_refs": [
            "#/texts/438",
            "#/texts/70",
            "#/texts/71",
            "#/texts/72",
            "#/texts/73"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003043::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003043::0035"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_073c1a9ed6a3",
      "model_name": "Mixtral 8x7b",
      "record_id": "full_2026-07-06__rec_000827",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_f500772cfe83",
      "paper_title": "Multimodal learning enables chat-based exploration of single-cell data",
      "doi": "10.1038/s41587-025-02857-9",
      "paper_url": "https://doi.org/10.1038/s41587-025-02857-9",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "pretraining"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "metadata_or_context"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_000827_figure_003.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_000827_745a8d0e1225/figure_003.png",
        "figure_index": 3,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel scientific figure about CellWhisperer, a joint embedding/model interface connecting transcriptomes and text.\n\nPanel a: Schematic workflow. Left shows dataset construction from “1 million pairs of human transcriptomes and text,” generating sample descriptions using an LLM, with metadata, expert annotations, GEO metadata, RNA-seq, single-cell RNA-seq, gene expression counts, and generated conversations. Center shows training and inference in a “multimodal embedding space (2,048 dimensions)” with language model adapter, transcriptome model adapter, contrastive alignment/minimize distance, supervised training, and generative LLM/Mistral interface. Right shows applications: free-text cell labeling from embedded dataset with prompt “immune cells,” term matching via embedding similarity, CellWhisperer score, and chat about user-provided transcriptome data, including question-answer bubbles about scRNA-seq dataset identity and characteristic genes/pathways.\n\nPanel b: UMAP landscape titled “Landscape of annotated human transcriptomes,” showing many colored clusters. Caption indicates CellWhisperer embedding of 705,430 transcriptomes from GEO.\n\nPanel c: UMAP query result titled “Querying the human transcriptome landscape with natural language,” showing red/blue CellWhisperer scores for a search term related to “inflammation.” Color bar labeled CellWhisperer score, roughly from negative blue to positive red. Dashed callouts point to regions matching textual query phrases such as inflamed immune response, heightened immune response to MTB infection, K562 leukemia cells, and active remodeling/immune-response cells.\n\nPanel d: Three histogram plots titled “GEO submission dates for the selected samples,” with x-axis GEO submission date and y-axis number of samples. The three rows correspond to selected query-associated sample groups, including CD34+ HSPCs with broad differentiation potential, K562 leukemia cells cultured in supplemented RPMI 1640, and active remodeling and immune-response cells.",
        "page_no": 3,
        "sha256": "808cb3e1740edc0e44a7ea55410cdcdfdc1c27b072d90e3a725a43dac17b8fe6",
        "pixel_width": 1034,
        "pixel_height": 1018,
        "crop_box": {
          "x": 0.0,
          "y": 0.04,
          "width": 0.29,
          "height": 0.48
        },
        "panel_label": "a",
        "visible_input_object": "Metadata from GEO and CELLxGENE Census, plus example sample-description text generated from a renal cell carcinoma sample",
        "visible_model_interface": "Generating sample descriptions with an LLM (GPT-4, Mixtral) and the resulting natural-language sample text / conversation bubbles",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This left-panel crop preserves the grounded input route: source metadata and annotations feed the LLM to generate sample-description text, with the prompt/output arrows still readable. It excludes the later training/inference and application-only panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_23ef5a05a115",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "metadata provided by GEO and CELLxGENE Census",
          "actual_model_visible_form": "sample metadata text"
        }
      ],
      "routes": [
        {
          "route_id": "route_23ef5a05a115",
          "configuration_id": "config_6f1ccbd9e068",
          "route_label": "metadata summarization for text annotations",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "generate a concise natural-language summary from the metadata",
          "source_object_verbatim": "metadata provided by GEO and CELLxGENE Census",
          "source_object_normalized": "sample metadata",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "metadata fields from GEO and CELLxGENE Census",
            "concise natural-language summary generation"
          ],
          "model_visible_form_verbatim": "sample metadata text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "guided by a prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "we generated a concise natural-language summary from the metadata using Mixtral 8x7b",
          "section_heading": "Multimodal training data of paired transcriptomes and text",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            12
          ],
          "doc_item_refs": [
            "#/texts/1085",
            "#/texts/1086",
            "#/texts/1087",
            "#/texts/1088",
            "#/texts/1089"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000827::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_7edc725404f8",
      "model_name": "Mixtral 8x7b",
      "record_id": "full_2026-07-06__rec_002105",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_266ca9b669f4",
      "paper_title": "Multimodal learning of transcriptomes and text enables interactive single-cell RNA-seq data exploration with natural-language chats",
      "doi": "10.1101/2024.10.15.618501",
      "paper_url": "https://doi.org/10.1101/2024.10.15.618501",
      "route_count": 2,
      "configuration_count": 2,
      "family_counts": {
        "text_native_token_stream": 2
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 2
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "inference"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "metadata_or_context"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_002105_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_002105_966eaaaee764/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1 . Overview of the CellWhisperer multimodal AI for natural-language analysis of transcriptome data",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel scientific figure describing a method called CellWhisperer for joint embedding of transcriptomes and text.\n\nPanel a: Workflow schematic titled “Joint embedding of transcriptomes and text for data exploration via natural-language chats.” It shows a dataset of 1 million paired human transcriptomes and text. Inputs include GEO metadata, expert annotations, bulk RNA-seq, single-cell RNA-seq, and gene expression counts. Sample descriptions and conversations are generated using LLMs including GPT-4 and Mixtral. The central diagram shows training and inference in a 2048-dimensional multimodal embedding space, with language and transcriptome models aligned via minimized distance and supervised training. The model interface includes text prompts, answers, transcriptome embeddings, and a generative LLM component. Applications shown on the right include free-text cell labeling, user-provided transcriptome data such as single-cell RNA-seq, and chat-style querying about cell samples.\n\nPanel b: UMAP visualization labeled “An annotated landscape of human transcriptomes.” It shows many colored clusters representing CellWhisperer embeddings of 705,430 transcriptomes from GEO. Dashed circles highlight selected regions. Biological labels connected to regions include “Activated CD8+ T Cells in Immune Response,” “Activated CD4+ T Cells in Immune Response,” “Active Myeloid Differentiation in HSPCs,” “K562 Erythroleukemia Cells in Culture,” and “Obese adipose tissue immune and metabolic state.”\n\nPanel c: UMAP visualization titled “Querying the human transcriptome landscape with natural language.” Points are colored by CellWhisperer score for the search term “leukemia,” with a blue-to-red scale. Red regions indicate higher similarity to the leukemia query.\n\nPanel d: Small histogram plots titled “GEO submission dates for the selected samples.” Three distributions are shown for selected biological query regions, with GEO submission dates spanning roughly 2013 to 2023.\n\nBiological source objects visible: human transcriptomes from GEO, bulk RNA-seq, single-cell RNA-seq, gene expression count matrices, renal carcinoma sample metadata, immune cells, HSPCs, K562 erythroleukemia cells, adipose tissue immune/metabolic samples.\n\nTransformations and interfaces: metadata and annotations are converted into sample descriptions and conversations by LLMs; transcriptomes and text are embedded into a shared multimodal space; natural-language prompts are used for cell labeling, sample search, and chat-based interpretation.",
        "page_no": 14,
        "sha256": "d836049e5b8ee5e302ee92c8b7045d9c05354705d6e883c60bd0b2a7c974fd85",
        "pixel_width": 962,
        "pixel_height": 949,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 0.335,
          "height": 0.54
        },
        "panel_label": "a",
        "visible_input_object": "GEO sample-associated metadata and expert annotations for transcriptome samples",
        "visible_model_interface": "LLM-based generation of sample descriptions and conversation prompts from metadata",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the left workflow in panel a, where GEO metadata is turned into text prompts/sample descriptions and then into training conversations. This keeps the source object, transformation, and model-visible text carrier while excluding the embedding-space overview and downstream application panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_4046842f6227",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "GEO sample-associated metadata",
          "actual_model_visible_form": "metadata fields turned into a text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_4046842f6227",
          "configuration_id": "config_97e791188e94",
          "route_label": "GEO metadata summary with Mixtral 8x7b",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "GEO sample metadata summarization",
          "source_object_verbatim": "GEO sample-associated metadata",
          "source_object_normalized": "GEO sample metadata",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "obtained the associated metadata using the Entrez API with either the sample's experiment accession, BioSample accession, or GEO accession",
            "removed binary data as well as special characters using the unidecode package",
            "generated a concise natural-language summary from the metadata"
          ],
          "model_visible_form_verbatim": "metadata fields turned into a text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "via the llama.cpp python bindings",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "For each sample (GEO) or pseudo-bulk transcriptome (CELLxGENE Census), we generated a concise naturallanguage summary from the metadata using Mixtral 8x7b (Jiang et al. 2024) (Q5_K_M quantized version), via the llama.cpp python bindings with a sampling temperature of 0.2, nucleus sampling (top_p) of 0.9, and top",
          "section_heading": "Creation of a multimodal training dataset with pairs of transcriptomes and text",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9
          ],
          "doc_item_refs": [
            "#/texts/131",
            "#/texts/132",
            "#/texts/133",
            "#/texts/134",
            "#/texts/135",
            "#/texts/136",
            "#/texts/137",
            "#/texts/138"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002105::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_42e3497047e3",
          "configuration_id": "config_1928e83b2412",
          "route_label": "CELLxGENE Census metadata summary with Mixtral 8x7b",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "CELLxGENE Census pseudo-bulk metadata summarization",
          "source_object_verbatim": "CELLxGENE Census cell-level and study-level metadata",
          "source_object_normalized": "CELLxGENE Census metadata",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "grouped the cells in each dataset based on the provided metadata",
            "calculated pseudo-bulk transcriptomes by taking the mean of the scRNA-seq count values across all cells in each metadata-defined group of cells",
            "generated a concise natural-language summary from the metadata"
          ],
          "model_visible_form_verbatim": "metadata fields turned into a text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "via the llama.cpp python bindings",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "For each sample (GEO) or pseudo-bulk transcriptome (CELLxGENE Census), we generated a concise naturallanguage summary from the metadata using Mixtral 8x7b (Jiang et al. 2024) (Q5_K_M quantized version), via the llama.cpp python bindings with a sampling temperature of 0.2, nucleus sampling (top_p) of 0.9, and top",
          "section_heading": "Creation of a multimodal training dataset with pairs of transcriptomes and text",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9
          ],
          "doc_item_refs": [
            "#/texts/131",
            "#/texts/132",
            "#/texts/133",
            "#/texts/134",
            "#/texts/135",
            "#/texts/136",
            "#/texts/137",
            "#/texts/138"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002105::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_59fd1ac5f56e",
      "model_name": "MUPAD",
      "record_id": "full_2026-07-06__rec_003852",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_d208190fc72d",
      "paper_title": "A Generative Foundation Model for Multimodal Histopathology",
      "doi": "",
      "paper_url": "",
      "route_count": 11,
      "configuration_count": 9,
      "family_counts": {
        "text_native_token_stream": 3,
        "dense_continuous_carrier": 6,
        "visual_raster_carrier": 1,
        "geometric_or_diffusion_state_carrier": 1
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 2,
        "pooled_or_aggregated_embedding": 3,
        "raw_slide_or_patch_input": 1,
        "noisy_diffusion_state": 1,
        "structured_biological_prompt_or_task_scaffold": 1,
        "direct_projected_embedding": 2,
        "connector_mediated_embedding": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier",
        "visual_raster_carrier",
        "geometric_or_diffusion_state_carrier"
      ],
      "subtypes": [
        "connector_mediated_embedding",
        "direct_projected_embedding",
        "noisy_diffusion_state",
        "plain_language_prompt_or_question",
        "pooled_or_aggregated_embedding",
        "raw_slide_or_patch_input",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "pooled_or_aggregated_embedding",
      "modalities": [
        "histology image",
        "histology section",
        "histopathology image",
        "text",
        "transcriptomics"
      ],
      "lifecycle_phases": [
        "evaluation",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "cross_attention",
        "side_or_generative_conditioning"
      ],
      "text_roles": [
        "instruction_or_query",
        "modality_or_task_selector",
        "no_text_on_this_route",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003852_figure_005.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003852_e33170bc6302/figure_005.png",
        "figure_index": 5,
        "caption": "Figure 5: Virtual H&E-to-IHC translation and clinical validation. a, Visual examples of multi-stain virtual IHC generation. Compared to CycleGAN and CUT, MUPAD provides more accurate spatial stain rendering. b, FID and KID scores demonstrate that MUPAD achieves better distributional fidelity and perceptual quality. c, Clinical utility on the IHC4BC dataset. When using virtual IHC images to predict ground-truth clinical biomarkers (H-scores for ER/PR/Ki67; amplification status for HER2), MUPAD yields the highest AUC performance across all markers. Sub-0.5 AUC values for baseline methods indicate systematic inverse correlations with clinical labels, as observed the semantic misalignment in stain generation.",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel biomedical figure comparing image-to-image translation methods for histopathology/IHC stain generation.\n\nPanel (a) shows example image tiles arranged in columns labeled `H&E (input)`, `IHC (true)`, `MUPAD`, `CycleGAN`, and `CUT`. Rows are grouped by biomarker/dataset labels on the left: `HER2Match`, `IHC4BC-ER`, `IHC4BC-PR`, `IHC4BC-Her2`, and `IHC4BC-Ki67`. The visible biological source objects are breast cancer histology tissue micrographs/cell fields with hematoxylin/eosin morphology and brown immunohistochemical staining patterns. The transformation shown is virtual IHC prediction from H&E input, with generated outputs from MUPAD, CycleGAN, and CUT compared against true IHC.\n\nPanel (b) contains bar charts for quantitative image-distribution metrics, with `FID ↓` and `KID ↓` columns. For each biomarker/dataset row, MUPAD generally has lower FID and KID than CycleGAN and CUT. Example visible values include HER2Match FID 103.92 for MUPAD, 117.74 CycleGAN, 189.16 CUT; and KID 0.1146, 0.1559, 0.2683 respectively.\n\nPanel (c) contains AUC bar charts for downstream biomarker classification tasks labeled `ER`, `PR`, `Ki67`, and `Her2`. MUPAD has the highest visible AUC in all four plots: ER 0.8872, PR 0.8842, Ki67 0.9772, Her2 0.9556. CycleGAN and CUT are lower for ER/PR/Her2, while for Ki67 CUT 0.9021 is higher than CycleGAN 0.8736 but below MUPAD.\n\nOverall finding shown: MUPAD better matches true IHC appearance and yields better quantitative FID/KID and classification AUC than CycleGAN and CUT in the displayed comparisons.",
        "page_no": 14,
        "sha256": "a19e998bec70399b41f81e4755f90d516aa43e50030e40dd12ba13aa789cc15a",
        "pixel_width": 950,
        "pixel_height": 1110,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 0.56,
          "height": 0.15
        },
        "panel_label": "(a)",
        "visible_input_object": "H&E-stained histology patch (source input)",
        "visible_model_interface": "H&E (input) -> IHC (true) / MUPAD / CycleGAN / CUT image-to-image translation comparison",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the panel (a) source and target example tiles plus the column labels needed to read the H&E-to-IHC translation route, while excluding the quantitative panels (b) and (c).",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "connector_mediated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_9ee52b4bb668",
          "example_input": "image / omics encoder states",
          "example_carrier": "Q-Former or adapter query vectors",
          "example_interface": "connector → LLM cross-modal interface",
          "actual_source": "H&E image",
          "actual_model_visible_form": "MUSK patch embeddings c sem"
        },
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_b3ac28847b13",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "H&E-stained slide image",
          "actual_model_visible_form": "translated IHC embedding"
        },
        {
          "subtype_id": "noisy_diffusion_state",
          "family_id": "geometric_or_diffusion_state_carrier",
          "route_id": "route_29724342cf05",
          "example_input": "biological state x₀ + noise ε",
          "example_carrier": "xₜ = √αₜx₀ + √(1−αₜ)ε",
          "example_interface": "conditioned denoiser / flow model",
          "actual_source": "fresh frozen section image",
          "actual_model_visible_form": "latent z_T under the source prompt"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_a6f084035330",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "descriptive histopathological caption",
          "actual_model_visible_form": "descriptive histopathological caption tokens"
        },
        {
          "subtype_id": "pooled_or_aggregated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_af3e8857f978",
          "example_input": "{gene/cell/patch embeddings}",
          "example_carrier": "mean/attention pool = one compact vector",
          "example_interface": "aggregator → generator",
          "actual_source": "bulk RNA-seq profile",
          "actual_model_visible_form": "331 pathway-level enrichment scores"
        },
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_94c33410fd8a",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "reference tissue image",
          "actual_model_visible_form": "reference image embeddings"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_a5983a60a443",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "text prompt ('fresh frozen' or 'ffpe image')",
          "actual_model_visible_form": "prompt tokens"
        }
      ],
      "routes": [
        {
          "route_id": "route_a6f084035330",
          "configuration_id": "config_9d2e426482f8",
          "route_label": "text-histology pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "1.6 million text-histology pairs",
          "source_object_verbatim": "descriptive histopathological caption",
          "source_object_normalized": "histopathological caption",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "descriptive histopathological caption tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "text cross-attention stream in decoupled cross-modal attention",
          "fusion_topology": "cross_attention",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "1.6 million text-histology pairs",
          "section_heading": "Abstract",
          "supporting_figure_or_table": "Figure 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            23
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/104",
            "#/texts/29"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003852::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f2a5f036cea6",
          "configuration_id": "config_dfa56cc0154a",
          "route_label": "text-to-image generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "zero-shot generation on 100K held-out captions from PathGen-1.6M",
          "source_object_verbatim": "held-out pathology caption",
          "source_object_normalized": "pathology caption",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "held-out pathology caption tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "text cross-attention stream in decoupled cross-modal attention",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "held-out captions",
          "section_heading": "Application: Text-to-Image and Image-to-Image Generation",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            6,
            7,
            8,
            10
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/texts/29",
            "#/texts/38",
            "#/texts/40",
            "#/texts/42",
            "#/texts/51"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003852::0001",
            "dense::full_2026-07-06__rec_003852::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_af3e8857f978",
          "configuration_id": "config_586e97e174a5",
          "route_label": "RNA-histology pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "10.8 million RNA-histology pairs",
          "source_object_verbatim": "bulk RNA-seq profile",
          "source_object_normalized": "bulk RNA-seq profile",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "normalized to TPM",
            "filtered to remove housekeeping and mitochondrial genes",
            "compressed into 331 pathway-level enrichment scores"
          ],
          "model_visible_form_verbatim": "331 pathway-level enrichment scores",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "RNA cross-attention stream in decoupled cross-modal attention",
          "fusion_topology": "cross_attention",
          "text_role": "no_text_on_this_route",
          "input_status": "paired_alignment_input",
          "evidence_quote": "10.8 million RNA-histology pairs",
          "section_heading": "Abstract",
          "supporting_figure_or_table": "Figure 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            23
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/104",
            "#/texts/29"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003852::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9a9d1cbd25a1",
          "configuration_id": "config_59a2ae1ec7a9",
          "route_label": "spatial transcriptomics to H&E generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "conditional H&E generation from spatial transcriptomics data",
          "source_object_verbatim": "spatial transcriptomics profile",
          "source_object_normalized": "spatial transcriptomics profile",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "compressed into the same 331 pathway-level scores"
          ],
          "model_visible_form_verbatim": "331 pathway-level scores",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "RNA conditioning stream in decoupled cross-modal attention",
          "fusion_topology": "cross_attention",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Although MUPAD is pretrained on bulk RNA-seq, we evaluate transcriptomics-conditioned generation on spatial transcriptomics (ST) data. Both bulk and ST profiles are compressed into the same 331 pathway-level scores, enabling zero-shot transfer of the pretrained priors. Fine-tuning on ST data from MOSAIC refines these priors toward spatially resolved generation. WSIs are tiled into 55 µ m spots ( 256 × 256 px); paired ST matrices are aligned to extract regional expression profiles per patch. We fine-tune with AdamW (lr = 10 -4 , 4 GPUs, 50K steps). Evaluation uses H-optimus-1 FID and cell-type distribution fidelity via Histoplus 36 (14 cell classes).",
          "section_heading": "Spatial transcriptomics to H&E image generation",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            24
          ],
          "doc_item_refs": [
            "#/texts/124"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_94c33410fd8a",
          "configuration_id": "config_86891985f771",
          "route_label": "image-to-image generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "image-to-image generation",
          "source_object_verbatim": "reference tissue image",
          "source_object_normalized": "histology image",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "reference image embeddings",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "image cross-attention stream in decoupled cross-modal attention",
          "fusion_topology": "cross_attention",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "reference tissue image",
          "section_heading": "Image-to-image generation",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            7,
            9
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/pictures/2",
            "#/texts/35",
            "#/texts/36",
            "#/texts/40",
            "#/texts/47"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003852::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_29724342cf05",
          "configuration_id": "config_bb70a2a0cfb3",
          "route_label": "fresh frozen to FFPE image translation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "virtual FF-to-FFPE conversion",
          "source_object_verbatim": "fresh frozen section image",
          "source_object_normalized": "fresh frozen section image",
          "source_modality_normalized": "histology section",
          "transformation_chain_verbatim": [
            "DDIM inversion",
            "cross-domain attention injection"
          ],
          "model_visible_form_verbatim": "latent z_T under the source prompt",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "noisy_diffusion_state",
          "insertion_or_fusion_verbatim": "source-conditioned reconstruction with self-attention map injection",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "The resulting model supports a broad range of cross-modal synthesis tasks with minimal or no task-specific fine-tuning. These include text-conditioned and RNA-conditioned histology generation for spatial transcriptomics exploration; translation of fresh-frozen (FF) sections to formalin-fixed paraffin-embedded (FFPE) appearance; and virtual staining of H&E images into IHC and mIF assays. Across all evaluated benchmarks, MUPAD substantially outperforms domain-specific generative models trained for individual tasks (Fig. 1d), a result we attribute to the cross-modal priors acquired through large-scale multimodal pretraining.",
          "section_heading": "Application: Fresh Frozen to FFPE Translation",
          "supporting_figure_or_table": "Extended Data Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            12,
            21,
            30,
            31,
            33
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/7",
            "#/tables/1",
            "#/texts/219",
            "#/texts/220",
            "#/texts/221",
            "#/texts/222",
            "#/texts/224",
            "#/texts/29",
            "#/texts/61",
            "#/texts/93",
            "#/texts/94",
            "#/texts/95",
            "#/texts/96"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003852::0003",
            "dense::full_2026-07-06__rec_003852::0014"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a5983a60a443",
          "configuration_id": "config_bb70a2a0cfb3",
          "route_label": "FF-to-FFPE text prompt conditioning",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "virtual FF-to-FFPE conversion",
          "source_object_verbatim": "text prompt ('fresh frozen' or 'ffpe image')",
          "source_object_normalized": "domain-control text prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "prompt tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "text conditioning with classifier-free guidance",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "modality_or_task_selector",
          "input_status": "actual_model_input",
          "evidence_quote": "text conditioning",
          "section_heading": "Application: Fresh Frozen to FFPE Translation",
          "supporting_figure_or_table": "Extended Data Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            12
          ],
          "doc_item_refs": [
            "#/texts/61"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b3ac28847b13",
          "configuration_id": "config_3fe7fd0f96bb",
          "route_label": "H&E to IHC translation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "virtual IHC staining from standard H&E slides",
          "source_object_verbatim": "H&E-stained slide image",
          "source_object_normalized": "H&E slide image",
          "source_modality_normalized": "histology image",
          "transformation_chain_verbatim": [
            "bridge the modality gap in MUSK latent space via flow matching",
            "a 6-layer MLP learns a continuous flow from H&E embeddings to IHC embeddings"
          ],
          "model_visible_form_verbatim": "translated IHC embedding",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "translated embedding conditions MUPAD",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "IHC staining remains indispensable for protein biomarker assessment and treatment stratification, but requires dedicated laboratory infrastructure, trained personnel, and multi-day turnaround times, limiting its availability in resource-constrained clinical settings. 42 . We evaluated MUPAD on the task of virtual IHC staining from standard H&E slides, focusing on five distinct markers in the weakly co-registered IHC4BC 43 and HER2Match 44 datasets. We benchmarked against widely used image-to-image translation methods CUT 40 and CycleGAN 41 .",
          "section_heading": "Application: H&E-to-IHC Translation",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13
          ],
          "doc_item_refs": [
            "#/texts/64",
            "#/texts/65",
            "#/texts/66"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_008"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c111c265cad5",
          "configuration_id": "config_b763e4e5a76a",
          "route_label": "H&E to mIF translation - structural latent conditioning",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "H&E-to-mIF image translation",
          "source_object_verbatim": "H&E image",
          "source_object_normalized": "H&E image",
          "source_modality_normalized": "histology image",
          "transformation_chain_verbatim": [
            "spatial concatenation of the H&E VAE latent z struct with the noisy mIF latent at each timestep"
          ],
          "model_visible_form_verbatim": "H&E VAE latent z struct",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "spatial concatenation of the H&E VAE latent z struct with the noisy mIF latent at each timestep",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "We fine-tune MUPAD on the ORION colorectal cancer dataset 46 (41 WSIs, 16 protein markers). H&E conditioning is applied through two complementary mechanisms: (1) spatial concatenation of the H&E VAE latent z struct with the noisy mIF latent at each timestep, providing a structural template; and (2) semantic cross-attention using MUSK patch embeddings c sem to modulate generation based on tissue pathology. Multiplex channels are grouped into 3-channel subsets and encoded with the VAE independently. The model is trained with flow-matching velocity prediction loss using BF16 mixed precision with HED-based H&E color augmentation. We evaluate Patch PCC and Slide-level PCC against GigaTIME 14 and HistoPlexer 47 .",
          "section_heading": "Application: H&E-to-mIF Translation",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            16,
            17,
            18,
            26
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/138",
            "#/texts/72",
            "#/texts/73",
            "#/texts/75",
            "#/texts/76",
            "#/texts/77",
            "#/texts/80"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003852::0009"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9ee52b4bb668",
          "configuration_id": "config_b763e4e5a76a",
          "route_label": "H&E to mIF translation - semantic cross-attention conditioning",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "H&E-to-mIF image translation",
          "source_object_verbatim": "H&E image",
          "source_object_normalized": "H&E image",
          "source_modality_normalized": "histology image",
          "transformation_chain_verbatim": [
            "semantic cross-attention using MUSK patch embeddings c sem"
          ],
          "model_visible_form_verbatim": "MUSK patch embeddings c sem",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "semantic cross-attention using MUSK patch embeddings c sem",
          "fusion_topology": "cross_attention",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "We fine-tune MUPAD on the ORION colorectal cancer dataset 46 (41 WSIs, 16 protein markers). H&E conditioning is applied through two complementary mechanisms: (1) spatial concatenation of the H&E VAE latent z struct with the noisy mIF latent at each timestep, providing a structural template; and (2) semantic cross-attention using MUSK patch embeddings c sem to modulate generation based on tissue pathology. Multiplex channels are grouped into 3-channel subsets and encoded with the VAE independently. The model is trained with flow-matching velocity prediction loss using BF16 mixed precision with HED-based H&E color augmentation. We evaluate Patch PCC and Slide-level PCC against GigaTIME 14 and HistoPlexer 47 .",
          "section_heading": "Application: H&E-to-mIF Translation",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            16,
            17,
            18,
            26
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/138",
            "#/texts/72",
            "#/texts/73",
            "#/texts/75",
            "#/texts/76",
            "#/texts/77",
            "#/texts/80"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003852::0009"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f61e030abbe9",
          "configuration_id": "config_2ed33cea1897",
          "route_label": "RNA-to-image generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "RNA-to-image generation",
          "source_object_verbatim": "bulk RNA-seq profile",
          "source_object_normalized": "bulk RNA-seq profile",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "331 pathway-level enrichment scores",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "RNA cross-attention stream in decoupled cross-modal attention",
          "fusion_topology": "cross_attention",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "RNA-to-image generation",
          "section_heading": "Ablation study of model design",
          "supporting_figure_or_table": "Figure 7a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            20,
            23
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/6",
            "#/texts/104",
            "#/texts/29",
            "#/texts/90"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003852::route_010"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003852::0010",
            "dense::full_2026-07-06__rec_003852::0002"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_686e999550fc"
    },
    {
      "model_id": "model_a95b39ba55bc",
      "model_name": "o1",
      "record_id": "full_2026-07-06__rec_001617",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4f442dcf4ed4",
      "paper_title": "Cell-o1: Training LLMs to Solve Single-Cell Reasoning Puzzles with Reinforcement Learning",
      "doi": "10.48550/arXiv.2506.02911",
      "paper_url": "https://doi.org/10.48550/arXiv.2506.02911",
      "route_count": 4,
      "configuration_count": 4,
      "family_counts": {
        "text_native_token_stream": 4
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 4
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell RNA sequencing batch with donor metadata",
        "single-cell RNA sequencing cell with donor metadata"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001617_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001617_76b0056d59c3/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of this work. Cell-o1 achieves state-of-the-art on the CellPuzzles task.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific schematic and results figure about the “CellPuzzles” task and a Cell-o1 training strategy.\n\nVisible panels and labels:\n- Left panel: “The CellPuzzles Task”\n  - Subpanel “(1) Previous LLM-based Annotation (Cell by Cell; No reasoning)”\n  - Input example lists marker genes: “MALAT1, RPS27, RPL10, RPL13, RPL41, EEF1A1, …”\n  - Output: “This is a CD8-positive, alpha-beta T cell.”\n  - Subpanel “(2) CellPuzzles (Batch-level; Reasoning required)”\n  - Input asks to jointly assign one unique cell type to each cell in a batch using expressed genes and donor context, and provide reasoning plus final answer.\n  - Output shows reasoning text and a small cell-to-cell assignment diagram with labels including:\n    - CD4-positive, alpha-beta T Cell\n    - CD8-positive, alpha-beta T Cell\n    - Capillary Endothelial Cell\n    - Plasma Cell\n    - Respiratory Basal Cell\n\n- Middle panel: “Cell-o1 Training Strategy”\n  - Subpanel “(1) Reasoning Distillation”\n  - Shows example reasoning boxes labeled “Reasoning 1” and “Reasoning 2” with corresponding “Answer 1” and “Answer 2” diagrams.\n  - Reasoning 1 states cell 9 shows strong immunoglobulin expression and is marked “Reject.”\n  - Reasoning 2 states there are 10 cells and 10 candidate types, so each type must match exactly one cell, marked “Accept.”\n  - Subpanel “(2) Reinforcement Learning”\n  - Flow diagram includes:\n    - Input Question\n    - Policy Model\n    - Reference Model\n    - Reward Function\n    - Advantage\n    - Group Computation\n    - Reward\n    - Update\n    - KL\n    - Cold Start\n\n- Right panel: “Results”\n  - Subpanel “(1) Cell-level Accuracy” shown as a radial/polar bar chart.\n  - Values visible include 0.42, 0.58, 0.65, 0.68, 0.26, 0.26, 0.14, 0.09.\n  - Cell-o1 is highlighted with “(+5.65%).”\n  - Subpanel “(2) Batch-level Accuracy” shown as another radial/polar bar chart.\n  - Values visible include 0.03, 0.14, 0.19, 0.33, and multiple 0.00/0.01 labels.\n  - Cell-o1 is highlighted with “(+73.05%).”\n\nBiological source objects:\n- Single-cell gene-expression examples and cell type annotations.\n- Cell types shown include T cells, endothelial cells, plasma cells, and respiratory basal cells.\n- Marker genes shown include MALAT1, RPS27, RPL10, RPL13, RPL41, EEF1A1, VIM, ADAMDEC1, CCL, and IGHA1/JCHAIN.\n\nModel interfaces and transformations:\n- The figure contrasts prior cell-by-cell LLM annotation with batch-level reasoning over multiple cells.\n- It depicts reasoning distillation where candidate reasoning-answer pairs are accepted or rejected.\n- It depicts reinforcement learning using a policy model, reference model, reward function, KL regularization, group computation, advantages, and rewards.\n\nFindings:\n- Cell-o1 appears to outperform comparison models in both cell-level and batch-level accuracy.\n- Reported improvement labels are +5.65% for cell-level accuracy and +73.05% for batch-level accuracy.\n- Legend compares Llama3.1-8B-Instruct, Qwen2.5-7B-Instruct, Llama3.3-70B-Instruct, GPT-4o-mini, GPT-4o, o3-mini, o1, and Cell-o1.",
        "page_no": 1,
        "sha256": "54b9bdf02b4a0ee126b6139a699bdb14ba90c40a9bcc4c6870cf609d32e78134",
        "pixel_width": 759,
        "pixel_height": 408,
        "crop_box": {
          "x": 0,
          "y": 0.36,
          "width": 0.35,
          "height": 0.64
        },
        "panel_label": "The CellPuzzles Task / (2) CellPuzzles (Batch-level; Reasoning required)",
        "visible_input_object": "A batch of cells from one donor with gene-expression markers and candidate cell types",
        "visible_model_interface": "Structured batch-level text prompt with donor context, constrained labels, and reasoning/final-answer instructions",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "Crops the left-lower CellPuzzles batch-level panel, keeping the prompt, donor-context instruction, and the assignment diagram while excluding the unrelated training-strategy and results panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_c9cb216e6226",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "the gene expression profile of a single cell from a specific donor",
          "actual_model_visible_form": "a single-cell prompt with donor context, candidate labels, and structured <think>/<answer> instructions"
        }
      ],
      "routes": [
        {
          "route_id": "route_c9cb216e6226",
          "configuration_id": "config_3470d1e24dfc",
          "route_label": "cell-level reasoning prompt (o1)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cell-level Reasoning Setup",
          "source_object_verbatim": "the gene expression profile of a single cell from a specific donor",
          "source_object_normalized": "single cell from a specific donor with top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing cell with donor metadata",
          "transformation_chain_verbatim": [
            "use the single cell's top expressed genes",
            "combine them with donor context",
            "provide a fixed candidate label set",
            "format the response with reasoning tags"
          ],
          "model_visible_form_verbatim": "a single-cell prompt with donor context, candidate labels, and structured <think>/<answer> instructions",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt template",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are an expert assistant specialized in cell type annotation. You will be given the gene expression profile of a single cell from a specific donor. The top expressed genes are listed in descending order. Use both gene expression and donor context to determine the correct cell type. You will also receive a list of candidate cell types-choose the one that best fits this cell . Include your detailed reasoning within <think> and </think> tags, and provide your final answer within <answer> and </answer> tags. The final answer should be a single string with exactly one cell type.",
          "section_heading": "C Cell-level vs. Batch-level Reasoning",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            21
          ],
          "doc_item_refs": [
            "#/texts/1015",
            "#/texts/1016",
            "#/texts/1017",
            "#/texts/1018",
            "#/texts/1019",
            "#/texts/1020"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_010"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ef7e448e3f6c",
          "configuration_id": "config_453d162792a5",
          "route_label": "batch-level reasoning prompt (o1)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Batch-level Reasoning Setup",
          "source_object_verbatim": "a batch of N cells from the same donor",
          "source_object_normalized": "batch of N cells from the same donor with ranked top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "combine with donor context",
            "present the candidate label set",
            "directly predict answers without reasoning traces"
          ],
          "model_visible_form_verbatim": "batch-level structured input without reasoning traces",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "direct prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are an expert assistant specialized in cell type annotation. You will be given a batch of N cells from the same donor, where each cell represents a unique cell type. For each cell, the top-expressed genes are provided in descending order of expression. Using both the gene expression data and donor information, determine the correct cell type for each cell. You will also receive a list of N candidate cell types, and each candidate must be assigned to exactly one cell. Ensure that you consider all cells and candidate types together, rather than annotating each cell individually. Include your detailed reasoning within <think> and </think> tags, and provide your final answer within <answer> and </answer> tags. The final answer should be a single string listing the assigned cell types in order, separated by ' | '.",
          "section_heading": "C Cell-level vs. Batch-level Reasoning",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/603",
            "#/texts/604",
            "#/texts/605"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_014"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f021d74d7c8a",
          "configuration_id": "config_c0e76edefa1d",
          "route_label": "open-ended QA prompt (o1)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Open-ended QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "provide gene expression profile and metadata",
            "remove the constrained label set",
            "ask for free-form cell type generation"
          ],
          "model_visible_form_verbatim": "free-form textual generation of a cell type name for each cell",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompting without constrained labels",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_020"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3bdf37b783eb",
          "configuration_id": "config_144bd7e41bc8",
          "route_label": "constrained QA prompt (o1)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Constrained QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "convert donor metadata into natural language context",
            "attach a predefined candidate label set",
            "require a single ordered answer string"
          ],
          "model_visible_form_verbatim": "a structured batch-level text prompt with candidate labels and ordered answer output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "predefined candidate label set with structured prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_026"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_c47c8bf166d6",
      "model_name": "o3-mini",
      "record_id": "full_2026-07-06__rec_001617",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4f442dcf4ed4",
      "paper_title": "Cell-o1: Training LLMs to Solve Single-Cell Reasoning Puzzles with Reinforcement Learning",
      "doi": "10.48550/arXiv.2506.02911",
      "paper_url": "https://doi.org/10.48550/arXiv.2506.02911",
      "route_count": 4,
      "configuration_count": 4,
      "family_counts": {
        "text_native_token_stream": 4
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 4
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell RNA sequencing batch with donor metadata",
        "single-cell RNA sequencing cell with donor metadata"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001617_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001617_76b0056d59c3/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of this work. Cell-o1 achieves state-of-the-art on the CellPuzzles task.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific schematic and results figure about the “CellPuzzles” task and a Cell-o1 training strategy.\n\nVisible panels and labels:\n- Left panel: “The CellPuzzles Task”\n  - Subpanel “(1) Previous LLM-based Annotation (Cell by Cell; No reasoning)”\n  - Input example lists marker genes: “MALAT1, RPS27, RPL10, RPL13, RPL41, EEF1A1, …”\n  - Output: “This is a CD8-positive, alpha-beta T cell.”\n  - Subpanel “(2) CellPuzzles (Batch-level; Reasoning required)”\n  - Input asks to jointly assign one unique cell type to each cell in a batch using expressed genes and donor context, and provide reasoning plus final answer.\n  - Output shows reasoning text and a small cell-to-cell assignment diagram with labels including:\n    - CD4-positive, alpha-beta T Cell\n    - CD8-positive, alpha-beta T Cell\n    - Capillary Endothelial Cell\n    - Plasma Cell\n    - Respiratory Basal Cell\n\n- Middle panel: “Cell-o1 Training Strategy”\n  - Subpanel “(1) Reasoning Distillation”\n  - Shows example reasoning boxes labeled “Reasoning 1” and “Reasoning 2” with corresponding “Answer 1” and “Answer 2” diagrams.\n  - Reasoning 1 states cell 9 shows strong immunoglobulin expression and is marked “Reject.”\n  - Reasoning 2 states there are 10 cells and 10 candidate types, so each type must match exactly one cell, marked “Accept.”\n  - Subpanel “(2) Reinforcement Learning”\n  - Flow diagram includes:\n    - Input Question\n    - Policy Model\n    - Reference Model\n    - Reward Function\n    - Advantage\n    - Group Computation\n    - Reward\n    - Update\n    - KL\n    - Cold Start\n\n- Right panel: “Results”\n  - Subpanel “(1) Cell-level Accuracy” shown as a radial/polar bar chart.\n  - Values visible include 0.42, 0.58, 0.65, 0.68, 0.26, 0.26, 0.14, 0.09.\n  - Cell-o1 is highlighted with “(+5.65%).”\n  - Subpanel “(2) Batch-level Accuracy” shown as another radial/polar bar chart.\n  - Values visible include 0.03, 0.14, 0.19, 0.33, and multiple 0.00/0.01 labels.\n  - Cell-o1 is highlighted with “(+73.05%).”\n\nBiological source objects:\n- Single-cell gene-expression examples and cell type annotations.\n- Cell types shown include T cells, endothelial cells, plasma cells, and respiratory basal cells.\n- Marker genes shown include MALAT1, RPS27, RPL10, RPL13, RPL41, EEF1A1, VIM, ADAMDEC1, CCL, and IGHA1/JCHAIN.\n\nModel interfaces and transformations:\n- The figure contrasts prior cell-by-cell LLM annotation with batch-level reasoning over multiple cells.\n- It depicts reasoning distillation where candidate reasoning-answer pairs are accepted or rejected.\n- It depicts reinforcement learning using a policy model, reference model, reward function, KL regularization, group computation, advantages, and rewards.\n\nFindings:\n- Cell-o1 appears to outperform comparison models in both cell-level and batch-level accuracy.\n- Reported improvement labels are +5.65% for cell-level accuracy and +73.05% for batch-level accuracy.\n- Legend compares Llama3.1-8B-Instruct, Qwen2.5-7B-Instruct, Llama3.3-70B-Instruct, GPT-4o-mini, GPT-4o, o3-mini, o1, and Cell-o1.",
        "page_no": 1,
        "sha256": "54b9bdf02b4a0ee126b6139a699bdb14ba90c40a9bcc4c6870cf609d32e78134",
        "pixel_width": 759,
        "pixel_height": 408,
        "crop_box": {
          "x": 0,
          "y": 0.34,
          "width": 0.33,
          "height": 0.66
        },
        "panel_label": "The CellPuzzles Task - Batch-level reasoning prompt",
        "visible_input_object": "A batch of cells with expressed genes plus donor context, framed as a joint unique-label assignment task",
        "visible_model_interface": "Structured text prompt with candidate labels and reasoning tags; the prompt text and batch-level instruction are the visible carrier",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop isolates the left lower CellPuzzles batch-level prompt and its immediate reasoning/assignment interface, which is the grounded input route needed. It excludes the results panel and the unrelated training-strategy panel while keeping the readable instruction text, arrows, and candidate-label context.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_781af690075d",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "the gene expression profile of a single cell from a specific donor",
          "actual_model_visible_form": "a single-cell prompt with donor context, candidate labels, and structured <think>/<answer> instructions"
        }
      ],
      "routes": [
        {
          "route_id": "route_781af690075d",
          "configuration_id": "config_6eca46c2607b",
          "route_label": "cell-level reasoning prompt (o3-mini)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cell-level Reasoning Setup",
          "source_object_verbatim": "the gene expression profile of a single cell from a specific donor",
          "source_object_normalized": "single cell from a specific donor with top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing cell with donor metadata",
          "transformation_chain_verbatim": [
            "use the single cell's top expressed genes",
            "combine them with donor context",
            "provide a fixed candidate label set",
            "format the response with reasoning tags"
          ],
          "model_visible_form_verbatim": "a single-cell prompt with donor context, candidate labels, and structured <think>/<answer> instructions",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt template",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are an expert assistant specialized in cell type annotation. You will be given the gene expression profile of a single cell from a specific donor. The top expressed genes are listed in descending order. Use both gene expression and donor context to determine the correct cell type. You will also receive a list of candidate cell types-choose the one that best fits this cell . Include your detailed reasoning within <think> and </think> tags, and provide your final answer within <answer> and </answer> tags. The final answer should be a single string with exactly one cell type.",
          "section_heading": "C Cell-level vs. Batch-level Reasoning",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            21
          ],
          "doc_item_refs": [
            "#/texts/1015",
            "#/texts/1016",
            "#/texts/1017",
            "#/texts/1018",
            "#/texts/1019",
            "#/texts/1020"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_009"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b3ec9d44728a",
          "configuration_id": "config_39814cbdf685",
          "route_label": "batch-level reasoning prompt (o3-mini)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Batch-level Reasoning Setup",
          "source_object_verbatim": "a batch of N cells from the same donor",
          "source_object_normalized": "batch of N cells from the same donor with ranked top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "combine with donor context",
            "present the candidate label set",
            "directly predict answers without reasoning traces"
          ],
          "model_visible_form_verbatim": "batch-level structured input without reasoning traces",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "direct prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are an expert assistant specialized in cell type annotation. You will be given a batch of N cells from the same donor, where each cell represents a unique cell type. For each cell, the top-expressed genes are provided in descending order of expression. Using both the gene expression data and donor information, determine the correct cell type for each cell. You will also receive a list of N candidate cell types, and each candidate must be assigned to exactly one cell. Ensure that you consider all cells and candidate types together, rather than annotating each cell individually. Include your detailed reasoning within <think> and </think> tags, and provide your final answer within <answer> and </answer> tags. The final answer should be a single string listing the assigned cell types in order, separated by ' | '.",
          "section_heading": "C Cell-level vs. Batch-level Reasoning",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/603",
            "#/texts/604",
            "#/texts/605"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_013"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_768e5f339019",
          "configuration_id": "config_1be8cc672eec",
          "route_label": "open-ended QA prompt (o3-mini)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Open-ended QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "provide gene expression profile and metadata",
            "remove the constrained label set",
            "ask for free-form cell type generation"
          ],
          "model_visible_form_verbatim": "free-form textual generation of a cell type name for each cell",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompting without constrained labels",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_019"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b74ebbc6e38c",
          "configuration_id": "config_ec5e7001d7e8",
          "route_label": "constrained QA prompt (o3-mini)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Constrained QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "convert donor metadata into natural language context",
            "attach a predefined candidate label set",
            "require a single ordered answer string"
          ],
          "model_visible_form_verbatim": "a structured batch-level text prompt with candidate labels and ordered answer output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "predefined candidate label set with structured prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_025"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_382b827ca741",
      "model_name": "o4-mini",
      "record_id": "full_2026-07-06__rec_003323",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_27fd0c39d9c3",
      "paper_title": "Teampath: Building multimodal pathology experts with reasoning ai copilots",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "inference"
      ],
      "fusion_topologies": [
        "concatenation"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003323_figure_005.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003323_ae216fea4cc3/figure_005.png",
        "figure_index": 5,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel scientific figure with panels labeled **a**, **b**, and **c**.\n\nPanel **a** shows schematic workflows involving histopathology image inputs and question/answer text boxes. The top workflow includes a **Verifier** module, an **Answer Corrector** module, a feedback loop, rejection/check icons, and a final **Corrected Answer** output. The lower workflow shows a **Reason Corrector** receiving a question, wrong reason, and proposed answer, then outputting **Correct Reason** and **Corrected Answer**. Small microscopy thumbnails and clinician/avatar icons are shown at the left.\n\nPanel **b** is a bar chart comparing **Expert** and **Corrector** performance across datasets labeled **PubMed**, **SocialPath**, **Atlas**, **EduContent**, **PathCLS**, and **Avg**. Green bars represent Expert and pale pink bars represent Corrector. The plot reports **P-value = 0.0004**, with Corrector bars generally higher than Expert bars.\n\nPanel **c** shows a large H&E-like histopathology microscopy image with dense purple/pink stained cells. Adjacent text boxes show a visual question-answering example about characteristic features of nuclei in the image. The question asks what features are observed in cell nuclei. Answer options include hyperchromatic round nuclei, small elongated nuclei, intranuclear inclusions, and large vesicular nuclei with prominent nucleoli. Color-coded annotations indicate green text as correct information and red text as wrong information. The example contrasts an expert answer/reason with a “TeamPath answer/reason,” discussing nuclear size, chromatin appearance, nucleoli prominence, and why answer A is selected instead of D.",
        "page_no": 10,
        "sha256": "41e1c150f31e495ff05f1f2610a84b400ae03eb6893def87605b20a3271b0e95",
        "pixel_width": 933,
        "pixel_height": 491,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.58,
          "height": 0.43
        },
        "panel_label": "a",
        "visible_input_object": "question and proposed answer text with the source histology thumbnail",
        "visible_model_interface": "Verifier prompt route from question-and-proposed-answer text into the verifier, with the arrow to Answer Corrector and corrected answer output",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop isolates the top-left self-verification workflow in panel a: source image plus question/proposed answer, the arrow into Verifier, the feedback-loop arrow toward Answer Corrector, and the corrected-answer output. It excludes the performance chart and the lower reasoning example.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_52285b2945fb",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "question and proposed solution",
          "actual_model_visible_form": "question-and-proposed-solution text"
        }
      ],
      "routes": [
        {
          "route_id": "route_52285b2945fb",
          "configuration_id": "config_f8beb1aa7e3d",
          "route_label": "o4-mini self-verifier text route",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "self-verification",
          "source_object_verbatim": "question and proposed solution",
          "source_object_normalized": "question and proposed solution",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "concatenation",
            "tokenization",
            "language encoding"
          ],
          "model_visible_form_verbatim": "question-and-proposed-solution text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed through the verifier prompt",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are given a QUESTION and a PROPOSED SOLUTION.",
          "section_heading": "A. Prompt list",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            26,
            27
          ],
          "doc_item_refs": [
            "#/texts/946",
            "#/texts/947",
            "#/texts/948",
            "#/texts/949",
            "#/texts/950",
            "#/texts/951",
            "#/texts/952",
            "#/texts/953"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003323::route_009"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_f924e02d892f",
      "model_name": "OCellus",
      "record_id": "update_2026-08-09__rec_000138",
      "collection_batch_id": "update_2026-08-09",
      "collection_date": "2026-08-09",
      "review_iteration": "2026-08-09",
      "study_id": "study_6673da673ba0",
      "paper_title": "OCellus: A Language-Model Framework for Single-Cell, Spatial, and Perturbation Biology with Natural-Language Reasoning",
      "doi": "",
      "paper_url": "",
      "route_count": 14,
      "configuration_count": 13,
      "family_counts": {
        "text_native_token_stream": 14
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 11,
        "structured_biological_prompt_or_task_scaffold": 3
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "graph/network",
        "single-cell perturbation transcriptomics",
        "single-cell transcriptomics",
        "spatial transcriptomics",
        "text",
        "transcriptomics"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "other_explicit",
        "prefix",
        "shared_latent_alignment",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/update_2026_08_09_rec_000138_figure_001.png",
        "source_path": "data/living_catalog_updates/update_2026-08-09/11_docling_vlm/profiles/figures/update_2026_08_09_rec_000138_796e32b2faec/figure_001.png",
        "figure_index": 1,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nPanel: labeled `a`.\n\nTitle: `OCellus: Training a virtual cell foundation model`.\n\nVisible structure:\n- Left dashed box: `Training from multi-modal single-cell data`.\n- Input modalities shown with icons and labels:\n  - `scRNA-seq`\n  - `scPerturb-seq`\n  - `Spatial transcriptomics`\n- These modalities are grouped by a brace and fed into an `OCellus Transformer Encoder`.\n- Output from the encoder is labeled `Multi-modal Cell Embedding`, shown as two vertical embedding/vector-like blocks.\n\nRight dashed box: `OCellus pretraining and alignment (based on Qwen LLM)`.\n- The multi-modal cell embedding flows into `Data integration`.\n- Data integration connects to a `Qwen LLM model`.\n- The Qwen LLM model connects to a `Tiny LLM`.\n- The Tiny LLM connects to the final `Aligned OCellus Base Model`, represented by an atom-like icon.\n\nBiological source objects:\n- Single-cell RNA sequencing data.\n- Perturbation single-cell sequencing data.\n- Spatial transcriptomics data.\n- Cell embeddings derived from multi-modal single-cell data.\n\nTransformations/model interfaces:\n- Multi-modal biological inputs are encoded by the OCellus Transformer Encoder.\n- Encoded outputs become multi-modal cell embeddings.\n- Embeddings are integrated with a Qwen LLM-based pretraining/alignment pipeline.\n- A Tiny LLM is used before producing the aligned OCellus base model.\n\nFindings/claim shown:\n- The figure schematizes the training workflow for OCellus, a virtual cell foundation model integrating multi-modal single-cell data with an LLM-based alignment pipeline.",
        "page_no": 7,
        "sha256": "166ae789ff33a3c17ee3e0cf2ea04c5dad6e51e92fb3d118ba4b692e70b593f1",
        "pixel_width": 896,
        "pixel_height": 214,
        "crop_box": {
          "x": 0.01,
          "y": 0.15,
          "width": 0.58,
          "height": 0.85
        },
        "panel_label": "a",
        "visible_input_object": "scRNA-seq, scPerturb-seq, spatial transcriptomics",
        "visible_model_interface": "OCellus Transformer Encoder feeding Multi-modal Cell Embedding",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "Crops the left pretraining pathway only: the three source modalities, the brace into the OCellus Transformer Encoder, and the resulting multi-modal cell embedding. This preserves a complete grounded input-to-interface route while excluding the downstream Qwen/Tiny LLM alignment chain.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_a9afe0a24336",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "single-cell RNA-seq data",
          "actual_model_visible_form": "tokenized cell sentence"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_f912c63c815f",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene pairs from STRING v12",
          "actual_model_visible_form": "gene-pair text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_a9afe0a24336",
          "configuration_id": "config_6756ccde762f",
          "route_label": "scRNA-seq to OCellus pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Stage 1 (pretraining) trains on seven representation-learning tasks",
          "source_object_verbatim": "single-cell RNA-seq data",
          "source_object_normalized": "single-cell RNA sequencing data",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "tokenized",
            "passed through a transformer encoder",
            "aligned to a Qwen LLM backbone"
          ],
          "model_visible_form_verbatim": "tokenized cell sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "Qwen3.5-9B backbone",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "The four-stage training pipeline executes on four NVIDIA A800-80GB GPUs with NVLink interconnect. Stage 1 (pretraining) trains on seven representation-learning tasks -cell-sentence prediction, gene-network node and link masking, and four spatial variants -for three epochs at learning rate 5 × 10⁻⁵ with cosine annealing, warmup ratio 0.1, weight decay 0.01, and an effective batch size of sixty-four. Stage 2 merges the pretrained adapters into the base weights via standard PEFT merging, producing the OCellus-Pretrain checkpoint. Stage 3 (multi-task finetuning) trains on twenty-six tasks -sixteen single-cell and ten spatial -covering perturbation prediction, gene-interaction classification (synthetic lethality, phenotypic, general, dosage),",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper presents this inside a shared multimodal encoder rather than an scRNA-seq-only branch.",
          "pages": [
            6,
            7,
            8,
            41,
            42,
            51
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/2",
            "#/texts/139",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/148",
            "#/texts/149",
            "#/texts/799",
            "#/texts/802",
            "#/texts/803",
            "#/texts/804",
            "#/texts/805",
            "#/texts/806",
            "#/texts/807",
            "#/texts/808"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_001"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f8991379a4c4",
          "configuration_id": "config_7884df2180b1",
          "route_label": "scRNA-seq to OCellus multi-task fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Stage 3 (multi-task finetuning) trains on twenty-six tasks",
          "source_object_verbatim": "single-cell RNA-seq data",
          "source_object_normalized": "single-cell RNA sequencing data",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "tokenized",
            "passed through a transformer encoder",
            "aligned to a Qwen LLM backbone"
          ],
          "model_visible_form_verbatim": "tokenized cell sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "Qwen3.5-9B backbone",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Stage 3 (multi-task finetuning) trains on twenty-six tasks",
          "section_heading": "Methods",
          "supporting_figure_or_table": "Supplementary Figure S1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes Stage 3 as a shared multi-task fine-tuning phase, not a standalone scRNA-seq branch.",
          "pages": [
            6,
            7,
            8,
            41,
            42
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/2",
            "#/texts/139",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/148",
            "#/texts/149",
            "#/texts/799",
            "#/texts/802",
            "#/texts/803",
            "#/texts/804",
            "#/texts/805",
            "#/texts/806",
            "#/texts/807",
            "#/texts/808"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_002"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_773640cd4c26",
          "configuration_id": "config_44fb4920b1fe",
          "route_label": "scPerturb-seq to OCellus multi-task fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "perturbation screens",
          "source_object_verbatim": "scPerturb-seq data",
          "source_object_normalized": "perturbation single-cell sequencing data",
          "source_modality_normalized": "single-cell perturbation transcriptomics",
          "transformation_chain_verbatim": [
            "tokenized",
            "passed through a transformer encoder",
            "aligned to a Qwen LLM backbone"
          ],
          "model_visible_form_verbatim": "tokenized cell sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "Qwen3.5-9B backbone",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "scRNA-seq, scPerturb-seq, Spatial transcriptomics",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper names scPerturb-seq as an input modality but does not spell out a standalone downstream route for it beyond the shared training path.",
          "pages": [
            6,
            7,
            8,
            41,
            42
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/2",
            "#/texts/139",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/148",
            "#/texts/149",
            "#/texts/799",
            "#/texts/802",
            "#/texts/803",
            "#/texts/804",
            "#/texts/805",
            "#/texts/806",
            "#/texts/807",
            "#/texts/808"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_003"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d07eb8c2a0c4",
          "configuration_id": "config_747b9838dbdb",
          "route_label": "spatial transcriptomics to OCellus pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "four spatial variants",
          "source_object_verbatim": "spatial transcriptomics data",
          "source_object_normalized": "spatial transcriptomics data",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "tokenized",
            "passed through a transformer encoder",
            "aligned to a Qwen LLM backbone"
          ],
          "model_visible_form_verbatim": "tokenized spatial cell sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "Qwen3.5-9B backbone",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Beyond the empirical results above, an underrecognized advantage of OCellus's training 3 pipeline is the integration of in-house spatial transcriptomics data alongside public sources. 4 Public spatial-transcriptomics training corpora remain scarce relative to single-cell atlases: 5 cellxgene and SToCorpus-88M together comprise the bulk of public spatial data, and most 6 existing spatial foundation models (SToFM, NicheFormer, HEIST) draw from the same narrow 7 pool. The salusSTS in-house 0.1-micrometer subcellular spatial transcriptomics platform 8 contributes approximately 17 percent of OCellus's current training samples (Supplementary 9 Figure S1c) and provides twenty million mouse-organ tissue cells with multi-scale bin40 / 10 bin100 / bin400 outputs at subcellular resolution. The strategic advantage is not data volume 11 alone but flexibility: because we generate spatial training data in-house, we can customize 12 training-data composition for new spatial protocols (Stereo-seq, Visium HD, Xenium Prime, 13 salusSTS), organ targets, or coordinate systems without waiting for public releases, and we can 14 deliberately balance under-represented tissues or technologies in the training mix. This is 15 particularly relevant given the rapid technological evolution of spatial transcriptomics, where 16 new platforms continuously generate data regimes that public atlases have not yet absorbed. 17 OCellus-Agent represents a conceptual shift in how virtual-cell models interact with researchers. 18 Current models require users to navigate complex computational pipelines -preprocessing data, 19 selecting parameters, running inference, and interpreting outputs -whereas OCellus-Agent 20 collapses these steps into a natural-language dialogue in which a researcher describes what they 21 want to know, and the trained Planner decomposes the query into an executable directed acyclic 22",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper folds spatial data into a shared multimodal pretraining path rather than a spatial-only branch.",
          "pages": [
            6,
            7,
            10,
            39,
            40
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/3",
            "#/texts/139",
            "#/texts/758",
            "#/texts/761",
            "#/texts/762",
            "#/texts/763",
            "#/texts/764"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_004"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_aec0f02313df",
          "configuration_id": "config_0398daef2be0",
          "route_label": "spatial transcriptomics to OCellus with EvenClock",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "spatial tasks",
          "source_object_verbatim": "spatial transcriptomics data",
          "source_object_normalized": "spatial transcriptomics data",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "compute the median nearest-neighbor distance",
            "build a cKDTree on the cell coordinates",
            "assign neighbors to eighteen sectors",
            "concatenate neighboring genes into a text prefix",
            "append the EvenClock prefix to the center cell's ranked-gene sentence"
          ],
          "model_visible_form_verbatim": "ranked-gene cell sentence plus EvenClock text prefix",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "EvenClock spatial-encoding module",
          "fusion_topology": "prefix",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "EvenClock discretizes the continuous two-dimensional plane into a clockface of six directions and three distance rings",
          "section_heading": "EvenClock spatial encoding",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            6,
            18,
            19
          ],
          "doc_item_refs": [
            "#/texts/117",
            "#/texts/120",
            "#/texts/121",
            "#/texts/122",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/339",
            "#/texts/340",
            "#/texts/341",
            "#/texts/342",
            "#/texts/343",
            "#/texts/344",
            "#/texts/345",
            "#/texts/346"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_005"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0068"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e8b754d26728",
          "configuration_id": "config_29bd08d4d609",
          "route_label": "mouse ortholog-gene sentence to OCellus cross-species cell-type annotation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "cross-species cell-type annotation",
          "source_object_verbatim": "mouse cell's top-ranked genes on the shared ortholog panel of approximately 250 genes",
          "source_object_normalized": "mouse ortholog gene list",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "top-ranked genes are harmonized onto the shared ortholog panel",
            "presented as a ranked gene sentence"
          ],
          "model_visible_form_verbatim": "ranked gene sentence on the shared ortholog panel",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "shared Qwen3.5-9B backbone with task-specific LoRA adapter",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "mouse cell's top-ranked genes on the shared ortholog panel of approximately 250 genes",
          "section_heading": "Cross-domain reasoning",
          "supporting_figure_or_table": "Figure 7",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            6,
            11,
            12,
            13,
            14,
            15,
            29,
            30,
            41,
            42
          ],
          "doc_item_refs": [
            "#/texts/117",
            "#/texts/120",
            "#/texts/121",
            "#/texts/122",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_010"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0106"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_017cdda5615a",
          "configuration_id": "config_13f52631e201",
          "route_label": "embryonic gene sentence to OCellus developmental-stage prediction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "developmental-stage prediction across mouse and human embryogenesis",
          "source_object_verbatim": "embryonic cell gene-expression profile",
          "source_object_normalized": "embryonic cell gene-expression profile",
          "source_modality_normalized": "transcriptomics",
          "transformation_chain_verbatim": [
            "ranked genes are converted into a cell sentence",
            "fed to the language model"
          ],
          "model_visible_form_verbatim": "ranked-gene cell sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "shared Qwen3.5-9B backbone with task-specific LoRA adapter",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "developmental-stage prediction across mouse embryogenesis (E9.5 -E16.5) and human embryogenesis (CS12 -CS23)",
          "section_heading": "Cross-domain reasoning",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            30,
            31
          ],
          "doc_item_refs": [
            "#/texts/594",
            "#/texts/595",
            "#/texts/596",
            "#/texts/597",
            "#/texts/598",
            "#/texts/599",
            "#/texts/600",
            "#/texts/601"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_011"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f912c63c815f",
          "configuration_id": "config_24bb096ba238",
          "route_label": "gene pairs from STRING v12 to OCellus binary PPI classification",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "binary protein-protein interaction classification on STRING v12",
          "source_object_verbatim": "gene pairs from STRING v12",
          "source_object_normalized": "gene pairs",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "gene pairs are formatted as a binary classification prompt"
          ],
          "model_visible_form_verbatim": "gene-pair text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "shared Qwen3.5-9B backbone with task-specific LoRA adapter",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "binary protein -protein interaction classification on STRING v12",
          "section_heading": "Cross-domain reasoning",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            24,
            30,
            31,
            34,
            35
          ],
          "doc_item_refs": [
            "#/pictures/7",
            "#/texts/461",
            "#/texts/464",
            "#/texts/465",
            "#/texts/466",
            "#/texts/467",
            "#/texts/468",
            "#/texts/469",
            "#/texts/594",
            "#/texts/595",
            "#/texts/596",
            "#/texts/597",
            "#/texts/598",
            "#/texts/599",
            "#/texts/600",
            "#/texts/601"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_012"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0045"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9daaabb1fd5e",
          "configuration_id": "config_bdfe195cea6f",
          "route_label": "gene name to OCellus gene-to-disease association",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "gene-disease association",
          "source_object_verbatim": "gene name",
          "source_object_normalized": "gene name",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "gene identity is encoded as a text prompt"
          ],
          "model_visible_form_verbatim": "gene-name text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "shared Qwen3.5-9B backbone with task-specific LoRA adapter",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "forward direction (gene to disease)",
          "section_heading": "Cross-domain reasoning",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            30,
            31
          ],
          "doc_item_refs": [
            "#/texts/594",
            "#/texts/595",
            "#/texts/596",
            "#/texts/597",
            "#/texts/598",
            "#/texts/599",
            "#/texts/600",
            "#/texts/601"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_013"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b68e65fecc46",
          "configuration_id": "config_bdfe195cea6f",
          "route_label": "disease label to OCellus disease-to-gene association",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "gene-disease association",
          "source_object_verbatim": "disease name",
          "source_object_normalized": "disease name",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "disease identity is encoded as a text prompt"
          ],
          "model_visible_form_verbatim": "disease-name text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "shared Qwen3.5-9B backbone with task-specific LoRA adapter",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "reverse direction (disease to gene)",
          "section_heading": "Cross-domain reasoning",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            30,
            31
          ],
          "doc_item_refs": [
            "#/texts/594",
            "#/texts/595",
            "#/texts/596",
            "#/texts/597",
            "#/texts/598",
            "#/texts/599",
            "#/texts/600",
            "#/texts/601"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_014"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6222a6dade31",
          "configuration_id": "config_c1b90d335ad3",
          "route_label": "EvenClock neighborhood text to OCellus spatial-neighborhood prediction",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "spatial-neighborhood task",
          "source_object_verbatim": "a center cell and EvenClock-encoded neighbors in one direction",
          "source_object_normalized": "spatial neighborhood context",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "center-cell neighborhood is encoded with EvenClock",
            "neighbor gene lists are concatenated into a text prefix"
          ],
          "model_visible_form_verbatim": "ranked-gene cell sentence plus EvenClock text prefix",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "EvenClock spatial-encoding module",
          "fusion_topology": "prefix",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "given a center cell and EvenClock-encoded neighbors in one direction",
          "section_heading": "EvenClock: text-only spatial reasoning",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            13,
            18,
            19,
            20,
            21
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/pictures/4",
            "#/texts/339",
            "#/texts/340",
            "#/texts/341",
            "#/texts/342",
            "#/texts/343",
            "#/texts/344",
            "#/texts/345",
            "#/texts/346",
            "#/texts/395",
            "#/texts/396",
            "#/texts/397",
            "#/texts/398",
            "#/texts/399",
            "#/texts/400"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_015"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0008"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_92d1fc7a989d",
          "configuration_id": "config_7ca57b1512ce",
          "route_label": "adjacent-cell gene profiles to OCellus spatial cell-chat classification",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "spatial-cellchat task",
          "source_object_verbatim": "gene profiles of two spatially adjacent cells",
          "source_object_normalized": "pair of adjacent spatial cell gene profiles",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "paired adjacent-cell gene profiles are compared for ligand-receptor communication"
          ],
          "model_visible_form_verbatim": "paired cell-gene profiles",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "EvenClock spatial-encoding module",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "given the gene profiles of two spatially adjacent cells and asked whether they engage in active ligand-receptor-mediated communication",
          "section_heading": "EvenClock: text-only spatial reasoning",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            18,
            19,
            20,
            21
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/texts/339",
            "#/texts/340",
            "#/texts/341",
            "#/texts/342",
            "#/texts/343",
            "#/texts/344",
            "#/texts/345",
            "#/texts/346",
            "#/texts/395",
            "#/texts/396",
            "#/texts/397",
            "#/texts/398",
            "#/texts/399",
            "#/texts/400",
            "#/texts/403"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_016"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0018"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_51ee68b9183a",
          "configuration_id": "config_a30607d9f434",
          "route_label": "marker-gene list to OCellus cell-marker identification",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "cell marker identification",
          "source_object_verbatim": "Markers MPZ, PAX3, PLP1, POSTN, SOX10, TFAP2A",
          "source_object_normalized": "marker-gene list",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "marker genes are presented as a ranked gene list"
          ],
          "model_visible_form_verbatim": "ranked-gene cell sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "shared Qwen3.5-9B backbone with task-specific LoRA adapter",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Cell marker identification | Markers MPZ, PAX3, PLP1, POSTN, SOX10, TFAP2A",
          "section_heading": "Multi-task competence across twenty-two biological tasks",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            11,
            12,
            13,
            14,
            15,
            41,
            42
          ],
          "doc_item_refs": [
            "#/texts/221",
            "#/texts/222",
            "#/texts/223",
            "#/texts/224",
            "#/texts/225",
            "#/texts/226",
            "#/texts/227",
            "#/texts/228"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_017"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0093"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ee2d489e0fab",
          "configuration_id": "config_3e3af05d3b15",
          "route_label": "protein-protein interaction databases to OCellus pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "multi-modal biological data (single-cell RNA-seq, spatial transcriptomics, perturbation screens, protein-protein interaction databases) is tokenized and passed through a transformer encoder, then aligned to a Qwen large-language-model backbone",
          "source_object_verbatim": "protein-protein interaction databases",
          "source_object_normalized": "protein-protein interaction databases",
          "source_modality_normalized": "graph/network",
          "transformation_chain_verbatim": [
            "protein-protein interaction databases",
            "tokenized and passed through a transformer encoder",
            "aligned to a Qwen large-language-model backbone"
          ],
          "model_visible_form_verbatim": "tokenized and passed through a transformer encoder",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "tokenized and passed through a transformer encoder, then aligned to a Qwen large-language-model backbone",
          "fusion_topology": "other_explicit",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "protein-protein interaction databases) is tokenized and passed through a transformer encoder",
          "section_heading": "The OCellus framework",
          "supporting_figure_or_table": "Figure 1a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8
          ],
          "doc_item_refs": [
            "#/texts/169"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0007"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_2bfb9536cf22"
    },
    {
      "model_id": "model_f12e33a1e764",
      "model_name": "OCellus-Agent",
      "record_id": "update_2026-08-09__rec_000138",
      "collection_batch_id": "update_2026-08-09",
      "collection_date": "2026-08-09",
      "review_iteration": "2026-08-09",
      "study_id": "study_6673da673ba0",
      "paper_title": "OCellus: A Language-Model Framework for Single-Cell, Spatial, and Perturbation Biology with Natural-Language Reasoning",
      "doi": "",
      "paper_url": "",
      "route_count": 2,
      "configuration_count": 2,
      "family_counts": {
        "text_native_token_stream": 2
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 2
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question"
      ],
      "primary_subtype": "plain_language_prompt_or_question",
      "modalities": [
        "natural-language query"
      ],
      "lifecycle_phases": [
        "inference"
      ],
      "fusion_topologies": [
        "retrieval_or_tool_context"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/update_2026_08_09_rec_000138_figure_003.png",
        "source_path": "data/living_catalog_updates/update_2026-08-09/11_docling_vlm/profiles/figures/update_2026_08_09_rec_000138_796e32b2faec/figure_003.png",
        "figure_index": 3,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic scientific workflow for “OCellus” applied to single-cell and spatial omics data.\n\nVisible structure:\n- Left panel shows input modalities feeding into an “OCellus Transformer Encoder”:\n  - `scRNA-seq` represented by colored cell icons.\n  - `scPerturb-seq` represented by a perturbation/gene-expression style grid.\n  - `Spatial Transcriptomics` represented by a tissue image.\n- The encoder outputs:\n  - `multi-modal cell embedding`\n  - `multi-model data-specific embedding`\n\nMiddle panel shows downstream task modules or prediction targets:\n- `Cell State`\n- `Drug Sensitivity`\n- `Perturbation`\n- `Spatial Neighbor`\n- `Spatial Deconvolution`\n- `Spatial Cell-Cell Communication`\n- `Spatial Imputation`\n\nOutputs illustrated on the right side of the middle panel include:\n- `Predicted spatial structure`\n- `Predicted perturbation profiles`\n- `Gene map`\n\nRight panel depicts a user interaction with OCellus:\n- User asks: “In this embryo brain slice, where would TP53 knockout have the strongest effect?”\n- OCellus returns a text response describing spatially heterogeneous predicted effects.\n- The visible finding states that ventricular zone progenitors show the strongest effect, with listed gene changes including `MDM2 -2.5 log2FC`, `CDKN1A -2.1`, `SERPINE1 +1.8`, while cortical plate neurons are largely resistant.\n- The response also mentions compensatory `ATM-CHEK2-RAD51` activation in the ventricular zone, suggesting region-specific DNA damage checkpoint recruitment.\n\nBiological/source objects visible:\n- Single cells, perturbation assay data, spatial transcriptomics tissue image, embryo brain slice context, predicted spatial maps, cell-neighborhood/network diagrams, and gene-expression/perturbation profiles.\n\nModel/interface elements:\n- Central transformer encoder labeled `OCellus Transformer Encoder`.\n- OCellus icon used for prediction and chat-style response.\n- A natural-language query interface connecting user questions to spatial perturbation predictions.",
        "page_no": 7,
        "sha256": "3d72c1a553a07149765ac7c037a323971493d043d656068810ced0a3be6692c5",
        "pixel_width": 890,
        "pixel_height": 459,
        "crop_box": {
          "x": 0.42,
          "y": 0.0,
          "width": 0.58,
          "height": 1.0
        },
        "panel_label": "left",
        "visible_input_object": "scRNA-seq, scPerturb-seq, and spatial transcriptomics feeding the OCellus Transformer Encoder",
        "visible_model_interface": "OCellus Transformer Encoder with multi-modal cell embedding and data-specific embedding outputs",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop isolates the grounded input route and immediate transformation interface: the three source omics modalities, arrows into the OCellus Transformer Encoder, and the embedding outputs. It excludes downstream task/output panels and the chat response, which are not needed to understand the input-to-model carrier path.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper__F7_adjusted_exact_preview_pass"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_3caf448da69c",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "TP53 KO in K562, what changes?",
          "actual_model_visible_form": "natural-language query"
        }
      ],
      "routes": [
        {
          "route_id": "route_3caf448da69c",
          "configuration_id": "config_3f59b83d864d",
          "route_label": "natural-language perturbation query to OCellus-Agent",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "perturbation with explanation",
          "source_object_verbatim": "TP53 KO in K562, what changes?",
          "source_object_normalized": "TP53 knockout in K562 query",
          "source_modality_normalized": "natural-language query",
          "transformation_chain_verbatim": [
            "Planner LoRA decomposes natural-language queries into executable directed acyclic graphs",
            "rule-based Router dispatches each step to the appropriate expert role",
            "three-layer Verifier provides biological plausibility checks"
          ],
          "model_visible_form_verbatim": "natural-language query",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "Coordinator containing a trained Planner LoRA, a rule-based Router, and a three-layer Verifier",
          "fusion_topology": "retrieval_or_tool_context",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The first -perturbation with explanation -takes the query 'TP53 KO in K562, what changes?'",
          "section_heading": "Agent system",
          "supporting_figure_or_table": "Figure 8",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            31,
            32,
            34,
            35
          ],
          "doc_item_refs": [
            "#/texts/607",
            "#/texts/610",
            "#/texts/672",
            "#/texts/675",
            "#/texts/676",
            "#/texts/677",
            "#/texts/678",
            "#/texts/679",
            "#/texts/680",
            "#/texts/681"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_007"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0074"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_276ab88653d9",
          "configuration_id": "config_59f9013b02f3",
          "route_label": "natural-language gene-function query to OCellus-Agent",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "gene-function question answering",
          "source_object_verbatim": "BRCA1 functions and interactors?",
          "source_object_normalized": "BRCA1 functions and interactors query",
          "source_modality_normalized": "natural-language query",
          "transformation_chain_verbatim": [
            "STRING protein-interaction query",
            "gene-embedding step",
            "explanation step"
          ],
          "model_visible_form_verbatim": "natural-language query",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "Coordinator containing a trained Planner LoRA, a rule-based Router, and a three-layer Verifier",
          "fusion_topology": "retrieval_or_tool_context",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The third workflow -gene-function question answering -takes 'BRCA1 functions and interactors?'",
          "section_heading": "Agent system",
          "supporting_figure_or_table": "Figure 8",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            31,
            32,
            34,
            35
          ],
          "doc_item_refs": [
            "#/texts/607",
            "#/texts/610",
            "#/texts/672",
            "#/texts/675",
            "#/texts/676",
            "#/texts/677",
            "#/texts/678",
            "#/texts/679",
            "#/texts/680",
            "#/texts/681"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_009"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0076"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_357f1bf88b0b"
    },
    {
      "model_id": "model_1bd0fe9134d7",
      "model_name": "OCellus-GNN",
      "record_id": "update_2026-08-09__rec_000138",
      "collection_batch_id": "update_2026-08-09",
      "collection_date": "2026-08-09",
      "review_iteration": "2026-08-09",
      "study_id": "study_6673da673ba0",
      "paper_title": "OCellus: A Language-Model Framework for Single-Cell, Spatial, and Perturbation Biology with Natural-Language Reasoning",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 1
      },
      "subtype_counts": {
        "pooled_or_aggregated_embedding": 1
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "pooled_or_aggregated_embedding"
      ],
      "primary_subtype": "pooled_or_aggregated_embedding",
      "modalities": [
        "gene symbol text"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "side_or_generative_conditioning"
      ],
      "text_roles": [
        "biological_payload"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/update_2026_08_09_rec_000138_figure_009.png",
        "source_path": "data/living_catalog_updates/update_2026-08-09/11_docling_vlm/profiles/figures/update_2026_08_09_rec_000138_796e32b2faec/figure_009.png",
        "figure_index": 9,
        "caption": "Figure 6 . Interpretable perturbation prediction: TP53 and BRCA1 knockout case studies, and the TP53 PPI response network. (a) TP53-knockout prediction case study. The model produces quantitative log2 fold-change values for the top response genes (MDM2, CDKN1A, SERPINE1) under LoRA-enabled mode; the explainer role (LoRA-disabled) then provides a natural-language interpretation linking the predicted downregulation of MDM2 and CDKN1A -direct transcriptional targets of TP53 -and the compensatory upregulation of SERPINE1 to TP53's known role as a tumor suppressor regulating cell -cycle arrest, apoptosis, and senescence-associated secretory activation. (b) BRCA1-knockout prediction case study. The model predicts upregulation of MDM2, RAD51, and FANCD2; the explainer interprets RAD51 and FANCD2 upregulation as compensatory activation of homologous-recombination repair mediated by the BRCA2-PALB2 axis, which functions independently of BRCA1. (c) Protein-interaction response network centered on TP53 (STRING v12, confidence ≥ 0.70): dark-blue node marks the knocked-out gene (TP53), grey nodes are direct PPI neighbors, and green nodes are predicted response genes identified across the TP53 and BRCA1 case studies. The network visualization shows that the response genes are not randomly distributed but cluster around direct PPI neighbors of TP53, consistent with the perturbation signal propagating through physical protein contacts.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel biological/scientific figure with panels labeled **a**, **b**, and **c**.\n\nPanel **a** shows a schematic of **gene knockout** leading to a **TP53 KO prediction**. A DNA/gene knockout icon points to a matrix-style prediction output. The matrix has response genes labeled **MDM2**, **CDKN1A**, and **SERPINE1** on the rows, and predictor/context genes labeled **MDM2** and **CDKN1A** on the columns. A color bar indicates **log2FC**. A callout states that an **interpretative model provides context-specific response gene prediction**.\n\nPanel **b** shows a similar gene knockout schematic leading to a **BRCA1 KO prediction**. The output is a bar chart with response genes **RAD51** and **FANCD2**, both labeled “up,” with a vertical **log2FC** axis. A callout says the prediction matches a **differential context explanation**.\n\nPanel **c** shows a protein-protein interaction/network diagram centered on the knockout gene **TP53**, colored teal. Surrounding nodes are **PPI neighbors** in gray and **response genes** in green. The legend defines teal as **Knockout gene**, gray as **PPI neighbors**, and green as **Response genes**, described as identified across panels a and b and other context-specific predictors. Visible network labels include **TP53**, **MDM2**, **RAD51**, **RAD52**, **FANCD2**, **TGF-BR**, **RPAR1**, **CLFX**, **COB3**, **CARO**, **RN1S1**, **RPNG**, **PURG9**, **SEEB1**, **MPRS1**, **HME3**, **TWRV2**, **CCT2**, and others.\n\nOverall, the figure illustrates an interpretable model for predicting gene-expression responses after knockout perturbations, using context-specific predictor-response relationships and mapping response genes onto a PPI network around **TP53**.",
        "page_no": 28,
        "sha256": "865a1a34b6128313a13f02a95aa37f896d56eb9e106627914a6b306fb63f373e",
        "pixel_width": 846,
        "pixel_height": 444,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 0.476,
          "height": 0.441
        },
        "panel_label": "a",
        "visible_input_object": "Gene knockout",
        "visible_model_interface": "TP53 KO prediction response-gene log2FC matrix with interpretative callout",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop isolates panel a and keeps the knockout source icon, arrow, response-gene matrix, log2FC labeling, and the interpretive callout needed to understand the input-to-prediction route. It excludes panels b and c and other nonessential content.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "pooled_or_aggregated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_0f9204eed8da",
          "example_input": "{gene/cell/patch embeddings}",
          "example_carrier": "mean/attention pool = one compact vector",
          "example_interface": "aggregator → generator",
          "actual_source": "individual gene symbols",
          "actual_model_visible_form": "4,096-dimensional gene embeddings"
        }
      ],
      "routes": [
        {
          "route_id": "route_0f9204eed8da",
          "configuration_id": "config_bc111698a5df",
          "route_label": "gene symbols to OCellus-GNN perturbation prediction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Replogle 2022 benchmark",
          "source_object_verbatim": "individual gene symbols",
          "source_object_normalized": "gene symbols",
          "source_modality_normalized": "gene symbol text",
          "transformation_chain_verbatim": [
            "passing individual gene symbols through the model",
            "averaging last-layer hidden states",
            "mapping frozen OCellus gene embeddings to the working dimension",
            "three graph-convolution layers",
            "MLP head produces continuous log2 fold-change predictions"
          ],
          "model_visible_form_verbatim": "4,096-dimensional gene embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "plug-in graph neural network over the STRING v12 protein-interaction graph",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "passing individual gene symbols through the model and averaging last-layer hidden states",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            6,
            46,
            47,
            48
          ],
          "doc_item_refs": [
            "#/texts/117",
            "#/texts/120",
            "#/texts/121",
            "#/texts/122",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/890",
            "#/texts/891",
            "#/texts/892",
            "#/texts/893",
            "#/texts/894",
            "#/texts/895",
            "#/texts/896",
            "#/texts/897"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000138::route_006"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000138::0105"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_998d40c02fca"
    },
    {
      "model_id": "model_5f2f0ba1126a",
      "model_name": "OKR-CELL",
      "record_id": "full_2026-07-06__rec_001277",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_55720b9d7702",
      "paper_title": "WITHDRAWN: OKR-Cell: Open World Knowledge Aided Single-Cell Foundation Model with Robust Cross-Modal Cell-Language Pre-training",
      "doi": "10.64898/2026.01.09.698573",
      "paper_url": "https://doi.org/10.64898/2026.01.09.698573",
      "route_count": 2,
      "configuration_count": 2,
      "family_counts": {
        "dense_continuous_carrier": 1,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "direct_projected_embedding": 1,
        "serialized_biological_context_or_ordered_profile": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "RNA",
        "text"
      ],
      "lifecycle_phases": [
        "pretraining"
      ],
      "fusion_topologies": [
        "other_explicit",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "no_text_on_this_route",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001277_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001277_8e87e5eae10d/figure_001.png",
        "figure_index": 1,
        "caption": "Fig. 1 (A) The schematic overview of the OKR-CELL method. (B) The illustration of several downstream tasks implemented via OKR-CELL, including cell clustering, batch affect correlation, cell-type annotation and cross-modal retrieval.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific schematic with two labeled panels: **A** and **B**.\n\n**Panel A:** Shows a workflow for constructing and pretraining a cross-modal cell-text model. Biological source objects include a human/body lung region illustration, dissociated cells, and **scRNA-seq** data represented as a gene-by-cell expression matrix. A text source labeled **Cell Description** is split into **Chunks**, sampled/retrieved, stored in a **Database**, processed by an **LLM**, and passed through a **Reliability Screen**. The cell modality is transformed from gene expression values into **Gene Tokens**, then passed into a **Cell Encoder**. The text modality is transformed into **Text Tokens**, then passed into a **Text Encoder**. The two encoders interface through a **Cross-modal Similarity** module. The pretraining block is labeled **Intra-modal Cellular Generative Pre-training** and includes masked gene expression values prediction. A second block is labeled **Cross-modal Cell-text Pre-training**, showing contrastive-style alignment between cells and text, with legend items including **Cell**, **Push**, **Real Positive Text**, **Fake Positive Text**, **Hard Negative Text**, and **Easy Negative Text**.\n\n**Panel B:** Shows downstream use of a **Pretrained Cell Encoder**. Biological inputs include organ/tissue icons and microscopy-like cellular imagery, followed by **scRNA-seq** gene expression matrices. Gene expression values are converted into **Gene Tokens** and passed into the pretrained cell encoder. The output is connected to multiple application panels: **Clustering**, **Batch-effect Correction**, **Cell type Annotation**, **Zero-shot Annotation**, **Few-shot Annotation**, and **Cross-modal Retrieval**.\n\nOverall, the figure depicts a computational biology / single-cell omics model pipeline that integrates scRNA-seq gene expression data with textual cell descriptions for cross-modal pretraining and downstream cell analysis tasks.",
        "page_no": 5,
        "sha256": "cdf370825d3d0bfcd7e4241b5d498050e78ed3dde00fe987b44c69c89b629577",
        "pixel_width": 745,
        "pixel_height": 520,
        "crop_box": {
          "x": 0.03,
          "y": 0.03,
          "width": 0.58,
          "height": 0.31
        },
        "panel_label": "A",
        "visible_input_object": "scRNA-seq cell-gene matrix / gene expression source with the source cell illustration",
        "visible_model_interface": "Gene Tokens feeding the Cell Encoder",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop isolates the left-to-center portion of panel A where the scRNA-seq source, gene expression values, gene tokens, and the cell encoder are all readable, while excluding the downstream output-only blocks and panel B.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_59c0f2ea15e4",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "single-cell RNA-seq data structured as a cell-gene matrix X",
          "actual_model_visible_form": "input embedding h(i) ∈ R^{M×D}"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_ff7cf939a1c2",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "LLM-enriched textual corpus",
          "actual_model_visible_form": "text tokens"
        }
      ],
      "routes": [
        {
          "route_id": "route_59c0f2ea15e4",
          "configuration_id": "config_0755fcfb4525",
          "route_label": "scRNA-seq cell-gene matrix to OKR-CELL cell encoder",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Intra-modal Cellular Generative Pre-training (IMGP)",
          "source_object_verbatim": "single-cell RNA-seq data structured as a cell-gene matrix X",
          "source_object_normalized": "single-cell RNA-seq cell-gene matrix",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "Gene Tokenization",
            "Gene Expression Binning",
            "Embedding Fusion"
          ],
          "model_visible_form_verbatim": "input embedding h(i) ∈ R^{M×D}",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "cell encoder / scGPT backbone",
          "fusion_topology": "other_explicit",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "The input embedding module transforms single-cell RNA-seq data-structured as a cell-gene matrix X",
          "section_heading": "4.1.1 Input Embeddings",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/377",
            "#/texts/378",
            "#/texts/379",
            "#/texts/380",
            "#/texts/381",
            "#/texts/383",
            "#/texts/384",
            "#/texts/385"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001277::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ff7cf939a1c2",
          "configuration_id": "config_11fafbe2ecd8",
          "route_label": "LLM-enriched text to OKR-CELL text encoder",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Cross-modal Cell-text Pre-training",
          "source_object_verbatim": "LLM-enriched textual corpus",
          "source_object_normalized": "LLM-enriched textual corpus",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "text tokenization"
          ],
          "model_visible_form_verbatim": "text tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "text encoder / Clinical-Longformer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "along with the textual corpus enriched by LLM",
          "section_heading": "2.1.2 Pre-training Objective and Pipeline",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/35",
            "#/texts/36"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001277::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_cf2af8e28dd7"
    },
    {
      "model_id": "model_064ca265f9bb",
      "model_name": "OmicsLM",
      "record_id": "june_update_2026-06-10__rec_000246",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_257d7c4ee751",
      "paper_title": "OmicsLM: A Multimodal Large Language Model for Multi-Sample Omics Reasoning",
      "doi": "",
      "paper_url": "",
      "route_count": 8,
      "configuration_count": 8,
      "family_counts": {
        "text_native_token_stream": 5,
        "dense_continuous_carrier": 3
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 2,
        "direct_projected_embedding": 3,
        "serialized_biological_context_or_ordered_profile": 1,
        "structured_biological_prompt_or_task_scaffold": 2
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "biological network knowledge",
        "bulk and multi-sample expression profiles",
        "mixed",
        "protein knowledgebase",
        "single-cell and bulk transcriptomic profiles",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning"
      ],
      "fusion_topologies": [
        "concatenation",
        "placeholder_replacement",
        "tokenizer_sequence",
        "unclear"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "metadata_or_context"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/june_update_2026_06_10_rec_000246_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/june_update_2026_06_10_rec_000246_55ef58661ddf/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: OmicsLM overview. Top: example questions from diverse omics datasets are formatted as multi-task conversations for OmicsLM, with each <omics> placeholder denoting a biological profile and gene symbols treated as explicit tokens. Bottom: each placeholder is built by mapping a single-cell or bulk expression profile, including profiles from perturbation screens, to a continuous vector and projecting it into the LLM token-embedding space, allowing multiple profiles to be interleaved in one prompt for sample-level and comparative reasoning.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic scientific figure describing **OmicsLM**, an LLM interface for omics tasks using explicit gene and omics tokenization.\n\nVisible components:\n\n- **Top panel: “MULTI-TASK QUESTIONS”**\n  - Shows example natural-language prompts for multiple datasets/tasks:\n    - **HCA:** “What cell type is `<omics>`?”\n    - **TCGA:** “Predict the cancer type from bulk profile `<omics>`.”\n    - **DEPMAP:** “Is *BRCA1* essential in cell line `<omics>`?”\n    - **GEO-OMICSQA:** “Do *CXCL10* and *ISG15* rise after cGAMP `<omics>` vs vehicle `<omics>`?”\n  - Indicates “70+ tasks across 10+ datasets.”\n\n- **Right model box: “OmicsLM”**\n  - Labeled as “LLM with explicit gene and omics tokenization.”\n  - Shows token types: `text`, `<omics>`, and `gene`.\n  - Lists downstream task categories:\n    - cell typing\n    - cancer phenotyping\n    - gene essentiality\n    - perturbation effects\n    - marker discovery\n    - pathway reasoning\n    - tissue annotation\n    - omics QA\n\n- **Bottom panel: “`<omics>` TOKEN CONSTRUCTION”**\n  - Starts from an **omics profile**, with icons for:\n    - single-cell\n    - bulk\n  - Converts the profile into a **continuous profile vector**.\n  - Example ranked/colored feature entries include:\n    - `countTPM`\n    - `expression`\n    - `FunomicsTO`\n    - `Geneformer`\n  - Vector dimensionality is shown as **D = 20,541**.\n  - A **linear omics projector** maps from profile space to LLM hidden space, labeled:\n    - \\( R^D \\rightarrow R^H \\)\n    - “replaces `<omics>` in the LLM input.”\n\nBiological source objects explicitly shown include omics profiles, genes *BRCA1*, *CXCL10*, and *ISG15*, cell lines, cancer/bulk profiles, and perturbation conditions involving cGAMP versus vehicle. The figure’s main finding/claim is that OmicsLM can convert single-cell or bulk omics profiles into projected `<omics>` tokens that are inserted into LLM inputs for multiple biological prediction and question-answering tasks.",
        "page_no": 2,
        "sha256": "20533e4095373818a3507e87177c8f50323443f2fa15a97ecfaefb30ab6f95c5",
        "pixel_width": 787,
        "pixel_height": 319,
        "crop_box": {
          "x": 0,
          "y": 0.62,
          "width": 1,
          "height": 0.38
        },
        "panel_label": "<omics> token construction",
        "visible_input_object": "omics profile (single-cell or bulk)",
        "visible_model_interface": "linear omics projector that replaces <omics> in the LLM input",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "The bottom panel alone shows the grounded input route: an omics profile is converted into a continuous profile vector and passed through a linear omics projector before replacing <omics> in the LLM input. It keeps the source object, transformation, arrow flow, and immediate insertion interface readable while excluding the output-only top examples.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_1812bd19fe75",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "single-cell or bulk transcriptomic profile",
          "actual_model_visible_form": "projected omics embedding"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_6190abf45a70",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "natural-language instructions and question prompts",
          "actual_model_visible_form": "text tokens"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_3a777c044617",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "canonical human gene symbols",
          "actual_model_visible_form": "atomic gene tokens"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_7ed0c41103cf",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "natural-language sample descriptions",
          "actual_model_visible_form": "Description of the sample"
        }
      ],
      "routes": [
        {
          "route_id": "route_6190abf45a70",
          "configuration_id": "config_8dfb03df08d3",
          "route_label": "Conversational text prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "standardized conversational format via a dynamic templating engine",
          "source_object_verbatim": "natural-language instructions and question prompts",
          "source_object_normalized": "natural-language prompts",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenized by the LLM tokenizer"
          ],
          "model_visible_form_verbatim": "text tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "standardized conversational format via a dynamic templating engine",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "standardized conversational format via a dynamic templating engine",
          "section_heading": "4 Training data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/100",
            "#/texts/101",
            "#/texts/102",
            "#/texts/103",
            "#/texts/104",
            "#/texts/330"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000246::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1812bd19fe75",
          "configuration_id": "config_8a7edf2a02d0",
          "route_label": "Projected omics embeddings",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "multi-task instruction-tuning corpus",
          "source_object_verbatim": "single-cell or bulk transcriptomic profile",
          "source_object_normalized": "transcriptomic profile",
          "source_modality_normalized": "single-cell and bulk transcriptomic profiles",
          "transformation_chain_verbatim": [
            "aligned to a fixed human gene panel",
            "assembled into a 20,541-dimensional omics vector",
            "linearly projected into the LLM input embedding space",
            "replaces <omics> in the LLM input"
          ],
          "model_visible_form_verbatim": "projected omics embedding",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "By replacing designated modality placeholders with these projected embeddings",
          "fusion_topology": "placeholder_replacement",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "By replacing designated modality placeholders with these projected embeddings",
          "section_heading": "3.1 Model Architecture",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            3,
            4,
            5,
            6,
            7,
            11,
            12,
            13,
            14,
            15,
            16,
            19,
            21,
            22,
            23,
            25,
            26
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/tables/10",
            "#/tables/12",
            "#/tables/13",
            "#/tables/5",
            "#/tables/6",
            "#/tables/7",
            "#/tables/8",
            "#/texts/100",
            "#/texts/101",
            "#/texts/102",
            "#/texts/103",
            "#/texts/104",
            "#/texts/113",
            "#/texts/114",
            "#/texts/115",
            "#/texts/116",
            "#/texts/118",
            "#/texts/175",
            "#/texts/176",
            "#/texts/177",
            "#/texts/178",
            "#/texts/179",
            "#/texts/180",
            "#/texts/213",
            "#/texts/214",
            "#/texts/230",
            "#/texts/232",
            "#/texts/237",
            "#/texts/303",
            "#/texts/304",
            "#/texts/306",
            "#/texts/320",
            "#/texts/322",
            "#/texts/330",
            "#/texts/6",
            "#/texts/7",
            "#/texts/70",
            "#/texts/71",
            "#/texts/72",
            "#/texts/73",
            "#/texts/75",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89",
            "#/texts/9",
            "#/texts/90",
            "#/texts/91",
            "#/texts/92",
            "#/texts/94",
            "#/texts/95"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000246::route_002"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000246::0002",
            "dense::june_update_2026-06-10__rec_000246::0004",
            "dense::june_update_2026-06-10__rec_000246::0005",
            "dense::june_update_2026-06-10__rec_000246::0007",
            "dense::june_update_2026-06-10__rec_000246::0008",
            "dense::june_update_2026-06-10__rec_000246::0011",
            "dense::june_update_2026-06-10__rec_000246::0012",
            "dense::june_update_2026-06-10__rec_000246::0057",
            "dense::june_update_2026-06-10__rec_000246::0060",
            "dense::june_update_2026-06-10__rec_000246::0061",
            "dense::june_update_2026-06-10__rec_000246::0062",
            "dense::june_update_2026-06-10__rec_000246::0063",
            "dense::june_update_2026-06-10__rec_000246::0064",
            "dense::june_update_2026-06-10__rec_000246::0093",
            "dense::june_update_2026-06-10__rec_000246::0094",
            "dense::june_update_2026-06-10__rec_000246::0096",
            "dense::june_update_2026-06-10__rec_000246::0134",
            "dense::june_update_2026-06-10__rec_000246::0136",
            "dense::june_update_2026-06-10__rec_000246::0137",
            "dense::june_update_2026-06-10__rec_000246::0138",
            "dense::june_update_2026-06-10__rec_000246::0148",
            "dense::june_update_2026-06-10__rec_000246::0149",
            "dense::june_update_2026-06-10__rec_000246::0158",
            "dense::june_update_2026-06-10__rec_000246::0159",
            "dense::june_update_2026-06-10__rec_000246::0160",
            "dense::june_update_2026-06-10__rec_000246::0161",
            "dense::june_update_2026-06-10__rec_000246::0162",
            "dense::june_update_2026-06-10__rec_000246::0164",
            "dense::june_update_2026-06-10__rec_000246::0165",
            "dense::june_update_2026-06-10__rec_000246::0166"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_93f3142beaf4",
          "configuration_id": "config_b1b8b58d8467",
          "route_label": "GEO-OmicsQA matched embeddings",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "GEO-OmicsQA multi-sample biological question answering",
          "source_object_verbatim": "one or more real expression profiles",
          "source_object_normalized": "real transcriptomic profiles",
          "source_modality_normalized": "bulk and multi-sample expression profiles",
          "transformation_chain_verbatim": [
            "paired with omics placeholders",
            "matched embeddings inserted into the prompt"
          ],
          "model_visible_form_verbatim": "<omics> placeholders with matched embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "placeholder embeddings are inserted into the prompt",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We introduce GEO-OmicsQA , a large-scale biological question-answering dataset constructed via automated literature scraping of the Gene Expression Omnibus (GEO) [Clough et al., 2024]. Test examples are publication-disjoint from the instruction-tuning corpus, so questions about a study are never evaluated on samples, metadata, or literature text from publications seen during training, and each question contains explicit references to one or more real expression profiles, requiring models to ground their answers in the linked omics data rather than textual priors.",
          "section_heading": "6.3 GEO-OmicsQA",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            8,
            19
          ],
          "doc_item_refs": [
            "#/tables/7",
            "#/texts/107",
            "#/texts/108",
            "#/texts/109",
            "#/texts/110",
            "#/texts/111",
            "#/texts/131",
            "#/texts/132",
            "#/texts/133"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000246::route_003"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000246::0003",
            "dense::june_update_2026-06-10__rec_000246::0009",
            "dense::june_update_2026-06-10__rec_000246::0010",
            "dense::june_update_2026-06-10__rec_000246::0059"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3a777c044617",
          "configuration_id": "config_0ba2f8168249",
          "route_label": "Gene-symbol tokens",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "explicit gene tokens",
          "source_object_verbatim": "canonical human gene symbols",
          "source_object_normalized": "human gene symbols",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "added to the base tokenizer",
            "initialized from the original subword embeddings",
            "trained together with the language model"
          ],
          "model_visible_form_verbatim": "atomic gene tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "gene-aware vocabulary augmentation",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "atomic tokens during instruction tuning",
          "section_heading": "3.1 Model Architecture",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/98"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000246::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_7ed0c41103cf",
          "configuration_id": "config_6263b435eb71",
          "route_label": "Sample description prompts",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "description-of-sample prompt with top-gene prediction",
          "source_object_verbatim": "natural-language sample descriptions",
          "source_object_normalized": "sample descriptions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "description of the sample",
            "identify the 134 genes with the highest expression levels in this sample"
          ],
          "model_visible_form_verbatim": "Description of the sample",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "Description of the sample ... Identify the 134 genes",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Identify the 134 genes",
          "section_heading": "A.7 Task Examples",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            21,
            22
          ],
          "doc_item_refs": [
            "#/tables/8",
            "#/texts/304",
            "#/texts/306"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000246::0067"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_505d12bcf3e7",
          "configuration_id": "config_0591e3c83efd",
          "route_label": "Cell-line genomics and annotations",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "cell-line genomics, disease, lineage, transformant, and similarity tasks",
          "source_object_verbatim": "cell-line expression profiles and metadata",
          "source_object_normalized": "cell-line expression profiles and annotations",
          "source_modality_normalized": "mixed",
          "transformation_chain_verbatim": [
            "retrieved annotations from the Cellosaurus REST API using the corresponding accession identifier ( CVCL_* )",
            "merged with existing cell-line metadata",
            "replaced by the projected omics embedding"
          ],
          "model_visible_form_verbatim": "projected omics embedding",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "replaced by the projected omics embedding",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "projected omics embedding",
          "section_heading": "4 Training data",
          "supporting_figure_or_table": "Table 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/92"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000246::0028",
            "dense::june_update_2026-06-10__rec_000246::0099",
            "dense::june_update_2026-06-10__rec_000246::0101",
            "dense::june_update_2026-06-10__rec_000246::0102",
            "dense::june_update_2026-06-10__rec_000246::0103",
            "dense::june_update_2026-06-10__rec_000246::0104",
            "dense::june_update_2026-06-10__rec_000246::0105",
            "dense::june_update_2026-06-10__rec_000246::0106",
            "dense::june_update_2026-06-10__rec_000246::0109",
            "dense::june_update_2026-06-10__rec_000246::0110",
            "dense::june_update_2026-06-10__rec_000246::0111",
            "dense::june_update_2026-06-10__rec_000246::0112"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e07beedf4bef",
          "configuration_id": "config_b465f476a320",
          "route_label": "Protein QA and localization",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "protein function, localization, and keyword annotations for protein QA tasks",
          "source_object_verbatim": "UniProt Swiss-Prot human protein records",
          "source_object_normalized": "UniProt Swiss-Prot human protein records",
          "source_modality_normalized": "protein knowledgebase",
          "transformation_chain_verbatim": [
            "downloaded the manually reviewed Swiss-Prot flat file",
            "filtered it to human proteins only",
            "extracted identity fields, functional descriptions, controlled-vocabulary keywords, subcellular localization, and protein class annotations"
          ],
          "model_visible_form_verbatim": "Protein function, localization, and keyword annotations",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "answer questions about protein properties given a gene name",
          "fusion_topology": "unclear",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "\"filtered it to human proteins only\"",
          "section_heading": "A.2.11 UniProt",
          "supporting_figure_or_table": "Table 10",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            17
          ],
          "doc_item_refs": [
            "#/texts/247"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000246::0029",
            "dense::june_update_2026-06-10__rec_000246::0133",
            "dense::june_update_2026-06-10__rec_000246::0143"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e4dd0a63096f",
          "configuration_id": "config_0724ff4ba9ee",
          "route_label": "Biological network reasoning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "natural-language graph-reasoning tasks over biological interaction networks and gene program annotations",
          "source_object_verbatim": "pathway graphs, protein-protein interaction networks, and gene set collections",
          "source_object_normalized": "biological network resources",
          "source_modality_normalized": "biological network knowledge",
          "transformation_chain_verbatim": [
            "pathway graphs, protein-protein interaction (PPI) networks, and gene set collections",
            "natural-language graph-reasoning tasks"
          ],
          "model_visible_form_verbatim": "natural-language graph-reasoning tasks",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "used solely to construct natural-language graph-reasoning tasks over biological interaction networks",
          "fusion_topology": "unclear",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Unlike the expression datasets above, pathway graphs, protein-protein interaction (PPI) networks, and gene set collections do not provide transcriptomic profiles; they are used solely to construct natural-language graph-reasoning tasks over biological interaction networks and gene program annotations.",
          "section_heading": "4 Training data",
          "supporting_figure_or_table": "Table 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            11,
            14,
            16,
            24,
            26
          ],
          "doc_item_refs": [
            "#/tables/11",
            "#/texts/151",
            "#/texts/152",
            "#/texts/153",
            "#/texts/154",
            "#/texts/155",
            "#/texts/156",
            "#/texts/157",
            "#/texts/239",
            "#/texts/240",
            "#/texts/241",
            "#/texts/242",
            "#/texts/331"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000246::0114",
            "dense::june_update_2026-06-10__rec_000246::0115",
            "dense::june_update_2026-06-10__rec_000246::0116",
            "dense::june_update_2026-06-10__rec_000246::0117",
            "dense::june_update_2026-06-10__rec_000246::0118",
            "dense::june_update_2026-06-10__rec_000246::0119",
            "dense::june_update_2026-06-10__rec_000246::0120",
            "dense::june_update_2026-06-10__rec_000246::0121",
            "dense::june_update_2026-06-10__rec_000246::0122",
            "dense::june_update_2026-06-10__rec_000246::0123",
            "dense::june_update_2026-06-10__rec_000246::0124",
            "dense::june_update_2026-06-10__rec_000246::0142",
            "dense::june_update_2026-06-10__rec_000246::0167"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_cab7fec0fcf3"
    },
    {
      "model_id": "model_b4d1be45d2bb",
      "model_name": "Omni-DNA",
      "record_id": "full_2026-07-06__rec_001773",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_c438b06dcc03",
      "paper_title": "Omni-DNA: A Unified Genomic Foundation Model for Cross-Modal and Multi-Task Learning",
      "doi": "10.48550/arXiv.2502.03499",
      "paper_url": "https://doi.org/10.48550/arXiv.2502.03499",
      "route_count": 6,
      "configuration_count": 6,
      "family_counts": {
        "discrete_biological_symbol_stream": 2,
        "text_native_token_stream": 4
      },
      "subtype_counts": {
        "native_biological_token_stream": 2,
        "structured_biological_prompt_or_task_scaffold": 3,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream",
        "plain_language_prompt_or_question",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "genomic sequence"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "other_explicit",
        "prefix",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001773_figure_003.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001773_811a9966d4e1/figure_003.png",
        "figure_index": 3,
        "caption": "Figure 4. Overview of of Omni-DNA architecture. In pretraining , Omni-DNA are trained on DNA only with next-token prediction. Multi-task finetuning enables the model to perform diverse tasks including classification, function prediction, and DNA-to-image.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic machine-learning workflow for an Omni-DNA auto-regressive transformer.\n\nVisible structure:\n- Two main sections: **Pretraining** on the left and **Multi-task Finetuning** on the right.\n- Central model block labeled **“Omni-DNA (Auto-Regressive Transformer)”**.\n- Left pretraining panel:\n  - Task labeled **“Next Token Prediction”**.\n  - DNA/token examples shown as boxed tokens such as `300`, `0`, `39`, and bases/k-mers such as `TATA`, `GCGC`, `G`.\n  - A **BPE Tokenizer** converts DNA sequence text into token IDs.\n- Right multi-task finetuning section:\n  - Three task panels labeled **Classification**, **Function Prediction**, and **DNA2Images**.\n  - Classification example asks whether DNA is an H3 or H4 sequence, with answers “Yes”.\n  - Function prediction example asks the function of DNA and gives an example response about mRNA for an olfactory receptor.\n  - DNA2Images example maps DNA to handwritten digits, with a small digit-like image shown.\n- Lower finetuning interface:\n  - Downstream tasks are unified into instruction-response format.\n  - Block labeled **“Task Unification + Expanded BPE Tokenizer”**.\n  - Inputs include **DNA**, **Label**, **Text**, and **Tensors**.\n  - Outputs are represented as **Instructions** and **Response** tokens.\n\nBiological source object:\n- DNA sequences are the primary biological input, represented as nucleotide/k-mer strings and tokenized DNA fragments.\n\nTransformation/model interface:\n- DNA is tokenized with BPE for pretraining.\n- Downstream DNA-related tasks are converted into instruction-response sequences using an expanded BPE tokenizer.\n- The same auto-regressive transformer is used for next-token prediction and multi-task finetuning.\n\nStated finding/claim visible in the figure:\n- The diagram presents a unified Omni-DNA framework that handles DNA sequence modeling, classification, function prediction, and DNA-to-tensor/image tasks through a shared transformer and unified instruction-response formatting.",
        "page_no": 4,
        "sha256": "94a37fafc43c437ff916296d712e3d2e561107e0f7c073dfddacf799b564aee4",
        "pixel_width": 892,
        "pixel_height": 332,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.27,
          "height": 1.0
        },
        "panel_label": "Pretraining",
        "visible_input_object": "DNA token sequences / unlabeled DNA data",
        "visible_model_interface": "BPE Tokenizer feeding the Omni-DNA auto-regressive transformer for next-token prediction",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the full left pretraining route: the DNA/token examples, the BPE Tokenizer block, the upward arrow into Omni-DNA, and the Next Token Prediction label. It excludes the downstream finetuning panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_ff58ef3678d5",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "unlabeled DNA data",
          "actual_model_visible_form": "DNA token sequences"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_b81c19c31deb",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "DNA sequence",
          "actual_model_visible_form": "What is the function of this sequence?GGCTG...TTTTCTGA"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_5a870e25295a",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "DNA sequences from labeled genomic tasks",
          "actual_model_visible_form": "instruction-response token sequence with expanded BPE tokenizer"
        }
      ],
      "routes": [
        {
          "route_id": "route_ff58ef3678d5",
          "configuration_id": "config_40d2635ba7e6",
          "route_label": "DNA-only pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "pretraining on DNA sequences with next token prediction objective",
          "source_object_verbatim": "unlabeled DNA data",
          "source_object_normalized": "DNA sequence",
          "source_modality_normalized": "genomic sequence",
          "transformation_chain_verbatim": [
            "remove exact duplicates from NCBI's multi-species genome dataset",
            "Byte-Pair Encoding (BPE) tokenizer",
            "autoregressive next token prediction"
          ],
          "model_visible_form_verbatim": "DNA token sequences",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "direct DNA-only tokenization into the auto-regressive transformer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "we pretrain a series of autoregressive transformer-based models using unlabeled DNA data",
          "section_heading": "3.1. Pretraining",
          "supporting_figure_or_table": "Figure 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            3,
            4
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/4",
            "#/texts/45",
            "#/texts/48",
            "#/texts/49",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001773::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001773::0001",
            "dense::full_2026-07-06__rec_001773::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5a870e25295a",
          "configuration_id": "config_d12728aa31ae",
          "route_label": "Task-unified DNA classification with prompts and label tokens",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "cross-modal multi-task finetuning; task unification",
          "source_object_verbatim": "DNA sequences from labeled genomic tasks",
          "source_object_normalized": "DNA sequences",
          "source_modality_normalized": "genomic sequence",
          "transformation_chain_verbatim": [
            "append a task-specific prompt Prompt_k to the input",
            "expand the tokenizer vocabulary to include unique new tokens",
            "replicate key label tokens by factor α",
            "add noise using NEFTune during loss computation"
          ],
          "model_visible_form_verbatim": "instruction-response token sequence with expanded BPE tokenizer",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "task-specific prompt appended to DNA input with expanded embedding matrix",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "paired_alignment_input",
          "evidence_quote": "each sample from task k is modified by appending a task-specific prompt Prompt k to its input x ( i )",
          "section_heading": "3.2. Cross-modal Multi-task Finetuning",
          "supporting_figure_or_table": "Algorithm 1 / Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            8,
            9
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/pictures/4",
            "#/pictures/5",
            "#/texts/141",
            "#/texts/142",
            "#/texts/143",
            "#/texts/144",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/257",
            "#/texts/260",
            "#/texts/281",
            "#/texts/89"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001773::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001773::0006",
            "dense::full_2026-07-06__rec_001773::0031"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6610039a3b9e",
          "configuration_id": "config_9d74fde07d18",
          "route_label": "DNA-to-functional-text generation",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "DNA2Func / Seq2Func functional annotation generation",
          "source_object_verbatim": "DNA sequences from 20 species",
          "source_object_normalized": "DNA sequences",
          "source_modality_normalized": "genomic sequence",
          "transformation_chain_verbatim": [
            "append a function-prediction prompt",
            "expand vocabulary with text tokens"
          ],
          "model_visible_form_verbatim": "instruction-response token sequence with textual annotations",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "task-specific prompt appended to DNA input with expanded BPE tokenizer",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "paired_alignment_input",
          "evidence_quote": "each sequence is annotated with a concise short annotation followed by a more detailed natural language description",
          "section_heading": "5.2. Functional Annotation Generation (DNA2Func)",
          "supporting_figure_or_table": "Figure 10 / Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            4,
            7
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/tables/4",
            "#/texts/174",
            "#/texts/175",
            "#/texts/176",
            "#/texts/177",
            "#/texts/178",
            "#/texts/4",
            "#/texts/89"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001773::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001773::0002",
            "dense::full_2026-07-06__rec_001773::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c4ad04e86ac4",
          "configuration_id": "config_065d161a999e",
          "route_label": "DNA-to-digit-image generation",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Needle-in-DNA task (DNA2Image)",
          "source_object_verbatim": "synthetic DNA sequences containing one of four functional motifs {TATAAA, CAAT, GGGCGG, TTAGGG}",
          "source_object_normalized": "synthetic DNA sequence",
          "source_modality_normalized": "genomic sequence",
          "transformation_chain_verbatim": [
            "append an image-mapping prompt",
            "discretize MNIST images with VQ-VAE into 49 discrete tokens"
          ],
          "model_visible_form_verbatim": "instruction-response token sequence with discretized image tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "task-specific prompt appended to DNA input; image output tokenized by VQ-VAE",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "paired_alignment_input",
          "evidence_quote": "generate the corresponding handwritten digit image",
          "section_heading": "5.3. Needle-in-DNA Task (DNA2Image)",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            1,
            4,
            7,
            8
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/pictures/3",
            "#/texts/180",
            "#/texts/183",
            "#/texts/247",
            "#/texts/248",
            "#/texts/249",
            "#/texts/250",
            "#/texts/4",
            "#/texts/89"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001773::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001773::0003",
            "dense::full_2026-07-06__rec_001773::0008"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_96f0223e6ea1",
          "configuration_id": "config_7e2bdb6bd7f0",
          "route_label": "DNA classification with a linear head",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "conventional full-size fine-tuning with a classification head for NT Downstream and Genomic Benchmark tasks",
          "source_object_verbatim": "DNA sequences from labeled genomic tasks",
          "source_object_normalized": "DNA sequences",
          "source_modality_normalized": "genomic sequence",
          "transformation_chain_verbatim": [
            "attach a linear layer on top for sequence classification",
            "full-size finetuning"
          ],
          "model_visible_form_verbatim": "DNA token sequences",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "pretrained model with AutoModelForSequenceClassification linear head",
          "fusion_topology": "other_explicit",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "automatically adds a linear layer on top for sequence classification",
          "section_heading": "D. Finetuning with Classificatoin Head",
          "supporting_figure_or_table": "Table 2 / Table 3 / Table 8 / Table 9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            6,
            8,
            9,
            14,
            15,
            16
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/pictures/5",
            "#/tables/2",
            "#/tables/7",
            "#/tables/8",
            "#/tables/9",
            "#/texts/155",
            "#/texts/156",
            "#/texts/257",
            "#/texts/260",
            "#/texts/281",
            "#/texts/420",
            "#/texts/421",
            "#/texts/422",
            "#/texts/423",
            "#/texts/424",
            "#/texts/425",
            "#/texts/428"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001773::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001773::0049",
            "dense::full_2026-07-06__rec_001773::0050",
            "dense::full_2026-07-06__rec_001773::0051",
            "dense::full_2026-07-06__rec_001773::0052",
            "dense::full_2026-07-06__rec_001773::0031"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b81c19c31deb",
          "configuration_id": "config_075a23c13205",
          "route_label": "DNA-to-functional-text inference",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "generate a natural language description for functional annotations",
          "source_object_verbatim": "DNA sequence",
          "source_object_normalized": "DNA sequence",
          "source_modality_normalized": "genomic sequence",
          "transformation_chain_verbatim": [
            "What is the function of this sequence?",
            "generate a natural language description for functional annotations"
          ],
          "model_visible_form_verbatim": "What is the function of this sequence?GGCTG...TTTTCTGA",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "What is the function of this sequence?",
          "fusion_topology": "prefix",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Omni-DNA could generate a natural language description for functional annotations.",
          "section_heading": "1. Introduction",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            2,
            4
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/texts/139",
            "#/texts/17",
            "#/texts/18",
            "#/texts/19",
            "#/texts/20"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001773::0004"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_8d5343810816"
    },
    {
      "model_id": "model_74e685d85fa8",
      "model_name": "OmniGene-4",
      "record_id": "june_update_2026-06-10__rec_000350",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_3f11aba2d4f5",
      "paper_title": "OmniGene-4: a unified bio-language MoE model with router-level interpretability",
      "doi": "10.64898/2026.05.12.724542",
      "paper_url": "https://doi.org/10.64898/2026.05.12.724542",
      "route_count": 18,
      "configuration_count": 6,
      "family_counts": {
        "discrete_biological_symbol_stream": 5,
        "text_native_token_stream": 13
      },
      "subtype_counts": {
        "native_biological_token_stream": 3,
        "plain_language_prompt_or_question": 3,
        "multi_track_structural_symbol_stream": 2,
        "structured_biological_prompt_or_task_scaffold": 10
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "multi_track_structural_symbol_stream",
        "native_biological_token_stream",
        "plain_language_prompt_or_question",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "DNA sequence",
        "DNA sequence pairs",
        "molecule descriptors / prompt text",
        "natural language",
        "natural language / protein mutation text",
        "natural language / protein question-answer text",
        "natural language / protein task text",
        "natural language / structural task text",
        "protein sequence",
        "protein sequence pair",
        "protein sequence pairs",
        "protein structure symbols",
        "single-cell transcriptomics / prompt text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "pretraining"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "Figures 1-3 are all analysis visuals about routing divergence or expert-usage shifts. None shows an actual source object, tokenized input, or immediate model interface for any listed route, so there is no grounded source figure to select.",
      "illustrative_examples": [
        {
          "subtype_id": "multi_track_structural_symbol_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_9b2b3ccb8d95",
          "example_input": "AA: M K T ...   SS: H H C ...",
          "example_carrier": "aligned sequence + structure tracks",
          "example_interface": "multi-track tokenizer → generator",
          "actual_source": "3Di (pdb_3di.fasta)",
          "actual_model_visible_form": "four-track parallel representation including text description, amino-acid sequence, DSSP backbone, and Foldseek letters"
        },
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_6eb1ed047213",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "DNA (dna_32g.txt)",
          "actual_model_visible_form": "DNA token sequences under the extended vocabulary"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_b1cf7879ee0b",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "OpenWebText",
          "actual_model_visible_form": "natural-language token sequences"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_5536cca39b1e",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "dnagpt/biopaws::protein_p",
          "actual_model_visible_form": "instruction-form examples labeled Homologous / Non-Homologous"
        }
      ],
      "routes": [
        {
          "route_id": "route_6eb1ed047213",
          "configuration_id": "config_73024cae174e",
          "route_label": "DNA CPT route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "CPT corpus",
          "source_object_verbatim": "DNA (dna_32g.txt)",
          "source_object_normalized": "DNA (dna_32g.txt)",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "sampled to 8GB (25%)",
            "tokenized and chunked to 1,024 tokens"
          ],
          "model_visible_form_verbatim": "DNA token sequences under the extended vocabulary",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "fed through the extended tokenizer and tied embedding table",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "DNA ( dna_32g.txt )",
          "section_heading": "2.3.1 Data",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            5
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/49",
            "#/texts/50",
            "#/texts/52"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c24183e12101",
          "configuration_id": "config_73024cae174e",
          "route_label": "Protein 1 CPT route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "CPT corpus",
          "source_object_verbatim": "Protein 1 (protein_uni_16.txt)",
          "source_object_normalized": "Protein 1 (protein_uni_16.txt)",
          "source_modality_normalized": "protein sequence",
          "transformation_chain_verbatim": [
            "used as 4GB of CPT data",
            "tokenized and chunked to 1,024 tokens"
          ],
          "model_visible_form_verbatim": "protein token sequences under the extended vocabulary",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "fed through the extended tokenizer and tied embedding table",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Protein 1",
          "section_heading": "2.3.1 Data",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            5
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/49",
            "#/texts/50",
            "#/texts/52"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b9bfb7a32c68",
          "configuration_id": "config_73024cae174e",
          "route_label": "Protein 2 CPT route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "CPT corpus",
          "source_object_verbatim": "Protein 2 (protein_lucaone_15g.txt)",
          "source_object_normalized": "Protein 2 (protein_lucaone_15g.txt)",
          "source_modality_normalized": "protein sequence",
          "transformation_chain_verbatim": [
            "used as 4GB of CPT data",
            "tokenized and chunked to 1,024 tokens"
          ],
          "model_visible_form_verbatim": "protein token sequences under the extended vocabulary",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "fed through the extended tokenizer and tied embedding table",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Protein 2",
          "section_heading": "2.3.1 Data",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            5
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/49",
            "#/texts/50",
            "#/texts/52"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b1cf7879ee0b",
          "configuration_id": "config_73024cae174e",
          "route_label": "OpenWebText CPT route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "CPT corpus",
          "source_object_verbatim": "OpenWebText",
          "source_object_normalized": "OpenWebText",
          "source_modality_normalized": "natural language",
          "transformation_chain_verbatim": [
            "retained in the CPT mixture at a 1:3 ratio against biological corpora",
            "tokenized and chunked to 1,024 tokens"
          ],
          "model_visible_form_verbatim": "natural-language token sequences",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "fed through the extended tokenizer and tied embedding table",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The 3Di [53] and DSSP [31] corpora are formatted as a four-track parallel representation - text description + 1-D amino acid sequence + 2-D DSSP backbone + 3-D Foldseek letters. OpenWebText [16] is retained at a 1 : 3 ratio against biological corpora to serve as a 'logical anchor' and prevent catastrophic forgetting of natural-language reasoning. After tokenization and chunking to 1,024 tokens, the binary corpus contains 11.7M chunks totaling 8.73B tokens.",
          "section_heading": "2.3.1 Data",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            5
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/49",
            "#/texts/50",
            "#/texts/52"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9b2b3ccb8d95",
          "configuration_id": "config_73024cae174e",
          "route_label": "3Di CPT route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "CPT corpus",
          "source_object_verbatim": "3Di (pdb_3di.fasta)",
          "source_object_normalized": "3Di (pdb_3di.fasta)",
          "source_modality_normalized": "protein structure symbols",
          "transformation_chain_verbatim": [
            "formatted as a four-track parallel representation",
            "tokenized and chunked to 1,024 tokens"
          ],
          "model_visible_form_verbatim": "four-track parallel representation including text description, amino-acid sequence, DSSP backbone, and Foldseek letters",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "multi_track_structural_symbol_stream",
          "insertion_or_fusion_verbatim": "formatted as a four-track parallel representation",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "3Di [53]",
          "section_heading": "2.3.1 Data",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/49",
            "#/texts/50"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0e9226a4885a",
          "configuration_id": "config_73024cae174e",
          "route_label": "DSSP CPT route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "CPT corpus",
          "source_object_verbatim": "DSSP (ss.txt)",
          "source_object_normalized": "DSSP (ss.txt)",
          "source_modality_normalized": "protein structure symbols",
          "transformation_chain_verbatim": [
            "formatted as a four-track parallel representation",
            "tokenized and chunked to 1,024 tokens"
          ],
          "model_visible_form_verbatim": "four-track parallel representation including text description, amino-acid sequence, DSSP backbone, and Foldseek letters",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "multi_track_structural_symbol_stream",
          "insertion_or_fusion_verbatim": "formatted as a four-track parallel representation",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "DSSP [31]",
          "section_heading": "2.3.1 Data",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/49",
            "#/texts/50"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d6115e2e8939",
          "configuration_id": "config_73024cae174e",
          "route_label": "Instruction replay CPT route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "CPT corpus",
          "source_object_verbatim": "Instruction replay (12 open corpora)",
          "source_object_normalized": "instruction replay corpus",
          "source_modality_normalized": "natural language",
          "transformation_chain_verbatim": [
            "mixed into the CPT corpus",
            "tokenized and chunked to 1,024 tokens"
          ],
          "model_visible_form_verbatim": "mixed instruction-style text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "mixed data corpus",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Instruction replay (12 open corpora)",
          "section_heading": "2.3.1 Data",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/tables/1",
            "#/texts/49",
            "#/texts/50"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5536cca39b1e",
          "configuration_id": "config_a4c451f9f087",
          "route_label": "Remote-homology SFT augmentation route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Bio-SFT v3 remote-homology ablation",
          "source_object_verbatim": "dnagpt/biopaws::protein_p",
          "source_object_normalized": "protein pair classification examples",
          "source_modality_normalized": "protein sequence pairs",
          "transformation_chain_verbatim": [
            "sampled 20,000 additional rows",
            "formatted as 5 rotating instructions",
            "labelled as Homologous / Non-Homologous"
          ],
          "model_visible_form_verbatim": "instruction-form examples labeled Homologous / Non-Homologous",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "wrapped as <User> ... <Assistant>",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Mixture-of-Experts (MoE) architectures offer a rare opportunity to probe the internal organization of large language models, but this affordance has not been systematically exploited in biological foundation modeling. We introduce OmniGene-4 , a unified bio-language foundation model built on Gemma-4-26B-A4B (30 layers, 128 experts per layer, top-8 routing) by injecting 28,028 biological tokens (DNA and protein BPE, Foldseek 3Di, DSSP secondary structure), continuing pretraining (CPT) on a 32.5 GB mixture of DNA, protein, natural-language and structural corpora, and supervised fine-tuning (SFT) on 199,576 instruction-format examples spanning eight task families. On a suite of standard benchmarks, the final model (v3) reaches 99.95% accuracy on BioPAWS standard protein homology (6,000 pairs), 59.50% on remote homology (2,000 pairs from protein_pair_remote ), and 93.66% on BixBench knowledge questions. Relative to its un-fine-tuned vocabulary-extended Gemma-4-Instruct baseline (85% / 60% / 87%), v3 gains +14 . 5 on Standard, is comparable on Remote ( -0 . 5, within statistical noise on this 2,000-pair sample), and gains +6 . 7 on BixBench. We do not claim parity with specialist remote-homology tools; published numbers for ESM-2, CATHe and PLMSearch on differently constructed splits reach 65-75%, and closing this gap is discussed as an open problem.",
          "section_heading": "2.4.1 Data assembly",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            6
          ],
          "doc_item_refs": [
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/6",
            "#/texts/7",
            "#/texts/70",
            "#/texts/71"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_009"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b22d2b402714",
          "configuration_id": "config_09422b5ee812",
          "route_label": "Homology SFT source route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "OmniGene-SFT-v1 data assembly",
          "source_object_verbatim": "homology_task_sft.jsonl",
          "source_object_normalized": "homology task SFT examples",
          "source_modality_normalized": "natural language / protein task text",
          "transformation_chain_verbatim": [
            "merged into OmniGene-SFT-v1",
            "converted to {instruction,input,output} triple form",
            "deduplicated by md5(instruction || input || output)"
          ],
          "model_visible_form_verbatim": "<User> instruction/input prompt with <Assistant> response target",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "wrapped as <User> ... <Assistant>",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "homology_task_sft.jsonl",
          "section_heading": "2.4.1 Data assembly",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            6
          ],
          "doc_item_refs": [
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/6",
            "#/texts/7",
            "#/texts/70",
            "#/texts/71"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_013"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_af45010fa9da",
          "configuration_id": "config_09422b5ee812",
          "route_label": "Structure SFT source route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "OmniGene-SFT-v1 data assembly",
          "source_object_verbatim": "structure_task_sft.jsonl",
          "source_object_normalized": "structure task SFT examples",
          "source_modality_normalized": "natural language / structural task text",
          "transformation_chain_verbatim": [
            "merged into OmniGene-SFT-v1",
            "converted to {instruction,input,output} triple form",
            "deduplicated by md5(instruction || input || output)"
          ],
          "model_visible_form_verbatim": "<User> instruction/input prompt with <Assistant> response target",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "wrapped as <User> ... <Assistant>",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "structure_task_sft.jsonl",
          "section_heading": "2.4.1 Data assembly",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            6
          ],
          "doc_item_refs": [
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/6",
            "#/texts/7",
            "#/texts/70",
            "#/texts/71"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_014"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_afa20a4bdb7f",
          "configuration_id": "config_09422b5ee812",
          "route_label": "UniProtQA SFT source route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "OmniGene-SFT-v1 data assembly",
          "source_object_verbatim": "uniprotqa_sft.jsonl",
          "source_object_normalized": "UniProtQA examples",
          "source_modality_normalized": "natural language / protein question-answer text",
          "transformation_chain_verbatim": [
            "merged into OmniGene-SFT-v1",
            "converted to {instruction,input,output} triple form",
            "deduplicated by md5(instruction || input || output)"
          ],
          "model_visible_form_verbatim": "<User> instruction/input prompt with <Assistant> response target",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "wrapped as <User> ... <Assistant>",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "uniprotqa_sft.jsonl",
          "section_heading": "2.4.1 Data assembly",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            6
          ],
          "doc_item_refs": [
            "#/tables/2",
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/6",
            "#/texts/7",
            "#/texts/70",
            "#/texts/71",
            "#/texts/72",
            "#/texts/73"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_015"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000350::0011"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8c0045aa548e",
          "configuration_id": "config_09422b5ee812",
          "route_label": "Mutation description SFT source route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "OmniGene-SFT-v1 data assembly",
          "source_object_verbatim": "mutadescribe_sft.jsonl",
          "source_object_normalized": "mutation description examples",
          "source_modality_normalized": "natural language / protein mutation text",
          "transformation_chain_verbatim": [
            "merged into OmniGene-SFT-v1",
            "converted to {instruction,input,output} triple form",
            "deduplicated by md5(instruction || input || output)"
          ],
          "model_visible_form_verbatim": "<User> instruction/input prompt with <Assistant> response target",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "wrapped as <User> ... <Assistant>",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "mutadescribe_sft.jsonl",
          "section_heading": "2.4.1 Data assembly",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            3,
            6
          ],
          "doc_item_refs": [
            "#/tables/2",
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/19",
            "#/texts/20",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/6",
            "#/texts/7",
            "#/texts/70",
            "#/texts/71"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_016"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000350::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_59bc33530f4c",
          "configuration_id": "config_09422b5ee812",
          "route_label": "Cell SFT source route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "OmniGene-SFT-v1 data assembly",
          "source_object_verbatim": "cell_sft_master.jsonl",
          "source_object_normalized": "cell SFT examples",
          "source_modality_normalized": "single-cell transcriptomics / prompt text",
          "transformation_chain_verbatim": [
            "constructed from PanglaoDB-style marker panels covering about 40 human cell types across 6 task templates",
            "merged into OmniGene-SFT-v1",
            "converted to {instruction,input,output} triple form",
            "deduplicated by md5(instruction || input || output)"
          ],
          "model_visible_form_verbatim": "<User> instruction/input prompt with <Assistant> response target",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "wrapped as <User> ... <Assistant>",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "cell_sft_master.jsonl",
          "section_heading": "2.4.1 Data assembly",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            6
          ],
          "doc_item_refs": [
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/6",
            "#/texts/7",
            "#/texts/70",
            "#/texts/71"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_017"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_755df4bfcf47",
          "configuration_id": "config_09422b5ee812",
          "route_label": "Molecule SFT source route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "OmniGene-SFT-v1 data assembly",
          "source_object_verbatim": "mol_sft_master.jsonl",
          "source_object_normalized": "molecule SFT examples",
          "source_modality_normalized": "molecule descriptors / prompt text",
          "transformation_chain_verbatim": [
            "built from nine MoleculeNet datasets with RDKit-derived physico-chemical descriptors across 6 task templates",
            "merged into OmniGene-SFT-v1",
            "converted to {instruction,input,output} triple form",
            "deduplicated by md5(instruction || input || output)"
          ],
          "model_visible_form_verbatim": "<User> instruction/input prompt with <Assistant> response target",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "wrapped as <User> ... <Assistant>",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "mol_sft_master.jsonl",
          "section_heading": "2.4.1 Data assembly",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            6
          ],
          "doc_item_refs": [
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/6",
            "#/texts/7",
            "#/texts/70",
            "#/texts/71"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_018"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a1ba49b312f1",
          "configuration_id": "config_a4c451f9f087",
          "route_label": "DNA pairs SFT source route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Bio-SFT v3 remote-homology ablation",
          "source_object_verbatim": "sampled DNA pairs",
          "source_object_normalized": "DNA pairs",
          "source_modality_normalized": "DNA sequence pairs",
          "transformation_chain_verbatim": [
            "sampled as paired examples",
            "merged into the SFT corpus",
            "converted to {instruction,input,output} triple form",
            "deduplicated by md5(instruction || input || output)"
          ],
          "model_visible_form_verbatim": "<User> instruction/input prompt with <Assistant> response target",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "wrapped as <User> ... <Assistant>",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "sampled DNA pairs",
          "section_heading": "2.4.1 Data assembly",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            6
          ],
          "doc_item_refs": [
            "#/tables/2",
            "#/texts/13",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/6",
            "#/texts/7",
            "#/texts/70",
            "#/texts/71",
            "#/texts/72",
            "#/texts/73"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_019"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000350::0009"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f090ee054645",
          "configuration_id": "config_75f5d99b2c82",
          "route_label": "Standard homology evaluation route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Standard homology",
          "source_object_verbatim": "dnagpt/biopaws::protein_pair_short",
          "source_object_normalized": "protein pair short benchmark",
          "source_modality_normalized": "protein sequence pair",
          "transformation_chain_verbatim": [
            "30% random sample",
            "balanced labels",
            "prompt matches the SFT training prompt"
          ],
          "model_visible_form_verbatim": "protein-pair classification prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt matches the SFT training prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Standard homology : 30% random sample of dnagpt/biopaws::protein_pair_short (6,000 pairs, balanced labels, seed 42). The prompt matches the SFT training prompt; the parser accepts Homologous / Non-Homologous in the first 40 characters of output (caseinsensitive), with non-homolog* matched before homolog* to avoid substring misclassification.",
          "section_heading": "2.5 Benchmark protocol",
          "supporting_figure_or_table": "Table 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            7
          ],
          "doc_item_refs": [
            "#/texts/79",
            "#/texts/80",
            "#/texts/81",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_010"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_82226f01fd7e",
          "configuration_id": "config_b416d8e878f8",
          "route_label": "Remote homology evaluation route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Remote homology",
          "source_object_verbatim": "dnagpt/biopaws::protein_pair_remote",
          "source_object_normalized": "protein pair remote benchmark",
          "source_modality_normalized": "protein sequence pair",
          "transformation_chain_verbatim": [
            "2,000 pairs sampled with seed 42",
            "balanced labels"
          ],
          "model_visible_form_verbatim": "protein-pair classification prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompt matches the SFT training prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Remote homology : 2,000 pairs (1,000 per label) sampled with seed 42 from dnagpt/biopaws::protein_pai",
          "section_heading": "2.5 Benchmark protocol",
          "supporting_figure_or_table": "Table 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            7
          ],
          "doc_item_refs": [
            "#/texts/79",
            "#/texts/80",
            "#/texts/81",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_011"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b91c60a6f8e5",
          "configuration_id": "config_00b7c42248ff",
          "route_label": "BixBench evaluation route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "BixBench Knowledge",
          "source_object_verbatim": "BixBench True / False questions",
          "source_object_normalized": "BixBench True / False questions",
          "source_modality_normalized": "natural language",
          "transformation_chain_verbatim": [
            "all True / False questions",
            "input truncated to 500 characters of the result field"
          ],
          "model_visible_form_verbatim": "truncated result-field question text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "standard prompt input with truncated result field",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "BixBench Knowledge [15]: all True / False questions ( ≈ 200). Input truncated to 500 characters of the result field.",
          "section_heading": "2.5 Benchmark protocol",
          "supporting_figure_or_table": "Table 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            6,
            7
          ],
          "doc_item_refs": [
            "#/texts/79",
            "#/texts/80",
            "#/texts/81",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000350::route_012"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_493db55fa177"
    },
    {
      "model_id": "model_fa039bd90d50",
      "model_name": "OmniGene-4 v3",
      "record_id": "june_update_2026-06-10__rec_000350",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_3f11aba2d4f5",
      "paper_title": "OmniGene-4: a unified bio-language MoE model with router-level interpretability",
      "doi": "10.64898/2026.05.12.724542",
      "paper_url": "https://doi.org/10.64898/2026.05.12.724542",
      "route_count": 5,
      "configuration_count": 3,
      "family_counts": {
        "text_native_token_stream": 5
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 4,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "biological sequence",
        "instruction-formatted biological text",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning"
      ],
      "fusion_topologies": [
        "concatenation",
        "placeholder_replacement"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "Figures 1-3 are internal routing-analysis plots: layer-wise JS divergence, gain decomposition, and expert-usage heatmap. None visibly shows a grounded source object, a raw DNA/protein/Cell/Mol/Structure input, or the neutral-template prompt scaffold ('Task: ... / Content: ... / End') required to support any listed route.",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_8fc9a178a272",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "OmniGene-SFT-v1",
          "actual_model_visible_form": "<User>\n### Instruction:\n{instr}\n\n{input}\n### Answer:\n<Assistant>\n{output}"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_6fd4f501ee5a",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "raw DNA and protein strings",
          "actual_model_visible_form": "Task: Protein / Content: {text} / End"
        }
      ],
      "routes": [
        {
          "route_id": "route_6fd4f501ee5a",
          "configuration_id": "config_771430a5d03e",
          "route_label": "Protein neutral-template control route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "raw DNA and protein strings",
          "source_object_verbatim": "raw DNA and protein strings",
          "source_object_normalized": "protein sequence",
          "source_modality_normalized": "biological sequence",
          "transformation_chain_verbatim": [
            "wrapped in a single neutral template"
          ],
          "model_visible_form_verbatim": "Task: Protein / Content: {text} / End",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "Task: {name} / Content: {text} / End",
          "fusion_topology": "placeholder_replacement",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "raw DNA and protein strings",
          "section_heading": "5 Limitations and future work",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15
          ],
          "doc_item_refs": [
            "#/texts/160",
            "#/texts/161",
            "#/texts/162",
            "#/texts/163",
            "#/texts/164",
            "#/texts/165",
            "#/texts/166"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000350::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3b056f06e7ba",
          "configuration_id": "config_4c8e5313c27b",
          "route_label": "Cell neutral-template control route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "instructionformatted Cell/Mol/Structure prompts",
          "source_object_verbatim": "instructionformatted Cell/Mol/Structure prompts",
          "source_object_normalized": "Cell prompt text",
          "source_modality_normalized": "instruction-formatted biological text",
          "transformation_chain_verbatim": [
            "wrapped in a single neutral template"
          ],
          "model_visible_form_verbatim": "Task: Cell / Content: {text} / End",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "Task: {name} / Content: {text} / End",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "instructionformatted Cell/Mol/Structure prompts",
          "section_heading": "5 Limitations and future work",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15
          ],
          "doc_item_refs": [
            "#/texts/160",
            "#/texts/161",
            "#/texts/162",
            "#/texts/163",
            "#/texts/164",
            "#/texts/165",
            "#/texts/166"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000350::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_7ce5b7aefdf7",
          "configuration_id": "config_4c8e5313c27b",
          "route_label": "Mol neutral-template control route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "instructionformatted Cell/Mol/Structure prompts",
          "source_object_verbatim": "instructionformatted Cell/Mol/Structure prompts",
          "source_object_normalized": "Mol prompt text",
          "source_modality_normalized": "instruction-formatted biological text",
          "transformation_chain_verbatim": [
            "wrapped in a single neutral template"
          ],
          "model_visible_form_verbatim": "Task: Mol / Content: {text} / End",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "Task: {name} / Content: {text} / End",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "instructionformatted Cell/Mol/Structure prompts",
          "section_heading": "5 Limitations and future work",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15
          ],
          "doc_item_refs": [
            "#/texts/160",
            "#/texts/161",
            "#/texts/162",
            "#/texts/163",
            "#/texts/164",
            "#/texts/165",
            "#/texts/166"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000350::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fd1ec85c31e1",
          "configuration_id": "config_4c8e5313c27b",
          "route_label": "Structure neutral-template control route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "instructionformatted Cell/Mol/Structure prompts",
          "source_object_verbatim": "instructionformatted Cell/Mol/Structure prompts",
          "source_object_normalized": "Structure prompt text",
          "source_modality_normalized": "instruction-formatted biological text",
          "transformation_chain_verbatim": [
            "wrapped in a single neutral template"
          ],
          "model_visible_form_verbatim": "Task: Structure / Content: {text} / End",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "Task: {name} / Content: {text} / End",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "instructionformatted Cell/Mol/Structure prompts",
          "section_heading": "5 Limitations and future work",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15
          ],
          "doc_item_refs": [
            "#/texts/160",
            "#/texts/161",
            "#/texts/162",
            "#/texts/163",
            "#/texts/164",
            "#/texts/165",
            "#/texts/166"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000350::0008"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8fc9a178a272",
          "configuration_id": "config_4d04e10669d4",
          "route_label": "General text prompt route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "general",
          "source_object_verbatim": "OmniGene-SFT-v1",
          "source_object_normalized": "instruction dataset",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "merged into OmniGene-SFT-v1",
            "converted to {instruction, input, output} triple form",
            "wrapped as a text prompt"
          ],
          "model_visible_form_verbatim": "<User>\n### Instruction:\n{instr}\n\n{input}\n### Answer:\n<Assistant>\n{output}",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "<User>\n### Instruction:\n{instr}\n\n{input}\n### Answer:\n<Assistant>\n{output}",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Bio-SFT v2 on 179K examples (homology, UniProtQA, structure, mutation, cell, molecule, general)",
          "section_heading": "2.4 Supervised fine-tuning (SFT)",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The paper supports a general SFT instruction-prompt family, but it does not separately name a dedicated source corpus for that family.",
          "pages": [
            3
          ],
          "doc_item_refs": [
            "#/texts/23"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000350::0010"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_85e0c6746826"
    },
    {
      "model_id": "model_74758c50e662",
      "model_name": "OmniNA",
      "record_id": "full_2026-07-06__rec_002327",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_c317900baf0b",
      "paper_title": "OmniNA: A foundation model for nucleotide sequences",
      "doi": "10.1101/2024.01.14.575543",
      "paper_url": "https://doi.org/10.1101/2024.01.14.575543",
      "route_count": 6,
      "configuration_count": 4,
      "family_counts": {
        "text_native_token_stream": 4,
        "discrete_biological_symbol_stream": 2
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 3,
        "plain_language_prompt_or_question": 1,
        "native_biological_token_stream": 2
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream",
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "nucleotide sequence",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "tokenizer_sequence",
        "unclear"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_002327_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_002327_accd0e8b6545/figure_001.png",
        "figure_index": 1,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific figure about a genomic foundation model, labeled with panels **A** and **B**.\n\n**Panel A** shows a workflow for training and applying a nucleotide sequence model called **OmniNA**.\n\nVisible stages:\n- **Stage 1: Pre-training on NCBI nucleotide database**\n  - Input described as an **NCBI nucleotide sequence database**.\n  - Text states a total of **1076.2 billion bases** from **91.7 million nucleotide sequences** spanning many species.\n  - A phylogenetic/tree-like graphic represents broad species diversity.\n  - A **data processing** box shows:\n    - Prefix: “Annotate the following sequence.”\n    - Input: a nucleotide sequence.\n    - Response: described as a Homo sapiens gene sequence related to acetyl cholinesterase.\n  - A **large-scale pre-training** box shows auto-regressive training, tokenized nucleotide text, transformer blocks, and “**1.7 billion parameters**.”\n\n- **Stage 2: Fine-tuning for genomic tasks**\n  - Inputs include genomic sequences such as:\n    - A sequence for promoter classification.\n    - A mutation example with a highlighted nucleotide substitution.\n    - A gRNA and target RNA pairing example.\n  - The pre-trained **OmniNA** model is shown with transformer blocks.\n  - Downstream task outputs include question-answer style interfaces:\n    - “Is the sequence a promoter?” Answer: No.\n    - “Classify the mutation as pathogenic/benign.” Answer: Pathogenic.\n    - “Does the gRNA will off-target?” Answer: Yes.\n\n**Panel B** is a biological/genomic overview illustration.\n\nVisible biological source objects and labels:\n- Chromosomes and chromatin/nucleosome structure.\n- DNA double helix with labeled genomic features.\n- Species phylogeny diagram labeled with domains/groups including **Bacteria**, **Archaea**, and **Eukarya**.\n- Functional genomic elements and molecular features labeled:\n  - **Species differentiation**\n  - **Histone occupation**\n  - **Chromosome accessibility**\n  - **Transcription factor**\n  - **Enhancer**\n  - **Translation initiation site**\n  - **DNA methylation**\n  - **poly(A) signal**\n  - **Splice site**\n  - **Promoter**\n  - **Functional genomic element characterization**\n\nOverall, the figure presents a nucleotide-sequence foundation model pipeline: large-scale pre-training on NCBI nucleotide data, fine-tuning for genomic tasks, and biological applications involving species differentiation and functional genomic element characterization.",
        "page_no": 32,
        "sha256": "4009bc09193a58c73e92264fbab8a20b45f4ca1a6fbd689fc9f4cc7b5062d76e",
        "pixel_width": 439,
        "pixel_height": 581,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 1.0,
          "height": 0.31
        },
        "panel_label": "A / Stage 1",
        "visible_input_object": "NCBI nucleotide sequence database; nucleotide sequence input",
        "visible_model_interface": "tokenized nucleotide sequence fed into the large-scale pre-training transformer",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the full Stage 1 pre-training route: source database at left, data-processing prompt in the middle, and the tokenized sequence plus transformer insertion on the right. It excludes the fine-tuning outputs and panel B.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_9c4898db67f0",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "mutated and wild-type nucleotide sequences",
          "actual_model_visible_form": "mutated ( s mut ) and wild-type ( s WT ) nucleotide sequences"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_02ce74ec15e8",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "test question",
          "actual_model_visible_form": "test question tokens"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_f25f4eadc58f",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "nucleotide sequence",
          "actual_model_visible_form": "tokenized nucleotide sequence"
        }
      ],
      "routes": [
        {
          "route_id": "route_f25f4eadc58f",
          "configuration_id": "config_9e37c67ea218",
          "route_label": "OmniNA pre-training nucleotide-sequence input",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "model pre-training with prompts composed of an instruction, a nucleotide sequence, and an annotation",
          "source_object_verbatim": "nucleotide sequence",
          "source_object_normalized": "nucleotide sequence",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "curate sequence data and corresponding textual annotations from the NCBI nt Database up to April, 2023",
            "organize the sequence data with the text annotations as feature-ordered input",
            "partition the data into sentence-based chunks",
            "tokenize with SentencePiece byte pair encoding",
            "add special tokens <s> and </s>",
            "embed the input tokens and add GPTNeo positional encodings"
          ],
          "model_visible_form_verbatim": "tokenized nucleotide sequence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "concatenated with position embeddings and fed into the transformer layers",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "We developed OmniNA as a foundation model for nucleotide sequence ( Figure 1 ). The cognitive core of OmniNA is a LLM developed through incremental training on an extensive nucleotide sequence corpus, which inherits the benefits of emergent abilities of LLM and domain-specific knowledge. The backbone of OmniNA is a transformer-based auto-regressive decoder with self-attention mechanism 10 . Three models with varying layer and head numbers were developed, spanning scales from 66 M to 330 M and up to 1.7 B parameters ( Supplementary Figure 1 ). The detailed network structure is delineated in Supplementary Figure 2 . We curated a dataset for model pre-training, which comprises 1076.2 B bases from 91.7 M nucleotide sequences across diverse species, extracted from the NCBI nt database ( Figure 1A ; see Method ). We designed prompts to structure the sequences and their corresponding natural language descriptions as model input ( Figure 1A ; Supplementary Table 1 ). The input underwent chunking into subword tokens, followed by embedding through the token embedding layer. Token embeddings were then concatenated with position embeddings and fed into the transformer layers and followed by a regression layer for auto-regressive training ( Figure 1A and Supplementary Figure 2 ) 5 . The pre-trained OmniNA can handle versatile downstream applications in a multi-task manner ( Figure 1B ). After fine-tuning, the model can manage taxonomy and genomic element identification tasks in a Question-Answer (QA) paradigm ( Figure 1A ).",
          "section_heading": "Methods Data processing for OmniNA pre-training",
          "supporting_figure_or_table": "Supplementary Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The canonical paper presents the sequence together with annotation in one shared pre-training prompt; isolating the sequence as its own route is a route-level split.",
          "pages": [
            4,
            5,
            12,
            25,
            26
          ],
          "doc_item_refs": [
            "#/texts/149",
            "#/texts/151",
            "#/texts/152",
            "#/texts/17",
            "#/texts/19",
            "#/texts/21",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/63"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002327::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002327::0001",
            "dense::full_2026-07-06__rec_002327::0010",
            "dense::full_2026-07-06__rec_002327::0019"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c9f58cc549fc",
          "configuration_id": "config_9e37c67ea218",
          "route_label": "OmniNA pre-training annotation input",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "model pre-training with prompts composed of an instruction, a nucleotide sequence, and an annotation",
          "source_object_verbatim": "text annotations",
          "source_object_normalized": "text annotations",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "curate sequence data and corresponding textual annotations from the NCBI nt Database up to April, 2023",
            "organize the sequence data with the text annotations as feature-ordered input",
            "partition the data into sentence-based chunks",
            "tokenize with SentencePiece byte pair encoding",
            "add special tokens <s> and </s>",
            "embed the input tokens and add GPTNeo positional encodings"
          ],
          "model_visible_form_verbatim": "annotation tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "concatenated with position embeddings and fed into the transformer layers",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "The prompt is composed of three modules: 1. an instruction; 2. a nucleotide sequence; 3. an annotation.",
          "section_heading": "Methods Data processing for OmniNA pre-training",
          "supporting_figure_or_table": "Supplementary Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The canonical paper presents the annotation together with sequence and instruction in one shared pre-training prompt; isolating the annotation as its own route is a route-level split.",
          "pages": [
            4,
            5,
            12,
            25,
            26
          ],
          "doc_item_refs": [
            "#/texts/149",
            "#/texts/15",
            "#/texts/151",
            "#/texts/152",
            "#/texts/17",
            "#/texts/19",
            "#/texts/21",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/63"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002327::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002327::0001",
            "dense::full_2026-07-06__rec_002327::0010",
            "dense::full_2026-07-06__rec_002327::0019"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_846f86009834",
          "configuration_id": "config_aadec28219bc",
          "route_label": "OmniNA fine-tuning nucleotide-sequence input",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "multi-task QA framework with an instruction, a nucleotide sequence, a test question, and an answer",
          "source_object_verbatim": "nucleotide sequence",
          "source_object_normalized": "nucleotide sequence",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "compose a prompt with an instruction, a nucleotide sequence, a test question, and an answer",
            "keep the first two parts fixed for all tasks",
            "train only the answer in an auto-regressive manner",
            "fine-tune with AdamW for 2 epochs"
          ],
          "model_visible_form_verbatim": "instruction, nucleotide sequence, and test question tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "fed as a prompt into the pre-trained decoder-only transformer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "The first two parts are fixed for all tasks, while the question and answer are task-specific.",
          "section_heading": "Apply the model for multi-task fine-tuning",
          "supporting_figure_or_table": "Supplementary Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper presents the sequence as one component of a four-part QA prompt; isolating the sequence as a separate route is a route-level split.",
          "pages": [
            10
          ],
          "doc_item_refs": [
            "#/texts/43"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002327::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002327::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_02ce74ec15e8",
          "configuration_id": "config_aadec28219bc",
          "route_label": "OmniNA fine-tuning test-question input",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "multi-task QA framework with an instruction, a nucleotide sequence, a test question, and an answer",
          "source_object_verbatim": "test question",
          "source_object_normalized": "test question",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "compose a prompt with an instruction, a nucleotide sequence, a test question, and an answer",
            "keep the first two parts fixed for all tasks",
            "train only the answer in an auto-regressive manner",
            "fine-tune with AdamW for 2 epochs"
          ],
          "model_visible_form_verbatim": "test question tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "fed as a prompt into the pre-trained decoder-only transformer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The first two parts are fixed for all tasks, while the question and answer are task-specific.",
          "section_heading": "Apply the model for multi-task fine-tuning",
          "supporting_figure_or_table": "Supplementary Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper presents the question as one component of a four-part QA prompt; isolating the question as a separate route is a route-level split.",
          "pages": [
            10
          ],
          "doc_item_refs": [
            "#/texts/43"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_002327::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002327::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9c4898db67f0",
          "configuration_id": "config_d47be6e5cd38",
          "route_label": "OmniNA saturation mutation scoring",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "We quantify mutation effects using log-likelihood (LL) differences between mutated ( s mut ) and wild-type ( s WT ) nucleotide sequences",
          "source_object_verbatim": "mutated and wild-type nucleotide sequences",
          "source_object_normalized": "paired mutated and wild-type nucleotide sequences",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "mutated nucleotide sequence",
            "wild-type nucleotide sequence",
            "log-likelihood (LL) differences",
            "predicted mutation effect"
          ],
          "model_visible_form_verbatim": "mutated ( s mut ) and wild-type ( s WT ) nucleotide sequences",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "log-likelihood (LL) differences between mutated ( s mut ) and wild-type ( s WT ) nucleotide sequences",
          "fusion_topology": "unclear",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "log-likelihood (LL) differences",
          "section_heading": "Saturation mutation analysis",
          "supporting_figure_or_table": "Figure 5A",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            11
          ],
          "doc_item_refs": [
            "#/texts/50"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002327::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b382ae08863d",
          "configuration_id": "config_c84f48c96136",
          "route_label": "OmniNA mutation-effect prediction in distinct genomic contexts",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "predict the mutation impact within varied genomic contexts",
          "source_object_verbatim": "mutated and wild-type nucleotide sequences",
          "source_object_normalized": "paired mutated and wild-type nucleotide sequences",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "paired wild-type and mutated nucleotide sequences",
            "estimate LL difference between the paired sequences",
            "derive a mutation effect score"
          ],
          "model_visible_form_verbatim": "mutated and wild-type nucleotide sequences",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "the OmniNA-1.7B estimated LL between mutated and wild-type nucleotide sequences",
          "fusion_topology": "unclear",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "the OmniNA-1.7B estimated LL",
          "section_heading": "Discriminating mutation effects in distinct genomic contexts with OmniNA",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            17
          ],
          "doc_item_refs": [
            "#/texts/80"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_002327::0014"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_ad9cb3ec7dd7"
    },
    {
      "model_id": "model_958497e0e4e2",
      "model_name": "OmniNA",
      "record_id": "full_2026-07-06__rec_003394",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_5b610ed9c766",
      "paper_title": "A foundation model for nucleotide sequences.",
      "doi": "10.1093/nar/gkag083",
      "paper_url": "https://doi.org/10.1093/nar/gkag083",
      "route_count": 5,
      "configuration_count": 4,
      "family_counts": {
        "discrete_biological_symbol_stream": 3,
        "text_native_token_stream": 2
      },
      "subtype_counts": {
        "native_biological_token_stream": 3,
        "serialized_biological_context_or_ordered_profile": 1,
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "native_biological_token_stream",
      "modalities": [
        "RNA sequence",
        "nucleotide sequence",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "paired_alignment_supervision",
        "semantic_annotation"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003394_figure_004.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003394_190500e27f28/figure_004.png",
        "figure_index": 4,
        "caption": "Figure 1 . Sc hematic of OmniNA. ( A ) Nucleotide sequences, sourced from the NCBI NT database, are amalgamated with their corresponding text annotations for model pre-training. The model, featuring a transformer-based decoder, undergoes pre-training through an autoregressive approach. Fine-tuning is e x ecuted across 23 tasks in a QA frame w ork. ( B ) Ov ervie w of the downstream tasks encompassing nucleotide sequence detection, ranging from tax onom y classification to genomic element detection.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific schematic about a genomic foundation model workflow and downstream genomics interpretation.\n\nPanel A/top: Shows a pipeline for “OmniNA” or similar genomic sequence model training. Left box labeled “NCBI nucleotide sequence database” states “1076.2 billion bases” and “91.7 million nucleotide sequences” spanning many species, with a phylogenetic/tree-like icon. Middle “Data processing” panel shows instruction-style formatting with “Prefix,” “Input,” and “Response,” converting nucleotide sequences into annotated text examples, e.g. a Homo sapiens gene description. Right “Large-scale pre-training” panel shows autoregressive training with input/output nucleotide tokens, a transformer decoder, and “1.7 billion parameters.”\n\nMiddle: “Stage2: Fine-tuning for genomic tasks” shows multiple genomic task inputs, including raw DNA sequence, mutation notation such as `CCATCGG[G>T]AATCGGC`, and gRNA/target RNA sequence pairs. These feed into “The pre-trained OmniNA,” represented as a transformer-like sequence model. Output task examples include promoter classification, pathogenic/benign mutation classification, and gRNA off-target prediction, with answers such as “No,” “Pathogenic,” and “Yes.”\n\nPanel B/bottom: A biological schematic of chromatin and regulatory genomics. Left shows nucleosomes/chromatin fibers with labeled “Transcription factor,” “Chromosome accessibility,” and histone components H2A, H2B, H3, H4, plus “Histone occupation.” Center/top shows a phylogenetic tree labeled “Species differentiation,” with groups “Bacteria,” “Archaea,” and “Eukaryota.” Right shows a DNA double helix and regulatory/functional elements labeled “Translation initiation site,” “DNA methylation,” “poly(A) signal,” “Splice site,” “Enhancer,” and “Promoter.” Bottom caption reads “Functional genomic element characterization.”\n\nOverall finding/concept: The figure presents a genomic language-model workflow: large-scale nucleotide database collection, instruction-style data processing, autoregressive transformer pre-training, task-specific fine-tuning, and downstream prediction or characterization of genomic functions, mutations, species differences, chromatin features, and regulatory elements.",
        "page_no": 6,
        "sha256": "b2e3e2144df6266f43e084c2d97ad80f5dbe291606388e66f30ed996d1c0fba6",
        "pixel_width": 868,
        "pixel_height": 1112,
        "crop_box": {
          "x": 0.013,
          "y": 0.0,
          "width": 0.625,
          "height": 0.206
        },
        "panel_label": "A",
        "visible_input_object": "NCBI NT nucleotide sequences with accompanying text annotations",
        "visible_model_interface": "Instruction-style pretraining prompt with Prefix/Input/Response fields",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the source database box and the data-processing prompt block, which together show the grounded input route from NT sequences and annotations into the pretraining format. It excludes the downstream task panels and the output-focused pretraining decoder area.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_df861ea65ff9",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "nucleotide sequences from the Nucleotide (NT) Database",
          "actual_model_visible_form": "tokenized nucleotide sequence chunks"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_e5dcb7e44939",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "textual annotations from the Nucleotide (NT) Database",
          "actual_model_visible_form": "tokenized annotation chunks"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_4b44933472fc",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "nucleotide sequences",
          "actual_model_visible_form": "prompt composed of instruction, nucleotide sequence, test question, and expected answer"
        }
      ],
      "routes": [
        {
          "route_id": "route_df861ea65ff9",
          "configuration_id": "config_67c7221374cf",
          "route_label": "Pretraining on NCBI NT sequences",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "sequence data with textual annotations as feature-ordered input",
          "source_object_verbatim": "nucleotide sequences from the Nucleotide (NT) Database",
          "source_object_normalized": "NCBI NT nucleotide sequences",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sliding window approach",
            "truncated to 3000 bp",
            "excluded sequences with length < 200 bp",
            "filtered out sequences containing two or more consecutive N nucleotides",
            "sentence-based chunks",
            "tokenized using the trained tokenizer"
          ],
          "model_visible_form_verbatim": "tokenized nucleotide sequence chunks",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "prompts to organize the sequence data with the text annotations as feature-ordered input for model pre-training",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "We curated sequence data along with corresponding textual annotations from the Nucleotide (NT) Database of the National Center for Biotechnology Information (NCBI) up to April, 2023. The NT database is composed of nucleotide sequences in FASTA format, accompanied by text annotations including organism names, taxonomic classifications, gene and gene product names, and functional descriptions. These annotations provide context for training, linking the sequences to relevant biological information. The annotations obtained from the NCBI NT database comply with the standards established by the International Nucleotide Sequence Database Collaboration [ 10 ]. This framework ensures consistency through controlled vocabularies and a standardized taxonomic hierarchy integrated with the NCBI Taxonomy database. The data encompass genomic DNA and RNA sequences derived from a wide range of species, enabling a diverse foundation for self-supervised learning tasks. To characterize the annotation landscape used during pre-training, we quantified the distribution of sequence types, taxonomic labels, and token embeddings derived from the curated annotations. The composition of sequence types and taxonomic ranks across the corpus is summarized in Supplementary Table S1 and visualized in Supplementary Fig. S1 A -C, providing an overview of the species diversity represented in the dataset. We further parsed all textual annotations into 20 token categories, forming a structured label space for model training. These token embeddings were analyzed using UMAP, revealing coherent clustering patterns that reflect shared biological or functional themes ( Supplementary Fig. S1 D). A detailed description of each token cluster, including representative themes and example annotations, is provided in Supplementary Table S2 . Employing a sliding window approach, sequences longer than 3000 bp were truncated to 3000 bp, while sequences with a length < 200 bp were excluded. We filtered out sequences containing two or more consecutive N nucleotides. Subsequently, a total of 91 732 311 sequences were included, collectively comprising 1 076 183 408 570 bases. The textual annotations totaled 197 089 040 words. We randomly selected 98% of the data for training and allocated 1% each for validation and testing. We designed prompts to organize the sequence data with textual annotations as feature-ordered input for model pre-training. As shown in Supplementary Table S3 , the prompt is composed of three modules: (i) an instruction, (ii) a nucleotide sequence, and (iii) an annotation.",
          "section_heading": "Data processing for OmniNA pre-training",
          "supporting_figure_or_table": "Supplementary Table S3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/texts/26",
            "#/texts/28",
            "#/texts/31",
            "#/texts/32",
            "#/texts/33",
            "#/texts/34",
            "#/texts/35",
            "#/texts/36",
            "#/texts/37"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003394::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0053",
            "dense::full_2026-07-06__rec_003394::0054"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e5dcb7e44939",
          "configuration_id": "config_67c7221374cf",
          "route_label": "Pretraining on NCBI NT annotations",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "sequence data with textual annotations as feature-ordered input",
          "source_object_verbatim": "textual annotations from the Nucleotide (NT) Database",
          "source_object_normalized": "NCBI NT textual annotations",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "parsed into 20 token categories",
            "sentence-based chunks",
            "tokenized using the trained tokenizer"
          ],
          "model_visible_form_verbatim": "tokenized annotation chunks",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "prompts to organize the sequence data with textual annotations as feature-ordered input for model pre-training",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "semantic_annotation",
          "input_status": "actual_model_input",
          "evidence_quote": "We curated sequence data along with corresponding textual annotations from the Nucleotide (NT) Database of the National Center for Biotechnology Information (NCBI) up to April, 2023. The NT database is composed of nucleotide sequences in FASTA format, accompanied by text annotations including organism names, taxonomic classifications, gene and gene product names, and functional descriptions. These annotations provide context for training, linking the sequences to relevant biological information. The annotations obtained from the NCBI NT database comply with the standards established by the International Nucleotide Sequence Database Collaboration [ 10 ]. This framework ensures consistency through controlled vocabularies and a standardized taxonomic hierarchy integrated with the NCBI Taxonomy database. The data encompass genomic DNA and RNA sequences derived from a wide range of species, enabling a diverse foundation for self-supervised learning tasks. To characterize the annotation landscape used during pre-training, we quantified the distribution of sequence types, taxonomic labels, and token embeddings derived from the curated annotations. The composition of sequence types and taxonomic ranks across the corpus is summarized in Supplementary Table S1 and visualized in Supplementary Fig. S1 A -C, providing an overview of the species diversity represented in the dataset. We further parsed all textual annotations into 20 token categories, forming a structured label space for model training. These token embeddings were analyzed using UMAP, revealing coherent clustering patterns that reflect shared biological or functional themes ( Supplementary Fig. S1 D). A detailed description of each token cluster, including representative themes and example annotations, is provided in Supplementary Table S2 . Employing a sliding window approach, sequences longer than 3000 bp were truncated to 3000 bp, while sequences with a length < 200 bp were excluded. We filtered out sequences containing two or more consecutive N nucleotides. Subsequently, a total of 91 732 311 sequences were included, collectively comprising 1 076 183 408 570 bases. The textual annotations totaled 197 089 040 words. We randomly selected 98% of the data for training and allocated 1% each for validation and testing. We designed prompts to organize the sequence data with textual annotations as feature-ordered input for model pre-training. As shown in Supplementary Table S3 , the prompt is composed of three modules: (i) an instruction, (ii) a nucleotide sequence, and (iii) an annotation.",
          "section_heading": "Data processing for OmniNA pre-training",
          "supporting_figure_or_table": "Supplementary Table S3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/texts/26",
            "#/texts/28",
            "#/texts/31",
            "#/texts/32",
            "#/texts/33",
            "#/texts/34",
            "#/texts/35",
            "#/texts/36",
            "#/texts/37"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003394::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0053",
            "dense::full_2026-07-06__rec_003394::0054"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4b44933472fc",
          "configuration_id": "config_2e7a71d91547",
          "route_label": "Multi-task QA fine-tuning on nucleotide sequences",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "question-answer (QA) paradigm",
          "source_object_verbatim": "nucleotide sequences",
          "source_object_normalized": "nucleotide sequence",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "prompts composed of four components",
            "autoregressive learning"
          ],
          "model_visible_form_verbatim": "prompt composed of instruction, nucleotide sequence, test question, and expected answer",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "each task instance includes a prompt composed of four components",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "In the fine-tuning phase, the model was further trained using supervised learning on task-specific datasets. Here, the model was trained to generate text annotations (e.g. biological responses or answers to questions) given nucleotide sequences as input. We adopted an autoregressive learning framework, where the model predicted the next token, with the targets being annotation tokens rather than nucleotide bases. As shown in Supplementary Table S3 , each task instance includes a prompt composed of four components: (i) an instruction, (ii) a nucleotide sequence, (iii) a test question, and (iv) an expected answer. The first two parts are fixed for all tasks, while the question and answer are task-specific. In the fine-tuning stage, only the answer was trained in an autoregressive manner. The model is fine-tuned using the AdamW optimizer for 2 epochs [ 18 ], with the following hyper-parameters: β 1 = 0.9 and β 2 = 0.999. The learning rate and batch size are varied with the size of the model. The learning rate is decayed with a cosine schedule and a warmup ratio of 0.03.",
          "section_heading": "Apply the model for multi-task fine-tuning",
          "supporting_figure_or_table": "Supplementary Table S3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/52"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003394::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c0eec7eef6c6",
          "configuration_id": "config_c1b046f46613",
          "route_label": "Sequence-only deployment on nucleotide sequences",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "sequence-only inputs",
          "source_object_verbatim": "nucleotide sequences",
          "source_object_normalized": "nucleotide sequence",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inference pathway"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "annotation tokens are neither required nor used",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "annotation tokens are neither required nor used",
          "section_heading": "Description of the downstream tasks",
          "supporting_figure_or_table": "Supplementary Fig. S4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/47",
            "#/texts/50"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003394::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0004"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ba08dd614cf3",
          "configuration_id": "config_39587c399935",
          "route_label": "cfRNA pancreas tumor-vs-normal classification",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "cell-free RNA sequences of pancreas patients and normal persons",
          "source_object_verbatim": "cell-free RNA sequences from GSE136651",
          "source_object_normalized": "cfRNA sequences from GSE136651",
          "source_modality_normalized": "RNA sequence",
          "transformation_chain_verbatim": [
            "adapter trimming",
            "randomly sample 30-bp fragments",
            "concatenate 70 fragments to form one input sequence per patient",
            "1000 inputs per patient are generated",
            "five-fold cross-validation with an 80/20 split stratified by patient"
          ],
          "model_visible_form_verbatim": "concatenated RNA fragments",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "after adapter trimming, we randomly sample 30-bp fragments and concatenate 70 fragments to form one input sequence per patient",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "11 categories. Negative examples were randomly sampled from protein-coding genes in NCBI RefSeq bacterial genomes and filtered by BLAST against CARD/VFDB to remove significantly high-similarity hits. Phage -host labels of bacteriophage were collected from PhageScope [ 38 ]. We retained 12 host taxa with ≥ 1000 sequences each. Host labels of eukaryotic viruses were collected from DeepHost dataset [ 39 ]. We retained 144 host taxa with ≥ 50 sequences each. For both bacterial functional sequence and host specificity prediction tasks, the data were split into training, validation, and test sets in an 8:1:1 ratio. cell-free RNA sequences of pancreas patients and normal persons were collected from GSE136651. After adapter trimming, we randomly sample 30-bp fragments and concatenate 70 fragments to form one input sequence per patient; 1000 inputs per patient are generated. Model logits are averaged across a patient's inputs (threshold 0.5) to obtain the patient-level label. We conduct five-fold cross-validation with an 80/20 split stratified by patient, ensuring no patient overlap between training and test folds.",
          "section_heading": "Description of the downstream tasks",
          "supporting_figure_or_table": "Supplementary Fig. S9 F",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            10
          ],
          "doc_item_refs": [
            "#/texts/47",
            "#/texts/50",
            "#/texts/90"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003394::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0045"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_aaacf24f224c"
    },
    {
      "model_id": "model_eb69df38d690",
      "model_name": "OmniNA-1.7B",
      "record_id": "full_2026-07-06__rec_003394",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_5b610ed9c766",
      "paper_title": "A foundation model for nucleotide sequences.",
      "doi": "10.1093/nar/gkag083",
      "paper_url": "https://doi.org/10.1093/nar/gkag083",
      "route_count": 16,
      "configuration_count": 16,
      "family_counts": {
        "discrete_biological_symbol_stream": 16
      },
      "subtype_counts": {
        "native_biological_token_stream": 16
      },
      "families": [
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream"
      ],
      "primary_subtype": "native_biological_token_stream",
      "modalities": [
        "nucleotide sequence"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003394_figure_008.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003394_190500e27f28/figure_008.png",
        "figure_index": 8,
        "caption": "Figure 5. Predicted mutation effects of clinically and functionally relevant missense variants. ( A ) Schematic diagram of the mutation effect score calculation. ( B ) Bar plot illustrates the correlation between mutation effect scores and the proportion of pathogenic and non-pathogenic mutations. Line plot depicts the OR between pathogenic and benign mutations. ( C ) Bar plot shows the mean predicted mutation effects for each gRNA position, with error bars representing the 95% confidence interval (CI) of the mutation score. ( D ) Bar plot shows the mean mutation effect score for mutations in the domain region compared with the non-domain region of BRCA1 coding gene. The error bar represents the 95% CI of the distributions of mutation scores. ( E ) Upper panel, schematic representation of the domains and non-domain protein coding regions of BRCA1 ; middle panel, maximum mutation effect score observed for each nucleotide base; lower panel, heatmap illustrating mutation effect scores across all possible 3 × 4 missense variants. ( F ) AlphaFold prediction of the three-dimensional str uct ure of the BRCA1 protein. The residues are colored by the predicted mutation effect scores. ( G ) Bar plot shows the mean mutation effect score for mutations in the domain region compared with the non-domain region of CHD7 coding gene. The error bar represents the 95% CI of the distributions of mutation scores. ( H ) Upper panel, schematic representation of the domains and non-domain protein coding regions of CHD7 ; middle panel, maximum mutation effect score observed for each nucleotide base; lower panel, heatmap illustrating mutation effect scores across all possible 3 × 4 missense variants. ( I ) AlphaFold prediction of the three-dimensional str uct ure of the CHD7 protein. The residues are colored by the predicted mutation effect scores.",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel genomics/protein biology figure about OmniNA predicted mutation effect scores.\n\nVisible content:\n- Panel A: Schematic definition of “OmniNA mutation effect score” for a single nucleotide variant `chr17:31221843:A>G`. Shows wild-type and mutant DNA sequence snippets (`sWT`, `smut`) with the altered base highlighted, plus a formula comparing log likelihoods of wild-type and mutant sequences.\n- Panel B: Two bar/line plots for PAS and TIS. X-axis is predicted mutation effect, left y-axis is percentage, right y-axis is odds ratio. Bars compare benign vs pathogenic percentages, with a red odds-ratio curve increasing as predicted mutation effect increases.\n- Panel C: Horizontal bar chart comparing predicted vs experimental mutation effect across gRNA sites 1-23. Purple bars show predicted mutation effect with error bars; red line shows experimental mutation effect.\n- Panels D and G: Bar charts comparing predicted mutation effect in domain vs non-domain regions for BRCA1 and CHD7, with `p-value < 1e-5`.\n- Panel E: BRCA1 amino acid sequence heatmap from residue 1 to 1863. Annotated protein domains include Zinc finger, BRCT1, and BRCT2. Heatmap color scale ranges from blue negative predicted mutation effect to red positive predicted mutation effect. Rows are labeled by bases.\n- Panel F: Protein structure visualization for BRCA1 regions/domains, with labeled Zinc finger, BRCT1, and BRCT2. Structure is colored blue to red according to predicted mutation effect.\n- Panel H: CHD7 amino acid sequence heatmap from residue 1 to 2961. Annotated domains include Chromo 1-2, ATP-binding, Helicase C-terminal, SANT, and BRK regions. Uses the same blue-to-red predicted mutation effect scale.\n- Panel I: Protein structure visualization for CHD7, labeled Chromo 1-2 and ATP-binding, colored by predicted mutation effect.\n\nBiological source objects:\n- DNA sequence variants.\n- gRNA sites.\n- BRCA1 protein sequence and domains.\n- CHD7 protein sequence and domains.\n- Protein structural models/representations for BRCA1 and CHD7 regions.\n\nTransformations/model interface:\n- Sequence mutation is evaluated by a model-derived log-likelihood difference between wild-type and mutant sequences.\n- Predicted mutation effects are compared with pathogenicity categories, experimental mutation effects, protein domains, amino acid positions, and structural locations.\n\nVisible findings:\n- Higher predicted mutation effect is associated with higher pathogenic odds ratio in PAS and TIS plots.\n- Predicted mutation effects broadly track experimental mutation effects across gRNA sites.\n- Protein domain regions in BRCA1 and CHD7 show higher predicted mutation effect than non-domain regions.\n- Elevated mutation effect regions appear concentrated around annotated functional domains and corresponding protein structural regions.",
        "page_no": 12,
        "sha256": "a7ceb7c87469569e8b30c0aea15ad2cfaa21625dbe24daf2eccfb6ce09495bea",
        "pixel_width": 965,
        "pixel_height": 963,
        "crop_box": {
          "x": 0.03,
          "y": 0.01,
          "width": 0.56,
          "height": 0.21
        },
        "panel_label": "A",
        "visible_input_object": "Wild-type and mutated nucleotide sequences for chr17:31221843:A>G, with the mutation arrow and log-likelihood formula visible",
        "visible_model_interface": "Sequence-only nucleotide input scored by a log-likelihood difference between sWT and smut",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "Panel A is the smallest self-contained region that shows the source object (WT/mut nucleotide strings), the mutation transformation arrow, and the scoring interface/formula needed to ground saturation mutation analysis. It excludes downstream result panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_e56839df26bb",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "mutated and wild-type nucleotide sequences",
          "actual_model_visible_form": "mutated ( s mut ) and wild-type ( s WT ) nucleotide sequences"
        }
      ],
      "routes": [
        {
          "route_id": "route_e56839df26bb",
          "configuration_id": "config_19b05e7c85c9",
          "route_label": "Saturation mutation analysis",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "saturation mutation analysis",
          "source_object_verbatim": "mutated and wild-type nucleotide sequences",
          "source_object_normalized": "paired mutated and wild-type nucleotide sequences",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "compare log-likelihood against the corresponding wild-type nucleotide sequence",
            "estimate the LL with the OmniNA-1.7B model"
          ],
          "model_visible_form_verbatim": "mutated ( s mut ) and wild-type ( s WT ) nucleotide sequences",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "mutation effects using log-likelihood (LL) differences",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "We quantify mutation effects using log-likelihood (LL) differences between mutated ( s mut ) and wild-type ( s WT ) nucleotide sequences:",
          "section_heading": "Saturation mutation analysis",
          "supporting_figure_or_table": "Figure 5A",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/57",
            "#/texts/58",
            "#/texts/59",
            "#/texts/60",
            "#/texts/61",
            "#/texts/62",
            "#/texts/63"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_69cbd9b2d88e",
          "configuration_id": "config_70f4f57394af",
          "route_label": "DNA element identification",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "human DNA element identification",
          "source_object_verbatim": "human DNA elements",
          "source_object_normalized": "human DNA elements",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "input sequences were centered on the region of interest",
            "resized by truncation or padding to meet input length requirement"
          ],
          "model_visible_form_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "To assess the generalization capabilities of OmniNA on nucleotide sequence tasks, we conducted fine-tuning experiments on 23 diverse tasks encompassing genome and species detection. All downstream fine-tuning and evaluations reported here use sequence-only inputs, with no annotations supplied during training or testing; hence, the reported accuracy reflects annotation-free deployment. OmniNA generally achieved better F1 scores than benchmark methods (Fig. 4 ). Specifically, for tasks related to human DNA element identification, OmniNA achieved comparable or superior performance as compared with the benchmark methods (Fig. 4 A). Besides, OmniNA-1.7B excelled in PAS, TIS signal, and splice site detection, achieving remarkable F1 scores of 0.88, 0.97, and 0.98, respectively (Fig. 4 B and C). The superior performance of the OmniNA-1.7B model can also be observed for gRNA off-target detection with an F1 score of 0.98 (Fig. 4 D and Supplementary Fig. S8 A). In methylated region detection tasks, the OmniNA-1.7B model achieved comparable F1 and higher precision scores across various cell types compared with the 66 M and 330 M models (Fig. 4 E). Additionally, for chromatin feature prediction, OmniNA-1.7B delivered high-accuracy predictions for chromatin accessibility and histone modification, surpassing current state-of-theart methods (Fig. 4 F, Supplementary Fig. S8 B and C, and Supplementary Table S8 ). Our method further demonstrated high performance in the prediction for TF occupation, achieving a median F1 score of 0.73 across all TF targets (Fig. 4 F, Supplementary Fig. S8 D, and Supplementary Table S8 ). Besides, the model's performance on pathogenic variation detection tasks improved with increasing model size (Fig. 4 G). We compared the OmniNA-1.7B model against the EVE method. Across 48 genes, OmniNA-1.7B achieved higher F1 scores in 27 genes compared with EVE (Fig. 4 H). Overall, OmniNA1.7B achieved a significantly higher mean F1 score of 0.90 compared with 0.81 for EVE (Fig. 4 I; two-sided t -test, P -value = 0.011). OmniNA also exhibited proficiency in species detection tasks (Fig. 4 J -L). In the pathogenic virus detection task (Fig. 4 J), the F1 scores of OmniNA-1.7B surpassed those of benchmark methods. Furthermore, on bacteria taxonomy classification tasks related to class, family, order, and phylum (Fig. 4 K and L and Supplementary Fig. S8 E), OmniNA1.7B outperformed the benchmark model. We also evaluated the model on non-B DNA structure classification, where OmniNA-1.7B achieved an Accuracy@1 of 0.78 and Accuracy@3 of 0.88 ( Supplementary Fig. S9 A). Furthermore, we extended our evaluation to microbiology tasks. For bacterial functional-sequence sub-classification, including ARGs and VGs, we report top-k performance across five ARG mechanisms and 11 VG types. OmniNA-1.7B achieved an Accuracy@1 = 0.59 and F1@1 = 0.56, which improved to Accuracy@3 = 0.92 / F1@3 = 0.54 for ARG mechanisms ( Supplementary Fig. S9 B and Supplementary Table S9 ). For VG classification, the model reached Accuracy@1 = 0.93 and F1@1 = 0.87 and further improved to Accuracy@3 = 0.96 / F1@3 = 0.63 ( Supplementary Fig. S9 C and Supplementary Table S9 ). For host specificity prediction, OmniNA-1.7B accurately classified both bacteriophage and eukaryotic virus hosts. In bacteriophage host prediction (12-way), the model achieved Accuracy@1 = 0.73 and F1@1 = 0.78, improving to Accuracy@3 = 0.97 and F1@3 = 0.60 under top-3 evaluation ( Supplementary Fig. S9 D and Supplementary Table S9 ). In eukaryotic virus host prediction (15-way), it achieved Accuracy@1 = 0.78 and F1@1 = 0.69, with Accuracy@3 = 0.91 and F1@3 = 0.51 ( Supplementary Fig. S9 E and Supplementary Table S9 ). On cfRNA pancreas tumor vs. normal distinguishing task, the model achieved near-perfect discrimination (Accuracy/F1/Precision/Recall/MCC ≈ 1.0), and its learned embeddings showed clear, complete separation between tumor and normal samples in UMAP space ( Supplementary Fig. S9 F). All detailed quantitative benchmark results are listed in Supplementary Table S10 .",
          "section_heading": "Benchmark OmniNA with task-specific deep learning methods",
          "supporting_figure_or_table": "Fig. 4 A",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            11,
            13
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/texts/107",
            "#/texts/90",
            "#/texts/94",
            "#/texts/97"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0024",
            "dense::full_2026-07-06__rec_003394::0046"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e5ae344ee988",
          "configuration_id": "config_5186cec1eb6f",
          "route_label": "PAS signal detection",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "PAS signal",
          "source_object_verbatim": "PAS signal",
          "source_object_normalized": "PAS signal",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "input sequences were centered on the region of interest",
            "resized by truncation or padding to meet input length requirement"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "To assess the generalization capabilities of OmniNA on nucleotide sequence tasks, we conducted fine-tuning experiments on 23 diverse tasks encompassing genome and species detection. All downstream fine-tuning and evaluations reported here use sequence-only inputs, with no annotations supplied during training or testing; hence, the reported accuracy reflects annotation-free deployment. OmniNA generally achieved better F1 scores than benchmark methods (Fig. 4 ). Specifically, for tasks related to human DNA element identification, OmniNA achieved comparable or superior performance as compared with the benchmark methods (Fig. 4 A). Besides, OmniNA-1.7B excelled in PAS, TIS signal, and splice site detection, achieving remarkable F1 scores of 0.88, 0.97, and 0.98, respectively (Fig. 4 B and C). The superior performance of the OmniNA-1.7B model can also be observed for gRNA off-target detection with an F1 score of 0.98 (Fig. 4 D and Supplementary Fig. S8 A). In methylated region detection tasks, the OmniNA-1.7B model achieved comparable F1 and higher precision scores across various cell types compared with the 66 M and 330 M models (Fig. 4 E). Additionally, for chromatin feature prediction, OmniNA-1.7B delivered high-accuracy predictions for chromatin accessibility and histone modification, surpassing current state-of-theart methods (Fig. 4 F, Supplementary Fig. S8 B and C, and Supplementary Table S8 ). Our method further demonstrated high performance in the prediction for TF occupation, achieving a median F1 score of 0.73 across all TF targets (Fig. 4 F, Supplementary Fig. S8 D, and Supplementary Table S8 ). Besides, the model's performance on pathogenic variation detection tasks improved with increasing model size (Fig. 4 G). We compared the OmniNA-1.7B model against the EVE method. Across 48 genes, OmniNA-1.7B achieved higher F1 scores in 27 genes compared with EVE (Fig. 4 H). Overall, OmniNA1.7B achieved a significantly higher mean F1 score of 0.90 compared with 0.81 for EVE (Fig. 4 I; two-sided t -test, P -value = 0.011). OmniNA also exhibited proficiency in species detection tasks (Fig. 4 J -L). In the pathogenic virus detection task (Fig. 4 J), the F1 scores of OmniNA-1.7B surpassed those of benchmark methods. Furthermore, on bacteria taxonomy classification tasks related to class, family, order, and phylum (Fig. 4 K and L and Supplementary Fig. S8 E), OmniNA1.7B outperformed the benchmark model. We also evaluated the model on non-B DNA structure classification, where OmniNA-1.7B achieved an Accuracy@1 of 0.78 and Accuracy@3 of 0.88 ( Supplementary Fig. S9 A). Furthermore, we extended our evaluation to microbiology tasks. For bacterial functional-sequence sub-classification, including ARGs and VGs, we report top-k performance across five ARG mechanisms and 11 VG types. OmniNA-1.7B achieved an Accuracy@1 = 0.59 and F1@1 = 0.56, which improved to Accuracy@3 = 0.92 / F1@3 = 0.54 for ARG mechanisms ( Supplementary Fig. S9 B and Supplementary Table S9 ). For VG classification, the model reached Accuracy@1 = 0.93 and F1@1 = 0.87 and further improved to Accuracy@3 = 0.96 / F1@3 = 0.63 ( Supplementary Fig. S9 C and Supplementary Table S9 ). For host specificity prediction, OmniNA-1.7B accurately classified both bacteriophage and eukaryotic virus hosts. In bacteriophage host prediction (12-way), the model achieved Accuracy@1 = 0.73 and F1@1 = 0.78, improving to Accuracy@3 = 0.97 and F1@3 = 0.60 under top-3 evaluation ( Supplementary Fig. S9 D and Supplementary Table S9 ). In eukaryotic virus host prediction (15-way), it achieved Accuracy@1 = 0.78 and F1@1 = 0.69, with Accuracy@3 = 0.91 and F1@3 = 0.51 ( Supplementary Fig. S9 E and Supplementary Table S9 ). On cfRNA pancreas tumor vs. normal distinguishing task, the model achieved near-perfect discrimination (Accuracy/F1/Precision/Recall/MCC ≈ 1.0), and its learned embeddings showed clear, complete separation between tumor and normal samples in UMAP space ( Supplementary Fig. S9 F). All detailed quantitative benchmark results are listed in Supplementary Table S10 .",
          "section_heading": "Description of the downstream tasks",
          "supporting_figure_or_table": "Fig. 4 B and C",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10
          ],
          "doc_item_refs": [
            "#/texts/90"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0025"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_868df09e4253",
          "configuration_id": "config_8bc02888dd08",
          "route_label": "TIS signal detection",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "TIS signal",
          "source_object_verbatim": "TIS signal",
          "source_object_normalized": "TIS signal",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "input sequences were centered on the region of interest",
            "resized by truncation or padding to meet input length requirement"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "TIS signal",
          "section_heading": "Description of the downstream tasks",
          "supporting_figure_or_table": "Fig. 4 B and C",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            10
          ],
          "doc_item_refs": [
            "#/texts/47",
            "#/texts/54",
            "#/texts/55",
            "#/texts/57",
            "#/texts/58",
            "#/texts/59",
            "#/texts/60",
            "#/texts/61"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0026"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f0f7e917ef10",
          "configuration_id": "config_a35fa6d72c9f",
          "route_label": "Splice site detection",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "splice site detection",
          "source_object_verbatim": "splice site detection",
          "source_object_normalized": "splice site detection",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "input sequences were centered on the region of interest",
            "resized by truncation or padding to meet input length requirement"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "splice site detection",
          "section_heading": "Description of the downstream tasks",
          "supporting_figure_or_table": "Fig. 4 B and C",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            10,
            11
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/texts/54",
            "#/texts/55",
            "#/texts/90",
            "#/texts/94",
            "#/texts/97"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0027"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_7a621ea3343b",
          "configuration_id": "config_7a991b79e1f1",
          "route_label": "gRNA off-target detection",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "gRNA off-target detection",
          "source_object_verbatim": "gRNA off-target effects",
          "source_object_normalized": "gRNA off-target effects",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "input sequences were centered on the region of interest",
            "resized by truncation or padding to meet input length requirement"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "gRNA off-target detection",
          "section_heading": "Results",
          "supporting_figure_or_table": "Fig. 4 D and Supplementary Fig. S8 A",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10
          ],
          "doc_item_refs": [
            "#/texts/90"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0028"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0eff9dfb4512",
          "configuration_id": "config_75afefd4f701",
          "route_label": "Methylated region detection",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "methylated region detection tasks",
          "source_object_verbatim": "methylated region detection",
          "source_object_normalized": "methylated region detection",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inputs",
            "no annotations supplied during training or testing"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "methylated region detection",
          "section_heading": "Results",
          "supporting_figure_or_table": "Fig. 4 E",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            11
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/texts/90",
            "#/texts/94",
            "#/texts/97"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0029"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1d8ba8ebb625",
          "configuration_id": "config_9876f232b0a8",
          "route_label": "Chromatin feature prediction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "chromatin feature prediction",
          "source_object_verbatim": "chromatin feature prediction",
          "source_object_normalized": "chromatin feature prediction",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inputs",
            "no annotations supplied during training or testing"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "To assess the generalization capabilities of OmniNA on nucleotide sequence tasks, we conducted fine-tuning experiments on 23 diverse tasks encompassing genome and species detection. All downstream fine-tuning and evaluations reported here use sequence-only inputs, with no annotations supplied during training or testing; hence, the reported accuracy reflects annotation-free deployment. OmniNA generally achieved better F1 scores than benchmark methods (Fig. 4 ). Specifically, for tasks related to human DNA element identification, OmniNA achieved comparable or superior performance as compared with the benchmark methods (Fig. 4 A). Besides, OmniNA-1.7B excelled in PAS, TIS signal, and splice site detection, achieving remarkable F1 scores of 0.88, 0.97, and 0.98, respectively (Fig. 4 B and C). The superior performance of the OmniNA-1.7B model can also be observed for gRNA off-target detection with an F1 score of 0.98 (Fig. 4 D and Supplementary Fig. S8 A). In methylated region detection tasks, the OmniNA-1.7B model achieved comparable F1 and higher precision scores across various cell types compared with the 66 M and 330 M models (Fig. 4 E). Additionally, for chromatin feature prediction, OmniNA-1.7B delivered high-accuracy predictions for chromatin accessibility and histone modification, surpassing current state-of-theart methods (Fig. 4 F, Supplementary Fig. S8 B and C, and Supplementary Table S8 ). Our method further demonstrated high performance in the prediction for TF occupation, achieving a median F1 score of 0.73 across all TF targets (Fig. 4 F, Supplementary Fig. S8 D, and Supplementary Table S8 ). Besides, the model's performance on pathogenic variation detection tasks improved with increasing model size (Fig. 4 G). We compared the OmniNA-1.7B model against the EVE method. Across 48 genes, OmniNA-1.7B achieved higher F1 scores in 27 genes compared with EVE (Fig. 4 H). Overall, OmniNA1.7B achieved a significantly higher mean F1 score of 0.90 compared with 0.81 for EVE (Fig. 4 I; two-sided t -test, P -value = 0.011). OmniNA also exhibited proficiency in species detection tasks (Fig. 4 J -L). In the pathogenic virus detection task (Fig. 4 J), the F1 scores of OmniNA-1.7B surpassed those of benchmark methods. Furthermore, on bacteria taxonomy classification tasks related to class, family, order, and phylum (Fig. 4 K and L and Supplementary Fig. S8 E), OmniNA1.7B outperformed the benchmark model. We also evaluated the model on non-B DNA structure classification, where OmniNA-1.7B achieved an Accuracy@1 of 0.78 and Accuracy@3 of 0.88 ( Supplementary Fig. S9 A). Furthermore, we extended our evaluation to microbiology tasks. For bacterial functional-sequence sub-classification, including ARGs and VGs, we report top-k performance across five ARG mechanisms and 11 VG types. OmniNA-1.7B achieved an Accuracy@1 = 0.59 and F1@1 = 0.56, which improved to Accuracy@3 = 0.92 / F1@3 = 0.54 for ARG mechanisms ( Supplementary Fig. S9 B and Supplementary Table S9 ). For VG classification, the model reached Accuracy@1 = 0.93 and F1@1 = 0.87 and further improved to Accuracy@3 = 0.96 / F1@3 = 0.63 ( Supplementary Fig. S9 C and Supplementary Table S9 ). For host specificity prediction, OmniNA-1.7B accurately classified both bacteriophage and eukaryotic virus hosts. In bacteriophage host prediction (12-way), the model achieved Accuracy@1 = 0.73 and F1@1 = 0.78, improving to Accuracy@3 = 0.97 and F1@3 = 0.60 under top-3 evaluation ( Supplementary Fig. S9 D and Supplementary Table S9 ). In eukaryotic virus host prediction (15-way), it achieved Accuracy@1 = 0.78 and F1@1 = 0.69, with Accuracy@3 = 0.91 and F1@3 = 0.51 ( Supplementary Fig. S9 E and Supplementary Table S9 ). On cfRNA pancreas tumor vs. normal distinguishing task, the model achieved near-perfect discrimination (Accuracy/F1/Precision/Recall/MCC ≈ 1.0), and its learned embeddings showed clear, complete separation between tumor and normal samples in UMAP space ( Supplementary Fig. S9 F). All detailed quantitative benchmark results are listed in Supplementary Table S10 .",
          "section_heading": "Results",
          "supporting_figure_or_table": "Fig. 4 F, Supplementary Fig. S8 B and C, and Supplementary Table S8",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            4,
            5,
            10,
            11,
            12
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/texts/100",
            "#/texts/47",
            "#/texts/54",
            "#/texts/55",
            "#/texts/90"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0030",
            "dense::full_2026-07-06__rec_003394::0031",
            "dense::full_2026-07-06__rec_003394::0032"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_19b8a1ffc223",
          "configuration_id": "config_e8c9ec743ff0",
          "route_label": "TF occupation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "TF occupation",
          "source_object_verbatim": "TF occupation",
          "source_object_normalized": "transcription factor binding",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inputs",
            "no annotations supplied during training or testing"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "To assess the generalization capabilities of OmniNA on nucleotide sequence tasks, we conducted fine-tuning experiments on 23 diverse tasks encompassing genome and species detection. All downstream fine-tuning and evaluations reported here use sequence-only inputs, with no annotations supplied during training or testing; hence, the reported accuracy reflects annotation-free deployment. OmniNA generally achieved better F1 scores than benchmark methods (Fig. 4 ). Specifically, for tasks related to human DNA element identification, OmniNA achieved comparable or superior performance as compared with the benchmark methods (Fig. 4 A). Besides, OmniNA-1.7B excelled in PAS, TIS signal, and splice site detection, achieving remarkable F1 scores of 0.88, 0.97, and 0.98, respectively (Fig. 4 B and C). The superior performance of the OmniNA-1.7B model can also be observed for gRNA off-target detection with an F1 score of 0.98 (Fig. 4 D and Supplementary Fig. S8 A). In methylated region detection tasks, the OmniNA-1.7B model achieved comparable F1 and higher precision scores across various cell types compared with the 66 M and 330 M models (Fig. 4 E). Additionally, for chromatin feature prediction, OmniNA-1.7B delivered high-accuracy predictions for chromatin accessibility and histone modification, surpassing current state-of-theart methods (Fig. 4 F, Supplementary Fig. S8 B and C, and Supplementary Table S8 ). Our method further demonstrated high performance in the prediction for TF occupation, achieving a median F1 score of 0.73 across all TF targets (Fig. 4 F, Supplementary Fig. S8 D, and Supplementary Table S8 ). Besides, the model's performance on pathogenic variation detection tasks improved with increasing model size (Fig. 4 G). We compared the OmniNA-1.7B model against the EVE method. Across 48 genes, OmniNA-1.7B achieved higher F1 scores in 27 genes compared with EVE (Fig. 4 H). Overall, OmniNA1.7B achieved a significantly higher mean F1 score of 0.90 compared with 0.81 for EVE (Fig. 4 I; two-sided t -test, P -value = 0.011). OmniNA also exhibited proficiency in species detection tasks (Fig. 4 J -L). In the pathogenic virus detection task (Fig. 4 J), the F1 scores of OmniNA-1.7B surpassed those of benchmark methods. Furthermore, on bacteria taxonomy classification tasks related to class, family, order, and phylum (Fig. 4 K and L and Supplementary Fig. S8 E), OmniNA1.7B outperformed the benchmark model. We also evaluated the model on non-B DNA structure classification, where OmniNA-1.7B achieved an Accuracy@1 of 0.78 and Accuracy@3 of 0.88 ( Supplementary Fig. S9 A). Furthermore, we extended our evaluation to microbiology tasks. For bacterial functional-sequence sub-classification, including ARGs and VGs, we report top-k performance across five ARG mechanisms and 11 VG types. OmniNA-1.7B achieved an Accuracy@1 = 0.59 and F1@1 = 0.56, which improved to Accuracy@3 = 0.92 / F1@3 = 0.54 for ARG mechanisms ( Supplementary Fig. S9 B and Supplementary Table S9 ). For VG classification, the model reached Accuracy@1 = 0.93 and F1@1 = 0.87 and further improved to Accuracy@3 = 0.96 / F1@3 = 0.63 ( Supplementary Fig. S9 C and Supplementary Table S9 ). For host specificity prediction, OmniNA-1.7B accurately classified both bacteriophage and eukaryotic virus hosts. In bacteriophage host prediction (12-way), the model achieved Accuracy@1 = 0.73 and F1@1 = 0.78, improving to Accuracy@3 = 0.97 and F1@3 = 0.60 under top-3 evaluation ( Supplementary Fig. S9 D and Supplementary Table S9 ). In eukaryotic virus host prediction (15-way), it achieved Accuracy@1 = 0.78 and F1@1 = 0.69, with Accuracy@3 = 0.91 and F1@3 = 0.51 ( Supplementary Fig. S9 E and Supplementary Table S9 ). On cfRNA pancreas tumor vs. normal distinguishing task, the model achieved near-perfect discrimination (Accuracy/F1/Precision/Recall/MCC ≈ 1.0), and its learned embeddings showed clear, complete separation between tumor and normal samples in UMAP space ( Supplementary Fig. S9 F). All detailed quantitative benchmark results are listed in Supplementary Table S10 .",
          "section_heading": "Benchmark OmniNA with task-specific deep learning methods",
          "supporting_figure_or_table": "Fig. 4 F, Supplementary Fig. S8 D, and Supplementary Table S8",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10
          ],
          "doc_item_refs": [
            "#/texts/90"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0033"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0d42dedd9d53",
          "configuration_id": "config_9bedfa97352f",
          "route_label": "Pathogenic variation detection",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "pathogenic variation detection tasks",
          "source_object_verbatim": "pathogenic variation detection",
          "source_object_normalized": "pathogenic variation detection",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inputs",
            "no annotations supplied during training or testing"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "pathogenic variation detection tasks",
          "section_heading": "Results",
          "supporting_figure_or_table": "Fig. 4 G, Fig. 4 H, and Fig. 4 I",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            11
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/texts/90",
            "#/texts/94",
            "#/texts/97"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0034",
            "dense::full_2026-07-06__rec_003394::0047"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_954d5b459054",
          "configuration_id": "config_8039ce04100c",
          "route_label": "Species detection",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "species detection tasks",
          "source_object_verbatim": "species detection tasks",
          "source_object_normalized": "species detection",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inputs",
            "no annotations supplied during training or testing"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "species detection tasks",
          "section_heading": "Results",
          "supporting_figure_or_table": "Fig. 4 J-L and Supplementary Fig. S8 E",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            10
          ],
          "doc_item_refs": [
            "#/texts/54",
            "#/texts/55",
            "#/texts/90"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0035"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b0ea0b18ac37",
          "configuration_id": "config_1461dc56551a",
          "route_label": "Pathogenic virus detection",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "pathogenic virus detection task",
          "source_object_verbatim": "pathogenic virus detection task",
          "source_object_normalized": "pathogenic virus detection",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inputs",
            "no annotations supplied during training or testing"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "pathogenic virus detection task",
          "section_heading": "Results",
          "supporting_figure_or_table": "Fig. 4 J",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            11
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/texts/90",
            "#/texts/94",
            "#/texts/97"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0036"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_65b1d7440e6c",
          "configuration_id": "config_639fb82cda96",
          "route_label": "Bacteria taxonomy classification",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "bacteria taxonomy classification tasks related to class, family, order, and phylum",
          "source_object_verbatim": "bacteria taxonomy classification tasks",
          "source_object_normalized": "bacteria taxonomy ranks",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inputs",
            "no annotations supplied during training or testing"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "class, family, order, and phylum",
          "section_heading": "Results",
          "supporting_figure_or_table": "Fig. 4 K and L and Supplementary Fig. S8 E",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            10,
            11
          ],
          "doc_item_refs": [
            "#/pictures/6",
            "#/texts/54",
            "#/texts/55",
            "#/texts/90",
            "#/texts/94",
            "#/texts/97"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0037",
            "dense::full_2026-07-06__rec_003394::0048",
            "dense::full_2026-07-06__rec_003394::0049"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e994e51504a9",
          "configuration_id": "config_c18674ca7394",
          "route_label": "Non-B DNA structure classification",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "non-B DNA structure classification",
          "source_object_verbatim": "non-B DNA structure",
          "source_object_normalized": "non-B DNA structure",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inputs",
            "no annotations supplied during training or testing"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "non-B DNA structure classification",
          "section_heading": "Results",
          "supporting_figure_or_table": "Supplementary Fig. S9 A",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10
          ],
          "doc_item_refs": [
            "#/texts/90"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0038"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_75d24b8cf603",
          "configuration_id": "config_0224cca50a71",
          "route_label": "Bacterial functional-sequence sub-classification",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "bacterial functional-sequence sub-classification, including ARGs and VGs",
          "source_object_verbatim": "ARGs and VGs",
          "source_object_normalized": "bacterial functional sequences",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inputs",
            "no annotations supplied during training or testing"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "To assess the generalization capabilities of OmniNA on nucleotide sequence tasks, we conducted fine-tuning experiments on 23 diverse tasks encompassing genome and species detection. All downstream fine-tuning and evaluations reported here use sequence-only inputs, with no annotations supplied during training or testing; hence, the reported accuracy reflects annotation-free deployment. OmniNA generally achieved better F1 scores than benchmark methods (Fig. 4 ). Specifically, for tasks related to human DNA element identification, OmniNA achieved comparable or superior performance as compared with the benchmark methods (Fig. 4 A). Besides, OmniNA-1.7B excelled in PAS, TIS signal, and splice site detection, achieving remarkable F1 scores of 0.88, 0.97, and 0.98, respectively (Fig. 4 B and C). The superior performance of the OmniNA-1.7B model can also be observed for gRNA off-target detection with an F1 score of 0.98 (Fig. 4 D and Supplementary Fig. S8 A). In methylated region detection tasks, the OmniNA-1.7B model achieved comparable F1 and higher precision scores across various cell types compared with the 66 M and 330 M models (Fig. 4 E). Additionally, for chromatin feature prediction, OmniNA-1.7B delivered high-accuracy predictions for chromatin accessibility and histone modification, surpassing current state-of-theart methods (Fig. 4 F, Supplementary Fig. S8 B and C, and Supplementary Table S8 ). Our method further demonstrated high performance in the prediction for TF occupation, achieving a median F1 score of 0.73 across all TF targets (Fig. 4 F, Supplementary Fig. S8 D, and Supplementary Table S8 ). Besides, the model's performance on pathogenic variation detection tasks improved with increasing model size (Fig. 4 G). We compared the OmniNA-1.7B model against the EVE method. Across 48 genes, OmniNA-1.7B achieved higher F1 scores in 27 genes compared with EVE (Fig. 4 H). Overall, OmniNA1.7B achieved a significantly higher mean F1 score of 0.90 compared with 0.81 for EVE (Fig. 4 I; two-sided t -test, P -value = 0.011). OmniNA also exhibited proficiency in species detection tasks (Fig. 4 J -L). In the pathogenic virus detection task (Fig. 4 J), the F1 scores of OmniNA-1.7B surpassed those of benchmark methods. Furthermore, on bacteria taxonomy classification tasks related to class, family, order, and phylum (Fig. 4 K and L and Supplementary Fig. S8 E), OmniNA1.7B outperformed the benchmark model. We also evaluated the model on non-B DNA structure classification, where OmniNA-1.7B achieved an Accuracy@1 of 0.78 and Accuracy@3 of 0.88 ( Supplementary Fig. S9 A). Furthermore, we extended our evaluation to microbiology tasks. For bacterial functional-sequence sub-classification, including ARGs and VGs, we report top-k performance across five ARG mechanisms and 11 VG types. OmniNA-1.7B achieved an Accuracy@1 = 0.59 and F1@1 = 0.56, which improved to Accuracy@3 = 0.92 / F1@3 = 0.54 for ARG mechanisms ( Supplementary Fig. S9 B and Supplementary Table S9 ). For VG classification, the model reached Accuracy@1 = 0.93 and F1@1 = 0.87 and further improved to Accuracy@3 = 0.96 / F1@3 = 0.63 ( Supplementary Fig. S9 C and Supplementary Table S9 ). For host specificity prediction, OmniNA-1.7B accurately classified both bacteriophage and eukaryotic virus hosts. In bacteriophage host prediction (12-way), the model achieved Accuracy@1 = 0.73 and F1@1 = 0.78, improving to Accuracy@3 = 0.97 and F1@3 = 0.60 under top-3 evaluation ( Supplementary Fig. S9 D and Supplementary Table S9 ). In eukaryotic virus host prediction (15-way), it achieved Accuracy@1 = 0.78 and F1@1 = 0.69, with Accuracy@3 = 0.91 and F1@3 = 0.51 ( Supplementary Fig. S9 E and Supplementary Table S9 ). On cfRNA pancreas tumor vs. normal distinguishing task, the model achieved near-perfect discrimination (Accuracy/F1/Precision/Recall/MCC ≈ 1.0), and its learned embeddings showed clear, complete separation between tumor and normal samples in UMAP space ( Supplementary Fig. S9 F). All detailed quantitative benchmark results are listed in Supplementary Table S10 .",
          "section_heading": "Results",
          "supporting_figure_or_table": "Supplementary Fig. S9 B and Supplementary Table S9",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            10
          ],
          "doc_item_refs": [
            "#/texts/47",
            "#/texts/50",
            "#/texts/90"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0039",
            "dense::full_2026-07-06__rec_003394::0040",
            "dense::full_2026-07-06__rec_003394::0041"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_69caefbd75a1",
          "configuration_id": "config_8b8affa3bd0c",
          "route_label": "Host specificity prediction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "host specificity prediction",
          "source_object_verbatim": "bacteriophage and eukaryotic virus hosts",
          "source_object_normalized": "virus hosts",
          "source_modality_normalized": "nucleotide sequence",
          "transformation_chain_verbatim": [
            "sequence-only inputs",
            "no annotations supplied during training or testing"
          ],
          "model_visible_form_verbatim": "sequence-only inputs",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "sequence-only inputs, with no annotations supplied during training or testing",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "bacteriophage and eukaryotic virus hosts",
          "section_heading": "Results",
          "supporting_figure_or_table": "Supplementary Fig. S9 D and Supplementary Fig. S9 E",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            10
          ],
          "doc_item_refs": [
            "#/texts/47",
            "#/texts/50",
            "#/texts/90"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003394::0042",
            "dense::full_2026-07-06__rec_003394::0043",
            "dense::full_2026-07-06__rec_003394::0044"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_de94be356588"
    },
    {
      "model_id": "model_b47a3fca0d58",
      "model_name": "OpenAI's o1",
      "record_id": "full_2026-07-06__rec_001617",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4f442dcf4ed4",
      "paper_title": "Cell-o1: Training LLMs to Solve Single-Cell Reasoning Puzzles with Reinforcement Learning",
      "doi": "10.48550/arXiv.2506.02911",
      "paper_url": "https://doi.org/10.48550/arXiv.2506.02911",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell RNA sequencing batch with donor metadata"
      ],
      "lifecycle_phases": [
        "inference"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001617_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001617_76b0056d59c3/figure_002.png",
        "figure_index": 2,
        "caption": "Figure 2: CellPuzzles formulates cell type annotation as a batch-level reasoning task that integrates gene expression and contextual metadata, inspired by how experts annotate cells in practice.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic comparison of two cell type annotation workflows with four main panel regions arranged in two rows.\n\nTop row: A human expert task workflow. The left “Input” panel shows clustered single-cell data with colored clusters labeled Cluster 2, Cluster 3, Cluster 4, and Cluster N, each associated with marker gene lists such as “SFTPC, SFTPA1, SFTPB, NAPSA, ...”, “C1QA, APOE, MARCO, LYZ, ...”, and “IGHM, MZB1, JCHAIN, XBP1, ...”. The center panel labeled “Human Expert Task” shows an expert assigning a cell type label to each cluster using representative marker genes, contextual metadata, biological knowledge, and reference sources. The right “Output” panel lists assigned cell types including Pulmonary Alveolar Type 2 (AT2) Cells, CD8+ Cytotoxic T Cells, Lung Pericyte, Alveolar Macrophages, and Plasma Cells.\n\nBottom row: A “CellPuzzles Task” workflow. The left “Input” panel shows individual cells from clusters, with representative sampled cells labeled [Cell 1], [Cell 2], [Cell 3], [Cell 4], and [Cell N], each paired with gene lists such as “S100A9, TMSB10, RPL37, ...”, “MALAT1, FTL, B2M, ...”, “MALAT1, FTL, AKR1B10, ...”, “B2M, MT2A, TMSB4X, ...”, and “IGLC3, IGLC2, IGHM, ...”. Arrows indicate selected representative cells from clusters. The center panel labeled “CellPuzzles Task” shows a model/interface labeled “Cell-o1” receiving N cells, contextual metadata, and N candidate cell types, then reasoning to determine the optimal label assignment and provide reasoning traces. The right “Output” panel shows colored cell-to-label assignment lines connecting cells to candidate labels including CD4-positive, alpha-beta T Cell; CD8-positive, alpha-beta T Cell; Smooth Muscle Cell; Lung Macrophage; Non-classical Monocyte; Capillary Endothelial Cell; Plasma Cell; and Respiratory Basal Cell.\n\nBiological source objects include single-cell clusters, individual cells, marker genes, and immune/lung-related cell type labels. The figure depicts a transformation from marker gene or cell-level input data to annotated biological cell type outputs, comparing expert manual annotation with an automated CellPuzzles/Cell-o1 reasoning task.",
        "page_no": 3,
        "sha256": "e74192b02a6c544c4d4ae6c01c1b71c75747f0270a5e4e6bc77aba0c8a2259ee",
        "pixel_width": 791,
        "pixel_height": 318,
        "crop_box": {
          "x": 0.0,
          "y": 0.47,
          "width": 0.675,
          "height": 0.53
        },
        "panel_label": "Bottom-left CellPuzzles input plus model interface",
        "visible_input_object": "Clustered single-cell input with representative sampled cells and top genes",
        "visible_model_interface": "CellPuzzles Task prompt for Cell-o1 batch-level reasoning",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the grounded source object (clusters and representative cells with gene lists) and the immediate insertion/interface where the batch-level task is posed, while excluding the output-only right panel and the unrelated top-row human-expert comparison.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_8db6326aa3e2",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "a batch of N cells from the same donor, each represented by a ranked list of top-expressed genes, together with donor-level contextual metadata and a candidate set of N cell types",
          "actual_model_visible_form": "a standardized text prompt with contextual metadata, gene lists, candidate labels, and <think> and <answer> tags"
        }
      ],
      "routes": [
        {
          "route_id": "route_8db6326aa3e2",
          "configuration_id": "config_9d14c1087c91",
          "route_label": "batch-level reasoning prompt (OpenAI's o1)",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "CellPuzzles batch-level reasoning task",
          "source_object_verbatim": "a batch of N cells from the same donor, each represented by a ranked list of top-expressed genes, together with donor-level contextual metadata and a candidate set of N cell types",
          "source_object_normalized": "batch of N cells from the same donor with ranked top-expressed genes, donor-level contextual metadata, and a candidate set of cell types",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "collect and curate Cell x Gene scRNA-seq datasets",
            "convert structured donor metadata into natural language context",
            "rank the top-expressed genes for each cell",
            "shuffle the candidate cell types",
            "apply the standardized prompt template"
          ],
          "model_visible_form_verbatim": "a standardized text prompt with contextual metadata, gene lists, candidate labels, and <think> and <answer> tags",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "standardized prompt template",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Each input is paired with the standardized prompt shown in Table 1.",
          "section_heading": "4.2.1 Reasoning Distillation via Rejection Sampling",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            22,
            23
          ],
          "doc_item_refs": [
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/603",
            "#/texts/604",
            "#/texts/605"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001617::0002"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_1bf0ccb8259f",
      "model_name": "OpticalDNA",
      "record_id": "full_2026-07-06__rec_001830",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4162fc52fa19",
      "paper_title": "Rethinking Genomic Modeling Through Optical Character Recognition",
      "doi": "",
      "paper_url": "",
      "route_count": 6,
      "configuration_count": 6,
      "family_counts": {
        "visual_raster_carrier": 6
      },
      "subtype_counts": {
        "raw_slide_or_patch_input": 6
      },
      "families": [
        "visual_raster_carrier"
      ],
      "subtypes": [
        "raw_slide_or_patch_input"
      ],
      "primary_subtype": "raw_slide_or_patch_input",
      "modalities": [
        "DNA sequence"
      ],
      "lifecycle_phases": [
        "pretraining"
      ],
      "fusion_topologies": [
        "placeholder_replacement"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001830_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001830_eb50f2bf58e6/figure_002.png",
        "figure_index": 2,
        "caption": "Figure 2. Overview of OpticalDNA. (a) Render a 1D genomic sequence into a multi-page DNA document with bounding-box annotations. (b) Construct six OCR-style prompted genomic tasks. (c) Pretrain a visual encoder-document decoder under prompt supervision.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel schematic describing a DNA document generation, visual encoding, OCR-style pretraining, and decoding pipeline.\n\nPanel (a), “DNA document generation and visual encoding,” shows a 1D DNA sequence beginning “ACGGTT…” transformed into a multi-page DNA document with spatial annotations. A highlighted sequence region is connected to a bounding-box annotation with fields including sequence text and coordinate-like values. The document is patchified into `p x n x 16 x 16` patches, then passed through a frozen visual encoder labeled “SAM-Conv → CLIP-L / 399.8M,” producing `p x (n/16) x d` visual tokens.\n\nPanel (b), “OCR-style prompt pretraining tasks,” lists six genomic reasoning primitives: T1 Reading, T2 Grounding, T3 ROI, T4 Completion, T5 Retrieval, and T6 Recognition. Each task is illustrated with small DNA document snippets and prompt examples such as “Read the sequence,” “Locate each line in the sequence,” “Read sequence within region,” “Complete masked-region,” “Locate ATG,” and “Classify document.” The tasks are aligned with capabilities labeled Perception, Localization, Scan, Inference, Query, and Global understanding.\n\nPanel (c), “Trainable projection, fusion, and decoding,” shows the visual tokens entering a trainable projector labeled 2.6M parameters, then a trainable multi-page fusion module labeled 6.6M, producing `(n/16) x d` visual tokens. These are passed into a trainable document decoder labeled “DeepSeek-3B MOE-A570M,” conditioned by a task instruction interface with “Prompt” and “Response.” The output is trained with a loss labeled `L_pt`.\n\nBiological source object: DNA/genome sequence text represented as document-like pages with genomic regions and coordinate annotations. The figure presents a model architecture for converting DNA sequences into visual document representations and training OCR-like tasks for genomic reasoning.",
        "page_no": 4,
        "sha256": "bc007bde37203be58bc237be638674f9ab31b216d98e731ff662d83e7ad280e8",
        "pixel_width": 956,
        "pixel_height": 314,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 0.6,
          "height": 0.5
        },
        "panel_label": "(a) DNA document generation and visual encoding",
        "visible_input_object": "1D DNA sequence rendered as a DNA document (e.g. \"ACGGTT...\")",
        "visible_model_interface": "Patchified document pages into the frozen visual encoder, yielding visual tokens",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop isolates panel (a) and keeps the source DNA sequence, document rendering, patchify step, frozen visual encoder, and visual-token output, which is enough to understand the grounded input route from sequence to model-visible carrier while excluding the task panels and decoder/output side.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_fdd5a06de00b",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "genomic sequence S loaded from FASTA",
          "actual_model_visible_form": "rendered page images plus the task prompt q, encoded into fused visual tokens Z"
        }
      ],
      "routes": [
        {
          "route_id": "route_fdd5a06de00b",
          "configuration_id": "config_0c7145e694de",
          "route_label": "free-form DNA transcription",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "T1: Free-form DNA transcription",
          "source_object_verbatim": "genomic sequence S loaded from FASTA",
          "source_object_normalized": "genomic sequence",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "Render the genomic sequence into a structured multi-page DNA document",
            "Partition each page into non-overlapping 16 × 16 patches",
            "Encode the patches with the SAM-Conv-CLIP-L visual front-end",
            "Apply the learned projector Πθ",
            "Aggregate pages with the multi-page fusion module Fθ"
          ],
          "model_visible_form_verbatim": "rendered page images plus the task prompt q, encoded into fused visual tokens Z",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "a single reserved <image> placeholder whose images field carries the ordered page list; the document decoder Gψ consumes the fused visual tokens Z",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "T1 performs free-form DNA transcription",
          "section_heading": "3.3. OCR-Style Prompted Pretraining Tasks",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            21,
            22,
            26
          ],
          "doc_item_refs": [
            "#/texts/47",
            "#/texts/48",
            "#/texts/485",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488",
            "#/texts/489",
            "#/texts/49",
            "#/texts/490",
            "#/texts/50",
            "#/texts/51",
            "#/texts/52",
            "#/texts/53",
            "#/texts/55"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001830::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001830::0033"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ce89ed8fc2f8",
          "configuration_id": "config_a09758a77aff",
          "route_label": "grounded DNA transcription",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "T2: DNA Transcription with Spatial Grounding",
          "source_object_verbatim": "genomic sequence S loaded from FASTA",
          "source_object_normalized": "genomic sequence",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "Render the genomic sequence into a structured multi-page DNA document",
            "Partition each page into non-overlapping 16 × 16 patches",
            "Encode the patches with the SAM-Conv-CLIP-L visual front-end",
            "Apply the learned projector Πθ",
            "Aggregate pages with the multi-page fusion module Fθ"
          ],
          "model_visible_form_verbatim": "rendered page images plus the grounding prompt q, encoded into fused visual tokens Z",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "a single reserved <image> placeholder whose images field carries the ordered page list; the document decoder Gψ consumes the fused visual tokens Z",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "T2 couples transcription with spatial grounding",
          "section_heading": "3.3. OCR-Style Prompted Pretraining Tasks",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            26
          ],
          "doc_item_refs": [
            "#/texts/47",
            "#/texts/48",
            "#/texts/485",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488",
            "#/texts/489",
            "#/texts/49",
            "#/texts/490",
            "#/texts/50",
            "#/texts/51",
            "#/texts/52",
            "#/texts/53"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001830::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001830::0034"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_167e8d61aaec",
          "configuration_id": "config_8f8a92b8feaf",
          "route_label": "ROI DNA transcription",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "T3: ROI-Based DNA Transcription",
          "source_object_verbatim": "genomic sequence S loaded from FASTA",
          "source_object_normalized": "genomic sequence",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "Render the genomic sequence into a structured multi-page DNA document",
            "Partition each page into non-overlapping 16 × 16 patches",
            "Encode the patches with the SAM-Conv-CLIP-L visual front-end",
            "Apply the learned projector Πθ",
            "Aggregate pages with the multi-page fusion module Fθ"
          ],
          "model_visible_form_verbatim": "rendered page images plus the ROI prompt q, encoded into fused visual tokens Z",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "a single reserved <image> placeholder whose images field carries the ordered page list; the document decoder Gψ consumes the fused visual tokens Z",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "T3 transcribes DNA within given ROIs",
          "section_heading": "3.3. OCR-Style Prompted Pretraining Tasks",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            22,
            23,
            26
          ],
          "doc_item_refs": [
            "#/texts/415",
            "#/texts/416",
            "#/texts/417",
            "#/texts/418",
            "#/texts/419",
            "#/texts/420",
            "#/texts/421",
            "#/texts/422",
            "#/texts/47",
            "#/texts/48",
            "#/texts/485",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488",
            "#/texts/489",
            "#/texts/49",
            "#/texts/490",
            "#/texts/50",
            "#/texts/51",
            "#/texts/52",
            "#/texts/53"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001830::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001830::0035",
            "dense::full_2026-07-06__rec_001830::0175"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4c3ebc517d6c",
          "configuration_id": "config_f6f3115cba8a",
          "route_label": "masked DNA completion",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "T4: Masked DNA Completion",
          "source_object_verbatim": "genomic sequence S loaded from FASTA",
          "source_object_normalized": "genomic sequence",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "Render the genomic sequence into a structured multi-page DNA document",
            "Partition each page into non-overlapping 16 × 16 patches",
            "Encode the patches with the SAM-Conv-CLIP-L visual front-end",
            "Apply the learned projector Πθ",
            "Aggregate pages with the multi-page fusion module Fθ"
          ],
          "model_visible_form_verbatim": "rendered page images with masked regions plus the completion prompt q, encoded into fused visual tokens Z",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "a single reserved <image> placeholder whose images field carries the ordered page list; the document decoder Gψ consumes the fused visual tokens Z",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "T4 completes masked regions within given ROIs",
          "section_heading": "3.3. OCR-Style Prompted Pretraining Tasks",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            24,
            26
          ],
          "doc_item_refs": [
            "#/pictures/9",
            "#/tables/16",
            "#/texts/434",
            "#/texts/435",
            "#/texts/436",
            "#/texts/439",
            "#/texts/440",
            "#/texts/485",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488",
            "#/texts/489",
            "#/texts/490"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001830::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001830::0036",
            "dense::full_2026-07-06__rec_001830::0179"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f7dea95d8f42",
          "configuration_id": "config_a23c2371e05c",
          "route_label": "subsequence localization",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "T5: DNA Subsequence Localization",
          "source_object_verbatim": "genomic sequence S loaded from FASTA",
          "source_object_normalized": "genomic sequence",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "Render the genomic sequence into a structured multi-page DNA document",
            "Partition each page into non-overlapping 16 × 16 patches",
            "Encode the patches with the SAM-Conv-CLIP-L visual front-end",
            "Apply the learned projector Πθ",
            "Aggregate pages with the multi-page fusion module Fθ"
          ],
          "model_visible_form_verbatim": "rendered page images plus the query prompt q, encoded into fused visual tokens Z",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "a single reserved <image> placeholder whose images field carries the ordered page list; the document decoder Gψ consumes the fused visual tokens Z",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "T5 localizes all occurrences of a query subsequence",
          "section_heading": "3.3. OCR-Style Prompted Pretraining Tasks",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5,
            24,
            25,
            26,
            27,
            28
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/texts/443",
            "#/texts/444",
            "#/texts/445",
            "#/texts/446",
            "#/texts/447",
            "#/texts/448",
            "#/texts/449",
            "#/texts/450",
            "#/texts/485",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488",
            "#/texts/489",
            "#/texts/490",
            "#/texts/55",
            "#/texts/56",
            "#/texts/59",
            "#/texts/60",
            "#/texts/61",
            "#/texts/62",
            "#/texts/63"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001830::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001830::0037",
            "dense::full_2026-07-06__rec_001830::0038",
            "dense::full_2026-07-06__rec_001830::0181"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_990b08560f17",
          "configuration_id": "config_d648d791d43e",
          "route_label": "chromosome classification",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "T6: Chromosome Classification",
          "source_object_verbatim": "genomic sequence S loaded from FASTA",
          "source_object_normalized": "genomic sequence",
          "source_modality_normalized": "DNA sequence",
          "transformation_chain_verbatim": [
            "Render the genomic sequence into a structured multi-page DNA document",
            "Partition each page into non-overlapping 16 × 16 patches",
            "Encode the patches with the SAM-Conv-CLIP-L visual front-end",
            "Apply the learned projector Πθ",
            "Aggregate pages with the multi-page fusion module Fθ"
          ],
          "model_visible_form_verbatim": "rendered page images plus the classification prompt q, encoded into fused visual tokens Z",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "a single reserved <image> placeholder whose images field carries the ordered page list; the document decoder Gψ consumes the fused visual tokens Z",
          "fusion_topology": "placeholder_replacement",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "T6 predicts a chromosome-level class label",
          "section_heading": "3.3. OCR-Style Prompted Pretraining Tasks",
          "supporting_figure_or_table": "Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            25,
            26,
            40
          ],
          "doc_item_refs": [
            "#/tables/18",
            "#/texts/460",
            "#/texts/461",
            "#/texts/462",
            "#/texts/463",
            "#/texts/464",
            "#/texts/465",
            "#/texts/468"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001830::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001830::0184"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_cb06ed43c31a"
    },
    {
      "model_id": "model_860d9511b312",
      "model_name": "P3GPT",
      "record_id": "full_2026-07-06__rec_001187",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_9645a7bdccf5",
      "paper_title": "Precious3GPT: Multimodal Multi-Species Multi-Omics Multi-Tissue Transformer for Aging Research and Drug Discovery",
      "doi": "10.1101/2024.07.25.605062",
      "paper_url": "https://doi.org/10.1101/2024.07.25.605062",
      "route_count": 12,
      "configuration_count": 11,
      "family_counts": {
        "text_native_token_stream": 8,
        "dense_continuous_carrier": 4
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 7,
        "serialized_biological_context_or_ordered_profile": 1,
        "connector_mediated_embedding": 2,
        "direct_projected_embedding": 1,
        "pooled_or_aggregated_embedding": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "connector_mediated_embedding",
        "direct_projected_embedding",
        "pooled_or_aggregated_embedding",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "DNA methylation",
        "chemical perturbation transcriptomics",
        "chemically induced expression perturbations",
        "clinical blood biomarkers",
        "disease indication",
        "expression profiles",
        "gene embeddings",
        "gene lists",
        "graph/network",
        "proteomics",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "cross_attention",
        "shared_latent_alignment",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001187_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001187_7214eb506e14/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1.  P3GPT features a novel architecture enabling efficient omics-data training and multimodal extensions.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic model architecture diagram for multimodal biological/OMICS data processing.\n\nVisible elements:\n- Left panel: “Model input” showing an instruction token sequence, OMICS input matrices, downregulated gene/protein-like entries, annotation metadata, and a species/context table. Labels include examples such as `IGF1R`, `CDK?`, `TCF?`, `PKA`, and annotations like Species: human, Age: 40, Sex: M, Tissue: liver, Health: T2D, and dataset identifiers.\n- Middle-left panel: “OMICS Data” with listed biological source modalities:\n  - Transcriptomics\n  - Epigenetics\n  - Proteomics\n- A green “Modality Mapper (MM)” block transforms OMICS data into a fixed embedding shape, shown as `[ n × k ] kG × Text embeddings`, using modules labeled `FF`, `B.norm`, `ReLU`, `FF`, producing `[ n × 256 ]`.\n- Lower middle inputs: separate “Text Data” and “Knowledge Graph Data” boxes, each passed through small green “MM” mapper blocks.\n- Center panel: “Learnable embeddings” and “Multi-modal Input”, both represented as colored token/embedding grids with dimension labels such as `×256`.\n- Right panel: stacked “Transformer Block” modules, each marked `×32`. One stack has a flame icon and note “Freeze weights”; another stack has a crossed/snowflake-like icon suggesting frozen or altered training behavior.\n- Top-right inset: transformer block internals labeled `MHA`, `Norm`, `FF`, `Norm`, and `TB`, with notes defining:\n  - ALiBi = Attention with Linear Biases\n  - MHA = Multi-Head Attention\n  - FF = Feed-Forward\n  - B.Norm = Batch Normalization\n- Top-right also shows ALiBi attention/bias matrices feeding into MHA.\n- Bottom-right “Novelty” section lists claimed contributions:\n  - Modalities first combined: multimodal OMICS, knowledge graph, and text\n  - New tokenization method for OMICS data\n  - Task solving in a language modelling way\n  - Unique multiomics multispecies dataset\n\nOverall, the figure depicts a multimodal transformer-based architecture that maps OMICS, text, and knowledge graph data into shared embeddings for language-model-style processing.",
        "page_no": 3,
        "sha256": "6451eeba54949e0870379d5c4ea9841d16445b48a07ed820ac630f234d2a5df4",
        "pixel_width": 787,
        "pixel_height": 420,
        "crop_box": {
          "x": 0.15,
          "y": 0.33,
          "width": 0.54,
          "height": 0.58
        },
        "panel_label": "source-to-MM",
        "visible_input_object": "Text Data and Knowledge Graph Data",
        "visible_model_interface": "Modality Mapper (MM) feeding the Multi-modal Input carrier",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the lower-left source inputs, their MM transformation blocks, and the adjacent Multi-modal Input column with readable labels and connector arrows. It excludes the novelty panel and transformer internals while preserving a grounded input route into the model.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "connector_mediated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_1e37d80b54f4",
          "example_input": "image / omics encoder states",
          "example_carrier": "Q-Former or adapter query vectors",
          "example_interface": "connector → LLM cross-modal interface",
          "actual_source": "biomedical texts",
          "actual_model_visible_form": "GenePT-derived embeddings"
        },
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_6ef939c09098",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "P3GPT's embeddings for all 25,332 human genes",
          "actual_model_visible_form": "embeddings"
        },
        {
          "subtype_id": "pooled_or_aggregated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_9ee2a23f3bb4",
          "example_input": "{gene/cell/patch embeddings}",
          "example_carrier": "mean/attention pool = one compact vector",
          "example_interface": "aggregator → generator",
          "actual_source": "GSE223748",
          "actual_model_visible_form": "stacked embeddings of the 1000 most methylated gene promoters"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_c4cbe3c8fb77",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "human clinical blood tests from NHANES-IV",
          "actual_model_visible_form": "blood biomarker values in structured sentences"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_1dc4de8f3ba0",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "LINCS chemical perturbation experiments",
          "actual_model_visible_form": "structured prompt sentences with differential gene lists and experimental metadata"
        }
      ],
      "routes": [
        {
          "route_id": "route_1dc4de8f3ba0",
          "configuration_id": "config_c7963bc5af93",
          "route_label": "LINCS chemical screening prompts",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "chemical omics screening (<cpd2diff> and <diff2cpd>)",
          "source_object_verbatim": "LINCS chemical perturbation experiments",
          "source_object_normalized": "LINCS chemical perturbation experiments",
          "source_modality_normalized": "chemical perturbation transcriptomics",
          "transformation_chain_verbatim": [
            "used level-5 data representing DEG signatures",
            "truncated the generated up- and down-regulated lists to the length of the corresponding level5 signatures",
            "formatted as sentences with a strictly defined structure"
          ],
          "model_visible_form_verbatim": "structured prompt sentences with differential gene lists and experimental metadata",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "submitted to training the transformer block (TB)",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Only LINCS chemical perturbation experiments were collected for the model's training.",
          "section_heading": "Methods > Data collection and processing",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            24
          ],
          "doc_item_refs": [
            "#/texts/141",
            "#/texts/144",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/148"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001187::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3a1dcc4d9cc9",
          "configuration_id": "config_6ff062e9843d",
          "route_label": "DNA methylation aging prompts",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cross-age group observational study (<age2diff> and <diff2age>) and aging-clock training",
          "source_object_verbatim": "DNA methylation experiments",
          "source_object_normalized": "DNA methylation experiments",
          "source_modality_normalized": "DNA methylation",
          "transformation_chain_verbatim": [
            "reduced to differentially methylated TSS1500 regions",
            "formatted as sentences with a strictly defined structure",
            "instruction added at the beginning of each sentence"
          ],
          "model_visible_form_verbatim": "structured prompt sentences with methylation features and metadata",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "submitted to training the transformer block (TB)",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "For methylation data, we used custom scripts to calculate the average i80-value",
          "section_heading": "Methods > Data collection and processing",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            24
          ],
          "doc_item_refs": [
            "#/texts/141",
            "#/texts/144",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/148"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001187::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fecdaf390901",
          "configuration_id": "config_a99ce089fff2",
          "route_label": "Proteomics target-discovery prompts",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "case-control condition study (<diff2disease> and <disease2diff>) and multimodal target enrichment",
          "source_object_verbatim": "proteome experiments",
          "source_object_normalized": "proteome experiments",
          "source_modality_normalized": "proteomics",
          "transformation_chain_verbatim": [
            "reduced to differentially abundant proteins",
            "formatted as sentences with a strictly defined structure",
            "instruction added at the beginning of each sentence"
          ],
          "model_visible_form_verbatim": "structured prompt sentences with protein abundance features and metadata",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "submitted to training the transformer block (TB)",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "differentially abundant proteins for proteome experiments",
          "section_heading": "Methods > Data collection and processing",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            24
          ],
          "doc_item_refs": [
            "#/texts/141",
            "#/texts/144",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/148"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001187::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c4cbe3c8fb77",
          "configuration_id": "config_36b6865a6803",
          "route_label": "Clinical blood-test prompts",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "human clinical blood tests",
          "source_object_verbatim": "human clinical blood tests from NHANES-IV",
          "source_object_normalized": "clinical blood tests",
          "source_modality_normalized": "clinical blood biomarkers",
          "transformation_chain_verbatim": [
            "30 blood biomarkers were included in the training procedure",
            "numeric values were min-max scaled from the original SI units"
          ],
          "model_visible_form_verbatim": "blood biomarker values in structured sentences",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "formatted as sentences with a strictly defined structure",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Human clinical blood tests used in the model's training included 30 blood biomarkers",
          "section_heading": "Methods > Data collection and processing",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            23,
            24
          ],
          "doc_item_refs": [
            "#/texts/141",
            "#/texts/144",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/148"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001187::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1e37d80b54f4",
          "configuration_id": "config_a61df29bcbc9",
          "route_label": "Biomedical text embeddings",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "multimodal extension",
          "source_object_verbatim": "biomedical texts",
          "source_object_normalized": "biomedical texts",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "embeddings from OpenAI's text-embedding-ada-002 accessed via GenePT",
            "mapped onto the language model's latent space"
          ],
          "model_visible_form_verbatim": "GenePT-derived embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "modality mapper units",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "The auxiliary component enriches P3GPT with multimodal data through three-layer feed-forward neural networks (modality mappers, MM). MMs take in embeddings from additional modalitiesbiomedical texts and KGs-and map them onto the language model's latent space, effectively fusing diverse knowledge sources. For the knowledge graph modality, we used embeddings generated by a heterogeneous graph transformer trained on data assembled using the Indra tool 51-53 . For biomedical text, we utilized embeddings from Open AI's text-embedding-ada-002 accessed via GenePT 54 . Each gene's embedding was averaged across the up- and down-regulated gene lists to create representations of gene lists. The training involved optimizing the language modeling objective using the AdamW optimizer with a dynamic learning rate, starting at 5e-3 and decaying by 0.01 after the 6th and 10th epochs, over a total of 15 epochs 55 . The training was conducted on five A6000 GPUs with a batch size of 17.",
          "section_heading": "Model architecture and training",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            25
          ],
          "doc_item_refs": [
            "#/texts/153",
            "#/texts/154",
            "#/texts/155"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001187::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_88869c561134",
          "configuration_id": "config_a61df29bcbc9",
          "route_label": "Knowledge-graph embeddings",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "multimodal extension",
          "source_object_verbatim": "knowledge graph data",
          "source_object_normalized": "knowledge graph data",
          "source_modality_normalized": "graph/network",
          "transformation_chain_verbatim": [
            "embeddings generated by a heterogeneous graph transformer trained on data assembled using the Indra tool",
            "mapped onto the language model's latent space"
          ],
          "model_visible_form_verbatim": "heterogeneous graph transformer embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "modality mapper (MM) units",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "For the knowledge graph modality, we used embeddings generated by a heterogeneous graph transformer",
          "section_heading": "Model architecture and training",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            25
          ],
          "doc_item_refs": [
            "#/texts/153",
            "#/texts/154",
            "#/texts/155"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001187::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_14d49b87f50f",
          "configuration_id": "config_5626f90cf2cb",
          "route_label": "LINCS holdout chemical screening prompts",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "chemical screening on the LINCS holdout set",
          "source_object_verbatim": "805 randomly selected chemical perturbations excluded from the LINCS training subset",
          "source_object_normalized": "LINCS holdout perturbation set",
          "source_modality_normalized": "chemically induced expression perturbations",
          "transformation_chain_verbatim": [
            "structured prompts used disease and compound tags plus tissue, cell, EFO, dose, time, species, and dataset_type fields",
            "generated up- and down-regulated gene lists were compared against the corresponding level-5 signatures"
          ],
          "model_visible_form_verbatim": "structured prompt sentences with tagged experimental metadata and blank fields",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "submitted to the transformer block (TB)",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The 805 randomly selected chemical perturbations were excluded from the LINCS training subset.",
          "section_heading": "Methods > Gene list generation",
          "supporting_figure_or_table": "Figure 3D",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            25
          ],
          "doc_item_refs": [
            "#/texts/159",
            "#/texts/160"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001187::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_42e13a712e55",
          "configuration_id": "config_50ebb50f276d",
          "route_label": "Age-group expression-signature prompting",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "age2diff instruction for geroprotector discovery",
          "source_object_verbatim": "human lung tissue expression signatures contrasting younger and older adults",
          "source_object_normalized": "age-group comparison expression profiles",
          "source_modality_normalized": "expression profiles",
          "transformation_chain_verbatim": [
            "applied the <age2diff> instruction to obtain expression signatures differentiating younger and older adults"
          ],
          "model_visible_form_verbatim": "instruction-bearing structured prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompted directly into P3GPT",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "To assess P3GPT's practical utility, we used it as a hypothesis generator in an in vitro anti-aging experiment. To obtain a list of potential geroprotectors from P3GPT, we sequentially executed it with two instructions. First, we applied the <age2diff> instruction to obtain expression signatures differentiating younger (20 years) and older (80 years) adults. Second, we applied the <diff2cpd> instruction to generate the compounds expected to reverse the signature identified in the first step in IMR90 cells. Upon manual curation to exclude toxic and commercially unavailable compounds, the 22 molecules listed in Table 4 were selected for screening in an in vitro senescence model (see Methods ).",
          "section_heading": "P3GPT-derived geroprotectors",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13,
            14
          ],
          "doc_item_refs": [
            "#/texts/72",
            "#/texts/73",
            "#/texts/76",
            "#/texts/77"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001187::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001187::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4ea7d2fffd2b",
          "configuration_id": "config_ce9056ba8d22",
          "route_label": "Geroprotector compound prompt generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "diff2cpd / diff2compound instruction for geroprotector discovery",
          "source_object_verbatim": "gene lists produced in Phase-1",
          "source_object_normalized": "age-reversal gene lists",
          "source_modality_normalized": "gene lists",
          "transformation_chain_verbatim": [
            "gene lists produced in Phase-1 were utilized as input for the P3GPT models' <diff2compound> instruction",
            "IMR90 cell line specified as an additional parameter in the model input"
          ],
          "model_visible_form_verbatim": "second-stage structured prompt carrying gene lists",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompted directly into P3GPT",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "(B) P3GPT can be used to generate novel geroprotectors. To achieve this, the model first needs to be instructed to generate the omics signatures differentiating older and younger individuals by applying the <age2diff> instruction to a prompt with the specified context. Then, the produced gene lists need to be supplied as input to P3GPT instructed to generate the compounds mimicking the reverse aging signature. Note that only age groups and gene lists are modified in the prompts throughout the workflow. Other prompt fields, such as the species, tissues, or omics type may be set to reflect a specific experimental setting.",
          "section_heading": "P3GPT-derived geroprotectors",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13,
            14,
            19,
            25,
            27
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/110",
            "#/texts/111",
            "#/texts/112",
            "#/texts/159",
            "#/texts/160",
            "#/texts/185",
            "#/texts/186",
            "#/texts/72",
            "#/texts/73",
            "#/texts/76",
            "#/texts/77"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001187::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001187::0036",
            "dense::full_2026-07-06__rec_001187::0005"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_eb4d5b439c28",
          "configuration_id": "config_c6fab3de6829",
          "route_label": "Disease target-enrichment prompts",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "multimodal target enrichment across proteomic, transcriptomic, and methylomic domains",
          "source_object_verbatim": "24 diseases that affect different tissues",
          "source_object_normalized": "disease indications",
          "source_modality_normalized": "disease indication",
          "transformation_chain_verbatim": [
            "constructed prompts with tissue, age, cell, EFO, drug, dose, time, case, control, dataset_type, gender, and species tags",
            "collected token probabilities to rank the 300 most likely up- or downregulated genes per omics domain",
            "checked the ranked genes for enrichment in clinical targets"
          ],
          "model_visible_form_verbatim": "structured prompt sentences with disease and experimental metadata",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "submitted to the transformer block (TB)",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "P3GPT was trained on biodata modalities representing different levels of regulation, including gene expression, DNA methylation, protein levels and protein interactions. To validate that the model successfully internalized entity relations across all omics levels, we implemented the following target identification test. First, we prepared a collection of 24 diseases that affect different tissues and instructed P3GPT to generate differentially methylated, expressed, and translated genes. The conditions in this experiment were selected based on a large number of known target genes (>10) and a sufficiently large representation of the affected tissue in the training set (see Methods).",
          "section_heading": "Methods > Multimodal target enrichment",
          "supporting_figure_or_table": "Figure 4C",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper reuses the same prompt scaffold once per omics domain; I keep the shared input format as one route.",
          "pages": [
            14,
            15
          ],
          "doc_item_refs": [
            "#/texts/79",
            "#/texts/80",
            "#/texts/81",
            "#/texts/84"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001187::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001187::0006"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6ef939c09098",
          "configuration_id": "config_e406945de0fc",
          "route_label": "GO-term binary classifiers on P3GPT gene embeddings",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "train binary classifiers for 18 high-level Gene Ontology (GO) terms",
          "source_object_verbatim": "P3GPT's embeddings for all 25,332 human genes",
          "source_object_normalized": "P3GPT gene embeddings",
          "source_modality_normalized": "gene embeddings",
          "transformation_chain_verbatim": [
            "P3GPT's embeddings for all 25,332 human genes",
            "train binary classifiers"
          ],
          "model_visible_form_verbatim": "embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "train binary classifiers",
          "fusion_topology": "cross_attention",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "we extracted P3GPT's embeddings for all 25,332 human genes it could interpret to train binary classifiers for 18 high-level Gene Ontology (GO) terms",
          "section_heading": "Results > P3GPT entities are associated with biological processes",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9
          ],
          "doc_item_refs": [
            "#/texts/46"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001187::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9ee2a23f3bb4",
          "configuration_id": "config_28a9a7d267af",
          "route_label": "Multi-species aging clock feature extraction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "multi-species aging clocks",
          "source_object_verbatim": "GSE223748",
          "source_object_normalized": "GEO dataset GSE223748",
          "source_modality_normalized": "DNA methylation",
          "transformation_chain_verbatim": [
            "processed ß-values",
            "stacked embeddings of the 1000 most methylated gene promoters",
            "CatBoost regressor"
          ],
          "model_visible_form_verbatim": "stacked embeddings of the 1000 most methylated gene promoters",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "For each cohort, a separate CatBoost regressor was trained using the stacked embeddings of the 1000 most methylated gene promoters.",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "we used processed ß-values from GSE223748 GEO dataset.",
          "section_heading": "P3GPT-derived aging clocks",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            26
          ],
          "doc_item_refs": [
            "#/texts/171",
            "#/texts/172",
            "#/texts/173",
            "#/texts/174"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001187::0106"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_6742df78bf4f"
    },
    {
      "model_id": "model_f8e766fddf66",
      "model_name": "PROCYON",
      "record_id": "full_2026-07-06__rec_000086",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_f6eff70c6863",
      "paper_title": "PROCYON: A multimodal foundation model for protein phenotypes.",
      "doi": "10.1101/2024.12.10.627665",
      "paper_url": "https://doi.org/10.1101/2024.12.10.627665",
      "route_count": 32,
      "configuration_count": 22,
      "family_counts": {
        "text_native_token_stream": 29,
        "dense_continuous_carrier": 3
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 18,
        "plain_language_prompt_or_question": 11,
        "direct_projected_embedding": 2,
        "connector_mediated_embedding": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "connector_mediated_embedding",
        "direct_projected_embedding",
        "plain_language_prompt_or_question",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "dense embeddings",
        "mixed",
        "multimodal",
        "protein",
        "protein/peptide",
        "structured biological database",
        "structured biological database record",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "inference"
      ],
      "fusion_topologies": [
        "concatenation",
        "other_explicit",
        "placeholder_replacement",
        "prefix",
        "query_bottleneck",
        "retrieval_or_tool_context",
        "shared_latent_alignment",
        "tokenizer_sequence",
        "unclear"
      ],
      "text_roles": [
        "instruction_or_query",
        "metadata_or_context",
        "no_text_on_this_route",
        "paired_alignment_supervision"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "The pixels clearly show a drug-target example, not the claimed protein-function retrieval prompt about catalytic proteins in glycolysis. This figure does not responsibly support the exact route, and there is no better rectangle in this source image for that route.",
      "illustrative_examples": [
        {
          "subtype_id": "connector_mediated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_058ac010dea2",
          "example_input": "image / omics encoder states",
          "example_carrier": "Q-Former or adapter query vectors",
          "example_interface": "connector → LLM cross-modal interface",
          "actual_source": "protein or prompt embeddings",
          "actual_model_visible_form": "protein or prompt embeddings"
        },
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_67104b9f17d9",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "protein",
          "actual_model_visible_form": "protein embeddings"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_9d3af98fe18f",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "a natural language prompt about catalytic proteins in glycolysis",
          "actual_model_visible_form": "the LLM-processed prompt projected into the unified latent space"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_9a2e2f038920",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "protein-phenotype pair",
          "actual_model_visible_form": "a natural language task definition, associated protein inputs, and an expected answer"
        }
      ],
      "routes": [
        {
          "route_id": "route_9a2e2f038920",
          "configuration_id": "config_a5406ac959eb",
          "route_label": "PROCYON-INSTRUCT instruction example",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "retrieval, QA, and phenotype generation tasks",
          "source_object_verbatim": "protein-phenotype pair",
          "source_object_normalized": "protein-phenotype pair",
          "source_modality_normalized": "mixed",
          "transformation_chain_verbatim": [
            "rephrase the original descriptions using OpenAI's GPT models",
            "create an instruction tuning example that consists of a natural language task definition, associated protein inputs, and an expected answer"
          ],
          "model_visible_form_verbatim": "a natural language task definition, associated protein inputs, and an expected answer",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "multimodal instruction tuning dataset",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "For each protein-phenotype pair, we create an instruction tuning example that consists of a natural language task definition, associated protein inputs, and an expected answer",
          "section_heading": "PROCYON-INSTRUCT dataset with 33 million protein-phenotype instructions",
          "supporting_figure_or_table": "Figure 1c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/47",
            "#/texts/50",
            "#/texts/51"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000086::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9d3af98fe18f",
          "configuration_id": "config_2f56a8eaf38c",
          "route_label": "Protein function retrieval prompt",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "protein function retrieval",
          "source_object_verbatim": "a natural language prompt about catalytic proteins in glycolysis",
          "source_object_normalized": "natural language prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "the LLM-processed prompt projected into the unified latent space",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "the LLM-processed prompt projected into the unified latent space by a query connector",
          "fusion_topology": "query_bottleneck",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "retrieving catalytic proteins in glycolysis",
          "section_heading": "Figure 1: Overview of PROCYON model architecture and PROCYON-INSTRUCT dataset",
          "supporting_figure_or_table": "Figure 1a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/291",
            "#/texts/292",
            "#/texts/293",
            "#/texts/294",
            "#/texts/295",
            "#/texts/296",
            "#/texts/297",
            "#/texts/298"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000086::route_012"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5d20ca16ef9b",
          "configuration_id": "config_54ef93b735da",
          "route_label": "Domain based protein retrieval prompt",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "domain based protein retrieval",
          "source_object_verbatim": "a natural language prompt about proteins with SH3 domains",
          "source_object_normalized": "natural language prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "the LLM-processed prompt projected into the unified latent space",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "the LLM-processed prompt projected into the unified latent space by a query connector",
          "fusion_topology": "query_bottleneck",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "proteins with SH3 domains",
          "section_heading": "Figure 1: Overview of PROCYON model architecture and PROCYON-INSTRUCT dataset",
          "supporting_figure_or_table": "Figure 1a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/291",
            "#/texts/292",
            "#/texts/293",
            "#/texts/294",
            "#/texts/295",
            "#/texts/296",
            "#/texts/297",
            "#/texts/298"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000086::route_013"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3e07d92b7f4a",
          "configuration_id": "config_e5d13f421ef7",
          "route_label": "Protein interaction prediction prompt",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "protein interaction prediction",
          "source_object_verbatim": "a natural language prompt about interacting proteins involved in autophagy",
          "source_object_normalized": "natural language prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "the LLM-processed prompt projected into the unified latent space",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "the LLM-processed prompt projected into the unified latent space by a query connector",
          "fusion_topology": "query_bottleneck",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "interacting proteins involved in autophagy",
          "section_heading": "Figure 1: Overview of PROCYON model architecture and PROCYON-INSTRUCT dataset",
          "supporting_figure_or_table": "Figure 1a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/291",
            "#/texts/292",
            "#/texts/293",
            "#/texts/294",
            "#/texts/295",
            "#/texts/296",
            "#/texts/297",
            "#/texts/298"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000086::route_014"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_35c4dd0a1c2f",
          "configuration_id": "config_f27afc5252c3",
          "route_label": "Disease associated protein retrieval prompt",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "disease associated protein retrieval",
          "source_object_verbatim": "a natural language prompt about proteins involved in Parkinson's disease pathology",
          "source_object_normalized": "natural language prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "the LLM-processed prompt projected into the unified latent space",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "the LLM-processed prompt projected into the unified latent space by a query connector",
          "fusion_topology": "query_bottleneck",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "proteins involved in Parkinson's disease pathology",
          "section_heading": "Figure 1: Overview of PROCYON model architecture and PROCYON-INSTRUCT dataset",
          "supporting_figure_or_table": "Figure 1a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/291",
            "#/texts/292",
            "#/texts/293",
            "#/texts/294",
            "#/texts/295",
            "#/texts/296",
            "#/texts/297",
            "#/texts/298"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000086::route_015"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_18a49a03ed5f",
          "configuration_id": "config_ef318b451e47",
          "route_label": "GO Function",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "PROCYON-INSTRUCT dataset",
          "source_object_verbatim": "GO Function",
          "source_object_normalized": "GO Function source records",
          "source_modality_normalized": "structured biological database record",
          "transformation_chain_verbatim": [
            "database curation",
            "instruction templating"
          ],
          "model_visible_form_verbatim": "instruction tuning example",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templating",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "instruction tuning example",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 1c",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/50"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0009"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ba110b719918",
          "configuration_id": "config_5c2cee1633c3",
          "route_label": "GtoP",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "PROCYON-INSTRUCT dataset",
          "source_object_verbatim": "GtoP",
          "source_object_normalized": "GtoP source records",
          "source_modality_normalized": "structured biological database record",
          "transformation_chain_verbatim": [
            "database curation",
            "instruction templating"
          ],
          "model_visible_form_verbatim": "instruction tuning example",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templating",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "instruction tuning example",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 1c",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/50"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_7d3bbc8ca354",
          "configuration_id": "config_7bae17ec6575",
          "route_label": "Drugbank Transporter",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "PROCYON-INSTRUCT dataset",
          "source_object_verbatim": "Drugbank Transporter",
          "source_object_normalized": "DrugBank transporter source records",
          "source_modality_normalized": "structured biological database record",
          "transformation_chain_verbatim": [
            "database curation",
            "instruction templating"
          ],
          "model_visible_form_verbatim": "instruction tuning example",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templating",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "instruction tuning example",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 1c",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/50"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0011"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_67d60756e8b4",
          "configuration_id": "config_06361596bc98",
          "route_label": "Drugbank Target",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "PROCYON-INSTRUCT dataset",
          "source_object_verbatim": "Drugbank Target",
          "source_object_normalized": "DrugBank target source records",
          "source_modality_normalized": "structured biological database record",
          "transformation_chain_verbatim": [
            "database curation",
            "instruction templating"
          ],
          "model_visible_form_verbatim": "instruction tuning example",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templating",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "instruction tuning example",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 1c",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/50"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_7b76d04a0605",
          "configuration_id": "config_991a56245b75",
          "route_label": "UniProt",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "A comprehensive PROCYON-INSTRUCT dataset with 33,899,528 protein-phenotype instructions is curated from five knowledge domains",
          "source_object_verbatim": "UniProt",
          "source_object_normalized": "UniProt",
          "source_modality_normalized": "structured biological database record",
          "transformation_chain_verbatim": [
            "protein-phenotype description pairs",
            "instruction tuning example"
          ],
          "model_visible_form_verbatim": "instruction tuning example",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templating",
          "fusion_topology": "other_explicit",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "instruction tuning example",
          "section_heading": "PROCYON-INSTRUCT dataset with 33 million protein-phenotype instructions",
          "supporting_figure_or_table": "Figure 1c",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/50"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_17d54e15b5c5",
          "configuration_id": "config_991a56245b75",
          "route_label": "Enzyme Commission",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "A comprehensive PROCYON-INSTRUCT dataset with 33,899,528 protein-phenotype instructions is curated from five knowledge domains",
          "source_object_verbatim": "Enzyme Commission",
          "source_object_normalized": "Enzyme Commission",
          "source_modality_normalized": "structured biological database record",
          "transformation_chain_verbatim": [
            "We transform the protein-phenotype pairs of PROCYON-INSTRUCT into a multimodal instruction tuning dataset."
          ],
          "model_visible_form_verbatim": "instruction tuning example that consists of a natural language task definition, associated protein inputs, and an expected answer",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templating",
          "fusion_topology": "prefix",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "We transform the protein-phenotype pairs of PROCYON-INSTRUCT into a multimodal instruction tuning dataset.",
          "section_heading": "PROCYON-INSTRUCT dataset with 33 million protein-phenotype instructions",
          "supporting_figure_or_table": "Figure 1c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/50"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0018"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6ea7a3e5f9c4",
          "configuration_id": "config_991a56245b75",
          "route_label": "DisGeNET",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "A comprehensive PROCYON-INSTRUCT dataset with 33,899,528 protein-phenotype instructions is curated from five knowledge domains",
          "source_object_verbatim": "DisGeNET",
          "source_object_normalized": "DisGeNET",
          "source_modality_normalized": "structured biological database record",
          "transformation_chain_verbatim": [
            "For each protein-phenotype pair, we create an instruction tuning example."
          ],
          "model_visible_form_verbatim": "a natural language task definition, associated protein inputs, and an expected answer",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templating",
          "fusion_topology": "prefix",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "For each protein-phenotype pair, we create an instruction tuning example",
          "section_heading": "PROCYON-INSTRUCT dataset with 33 million protein-phenotype instructions",
          "supporting_figure_or_table": "Figure 1c",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The route is explicit; the prior serialized-profile subtype overstates the paper.",
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/50"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0019"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_935a12c569e2",
          "configuration_id": "config_991a56245b75",
          "route_label": "OMIM",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "PROCYON-INSTRUCT dataset",
          "source_object_verbatim": "OMIM",
          "source_object_normalized": "OMIM",
          "source_modality_normalized": "structured biological database record",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "instruction tuning example",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "instruction templating",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "instruction tuning example",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 1c",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/50"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0020"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_010f2669b2a2",
          "configuration_id": "config_5cc6d5f156ab",
          "route_label": "GDA",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "disease-associated protein retrieval across different lines of evidence for association",
          "source_object_verbatim": "GDA",
          "source_object_normalized": "gene-disease association evidence",
          "source_modality_normalized": "structured biological database",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Disease-associated protein retrieval",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "processed into the LLM and compared against a library of proteins",
          "fusion_topology": "retrieval_or_tool_context",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "GDA",
          "section_heading": "Discussion",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            16
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/290"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0021"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_51fc1b08e24b",
          "configuration_id": "config_5cc6d5f156ab",
          "route_label": "STRING Experimental",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "disease-associated protein retrieval across different lines of evidence for association",
          "source_object_verbatim": "STRING Experimental",
          "source_object_normalized": "STRING experimental interaction evidence",
          "source_modality_normalized": "structured biological database",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Disease-associated protein retrieval",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "processed into the LLM and compared against a library of proteins",
          "fusion_topology": "retrieval_or_tool_context",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "STRING Experimental",
          "section_heading": "Discussion",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            16
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/290"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0022"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3d1f794c8397",
          "configuration_id": "config_5cc6d5f156ab",
          "route_label": "STRING Textmining",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "disease-associated protein retrieval across different lines of evidence for association",
          "source_object_verbatim": "STRING Textmining",
          "source_object_normalized": "STRING textmining interaction evidence",
          "source_modality_normalized": "structured biological database",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Disease-associated protein retrieval",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "processed into the LLM and compared against a library of proteins",
          "fusion_topology": "retrieval_or_tool_context",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "STRING Textmining",
          "section_heading": "Discussion",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            16
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/290"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0023"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0b1b43567b95",
          "configuration_id": "config_c81d5d1f9649",
          "route_label": "Protein function",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Protein function retrieval",
          "source_object_verbatim": "GO Function",
          "source_object_normalized": "protein function annotations",
          "source_modality_normalized": "structured biological database",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Protein function retrieval",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "processed into the LLM and compared against a library of proteins",
          "fusion_topology": "retrieval_or_tool_context",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Protein function",
          "section_heading": "Discussion",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/291",
            "#/texts/292",
            "#/texts/293",
            "#/texts/294",
            "#/texts/295",
            "#/texts/296",
            "#/texts/297",
            "#/texts/298"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0024"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_67104b9f17d9",
          "configuration_id": "config_0f578b47e332",
          "route_label": "Cellular component",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cellular component retrieval benchmark",
          "source_object_verbatim": "protein",
          "source_object_normalized": "protein",
          "source_modality_normalized": "protein",
          "transformation_chain_verbatim": [
            "projects them into the unified latent space"
          ],
          "model_visible_form_verbatim": "protein embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "projects them into the unified latent space via a retrieval-specific projector",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 2b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/44"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0025"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_45c11f118d39",
          "configuration_id": "config_fd6d26b9af6e",
          "route_label": "Biological pathway",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Biological pathway retrieval benchmark",
          "source_object_verbatim": "protein",
          "source_object_normalized": "protein",
          "source_modality_normalized": "protein",
          "transformation_chain_verbatim": [
            "projects them into the unified latent space"
          ],
          "model_visible_form_verbatim": "protein embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "projects them into the unified latent space via a retrieval-specific projector",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 2b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/44"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0026"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_058ac010dea2",
          "configuration_id": "config_dee28cec603c",
          "route_label": "Drug-protein interaction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Benchmark retrieval across task types",
          "source_object_verbatim": "protein or prompt embeddings",
          "source_object_normalized": "protein/prompt embeddings",
          "source_modality_normalized": "dense embeddings",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "protein or prompt embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "projected into the unified latent space",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "retrieval of biological functions/components/pathways/diseases/drugs from protein or prompt embeddings",
          "section_heading": "PROCYON accurately retrieves proteins from flexible prompts of phenotypes",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The latent-space retrieval mechanism is explicit; the precise row-specific prompt construction is not isolated in the text.",
          "pages": [
            17
          ],
          "doc_item_refs": [
            "#/pictures/2"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0029"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_23a47183c915",
          "configuration_id": "config_5f87d2b37166",
          "route_label": "Many-shot",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "ProCyon instruction split evaluation",
          "source_object_verbatim": "PROCYON-INSTRUCT",
          "source_object_normalized": "PROCYON-INSTRUCT",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Many-shot",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "ProCyon's retrieval performance across splits of PROCYON-INSTRUCT",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "pairs from many-shot phenotypes are split randomly between train and test",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": "Split label from panel c; exact prompt formatting is not specified.",
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/51"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0030"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1372c917a03e",
          "configuration_id": "config_5f87d2b37166",
          "route_label": "Few-shot",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "ProCyon instruction split evaluation",
          "source_object_verbatim": "PROCYONINSTRUCT",
          "source_object_normalized": "PROCYONINSTRUCT",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "We partition PROCYON-INSTRUCT into three subsets"
          ],
          "model_visible_form_verbatim": "Few-shot",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "ProCyon's retrieval performance across splits of PROCYONINSTRUCT",
          "fusion_topology": "unclear",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We partition PROCYON-INSTRUCT into three subsets",
          "section_heading": "PROCYON accurately retrieves proteins from flexible prompts of phenotypes",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": "Exact few-shot prompt formatting is not specified.",
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/51"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0031"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_205dede67b6c",
          "configuration_id": "config_5f87d2b37166",
          "route_label": "Zero-shot",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "ProCyon instruction split evaluation",
          "source_object_verbatim": "PROCYON-INSTRUCT",
          "source_object_normalized": "PROCYON-INSTRUCT",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Zero-shot",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "ProCyon's retrieval performance across splits of PROCYON-INSTRUCT",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "all pairs from zero-shot phenotypes are assigned to the test set",
          "section_heading": "Results",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": "Split label from panel c; exact prompt formatting is not specified.",
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/51"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0032"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_151e451d06e7",
          "configuration_id": "config_242b4563b4b5",
          "route_label": "Four functions",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "composite prompts composed of four pathways",
          "source_object_verbatim": "four pathways",
          "source_object_normalized": "four pathways",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "composite prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "using composite prompts composed of four pathways",
          "fusion_topology": "retrieval_or_tool_context",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Percentiles of potentially pleiotropic proteins ranked by PROCYON using composite prompts composed of two, three, and four pathways",
          "section_heading": "Discussion",
          "supporting_figure_or_table": "Figure 2e",
          "evidence_status": "text_plus_figure",
          "uncertainty": "Route label inferred from the pathway count in panel e.",
          "pages": [
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/291",
            "#/texts/292",
            "#/texts/293",
            "#/texts/294",
            "#/texts/295",
            "#/texts/296",
            "#/texts/297",
            "#/texts/298"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0035"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5586550693cf",
          "configuration_id": "config_d8d53784b84e",
          "route_label": "Literature",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "External validation of disease-associated protein retrieval across different lines of evidence for association.",
          "source_object_verbatim": "Literature",
          "source_object_normalized": "literature",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Literature",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "different lines of evidence for association",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "External validation of disease-associated protein retrieval across different lines of evidence for association.",
          "section_heading": "Discussion",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            2,
            17
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/30",
            "#/texts/31",
            "#/texts/32",
            "#/texts/33"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0040"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6b0e6e9a158c",
          "configuration_id": "config_2a0fd9641f9f",
          "route_label": "Disease",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Protein retrieval with different disease descriptions.",
          "source_object_verbatim": "Disease",
          "source_object_normalized": "disease",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Disease",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "prompts with gradually increasing information",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "for Autistic Spectrum Disorder",
          "section_heading": "Discussion",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The document gives the example disease as Autistic Spectrum Disorder; the route label itself is Disease.",
          "pages": [
            17
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/541",
            "#/texts/542",
            "#/texts/543",
            "#/texts/544",
            "#/texts/545",
            "#/texts/546",
            "#/texts/547"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0042"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_daf657937cea",
          "configuration_id": "config_2a0fd9641f9f",
          "route_label": "Associated Features supporting Diagnosis",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Protein retrieval with different disease descriptions.",
          "source_object_verbatim": "Associated Features supporting Diagnosis",
          "source_object_normalized": "associated features supporting diagnosis",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Associated Features supporting Diagnosis",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "prompts with gradually increasing information",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Associated Features supporting Diagnosis",
          "section_heading": "Discussion",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The document lists this as a prompt component and does not isolate the full composed prompt.",
          "pages": [
            6,
            7,
            17
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/541",
            "#/texts/542",
            "#/texts/543",
            "#/texts/544",
            "#/texts/57",
            "#/texts/60",
            "#/texts/61"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0044"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d5fcb8c41818",
          "configuration_id": "config_7612559b9f05",
          "route_label": "Precise function",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Precise function",
          "source_object_verbatim": "STING",
          "source_object_normalized": "protein",
          "source_modality_normalized": "protein",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "Precise function",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "more precise prompts are provided as input",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Labels include Basic phenomenon, Mechanistic insights, and Precise function.",
          "section_heading": "Discussion",
          "supporting_figure_or_table": "Figure 2h",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            6,
            7,
            17
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/57",
            "#/texts/60",
            "#/texts/61"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0047"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a141e53f4d01",
          "configuration_id": "config_2b2272ce78af",
          "route_label": "Question answering",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Question answering",
          "source_object_verbatim": "protein sequence and disease description",
          "source_object_normalized": "protein sequence and phenotype description",
          "source_modality_normalized": "multimodal",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "multimodal prompt (protein + phenotype description)",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "multimodal prompt (protein + phenotype description)",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "PROCYON receives a multimodal prompt (protein + phenotype description) and outputs a yes/no answer.",
          "section_heading": "Discussion",
          "supporting_figure_or_table": "Figure 3a",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            17,
            18
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/texts/36",
            "#/texts/37",
            "#/texts/38",
            "#/texts/549",
            "#/texts/551"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0048"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f42fa0ac09e5",
          "configuration_id": "config_404ae1a44c21",
          "route_label": "therapeutics",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "PROCYON-INSTRUCT, an instruction tuning dataset that spans 12 data sources and 5 knowledge domains",
          "source_object_verbatim": "protein-drug associations",
          "source_object_normalized": "protein-drug associations",
          "source_modality_normalized": "mixed",
          "transformation_chain_verbatim": [
            "protein-phenotype pairs",
            "instruction tuning example"
          ],
          "model_visible_form_verbatim": "a natural language task definition, associated protein inputs, and an expected answer",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "associated protein inputs",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "therapeutics , covering protein-drug associations",
          "section_heading": "PROCYON-INSTRUCT dataset with 33 million protein-phenotype instructions",
          "supporting_figure_or_table": "Supplementary Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The document does not give a therapeutics-specific template, so the task/form fields are inferred from the general instruction-tuning description.",
          "pages": [
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/47"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0053"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_78cbb264c279",
          "configuration_id": "config_404ae1a44c21",
          "route_label": "protein domains",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "PROCYON-INSTRUCT, an instruction tuning dataset that spans 12 data sources and 5 knowledge domains",
          "source_object_verbatim": "evolutionarily-conserved functional sub-units of proteins",
          "source_object_normalized": "protein domains",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "protein-phenotype pairs",
            "instruction tuning example"
          ],
          "model_visible_form_verbatim": "a natural language task definition, associated protein inputs, and an expected answer",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "associated protein inputs",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "protein domains , covering evolutionarily-conserved functional sub-units of proteins",
          "section_heading": "PROCYON-INSTRUCT dataset with 33 million protein-phenotype instructions",
          "supporting_figure_or_table": "Supplementary Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The document does not give a protein-domains-specific template, so the task/form fields are inferred from the general instruction-tuning description.",
          "pages": [
            4,
            5,
            8,
            9,
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/291",
            "#/texts/292",
            "#/texts/293",
            "#/texts/47",
            "#/texts/50",
            "#/texts/51",
            "#/texts/69",
            "#/texts/72"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0054"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e03e955f25bb",
          "configuration_id": "config_404ae1a44c21",
          "route_label": "interaction",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "PROCYON-INSTRUCT, an instruction tuning dataset that spans 12 data sources and 5 knowledge domains",
          "source_object_verbatim": "protein-protein and protein-peptide interactions",
          "source_object_normalized": "protein interactions",
          "source_modality_normalized": "protein/peptide",
          "transformation_chain_verbatim": [
            "protein-phenotype pairs",
            "instruction tuning example"
          ],
          "model_visible_form_verbatim": "a natural language task definition, associated protein inputs, and an expected answer",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "associated protein inputs",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "interaction , covering protein-protein and protein-peptide interactions",
          "section_heading": "PROCYON-INSTRUCT dataset with 33 million protein-phenotype instructions",
          "supporting_figure_or_table": "Supplementary Table 1",
          "evidence_status": "explicit_text",
          "uncertainty": "The document does not give an interaction-specific template, so the task/form fields are inferred from the general instruction-tuning description.",
          "pages": [
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/47"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0055"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_aab758294f91"
    },
    {
      "model_id": "model_b1f724b128da",
      "model_name": "PROCYON-SPLIT",
      "record_id": "full_2026-07-06__rec_000086",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_f6eff70c6863",
      "paper_title": "PROCYON: A multimodal foundation model for protein phenotypes.",
      "doi": "10.1101/2024.12.10.627665",
      "paper_url": "https://doi.org/10.1101/2024.12.10.627665",
      "route_count": 6,
      "configuration_count": 3,
      "family_counts": {
        "text_native_token_stream": 3,
        "discrete_biological_symbol_stream": 2,
        "geometric_or_diffusion_state_carrier": 1
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 3,
        "native_biological_token_stream": 2,
        "coordinate_backbone_or_shape_conditioning": 1
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream",
        "geometric_or_diffusion_state_carrier"
      ],
      "subtypes": [
        "coordinate_backbone_or_shape_conditioning",
        "native_biological_token_stream",
        "plain_language_prompt_or_question"
      ],
      "primary_subtype": "plain_language_prompt_or_question",
      "modalities": [
        "protein",
        "protein structure",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "concatenation",
        "query_bottleneck",
        "shared_latent_alignment",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_000086_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_000086_f51592e07fe3/figure_002.png",
        "figure_index": 2,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific figure labeled **a**, **b**, and **c** describing **ProCyon**, a unified latent-space framework for protein and phenotype tasks.\n\nPanel **a** shows a table-like overview of tasks under “ProCyon” and “Unified latent space of proteins and phenotypes.” Visible task rows include:\n- **Protein function retrieval**\n- **Domain based protein retrieval**\n- **Protein interaction prediction**\n- **Disease associated protein retrieval**\n- **Contextual target retrieval**\n\nEach row pairs a task name with example prompts and outputs, such as retrieving catalytic proteins in glycolysis, proteins with SH3 domains, interacting proteins involved in autophagy, proteins involved in Parkinson’s disease pathology, and drug-target domain identification.\n\nPanel **b** shows two workflow diagrams:\n- **Protein Retrieval Prompt**: a user provides a disease description, which is processed through a **Protein LM**, **LLM**, and retrieval model against a large protein library, producing retrieved proteins with scores.\n- **Phenotype Generation Prompt**: a user provides a protein input, processed through a **Protein LM**, **structural encoder**, **LLM**, and an **autoregressive text decoder**, producing phenotype descriptions such as disease associations.\n\nPanel **c** shows a radial bar chart summarizing task/data categories by sample count. Categories are grouped around the chart as:\n- **Function**\n- **Disease**\n- **Therapeutics**\n- **Domain**\n- **Interaction**\n\nVisible sources or labels include **GO Function**, **GO Process**, **GO Component**, **Reactome**, **UniProt**, **Enzyme Commission**, **DisGeNET**, **OMIM**, **DrugBank Carrier**, **DrugBank Enzyme**, **DrugBank Target**, **DrugBank Transporter**, **GDA**, **Pfam**, **STRING Experimental**, **STRING Textmining**, and **STRING Coexpression**. The radial scale indicates sample counts from about **1,000** up to **3,000,000**.",
        "page_no": 16,
        "sha256": "fe9bee26ab45b4232708aaa1db16267880ab322b26c3afe85e7ee283bbc9a6e2",
        "pixel_width": 853,
        "pixel_height": 415,
        "crop_box": {
          "x": 0.0,
          "y": 0.39,
          "width": 0.57,
          "height": 0.31
        },
        "panel_label": "b",
        "visible_input_object": "Natural-language disease prompt for protein retrieval",
        "visible_model_interface": "Protein LM and LLM projection into retrieval pipeline with the Protein retrieval block",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop isolates panel b’s top protein-retrieval workflow, keeping the source prompt, processing arrows, Protein LM/LLM blocks, and the retrieval interface while excluding panel a/c and output-only result listings.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "coordinate_backbone_or_shape_conditioning",
          "family_id": "geometric_or_diffusion_state_carrier",
          "route_id": "route_4c1c9b303c55",
          "example_input": "residue/atom coordinates (xᵢ,yᵢ,zᵢ)",
          "example_carrier": "equivariant geometric state",
          "example_interface": "geometry-aware generator",
          "actual_source": "protein structure",
          "actual_model_visible_form": "protein structure encoded via the modality-specific encoders"
        },
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_de75413d9ba2",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "protein sequences",
          "actual_model_visible_form": "protein-sequence embeddings projected into the unified latent space"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_440ad7fad63b",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "natural language prompt",
          "actual_model_visible_form": "the LLM-processed prompt projected into the unified latent space"
        }
      ],
      "routes": [
        {
          "route_id": "route_440ad7fad63b",
          "configuration_id": "config_f42dcc13c120",
          "route_label": "Protein retrieval query prompt",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "protein retrieval",
          "source_object_verbatim": "natural language prompt",
          "source_object_normalized": "natural language prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "the LLM-processed prompt projected into the unified latent space by a query connector"
          ],
          "model_visible_form_verbatim": "the LLM-processed prompt projected into the unified latent space",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "the natural language prompt is then processed by the LLM and projected into the unified latent space by a query connector",
          "fusion_topology": "query_bottleneck",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "the natural language prompt is then processed by the LLM and projected into the unified latent space by a query connector",
          "section_heading": "PROCYON model architecture",
          "supporting_figure_or_table": "Figure 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            12,
            13,
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/36",
            "#/texts/37",
            "#/texts/38",
            "#/texts/43",
            "#/texts/44",
            "#/texts/45",
            "#/texts/96",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000086::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0052"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_de75413d9ba2",
          "configuration_id": "config_f42dcc13c120",
          "route_label": "Protein retrieval protein sequence",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "protein retrieval",
          "source_object_verbatim": "protein sequences",
          "source_object_normalized": "protein sequence",
          "source_modality_normalized": "protein",
          "transformation_chain_verbatim": [
            "the protein language model processes protein sequences",
            "projects them into the unified latent space via a retrieval-specific projector"
          ],
          "model_visible_form_verbatim": "protein-sequence embeddings projected into the unified latent space",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "the protein language model processes protein sequences and projects them into the unified latent space via a retrieval-specific projector",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "the protein language model processes protein sequences and projects them into the unified latent space via a retrieval-specific projector",
          "section_heading": "PROCYON model architecture",
          "supporting_figure_or_table": "Figure 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            12,
            13,
            16,
            17
          ],
          "doc_item_refs": [
            "#/texts/36",
            "#/texts/37",
            "#/texts/38",
            "#/texts/43",
            "#/texts/44",
            "#/texts/45",
            "#/texts/96",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000086::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0052"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_caa6f466af38",
          "configuration_id": "config_d9c453530470",
          "route_label": "Phenotype generation text prompt",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "phenotype generation",
          "source_object_verbatim": "the user's prompt",
          "source_object_normalized": "user prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "the text is tokenized via a standard text tokenizer"
          ],
          "model_visible_form_verbatim": "text tokenized via a standard text tokenizer",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "the text is tokenized via a standard text tokenizer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "the text is tokenized via a standard text tokenizer",
          "section_heading": "PROCYON model architecture",
          "supporting_figure_or_table": "Figure 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            17,
            18
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/texts/43",
            "#/texts/44",
            "#/texts/45",
            "#/texts/549",
            "#/texts/551"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000086::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0049",
            "dense::full_2026-07-06__rec_000086::0050",
            "dense::full_2026-07-06__rec_000086::0051"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c4afa80469f5",
          "configuration_id": "config_d9c453530470",
          "route_label": "Phenotype generation protein sequence",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "phenotype generation",
          "source_object_verbatim": "protein sequence",
          "source_object_normalized": "protein sequence",
          "source_modality_normalized": "protein",
          "transformation_chain_verbatim": [
            "the protein sequence is encoded via the modality-specific encoders"
          ],
          "model_visible_form_verbatim": "protein sequence encoded via the modality-specific encoders",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "the protein sequence and structure are encoded via the modality-specific encoders",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "the protein sequence and structure are encoded via the modality-specific encoders",
          "section_heading": "PROCYON model architecture",
          "supporting_figure_or_table": "Figure 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            17,
            18
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/texts/43",
            "#/texts/44",
            "#/texts/45",
            "#/texts/549",
            "#/texts/551"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000086::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0049",
            "dense::full_2026-07-06__rec_000086::0050",
            "dense::full_2026-07-06__rec_000086::0051"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4c1c9b303c55",
          "configuration_id": "config_d9c453530470",
          "route_label": "Phenotype generation protein structure",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "phenotype generation",
          "source_object_verbatim": "protein structure",
          "source_object_normalized": "protein structure",
          "source_modality_normalized": "protein structure",
          "transformation_chain_verbatim": [
            "the protein structure is encoded via the modality-specific encoders"
          ],
          "model_visible_form_verbatim": "protein structure encoded via the modality-specific encoders",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "coordinate_backbone_or_shape_conditioning",
          "insertion_or_fusion_verbatim": "the protein sequence and structure are encoded via the modality-specific encoders",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "the protein sequence and structure are encoded via the modality-specific encoders",
          "section_heading": "PROCYON model architecture",
          "supporting_figure_or_table": "Figure 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            17,
            18
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/texts/43",
            "#/texts/44",
            "#/texts/45",
            "#/texts/549",
            "#/texts/551"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000086::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0049",
            "dense::full_2026-07-06__rec_000086::0050",
            "dense::full_2026-07-06__rec_000086::0051"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4b1ea824da6c",
          "configuration_id": "config_72c35ec819bd",
          "route_label": "Diagnostic Features",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Using descriptions from the Diagnostic and Statistical Manual of Mental Disorders (DSM-5), we incrementally expand prompts from the disease name alone ('Disease') to include 'Diagnostic Features' and 'Associated Features Supporting Diagnosis.'",
          "source_object_verbatim": "Diagnostic Features",
          "source_object_normalized": "diagnostic features",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "Disease",
            "Diagnostic Features",
            "Associated Features Supporting Diagnosis"
          ],
          "model_visible_form_verbatim": "descriptions from the Diagnostic and Statistical Manual of Mental Disorders (DSM-5)",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "expand prompts from the disease name alone",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "include 'Diagnostic Features' and 'Associated Features Supporting Diagnosis.'",
          "section_heading": "PROCYON accurately retrieves proteins from flexible prompts of phenotypes",
          "supporting_figure_or_table": "Figure 2g",
          "evidence_status": "text_plus_figure",
          "uncertainty": "This route is tied to the evaluation prompt-expansion experiment rather than a training example.",
          "pages": [
            6,
            7,
            17
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/541",
            "#/texts/542",
            "#/texts/543",
            "#/texts/544",
            "#/texts/57",
            "#/texts/60",
            "#/texts/61"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000086::0056"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_af3baba835a0"
    },
    {
      "model_id": "model_ca02cecd45be",
      "model_name": "Qwen2.5-7B-Instruct",
      "record_id": "full_2026-07-06__rec_001617",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_4f442dcf4ed4",
      "paper_title": "Cell-o1: Training LLMs to Solve Single-Cell Reasoning Puzzles with Reinforcement Learning",
      "doi": "10.48550/arXiv.2506.02911",
      "paper_url": "https://doi.org/10.48550/arXiv.2506.02911",
      "route_count": 3,
      "configuration_count": 3,
      "family_counts": {
        "text_native_token_stream": 3
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 3
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell RNA sequencing batch with donor metadata",
        "single-cell RNA sequencing cell with donor metadata"
      ],
      "lifecycle_phases": [
        "evaluation"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001617_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001617_76b0056d59c3/figure_002.png",
        "figure_index": 2,
        "caption": "Figure 2: CellPuzzles formulates cell type annotation as a batch-level reasoning task that integrates gene expression and contextual metadata, inspired by how experts annotate cells in practice.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic comparison of two cell type annotation workflows with four main panel regions arranged in two rows.\n\nTop row: A human expert task workflow. The left “Input” panel shows clustered single-cell data with colored clusters labeled Cluster 2, Cluster 3, Cluster 4, and Cluster N, each associated with marker gene lists such as “SFTPC, SFTPA1, SFTPB, NAPSA, ...”, “C1QA, APOE, MARCO, LYZ, ...”, and “IGHM, MZB1, JCHAIN, XBP1, ...”. The center panel labeled “Human Expert Task” shows an expert assigning a cell type label to each cluster using representative marker genes, contextual metadata, biological knowledge, and reference sources. The right “Output” panel lists assigned cell types including Pulmonary Alveolar Type 2 (AT2) Cells, CD8+ Cytotoxic T Cells, Lung Pericyte, Alveolar Macrophages, and Plasma Cells.\n\nBottom row: A “CellPuzzles Task” workflow. The left “Input” panel shows individual cells from clusters, with representative sampled cells labeled [Cell 1], [Cell 2], [Cell 3], [Cell 4], and [Cell N], each paired with gene lists such as “S100A9, TMSB10, RPL37, ...”, “MALAT1, FTL, B2M, ...”, “MALAT1, FTL, AKR1B10, ...”, “B2M, MT2A, TMSB4X, ...”, and “IGLC3, IGLC2, IGHM, ...”. Arrows indicate selected representative cells from clusters. The center panel labeled “CellPuzzles Task” shows a model/interface labeled “Cell-o1” receiving N cells, contextual metadata, and N candidate cell types, then reasoning to determine the optimal label assignment and provide reasoning traces. The right “Output” panel shows colored cell-to-label assignment lines connecting cells to candidate labels including CD4-positive, alpha-beta T Cell; CD8-positive, alpha-beta T Cell; Smooth Muscle Cell; Lung Macrophage; Non-classical Monocyte; Capillary Endothelial Cell; Plasma Cell; and Respiratory Basal Cell.\n\nBiological source objects include single-cell clusters, individual cells, marker genes, and immune/lung-related cell type labels. The figure depicts a transformation from marker gene or cell-level input data to annotated biological cell type outputs, comparing expert manual annotation with an automated CellPuzzles/Cell-o1 reasoning task.",
        "page_no": 3,
        "sha256": "e74192b02a6c544c4d4ae6c01c1b71c75747f0270a5e4e6bc77aba0c8a2259ee",
        "pixel_width": 791,
        "pixel_height": 318,
        "crop_box": {
          "x": 0,
          "y": 0.46,
          "width": 0.72,
          "height": 0.54
        },
        "panel_label": "Bottom-left CellPuzzles input plus task interface",
        "visible_input_object": "Representative cells sampled from clusters, with top-expressed gene lists and arrows from clusters to selected cells.",
        "visible_model_interface": "CellPuzzles task box showing Cell-o1 receiving N cells, contextual metadata, and candidate cell types for batch-level label assignment.",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the lower-left batch input and the adjacent CellPuzzles task panel, so the cluster-to-representative-cell arrows, sampled cell labels, gene lists, and the model-facing batch prompt scaffold remain readable while the output-only right panel is excluded.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_f6a3945350e8",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "the gene expression profile of a single cell from a specific donor",
          "actual_model_visible_form": "a single-cell prompt with donor context and candidate labels, without reasoning traces"
        }
      ],
      "routes": [
        {
          "route_id": "route_f6a3945350e8",
          "configuration_id": "config_7de96db454db",
          "route_label": "cell-level prediction prompt (Qwen2.5-7B-Instruct)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Cell-level Prediction Setup",
          "source_object_verbatim": "the gene expression profile of a single cell from a specific donor",
          "source_object_normalized": "single cell from a specific donor with top-expressed genes, donor context, and candidate labels",
          "source_modality_normalized": "single-cell RNA sequencing cell with donor metadata",
          "transformation_chain_verbatim": [
            "use the single cell's top expressed genes",
            "combine them with donor context",
            "provide the fixed candidate label set",
            "directly classify without reasoning traces"
          ],
          "model_visible_form_verbatim": "a single-cell prompt with donor context and candidate labels, without reasoning traces",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "direct prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We consider two variants under this cell-level setup: Cell-level Prediction Setup : Instruction-tuned models that have not been exposed to any reasoning tasks. These models directly predict answers from the input without generating reasoning traces. Cell-level Reasoning Setup : Models capable of generating reasoning traces are evaluated on individual cells in isolation, requiring them to infer labels without the help of contextual comparisons among cells.",
          "section_heading": "B.2 Baselines",
          "supporting_figure_or_table": "Table 11",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            21
          ],
          "doc_item_refs": [
            "#/texts/1015",
            "#/texts/1016",
            "#/texts/1017",
            "#/texts/1018",
            "#/texts/1019",
            "#/texts/1020"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d1359ce2ae2e",
          "configuration_id": "config_0c32a45ad13a",
          "route_label": "open-ended QA prompt (Qwen2.5-7B-Instruct)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Open-ended QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "provide gene expression profile and metadata",
            "remove the constrained label set",
            "ask for free-form cell type generation"
          ],
          "model_visible_form_verbatim": "free-form textual generation of a cell type name for each cell",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompting without constrained labels",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_016"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e8e195e0b315",
          "configuration_id": "config_aa11e3609f0b",
          "route_label": "constrained QA prompt (Qwen2.5-7B-Instruct)",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Constrained QA Setup",
          "source_object_verbatim": "a batch of cells in a given batch, based on its gene expression profile and metadata",
          "source_object_normalized": "batch of cells with gene expression profiles and donor metadata",
          "source_modality_normalized": "single-cell RNA sequencing batch with donor metadata",
          "transformation_chain_verbatim": [
            "rank top-expressed genes per cell",
            "convert donor metadata into natural language context",
            "attach a predefined candidate label set",
            "require a single ordered answer string"
          ],
          "model_visible_form_verbatim": "a structured batch-level text prompt with candidate labels and ordered answer output",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "predefined candidate label set with structured prompt",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Although LLMs are naturally suited for open-ended question answering (QA), we find this formulation to be suboptimal for the task of cell type annotation. In the open-ended QA setup, the model is prompted to freely generate a cell type name for each cell in a given batch, based on its gene expression profile and metadata, without access to a constrained label set.",
          "section_heading": "D Open-ended QA vs. Constrained QA",
          "supporting_figure_or_table": "Table 12",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/9",
            "#/texts/1025",
            "#/texts/1026",
            "#/texts/1027",
            "#/texts/1028",
            "#/texts/1029",
            "#/texts/1031"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001617::route_022"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_9d05ddb97916",
      "model_name": "Qwen3",
      "record_id": "full_2026-07-06__rec_001074",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_08046f30aead",
      "paper_title": "BIOREASON: Incentivizing Multimodal Biological Reasoning within a DNA-LLM Model",
      "doi": "10.48550/arXiv.2505.23579",
      "paper_url": "https://doi.org/10.48550/arXiv.2505.23579",
      "route_count": 7,
      "configuration_count": 4,
      "family_counts": {
        "dense_continuous_carrier": 4,
        "text_native_token_stream": 3
      },
      "subtype_counts": {
        "direct_projected_embedding": 4,
        "structured_biological_prompt_or_task_scaffold": 3
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "DNA",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "concatenation",
        "unclear"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001074_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001074_c72d500e989d/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: BIOREASON Architecture. Schematic representation of our novel multimodal framework that integrates a DNA foundation model with a Large Language Model.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow for a multimodal genomics/text language-model pipeline.\n\nVisible components:\n- Left panel: DNA sequences are shown as stacked nucleotide strings with one highlighted base in red. These sequences feed upward into a frozen `Evo2-1B` model.\n- A “Learnable Linear Projection” maps representations from `dim Evo2` to `dim Qwen3`.\n- Top panel: “Stacked DNA and Text Embeddings” shows projected DNA embeddings combined with text-token embeddings in a single sequence.\n- Middle interface: a `Qwen3 Tokenizer` converts a BioReason prompt/template into tokens, including special placeholders such as `<|dna_start|>`, `<|dna_pad|>`, and `<|dna_end|>`.\n- The tokens are mapped through the `Qwen3 Embedding Matrix`.\n- Bottom middle panel: a “BioReason Chat Template” contains a structured biological question:\n  - Chromosome Number: 12\n  - Pathway: “LRRK2* -> CYCS -- APAF1 -> CASP9 -> CASP3”\n  - Genes in pathway include LRRK2, CYCS, APAF1, CASP9, and CASP3.\n  - The question asks about the biological effect of an `LRRK2` allele and its disease contribution.\n- Right panel: the combined embeddings are passed into `Qwen3-4B`.\n- Output panel: model response with reasoning identifies a `C>G` substitution at position `40310433` on chromosome 12 in the `LRRK2` gene, describes it as likely gain-of-function affecting kinase activity, and concludes: “Parkinson’s disease.”\n\nBiological source objects:\n- DNA sequence snippets\n- Chromosome 12\n- LRRK2 allele/variant\n- Apoptosis-related pathway genes: CYCS, APAF1, CASP9, CASP3\n\nTransformation shown:\nDNA sequence → Evo2-1B embeddings → learned linear projection → Qwen3-compatible embedding space → stacked with text embeddings → Qwen3-4B reasoning output.\n\nModel interfaces:\n- Frozen Evo2-1B encoder\n- Learnable linear projection\n- Qwen3 tokenizer\n- Qwen3 embedding matrix\n- Qwen3-4B language model\n- BioReason chat template with embedded DNA placeholders\n\nFinding depicted:\nThe example response associates a chromosome 12 `LRRK2` mutation with Parkinson’s disease through altered LRRK2 kinase activity.",
        "page_no": 4,
        "sha256": "fffbebce383be4f4065f5cd948aab8480eda41347d6e4c47569b8ab4c6d35af4",
        "pixel_width": 773,
        "pixel_height": 402,
        "crop_box": {
          "x": 0.0,
          "y": 0.0,
          "width": 0.79,
          "height": 0.83
        },
        "panel_label": "DNA-to-Qwen3 input pathway",
        "visible_input_object": "DNA sequences plus the BioReason chat-template query for chromosome 12 / LRRK2",
        "visible_model_interface": "Frozen Evo2-1B encoder, learnable linear projection, Qwen3 tokenizer and embedding matrix, stacked DNA and text embeddings",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the source DNA panel, the Evo2-to-Qwen3 projection, the tokenizer/template interface, and the stacked embedding pathway while excluding the output-only answer panel.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_6b88a24df658",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "paired reference and variant DNA sequences (S_DNA)",
          "actual_model_visible_form": "projected DNA embeddings stacked into the multimodal input sequence X_LLM"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_63d3232eb95a",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "textual queries, Q_TEXT",
          "actual_model_visible_form": "tokenized query embeddings E_Q text"
        }
      ],
      "routes": [
        {
          "route_id": "route_6b88a24df658",
          "configuration_id": "config_9cb0e556b0e9",
          "route_label": "BIOREASON KEGG DNA sequence route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "KEGG-derived biological reasoning dataset",
          "source_object_verbatim": "paired reference and variant DNA sequences (S_DNA)",
          "source_object_normalized": "paired reference and variant DNA sequences",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "DNA-specific tokenizer T_DNA",
            "f_DNA encoder such as StripedHyena2 (Evo2) or the Nucleotide Transformer (NT)",
            "learnable linear projection Proj",
            "stacking with text embeddings and special tokens <dna_start> and <dna_end>",
            "Rotary Position Embedding (RoPE) within Qwen3"
          ],
          "model_visible_form_verbatim": "projected DNA embeddings stacked into the multimodal input sequence X_LLM",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "stacked with embeddings of the user's query Q_TEXT to form X_LLM for the core LLM",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "BIOREASON operates on two primary input streams: (i) one or more genomic sequences, denoted S DNA; and (ii) textual queries, Q TEXT.",
          "section_heading": "3 BioReason Model",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            20
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/texts/266",
            "#/texts/267",
            "#/texts/268",
            "#/texts/53",
            "#/texts/55",
            "#/texts/63",
            "#/texts/65",
            "#/texts/66",
            "#/texts/67",
            "#/texts/75",
            "#/texts/76",
            "#/texts/77",
            "#/texts/79",
            "#/texts/81",
            "#/texts/89",
            "#/texts/90"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001074::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001074::0003",
            "dense::full_2026-07-06__rec_001074::0004",
            "dense::full_2026-07-06__rec_001074::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_63d3232eb95a",
          "configuration_id": "config_9cb0e556b0e9",
          "route_label": "BIOREASON KEGG text query route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "KEGG-derived biological reasoning dataset",
          "source_object_verbatim": "textual queries, Q_TEXT",
          "source_object_normalized": "textual queries",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "LLM-specific tokenizer T_LLM",
            "Qwen3 input embeddings",
            "stacking with projected DNA embeddings and special tokens",
            "Rotary Position Embedding (RoPE) within Qwen3"
          ],
          "model_visible_form_verbatim": "tokenized query embeddings E_Q text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "stacked with DNA embeddings to form X_LLM for the core LLM",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "BIOREASON operates on two primary input streams: (i) one or more genomic sequences, denoted S DNA; and (ii) textual queries, Q TEXT.",
          "section_heading": "3 BioReason Model",
          "supporting_figure_or_table": "Figure 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            20
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/texts/266",
            "#/texts/267",
            "#/texts/268",
            "#/texts/53",
            "#/texts/55",
            "#/texts/63",
            "#/texts/65",
            "#/texts/66",
            "#/texts/67",
            "#/texts/75",
            "#/texts/76",
            "#/texts/77",
            "#/texts/79",
            "#/texts/81",
            "#/texts/89",
            "#/texts/90"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001074::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001074::0003",
            "dense::full_2026-07-06__rec_001074::0004",
            "dense::full_2026-07-06__rec_001074::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_9f2e4378ab4e",
          "configuration_id": "config_ab760363a2a6",
          "route_label": "BIOREASON VEP-Coding DNA sequence route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Variant Effect Prediction of Coding Sequences",
          "source_object_verbatim": "paired reference and variant DNA sequences (S_DNA)",
          "source_object_normalized": "paired reference and variant DNA sequences",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "DNA-specific tokenizer T_DNA",
            "f_DNA encoder such as Evo2-1B or Nucleotide Transformer (NT-500M)",
            "learnable linear projection Proj",
            "stacking with text embeddings and special tokens",
            "Rotary Position Embedding (RoPE) within Qwen3"
          ],
          "model_visible_form_verbatim": "projected DNA embeddings stacked into the multimodal input sequence X_LLM",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "stacked with embeddings of the user's query Q_TEXT to form X_LLM for the core LLM",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Variant Effect Prediction of Coding Sequences (VEP-Coding). Comprising 50,083 core variant entries, this dataset tests classifying coding variants. Input: paired reference and variant DNA sequences ( S DNA ), and a textual query ( Q TEXT ) providing gene and chromosome context. Task: sequence generation to predict if a variant is benign, or pathogenic with its associated disease. Split: Chromosomes (Chr) 1-7, 9-22, X, Y for train/validation; Chr 8 for testing.",
          "section_heading": "5.1 Datasets",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            7,
            20
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/266",
            "#/texts/267",
            "#/texts/268",
            "#/texts/53",
            "#/texts/55",
            "#/texts/63",
            "#/texts/65",
            "#/texts/66",
            "#/texts/67",
            "#/texts/89",
            "#/texts/90",
            "#/texts/91",
            "#/texts/92"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001074::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001074::0080",
            "dense::full_2026-07-06__rec_001074::0003",
            "dense::full_2026-07-06__rec_001074::0004",
            "dense::full_2026-07-06__rec_001074::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f1557c915aec",
          "configuration_id": "config_ab760363a2a6",
          "route_label": "BIOREASON VEP-Coding text query route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Variant Effect Prediction of Coding Sequences",
          "source_object_verbatim": "textual query providing gene and chromosome context",
          "source_object_normalized": "textual query",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "LLM-specific tokenizer T_LLM",
            "Qwen3 input embeddings",
            "stacking with projected DNA embeddings and special tokens",
            "Rotary Position Embedding (RoPE) within Qwen3"
          ],
          "model_visible_form_verbatim": "tokenized query embeddings",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "stacked with DNA embeddings to form X_LLM",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Variant Effect Prediction of Coding Sequences (VEP-Coding). Comprising 50,083 core variant entries, this dataset tests classifying coding variants. Input: paired reference and variant DNA sequences ( S DNA ), and a textual query ( Q TEXT ) providing gene and chromosome context. Task: sequence generation to predict if a variant is benign, or pathogenic with its associated disease. Split: Chromosomes (Chr) 1-7, 9-22, X, Y for train/validation; Chr 8 for testing.",
          "section_heading": "5.1 Datasets",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            7,
            20
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/266",
            "#/texts/267",
            "#/texts/268",
            "#/texts/53",
            "#/texts/55",
            "#/texts/63",
            "#/texts/65",
            "#/texts/66",
            "#/texts/67",
            "#/texts/89",
            "#/texts/90",
            "#/texts/91",
            "#/texts/92"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001074::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001074::0080",
            "dense::full_2026-07-06__rec_001074::0003",
            "dense::full_2026-07-06__rec_001074::0004",
            "dense::full_2026-07-06__rec_001074::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_909fab50c354",
          "configuration_id": "config_b328edf6e110",
          "route_label": "BIOREASON VEP-Non-SNV DNA sequence route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Variant Effect Prediction of Coding Non-SNVs",
          "source_object_verbatim": "paired reference and variant DNA sequences (S_DNA)",
          "source_object_normalized": "paired reference and variant DNA sequences",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "DNA-specific tokenizer T_DNA",
            "f_DNA encoder such as Evo2-1B or Nucleotide Transformer (NT-500M)",
            "learnable linear projection Proj",
            "stacking with text embeddings and special tokens",
            "Rotary Position Embedding (RoPE) within Qwen3"
          ],
          "model_visible_form_verbatim": "projected DNA embeddings stacked into the multimodal input sequence X_LLM",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "stacked with embeddings of the user's query Q_TEXT to form X_LLM for the core LLM",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Variant Effect Prediction of Coding Non-SNVs (VEP-Non-SNV). Containing 36,088 core nonSNV entries, this dataset addresses non-SNV alterations (e.g., indels <64 bp). Input: paired reference and variant DNA sequences ( S DNA ), and an augmented textual query ( Q TEXT ) providing gene and chromosome context. Task: sequence generation to predict if a non-SNV is benign, or pathogenic with its associated disease(s). We used stratified train/test splits to ensure balanced disease representation.",
          "section_heading": "5.1 Datasets",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            20
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/266",
            "#/texts/267",
            "#/texts/268",
            "#/texts/53",
            "#/texts/55",
            "#/texts/63",
            "#/texts/65",
            "#/texts/66",
            "#/texts/67",
            "#/texts/85",
            "#/texts/89",
            "#/texts/90",
            "#/texts/91",
            "#/texts/92"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001074::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001074::0081",
            "dense::full_2026-07-06__rec_001074::0003",
            "dense::full_2026-07-06__rec_001074::0004",
            "dense::full_2026-07-06__rec_001074::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d16851ea3c52",
          "configuration_id": "config_b328edf6e110",
          "route_label": "BIOREASON VEP-Non-SNV text query route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Variant Effect Prediction of Coding Non-SNVs",
          "source_object_verbatim": "augmented textual query providing gene and chromosome context",
          "source_object_normalized": "textual query",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "LLM-specific tokenizer T_LLM",
            "Qwen3 input embeddings",
            "stacking with projected DNA embeddings and special tokens",
            "Rotary Position Embedding (RoPE) within Qwen3"
          ],
          "model_visible_form_verbatim": "tokenized query embeddings",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "stacked with DNA embeddings to form X_LLM",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Variant Effect Prediction of Coding Non-SNVs (VEP-Non-SNV). Containing 36,088 core nonSNV entries, this dataset addresses non-SNV alterations (e.g., indels <64 bp). Input: paired reference and variant DNA sequences ( S DNA ), and an augmented textual query ( Q TEXT ) providing gene and chromosome context. Task: sequence generation to predict if a non-SNV is benign, or pathogenic with its associated disease(s). We used stratified train/test splits to ensure balanced disease representation.",
          "section_heading": "5.1 Datasets",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            20
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/266",
            "#/texts/267",
            "#/texts/268",
            "#/texts/53",
            "#/texts/55",
            "#/texts/63",
            "#/texts/65",
            "#/texts/66",
            "#/texts/67",
            "#/texts/85",
            "#/texts/89",
            "#/texts/90",
            "#/texts/91",
            "#/texts/92"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001074::route_010"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001074::0081",
            "dense::full_2026-07-06__rec_001074::0003",
            "dense::full_2026-07-06__rec_001074::0004",
            "dense::full_2026-07-06__rec_001074::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1945b7def38e",
          "configuration_id": "config_0e1cf756a45e",
          "route_label": "BIOREASON Enformer chromatin accessibility route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "chromatin accessibility prediction across 20 DNase-seq tracks",
          "source_object_verbatim": "human genome segmented into 200bp bins",
          "source_object_normalized": "human genome segmented into 200bp bins",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "segmentation into 200bp bins",
            "Enformer encoder",
            "integrating Enformer embeddings with Qwen3 backbones"
          ],
          "model_visible_form_verbatim": "Enformer embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "integrating Enformer embeddings with Qwen3 backbones",
          "fusion_topology": "unclear",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Enformer-Qwen3-1.7B and Enformer-Qwen3-4B (DNA-LLM): BIOREASON hybrids integrating Enformer embeddings with Qwen3 backbones.",
          "section_heading": "B.2 Model Configurations",
          "supporting_figure_or_table": "Table 3",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper confirms Enformer embeddings are integrated with Qwen3 backbones, but it does not spell out the exact fusion operator.",
          "pages": [
            7,
            8,
            20
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/tables/1",
            "#/texts/100",
            "#/texts/102",
            "#/texts/103",
            "#/texts/104",
            "#/texts/257",
            "#/texts/258",
            "#/texts/260",
            "#/texts/261",
            "#/texts/262",
            "#/texts/263",
            "#/texts/264",
            "#/texts/98",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001074::route_013"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001074::0046",
            "dense::full_2026-07-06__rec_001074::0083"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_c17be9263882"
    },
    {
      "model_id": "model_ac7794e51425",
      "model_name": "Qwen3-1.7B",
      "record_id": "july_update_2026-07-06__rec_000050",
      "collection_batch_id": "july_update_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_cccbb8d94eda",
      "paper_title": "How Post-Training Shapes Biological Reasoning Models",
      "doi": "",
      "paper_url": "",
      "route_count": 2,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 1,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "direct_projected_embedding": 1,
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "DNA",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "concatenation",
        "prefix"
      ],
      "text_roles": [
        "biological_payload",
        "metadata_or_context"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "No figure visibly supports either VEP input route. Figure 1 is a broad training/modality schematic with generic DNA/RNA/protein examples, but it does not show paired reference and variant DNA sequences, DNA hidden-state prepending, or gene/chromosome context for VEP-Non-SNV. Figures 2-10 are downstream performance plots or heatmaps, not input routes or immediate model interfaces.",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_93769abd2260",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "paired reference and variant DNA sequences",
          "actual_model_visible_form": "DNA hidden state"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_8974deca3575",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "gene and chromosome context",
          "actual_model_visible_form": "tokenized text"
        }
      ],
      "routes": [
        {
          "route_id": "route_93769abd2260",
          "configuration_id": "config_9283059da9db",
          "route_label": "VEP DNA input route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "variant effect prediction (VEP-Non-SNV)",
          "source_object_verbatim": "paired reference and variant DNA sequences",
          "source_object_normalized": "paired reference and variant DNA sequences",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "DNA-specific start and padding tokens",
            "frozen Evo2-1B encoder",
            "trainable linear projection"
          ],
          "model_visible_form_verbatim": "DNA hidden state",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "the DNA hidden state is prepended to the text embeddings",
          "fusion_topology": "prefix",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "the DNA hidden state is prepended to the text embeddings",
          "section_heading": "B.1 Base Models, Tokenization, and Input Representations",
          "supporting_figure_or_table": "Table 8",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            18,
            22,
            23,
            28,
            29
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/tables/8",
            "#/texts/404",
            "#/texts/405",
            "#/texts/406",
            "#/texts/407",
            "#/texts/408",
            "#/texts/415",
            "#/texts/416",
            "#/texts/417",
            "#/texts/418",
            "#/texts/419",
            "#/texts/484",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000050::route_005"
          ],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000050::0008",
            "dense::july_update_2026-07-06__rec_000050::0078",
            "dense::july_update_2026-07-06__rec_000050::0084"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8974deca3575",
          "configuration_id": "config_9283059da9db",
          "route_label": "VEP gene/chromosome context route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "variant effect prediction (VEP-Non-SNV)",
          "source_object_verbatim": "gene and chromosome context",
          "source_object_normalized": "gene and chromosome context",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenized text",
            "text embeddings"
          ],
          "model_visible_form_verbatim": "tokenized text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "paired with the DNA sequence representation",
          "fusion_topology": "concatenation",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "given paired reference and variant DNA sequences together with gene and chromosome context",
          "section_heading": "D.1 Scaling Post-Training for Biological Non-Reasoning Tasks",
          "supporting_figure_or_table": "Table 8",
          "evidence_status": "explicit_text",
          "uncertainty": "Split from a hybrid VEP example that also includes DNA sequence inputs; the paper states the combined input rather than isolating this context alone.",
          "pages": [
            28,
            29
          ],
          "doc_item_refs": [
            "#/tables/8",
            "#/texts/484",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000050::route_005"
          ],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000050::0008"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_c17be9263882"
    },
    {
      "model_id": "model_b8eba3c26c46",
      "model_name": "Qwen3-1.7B and Qwen3-4B",
      "record_id": "july_update_2026-07-06__rec_000050",
      "collection_batch_id": "july_update_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_cccbb8d94eda",
      "paper_title": "How Post-Training Shapes Biological Reasoning Models",
      "doi": "",
      "paper_url": "",
      "route_count": 3,
      "configuration_count": 2,
      "family_counts": {
        "text_native_token_stream": 2,
        "dense_continuous_carrier": 1
      },
      "subtype_counts": {
        "plain_language_prompt_or_question": 1,
        "direct_projected_embedding": 1,
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "plain_language_prompt_or_question",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "DNA",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "prefix",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/july_update_2026_07_06_rec_000050_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/july_update_2026_07_06_rec_000050_23824e963093/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Training dynamics define distinct generalization regimes in biological reasoning models. We compare backbone choice, continued pre-training (CPT), supervised fine-tuning (SFT), and reinforcement learning (RL) across genomics, transcriptomics, and protein tasks, and evaluate each stage on biologically meaningful in-domain (ID) and out-of-domain (OOD) splits.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic overview of a biological foundation model training and evaluation pipeline.\n\nVisible panels and labels:\n- **Backbones**: lists general-purpose LLM backbones `Qwen3-1.7B`, `Qwen3-4B`, and `Gemma4`; lists biological foundation models for `DNA: Evo 2`, `RNA: TranscriptFormer`, and `Proteins: ESM3`.\n- **Training Stages**: three sequential stages connected by arrows:\n  - `Continued pre-training (CPT)`: domain-adaptive pre-training on biological corpora.\n  - `Supervised fine-tuning (SFT)`: task-specific instruction tuning with curated datasets.\n  - `Reinforcement learning (RL)`: optimize reasoning and final answers with task-aware rewards.\n- **Biological Modalities**: three vertical modality sections:\n  - `DNA`: example token/string `ATCG`; task `Pathway prediction`; source object appears as genomic alteration nodes/network; output described as predicting pathway activation from genomic alterations.\n  - `RNA`: example token/string `AUGC`; task `Drug target identification`; source object includes a `GENE` label pointing to a target symbol; output described as identifying genes as drug targets from transcriptomic profiles.\n  - `Proteins`: example amino-acid token sequence shown as boxed letters; task `Function prediction`; output label `Function`; described as predicting protein function from amino acid sequence.\n- **Evaluation**: compares:\n  - `In-domain (ID)`: same biological pathway, same diseases, same species.\n  - `Out-of-domain (OOD)`: unseen pathways, unseen diseases, unseen species.\n  - `Findings`: SFT improves in-domain reasoning; RL enhances out-of-domain performance; trade-offs across stages.\n\nModel/interface flow:\n- Backbone models feed into training stages.\n- Training stages feed into modality-specific biological tasks.\n- Biological modality outputs are evaluated under ID and OOD settings.\n\nBiological source objects:\n- DNA nucleotide sequence example.\n- RNA nucleotide sequence example.\n- Protein amino-acid sequence example.\n- Genomic alteration/network representation.\n- Transcriptomic gene/drug-target representation.\n- Protein sequence-to-function representation.",
        "page_no": 2,
        "sha256": "ec06be41498384e3d1cc6b9c2a234068341f8d5a592d11ceb6e4b139c9840bac",
        "pixel_width": 784,
        "pixel_height": 366,
        "crop_box": {
          "x": 0.435,
          "y": 0.011,
          "width": 0.13,
          "height": 0.989
        },
        "panel_label": "Biological Modalities: DNA route",
        "visible_input_object": "ATCG DNA token with pathway-prediction network graphic",
        "visible_model_interface": "DNA modality column showing input token, task label, and immediate pathway-prediction schematic",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the DNA modality column only, keeping the ATCG source token, pathway-prediction label, node-network schematic, and explanatory caption needed to ground one input route while excluding RNA, proteins, evaluation, and other irrelevant panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_b1629b0a2392",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "paired reference and variant DNA sequences",
          "actual_model_visible_form": "DNA hidden state"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_f63a340967cb",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "biology subset of FineFineWeb",
          "actual_model_visible_form": "tokenized biological free-text"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_f4b298ad02fb",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "pathway network definition and gene annotations",
          "actual_model_visible_form": "tokenized text"
        }
      ],
      "routes": [
        {
          "route_id": "route_f63a340967cb",
          "configuration_id": "config_0a6978f55289",
          "route_label": "Biology CPT text",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "biology subset of FineFineWeb",
          "source_object_verbatim": "biology subset of FineFineWeb",
          "source_object_normalized": "biology subset of FineFineWeb",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenized biological free-text",
            "causal next-token prediction"
          ],
          "model_visible_form_verbatim": "tokenized biological free-text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "language model alone",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "We mid-train the two Qwen3 backbones on the biology subset of FineFineWeb [77].",
          "section_heading": "B.2 Continued Pre-training (Mid-training) Setup",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes ordinary biological web text rather than a literal prompt or question; the closest frozen text-native leaf is used.",
          "pages": [
            18,
            22,
            32
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/tables/13",
            "#/texts/404",
            "#/texts/405",
            "#/texts/406",
            "#/texts/407",
            "#/texts/408",
            "#/texts/410",
            "#/texts/411",
            "#/texts/412",
            "#/texts/575"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000050::route_001"
          ],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000050::0052",
            "dense::july_update_2026-07-06__rec_000050::0077",
            "dense::july_update_2026-07-06__rec_000050::0087"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b1629b0a2392",
          "configuration_id": "config_751632616ab0",
          "route_label": "Pathway DNA encoder route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "pathway prediction",
          "source_object_verbatim": "paired reference and variant DNA sequences",
          "source_object_normalized": "paired reference and variant DNA sequences",
          "source_modality_normalized": "DNA",
          "transformation_chain_verbatim": [
            "DNA-specific start and padding tokens",
            "frozen Evo2-1B encoder",
            "trainable linear projection",
            "prepended to the text embeddings"
          ],
          "model_visible_form_verbatim": "DNA hidden state",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "DNA hidden state is prepended to the text embeddings",
          "fusion_topology": "prefix",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "DNA hidden state is prepended",
          "section_heading": "B.1 Base Models, Tokenization, and Input Representations",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            18,
            20,
            21,
            22,
            23
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/texts/369",
            "#/texts/370",
            "#/texts/371",
            "#/texts/372",
            "#/texts/374",
            "#/texts/375",
            "#/texts/376",
            "#/texts/377",
            "#/texts/404",
            "#/texts/405",
            "#/texts/406",
            "#/texts/407",
            "#/texts/408",
            "#/texts/415",
            "#/texts/416",
            "#/texts/417",
            "#/texts/418",
            "#/texts/419"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000050::route_002"
          ],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000050::0018",
            "dense::july_update_2026-07-06__rec_000050::0078",
            "dense::july_update_2026-07-06__rec_000050::0084"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f4b298ad02fb",
          "configuration_id": "config_751632616ab0",
          "route_label": "Pathway prompt route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "pathway prediction",
          "source_object_verbatim": "pathway network definition and gene annotations",
          "source_object_normalized": "pathway network definition and gene annotations",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenized text",
            "text embeddings"
          ],
          "model_visible_form_verbatim": "tokenized text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "followed by a pathway network definition and gene annotations",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Pathway prediction. A representative input consists of two versions of the same DNA sequence region, with and without the mutation, including DNA-specific start and padding tokens, followed by a pathway network definition and gene annotations. The model is then asked to infer the biological or disease effect associated with the allele. For example:",
          "section_heading": "A.4 Example Prompts and Inputs for Each Task",
          "supporting_figure_or_table": null,
          "evidence_status": "inferred",
          "uncertainty": "The text carrier is explicit, but tokenization/embedding is inferred from the model interface description; this route is split from a hybrid example that also includes DNA embeddings.",
          "pages": [
            20,
            21,
            22
          ],
          "doc_item_refs": [
            "#/texts/369",
            "#/texts/370",
            "#/texts/371",
            "#/texts/372",
            "#/texts/374",
            "#/texts/375",
            "#/texts/376",
            "#/texts/377",
            "#/texts/404",
            "#/texts/405",
            "#/texts/406",
            "#/texts/407",
            "#/texts/408"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000050::route_002"
          ],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000050::0018"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_db6432d4d322"
    },
    {
      "model_id": "model_a493b5670a18",
      "model_name": "Qwen3-1.7B and Qwen3-4B-Thinking",
      "record_id": "july_update_2026-07-06__rec_000050",
      "collection_batch_id": "july_update_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_cccbb8d94eda",
      "paper_title": "How Post-Training Shapes Biological Reasoning Models",
      "doi": "",
      "paper_url": "",
      "route_count": 2,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 1,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "direct_projected_embedding": 1,
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "protein/peptide",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "concatenation",
        "placeholder_replacement"
      ],
      "text_roles": [
        "instruction_or_query",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/july_update_2026_07_06_rec_000050_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/july_update_2026_07_06_rec_000050_23824e963093/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Training dynamics define distinct generalization regimes in biological reasoning models. We compare backbone choice, continued pre-training (CPT), supervised fine-tuning (SFT), and reinforcement learning (RL) across genomics, transcriptomics, and protein tasks, and evaluate each stage on biologically meaningful in-domain (ID) and out-of-domain (OOD) splits.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic overview of a biological foundation model training and evaluation pipeline.\n\nVisible panels and labels:\n- **Backbones**: lists general-purpose LLM backbones `Qwen3-1.7B`, `Qwen3-4B`, and `Gemma4`; lists biological foundation models for `DNA: Evo 2`, `RNA: TranscriptFormer`, and `Proteins: ESM3`.\n- **Training Stages**: three sequential stages connected by arrows:\n  - `Continued pre-training (CPT)`: domain-adaptive pre-training on biological corpora.\n  - `Supervised fine-tuning (SFT)`: task-specific instruction tuning with curated datasets.\n  - `Reinforcement learning (RL)`: optimize reasoning and final answers with task-aware rewards.\n- **Biological Modalities**: three vertical modality sections:\n  - `DNA`: example token/string `ATCG`; task `Pathway prediction`; source object appears as genomic alteration nodes/network; output described as predicting pathway activation from genomic alterations.\n  - `RNA`: example token/string `AUGC`; task `Drug target identification`; source object includes a `GENE` label pointing to a target symbol; output described as identifying genes as drug targets from transcriptomic profiles.\n  - `Proteins`: example amino-acid token sequence shown as boxed letters; task `Function prediction`; output label `Function`; described as predicting protein function from amino acid sequence.\n- **Evaluation**: compares:\n  - `In-domain (ID)`: same biological pathway, same diseases, same species.\n  - `Out-of-domain (OOD)`: unseen pathways, unseen diseases, unseen species.\n  - `Findings`: SFT improves in-domain reasoning; RL enhances out-of-domain performance; trade-offs across stages.\n\nModel/interface flow:\n- Backbone models feed into training stages.\n- Training stages feed into modality-specific biological tasks.\n- Biological modality outputs are evaluated under ID and OOD settings.\n\nBiological source objects:\n- DNA nucleotide sequence example.\n- RNA nucleotide sequence example.\n- Protein amino-acid sequence example.\n- Genomic alteration/network representation.\n- Transcriptomic gene/drug-target representation.\n- Protein sequence-to-function representation.",
        "page_no": 2,
        "sha256": "ec06be41498384e3d1cc6b9c2a234068341f8d5a592d11ceb6e4b139c9840bac",
        "pixel_width": 784,
        "pixel_height": 366,
        "crop_box": {
          "x": 0.655,
          "y": 0.126,
          "width": 0.12,
          "height": 0.758
        },
        "panel_label": "Biological Modalities / Proteins",
        "visible_input_object": "protein amino-acid sequence",
        "visible_model_interface": "boxed residue tokens feeding the function-prediction task",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop isolates the protein modality route: the sequence token boxes, the Proteins label, the function-prediction task, and the immediate arrow/interface needed to read the grounded input path. It excludes the backbone, training-stages, and evaluation panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_1deed07c566b",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "protein sequence",
          "actual_model_visible_form": "per-residue embeddings"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_526a80e3bb7b",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "organism metadata, InterPro domain annotations, protein-protein interaction context, and initial GO term speculations",
          "actual_model_visible_form": "text tokens"
        }
      ],
      "routes": [
        {
          "route_id": "route_1deed07c566b",
          "configuration_id": "config_04b947e1ebe2",
          "route_label": "Protein embedding route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "protein function prediction",
          "source_object_verbatim": "protein sequence",
          "source_object_normalized": "protein sequence",
          "source_modality_normalized": "protein/peptide",
          "transformation_chain_verbatim": [
            "frozen ESM-3 small protein encoder",
            "projected through a trainable linear layer"
          ],
          "model_visible_form_verbatim": "per-residue embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "inserted at the protein placeholder positions before the text tokens",
          "fusion_topology": "placeholder_replacement",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "For SFT and GRPO we use the same BioReason-Pro-style protein-conditioned interface, except that we omit the GO-graph encoder [3]. The text backbone is paired with a frozen ESM-3 small protein encoder [52]. Per-residue embeddings are extracted from layer 37 of ESM-3, projected through a trainable linear layer into the text-embedding space, and inserted at the protein placeholder positions before the text tokens [52]. The protein encoder is kept frozen; the protein projection layer and the LoRA adapter on the text model receive gradients [93].",
          "section_heading": "B.1 Base Models, Tokenization, and Input Representations",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            18,
            20,
            22,
            23,
            24,
            25,
            26,
            27
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/texts/365",
            "#/texts/366",
            "#/texts/367",
            "#/texts/404",
            "#/texts/405",
            "#/texts/406",
            "#/texts/407",
            "#/texts/408",
            "#/texts/415",
            "#/texts/416",
            "#/texts/417",
            "#/texts/418",
            "#/texts/419",
            "#/texts/425",
            "#/texts/426",
            "#/texts/427",
            "#/texts/428",
            "#/texts/429",
            "#/texts/430",
            "#/texts/431",
            "#/texts/432"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000050::route_004"
          ],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000050::0016",
            "dense::july_update_2026-07-06__rec_000050::0023",
            "dense::july_update_2026-07-06__rec_000050::0078",
            "dense::july_update_2026-07-06__rec_000050::0086"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_526a80e3bb7b",
          "configuration_id": "config_04b947e1ebe2",
          "route_label": "Protein metadata prompt route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "protein function prediction",
          "source_object_verbatim": "organism metadata, InterPro domain annotations, protein-protein interaction context, and initial GO term speculations",
          "source_object_normalized": "protein metadata and prompt scaffold",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "concatenate system instruction, header, and user instruction",
            "text tokens"
          ],
          "model_visible_form_verbatim": "text tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "concatenate system instruction, header, and user instruction",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Protein experiments. The text backbones for the protein experiments are Qwen3-1.7B and Qwen34B-Thinking, loaded with their native BPE tokenizer [75]. The padding token is aliased to the end-of-sequence token. Both the SFT and RL prompts concatenate (i) a system instruction describing the task and available biological context, (ii) a header containing the protein name, organism, and amino-acid sequence, and (iii) a user instruction asking the model to emit GO identifiers across the molecular function, biological process, and cellular component aspects [89].",
          "section_heading": "B.1 Base Models, Tokenization, and Input Representations",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "Split from a hybrid example that also includes projected ESM-3 protein embeddings; the paper describes the protein text context and dense carrier together.",
          "pages": [
            18,
            20,
            22,
            23,
            24,
            25,
            26,
            27
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/texts/365",
            "#/texts/366",
            "#/texts/367",
            "#/texts/404",
            "#/texts/405",
            "#/texts/406",
            "#/texts/407",
            "#/texts/408",
            "#/texts/415",
            "#/texts/416",
            "#/texts/417",
            "#/texts/418",
            "#/texts/419",
            "#/texts/425",
            "#/texts/426",
            "#/texts/427",
            "#/texts/428",
            "#/texts/429",
            "#/texts/430",
            "#/texts/431",
            "#/texts/432"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000050::route_004"
          ],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000050::0016",
            "dense::july_update_2026-07-06__rec_000050::0023",
            "dense::july_update_2026-07-06__rec_000050::0078",
            "dense::july_update_2026-07-06__rec_000050::0086"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_c17be9263882"
    },
    {
      "model_id": "model_ab432decc9aa",
      "model_name": "Qwen3-1.7B, Qwen3-4B, and Gemma 4 E2B",
      "record_id": "july_update_2026-07-06__rec_000050",
      "collection_batch_id": "july_update_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_cccbb8d94eda",
      "paper_title": "How Post-Training Shapes Biological Reasoning Models",
      "doi": "",
      "paper_url": "",
      "route_count": 2,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 1,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "direct_projected_embedding": 1,
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "RNA",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "concatenation",
        "prefix"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "The contact sheet only shows a high-level training-and-evaluation overview plus downstream performance plots. None of the figures visibly shows the exact RNA embedding-to-text interface, projected RNA hidden states, or prompt-token fusion required for the named route, so no figure can be used responsibly.",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_9f035bc3c1c6",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "TranscriptFormer representations for the corresponding normal and disease states",
          "actual_model_visible_form": "projected RNA hidden states"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_e1bb17396093",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "natural-language disease and cell-type context, five candidate genes",
          "actual_model_visible_form": "tokenized natural-language prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_9f035bc3c1c6",
          "configuration_id": "config_c5a701c45f25",
          "route_label": "RNA transcriptomic embedding route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "target identification",
          "source_object_verbatim": "TranscriptFormer representations for the corresponding normal and disease states",
          "source_object_normalized": "TranscriptFormer representations for corresponding normal and disease states",
          "source_modality_normalized": "RNA",
          "transformation_chain_verbatim": [
            "frozen TranscriptFormer encoder",
            "trainable linear projection",
            "prepended to the text-token embeddings before the prompt tokens"
          ],
          "model_visible_form_verbatim": "projected RNA hidden states",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "The projected RNA hidden states are prepended to the text-token embeddings before the prompt tokens",
          "fusion_topology": "prefix",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "The projected RNA hidden states are prepended to the text-token embeddings",
          "section_heading": "B.1 Base Models, Tokenization, and Input Representations",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            2,
            3,
            4,
            8,
            9,
            10,
            15,
            18,
            19,
            20,
            21,
            22,
            23,
            30,
            31,
            32
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/tables/0",
            "#/tables/12",
            "#/texts/25",
            "#/texts/26",
            "#/texts/28",
            "#/texts/314",
            "#/texts/315",
            "#/texts/316",
            "#/texts/317",
            "#/texts/360",
            "#/texts/361",
            "#/texts/362",
            "#/texts/404",
            "#/texts/405",
            "#/texts/406",
            "#/texts/407",
            "#/texts/408",
            "#/texts/415",
            "#/texts/416",
            "#/texts/417",
            "#/texts/418",
            "#/texts/419",
            "#/texts/42",
            "#/texts/44",
            "#/texts/45",
            "#/texts/46",
            "#/texts/568",
            "#/texts/570",
            "#/texts/571",
            "#/texts/572"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000050::route_003"
          ],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000050::0015",
            "dense::july_update_2026-07-06__rec_000050::0019",
            "dense::july_update_2026-07-06__rec_000050::0046",
            "dense::july_update_2026-07-06__rec_000050::0048",
            "dense::july_update_2026-07-06__rec_000050::0078",
            "dense::july_update_2026-07-06__rec_000050::0085"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e1bb17396093",
          "configuration_id": "config_c5a701c45f25",
          "route_label": "RNA task prompt route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "target identification",
          "source_object_verbatim": "natural-language disease and cell-type context, five candidate genes",
          "source_object_normalized": "disease and cell-type context with five candidate genes",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenized natural-language prompt",
            "text-token embeddings"
          ],
          "model_visible_form_verbatim": "tokenized natural-language prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "jointly conditions on transcriptomic representations and the textual task description",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "For the downstream SFT and RL stages, we couple the text backbone to a frozen TranscriptFormer encoder through a trainable linear projection [45]. Each target-identification example contains a natural-language disease and cell-type context, a five-gene candidate set, and TranscriptFormer representations for the candidate genes in normal and disease states [45, 73]. The projected RNA hidden states are prepended to the text-token embeddings before the prompt tokens, so that the language model conditions jointly on transcriptomic representations and the textual task description.",
          "section_heading": "B.1 Base Models, Tokenization, and Input Representations",
          "supporting_figure_or_table": null,
          "evidence_status": "inferred",
          "uncertainty": "Split from a hybrid example that also includes projected TranscriptFormer embeddings; the paper presents the text context and dense RNA carrier together.",
          "pages": [
            1,
            2,
            3,
            4,
            8,
            9,
            10,
            15,
            18,
            19,
            20,
            21,
            22,
            30,
            31,
            32
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/tables/12",
            "#/texts/25",
            "#/texts/26",
            "#/texts/28",
            "#/texts/314",
            "#/texts/315",
            "#/texts/316",
            "#/texts/317",
            "#/texts/360",
            "#/texts/361",
            "#/texts/362",
            "#/texts/404",
            "#/texts/405",
            "#/texts/406",
            "#/texts/407",
            "#/texts/408",
            "#/texts/42",
            "#/texts/44",
            "#/texts/45",
            "#/texts/46",
            "#/texts/568",
            "#/texts/570",
            "#/texts/571",
            "#/texts/572"
          ],
          "source_candidate_refs": [
            "july_update_2026-07-06__rec_000050::route_003"
          ],
          "dense_candidate_refs": [
            "dense::july_update_2026-07-06__rec_000050::0015",
            "dense::july_update_2026-07-06__rec_000050::0019",
            "dense::july_update_2026-07-06__rec_000050::0046",
            "dense::july_update_2026-07-06__rec_000050::0048",
            "dense::july_update_2026-07-06__rec_000050::0085"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_c17be9263882"
    },
    {
      "model_id": "model_4ba985125332",
      "model_name": "replaceable LLM backend",
      "record_id": "june_update_2026-06-10__rec_000121",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_9cc3c78888c3",
      "paper_title": "RVQ-Alpha: Bridging Single-Cell Transcriptomics and Large Language Models via Discrete Tokenization and Verifiable Reinforcement Learning",
      "doi": "10.64898/2026.04.20.719773",
      "paper_url": "https://doi.org/10.64898/2026.04.20.719773",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell transcriptomics"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "concatenation"
      ],
      "text_roles": [
        "metadata_or_context"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "Neither blind selection is supported by the contact sheet. Figure 1 is a high-level RVQ-Alpha overview with tokenization and downstream tasks, and Figure 2 shows BPE/RVQ tokenization and LLM fusion, but neither visibly shows the teacher-route interface of curated gene features received alongside c and t in a teacher prompt. The remaining figures are performance plots or generic schematics, not the specific input route.",
      "illustrative_examples": [
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_d6173536f567",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "curated gene features",
          "actual_model_visible_form": "curated gene features"
        }
      ],
      "routes": [
        {
          "route_id": "route_d6173536f567",
          "configuration_id": "config_c86edc520e6b",
          "route_label": "teacher gene-feature scaffolding route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "constraint-conditioned rationale synthesis",
          "source_object_verbatim": "curated gene features",
          "source_object_normalized": "curated gene features",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "curated gene features",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "received alongside c and t in the teacher prompt",
          "fusion_topology": "concatenation",
          "text_role": "metadata_or_context",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Athird constraint compounds the first two: cell tokens must enter the LLM through its native token embedding table rather than through a separately learned cross-modal adapter, which would relegate gene expression to a secondary modality and add a trainable bottleneck between cell identity and the reasoning path.",
          "section_heading": "3.3 scCoT-Synth: Evidence-Grounded Supervision Engine for Cold-Start RVQ Alignment",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "Teacher-side scaffolding rather than the final student path; the backend is described as replaceable rather than named.",
          "pages": [
            3,
            4,
            7,
            8,
            9
          ],
          "doc_item_refs": [
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/127",
            "#/texts/128",
            "#/texts/130",
            "#/texts/131",
            "#/texts/132",
            "#/texts/61",
            "#/texts/63",
            "#/texts/64"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000121::route_001"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000121::0101"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_1b67d9407546"
    },
    {
      "model_id": "model_7ccbcc7f850b",
      "model_name": "RVQ-Alpha",
      "record_id": "june_update_2026-06-10__rec_000121",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_9cc3c78888c3",
      "paper_title": "RVQ-Alpha: Bridging Single-Cell Transcriptomics and Large Language Models via Discrete Tokenization and Verifiable Reinforcement Learning",
      "doi": "10.64898/2026.04.20.719773",
      "paper_url": "https://doi.org/10.64898/2026.04.20.719773",
      "route_count": 6,
      "configuration_count": 5,
      "family_counts": {
        "text_native_token_stream": 3,
        "discrete_biological_symbol_stream": 3
      },
      "subtype_counts": {
        "structured_biological_prompt_or_task_scaffold": 2,
        "learned_quantized_id_or_codebook_token": 2,
        "multi_track_structural_symbol_stream": 1,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "learned_quantized_id_or_codebook_token",
        "multi_track_structural_symbol_stream",
        "plain_language_prompt_or_question",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "single-cell transcriptomics",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "metadata_or_context"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/june_update_2026_06_10_rec_000121_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/june_update_2026_06_10_rec_000121_8f4709456aba/figure_002.png",
        "figure_index": 2,
        "caption": "Figure 2: Core dual-tokenization architecture. Text instructions are processed by a BPE tokenizer (top), while gene expression data passes through the RVQ tokenizer (bottom). Both token streams are fused at the LLM layer ( p ( t | h , c ) ), enabling bidirectional encoding and decoding: the BPE and RVQ detokenizers reconstruct the original modalities from the unified representation.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow figure for tokenizing and modeling single-cell data with LLM-style discrete tokens.\n\nVisible structure:\n- Two parallel input streams are shown on the left:\n  - **Meta data**, represented as small document-like panels.\n  - **Expression data**, represented as multiple single-cell study plots labeled with studies such as “Study 1,” “Study 2,” “Study 3,” and “Study-N.”\n- The metadata stream passes through a blue block labeled **“BPE Tokenize”**, producing a **Words Vocabulary** example with tokens such as “Nanocells,” “are,” “ultra-small,” “cells,” “~10nm,” and “diameter.”\n- The expression data stream passes through an orange block labeled **“RVQ Tokenize”**, producing a colored grid labeled **“Single Cell codebook.”**\n- Both streams are converted into vertical sequences labeled as discrete tokens. The lower caption reads **“Single Cell Discrete Tokens.”**\n- A central model block labeled **“LLMs”** receives token sequences and shows a next-token prediction style formula/interface, indicating language-model processing over mixed metadata and expression-derived tokens.\n- On the right, generated or decoded token sequences pass through:\n  - **“BPE Detokenize”** for metadata, producing text output: “Nanocells are ultra-small cells.”\n  - **“RVQ Detokenize”** for expression data, reconstructing study-like expression data panels.\n\nBiological source objects:\n- Single-cell expression data from multiple studies.\n- Associated metadata text describing biological/cellular concepts.\n\nTransformations:\n- Text metadata is transformed into BPE tokens and later detokenized back into text.\n- Single-cell expression data is transformed into RVQ discrete tokens via a single-cell codebook and later detokenized back into expression data.\n\nModel interface:\n- A central LLM operates on the combined discrete token sequences from metadata and expression data.\n\nVisible finding/claim:\n- The figure illustrates a multimodal tokenization framework where both textual metadata and single-cell expression data are represented as discrete tokens that can be modeled by LLMs and decoded back into their original modalities.",
        "page_no": 9,
        "sha256": "a2d6d0213d177e3a932e1bef4b0a0239a8463ef9f1ccac53fc689bfc87a01262",
        "pixel_width": 876,
        "pixel_height": 341,
        "crop_box": {
          "x": 0.0,
          "y": 0.02,
          "width": 0.67,
          "height": 0.6
        },
        "panel_label": "top-left BPE-to-LLM input route",
        "visible_input_object": "Meta data",
        "visible_model_interface": "LLMs layer receiving BPE tokens from the text stream",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "Crops the metadata source, BPE tokenizer, word-vocabulary example, token stream, and LLM fusion/interface while excluding the right-side detokenized output panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "learned_quantized_id_or_codebook_token",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_3ce30344d54f",
          "example_input": "continuous biology → quantizer",
          "example_carrier": "[BIO_187] [BIO_042] [BIO_913]",
          "example_interface": "VQ/RVQ codebook IDs → generator",
          "actual_source": "gene expression data",
          "actual_model_visible_form": "Discrete Tokens"
        },
        {
          "subtype_id": "multi_track_structural_symbol_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_2dae09106751",
          "example_input": "AA: M K T ...   SS: H H C ...",
          "example_carrier": "aligned sequence + structure tracks",
          "example_interface": "multi-track tokenizer → generator",
          "actual_source": "a cell population P = { x1 , . . . , xN }",
          "actual_model_visible_form": "multiple RVQ-encoded cell sequences"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_c143fc62faab",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "text descriptions and biomedical literature",
          "actual_model_visible_form": "BPE tokens"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_6b322de9c08a",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "metadata and text instructions",
          "actual_model_visible_form": "BPE tokens"
        }
      ],
      "routes": [
        {
          "route_id": "route_6b322de9c08a",
          "configuration_id": "config_ec5b6d1b8e51",
          "route_label": "instruction-response BPE route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "instruction-response format",
          "source_object_verbatim": "metadata and text instructions",
          "source_object_normalized": "metadata and instruction text",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "BPE tokenizer"
          ],
          "model_visible_form_verbatim": "BPE tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fused at the LLM layer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Figure 2: Core dual-tokenization architecture. Text instructions are processed by a BPE tokenizer (top), while gene expression data passes through the RVQ tokenizer (bottom). Both token streams are fused at the LLM layer ( p ( t | h , c ) ), enabling bidirectional encoding and decoding: the BPE and RVQ detokenizers reconstruct the original modalities from the unified representation.",
          "section_heading": "Figure 2: Core dual-tokenization architecture",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            8,
            9,
            57,
            58
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/pictures/20",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/148",
            "#/texts/149",
            "#/texts/151",
            "#/texts/152",
            "#/texts/153",
            "#/texts/2101",
            "#/texts/281",
            "#/texts/282",
            "#/texts/283",
            "#/texts/284"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000121::route_002"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000121::0107"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3ce30344d54f",
          "configuration_id": "config_beac415936bb",
          "route_label": "single-cell RVQ route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "single-cell tasks",
          "source_object_verbatim": "gene expression data",
          "source_object_normalized": "gene expression data",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "RVQ Encoder",
            "RVQ tokenizer"
          ],
          "model_visible_form_verbatim": "Discrete Tokens",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "learned_quantized_id_or_codebook_token",
          "insertion_or_fusion_verbatim": "embedded directly into the LLM vocabulary",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "gene expression data passes through the RVQ tokenizer",
          "section_heading": "Figure 2: Core dual-tokenization architecture",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            2,
            3,
            4,
            7,
            8,
            9,
            37,
            38,
            48,
            49
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/tables/13",
            "#/texts/1013",
            "#/texts/1014",
            "#/texts/1015",
            "#/texts/1018",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/127",
            "#/texts/128",
            "#/texts/130",
            "#/texts/131",
            "#/texts/132",
            "#/texts/145",
            "#/texts/146",
            "#/texts/147",
            "#/texts/148",
            "#/texts/149",
            "#/texts/151",
            "#/texts/152",
            "#/texts/153",
            "#/texts/1667",
            "#/texts/1668",
            "#/texts/1669",
            "#/texts/1670",
            "#/texts/1671",
            "#/texts/1672",
            "#/texts/1673",
            "#/texts/1675",
            "#/texts/1677",
            "#/texts/1678",
            "#/texts/1679",
            "#/texts/1680",
            "#/texts/21",
            "#/texts/25",
            "#/texts/281",
            "#/texts/282",
            "#/texts/283",
            "#/texts/65",
            "#/texts/66",
            "#/texts/67"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000121::route_003"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000121::0007",
            "dense::june_update_2026-06-10__rec_000121::0100",
            "dense::june_update_2026-06-10__rec_000121::0105",
            "dense::june_update_2026-06-10__rec_000121::0106",
            "dense::june_update_2026-06-10__rec_000121::0070",
            "dense::june_update_2026-06-10__rec_000121::0074"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2dae09106751",
          "configuration_id": "config_b7714aaf8697",
          "route_label": "population RVQ route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Population-level tasks",
          "source_object_verbatim": "a cell population P = { x1 , . . . , xN }",
          "source_object_normalized": "cell population",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "RVQ tokenizer",
            "per-cell and population-level delimiters"
          ],
          "model_visible_form_verbatim": "multiple RVQ-encoded cell sequences",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "multi_track_structural_symbol_stream",
          "insertion_or_fusion_verbatim": "fused at the shared embedding layer",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "RVQ-Alpha: Bridging Single-Cell Transcriptomics and Large Language Models via Discrete Tokenization and Verifiable Reinforcement Learning",
          "section_heading": "3.2 Residual Quantization for LLM-Native Cell Tokens",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            9,
            10,
            11
          ],
          "doc_item_refs": [
            "#/texts/286",
            "#/texts/287",
            "#/texts/288",
            "#/texts/289",
            "#/texts/291",
            "#/texts/292",
            "#/texts/293",
            "#/texts/294"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000121::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c143fc62faab",
          "configuration_id": "config_092110ae9b75",
          "route_label": "continued-pretraining text-description route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "RVQ-text paired descriptions spanning diverse linguistic registers",
          "source_object_verbatim": "text descriptions and biomedical literature",
          "source_object_normalized": "text descriptions and biomedical literature",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "text tokenization"
          ],
          "model_visible_form_verbatim": "BPE tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "fused at the LLM layer",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "The training data consists of RVQ-text paired descriptions spanning diverse linguistic registers (nine template families; §3.3), which are interleaved with biomedical literature and explicitly cover the top 500 gene names.",
          "section_heading": "3.4 Multi-Stage Training Pipeline",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "This route captures the text side of the paired continued-pretraining corpus.",
          "pages": [
            12
          ],
          "doc_item_refs": [
            "#/texts/334",
            "#/texts/335",
            "#/texts/336",
            "#/texts/337",
            "#/texts/338",
            "#/texts/339"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000121::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8216d074f753",
          "configuration_id": "config_e2b3062a0099",
          "route_label": "student RVQ-code input route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "student training",
          "source_object_verbatim": "RVQ codes c",
          "source_object_normalized": "RVQ codes",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "RVQ tokenizer"
          ],
          "model_visible_form_verbatim": "RVQ code sequence c",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "learned_quantized_id_or_codebook_token",
          "insertion_or_fusion_verbatim": "the input to the student contains only c ⊕ t",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "the input to the student contains only c ⊕ t",
          "section_heading": "3.3 scCoT-Synth: Evidence-Grounded Supervision Engine for Cold-Start RVQ Alignment",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": "Split from the combined student-input candidate; the paper states the joint c ⊕ t input.",
          "pages": [
            9,
            10,
            11,
            12
          ],
          "doc_item_refs": [
            "#/texts/286",
            "#/texts/287",
            "#/texts/288",
            "#/texts/289",
            "#/texts/291",
            "#/texts/292",
            "#/texts/293",
            "#/texts/294",
            "#/texts/334",
            "#/texts/335",
            "#/texts/336",
            "#/texts/337",
            "#/texts/338",
            "#/texts/339"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000121::route_006"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000121::0008"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a9694890c7bc",
          "configuration_id": "config_e2b3062a0099",
          "route_label": "student task-instruction input route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "student training",
          "source_object_verbatim": "task instruction t",
          "source_object_normalized": "task instruction",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "BPE tokenizer"
          ],
          "model_visible_form_verbatim": "task instruction text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "the input to the student contains only c ⊕ t",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "the input to the student contains only c ⊕ t",
          "section_heading": "3.3 scCoT-Synth: Evidence-Grounded Supervision Engine for Cold-Start RVQ Alignment",
          "supporting_figure_or_table": "Table 2",
          "evidence_status": "explicit_text",
          "uncertainty": "Split from the combined student-input candidate; the paper states the joint c ⊕ t input.",
          "pages": [
            9,
            10,
            11,
            12
          ],
          "doc_item_refs": [
            "#/texts/286",
            "#/texts/287",
            "#/texts/288",
            "#/texts/289",
            "#/texts/291",
            "#/texts/292",
            "#/texts/293",
            "#/texts/294",
            "#/texts/334",
            "#/texts/335",
            "#/texts/336",
            "#/texts/337",
            "#/texts/338",
            "#/texts/339"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000121::route_006"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000121::0008"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_fbfad76cabfd"
    },
    {
      "model_id": "model_2a4d0a37a0ad",
      "model_name": "scDiff",
      "record_id": "full_2026-07-06__rec_001343",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_7bc6678864bf",
      "paper_title": "A General Single-Cell Analysis Framework via Conditional Diffusion Generative Models",
      "doi": "10.1101/2023.10.13.562243",
      "paper_url": "https://doi.org/10.1101/2023.10.13.562243",
      "route_count": 8,
      "configuration_count": 8,
      "family_counts": {
        "dense_continuous_carrier": 8
      },
      "subtype_counts": {
        "direct_projected_embedding": 6,
        "connector_mediated_embedding": 2
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "connector_mediated_embedding",
        "direct_projected_embedding"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "batch metadata",
        "condition metadata",
        "gene expression",
        "gene ontology graph",
        "source modality expression measures",
        "spatial transcriptomics",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "pretraining",
        "unclear"
      ],
      "fusion_topologies": [
        "concatenation",
        "cross_attention",
        "encoder_decoder",
        "side_or_generative_conditioning"
      ],
      "text_roles": [
        "metadata_or_context",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001343_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001343_6968c1123911/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: An Overview of scDiff .",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic model architecture diagram for a diffusion-style generative framework.\n\nVisible elements:\n- A top diffusion trajectory shows variables progressing from `x_T` through `x_t`, `x_{t-1}`, to `x_0`.\n- The main boxed architecture takes `x_t` as input and outputs `x_{t-1}`.\n- Left side: an embedder labeled `φ` processes `x_t`.\n- Conditioning inputs are shown as biological/context sources:\n  - “cell types” illustrated with colored cell-like clusters.\n  - “text” represented by a document icon.\n  - “perturbation” represented by a small signal/trace plot.\n  - An ellipsis indicates additional possible conditioning inputs.\n- These inputs feed into a conditioner labeled `ψ`.\n- The central encoder labeled `𝓔` contains cross-attention blocks, each marked with `Q`, `K`, `V`.\n- Arrows indicate cross-attention connections from the conditioner output into encoder blocks.\n- A standalone legend-like cross-attention block labeled `Q K V` appears below the encoder.\n- Right side: a decoder labeled `𝓓` produces `x_θ`, which is combined with an epsilon/noise term `ϵ` to produce `x_{t-1}`.\n- A legend defines:\n  - `φ`: embedder\n  - `ψ`: conditioner\n  - `𝓔`: encoder\n  - `𝓓`: decoder\n\nNo experimental results or quantitative findings are visible; the figure communicates model components and data/model interfaces.",
        "page_no": 4,
        "sha256": "189e63bbead49f09efe6bf1f2f9413b9cdfda31b2e8a41fc24ecb424297a2026",
        "pixel_width": 774,
        "pixel_height": 365,
        "crop_box": {
          "x": 0,
          "y": 0.2,
          "width": 0.52,
          "height": 0.8
        },
        "panel_label": "left input-conditioning and encoder interface",
        "visible_input_object": "noised gene expression x_t, cell types, text, perturbation",
        "visible_model_interface": "input embedder φ feeding conditioner ψ into the first cross-attention QKV block of encoder 𝓔",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop preserves the grounded input route from x_t through φ, the visible conditioning sources, and the immediate ψ→QKV fusion interface. It excludes the decoder/output side and the top diffusion trajectory, which are not needed to understand the input path.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "connector_mediated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_b441718bb9ae",
          "example_input": "image / omics encoder states",
          "example_carrier": "Q-Former or adapter query vectors",
          "example_interface": "connector → LLM cross-modal interface",
          "actual_source": "cell type definitions from the cell ontology",
          "actual_model_visible_form": "class token embeddings"
        },
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_341400d84a00",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "noised gene expression x_t",
          "actual_model_visible_form": "d-dimensional input embedding"
        }
      ],
      "routes": [
        {
          "route_id": "route_341400d84a00",
          "configuration_id": "config_7864e4c05d65",
          "route_label": "corrupted gene expression input",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cell labeling, expression completion and knowledge transfer",
          "source_object_verbatim": "noised gene expression x_t",
          "source_object_normalized": "corrupted gene expression x_t",
          "source_modality_normalized": "gene expression",
          "transformation_chain_verbatim": [
            "linear mapping W",
            "sinusoidal time embedding"
          ],
          "model_visible_form_verbatim": "d-dimensional input embedding",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "input expression embedder φ",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Formally, we denote the expression of all cells as X ∈ R n × m , where n is the number of cells and m stands for the number of genes. We categorize the common tasks into three classes: cell labeling, expression completion and knowledge transfer. The notations for task-specific conditions are detailed subsequently.",
          "section_heading": "Embedder.",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/texts/21",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/29",
            "#/texts/30"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001343::route_001"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_33c428778637",
          "configuration_id": "config_c09a1eee8b62",
          "route_label": "masked expression context",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "expression completion",
          "source_object_verbatim": "randomly masked expression ˜x",
          "source_object_normalized": "randomly masked expression",
          "source_modality_normalized": "gene expression",
          "transformation_chain_verbatim": [
            "linear projection W_ctxt"
          ],
          "model_visible_form_verbatim": "projected masked-expression vector",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "context conditioner ψ_f^ctxt",
          "fusion_topology": "cross_attention",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "Expression completion. A crucial category of single-cell analysis tasks is expression completion. It includes both filling in missing values or predicting the whole expression. In some scenarios, the task may require external information from reference datasets. To account for the majority of the settings, we denote the observed expression as M ⊙ X , where ⊙ denotes matrix element-wise multiplication and M ∈ { 0 , 1 } n × m is the element-wise indicator with ones be observed and zeros be missing. The task is defined to estimate the posterior p (( J n,m -M ) ⊙ X | M ⊙ X ) , where J n,m is an all-ones matrix of dimensions n × m . Equivalently, we write the objective as",
          "section_heading": "Context.",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3
          ],
          "doc_item_refs": [
            "#/texts/21",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/25",
            "#/texts/26",
            "#/texts/29",
            "#/texts/30"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001343::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_60811d2ca646",
          "configuration_id": "config_3b6517c35c69",
          "route_label": "class-conditioning embeddings",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cell type or the perturbation state",
          "source_object_verbatim": "each class",
          "source_object_normalized": "class label",
          "source_modality_normalized": "condition metadata",
          "transformation_chain_verbatim": [
            "learnable d dimensional embeddings"
          ],
          "model_visible_form_verbatim": "class embeddings h_c",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "class conditioner",
          "fusion_topology": "cross_attention",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Next we introduce our scDiff model architecture, which is depicted in Fig 1. From a high level, scDiff aims to recover the clean single cell gene expression x 0 given the corrupted signal x t with added Gaussian noise to time step t . The associated conditions of the cell are also fed into the model to provide conditional information. Specifically, scDiff follows a general encoder-decoder design and consists of four main components: (1) input expression embedder ϕ ; (2) various conditioners , ψ ∗ , where each converts a specific condition of the input cell into a sequence of dense numerical vectors; (3) a cross-attention encoder , E , which combines the input embeddings with the corresponding conditioners and transforms them into the hidden representation of the input cell; and (4) a linear decoder , D , that projects the hidden representation back to the gene expression space to recover the input cell's noise free expression. We detail each component in the followings.",
          "section_heading": "Class.",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/100",
            "#/texts/103",
            "#/texts/104",
            "#/texts/105",
            "#/texts/106",
            "#/texts/97",
            "#/texts/98",
            "#/texts/99"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001343::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b441718bb9ae",
          "configuration_id": "config_295e12629e55",
          "route_label": "LLM cell-type descriptions",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "one-shot cell type annotation",
          "source_object_verbatim": "cell type definitions from the cell ontology",
          "source_object_normalized": "cell ontology cell type definitions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "BioLinkBERT"
          ],
          "model_visible_form_verbatim": "class token embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "LLM conditioner",
          "fusion_topology": "cross_attention",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "While some rare cell types play a crucial role in particular researches (Khalilia et al., 2011), accurately annotating them is incredibly challenging because of the limited availability of labeled samples (Jindal et al., 2018). Under the few-shot setting, prior information on cell types would significantly enhance the model. The cell ontology provides a comprehensive vocabulary and definitions of different cell types written in natural language, which can be readily encoded by LLMs into embeddings. We use BioLinkBERT (Yasunaga et al., 2022) as the backbone LLM since it is specifically trained on the biomedical corpuses. We extracted textual descriptions of all cell types from the cell ontology terms that appeared in our datasets except for the mucus-secreting cell (CL:0000319) and pulmonary artery endothelial cell (CL:1001568). For these two terms, we utilize GPT4 (OpenAI, 2023) to depict their descriptions given available definitions as query contexts.",
          "section_heading": "LLM.",
          "supporting_figure_or_table": "Figure 2; Figure 5; Figure 6; Figure 7",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            9,
            17,
            18
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/143",
            "#/texts/146",
            "#/texts/179",
            "#/texts/181",
            "#/texts/182",
            "#/texts/185",
            "#/texts/362"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001343::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_db014e40a27b",
          "configuration_id": "config_6207fb630733",
          "route_label": "gene-perturbation graph embeddings",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "zero-shot gene perturbation prediction",
          "source_object_verbatim": "gene perturbation information",
          "source_object_normalized": "gene perturbation information",
          "source_modality_normalized": "gene ontology graph",
          "transformation_chain_verbatim": [
            "gene similarity graph G",
            "simple graph convolution (SGC)"
          ],
          "model_visible_form_verbatim": "gene perturbation embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "GEARS conditioner",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "In Section 4.1, we evaluated scDiff in three representative tasks where all individual conditions are observed. In practice, we may encounter extreme cases when only a few or even no labeled samples are available for the query conditions. Given only the internal information, the few-shot or zero-shot settings are challenging, if not intractable. Here, we showcase ways to extend scDiff by incorporating prior information as external conditions to enable the handling of unseen conditions. To test the performance of scDiff under these settings, we conduct experiments with one-shot cell type annotation and zero-shot gene perturbation prediction.",
          "section_heading": "GEARS.",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            9,
            19
          ],
          "doc_item_refs": [
            "#/texts/179",
            "#/texts/187",
            "#/texts/188",
            "#/texts/189",
            "#/texts/448",
            "#/texts/449"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001343::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001343::0009"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_7a11d879f520",
          "configuration_id": "config_944b120e0e26",
          "route_label": "batch-label decoder mix-in",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cell expression recovery",
          "source_object_verbatim": "batch label",
          "source_object_normalized": "batch label",
          "source_modality_normalized": "batch metadata",
          "transformation_chain_verbatim": [
            "learnable batch embedding"
          ],
          "model_visible_form_verbatim": "batch embedding",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "decoder latent embedding mix-in",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Decoder. Finally, the cell latent embedding is linearly projected back to the gene expression space to recover the clean expression signals x 0 . We follow Lopez et al. (2018) and mix in an additional learnable batch embedding with the latent embedding according to the batch label of the input cell. This approach can better disentangle the non-biological variations in the data.",
          "section_heading": "Decoder.",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5
          ],
          "doc_item_refs": [
            "#/texts/109",
            "#/texts/110",
            "#/texts/111",
            "#/texts/112",
            "#/texts/113",
            "#/texts/114",
            "#/texts/115",
            "#/texts/116"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001343::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c88dc1223aa2",
          "configuration_id": "config_7ca2f221b08f",
          "route_label": "cell type deconvolution",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "estimate the proportions of different cell types within mixed-cell spatial transcriptomics data",
          "source_object_verbatim": "mixed-cell spatial transcriptomics data",
          "source_object_normalized": "mixed-cell spatial transcriptomics data",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "use annotated scRNA-seq data as a reference",
            "estimate the proportions of different cell types"
          ],
          "model_visible_form_verbatim": "mixed-cell spatial transcriptomics data",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "This task typically requires some external reference, in most cases using annotated scRNA-seq data as a reference",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "estimate the proportions of different cell types",
          "section_heading": "D FURTHER DETAILS OF SINGLE-CELL TASKS",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The appendix defines the task and reference source, but the exact scDiff fusion operator is not spelled out.",
          "pages": [
            16
          ],
          "doc_item_refs": [
            "#/texts/345"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001343::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fe676aa8da45",
          "configuration_id": "config_5419b3118b6a",
          "route_label": "modality prediction",
          "lifecycle_phase": "unclear",
          "task_or_configuration_verbatim": "estimate expression levels of target modality from the input modality",
          "source_object_verbatim": "input modality",
          "source_object_normalized": "source modality",
          "source_modality_normalized": "source modality expression measures",
          "transformation_chain_verbatim": [
            "concatenation of measures of source modality X source and target modality X target",
            "estimate expression levels of target modality from the input modality"
          ],
          "model_visible_form_verbatim": "X = [ X source , X target ]",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "let X = [ X source , X target ] be the concatenation of measures of source modality X source and target modality X target",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "estimate expression levels of target modality",
          "section_heading": "D FURTHER DETAILS OF SINGLE-CELL TASKS",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The matrix notation is explicit, but the exact model-visible carrier is inferred from the continuous expression representation.",
          "pages": [
            16
          ],
          "doc_item_refs": [
            "#/texts/346"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001343::0011"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_20ed14556e30"
    },
    {
      "model_id": "model_58c5ce11b66a",
      "model_name": "scGPT",
      "record_id": "full_2026-07-06__rec_001381",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_d0e026b8e9c9",
      "paper_title": "Language-Enhanced Representation Learning for Single-Cell Transcriptomics",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "discrete_biological_symbol_stream": 1
      },
      "subtype_counts": {
        "multi_track_structural_symbol_stream": 1
      },
      "families": [
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "multi_track_structural_symbol_stream"
      ],
      "primary_subtype": "multi_track_structural_symbol_stream",
      "modalities": [
        "single-cell transcriptomics"
      ],
      "lifecycle_phases": [
        "pretraining"
      ],
      "fusion_topologies": [
        "unclear"
      ],
      "text_roles": [
        "no_text_on_this_route"
      ],
      "figure": null,
      "figure_status": "no_suitable_figure",
      "no_figure_rationale": "None of the visible panels supports the exact scGPT cell-representation-learning input path with sufficient model-specific evidence. The contact sheet is dominated by scMMGPT and downstream evaluation/summary figures; the only generic matrix panel lacks an scGPT-specific interface or model-visible carrier for the requested route.",
      "illustrative_examples": [
        {
          "subtype_id": "multi_track_structural_symbol_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_18f47c655714",
          "example_input": "AA: M K T ...   SS: H H C ...",
          "example_carrier": "aligned sequence + structure tracks",
          "example_interface": "multi-track tokenizer → generator",
          "actual_source": "raw count matrix X ∈ N N × M",
          "actual_model_visible_form": "gene symbols and their quantitative expression levels"
        }
      ],
      "routes": [
        {
          "route_id": "route_18f47c655714",
          "configuration_id": "config_b1292816fcbd",
          "route_label": "scGPT cell representation learning",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cell representation learning",
          "source_object_verbatim": "raw count matrix X ∈ N N × M",
          "source_object_normalized": "raw count matrix",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "raw count matrix X ∈ N N × M",
            "row-wise normalization",
            "retain the top 2,048 most expressed genes per cell"
          ],
          "model_visible_form_verbatim": "gene symbols and their quantitative expression levels",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "multi_track_structural_symbol_stream",
          "insertion_or_fusion_verbatim": "row-wise normalization before feeding into the model",
          "fusion_topology": "unclear",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "jointly model gene symbols and their quantitative expression levels",
          "section_heading": "3.2 Cell Representation Learning & Language Generation with Pre-Trained Models",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/texts/143"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001381::0066",
            "dense::full_2026-07-06__rec_001381::0069"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_951dfcefa907"
    },
    {
      "model_id": "model_498b709f811c",
      "model_name": "scGPT-FT",
      "record_id": "june_update_2026-06-10__rec_000152",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_1e6901520f8e",
      "paper_title": "H2O: A Foundation Model Bridging Histopathology to Spatial Multi-Omics Profiling",
      "doi": "10.64898/2026.04.21.717342",
      "paper_url": "https://doi.org/10.64898/2026.04.21.717342",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "discrete_biological_symbol_stream": 1
      },
      "subtype_counts": {
        "native_biological_token_stream": 1
      },
      "families": [
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream"
      ],
      "primary_subtype": "native_biological_token_stream",
      "modalities": [
        "spatial transcriptomics"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/june_update_2026_06_10_rec_000152_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/june_update_2026_06_10_rec_000152_5827375351f8/figure_002.png",
        "figure_index": 2,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a scientific schematic figure, panel labeled **“b.”**, titled **“Model Architecture and Training Procedure.”**\n\nIt depicts a computational biology pipeline combining **H&E imaging** and **spatial sequencing** data for **spatial transcriptomics prediction**.\n\nVisible elements include:\n\n- **Input biological data sources**\n  - H&E imaging, shown with a microscope icon and tissue image tiles.\n  - Spatial sequencing, shown with an instrument icon and a grid-like spatial transcriptomics matrix.\n\n- **Feature extraction / model components**\n  - An **Image FM** branch processes histology image patches.\n  - An **ST FM** branch processes spatial transcriptomics features.\n  - A diameter-based feature extraction inset shows circular regions at approximate scales labeled **2 µm**, **55 µm**, and **150 µm**.\n  - Extracted image and ST feature vectors are fed into a **Contrastive Learning** module.\n\n- **Contrastive learning**\n  - A similarity matrix is shown, with diagonal matching entries highlighted.\n  - Feature vectors from image and spatial transcriptomics modalities are aligned.\n  - The output connects to a convolutional block labeled **“8 Neighbors”**, followed by **“Conv & Concat.”**\n\n- **Fusion and prediction**\n  - A large yellow module labeled **FiLM** receives concatenated convolutional features and diameter/context features.\n  - The FiLM-modulated representation is passed to **ST Prediction**, producing a predicted spatial transcriptomics grid.\n\n- **Legend / training status**\n  - Gray arrows indicate **Input**.\n  - Green arrows indicate **Output**.\n  - Flame icon indicates **Trainable**.\n  - Snowflake icon indicates **Untrainable**.\n\n- **Lower training procedure panels**\n  - **Image FM** panel: shows **DINO V2** with **Student Image Transformer** and **Teacher Image Transformer**, fine-tuned using datasets labeled **TCGA**, **GTEx**, and **In-house**, producing an Image FM.\n  - **ST FM** panel: shows a **Single Cell FM** fine-tuned using **HEST1K** to produce a **Spatial Transcriptomics FM**.\n\nNo quantitative results or empirical findings are shown; the figure presents the architecture and training workflow.",
        "page_no": 4,
        "sha256": "370c807d40891c12e7df381d0f5e155b32c3952421ff2c119b3fdfbc026ffb51",
        "pixel_width": 1162,
        "pixel_height": 648,
        "crop_box": {
          "x": 0.03,
          "y": 0.05,
          "width": 0.84,
          "height": 0.58
        },
        "panel_label": "b.",
        "visible_input_object": "H&E imaging and spatial sequencing inputs with their extracted image patch and ST grid features",
        "visible_model_interface": "Contrastive Learning similarity matrix, Conv & Concat, and the FiLM fusion block before ST prediction",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps one grounded input route end-to-end: source modalities on the left, their transformed feature carriers, and the immediate fusion interface in FiLM, while excluding the output-only prediction panel and the lower training-summary panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_5531f04067d0",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "paired spatial transcriptomics gene expression data from HEST-1k",
          "actual_model_visible_form": "gene tokens, expression values, and condition tokens"
        }
      ],
      "routes": [
        {
          "route_id": "route_5531f04067d0",
          "configuration_id": "config_7a9090c449be",
          "route_label": "ST gene-expression branch for contrastive alignment",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "align a trainable histopathology foundation model and a parameter-frozen ST foundation model using paired image and gene expression data",
          "source_object_verbatim": "paired spatial transcriptomics gene expression data from HEST-1k",
          "source_object_normalized": "paired spatial transcriptomics gene expression data",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "log1p transformation",
            "gene tokenization",
            "embedding layers",
            "scGPT fine-tuning on ST data"
          ],
          "model_visible_form_verbatim": "gene tokens, expression values, and condition tokens",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "The input to fine-tune scGPT consists of three components: (1) gene (or peak) tokens, (2) expression values, and (3) condition tokens.",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "paired_alignment_input",
          "evidence_quote": "H2O is a multi-omics-guided histopathology foundation model designed to align whole-slide imaging features with  spatially  resolved  transcriptomics  and  proteomics  profiles  through  contrastive  learning.  To  enable comprehensive coverage of various tissues, organs, and diseases (Extended Data Fig. 2) across both H&E and ST modalities, we curated a diverse dataset, which contains three major parts for H2O training and evaluation (Fig. 1a). Samples from The Cancer Genome Atlas (TCGA) [48], the Genotype-Tissue Expression (GTEx) [49] project, and additional in-house breast cancer collections (Methods) were used to train the histopathology FM. Then we employed a gene expression FM based on scGPT [35], initialized from whole-human pretrained checkpoints and further fine-tuned on ST data from the HEST-1k [47] dataset to encode spatial transcriptomics knowledge. Subsequently, we trained H2O using matched H&E and ST samples from the HEST-1k dataset. To infer transcriptomics profiles from a central image patch, H2O leverages morphological context from its local neighborhood and incorporates a Feature-wise Linear Modulation (FiLM) layer for resolution-aware feature calibration. See Extended Data Fig. 3 for ablation study of these modules. We further collected three additional datasets, HTSA, OpenST, and HTAPP, to investigate the ability of H2O in imparting molecular features to H&E features.",
          "section_heading": "A transcriptomics FM fine-tuned with ST data.",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            6
          ],
          "doc_item_refs": [
            "#/texts/228",
            "#/texts/229",
            "#/texts/230",
            "#/texts/231",
            "#/texts/233",
            "#/texts/234",
            "#/texts/235",
            "#/texts/236"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000152::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_de94be356588"
    },
    {
      "model_id": "model_f67272a54ac0",
      "model_name": "SciCore-Omics",
      "record_id": "june_update_2026-06-10__rec_000148",
      "collection_batch_id": "june_update_2026-06-10",
      "collection_date": "2026-06-10",
      "review_iteration": "2026-06-10",
      "study_id": "study_b476c65521c3",
      "paper_title": "SciCore-Omics: a tri-modal foundation model unifying histology, spatial transcriptomics and language for spatial biology",
      "doi": "10.64898/2026.05.30.728937",
      "paper_url": "https://doi.org/10.64898/2026.05.30.728937",
      "route_count": 11,
      "configuration_count": 9,
      "family_counts": {
        "visual_raster_carrier": 7,
        "dense_continuous_carrier": 4
      },
      "subtype_counts": {
        "raw_slide_or_patch_input": 6,
        "virtual_token_prefix": 4,
        "patch_context_or_case_level_visual_reasoning": 1
      },
      "families": [
        "dense_continuous_carrier",
        "visual_raster_carrier"
      ],
      "subtypes": [
        "patch_context_or_case_level_visual_reasoning",
        "raw_slide_or_patch_input",
        "virtual_token_prefix"
      ],
      "primary_subtype": "raw_slide_or_patch_input",
      "modalities": [
        "histology image",
        "spatial transcriptomics"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "other_explicit",
        "placeholder_replacement",
        "side_or_generative_conditioning"
      ],
      "text_roles": [
        "generated_output",
        "instruction_or_query",
        "no_text_on_this_route",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/june_update_2026_06_10_rec_000148_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/june_update_2026_06_10_rec_000148_32d2df447629/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of SciCore-Omics. a , Model architecture of SciCore-Omics. Histological images and spatial gene expression profiles are encoded by modality-specific encoders, and projected into the language-model space. b , Unified tri-modal token representation. Image, gene virtual tokens and text tokens are jointly modelled in a shared decoder-based backbone. c , Representative downstream tasks supported by SciCore-Omics across multiple biological scales.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel schematic labeled **a**, **b**, and **c** describing a unified tri-modal model workflow for histology image, gene expression, and text inputs.\n\nPanel **a** shows input modalities from a biological sample: an **image** branch with a histology tissue patch and zoomed region, a **gene** branch with a gene expression matrix/heatmap, and a **text prompt** reading approximately “Describe the biological state of this sample using the image patch and gene expression profile.” These inputs are passed through separate components labeled **Vision Encoder**, **Gene Encoder**, and **LLM token embedding**, followed by **Resampler** and **Q-former Projector** modules for the image and gene streams.\n\nPanel **b** shows a **Unified Tri-Modal Token Space** containing token sequences for **Vision Tokens**, **Gene Tokens**, and **Text Tokens**. The token examples include image tokens such as `<image>`, `VIS01`, `VIS02`, `VIS03`; gene tokens such as `<genes>`, `GEN01`, `GEN02`, `GEN03`; and text tokens forming a prompt-like sequence such as “Describe the histology …”.\n\nPanel **c** illustrates four downstream tasks: **Task1: Gene Expression Prediction**, showing a histology patch transformed into a predicted expression/bar profile; **Task2: Spatial Domain Recognition**, showing gene/spatial features mapped to colored tissue domains; **Task3: Histopathology Classification**, showing a histology patch passed to a model that outputs text such as “The patch is Breast tissue …”; and **Task4: Sample Analysis**, showing an H&E tissue overview image passed to a model producing descriptive text beginning “The H&E overview shows …”.\n\nNo quantitative findings or experimental results are shown; this is a conceptual architecture and task overview figure.",
        "page_no": 3,
        "sha256": "ac3c609f90a2dc2bfc20262f7ec81fa5bd98e51f44a603fd99628e02ef577f49",
        "pixel_width": 891,
        "pixel_height": 442,
        "crop_box": {
          "x": 0.0,
          "y": 0.07,
          "width": 0.54,
          "height": 0.38
        },
        "panel_label": "a",
        "visible_input_object": "histology image patch / image branch",
        "visible_model_interface": "vision encoder -> resampler -> insertion into the unified token space (image placeholders / image tokens)",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the image-input path in panel a, including the histology patch source, the visual encoder/resampler transformation, and the arrow into the shared token space; excludes the gene/text-only and downstream task panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "patch_context_or_case_level_visual_reasoning",
          "family_id": "visual_raster_carrier",
          "route_id": "route_eb2576517cc5",
          "example_input": "ROI + neighboring patches + case context",
          "example_carrier": "ordered visual token bank",
          "example_interface": "context aggregator → multimodal LLM",
          "actual_source": "whole-slide H&E images",
          "actual_model_visible_form": "whole-slide H&E images"
        },
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_8d99a519d6eb",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "histology image patch",
          "actual_model_visible_form": "histology image patch"
        },
        {
          "subtype_id": "virtual_token_prefix",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_bf7aa611d54c",
          "example_input": "cell embedding + prompt",
          "example_carrier": "<bio₁> <bio₂> ... <bioₖ> [prompt tokens]",
          "example_interface": "soft prefix → LLM stream",
          "actual_source": "gene expression profile",
          "actual_model_visible_form": "32 query virtual tokens"
        }
      ],
      "routes": [
        {
          "route_id": "route_8d99a519d6eb",
          "configuration_id": "config_7602ba38914a",
          "route_label": "histology image patch to stage I image-text pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Stage I: histology-text alignment",
          "source_object_verbatim": "histology image patch",
          "source_object_normalized": "histology image patch",
          "source_modality_normalized": "histology image",
          "transformation_chain_verbatim": [
            "visual encoder",
            "resampler module"
          ],
          "model_visible_form_verbatim": "histology image patch",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "inserted into placeholders",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "The first stage adapted the visual branch and language backbone to pathology image-text semantics.",
          "section_heading": "3.4 Training corpus construction and progressive multimodal alignment",
          "supporting_figure_or_table": "Extended Data Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            12,
            13,
            14,
            17,
            24,
            25
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/105",
            "#/texts/108",
            "#/texts/109",
            "#/texts/110",
            "#/texts/30",
            "#/texts/85",
            "#/texts/86",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89",
            "#/texts/90"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_001"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000148::0001",
            "dense::june_update_2026-06-10__rec_000148::0027"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_bf7aa611d54c",
          "configuration_id": "config_aaa9b7224fe0",
          "route_label": "gene expression profile to stage II gene-text alignment",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Stage II: gene-text alignment",
          "source_object_verbatim": "gene expression profile",
          "source_object_normalized": "gene expression profile",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "gene encoder",
            "Q-Former module",
            "projection layer"
          ],
          "model_visible_form_verbatim": "32 query virtual tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "virtual_token_prefix",
          "insertion_or_fusion_verbatim": "projected into the language-model hidden space",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "the gene encoder first maps the gene expression profile into molecular embeddings",
          "section_heading": "3.2 Model architecture",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": "Stage II is named as an alignment stage in the paper and normalized here to fine_tuning.",
          "pages": [
            3,
            12,
            17,
            24,
            25
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/30",
            "#/texts/85",
            "#/texts/86",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89",
            "#/texts/90",
            "#/texts/91",
            "#/texts/92"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_002"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000148::0001",
            "dense::june_update_2026-06-10__rec_000148::0027"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b70635dcdd77",
          "configuration_id": "config_7cc664b41139",
          "route_label": "histology image patch in stage III tri-modal alignment",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Stage III: tri-modal alignment",
          "source_object_verbatim": "histology image patch",
          "source_object_normalized": "histology image patch",
          "source_modality_normalized": "histology image",
          "transformation_chain_verbatim": [
            "visual encoder",
            "resampler module",
            "projection layer"
          ],
          "model_visible_form_verbatim": "histology image patch",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "projected into the same token space as the molecular virtual tokens",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Stage III jointly optimizes image, gene and text representations using paired tri-modal data.",
          "section_heading": "3.4 Training corpus construction and progressive multimodal alignment",
          "supporting_figure_or_table": "Extended Data Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": "Stage III is a joint alignment training stage and normalized here to fine_tuning.",
          "pages": [
            3,
            12,
            17,
            24,
            25
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/6",
            "#/texts/251",
            "#/texts/30",
            "#/texts/85",
            "#/texts/86",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89",
            "#/texts/90"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_003"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000148::0001",
            "dense::june_update_2026-06-10__rec_000148::0027"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_01fead809e1c",
          "configuration_id": "config_7cc664b41139",
          "route_label": "gene expression profile in stage III tri-modal alignment",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Stage III: tri-modal alignment",
          "source_object_verbatim": "gene expression profile",
          "source_object_normalized": "gene expression profile",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "gene encoder",
            "Q-Former module",
            "projection layer"
          ],
          "model_visible_form_verbatim": "32 query virtual tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "virtual_token_prefix",
          "insertion_or_fusion_verbatim": "projected into the same token space as the visual virtual tokens",
          "fusion_topology": "placeholder_replacement",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Stage III jointly optimizes image, gene and text representations using paired tri-modal data.",
          "section_heading": "3.4 Training corpus construction and progressive multimodal alignment",
          "supporting_figure_or_table": "Extended Data Fig. 1",
          "evidence_status": "explicit_text",
          "uncertainty": "Stage III is a joint alignment training stage and normalized here to fine_tuning.",
          "pages": [
            3,
            12,
            17,
            24,
            25
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/6",
            "#/texts/251",
            "#/texts/30",
            "#/texts/85",
            "#/texts/86",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89",
            "#/texts/90"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_004"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000148::0001",
            "dense::june_update_2026-06-10__rec_000148::0027"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f4eb22871096",
          "configuration_id": "config_76d1d94582d2",
          "route_label": "gene expression profile to transcriptome-to-language generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "Transcriptome-to-language generation",
          "source_object_verbatim": "gene expression profile",
          "source_object_normalized": "spot-level gene expression profile",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "gene encoder",
            "Q-Former module",
            "projection layer"
          ],
          "model_visible_form_verbatim": "gene virtual tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "virtual_token_prefix",
          "insertion_or_fusion_verbatim": "Each model was prompted with transcriptomic input to generate a biological description",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "generated_output",
          "input_status": "actual_model_input",
          "evidence_quote": "Each model was prompted with transcriptomic input to generate a biological description",
          "section_heading": "3.7 Transcriptome-to-language evaluation",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            3,
            4,
            12,
            15,
            17,
            24,
            25
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/texts/132",
            "#/texts/133",
            "#/texts/26",
            "#/texts/27",
            "#/texts/30",
            "#/texts/31",
            "#/texts/36",
            "#/texts/85",
            "#/texts/86",
            "#/texts/87",
            "#/texts/88",
            "#/texts/89",
            "#/texts/90"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_005"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000148::0002",
            "dense::june_update_2026-06-10__rec_000148::0013",
            "dense::june_update_2026-06-10__rec_000148::0027",
            "dense::june_update_2026-06-10__rec_000148::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_cef5e0303c7a",
          "configuration_id": "config_1fff1f8ba2ec",
          "route_label": "histological image patch to generative spatial domain recognition",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "image-only, gene-only or joint image-gene input",
          "source_object_verbatim": "histological image patch",
          "source_object_normalized": "histological image patch",
          "source_modality_normalized": "histology image",
          "transformation_chain_verbatim": [
            "visual encoder",
            "resampler module",
            "projection layer"
          ],
          "model_visible_form_verbatim": "histological image patch",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "takes histological image patches and/or gene expression profiles as input to generate biological descriptions",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "generated_output",
          "input_status": "actual_model_input",
          "evidence_quote": "for each spot, SciCoreOmics takes histological image patches and/or gene expression profiles as input to generate biological descriptions",
          "section_heading": "1.4 SciCore-Omics improves spot-level spatial domain recognition",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            16,
            26
          ],
          "doc_item_refs": [
            "#/pictures/9",
            "#/texts/140",
            "#/texts/141",
            "#/texts/142",
            "#/texts/258",
            "#/texts/43",
            "#/texts/44",
            "#/texts/45",
            "#/texts/46"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_006"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000148::0006",
            "dense::june_update_2026-06-10__rec_000148::0015"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2abfc684ce18",
          "configuration_id": "config_1fff1f8ba2ec",
          "route_label": "gene expression profile to generative spatial domain recognition",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "image-only, gene-only or joint image-gene input",
          "source_object_verbatim": "gene expression profile",
          "source_object_normalized": "gene expression profile",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "gene encoder",
            "Q-Former module",
            "projection layer"
          ],
          "model_visible_form_verbatim": "gene virtual tokens",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "virtual_token_prefix",
          "insertion_or_fusion_verbatim": "takes histological image patches and/or gene expression profiles as input to generate biological descriptions",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "generated_output",
          "input_status": "actual_model_input",
          "evidence_quote": "for each spot, SciCoreOmics takes histological image patches and/or gene expression profiles as input to generate biological descriptions",
          "section_heading": "1.4 SciCore-Omics improves spot-level spatial domain recognition",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            5,
            16,
            26
          ],
          "doc_item_refs": [
            "#/pictures/9",
            "#/texts/140",
            "#/texts/141",
            "#/texts/142",
            "#/texts/258",
            "#/texts/43",
            "#/texts/44",
            "#/texts/45",
            "#/texts/46"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_007"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000148::0006",
            "dense::june_update_2026-06-10__rec_000148::0015"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2f13a5f5d52d",
          "configuration_id": "config_7a0bdd4743ab",
          "route_label": "H&E image patch to gene expression prediction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "histology-based gene expression prediction",
          "source_object_verbatim": "H&E image patches",
          "source_object_normalized": "H&E image patches",
          "source_modality_normalized": "histology image",
          "transformation_chain_verbatim": [
            "visual encoder",
            "regression head"
          ],
          "model_visible_form_verbatim": "H&E image patches",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "regression head on top of SciCore-Omics visual embeddings",
          "fusion_topology": "other_explicit",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "To evaluate whether histology-derived representations retained molecular information, we trained a lightweight regression head on top of SciCore-Omics visual embeddings to predict spot-level gene expression from H&E image patches. Experiments were conducted on the normal human heart spatial transcriptomics dataset comprising 39 sections. To avoid information leakage caused by spatial correlations between neighbouring spots within the same tissue section, we used section-level 10-fold crossvalidation. Specifically, the 39 sections were partitioned into 10 folds at the section level. In each fold, the model was trained only on spots from the training sections and tested on spots from the held-out section or sections. The prediction targets were the top 50 genes with the highest expression in the training set, and performance was evaluated on the held-out spots from the test sections. Before training and evaluation, all spots were subjected to quality-control filtering, retaining only spots with at least 500 detected genes.",
          "section_heading": "3.8 Spatial gene expression prediction from histology",
          "supporting_figure_or_table": "Figure 3",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            16
          ],
          "doc_item_refs": [
            "#/texts/135",
            "#/texts/138"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_008"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1fd61cdc95ef",
          "configuration_id": "config_b78e26b4d1c2",
          "route_label": "H&E image to zero-shot histopathology classification",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "zero-shot histopathological classification",
          "source_object_verbatim": "H&E images",
          "source_object_normalized": "H&E images",
          "source_modality_normalized": "histology image",
          "transformation_chain_verbatim": [
            "H&E image",
            "prompt specifying dataset-specific candidate labels",
            "select one category"
          ],
          "model_visible_form_verbatim": "H&E images",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "the prompt specified the dataset-specific candidate labels",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "For each image, the prompt specified the dataset-specific candidate labels",
          "section_heading": "1.5 SciCore-Omics supports zero-shot tissue-level histopathology classification",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            9
          ],
          "doc_item_refs": [
            "#/pictures/3",
            "#/pictures/4",
            "#/texts/53",
            "#/texts/54",
            "#/texts/64"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_009"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000148::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8b527bbd7b4b",
          "configuration_id": "config_a0e1310c2cc0",
          "route_label": "histopathology image to PathVQA",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "PathVQA",
          "source_object_verbatim": "histopathology images",
          "source_object_normalized": "histopathology images",
          "source_modality_normalized": "histology image",
          "transformation_chain_verbatim": [
            "visual encoder",
            "resampler module",
            "projection layer"
          ],
          "model_visible_form_verbatim": "histopathology images paired with medically grounded questions and answers",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "histopathology images paired with medically grounded questions and answers",
          "fusion_topology": "concatenation",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "PathVQA, a pathology visual question answering benchmark consisting of histopathology images paired with medically grounded questions and answers",
          "section_heading": "1.6 SciCore-Omics enables H&E-based pathology reasoning",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            12,
            17
          ],
          "doc_item_refs": [
            "#/texts/151",
            "#/texts/56",
            "#/texts/57",
            "#/texts/58",
            "#/texts/61",
            "#/texts/81",
            "#/texts/82",
            "#/texts/83"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_010"
          ],
          "dense_candidate_refs": [
            "dense::june_update_2026-06-10__rec_000148::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_eb2576517cc5",
          "configuration_id": "config_9a3d22deffed",
          "route_label": "whole-slide H&E image to case-level pathology reasoning",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "H&E-only case-level pathology workflow",
          "source_object_verbatim": "whole-slide H&E images",
          "source_object_normalized": "whole-slide H&E images",
          "source_modality_normalized": "histology image",
          "transformation_chain_verbatim": [
            "low-magnification overview",
            "high-magnification patch assessment",
            "case-level evidence synthesis"
          ],
          "model_visible_form_verbatim": "whole-slide H&E images",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "patch_context_or_case_level_visual_reasoning",
          "insertion_or_fusion_verbatim": "only H&E whole-slide images were provided as model input",
          "fusion_topology": "other_explicit",
          "text_role": "generated_output",
          "input_status": "actual_model_input",
          "evidence_quote": "We next examined whether the model could extend this capability to a more clinically oriented caselevel reasoning setting. In this evaluation, only the whole-slide H&E images were used as the model input, while immunohistochemistry and pathology reports were withheld from the model and referenced solely for expert assessment (Fig. 6b). The inference workflow was designed to simulate a staged pathological review process. First, SciCore-Omics reviewed the low-magnification overview of the whole-slide image to identify tumour-rich regions and representative tumour-stroma structures. Second, representative tumour patches were selected from these regions for high-magnification assessment, with attention to interpretable morphological features, including tumour cell density, nuclear atypia, hyperchromasia, suspected mitotic activity, necrosis and stromal reaction. Finally, the model integrated evidence across multiple regions to generate a case-level interpretation of the proliferative status.",
          "section_heading": "3.12 H&E-only case-level pathology workflow",
          "supporting_figure_or_table": "Figure 6b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8
          ],
          "doc_item_refs": [
            "#/texts/56",
            "#/texts/57",
            "#/texts/58",
            "#/texts/61"
          ],
          "source_candidate_refs": [
            "june_update_2026-06-10__rec_000148::route_011"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_887038a8a19a"
    },
    {
      "model_id": "model_e0f70ec41e97",
      "model_name": "scMIR",
      "record_id": "update_2026-08-09__rec_000106",
      "collection_batch_id": "update_2026-08-09",
      "collection_date": "2026-08-09",
      "review_iteration": "2026-08-09",
      "study_id": "study_ab49825e939e",
      "paper_title": "scMIR: a vision-language foundation model for single-cell light microscopy image representation",
      "doi": "",
      "paper_url": "",
      "route_count": 10,
      "configuration_count": 9,
      "family_counts": {
        "visual_raster_carrier": 9,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "raw_slide_or_patch_input": 9,
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream",
        "visual_raster_carrier"
      ],
      "subtypes": [
        "raw_slide_or_patch_input",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "raw_slide_or_patch_input",
      "modalities": [
        "bright-field single-cell microscopy images",
        "confocal fluorescence microscopy images",
        "differential interference contrast microscopy images",
        "phase-contrast microscopy images",
        "quantitative phase imaging",
        "single-cell light microscopy images",
        "single-cell microscopy images",
        "text",
        "widefield fluorescence microscopy images"
      ],
      "lifecycle_phases": [
        "evaluation",
        "pretraining"
      ],
      "fusion_topologies": [
        "encoder_decoder",
        "query_bottleneck",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "no_text_on_this_route",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/update_2026_08_09_rec_000106_figure_001.png",
        "source_path": "data/living_catalog_updates/update_2026-08-09/11_docling_vlm/profiles/figures/update_2026_08_09_rec_000106_c9bc52778731/figure_001.png",
        "figure_index": 1,
        "caption": "Fig. 1 | Overview of framework. a . We curated a diverse collection of light microscopy datasets.  The  left  panel  summarizes  dataset  diversity  across  microscopy  types,  cell types,  perturbation  conditions  and  species.  The  right  panel  presents  a  hierarchical organization  of  the  datasets,  grouped  into  pre-training  and  evaluation  sets,  with  the number  of  images  annotated  at  each  level  of  the  hierarchy.  Microscopy  types  are",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific figure with three labeled panels: `a. Dataset characterization`, `b. Model architecture`, and `c. Downstream analysis based on learned image embedding`.\n\nPanel `a` summarizes dataset characterization. A circular chart shows microscopy/data categories including `BF`, `DIC`, `Conf`, `WF`, `SR`, `QPI`, `PC`, `Cell type (100+)`, `Perturbation (300+)`, and species labels including `Human`, `Mouse`, `Yeast`, and `Bacteria`. The center contains example microscopy cell images. A right-side hierarchy organizes data by `Granularity`, `Scale`, `Attribute`, and `Number`, mapping `Single-cell`, `Full-field`, `Subcellular`, `Cell-level`, and `Population` to attributes such as `Protein localization`, `Cell identity/state`, and `Perturbation response`. Counts are shown for `Pre-training` and `Evaluation`, including values such as `72,832`, `2,622,150`, `40,000`, `156,669`, `95,125`, and `117,628`.\n\nPanel `b` shows a model architecture for single-cell image-text learning. Inputs include a generated `Caption` with keywords such as `Species`, `Source`, `Microscopy`, `Cell type`, and `Biological attribute`, plus a `Single-cell image`. The caption is passed through a `Tokenizer` into `Text tokens`. The image is passed through an `Image encoder` into `Image tokens`. A `Q-Former` module connects text/image tokens to image-text alignment, represented by a matrix comparing `Image feature` entries `I1 ... Im` with `Text feature` entries `T1 ... Tn`. A second shared-weight `Q-Former` receives masked image tokens, followed by an `Image decoder`, performing `Mask tokens reconstruction`.\n\nPanel `c` illustrates downstream analysis using learned image embeddings. `Evaluation images` are input into `scMIR`, producing a learned image embedding matrix. The embeddings are used for `Cell classification`, `Morphological inference`, `Cell clustering`, and `Batch correction`, shown as schematic point clusters and transformed groupings.\n\nVisible biological source objects include single-cell microscopy images and schematic cells. The figure describes transformations from microscopy images and captions into tokenized text/image representations, learned embeddings, reconstructed image tokens, and downstream biological analysis outputs.",
        "page_no": 4,
        "sha256": "e32f5083d9f44df98cba814988d7d36f771f79c915f4c4305217d0bd849c0c84",
        "pixel_width": 822,
        "pixel_height": 1034,
        "crop_box": {
          "x": 0,
          "y": 0.368,
          "width": 0.742,
          "height": 0.334
        },
        "panel_label": "b",
        "visible_input_object": "single-cell microscopy image and structured caption keywords",
        "visible_model_interface": "caption -> Tokenizer -> text tokens; image -> frozen image encoder -> image tokens -> shared Q-Former",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop isolates panel b and keeps the source image, caption path, tokenization, image encoder, image tokens, and shared Q-Former visible while excluding panels a/c and the downstream-only outputs.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_5963db69cdb5",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "single-cell microscopy images",
          "actual_model_visible_form": "visual token embeddings / query embeddings"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_d02dbf078b83",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "structured textual captions",
          "actual_model_visible_form": "text tokens"
        }
      ],
      "routes": [
        {
          "route_id": "route_5963db69cdb5",
          "configuration_id": "config_b14160e930e3",
          "route_label": "pretraining single-cell microscopy image to multimodal representation",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "paired single-cell images and structured textual captions",
          "source_object_verbatim": "single-cell microscopy images",
          "source_object_normalized": "single-cell microscopy images in paired image-text pretraining",
          "source_modality_normalized": "single-cell light microscopy images",
          "transformation_chain_verbatim": [
            "frozen image encoder",
            "Q-Former initialized from BLIP-2",
            "image-text alignment objective"
          ],
          "model_visible_form_verbatim": "visual token embeddings / query embeddings",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "frozen image encoder coupled with Q-Former",
          "fusion_topology": "query_bottleneck",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "During pre-training, scMIR is trained on paired single-cell images and structured textual captions",
          "section_heading": "scMIR model",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            16
          ],
          "doc_item_refs": [
            "#/texts/1",
            "#/texts/2",
            "#/texts/3",
            "#/texts/4",
            "#/texts/49",
            "#/texts/5",
            "#/texts/6"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000106::route_001"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000106::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d02dbf078b83",
          "configuration_id": "config_fc60f1c1dee8",
          "route_label": "pretraining structured caption text to semantic alignment",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "structured textual descriptions constructed from controlled keywords",
          "source_object_verbatim": "structured textual captions",
          "source_object_normalized": "structured textual captions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "controlled keywords (species, data source, microscopy type, cell type, and related annotations)",
            "structured textual descriptions",
            "Tokenizer",
            "Text tokens"
          ],
          "model_visible_form_verbatim": "text tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "text transformer in the Q-Former image-text alignment path",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "structured textual descriptions were constructed using controlled keywords",
          "section_heading": "Overview of framework",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes the captions as both alignment input and semantic supervision; token-level routing is only described at a high level.",
          "pages": [
            16
          ],
          "doc_item_refs": [
            "#/texts/49"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000106::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_afa5192ae8e0",
          "configuration_id": "config_ca822a1cc5d2",
          "route_label": "pretraining masked image reconstruction route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "image reconstruction objective on masked image features",
          "source_object_verbatim": "single-cell microscopy images",
          "source_object_normalized": "single-cell microscopy images",
          "source_modality_normalized": "single-cell light microscopy images",
          "transformation_chain_verbatim": [
            "frozen image encoder",
            "random masking of visual tokens",
            "shared-weight Q-Former",
            "lightweight Transformer decoder",
            "reconstruct original embeddings"
          ],
          "model_visible_form_verbatim": "masked visual token embeddings",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "shared-weight Q-Former followed by a lightweight Transformer decoder",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "an image reconstruction objective operates on masked image features",
          "section_heading": "Overview of framework",
          "supporting_figure_or_table": "Fig. 1b",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/14",
            "#/texts/15",
            "#/texts/16",
            "#/texts/17"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000106::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_221cccfc895d",
          "configuration_id": "config_6201c49942b7",
          "route_label": "downstream image-only embedding extraction",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "the model takes only single-cell microscopy images as input and outputs task-agnostic image embeddings",
          "source_object_verbatim": "single-cell microscopy images",
          "source_object_normalized": "single-cell microscopy images used for downstream evaluation",
          "source_modality_normalized": "single-cell microscopy images",
          "transformation_chain_verbatim": [
            "frozen image encoder",
            "Q-Former",
            "fixed image-level representations"
          ],
          "model_visible_form_verbatim": "task-agnostic image embeddings",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "pretrained image encoder and Q-Former",
          "fusion_topology": "query_bottleneck",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "the model takes only single-cell microscopy images as input and outputs task-agnostic image embeddings",
          "section_heading": "scMIR model",
          "supporting_figure_or_table": "Fig. 1c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            20,
            32,
            33
          ],
          "doc_item_refs": [
            "#/texts/252",
            "#/texts/254",
            "#/texts/256",
            "#/texts/258",
            "#/texts/260",
            "#/texts/262",
            "#/texts/96"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000106::route_004"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000106::0006",
            "dense::update_2026-08-09__rec_000106::0007",
            "dense::update_2026-08-09__rec_000106::0008",
            "dense::update_2026-08-09__rec_000106::0009",
            "dense::update_2026-08-09__rec_000106::0010",
            "dense::update_2026-08-09__rec_000106::0011"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e2bec151f3ad",
          "configuration_id": "config_bc580a431213",
          "route_label": "pretraining LIVECell cell instances to multimodal representation",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "multimodal pretraining on curated public microscopy datasets",
          "source_object_verbatim": "cell instances provided with the original dataset",
          "source_object_normalized": "LIVECell cell instances",
          "source_modality_normalized": "phase-contrast microscopy images",
          "transformation_chain_verbatim": [
            "directly used the cell instances provided with the original dataset",
            "frozen image encoder",
            "Q-Former initialized from BLIP-2",
            "image-text alignment objective"
          ],
          "model_visible_form_verbatim": "visual token embeddings / query embeddings",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "frozen image encoder coupled with Q-Former",
          "fusion_topology": "query_bottleneck",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "For LIVECell, we directly used the cell instances provided with the original dataset.",
          "section_heading": "Pre-training datasets",
          "supporting_figure_or_table": "Table S1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            31
          ],
          "doc_item_refs": [
            "#/texts/236"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000106::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6ad595695f15",
          "configuration_id": "config_9ca79c8ef17c",
          "route_label": "pretraining cpg0000 full-field fluorescence images to single-cell instances",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "full-field images segmented into single-cell instances for multimodal pretraining",
          "source_object_verbatim": "full-field images",
          "source_object_normalized": "cpg0000 full-field fluorescence images",
          "source_modality_normalized": "widefield fluorescence microscopy images",
          "transformation_chain_verbatim": [
            "Cellpose segmentation with the Hoechst channel serving as the nuclear reference",
            "zero-padding to square inputs when needed",
            "frozen image encoder",
            "Q-Former initialized from BLIP-2",
            "image-text alignment objective"
          ],
          "model_visible_form_verbatim": "visual token embeddings / query embeddings",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "frozen image encoder coupled with Q-Former",
          "fusion_topology": "query_bottleneck",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "We  curated  six  publicly  available  microscopy  datasets  for  multimodal  pretraining, including  LIVECell,  cpg0000,  PBC,  HPA,  HepG2,  and  B_ALL.  Among  them, LIVECell and cpg0000 consist of full-field microscopy images, whereas the remaining datasets provide single-cell images. For LIVECell, we directly used the cell instances provided with the original dataset. For cpg0000, full-field images were segmented into single-cell  instances  using  the  Cellpose  [1]  framework,  with  the  Hoechst  channel serving as the nuclear reference. When segmented cell images did not conform to a square shape, zero-padding was applied to obtain square inputs. For LIVECell, cpg0000, and HPA, the original pretraining and evaluation splits provided by the datasets were preserved, ensuring no overlap between the two subsets. Detailed descriptions of each dataset,  along  with  a  comprehensive  summary  of  the  pretraining  data  statistics,  are provided in Table S1.",
          "section_heading": "Pre-training datasets",
          "supporting_figure_or_table": "Table S1",
          "evidence_status": "explicit_text",
          "uncertainty": "The original discovery inventory marked this candidate grounding_valid=false, but the canonical supplement explicitly describes the cpg0000 segmentation route.",
          "pages": [
            31
          ],
          "doc_item_refs": [
            "#/texts/236"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000106::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5889232d680f",
          "configuration_id": "config_bc580a431213",
          "route_label": "pretraining PBC bright-field peripheral blood cell images to multimodal representation",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "multimodal pretraining on curated public microscopy datasets",
          "source_object_verbatim": "bright-field single-cell microscopy images of peripheral blood cells",
          "source_object_normalized": "PBC peripheral blood cells",
          "source_modality_normalized": "bright-field single-cell microscopy images",
          "transformation_chain_verbatim": [
            "frozen image encoder",
            "Q-Former initialized from BLIP-2",
            "image-text alignment objective"
          ],
          "model_visible_form_verbatim": "visual token embeddings / query embeddings",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "frozen image encoder coupled with Q-Former",
          "fusion_topology": "query_bottleneck",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "PBC is a curated bright-field single-cell microscopy dataset comprising 17,092 images of peripheral blood cells",
          "section_heading": "PBC",
          "supporting_figure_or_table": "Table S1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            31,
            50
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/texts/236",
            "#/texts/242"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000106::route_007"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000106::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f65e8bf88650",
          "configuration_id": "config_2fca27311727",
          "route_label": "pretraining HPA confocal subcellular localization images to multimodal representation",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "multimodal pretraining on the HPA subcellular localization subset",
          "source_object_verbatim": "single-cell confocal microscopy images from the HPA subcellular localization subset",
          "source_object_normalized": "HPA subcellular localization subset",
          "source_modality_normalized": "confocal fluorescence microscopy images",
          "transformation_chain_verbatim": [
            "frozen image encoder",
            "Q-Former initialized from BLIP-2",
            "image-text alignment objective"
          ],
          "model_visible_form_verbatim": "visual token embeddings / query embeddings",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "frozen image encoder coupled with Q-Former",
          "fusion_topology": "query_bottleneck",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "We used single-cell confocal microscopy images from the HPA subcellular localization subset",
          "section_heading": "HPA",
          "supporting_figure_or_table": "Table S1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            32
          ],
          "doc_item_refs": [
            "#/texts/244"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000106::route_008"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000106::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0671386d0d64",
          "configuration_id": "config_283ba1300733",
          "route_label": "pretraining HepG2 DIC cell images to multimodal representation",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "multimodal pretraining on segmented HepG2 single-cell images",
          "source_object_verbatim": "segmented single-cell images derived from the HepG2 dataset",
          "source_object_normalized": "HepG2 segmented single-cell images",
          "source_modality_normalized": "differential interference contrast microscopy images",
          "transformation_chain_verbatim": [
            "subset selection of segmented single-cell images",
            "frozen image encoder",
            "Q-Former initialized from BLIP-2",
            "image-text alignment objective"
          ],
          "model_visible_form_verbatim": "visual token embeddings / query embeddings",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "frozen image encoder coupled with Q-Former",
          "fusion_topology": "query_bottleneck",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "The HepG2 dataset comprises 520 differential interference contrast (DIC) microscopy images containing 12,198 HepG2 human liver cancer cells",
          "section_heading": "HepG2",
          "supporting_figure_or_table": "Table S1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            31,
            32,
            50
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/texts/236",
            "#/texts/246"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000106::route_009"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000106::0004"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_303a4118d76c",
          "configuration_id": "config_670c921975a0",
          "route_label": "pretraining B_ALL QPI single-cell images to multimodal representation",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "multimodal pretraining on quantitative phase imaging single-cell data",
          "source_object_verbatim": "quantitative phase imaging single-cell images collected from four healthy donors",
          "source_object_normalized": "B_ALL single-cell quantitative phase images",
          "source_modality_normalized": "quantitative phase imaging",
          "transformation_chain_verbatim": [
            "frozen image encoder",
            "Q-Former initialized from BLIP-2",
            "image-text alignment objective"
          ],
          "model_visible_form_verbatim": "quantitative phase imaging single-cell images",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "frozen image encoder coupled with Q-Former",
          "fusion_topology": "query_bottleneck",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "B_ALL is a quantitative phase imaging (QPI) dataset comprising single-cell images collected from four healthy donors.",
          "section_heading": "B_ALL",
          "supporting_figure_or_table": "Table S1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            31,
            32,
            50
          ],
          "doc_item_refs": [
            "#/tables/0",
            "#/texts/236",
            "#/texts/248"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__rec_000106::route_010"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__rec_000106::0005"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_d5e749b180fc"
    },
    {
      "model_id": "model_ffaa8b52273c",
      "model_name": "scMMGPT",
      "record_id": "full_2026-07-06__rec_001381",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_d0e026b8e9c9",
      "paper_title": "Language-Enhanced Representation Learning for Single-Cell Transcriptomics",
      "doi": "",
      "paper_url": "",
      "route_count": 5,
      "configuration_count": 4,
      "family_counts": {
        "dense_continuous_carrier": 3,
        "text_native_token_stream": 2
      },
      "subtype_counts": {
        "connector_mediated_embedding": 2,
        "serialized_biological_context_or_ordered_profile": 2,
        "direct_projected_embedding": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "connector_mediated_embedding",
        "direct_projected_embedding",
        "serialized_biological_context_or_ordered_profile"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "single-cell transcriptomics",
        "text"
      ],
      "lifecycle_phases": [
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "prefix",
        "query_bottleneck",
        "unclear"
      ],
      "text_roles": [
        "generated_output",
        "no_text_on_this_route",
        "paired_alignment_supervision",
        "semantic_annotation"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001381_figure_009.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001381_9e666a0934af/figure_009.png",
        "figure_index": 9,
        "caption": "Figure 5: The two-stage cross-modal pre-training scheme. (a) In cross-modal discriminative pretraining, the model achieves cell-text integration by distinguishing matched cell-text pairs from unrelated pairs through contrastive and matching objectives. (b) In cross-modal generative pretraining, the model continues knowledge integration via unified generative tasks, including cell-to-text and text-to-cell generation objectives.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic scientific figure with two labeled panels describing a cross-modal pretraining framework for single-cell and text models.\n\nPanel (a), “Stage 1: Cross-modal Discriminative Pre-training,” shows scRNA-seq input for “a B cell” represented as gene tokens/expression bars feeding into an scLLM. Text descriptions are divided into negative descriptions, such as mismatched cell types including megakaryocyte and natural killer cell, and a positive description identifying the cell as a B cell. The scLLM output and text descriptions connect through a Q-former projector with self-attention, cross-attention, and FFN blocks. The objective is labeled “Cell-Text Matching” and “Cell-Text Contrasting.”\n\nPanel (b), “Stage 2: Cross-modal Generative Pre-training,” has two workflows. The top “Cell-to-Text Generation” path sends scRNA/gene information through an scLLM, then a Q-former projector, then “Cell Features” into a Text LLM to generate a textual cell description. The bottom “Text-to-Cell Generation” path sends a B-cell text description through a Text LLM to produce embeddings, then through a cross-attention projector into an scLLM interface with gene tokens and expression levels.\n\nBiological source objects include scRNA-seq measurements from a B cell, gene tokens, expression-level bars, and cell-type descriptions. The figure illustrates transformations between single-cell gene-expression representations and natural-language descriptions using scLLM, Text LLM, Q-former, and cross-attention projection modules.",
        "page_no": 6,
        "sha256": "80bb030c04eda3d17e5465d232fad0afe4f21a558c00614307afd829fa4786f8",
        "pixel_width": 785,
        "pixel_height": 284,
        "crop_box": {
          "x": 0.49,
          "y": 0,
          "width": 0.51,
          "height": 0.49
        },
        "panel_label": "(b) Stage 2: Cross-modal Generative Pre-training, cell-to-text route",
        "visible_input_object": "scRNA-seq / gene-token cell input for a B cell",
        "visible_model_interface": "scLLM -> Q-former Projector -> Cell Features -> Text LLM",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop isolates the upper-right cell-to-text generation pathway with the source cell input, the scLLM, Q-former projector, cell features, and the Text LLM arrow flow still readable, while excluding the left discriminative panel and the lower text-to-cell branch.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "connector_mediated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_d9d1fbe0e6f2",
          "example_input": "image / omics encoder states",
          "example_carrier": "Q-Former or adapter query vectors",
          "example_interface": "connector → LLM cross-modal interface",
          "actual_source": "normalized single-cell expression vector",
          "actual_model_visible_form": "projected cell embedding / cell features"
        },
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_0e7af367b8f6",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "a set of cells { x ( i ) }",
          "actual_model_visible_form": "c ( i ) = scGPT ( x ( i ) )"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_2d321e59d60e",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "textual descriptions of cells",
          "actual_model_visible_form": "textual descriptions of cells"
        }
      ],
      "routes": [
        {
          "route_id": "route_d9d1fbe0e6f2",
          "configuration_id": "config_da094946c522",
          "route_label": "scMMGPT cell-to-text generation",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Cell → Text Generation",
          "source_object_verbatim": "normalized single-cell expression vector",
          "source_object_normalized": "normalized single-cell expression vector",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "row-wise normalization",
            "retain the top 2,048 most expressed genes per cell",
            "gene tokens",
            "scLLM",
            "Q-Former projector",
            "Text LLM"
          ],
          "model_visible_form_verbatim": "projected cell embedding / cell features",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "cell-to-text module",
          "fusion_topology": "query_bottleneck",
          "text_role": "generated_output",
          "input_status": "actual_model_input",
          "evidence_quote": "cell-to-text module",
          "section_heading": "3.3.2 Cross-Modal Pre-training for Cell-Text Integration",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            1,
            2,
            4,
            5,
            6,
            7
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/pictures/6",
            "#/pictures/8",
            "#/texts/211",
            "#/texts/213",
            "#/texts/28",
            "#/texts/30",
            "#/texts/314",
            "#/texts/315",
            "#/texts/316",
            "#/texts/317",
            "#/texts/318",
            "#/texts/46",
            "#/texts/8",
            "#/texts/9"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001381::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001381::0076",
            "dense::full_2026-07-06__rec_001381::0077",
            "dense::full_2026-07-06__rec_001381::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2d321e59d60e",
          "configuration_id": "config_999e5ded8642",
          "route_label": "scMMGPT text-to-cell generation",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "Text → Cell Generation",
          "source_object_verbatim": "textual descriptions of cells",
          "source_object_normalized": "textual descriptions of cells",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenize the textual description into a sequence of tokens",
            "Text LLM",
            "intermediate embedding c'",
            "text-to-cell projector",
            "soft prompts",
            "scLLM"
          ],
          "model_visible_form_verbatim": "textual descriptions of cells",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "A lightweight text-to-cell projector then transforms this intermediate embedding into a soft prompt for the scLLM.",
          "fusion_topology": "prefix",
          "text_role": "semantic_annotation",
          "input_status": "actual_model_input",
          "evidence_quote": "soft prompt",
          "section_heading": "3.3.2 Cross-Modal Pre-training for Cell-Text Integration",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The exact prompt template is not shown; the route is reconstructed at the text-input boundary.",
          "pages": [
            1,
            2,
            4,
            5,
            6,
            7
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/pictures/1",
            "#/pictures/6",
            "#/pictures/8",
            "#/texts/207",
            "#/texts/208",
            "#/texts/209",
            "#/texts/211",
            "#/texts/213",
            "#/texts/28",
            "#/texts/30",
            "#/texts/314",
            "#/texts/315",
            "#/texts/316",
            "#/texts/317",
            "#/texts/318",
            "#/texts/46",
            "#/texts/8",
            "#/texts/9"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001381::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001381::0076",
            "dense::full_2026-07-06__rec_001381::0078",
            "dense::full_2026-07-06__rec_001381::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_82d0d8cf6fc5",
          "configuration_id": "config_a2cb5fda0aed",
          "route_label": "scMMGPT cell-to-text discrimination",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cross-modal discriminative pretraining",
          "source_object_verbatim": "normalized single-cell expression vector",
          "source_object_normalized": "normalized single-cell expression vector",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "row-wise normalization",
            "retain the top 2,048 most expressed genes per cell",
            "scGPT",
            "Q-Former"
          ],
          "model_visible_form_verbatim": "cell feature c",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "connector_mediated_embedding",
          "insertion_or_fusion_verbatim": "Q-Former",
          "fusion_topology": "query_bottleneck",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "Stage 1: Cross-modal discriminative pre-training. We conduct cross-modal discriminative pretraining to create a shared latent space that captures semantic correspondences between scRNA-seq profiles and biomedical texts. Given a normalized single-cell expression vector ˜ x ( i ) ∈ R M , the scLLM produces a contextualized embedding h cell = scGPT ( ˜ x ( i ) ) . This is passed through a Qformer to yield the cell feature c = QFormer ( h cell ) . Similarly, for a textual description represented by the token sequence t ( i ) = { t 1 , . . . , t L } , we extract the text embedding via the BERT module within the Q-Former as h text = BERT ( t ( i ) ) .",
          "section_heading": "3.3.2 Cross-Modal Pre-training for Cell-Text Integration",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5,
            6,
            7,
            20
          ],
          "doc_item_refs": [
            "#/pictures/8",
            "#/texts/211",
            "#/texts/213",
            "#/texts/314",
            "#/texts/315",
            "#/texts/316",
            "#/texts/317",
            "#/texts/318"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001381::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001381::0073",
            "dense::full_2026-07-06__rec_001381::0074",
            "dense::full_2026-07-06__rec_001381::0075"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e85e6aec8370",
          "configuration_id": "config_a2cb5fda0aed",
          "route_label": "scMMGPT text-to-text discrimination",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cross-modal discriminative pretraining",
          "source_object_verbatim": "textual description of a cell",
          "source_object_normalized": "textual description of a cell",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenize the textual description into a sequence of tokens",
            "BERT module within the Q-Former"
          ],
          "model_visible_form_verbatim": "text token sequence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "BERT module within the Q-Former",
          "fusion_topology": "query_bottleneck",
          "text_role": "paired_alignment_supervision",
          "input_status": "actual_model_input",
          "evidence_quote": "Stage 1: Cross-modal discriminative pre-training. We conduct cross-modal discriminative pretraining to create a shared latent space that captures semantic correspondences between scRNA-seq profiles and biomedical texts. Given a normalized single-cell expression vector ˜ x ( i ) ∈ R M , the scLLM produces a contextualized embedding h cell = scGPT ( ˜ x ( i ) ) . This is passed through a Qformer to yield the cell feature c = QFormer ( h cell ) . Similarly, for a textual description represented by the token sequence t ( i ) = { t 1 , . . . , t L } , we extract the text embedding via the BERT module within the Q-Former as h text = BERT ( t ( i ) ) .",
          "section_heading": "3.3.2 Cross-Modal Pre-training for Cell-Text Integration",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            5,
            6
          ],
          "doc_item_refs": [
            "#/pictures/8",
            "#/texts/211",
            "#/texts/213",
            "#/texts/314"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001381::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0e7af367b8f6",
          "configuration_id": "config_11f0ba65eaae",
          "route_label": "scMMGPT batch correction and clustering",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Batch Effect Correction and Cell Clustering (§4.1.2).",
          "source_object_verbatim": "a set of cells { x ( i ) }",
          "source_object_normalized": "cells",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "zero-shot cell feature extraction",
            "clustering"
          ],
          "model_visible_form_verbatim": "c ( i ) = scGPT ( x ( i ) )",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "we obtain their representations as c ( i ) = scGPT ( x ( i ) )",
          "fusion_topology": "unclear",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "given a set of cells { x ( i ) }",
          "section_heading": "3.4 Adapting scMMGPT to Downstream Tasks",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "Inference-stage feature extraction for clustering and batch-effect correction.",
          "pages": [
            7
          ],
          "doc_item_refs": [
            "#/texts/330"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001381::0079"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_433657cba128"
    },
    {
      "model_id": "model_5fd880eafa1a",
      "model_name": "scMOBA",
      "record_id": "full_2026-07-06__rec_001889",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_d28c97a665ce",
      "paper_title": "scMOBA: A conversational single-cell Multi-Omics Brain Agent across species",
      "doi": "10.64898/2025.12.01.691565",
      "paper_url": "https://doi.org/10.64898/2025.12.01.691565",
      "route_count": 13,
      "configuration_count": 9,
      "family_counts": {
        "discrete_biological_symbol_stream": 12,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "native_biological_token_stream": 12,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "discrete_biological_symbol_stream"
      ],
      "subtypes": [
        "native_biological_token_stream",
        "plain_language_prompt_or_question"
      ],
      "primary_subtype": "native_biological_token_stream",
      "modalities": [
        "chromatin accessibility",
        "single-cell transcriptomics",
        "spatial transcriptomics",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "cross_attention"
      ],
      "text_roles": [
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001889_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001889_415cb5daeaca/figure_001.png",
        "figure_index": 1,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific figure labeled **a–d** describing a multimodal, multispecies biological foundation model workflow for brain single-cell/spatial data.\n\n**Panel a:** Circular multilayer summary plot of dataset composition. Rings encode categories including **Modality**, **Species**, **Age group**, **Region**, and **Disease**. Visible labels include:\n- Modalities: epigenomics, transcriptomics, spatial, transcriptomics\n- Species: mouse, marmoset, macaque, human\n- Age groups: prenatal, juvenile, adult, aged\n- Brain regions: entorhinal cortex, frontal cortex, insular cortex, occipital cortex, parietal cortex, retrosplenial cortex, subcortex, superior temporal cortex, temporal cortex, and other abbreviated regions\n- Disease categories: brain tumors, cognitive dysfunction, epilepsy, immune disorders, neurodegeneration, normal, others, psychiatric\n\n**Panel b:** Word-cloud summaries for metadata fields. Six labeled word clouds show prominent terms for:\n- Question type\n- Class\n- Subclass\n- Region\n- Age group\n- Disease  \nVisible frequent terms include **GABA**, **GL**, **oligo**, **Astro**, **Vip**, **Pvalb**, **cerebral cortex**, **frontal**, **temporal**, **aged**, **Normal**, **disease**, and **Alzheimer**.\n\n**Panel c:** Main model architecture and task interface. Biological inputs are shown as **cell and feature input** from **multi-omics** and **multi-species** sources, represented by molecular/omics icons and species silhouettes including mouse, monkey/primate, and human. The pipeline proceeds through:\n- Tokenization\n- Gene Encoder\n- Projector\n- Graph Convolutional Networks, labeled as **2 layers**\n- Learnable query\n- Cross attention\n- Large language model\n- Text input and text output interface\n\nThe right side shows six example question-answer task panels:\n1. **Cell type annotation:** asks what cell type corresponds to a gene expression pattern; answer says astrocyte.\n2. **Cell origin prediction:** predicts conserved or species-specific cell origin; answer indicates a primate-specific cell type.\n3. **Aging clock:** predicts age from feature abundance; answer gives biological age as 90 years old.\n4. **Disease status inference:** asks whether the cell is involved in Alzheimer’s disease pathogenesis; answer says the cell belongs to an AD-related subpopulation.\n5. **Gene mask prediction:** masked genes are predicted in expression order; visible answer lists **Gad1, Gad2, Atf4, Neurod1, Syt**.\n6. **Spatial mask prediction:** predicts top genes for neighboring spatial spots; visible answer lists **Gad1, Gad2, Atf4, Neurod1, Syt**.\n\n**Panel d:** Training strategy with two stages:\n- **Stage 1:** Cell & feature input passes through a frozen gene encoder, a trainable projector, text input, and a frozen LLM.\n- **Stage 2:** Cell & feature input passes through a trainable gene encoder, trainable projector, text input, and an LLM with LoRA.\n\nOverall, the figure presents dataset metadata distributions, task/category vocabulary, a biological feature-to-language model architecture, example biomedical question-answer tasks, and a two-stage training scheme for integrating gene/cell features with a large language model.",
        "page_no": 40,
        "sha256": "c791d6b708db5831604d55a4260039401bf005797269da852224d598c552e137",
        "pixel_width": 882,
        "pixel_height": 1279,
        "crop_box": {
          "x": 0.0,
          "y": 0.3,
          "width": 0.46,
          "height": 0.44
        },
        "panel_label": "c",
        "visible_input_object": "Cell & Feature Input from multi-omics / multi-species sources",
        "visible_model_interface": "Tokenization -> Gene Encoder -> Projector with Learnable Query and Cross Attention -> LLM",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop isolates the left architecture in panel c, keeping the biological input, the transformation stack, and the fusion/interface path into the LLM while excluding the output-only example tasks and the separate training-strategy panel.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "native_biological_token_stream",
          "family_id": "discrete_biological_symbol_stream",
          "route_id": "route_3cfe128dfd30",
          "example_input": "A C G T G C A ...",
          "example_carrier": "native nucleotide/amino-acid token IDs",
          "example_interface": "biological tokenizer → generator",
          "actual_source": "gene expression profiles",
          "actual_model_visible_form": "feature abundance representations and text-formatted queries"
        },
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_09b4e223ce85",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "corresponding textual questions",
          "actual_model_visible_form": "text-formatted queries"
        }
      ],
      "routes": [
        {
          "route_id": "route_3cfe128dfd30",
          "configuration_id": "config_6cd68fbfc84e",
          "route_label": "snRNA-seq expression profile",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "feature-question-answer pretraining",
          "source_object_verbatim": "gene expression profiles",
          "source_object_normalized": "single-cell gene expression profiles",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "tokenized cell and feature inputs",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual questions",
            "large language model"
          ],
          "model_visible_form_verbatim": "feature abundance representations and text-formatted queries",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "the LLM received and concatenated the gene-derived features with the corresponding textual questions",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The architecture began by processing tokenized cell and feature inputs (e.g., gene expression or snATAC-seq-derived gene activity score) through a dedicated gene encoder.",
          "section_heading": "The architecture of scMOBA and FQA pretraining",
          "supporting_figure_or_table": "Fig. 1c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/16",
            "#/texts/29",
            "#/texts/31"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f1a7508064a3",
          "configuration_id": "config_6cd68fbfc84e",
          "route_label": "snATAC-seq gene activity score",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "feature-question-answer pretraining",
          "source_object_verbatim": "snATAC-seq-derived gene activity score",
          "source_object_normalized": "snATAC-seq-derived gene activity scores",
          "source_modality_normalized": "chromatin accessibility",
          "transformation_chain_verbatim": [
            "tokenized cell and feature inputs",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual questions",
            "large language model"
          ],
          "model_visible_form_verbatim": "feature abundance representations and text-formatted queries",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "the LLM received and concatenated the gene-derived features with the corresponding textual questions",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The architecture began by processing tokenized cell and feature inputs (e.g., gene expression or snATAC-seq-derived gene activity score) through a dedicated gene encoder.",
          "section_heading": "The architecture of scMOBA and FQA pretraining",
          "supporting_figure_or_table": "Fig. 1c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/16",
            "#/texts/29",
            "#/texts/31"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_926edf3c6519",
          "configuration_id": "config_6cd68fbfc84e",
          "route_label": "spatial transcriptomics profile",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "feature-question-answer pretraining",
          "source_object_verbatim": "processed spatial transcriptomics datasets",
          "source_object_normalized": "processed spatial transcriptomics datasets",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "same preprocessing pipeline as snRNA-seq data",
            "tokenization",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual questions",
            "large language model"
          ],
          "model_visible_form_verbatim": "tokenized feature abundance representations",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "spatial feature representations are mapped into the language space and concatenated with the question text",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We downloaded the processed spatial transcriptomics datasets and applied the same preprocessing pipeline used for the snRNA-seq data.",
          "section_heading": "Spatial transcriptome and preprocessing",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            19
          ],
          "doc_item_refs": [
            "#/texts/16",
            "#/texts/96"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_09b4e223ce85",
          "configuration_id": "config_93bfc5c4c2de",
          "route_label": "textual question prompt",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "question-answer pretraining",
          "source_object_verbatim": "corresponding textual questions",
          "source_object_normalized": "text questions",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "question template selection",
            "tokenization",
            "concatenation with feature abundance representations",
            "large language model"
          ],
          "model_visible_form_verbatim": "text-formatted queries",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "concatenated with the feature abundance representations as input to the LLM",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The LLM received and concatenated both the feature abundance representations and the corresponding textual questions as input",
          "section_heading": "The architecture of scMOBA and FQA pretraining",
          "supporting_figure_or_table": "Fig. 1c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            4,
            5
          ],
          "doc_item_refs": [
            "#/texts/16",
            "#/texts/29",
            "#/texts/31"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2837ee1774c1",
          "configuration_id": "config_fcb24407df50",
          "route_label": "snRNA-seq cell type annotation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "zero-shot cell type annotation",
          "source_object_verbatim": "gene-expression vector",
          "source_object_normalized": "single-cell gene-expression vector",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "tokenization",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual question",
            "large language model"
          ],
          "model_visible_form_verbatim": "feature abundance representations and text-formatted queries",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "the LLM received and concatenated the gene-derived features with the corresponding textual questions",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "For each cell, the model received its gene-expression (or ATAC-derived gene activity) vector and a textual question such as 'What is the cell type of this cell?'.",
          "section_heading": "Downstream tasks",
          "supporting_figure_or_table": "Fig. 1c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            6,
            7
          ],
          "doc_item_refs": [
            "#/texts/16",
            "#/texts/35",
            "#/texts/37"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a4c774fd1b5b",
          "configuration_id": "config_fcb24407df50",
          "route_label": "snATAC-seq cell type annotation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "zero-shot cell type annotation",
          "source_object_verbatim": "ATAC-derived gene activity vector",
          "source_object_normalized": "ATAC-derived gene activity vector",
          "source_modality_normalized": "chromatin accessibility",
          "transformation_chain_verbatim": [
            "tokenization",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual question",
            "large language model"
          ],
          "model_visible_form_verbatim": "feature abundance representations and text-formatted queries",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "the LLM received and concatenated the gene-derived features with the corresponding textual questions",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "For each cell, the model received its gene-expression (or ATAC-derived gene activity) vector and a textual question such as 'What is the cell type of this cell?'.",
          "section_heading": "Downstream tasks",
          "supporting_figure_or_table": "Fig. 1c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            6,
            7
          ],
          "doc_item_refs": [
            "#/texts/16",
            "#/texts/35",
            "#/texts/37"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d9808b3f41b5",
          "configuration_id": "config_fcb24407df50",
          "route_label": "spatial transcriptome cell type annotation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "zero-shot cell type annotation",
          "source_object_verbatim": "spatial transcriptome profile",
          "source_object_normalized": "spatial transcriptome profile",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "same preprocessing pipeline as snRNA-seq data",
            "tokenization",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual question",
            "large language model"
          ],
          "model_visible_form_verbatim": "tokenized feature abundance representations",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "spatial feature representations are mapped into the language space and concatenated with the question text",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "scMOBA also accurately predicted cell types for the snATAC-seq dataset and the two spatial transcriptome datasets from human and mouse brain.",
          "section_heading": "Fine-grained cell type prediction",
          "supporting_figure_or_table": "Fig. 2e",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            6,
            7,
            8
          ],
          "doc_item_refs": [
            "#/texts/16",
            "#/texts/35",
            "#/texts/37",
            "#/texts/38",
            "#/texts/40"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6c33fca90b79",
          "configuration_id": "config_7e452383cac4",
          "route_label": "cell origin prediction",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cell origin prediction",
          "source_object_verbatim": "gene-expression profile",
          "source_object_normalized": "single-cell gene-expression profile",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "tokenization",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual question",
            "large language model"
          ],
          "model_visible_form_verbatim": "feature abundance representations and text-formatted queries",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "the LLM received and concatenated the gene-derived features with the corresponding textual questions",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The architecture of scMOBA offered flexible downstream tasks suitable for facile cell type annotation, cell origin prediction, aging clock construction and disease status inference.",
          "section_heading": "The architecture of scMOBA and FQA pretraining",
          "supporting_figure_or_table": "Fig. 1c",
          "evidence_status": "explicit_text",
          "uncertainty": "Named as a supported downstream task in the architecture section, but not isolated as a separate benchmark subsection.",
          "pages": [
            1,
            4,
            5,
            40
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/16",
            "#/texts/29",
            "#/texts/31"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_12d12f1ba884",
          "configuration_id": "config_80d44a9b554e",
          "route_label": "disease status inference",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "disease status inference",
          "source_object_verbatim": "gene-expression profile",
          "source_object_normalized": "single-cell gene-expression profile",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "tokenization",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual question",
            "large language model"
          ],
          "model_visible_form_verbatim": "feature abundance representations and text-formatted queries",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "the LLM received and concatenated the gene-derived features with the corresponding textual questions",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "The architecture of scMOBA offered flexible downstream tasks suitable for facile cell type annotation, cell origin prediction, aging clock construction and disease status inference.",
          "section_heading": "The architecture of scMOBA and FQA pretraining",
          "supporting_figure_or_table": "Fig. 1c",
          "evidence_status": "explicit_text",
          "uncertainty": "Named as a supported downstream task in the architecture section, but not isolated as a separate benchmark subsection.",
          "pages": [
            1,
            4,
            5,
            40
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/16",
            "#/texts/29",
            "#/texts/31"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_db10859c8abb",
          "configuration_id": "config_62ed1825501a",
          "route_label": "human snRNA-seq aging clock",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "aging clock construction",
          "source_object_verbatim": "aged human prefrontal cortex snRNA-seq data",
          "source_object_normalized": "aged human prefrontal cortex snRNA-seq data",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "train/test split",
            "gene-expression preprocessing",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual question",
            "large language model"
          ],
          "model_visible_form_verbatim": "feature abundance representations and text-formatted queries",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "the LLM received and concatenated the gene-derived features with the corresponding textual questions",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "We retrieved a unseen snRNA-seq data from the aged human prefrontal cortex",
          "section_heading": "Accurate prediction of cellular aging with scMOBA",
          "supporting_figure_or_table": "Fig. 5a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            12
          ],
          "doc_item_refs": [
            "#/texts/16",
            "#/texts/56",
            "#/texts/57"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_010"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b41e139a6df2",
          "configuration_id": "config_3d89c24f9175",
          "route_label": "mouse spatial MERFISH aging clock",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "aging clock construction",
          "source_object_verbatim": "mouse coronal MERFISH dataset",
          "source_object_normalized": "mouse coronal MERFISH dataset",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "spatial smoothing",
            "feature preprocessing",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual question",
            "large language model"
          ],
          "model_visible_form_verbatim": "tokenized feature abundance representations",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "spatial feature representations are mapped into the language space and concatenated with the question text",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "In the mouse coronal MERFISH dataset, scMOBA predicted ages showed an average of 0.92 correlation coefficient with the actual ages of mice across 14 cell types",
          "section_heading": "Accurate prediction of cellular aging with scMOBA",
          "supporting_figure_or_table": "Fig. 5f",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            13,
            14
          ],
          "doc_item_refs": [
            "#/texts/16",
            "#/texts/59",
            "#/texts/61"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_011"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d741b667bdc9",
          "configuration_id": "config_3f9f9e383486",
          "route_label": "masked gene reconstruction",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "masked gene reconstruction",
          "source_object_verbatim": "a subset of genes within each cell",
          "source_object_normalized": "subset of genes within each cell",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "random masking of gene tokens",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual question",
            "large language model"
          ],
          "model_visible_form_verbatim": "masked gene tokens and feature abundance representations",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "masked gene features are passed through the gene encoder and then fused with the textual query",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "a subset of genes within each cell is randomly masked, and the model is trained to predict their expression values.",
          "section_heading": "Construction of question-answer pairs",
          "supporting_figure_or_table": "Fig. 1c",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            1,
            14,
            15
          ],
          "doc_item_refs": [
            "#/texts/16",
            "#/texts/63",
            "#/texts/64",
            "#/texts/66"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_012"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fcd011353b68",
          "configuration_id": "config_6eef6e2a23f7",
          "route_label": "spatial mask prediction",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "spatial prediction task",
          "source_object_verbatim": "expression profiles of neighboring spots",
          "source_object_normalized": "neighboring spatial expression profiles",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "context from neighboring spots",
            "gene encoder",
            "projector",
            "cross-attention",
            "concatenation with textual question",
            "large language model"
          ],
          "model_visible_form_verbatim": "tokenized feature abundance representations",
          "carrier_family": "discrete_biological_symbol_stream",
          "carrier_subtype": "native_biological_token_stream",
          "insertion_or_fusion_verbatim": "spatial feature representations are mapped into the language space and concatenated with the question text",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "the expression profiles of neighboring spots are provided as context, and the model predicts highly expressed genes in the central cell.",
          "section_heading": "Construction of question-answer pairs",
          "supporting_figure_or_table": "Fig. 1c",
          "evidence_status": "explicit_text",
          "uncertainty": "Discovery provenance was figure-only, but the Methods text independently confirms the neighboring-spot context route.",
          "pages": [
            1,
            40
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/16"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001889::route_013"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001889::0001"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_fdbd3ba264a9"
    },
    {
      "model_id": "model_d3d6352d5bcb",
      "model_name": "Shusi",
      "record_id": "full_2026-07-06__rec_001200",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_ceeeabf3583a",
      "paper_title": "Systematic discovery of single-cell protein networks in cancer with Shusi",
      "doi": "10.1101/2025.04.27.649905",
      "paper_url": "https://doi.org/10.1101/2025.04.27.649905",
      "route_count": 6,
      "configuration_count": 5,
      "family_counts": {
        "geometric_or_diffusion_state_carrier": 4,
        "dense_continuous_carrier": 2
      },
      "subtype_counts": {
        "symbolic_structural_constraint": 4,
        "direct_projected_embedding": 2
      },
      "families": [
        "dense_continuous_carrier",
        "geometric_or_diffusion_state_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "symbolic_structural_constraint"
      ],
      "primary_subtype": "symbolic_structural_constraint",
      "modalities": [
        "functional text",
        "protein sequence text",
        "protein-protein interaction network",
        "single-cell RNA sequencing"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "encoder_decoder"
      ],
      "text_roles": [
        "biological_payload",
        "no_text_on_this_route",
        "semantic_annotation"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001200_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001200_07a2a3f7e205/figure_001.png",
        "figure_index": 1,
        "caption": "Fig. 1 Overview of Shusi. a, Pipeline for constructing single-cell protein interaction networks from scRNA-seq data. b, Architecture of the Shusi model for predicting novel interactions on single cell level. d, Fine-tuning and application of Shusi to AML patient scRNA-seq data.",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel scientific workflow figure with panels labeled **a**, **b**, and **c**.\n\n**Panel a: Data Collection & Processing**\n- Shows inputs for network construction:\n  - **23 cancer types / 234 cell lines** illustrated with human organ/cancer-related icons.\n  - **12,350 proteins** illustrated by a protein ribbon structure.\n  - **scRNA-seq** shown as a t-SNE scatter plot with colored cell clusters.\n  - **Experimentally validated network (STRING-E)** shown as a node-link graph.\n- Transformation arrows indicate combining scRNA-seq data and STRING-E into a stack of constructed networks.\n- Output labeled **75,010 Networks** under **Network Construction**.\n\n**Panel b: Model Training**\n- Left side labeled **Large Language Model**.\n- Inputs include:\n  - **Amino acid sequence** represented by sequence/structure icons.\n  - **Functional description** represented by molecular/annotation icons.\n- These pass through an **LLM** block to produce **Protein Embeddings**.\n- Protein embeddings are used to create **Network per cell**, with example cell-specific graphs labeled **Cell 1**, **Cell 2**, **Cell 3**, and **Cell N**.\n- Right side labeled **Variational Graph Autoencode**.\n- Shows an edge feature formula: **Edgeᵢ = Eᵢ * gᵢ * Iᵢ**.\n- Cell-specific networks feed into a **Graph Isomorphism Network** with encoder, latent variable **Z**, and decoder.\n- Output task is **Link Prediction**, shown as predicted node-link graphs for multiple cells.\n\n**Panel c: Drug Target Identification**\n- Shows a **Shusi Model** fine-tuning workflow using **AML Patient scRNA-seq**, represented by a t-SNE plot and patient icons.\n- Fine-tuned model produces graph/link prediction outputs grouped into colored clusters.\n- These feed into **Target Identification**, shown as a molecular network with drug icons.\n- Final step is **Experimental Validation**, illustrated by a human body/cancer icon and a multiwell plate.\n\nOverall, the figure depicts a computational biology pipeline integrating cancer cell line data, protein information, scRNA-seq, experimentally validated protein networks, language-model-derived protein embeddings, and a variational graph autoencoder/GIN model for cell-specific link prediction and drug target identification, followed by experimental validation.",
        "page_no": 5,
        "sha256": "096db91d754e4ab3949e466a513debdf24ed09c17b9d18d33ee4249263b28c35",
        "pixel_width": 725,
        "pixel_height": 743,
        "crop_box": {
          "x": 0.03,
          "y": 0.4,
          "width": 0.62,
          "height": 0.29
        },
        "panel_label": "b",
        "visible_input_object": "Amino acid sequence and functional description",
        "visible_model_interface": "Large Language Model to protein embeddings, with the start of the cell-specific graph input interface",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the source features, the LLM transformation, the protein-embedding carrier, and the immediate graph-input side needed to ground the pretraining input route, while excluding the output-only decoder/link-prediction side and other panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_1df4fd7ee33b",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "amino acid sequences",
          "actual_model_visible_form": "sequence embeddings / protein embeddings"
        },
        {
          "subtype_id": "symbolic_structural_constraint",
          "family_id": "geometric_or_diffusion_state_carrier",
          "route_id": "route_68c5a50f865a",
          "example_input": "motif anchors + distance constraints",
          "example_carrier": "symbolic geometry/structure constraints",
          "example_interface": "constraint-conditioned generator",
          "actual_source": "75,010 single-cell transcriptomic profiles from 23 cancer types",
          "actual_model_visible_form": "single-cell-specific biological networks"
        }
      ],
      "routes": [
        {
          "route_id": "route_68c5a50f865a",
          "configuration_id": "config_ca8e89be6fba",
          "route_label": "Pan-cancer single-cell graph construction",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "single-cell protein interaction prediction",
          "source_object_verbatim": "75,010 single-cell transcriptomic profiles from 23 cancer types",
          "source_object_normalized": "pan-cancer single-cell transcriptomic profiles",
          "source_modality_normalized": "single-cell RNA sequencing",
          "transformation_chain_verbatim": [
            "integrated with prior knowledge of protein-protein interactions",
            "constructed single-cell-specific networks",
            "applied a variational graph auto-encoder with a graph isomorphism network",
            "link prediction"
          ],
          "model_visible_form_verbatim": "single-cell-specific biological networks",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "graph input to the VGAE-GIN encoder-decoder",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "To enable cell-specific interaction prediction, we integrated 75,010 single-cell transcriptomic profiles from 23 cancer types, generated using two complementary sequencing platforms 29,30 , the experimental-validated STRING database (STRINGE) as reference PPI network 15 (Fig. 1a, Fig. 2a). The reference interactome comprised 12,350 protein-coding genes curated from the STRING database (v11.0) 15,31 , providing experimentally validated interactions, functional annotations, and amino acid sequences. We derived single-cell-specific networks containing an average of 3,681 proteins connected by about 496,179 high-confidence interactions for model training (Methods). This approach allowed us to capture biological heterogeneity at superior resolution, enabling precise identification of context-dependent interactions. Meanwhile, our single-cell network construction approach ensured that the model could handle variable input sizes, accommodating any number of cells while maintaining computational scalability.",
          "section_heading": "The Shusi model",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper’s model-visible object is a graph/network; the frozen taxonomy has no dedicated graph family, so symbolic_structural_constraint is the closest fit.",
          "pages": [
            3,
            4,
            19,
            20
          ],
          "doc_item_refs": [
            "#/texts/1274",
            "#/texts/1280",
            "#/texts/1281",
            "#/texts/1282",
            "#/texts/1283",
            "#/texts/1284",
            "#/texts/1285",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/27",
            "#/texts/28"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001200::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001200::0012",
            "dense::full_2026-07-06__rec_001200::0014"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1df4fd7ee33b",
          "configuration_id": "config_4ff6c3f2b5bd",
          "route_label": "Amino acid sequence embeddings",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "feature extraction module",
          "source_object_verbatim": "amino acid sequences",
          "source_object_normalized": "protein sequences",
          "source_modality_normalized": "protein sequence text",
          "transformation_chain_verbatim": [
            "Sentence-BERT",
            "protein embeddings",
            "concatenating the embeddings generated by both encoders",
            "node embeddings in the graph"
          ],
          "model_visible_form_verbatim": "sequence embeddings / protein embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "concatenated with functional-description embeddings",
          "fusion_topology": "concatenation",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Sentence-BERT 32 effectively encoded amino acid sequences",
          "section_heading": "The Shusi model",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            20,
            21
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/143",
            "#/texts/146",
            "#/texts/147",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/27",
            "#/texts/28",
            "#/texts/29",
            "#/texts/32"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001200::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001200::0001",
            "dense::full_2026-07-06__rec_001200::0016"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b63107a7a758",
          "configuration_id": "config_4ff6c3f2b5bd",
          "route_label": "Functional-description embeddings",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "feature extraction module",
          "source_object_verbatim": "functional descriptions",
          "source_object_normalized": "functional descriptions",
          "source_modality_normalized": "functional text",
          "transformation_chain_verbatim": [
            "Galactica",
            "protein embeddings",
            "concatenating the embeddings generated by both encoders",
            "node embeddings in the graph"
          ],
          "model_visible_form_verbatim": "functional-description embeddings / protein embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "concatenated with amino-acid-sequence embeddings",
          "fusion_topology": "concatenation",
          "text_role": "semantic_annotation",
          "input_status": "actual_model_input",
          "evidence_quote": "the functional descriptions were encoded using Galactica 33",
          "section_heading": "The Shusi model",
          "supporting_figure_or_table": "Fig. 1",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            5,
            6,
            7,
            20,
            21
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/143",
            "#/texts/146",
            "#/texts/147",
            "#/texts/22",
            "#/texts/23",
            "#/texts/24",
            "#/texts/27",
            "#/texts/28",
            "#/texts/29",
            "#/texts/32"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001200::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001200::0001",
            "dense::full_2026-07-06__rec_001200::0016"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e3362f03567b",
          "configuration_id": "config_e87a089c0b5c",
          "route_label": "Primary AML single-cell graph fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "fine-tuning Shusi for AML datasets",
          "source_object_verbatim": "scRNA-seq profiles from 32,146 cells across 27 primary AML specimens",
          "source_object_normalized": "primary AML single-cell RNA-seq profiles",
          "source_modality_normalized": "single-cell RNA sequencing",
          "transformation_chain_verbatim": [
            "constructed AML-specific single-cell protein networks",
            "computed single-cell network degree matrices",
            "applied topological filtering for monocyte-specific target identification"
          ],
          "model_visible_form_verbatim": "AML cell graphs / degree matrices",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "fine-tuning on AML cell graphs",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "we first fine-tuned Shusi using scRNA-seq profiles from 32,146 cells across 27 primary AML patients",
          "section_heading": "Fine-tuning Shusi for AML datasets",
          "supporting_figure_or_table": "Fig. 6",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper’s fine-tuning input is a graph/network representation; the frozen taxonomy has no dedicated graph family, so symbolic_structural_constraint is the closest fit.",
          "pages": [
            14,
            15,
            16,
            17,
            20,
            24
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/1278",
            "#/texts/1342",
            "#/texts/1343",
            "#/texts/1344",
            "#/texts/981",
            "#/texts/984",
            "#/texts/985",
            "#/texts/986",
            "#/texts/987",
            "#/texts/990",
            "#/texts/993",
            "#/texts/994"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001200::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001200::0007",
            "dense::full_2026-07-06__rec_001200::0008",
            "dense::full_2026-07-06__rec_001200::0013",
            "dense::full_2026-07-06__rec_001200::0024"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5de733be5449",
          "configuration_id": "config_4f143b5265d5",
          "route_label": "STRING-E reference PPI graph",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "single-cell network construction for link prediction",
          "source_object_verbatim": "experimentally validated interactions from the STRING database (STRING-E)",
          "source_object_normalized": "STRING-E human physical protein interaction network",
          "source_modality_normalized": "protein-protein interaction network",
          "transformation_chain_verbatim": [
            "filtered to high-confidence physical associations",
            "used as prior knowledge of protein-protein interactions",
            "combined with gene expression profiles to compute edge weights",
            "constructed single-cell-specific interaction networks"
          ],
          "model_visible_form_verbatim": "reference interaction graph / weighted PPI edges",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "prior knowledge fused into the graph input",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "The reference PPI network was constructed using experimentally validated interactions from the STRING database (STRING-E)",
          "section_heading": "Reference human physical protein interaction network",
          "supporting_figure_or_table": "Methods",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper models a network prior entering a graph model; the frozen taxonomy has no graph-specific leaf, so symbolic_structural_constraint is the closest fit.",
          "pages": [
            19
          ],
          "doc_item_refs": [
            "#/texts/1272"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001200::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001200::0011"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_c75010de41e1",
          "configuration_id": "config_e01a80179e5c",
          "route_label": "Held-out single-cell graph evaluation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "benchmark analysis against baseline models",
          "source_object_verbatim": "15,010 single cells held out as an independent test set",
          "source_object_normalized": "held-out single-cell graph benchmark set",
          "source_modality_normalized": "single-cell RNA sequencing",
          "transformation_chain_verbatim": [
            "randomly partitioned as an independent test set",
            "masked 20% of protein interactions in each cellular network",
            "used held-out edges for prediction evaluation"
          ],
          "model_visible_form_verbatim": "held-out cell graphs with masked interactions",
          "carrier_family": "geometric_or_diffusion_state_carrier",
          "carrier_subtype": "symbolic_structural_constraint",
          "insertion_or_fusion_verbatim": "evaluation-time graph input to the same VGAE-GIN predictor",
          "fusion_topology": "encoder_decoder",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "randomly partitioned 15,010 single cells (20% of the total dataset) as an independent test set",
          "section_heading": "Performance evaluations of Shusi",
          "supporting_figure_or_table": "Fig. 2",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The evaluation split is still a graph/network carrier, but the frozen taxonomy has no dedicated graph family; symbolic_structural_constraint is the closest fit.",
          "pages": [
            5,
            6,
            7
          ],
          "doc_item_refs": [
            "#/texts/143",
            "#/texts/146",
            "#/texts/147",
            "#/texts/148",
            "#/texts/151"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_001200::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_2e20f0c5603e"
    },
    {
      "model_id": "model_6a1d79c2e81c",
      "model_name": "TeamPath-7B",
      "record_id": "full_2026-07-06__rec_003323",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_27fd0c39d9c3",
      "paper_title": "Teampath: Building multimodal pathology experts with reasoning ai copilots",
      "doi": "",
      "paper_url": "",
      "route_count": 9,
      "configuration_count": 5,
      "family_counts": {
        "visual_raster_carrier": 4,
        "text_native_token_stream": 5
      },
      "subtype_counts": {
        "raw_slide_or_patch_input": 4,
        "structured_biological_prompt_or_task_scaffold": 4,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "visual_raster_carrier"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "raw_slide_or_patch_input",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "histopathology image",
        "text"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "inference"
      ],
      "fusion_topologies": [
        "concatenation",
        "encoder_decoder",
        "tokenizer_sequence"
      ],
      "text_roles": [
        "instruction_or_query",
        "metadata_or_context"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003323_figure_002.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003323_ae216fea4cc3/figure_002.png",
        "figure_index": 2,
        "caption": "Figure 1: Landscape of TeamPath (a) Steps of dataset curation. We extract image-text pairs from a processed TCGA dataset (PathGen-1.6M). (b) Word cloud visualization of ROI captions (upper) and questions (bottom). (c) The core visual language model architecture of TeamPath. (d) TeamPath as a system with an LLM-enhanced router (with over 80% accuracy in choosing the correct approach) and the corresponding capacities in various downstream applications. The logo fire means that we need to adjust the parameters of models, and the logo snowflake means that we do not change the parameters. (e) Overall ranking list of different methods across tasks and metrics. A lower rank (larger bubble) means a better method.",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel scientific figure about pathology vision-language model dataset curation, training, and evaluation.\n\nPanel **a**: Workflow schematic starting from a **whole-slide histopathology image** and zoomed **regions of interest**. It shows data retrieval and agent summary steps, then PathGen-1.6M / PathInstruct-Reason style instruction data generation, including QA pairs, transcriptomic pairs, and QA pairs with reasoning. A dataset statistics box lists HEST-1K/STImage1K4M multi-omic datasets and measured genes.\n\nPanel **c**: VLM training architecture diagram. It includes an **Expert Vis Encoder**, **Expert Lang Encoder**, and **Expert Lang Decoder**, with pathology image patches/tokens and a text prompt leading to an answer output.\n\nPanel **d**: TeamPath router-based multi-task system. A task-specific instruction and pathology image region are sent to an SFT-trained router, which routes to Expert A, Expert B, or Expert C with RL/SFT/TTS indicators. The right box lists downstream tasks: spatial transcriptomic generation, reasoning-driven VQA, reasoning-driven summary, and AI-physician collaboration.\n\nPanel **e**: Bubble plot comparing models across pathology VQA, pathology summary, and generation metrics. Rows include TeamPath-7B, PathGen-LLaVA-13B, Patho-R1-7B, Qwen2.5VL-7B, MedGemma-4B, InternVL3-8B, Qwen2.5VL-3B, and MedVLThinker-7B. Columns include PubMed, SocialPath, Atlas, EduContent, PathCLS, BLEU, ROUGE-1/2/L, BERT, MEDCON, SPCC, GPCC, MSE, and AvgRank. Bubble size encodes rank from 1 to 8, with TeamPath-7B visually strongest across many metrics.",
        "page_no": 6,
        "sha256": "fde2882e75e7e8884513794d0ee1f9af9cc06d7702afbe6d0c71f5850f4294fa",
        "pixel_width": 923,
        "pixel_height": 962,
        "crop_box": {
          "x": 0.026,
          "y": 0.272,
          "width": 0.456,
          "height": 0.275
        },
        "panel_label": "c",
        "visible_input_object": "pathology image patches/tokens and text prompt tokens",
        "visible_model_interface": "Expert Vis Encoder and Expert Lang Encoder feeding the Expert Lang Decoder",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop isolates panel c, where the readable labels and arrows show the actual input routes into the VLM: pathology image patches/tokens and text prompt tokens enter the Expert Vis Encoder / Expert Lang Encoder, then flow into the Expert Lang Decoder. It excludes the output-only panel e and the unrelated curation/router panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_1b204f10d57b",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "summarization prompt",
          "actual_model_visible_form": "tokenized summarization prompt"
        },
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_f3c4de37c76d",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "pathology image",
          "actual_model_visible_form": "pathology image patches/tokens"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_7de6129d042f",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "question prompt",
          "actual_model_visible_form": "tokenized text prompt"
        }
      ],
      "routes": [
        {
          "route_id": "route_f3c4de37c76d",
          "configuration_id": "config_3c89875f2614",
          "route_label": "TeamPath visual route for Pathology VQA",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Pathology VQA",
          "source_object_verbatim": "pathology image",
          "source_object_normalized": "pathology image",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [
            "image preprocessing",
            "patch/token extraction",
            "visual encoding"
          ],
          "model_visible_form_verbatim": "pathology image patches/tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "fed through the Expert Vis Encoder",
          "fusion_topology": "encoder_decoder",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "accepts text prompts u1D447 and pathology image u1D443 as inputs.",
          "section_heading": "4. Methods",
          "supporting_figure_or_table": "Figure 1(c)",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            6,
            7,
            15,
            16
          ],
          "doc_item_refs": [
            "#/pictures/1",
            "#/texts/114",
            "#/texts/220",
            "#/texts/221",
            "#/texts/222",
            "#/texts/223",
            "#/texts/224",
            "#/texts/225",
            "#/texts/606",
            "#/texts/607",
            "#/texts/610",
            "#/texts/611",
            "#/texts/612",
            "#/texts/613",
            "#/texts/614",
            "#/texts/615"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003323::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003323::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_7de6129d042f",
          "configuration_id": "config_3c89875f2614",
          "route_label": "TeamPath text route for Pathology VQA",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "Pathology VQA",
          "source_object_verbatim": "question prompt",
          "source_object_normalized": "question prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenization",
            "language encoding"
          ],
          "model_visible_form_verbatim": "tokenized text prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed through the Expert Lang Encoder",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Your task: 1. Think through the question step by step, enclose your reasoning process in <think>...</think> tags. 2. Then provide the correct single-letter choice (A, B, C, D,...) inside <answer>...</answer> tags. 3. No extra information or text outside of these tags.",
          "section_heading": "A. Prompt list",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            26,
            27
          ],
          "doc_item_refs": [
            "#/texts/946",
            "#/texts/947",
            "#/texts/948",
            "#/texts/949",
            "#/texts/950",
            "#/texts/951",
            "#/texts/952",
            "#/texts/953"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003323::route_002"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2c2128dcddb3",
          "configuration_id": "config_fe46364f26ee",
          "route_label": "TeamPath visual route for caption summary",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "caption summary",
          "source_object_verbatim": "histopathology image",
          "source_object_normalized": "histopathology image",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [
            "image preprocessing",
            "patch/token extraction",
            "visual encoding"
          ],
          "model_visible_form_verbatim": "histopathology image patches/tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "fed through the Expert Vis Encoder",
          "fusion_topology": "encoder_decoder",
          "text_role": "instruction_or_query",
          "input_status": "paired_alignment_input",
          "evidence_quote": "You are also given a question and an analysis for the question. Your job is to outline your step-by-step thought process for deriving a correct solution and also write down the correct solution. Using this format: <think> Your step-by-step reasoning of the question and solution < / think><answer> Your final answer < / answer> Question: question Solution: out_verifier.",
          "section_heading": "2. Results",
          "supporting_figure_or_table": "Figure 5",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper uses this route in both training and testing; I anchored it to the paired finetuning input side.",
          "pages": [
            12,
            26,
            27
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/946",
            "#/texts/947",
            "#/texts/948",
            "#/texts/949",
            "#/texts/950",
            "#/texts/951",
            "#/texts/952"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003323::route_003"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_1b204f10d57b",
          "configuration_id": "config_fe46364f26ee",
          "route_label": "TeamPath text route for caption summary",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "caption summary",
          "source_object_verbatim": "summarization prompt",
          "source_object_normalized": "summarization prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "tokenization",
            "language encoding"
          ],
          "model_visible_form_verbatim": "tokenized summarization prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "fed through the Expert Lang Encoder",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Your task: 1. Think through the question step by step, enclose your reasoning process in <think>...</think> tags. 2. Then provide the correct single-letter choice (A, B, C, D,...) inside <answer>...</answer> tags. 3. No extra information or text outside of these tags.",
          "section_heading": "A. Prompt list",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The prompt family is also reused in the paper's caption-summary setup; I anchored it to the finetuning-side prompt input.",
          "pages": [
            26,
            27
          ],
          "doc_item_refs": [
            "#/texts/946",
            "#/texts/947",
            "#/texts/948",
            "#/texts/949",
            "#/texts/950",
            "#/texts/951",
            "#/texts/952",
            "#/texts/953"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003323::route_004"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4487dad29d73",
          "configuration_id": "config_72100b5b5b86",
          "route_label": "TeamPath visual route for cross-modality generation",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "spatial transcriptomic profile generation",
          "source_object_verbatim": "histopathology image ROI",
          "source_object_normalized": "histopathology image ROI",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [
            "ROI extraction",
            "image preprocessing",
            "patch/token extraction",
            "visual encoding"
          ],
          "model_visible_form_verbatim": "ROI image patches/tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "fed through the Expert Vis Encoder",
          "fusion_topology": "encoder_decoder",
          "text_role": "instruction_or_query",
          "input_status": "paired_alignment_input",
          "evidence_quote": "paired histopathology images and transcriptomic profiles generated with the Visium technology",
          "section_heading": "2. Results",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "text_plus_figure",
          "uncertainty": "The paper uses this route in both training and evaluation; I anchored it to the finetuning-side paired-input configuration.",
          "pages": [
            12,
            13
          ],
          "doc_item_refs": [
            "#/pictures/5",
            "#/texts/488",
            "#/texts/489",
            "#/texts/490",
            "#/texts/491",
            "#/texts/492",
            "#/texts/493",
            "#/texts/494"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003323::route_005"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fd38a0d4631c",
          "configuration_id": "config_72100b5b5b86",
          "route_label": "TeamPath text route for cross-modality generation",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "spatial transcriptomic profile generation",
          "source_object_verbatim": "gene-expression prompt",
          "source_object_normalized": "gene-expression prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "prompt construction from ranked genes",
            "tokenization",
            "language encoding"
          ],
          "model_visible_form_verbatim": "text prompt describing ranked genes / spot sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed through the Expert Lang Encoder",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Generate a list of 100 genes in order of descending expression from one spot shown in the histopathology image in IDC disease. Cell sentence:",
          "section_heading": "A. Prompt list",
          "supporting_figure_or_table": "Figure 6",
          "evidence_status": "explicit_text",
          "uncertainty": "The prompt family is used in the cross-modality generation training setup and also referenced for evaluation; I anchored it to the finetuning-side instruction input.",
          "pages": [
            27
          ],
          "doc_item_refs": [
            "#/texts/983"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003323::route_006"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_46dffe37ada4",
          "configuration_id": "config_6a1be0cac103",
          "route_label": "TeamPath visual route for verifier-corrector collaboration",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "self-verification/correction pipeline",
          "source_object_verbatim": "pathology image",
          "source_object_normalized": "pathology image",
          "source_modality_normalized": "histopathology image",
          "transformation_chain_verbatim": [
            "image preprocessing",
            "patch/token extraction",
            "visual encoding"
          ],
          "model_visible_form_verbatim": "pathology image patches/tokens",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "fed into the verifier-corrector pipeline together with the human answer and reasoning path",
          "fusion_topology": "encoder_decoder",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Using TeamPath as an AI Copilot to help pathologists. To formalize our method as an AI Assistant, we consider two case studies inspired by the Path VQA experiments with pathologists. We invite the pathologists to answer 50 questions extracted from PathMMU from the five categories, and record their answers as well as reasoning steps. Our first case is a verifier-corrector pipeline, which can detect the incorrect answers made by pathologists and generate the correct answers. Our pipeline utilizes one verifier (a VLM, default as o4-mini) to verify whether the answers and questions proposed by pathologists are correct or not. If it is justified as wrong, we will call the corrector (also a VLM, default as TeamPath used for Pathology VQA) to fix it. Otherwise, the correct answer will be returned. We have a specific threshold to limit the number of epochs in this loop. Our algorithm is summarized in Algorithm 1. We define the success of a self-verification/correction system as follows: if the expert answer is correct, or the expert answer is wrong but the answer produced by this system is correct.",
          "section_heading": "2. Results",
          "supporting_figure_or_table": "Figure 4",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            17,
            18
          ],
          "doc_item_refs": [
            "#/texts/673",
            "#/texts/674",
            "#/texts/675",
            "#/texts/676",
            "#/texts/679",
            "#/texts/680",
            "#/texts/681",
            "#/texts/682"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003323::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_fb28f16be37b",
          "configuration_id": "config_6a1be0cac103",
          "route_label": "TeamPath text route for verifier-corrector collaboration",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "self-verification/correction pipeline",
          "source_object_verbatim": "question, human answer, and reasoning path",
          "source_object_normalized": "question-answer-reasoning context",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "concatenation",
            "tokenization",
            "language encoding"
          ],
          "model_visible_form_verbatim": "concatenated question-answer-reasoning text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed into the verifier-corrector pipeline as textual context",
          "fusion_topology": "concatenation",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Question u1D444 u1D440 , pathology image u1D43C u1D446 , human answer u1D442 u1D434 , reasoning path u1D442 u1D445",
          "section_heading": "4. Methods",
          "supporting_figure_or_table": "Algorithm 1",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            17,
            18
          ],
          "doc_item_refs": [
            "#/texts/673",
            "#/texts/674",
            "#/texts/675",
            "#/texts/676",
            "#/texts/679",
            "#/texts/680",
            "#/texts/681",
            "#/texts/682",
            "#/texts/698",
            "#/texts/699"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003323::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003323::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_22f773f4ee00",
          "configuration_id": "config_a7d6fa3454ba",
          "route_label": "TeamPath self-corrector text route",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "self-correction",
          "source_object_verbatim": "question and analysis",
          "source_object_normalized": "question and analysis",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "concatenation",
            "tokenization",
            "language encoding"
          ],
          "model_visible_form_verbatim": "question-and-analysis text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "fed through the self-corrector prompt",
          "fusion_topology": "concatenation",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "You are also given a question and an analysis for the question.",
          "section_heading": "A. Prompt list",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            26,
            27
          ],
          "doc_item_refs": [
            "#/texts/946",
            "#/texts/947",
            "#/texts/948",
            "#/texts/949",
            "#/texts/950",
            "#/texts/951",
            "#/texts/952",
            "#/texts/953"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003323::route_010"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_7ec27bf438d4"
    },
    {
      "model_id": "model_b4a92e5e540d",
      "model_name": "text LLM",
      "record_id": "full_2026-07-06__rec_001381",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_d0e026b8e9c9",
      "paper_title": "Language-Enhanced Representation Learning for Single-Cell Transcriptomics",
      "doi": "",
      "paper_url": "",
      "route_count": 1,
      "configuration_count": 1,
      "family_counts": {
        "dense_continuous_carrier": 1
      },
      "subtype_counts": {
        "direct_projected_embedding": 1
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "dense continuous embedding"
      ],
      "lifecycle_phases": [
        "fine_tuning"
      ],
      "fusion_topologies": [
        "side_or_generative_conditioning"
      ],
      "text_roles": [
        "generated_output"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_001381_figure_007.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_001381_9e666a0934af/figure_007.png",
        "figure_index": 7,
        "caption": "Figure 3: Overview of scMMGPT. (1) Cross-modal Discriminative Objective: Given paired cell and text inputs, the model learns to identify the correct textual description of a cell by aligning the outputs of the scLLM and text LLM. (2, 3) Cross-modal Generative Objectives: scMMGPT strengthens multimodal alignment through a unified generative pre-training strategy, jointly optimizing cell-to-text and text-to-cell translation tasks to facilitate bidirectional knowledge transfer.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic workflow for multimodal cell information modeling using single-cell RNA-seq data and text.\n\nVisible components:\n\n- Left panel: “Multimodal Cell Information”\n  - Shows a human body/lung illustration with arrows labeled “scRNA-seq” and “Biological Databases.”\n  - Shows an “Experimental Evidence in scRNA-seq Matrices” block: a heatmap-like matrix with rows labeled Cell 1, Cell 2, …, Cell N and columns labeled Gene 1, Gene 2, …, Gene M.\n  - Shows “Domain Knowledge in Textual Descriptions”: an example text box describing a “classical monocyte,” sourced from human lung female donor samples.\n\n- Middle panel: “scRNA-seq Data”\n  - Gene expression values are transformed into “Gene Tokens.”\n  - Example gene-expression bar/sequence includes gene names such as SEC23B, MT-CO2, RPL37, MT-CYB and values like 127, 3, 18, 69.\n  - Text descriptions are transformed into a “Text Token Sequence,” with tokens such as “[BOS], This, cell, is, a, …, [EOS].”\n\n- Main model panel:\n  - A blue “scLLM” block processes gene tokens.\n  - An orange “Text LLM” block processes text tokens.\n  - Cross-modal projectors connect the two modalities:\n    - “Q-former Projector”\n    - “Cross-Attn Projection”\n  - Intermediate feature boxes are labeled “Cell Features” and “Text Features.”\n  - Arrows indicate cell-to-text and text-to-cell projection paths.\n  - A central label reads “Cross-modal Projectors.”\n\n- Right panel: output tasks/findings\n  - “1. Cell-Text Discrimination”: produces a “Relevance Score.”\n  - “2. Text-to-Cell Generation”: decodes text-derived information into a gene-expression/cell representation.\n  - “3. Cell-To-Text Generation”: decodes cell representation into a textual description, shown as the classical monocyte example.\n\nOverall, the figure depicts a multimodal architecture aligning scRNA-seq gene-token representations with textual biological descriptions using an scLLM, a text LLM, and cross-modal projection modules for discrimination and bidirectional generation tasks.",
        "page_no": 4,
        "sha256": "c1acaeb6141c4bb543733f074f0bcdef463979db71492c67f4ffa275d4491c4d",
        "pixel_width": 786,
        "pixel_height": 253,
        "crop_box": {
          "x": 0.29,
          "y": 0.08,
          "width": 0.53,
          "height": 0.74
        },
        "panel_label": "scRNA-seq Data and cross-modal projectors",
        "visible_input_object": "Gene Expression Levels; Gene Tokens; Text Tokens",
        "visible_model_interface": "scLLM, Text LLM, Q-former Projector, Cross-Attn Projection",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the central route where scRNA-seq gene-expression values become gene tokens and connect into scLLM and the cross-modal projector interface with Text LLM visible. This keeps the grounded input/transformation path and excludes the right-side output tasks and the left explanatory panel.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_bbae04cee963",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "the cell embedding c",
          "actual_model_visible_form": "cell embedding c"
        }
      ],
      "routes": [
        {
          "route_id": "route_bbae04cee963",
          "configuration_id": "config_3c8409d79794",
          "route_label": "scMMGPT cell description generation",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "Cell Description Generation (§4.1.3).",
          "source_object_verbatim": "the cell embedding c",
          "source_object_normalized": "cell embedding",
          "source_modality_normalized": "dense continuous embedding",
          "transformation_chain_verbatim": [
            "we further fine-tune the text LLM with the cell-to-text translation loss L c2t",
            "autoregressively generate descriptions conditioned on the cell embedding c until the end-of-sentence token"
          ],
          "model_visible_form_verbatim": "cell embedding c",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "conditioned on the cell embedding c",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "generated_output",
          "input_status": "actual_model_input",
          "evidence_quote": "cell embedding c",
          "section_heading": "3.4 Adapting scMMGPT to Downstream Tasks",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            9,
            10
          ],
          "doc_item_refs": [
            "#/tables/3",
            "#/texts/328",
            "#/texts/329",
            "#/texts/330",
            "#/texts/331",
            "#/texts/349",
            "#/texts/350",
            "#/texts/355"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_001381::0030"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_000c4dd8661a"
    },
    {
      "model_id": "model_ff0ac12f6b2a",
      "model_name": "TissueCraftAI",
      "record_id": "full_2026-07-06__rec_000060",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_b4bb1defb3cb",
      "paper_title": "Query-driven generative AI synthesizes multi-modal spatial omics from histology.",
      "doi": "10.64898/2025.12.11.693669",
      "paper_url": "https://doi.org/10.64898/2025.12.11.693669",
      "route_count": 4,
      "configuration_count": 2,
      "family_counts": {
        "visual_raster_carrier": 2,
        "text_native_token_stream": 2
      },
      "subtype_counts": {
        "raw_slide_or_patch_input": 2,
        "structured_biological_prompt_or_task_scaffold": 1,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream",
        "visual_raster_carrier"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "raw_slide_or_patch_input",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "raw_slide_or_patch_input",
      "modalities": [
        "histology/slide image",
        "text"
      ],
      "lifecycle_phases": [
        "inference"
      ],
      "fusion_topologies": [
        "cross_attention",
        "side_or_generative_conditioning"
      ],
      "text_roles": [
        "instruction_or_query",
        "modality_or_task_selector",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_000060_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_000060_39a83b77f2f1/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1 -An overview of the PRISM-12M dataset and the TissueCraftAI framework. (a) Representative image pairs from the PRISM-12M dataset. Left, whole slide image (WSI) of spatial proteomics data overlaid with a pixel coordinate grid. The tissue was stained with antiCD45 (red, a pan-hematopoietic marker), anti-vimentin (green, a mesenchymal/stromal cell marker) antibodies, and DAPI (blue, a nuclear stain). Right, the corresponding WSI of Hematoxylin and Eosin (H&E) staining, showing tissue morphology. The same coordinate grid is shown, demonstrating spatial alignment with the spatial omics data. Three representative regions of interest (Patches 1-3) are shown at identical pixel coordinates (Width, Height) from both images. (b) Pie chart showing the composition of the PRISM-12M dataset, consisting of over 12 million patches of matched H&E and spatial omics images covering 14 human and mouse tissues. Mouse tissue is indicated with a 'mm' in the tissue name. The proportion of each dataset is indicated in parentheses. (c) Breakdown of the PRISM-12M dataset by tissue source and spatial omics data modality. CyCIF, cyclic Immunofluorescence; CODEX, CO-Detection by",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel scientific figure labeled **Fig. 1** with panels **a-e**.\n\nPanel **a** compares a **Spatial Omics Whole Slide Image** labeled **CD45, Vimentin, DAPI** with an **H&E Whole Slide Image**. Both show whole-slide tissue images with pixel axes and colored boxed regions. Below each whole-slide image are three extracted image patches labeled **Patch 1**, **Patch 2**, and **Patch 3**, with coordinate annotations. The lower row is labeled **Spatial Omics Image Patch** on the left and **H&E Image Patch** on the right.\n\nPanel **b** is a pie chart titled **PRISM-12M**, showing tissue or disease source categories including **Colon**, **Skin**, **Thymus**, **Tonsil**, **Heart**, **Kidney**, **Lung**, **Sarcoma**, **Liver**, **Bone marrow**, **Bone marrow**, **Pancreas**, **Glioma**, and **Neuroblastoma**. Colon is the largest segment at **45.4%**, followed by Skin at **10.1%**.\n\nPanel **c** is a horizontal bar chart of **Number of Patches** by source/tissue and technology. The legend includes **CODEX**, **CyCIF**, and **Xenium**. The largest category is **Colon (CyCIF)** with about **5,010,049** patches, followed by **Skin (CyCIF)** with about **1,260,894**. Other listed sources include Thymus, Tonsil, Heart, Colon, Kidney, Sarcoma, Liver, Mouse_Bone, Lung, Bone_marrow, Pancreas, and Glioma.\n\nPanel **d** is a model workflow diagram for **TissueCraftAI**. Inputs include a **Conditioning Image** labeled **H&E / DAPI** and a **Text Prompt** example: “Predict CD18 protein expression”. The pipeline shows a **ControlNet Module** with **Image Encoder** producing **Morphological Features**, a **Text Encoding** block with **Text Encoder** producing **Semantic Features**, and a **Latent Diffusion Module** with **Initial Noise Vector**, **Cross Attention**, **Iterative Denoising**, **Conditioned Representation**, and **VAE Decoder**. Outputs are labeled **Proteomics / Transcriptomics Map / Pathology Image**.\n\nPanel **e** lists applications with example thumbnails: **H&E generation**, **Protein expression prediction**, **RNA expression prediction**, **Cell type clustering/annotation**, and **Outcome prediction**.",
        "page_no": 18,
        "sha256": "ca296b44fc8764d23271be36ed764873aa392ea6c1b64ca49b8557d26cbd4765",
        "pixel_width": 621,
        "pixel_height": 919,
        "crop_box": {
          "x": 0.03,
          "y": 0.69,
          "width": 0.66,
          "height": 0.31
        },
        "panel_label": "d",
        "visible_input_object": "H&E/DAPI conditioning image and text prompt",
        "visible_model_interface": "ControlNet image encoder feeding morphological features into cross-attention with the text encoder",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crop panel d around the left and middle of the TissueCraftAI workflow so the conditioning image, text prompt, arrows, ControlNet/image encoder, text encoder, and cross-attention fusion are readable, while excluding the output-only panel e.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_39890e09066d",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "text prompt",
          "actual_model_visible_form": "plain-language prompt tokens"
        },
        {
          "subtype_id": "raw_slide_or_patch_input",
          "family_id": "visual_raster_carrier",
          "route_id": "route_49e2cc9a5c64",
          "example_input": "whole-slide image",
          "example_carrier": "224×224 RGB tissue patches",
          "example_interface": "patch encoder → multimodal generator",
          "actual_source": "H&E histology image",
          "actual_model_visible_form": "conditioning image"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_0d733b511958",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "structured text prompt",
          "actual_model_visible_form": "structured text prompt tokens"
        }
      ],
      "routes": [
        {
          "route_id": "route_49e2cc9a5c64",
          "configuration_id": "config_3f6b2df58274",
          "route_label": "TissueCraftAI histology image conditioning",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "For predicting protein and RNA expression",
          "source_object_verbatim": "H&E histology image",
          "source_object_normalized": "H&E histology image",
          "source_modality_normalized": "histology/slide image",
          "transformation_chain_verbatim": [
            "serves as a conditioning image",
            "processed by a ControlNet neural network",
            "merged with a text encoder output via cross-attention",
            "decoded by a VAE"
          ],
          "model_visible_form_verbatim": "conditioning image",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "A ControlNet neural network encodes the conditioning image",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "For predicting protein and RNA expression",
          "section_heading": "The TissueCraftAI Framework",
          "supporting_figure_or_table": "Figure 1d",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            11,
            18,
            19
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/483",
            "#/texts/484",
            "#/texts/485",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488",
            "#/texts/489",
            "#/texts/490",
            "#/texts/865",
            "#/texts/866",
            "#/texts/867",
            "#/texts/868",
            "#/texts/869",
            "#/texts/870",
            "#/texts/871"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000060::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000060::0042",
            "dense::full_2026-07-06__rec_000060::0043"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_0d733b511958",
          "configuration_id": "config_3f6b2df58274",
          "route_label": "TissueCraftAI molecular-target text prompting",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "For predicting protein and RNA expression",
          "source_object_verbatim": "structured text prompt",
          "source_object_normalized": "structured text prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "formatted as a molecular-target prompt",
            "processed by a text encoder",
            "merged with morphological features via cross-attention",
            "decoded by a VAE"
          ],
          "model_visible_form_verbatim": "structured text prompt tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "a user-defined text prompt specifies the molecular target",
          "fusion_topology": "cross_attention",
          "text_role": "modality_or_task_selector",
          "input_status": "actual_model_input",
          "evidence_quote": "For predicting protein and RNA expression",
          "section_heading": "Task formulation and text prompts",
          "supporting_figure_or_table": "Figure 1d",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            11,
            18,
            19
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/483",
            "#/texts/484",
            "#/texts/485",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488",
            "#/texts/489",
            "#/texts/490",
            "#/texts/865",
            "#/texts/866",
            "#/texts/867",
            "#/texts/868",
            "#/texts/869",
            "#/texts/870",
            "#/texts/871"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000060::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000060::0042",
            "dense::full_2026-07-06__rec_000060::0043"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_226242be284c",
          "configuration_id": "config_df2e5ea900f1",
          "route_label": "TissueCraftAI DAPI image conditioning",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "For the DAPI-to-H&E translation task",
          "source_object_verbatim": "DAPI nuclear-staining slide",
          "source_object_normalized": "DAPI nuclear-staining slide",
          "source_modality_normalized": "histology/slide image",
          "transformation_chain_verbatim": [
            "serves as a conditioning image",
            "processed by a ControlNet neural network",
            "merged with a text encoder output via cross-attention",
            "decoded by a VAE"
          ],
          "model_visible_form_verbatim": "conditioning image",
          "carrier_family": "visual_raster_carrier",
          "carrier_subtype": "raw_slide_or_patch_input",
          "insertion_or_fusion_verbatim": "a 4',6 -diamidino-2-phenylindole (DAPI) nuclear-staining slide, serves as a conditioning image",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "DAPI nuclear-staining slide",
          "section_heading": "The TissueCraftAI Framework",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            11,
            18,
            19
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/483",
            "#/texts/484",
            "#/texts/485",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488",
            "#/texts/489",
            "#/texts/490",
            "#/texts/865",
            "#/texts/866",
            "#/texts/867",
            "#/texts/868",
            "#/texts/869",
            "#/texts/870",
            "#/texts/871"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000060::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000060::0041"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_39890e09066d",
          "configuration_id": "config_df2e5ea900f1",
          "route_label": "TissueCraftAI H&E generation text prompting",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "For the DAPI-to-H&E translation task",
          "source_object_verbatim": "text prompt",
          "source_object_normalized": "text prompt",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "prompt specifies the target image",
            "processed by a text encoder",
            "merged with the conditioning image via cross-attention",
            "decoded by a VAE"
          ],
          "model_visible_form_verbatim": "plain-language prompt tokens",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "we used the prompt \"predict corresponding H&E image\"",
          "fusion_topology": "cross_attention",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "semantic feature vectors",
          "section_heading": "Task formulation and text prompts",
          "supporting_figure_or_table": "Figure 2",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            11,
            18,
            19
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/483",
            "#/texts/484",
            "#/texts/485",
            "#/texts/486",
            "#/texts/487",
            "#/texts/488",
            "#/texts/489",
            "#/texts/490",
            "#/texts/865",
            "#/texts/866",
            "#/texts/867",
            "#/texts/868",
            "#/texts/869",
            "#/texts/870",
            "#/texts/871"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000060::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000060::0041"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_7ec27bf438d4"
    },
    {
      "model_id": "model_df8ed6559ceb",
      "model_name": "TISSUENARRATOR",
      "record_id": "full_2026-07-06__rec_000063",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_b53ceb5b4db2",
      "paper_title": "TISSUENARRATOR: Generative Modeling of Spatial Transcriptomics with Large Language Models.",
      "doi": "10.1101/2025.11.24.690325",
      "paper_url": "https://doi.org/10.1101/2025.11.24.690325",
      "route_count": 7,
      "configuration_count": 7,
      "family_counts": {
        "text_native_token_stream": 7
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 4,
        "structured_biological_prompt_or_task_scaffold": 3
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "serialized_biological_context_or_ordered_profile",
      "modalities": [
        "spatial transcriptomics",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "inference"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_000063_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_000063_b081bf83cc88/figure_001.png",
        "figure_index": 1,
        "caption": "Figure 1: Overview of the TISSUENARRATOR framework. A. TISSUENARRATOR takes as input a tissue patch, encodes individual cells into cell sentences, and assembles a spatial sentence from neighboring cells. The spatial sentence can be interpreted by an LLM with biological prior knowledge and contextual understanding to predict gene expressions in a cell and to answer questions about tissues. Text output of the LLM can be further transformed into gene expression levels. B. TISSUENARRATOR supports four downstream tasks: cell generation, neighborhood perturbation, genetic perturbation, and conversational tissue querying. For cell generation tasks, model-generated cell sentences can be converted back into gene expression space.",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a schematic scientific figure with two labeled panels, **A** and **B**, describing an LLM-based framework for spatial/tissue data representation and prediction.\n\n**Panel A: Workflow**\n- Starts from a **tissue patch** shown as a small cluster of stylized cells, with a zoomed region indicated from a brain-like tissue icon.\n- The tissue patch is converted into **cell sentences**, shown as an **ordered gene list** per cell, with example gene tokens such as `G1, G2, G3, ...`, `G2, G1, G3, ...`, and `G4, G2, G5, ...`.\n- These cell sentences are then transformed into a **spatial sentence**, represented as an **ordered cell list**.\n- Each cell entry includes structured fields: `<x, y>` position, `cell type`, metadata, and gene list.\n- An **LLM** is shown consuming the spatial sentence, with annotations for **biological prior knowledge** and **contextual understanding**.\n- Outputs are split into:\n  - **Generative prediction**, including:\n    - **Cell prediction**, where a missing or unknown cell representation is predicted.\n    - **Text prediction**, with example text: “The major cell types in this region are microglia, astrocytes, ________”\n  - **Quantitative prediction**, showing a table predicting gene expression values for genes `G1`, `G2`, `G3`, with example values `.9`, `.6`, `.2`.\n\n**Panel B: Applications**\nShows four application boxes:\n- **Cell generation**: generating cells conditioned on optional metadata, including type.\n- **Conversational tissue querying**: a brain/tissue icon connected to chat bubbles and a user icon, implying question-answer interaction over tissue data.\n- **Genetic perturbation**: altered gene expression in a highlighted cell or cell group.\n- **Neighborhood perturbation**: altered neighborhood composition among nearby cells.\n\n**Biological source objects**\n- Tissue patch / brain-region context.\n- Individual cells with different colors and shapes.\n- Cell types including explicitly mentioned **microglia** and **astrocytes**.\n- Gene lists and gene expression values.\n\n**Model interface and transformations**\n- Biological tissue/spatial cell data are serialized into text-like “sentences.”\n- Cell-level gene lists become ordered cell representations with position, type, metadata, and genes.\n- These structured spatial sentences are passed to an LLM for generative, textual, and quantitative prediction tasks.\n\n**Findings or claim conveyed**\nThe figure proposes that tissue patches can be encoded as structured spatial sentences for LLM-based modeling, enabling cell prediction, text prediction, gene-expression prediction, cell generation, tissue querying, and perturbation analysis.",
        "page_no": 3,
        "sha256": "0befdb410abb4e546974eadd4f1986311baf5682f0c870a9c16a86bb69266aa4",
        "pixel_width": 993,
        "pixel_height": 308,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 0.635,
          "height": 0.62
        },
        "panel_label": "A",
        "visible_input_object": "tissue patch -> cell sentences -> spatial sentence with LLM interface",
        "visible_model_interface": "LLM consuming the serialized spatial sentence with biological prior knowledge/contextual understanding",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crops the left workflow in Figure 1A, keeping the source tissue patch, ordered gene-list cell sentences, the assembled spatial sentence, and the LLM insertion point. It excludes the downstream output panels and all of panel B while preserving the arrows needed to understand the serialization route.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_de59b0fceddc",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "an input tissue section",
          "actual_model_visible_form": "1D spatial sentence"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_24ffc126eb25",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "the name of an anatomical structure",
          "actual_model_visible_form": "natural-language query"
        }
      ],
      "routes": [
        {
          "route_id": "route_de59b0fceddc",
          "configuration_id": "config_5e8d49c69552",
          "route_label": "Spatial sentence fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "fine-tuning on spatial sentences",
          "source_object_verbatim": "an input tissue section",
          "source_object_normalized": "tissue section",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "transform each cell into cell sentences",
            "assemble a 1D spatial sentence"
          ],
          "model_visible_form_verbatim": "1D spatial sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "fine-tune an LLM using next-token prediction",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "An overview of the TISSUENARRATOR framework is shown in Fig. 1A. On an input tissue section, TISSUENARRATOR starts by transforming each cell into the natural language space by encoding gene expression as 'cell sentences', a list of genes ranked by gene expression level. All cell sentences in a tissue patch are then assembled into a 1D 'spatial sentence', where cells are serialized by proximitybased traversal. This spatial sentence then serves as the input for fine-tuning an LLM to learn cell generation through next-token prediction. The fine-tuned model fits multiple downstream applications, including cell generation, neighborhood perturbation, genetic perturbation, and conversational tissue querying. Further details of TISSUENARRATOR can be found in the Methods section.",
          "section_heading": "Overview of generative modeling in TISSUENARRATOR",
          "supporting_figure_or_table": "Fig. 1A",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            11,
            12
          ],
          "doc_item_refs": [
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/129",
            "#/texts/130",
            "#/texts/131",
            "#/texts/132",
            "#/texts/221",
            "#/texts/222",
            "#/texts/223",
            "#/texts/225",
            "#/texts/226",
            "#/texts/227",
            "#/texts/228",
            "#/texts/229",
            "#/texts/232",
            "#/texts/233",
            "#/texts/234"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000063::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000063::0014",
            "dense::full_2026-07-06__rec_000063::0031"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a5e96dd9a19b",
          "configuration_id": "config_0483e926ac88",
          "route_label": "Cell generation from spatial neighborhood",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "cell generation mode (conditional generation / unconditional generation)",
          "source_object_verbatim": "a tissue neighborhood around a target cell",
          "source_object_normalized": "tissue neighborhood around a target cell",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "encode neighboring cells as cell sentences",
            "assemble them into a spatial sentence"
          ],
          "model_visible_form_verbatim": "spatial sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conditioned generation through next-token prediction",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Given the cell sentences from neighboring cells and the spatial location of a target cell",
          "section_heading": "TISSUENARRATOR inference",
          "supporting_figure_or_table": "Fig. 1A",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            12,
            13,
            14
          ],
          "doc_item_refs": [
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/129",
            "#/texts/130",
            "#/texts/131",
            "#/texts/132",
            "#/texts/242",
            "#/texts/243",
            "#/texts/244",
            "#/texts/245",
            "#/texts/247",
            "#/texts/248",
            "#/texts/250",
            "#/texts/251",
            "#/texts/252",
            "#/texts/253",
            "#/texts/254",
            "#/texts/256",
            "#/texts/257",
            "#/texts/258"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000063::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000063::0001",
            "dense::full_2026-07-06__rec_000063::0016",
            "dense::full_2026-07-06__rec_000063::0017",
            "dense::full_2026-07-06__rec_000063::0018"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_dbc6b7ab297a",
          "configuration_id": "config_597a8a8b8d6b",
          "route_label": "Neighborhood perturbation generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "predicting response to neighborhood composition",
          "source_object_verbatim": "a neighborhood with counterfactual cell-type composition",
          "source_object_normalized": "counterfactual neighborhood cell-type composition",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "serialize the neighborhood as a spatial sentence",
            "predict the target cell within the altered neighborhood"
          ],
          "model_visible_form_verbatim": "spatial sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conditional generation through next-token prediction",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "TISSUENARRATOR predicts a target cell within a neighborhood with counterfactual cell-type composition",
          "section_heading": "Real data applications",
          "supporting_figure_or_table": "Fig. 1B",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            6,
            7,
            11,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/129",
            "#/texts/130",
            "#/texts/131",
            "#/texts/250",
            "#/texts/251",
            "#/texts/252",
            "#/texts/253",
            "#/texts/254",
            "#/texts/256",
            "#/texts/257",
            "#/texts/258"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000063::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000063::0004",
            "dense::full_2026-07-06__rec_000063::0017",
            "dense::full_2026-07-06__rec_000063::0019"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_5a170efb9d64",
          "configuration_id": "config_1dbd78fd297a",
          "route_label": "Genetic perturbation generation",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "predicting effects of genetic perturbation",
          "source_object_verbatim": "an edited or perturbed neighboring cell",
          "source_object_normalized": "edited or perturbed neighboring cell",
          "source_modality_normalized": "spatial transcriptomics",
          "transformation_chain_verbatim": [
            "serialize the perturbed cell and its context",
            "predict the target cell given the perturbed neighbor"
          ],
          "model_visible_form_verbatim": "spatial sentence",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "conditioned generation through next-token prediction",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "TISSUENARRATOR predicts a target cell given an edited or perturbed neighboring cell",
          "section_heading": "Real data applications",
          "supporting_figure_or_table": "Fig. 1B",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            6,
            7,
            11,
            13,
            14
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/129",
            "#/texts/130",
            "#/texts/131",
            "#/texts/250",
            "#/texts/251",
            "#/texts/252",
            "#/texts/253",
            "#/texts/254",
            "#/texts/256",
            "#/texts/257",
            "#/texts/258"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000063::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000063::0003",
            "dense::full_2026-07-06__rec_000063::0017",
            "dense::full_2026-07-06__rec_000063::0020"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_24ffc126eb25",
          "configuration_id": "config_bc93787e451a",
          "route_label": "Spatial Q&A: top cell types",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "given the name of an anatomical structure, return the top 5 most abundant cell types",
          "source_object_verbatim": "the name of an anatomical structure",
          "source_object_normalized": "anatomical structure name",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "form a natural-language question",
            "fine-tune on questions derived from held-out structures"
          ],
          "model_visible_form_verbatim": "natural-language query",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "conversational interface",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "given the name of an anatomical structure, (1) return the top 5 most abundant cell types",
          "section_heading": "TISSUENARRATOR enables interactive Q&A about spatial regions",
          "supporting_figure_or_table": "Fig. 5A",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            8,
            11,
            13
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/129",
            "#/texts/130",
            "#/texts/131",
            "#/texts/196"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000063::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000063::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_6ccf596636de",
          "configuration_id": "config_c72b62f6828f",
          "route_label": "Spatial Q&A: top genes",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "given the name of an anatomical structure, list the top 20 highly expressed genes",
          "source_object_verbatim": "the name of an anatomical structure",
          "source_object_normalized": "anatomical structure name",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "form a natural-language question",
            "fine-tune on questions derived from held-out structures"
          ],
          "model_visible_form_verbatim": "natural-language query",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "conversational interface",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "To evaluate TISSUENARRATOR's ability to apply biological prior knowledge and learned spatial knowledge in a Q&A setting, we constructed a proof-of-concept dataset covering three tasks: given the name of an anatomical structure, (1) return the top 5 most abundant cell types, (2) list the top 20 highly expressed genes, and (3) generate a natural-language description of the structure. Details of data curation and task examples are provided in Supplementary Note D. We split the dataset by structure name, fine-tuned TISSUENARRATOR, and evaluated the model on questions derived from held-out structures. TISSUENARRATOR generalizes to unseen anatomical structure names in natural language queries by building on the pretrained knowledge of its LLM backbone. For example, it can generate the exact top five cell types for the Dorsal peduncular area ( Fig. 5A) and also produces a coherent, biologically grounded paragraph describing cranial nerves ( Fig. 5B), These examples highlight the model's ability to link spatial context with natural language explanations, a bridge between data-driven cell modeling and user-friendly querying. Moreover, fine-tuning TISSUENARRATOR outperformed directly fine-tuning Qwen-4B-Base . This suggests that the spatial sentence training stage not only teaches the model cell generation, but also enables a deeper spatial understanding that transfers to Q&A tasks ( Supplementary Note D).",
          "section_heading": "TISSUENARRATOR enables interactive Q&A about spatial regions",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4,
            8,
            11,
            13
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/129",
            "#/texts/130",
            "#/texts/131",
            "#/texts/196",
            "#/texts/198"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000063::route_007"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000063::0002"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a3f58fcbfec1",
          "configuration_id": "config_bdaf00473df9",
          "route_label": "Spatial Q&A: region summarization",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "given the name of an anatomical structure, generate a natural-language description of the structure",
          "source_object_verbatim": "the name of an anatomical structure",
          "source_object_normalized": "anatomical structure name",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "form a natural-language question",
            "fine-tune on questions derived from held-out structures"
          ],
          "model_visible_form_verbatim": "natural-language query",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "conversational interface",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "To evaluate TISSUENARRATOR's ability to apply biological prior knowledge and learned spatial knowledge in a Q&A setting, we constructed a proof-of-concept dataset covering three tasks: given the name of an anatomical structure, (1) return the top 5 most abundant cell types, (2) list the top 20 highly expressed genes, and (3) generate a natural-language description of the structure. Details of data curation and task examples are provided in Supplementary Note D. We split the dataset by structure name, fine-tuned TISSUENARRATOR, and evaluated the model on questions derived from held-out structures. TISSUENARRATOR generalizes to unseen anatomical structure names in natural language queries by building on the pretrained knowledge of its LLM backbone. For example, it can generate the exact top five cell types for the Dorsal peduncular area ( Fig. 5A) and also produces a coherent, biologically grounded paragraph describing cranial nerves ( Fig. 5B), These examples highlight the model's ability to link spatial context with natural language explanations, a bridge between data-driven cell modeling and user-friendly querying. Moreover, fine-tuning TISSUENARRATOR outperformed directly fine-tuning Qwen-4B-Base . This suggests that the spatial sentence training stage not only teaches the model cell generation, but also enables a deeper spatial understanding that transfers to Q&A tasks ( Supplementary Note D).",
          "section_heading": "TISSUENARRATOR enables interactive Q&A about spatial regions",
          "supporting_figure_or_table": "Fig. 5B",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            3,
            4,
            8,
            11,
            13
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/123",
            "#/texts/124",
            "#/texts/125",
            "#/texts/126",
            "#/texts/129",
            "#/texts/130",
            "#/texts/131",
            "#/texts/196",
            "#/texts/198"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_000063::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_000063::0002"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_2bfb9536cf22"
    },
    {
      "model_id": "model_4361ad5c9fbb",
      "model_name": "X-Cell",
      "record_id": "full_2026-07-06__rec_003517",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_bcfca4a74dd0",
      "paper_title": "X-Cell: Scaling Causal Perturbation Prediction Across Diverse Cellular Contexts via Diffusion Language Models",
      "doi": "10.64898/2026.03.18.712807",
      "paper_url": "https://doi.org/10.64898/2026.03.18.712807",
      "route_count": 19,
      "configuration_count": 14,
      "family_counts": {
        "dense_continuous_carrier": 18,
        "text_native_token_stream": 1
      },
      "subtype_counts": {
        "direct_projected_embedding": 16,
        "pooled_or_aggregated_embedding": 2,
        "structured_biological_prompt_or_task_scaffold": 1
      },
      "families": [
        "text_native_token_stream",
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "pooled_or_aggregated_embedding",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "gene dependency profiles",
        "gene symbol conditioning",
        "ligand embeddings",
        "morphological phenotype embeddings",
        "protein sequence",
        "protein-protein interaction network",
        "single-cell chemical perturbation data",
        "single-cell perturbation data",
        "single-cell transcriptomic perturbation profiles",
        "single-cell transcriptomics",
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "inference",
        "pretraining"
      ],
      "fusion_topologies": [
        "cross_attention",
        "retrieval_or_tool_context",
        "side_or_generative_conditioning",
        "unclear"
      ],
      "text_roles": [
        "biological_payload",
        "metadata_or_context",
        "modality_or_task_selector",
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003517_figure_001.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003517_bdcbcbd932f3/figure_001.png",
        "figure_index": 1,
        "caption": "Fig. 1 Overview of X-Atlas/Pisces and X-Cell. (A) Schematic of X-Atlas/Pisces, which contains seven genome-scale CRISPRi Perturb-seq screens in HCT116, HEK293T, HepG2, iPSC, Jurkat Resting, Jurkat Active, and iPSC MultiDifferentiation (iPSC Multi-Diff) (left). Size comparison of X-Atlas/Orion (8M cells) versus X-Atlas/Pisces (25.6M cells) (middle). UMAP of cells in X-Atlas/Pisces, colored by screen. For visualization, the dataset was downsampled to 1M total cells while preserving the relative proportions of each original screen. Within this subset, 5% of the cells are nontargeting controls and the remainder are perturbed cells. (B) X-Cell combines diffusion LM training and cross-attention to prior knowledge to predict perturbed cell states from control cell sets. Six prior knowledge sources include pre-trained embeddings from LLM [19], ESM-2 [38], STRING [40, 41], DepMap [42], JUMP-Cell Painting [43], and scGPT [8] (Section 4.1.2). The main architecture consists of stacked self-attention blocks that encode control cell sets, with interleaved crossattention to prior knowledge embeddings. X-Cell enables diffusion-style training by randomly replacing 25%, 50%, or 75% of control gene expression values with ground-truth perturbed values, and providing a binary Diff Mask to indicate the revealed positions (Section 4.1.3). (C) During inference, X-Cell refines predictions via iterative diffusion by remasking part of its output as input for subsequent generation steps. (D) X-Cell scales from 55M parameters (X-Cell) to 4.9B parameters (X-Cell-Ultra), exceeding the size of existing single-cell foundation models.",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel scientific figure describing the X-Cell model and datasets.\n\nPanel A: Dataset and screen overview. Shows “X-Atlas” with two sources: Orion (8M) and Pisces (25.6M). Screens listed include HCT116, HEK293T, HepG2, iPSC, Jurkat Resting, Jurkat Active, and iPSC Multi-Diff. Purposes are grouped as Architecture Validation, Context Generalization, and Cell Type Generalization. A stacked bar chart compares cell counts in Orion and Pisces by screen, and a UMAP labeled “Cell x Screen” shows colored cell clusters by dataset/screen.\n\nPanel B: X-Cell architecture schematic. Input consists of X-Atlas/Pisces perturbation examples. Prior knowledge is injected through “Gene X KD” using sources/interfaces labeled LLM, ESM2, STRING, DepMap, JUMP-Cell Painting, and scGPT. The central architecture shows control and perturbed cell sets with gene tokens, expression tokens, and diffusion masks. A “Diffusion LM Training” block indicates replacing expression values with perturbed expression for t > 0 using a diffusion mask. The model combines control encoding and perturbation conditioning, with cross-attention over prior knowledge embeddings, self-attention layers, and a perturbation decoder. Translational applications shown at right are Context Generalization, Perturbation Prediction, and Mechanism of Action.\n\nPanel C: Iterative prediction/remasking workflow for “Gene X KD.” Starts from input all-control expression plus diffusion mask at t=0, predicts intermediate values, remasks with 50% predicted and 50% control values at t=0.5, then produces a final prediction at t=1.0. Color legend indicates control versus predicted values.\n\nPanel D: Model parameter comparison bar chart. Models shown include scGPT, X-Cell, Geneformer V2, STATE-SE, scFoundation, TranscriptFormer, Tahoe-x1, and X-Cell-Ultra. Bars indicate parameter counts and years. X-Cell is labeled 55M 2026, X-Cell-Ultra 4.9B 2026, and an annotation indicates “92x.”",
        "page_no": 4,
        "sha256": "127b8093c250015a3bff1152f2ab68c840594b41f92265d6231a50b94ca5a83f",
        "pixel_width": 878,
        "pixel_height": 757,
        "crop_box": {
          "x": 0,
          "y": 0.27,
          "width": 0.86,
          "height": 0.49
        },
        "panel_label": "B",
        "visible_input_object": "X-Atlas/Pisces control cell sets with Gene X KD prior-knowledge injection",
        "visible_model_interface": "Control Cell Set (t=0) feeding X-Cell control encoding and perturbation conditioning with diffusion LM training / cross-attention",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "This crop keeps the grounded input path in Fig. 1B: the X-Atlas/Pisces input panel, the control cell set at t=0, the Gene X KD prior-knowledge injection, and the central control-encoding / perturbation-conditioning interface. It excludes the output-only application column and the unrelated bottom panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_4481082b958d",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "control cell sets",
          "actual_model_visible_form": "gene identity embeddings, continuous value encodings, and a binary perturbation mask"
        },
        {
          "subtype_id": "pooled_or_aggregated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_f8ea4976270b",
          "example_input": "{gene/cell/patch embeddings}",
          "example_carrier": "mean/attention pool = one compact vector",
          "example_interface": "aggregator → generator",
          "actual_source": "JUMP Cell Painting dataset",
          "actual_model_visible_form": "morphological embeddings"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_ff0c287e0a54",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "the MOA gene",
          "actual_model_visible_form": "perturbed target gene identity"
        }
      ],
      "routes": [
        {
          "route_id": "route_4481082b958d",
          "configuration_id": "config_af51668f2836",
          "route_label": "Control cell set diffusion input",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "predict perturbed cell states from control cell sets",
          "source_object_verbatim": "control cell sets",
          "source_object_normalized": "control cell sets from X-Atlas/Pisces",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "CP10K normalization",
            "log1p",
            "protein-coding gene filter",
            "gene subsampling to context length G' ≤ 4,000",
            "discrete masking with 25%, 50%, 75%, or 100% revealed values"
          ],
          "model_visible_form_verbatim": "gene identity embeddings, continuous value encodings, and a binary perturbation mask",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "stacked self-attention blocks and interleaved cross-attention",
          "fusion_topology": "cross_attention",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "X-Cell combines diffusion LM training and cross-attention to prior knowledge to predict perturbed cell states from control cell sets.",
          "section_heading": "Fig. 1 Overview of X-Atlas/Pisces and X-Cell",
          "supporting_figure_or_table": "Figure 1B",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            3,
            4
          ],
          "doc_item_refs": [
            "#/pictures/0",
            "#/texts/19",
            "#/texts/20"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_001"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0001"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_87d509a9589c",
          "configuration_id": "config_ccd0df06e096",
          "route_label": "GenePT prior embedding route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cross-attention to prior knowledge embeddings",
          "source_object_verbatim": "NCBI gene summaries and related annotations",
          "source_object_normalized": "NCBI gene summaries and related annotations",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "large language model text embeddings"
          ],
          "model_visible_form_verbatim": "GenePT gene embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "cross-attention to prior knowledge embeddings",
          "fusion_topology": "cross_attention",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "GenePT embeddings encode semantic representations of gene function by applying large language model text embeddings to NCBI gene summaries and related annotations.",
          "section_heading": "4.1.2 Prior Knowledge Sources",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9,
            15,
            23,
            24,
            54,
            55
          ],
          "doc_item_refs": [
            "#/texts/1158",
            "#/texts/1159",
            "#/texts/1160",
            "#/texts/1161",
            "#/texts/1162",
            "#/texts/1163",
            "#/texts/1164",
            "#/texts/1287",
            "#/texts/1288",
            "#/texts/1289",
            "#/texts/1290",
            "#/texts/1291",
            "#/texts/1292",
            "#/texts/1293",
            "#/texts/1294",
            "#/texts/2096",
            "#/texts/2098",
            "#/texts/666",
            "#/texts/667",
            "#/texts/668"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_002"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0130",
            "dense::full_2026-07-06__rec_003517::0161"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4d04bd23e2d1",
          "configuration_id": "config_ccd0df06e096",
          "route_label": "ESM-2 prior embedding route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cross-attention to prior knowledge embeddings",
          "source_object_verbatim": "protein sequences of protein-coding genes",
          "source_object_normalized": "protein sequences of protein-coding genes",
          "source_modality_normalized": "protein sequence",
          "transformation_chain_verbatim": [
            "ESM-2 protein language model embeddings"
          ],
          "model_visible_form_verbatim": "ESM-2 protein embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "cross-attention to prior knowledge embeddings",
          "fusion_topology": "cross_attention",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "We incorporated protein sequence embeddings from the ESM-2 protein language model [38] for protein-coding genes",
          "section_heading": "4.1.2 Prior Knowledge Sources",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9,
            15,
            23,
            24,
            54,
            55
          ],
          "doc_item_refs": [
            "#/texts/1158",
            "#/texts/1159",
            "#/texts/1160",
            "#/texts/1161",
            "#/texts/1162",
            "#/texts/1163",
            "#/texts/1164",
            "#/texts/1287",
            "#/texts/1288",
            "#/texts/1289",
            "#/texts/1290",
            "#/texts/1291",
            "#/texts/1292",
            "#/texts/1293",
            "#/texts/1294",
            "#/texts/2096",
            "#/texts/2098",
            "#/texts/666",
            "#/texts/667",
            "#/texts/668"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_003"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0130",
            "dense::full_2026-07-06__rec_003517::0161"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_923c186d7d07",
          "configuration_id": "config_ccd0df06e096",
          "route_label": "STRING interaction embedding route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cross-attention to prior knowledge embeddings",
          "source_object_verbatim": "STRING protein-protein interaction network",
          "source_object_normalized": "STRING protein-protein interaction network",
          "source_modality_normalized": "protein-protein interaction network",
          "transformation_chain_verbatim": [
            "SPACE graph embeddings",
            "protein-protein interaction network"
          ],
          "model_visible_form_verbatim": "STRING network embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "cross-attention to prior knowledge embeddings",
          "fusion_topology": "cross_attention",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Protein interaction priors were obtained from embeddings derived from the STRING protein-protein interaction (PPI) network [41].",
          "section_heading": "4.1.2 Prior Knowledge Sources",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9,
            15,
            23,
            24,
            54,
            55
          ],
          "doc_item_refs": [
            "#/texts/1158",
            "#/texts/1159",
            "#/texts/1160",
            "#/texts/1161",
            "#/texts/1162",
            "#/texts/1163",
            "#/texts/1164",
            "#/texts/1287",
            "#/texts/1288",
            "#/texts/1289",
            "#/texts/1290",
            "#/texts/1291",
            "#/texts/1292",
            "#/texts/1293",
            "#/texts/1294",
            "#/texts/2096",
            "#/texts/2098",
            "#/texts/666",
            "#/texts/667",
            "#/texts/668"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_004"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0130",
            "dense::full_2026-07-06__rec_003517::0161"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4949b4e43333",
          "configuration_id": "config_ccd0df06e096",
          "route_label": "DepMap dependency embedding route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cross-attention to prior knowledge embeddings",
          "source_object_verbatim": "DepMap CRISPR gene dependency dataset",
          "source_object_normalized": "DepMap CRISPR gene dependency dataset",
          "source_modality_normalized": "gene dependency profiles",
          "transformation_chain_verbatim": [
            "Chronos-inferred gene effect scores",
            "gene dependency embeddings"
          ],
          "model_visible_form_verbatim": "DepMap gene effect embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "cross-attention to prior knowledge embeddings",
          "fusion_topology": "cross_attention",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "Functional genomics priors were derived from the DepMap CRISPR gene dependency dataset [42]",
          "section_heading": "4.1.2 Prior Knowledge Sources",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9,
            15,
            23,
            24,
            54,
            55
          ],
          "doc_item_refs": [
            "#/texts/1158",
            "#/texts/1159",
            "#/texts/1160",
            "#/texts/1161",
            "#/texts/1162",
            "#/texts/1163",
            "#/texts/1164",
            "#/texts/1287",
            "#/texts/1288",
            "#/texts/1289",
            "#/texts/1290",
            "#/texts/1291",
            "#/texts/1292",
            "#/texts/1293",
            "#/texts/1294",
            "#/texts/2096",
            "#/texts/2098",
            "#/texts/666",
            "#/texts/667",
            "#/texts/668"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_005"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0130",
            "dense::full_2026-07-06__rec_003517::0161"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f8ea4976270b",
          "configuration_id": "config_ccd0df06e096",
          "route_label": "JUMP Cell Painting morphology route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "cross-attention to prior knowledge embeddings",
          "source_object_verbatim": "JUMP Cell Painting dataset",
          "source_object_normalized": "JUMP Cell Painting dataset",
          "source_modality_normalized": "morphological phenotype embeddings",
          "transformation_chain_verbatim": [
            "batch-corrected embeddings",
            "Harmony batch correction",
            "PCA dimensionality reduction"
          ],
          "model_visible_form_verbatim": "morphological embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "cross-attention to prior knowledge embeddings",
          "fusion_topology": "cross_attention",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "We incorporated morphological embeddings derived from the JUMP Cell Painting dataset [43]",
          "section_heading": "4.1.2 Prior Knowledge Sources",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            9,
            15,
            23,
            24,
            46,
            54,
            55
          ],
          "doc_item_refs": [
            "#/texts/1158",
            "#/texts/1159",
            "#/texts/1160",
            "#/texts/1161",
            "#/texts/1162",
            "#/texts/1163",
            "#/texts/1164",
            "#/texts/1287",
            "#/texts/1288",
            "#/texts/1289",
            "#/texts/1290",
            "#/texts/1291",
            "#/texts/1292",
            "#/texts/1293",
            "#/texts/1294",
            "#/texts/1689",
            "#/texts/2096",
            "#/texts/2098",
            "#/texts/666",
            "#/texts/667",
            "#/texts/668"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_006"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0035",
            "dense::full_2026-07-06__rec_003517::0130",
            "dense::full_2026-07-06__rec_003517::0161"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_51803f2831b6",
          "configuration_id": "config_2088ccbdb12b",
          "route_label": "Parse-1M ligand cross-attention route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "incorporated ligand information via cross-attention using the Parse-1M dataset",
          "source_object_verbatim": "ligand information",
          "source_object_normalized": "ligand information from Parse-1M",
          "source_modality_normalized": "ligand embeddings",
          "transformation_chain_verbatim": [
            "Parse-1M dataset",
            "ligand embeddings"
          ],
          "model_visible_form_verbatim": "ligand information embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "cross-attention",
          "fusion_topology": "cross_attention",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "we incorporated ligand information via cross-attention using the Parse-1M dataset [53]",
          "section_heading": "2.3 X-Cell generalizes perturbation effects across cellular contexts",
          "supporting_figure_or_table": "Figure 3C",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            40,
            41,
            42
          ],
          "doc_item_refs": [
            "#/texts/1591",
            "#/texts/1593",
            "#/texts/1595",
            "#/texts/1596",
            "#/texts/1597",
            "#/texts/1598",
            "#/texts/1600",
            "#/texts/1601",
            "#/texts/334",
            "#/texts/337"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_008"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0113",
            "dense::full_2026-07-06__rec_003517::0115"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ff0c287e0a54",
          "configuration_id": "config_63c0863f768c",
          "route_label": "MOA gene perturbation-target conditioning",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "predict drug-induced expression responses",
          "source_object_verbatim": "the MOA gene",
          "source_object_normalized": "MOA gene",
          "source_modality_normalized": "gene symbol conditioning",
          "transformation_chain_verbatim": [
            "encode the MOA gene as the perturbed target"
          ],
          "model_visible_form_verbatim": "perturbed target gene identity",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "perturbation target conditioning",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "modality_or_task_selector",
          "input_status": "actual_model_input",
          "evidence_quote": "We further evaluated X-Cell's ability to transfer knowledge from genetic perturbations to signaling and chemical perturbations. In a fine-tuning setting, we incorporated ligand information via cross-attention using the Parse-1M dataset [53], achieving strong performance on a held-out set of 59 signaling perturbations (Figure 3C, row iii; Section 4.1.7). Next, we tested X-Cell's generalization to predicting the effects of drug perturbations in a zero-shot setting. We constructed a subset of the Tahoe-100M dataset [54] containing inhibitor drugs with single-target mechanisms of action (MOA), establishing a direct link between genetic perturbations and drug effects (Section 4.1.7). In this setup, X-Cell encodes the MOA gene as the perturbed target and the control cell-line transcriptome as input to predict drug-induced expression responses (Figure 3E). On zero-shot tasks, we benchmarked against a STATE model trained on 3.1M cells from public datasets only (Section 4.1.5) [12, 52, 55]. Despite never observing Tahoe perturbations during training, X-Cell achieves stronger performance in both Pearson ∆ (mean 0.31 versus 0.22) and MAE (mean 0.08 versus 0.16). A stratified analysis further suggests that predictions are more effective for cell lines that are similar to the X-Cell training corpus, highlighting the importance of biologically relevant training contexts for successful cross-modality generalization (Supplementary Figure 6). These results highlight X-Cell's ability to generalize across perturbation modalities and suggest a path toward integrating perturbation modeling with drug discovery.",
          "section_heading": "2.3 X-Cell generalizes perturbation effects across cellular contexts",
          "supporting_figure_or_table": "Figure 3E",
          "evidence_status": "explicit_text",
          "uncertainty": "Split from a hybrid route that also includes the control transcriptome.",
          "pages": [
            7,
            8,
            18,
            20,
            21,
            22,
            46,
            55,
            56
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/1202",
            "#/texts/1203",
            "#/texts/1204",
            "#/texts/1238",
            "#/texts/1239",
            "#/texts/1240",
            "#/texts/1241",
            "#/texts/1251",
            "#/texts/334",
            "#/texts/337"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0022",
            "dense::full_2026-07-06__rec_003517::0040"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_3df5842842fc",
          "configuration_id": "config_63c0863f768c",
          "route_label": "Control transcriptome drug-response input",
          "lifecycle_phase": "inference",
          "task_or_configuration_verbatim": "predict drug-induced expression responses",
          "source_object_verbatim": "control cell-line transcriptome",
          "source_object_normalized": "control cell-line transcriptome",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [
            "use the control cell-line transcriptome as input"
          ],
          "model_visible_form_verbatim": "control cell-line transcriptome",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "input to predict drug-induced expression responses",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "We further evaluated X-Cell's ability to transfer knowledge from genetic perturbations to signaling and chemical perturbations. In a fine-tuning setting, we incorporated ligand information via cross-attention using the Parse-1M dataset [53], achieving strong performance on a held-out set of 59 signaling perturbations (Figure 3C, row iii; Section 4.1.7). Next, we tested X-Cell's generalization to predicting the effects of drug perturbations in a zero-shot setting. We constructed a subset of the Tahoe-100M dataset [54] containing inhibitor drugs with single-target mechanisms of action (MOA), establishing a direct link between genetic perturbations and drug effects (Section 4.1.7). In this setup, X-Cell encodes the MOA gene as the perturbed target and the control cell-line transcriptome as input to predict drug-induced expression responses (Figure 3E). On zero-shot tasks, we benchmarked against a STATE model trained on 3.1M cells from public datasets only (Section 4.1.5) [12, 52, 55]. Despite never observing Tahoe perturbations during training, X-Cell achieves stronger performance in both Pearson ∆ (mean 0.31 versus 0.22) and MAE (mean 0.08 versus 0.16). A stratified analysis further suggests that predictions are more effective for cell lines that are similar to the X-Cell training corpus, highlighting the importance of biologically relevant training contexts for successful cross-modality generalization (Supplementary Figure 6). These results highlight X-Cell's ability to generalize across perturbation modalities and suggest a path toward integrating perturbation modeling with drug discovery.",
          "section_heading": "2.3 X-Cell generalizes perturbation effects across cellular contexts",
          "supporting_figure_or_table": "Figure 3E",
          "evidence_status": "explicit_text",
          "uncertainty": "Split from a hybrid route that also includes the MOA-gene target label.",
          "pages": [
            7,
            8,
            18,
            20,
            21,
            22,
            46,
            55,
            56
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/1202",
            "#/texts/1203",
            "#/texts/1204",
            "#/texts/1238",
            "#/texts/1239",
            "#/texts/1240",
            "#/texts/1241",
            "#/texts/1251",
            "#/texts/334",
            "#/texts/337"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_009"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0022",
            "dense::full_2026-07-06__rec_003517::0040"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_031891f72cba",
          "configuration_id": "config_83ceba3558a5",
          "route_label": "Replogle-Nadig fine-tuning route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "fine-tuned on the Replogle-Nadig dataset",
          "source_object_verbatim": "Replogle-Nadig Perturb-seq dataset",
          "source_object_normalized": "Replogle-Nadig Perturb-seq dataset",
          "source_modality_normalized": "single-cell perturbation data",
          "transformation_chain_verbatim": [
            "processed external Perturb-seq dataset",
            "normalized gene expression",
            "2,000 highly variable genes",
            "HepG2 perturbation holdout split"
          ],
          "model_visible_form_verbatim": "control and perturbed cell sets with perturbation labels",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "set-level diffusion transformer with stacked self-attention blocks and interleaved cross-attention",
          "fusion_topology": "cross_attention",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "We then evaluated X-Cell's generalizability to external CRISPR perturbation datasets by fine-tuning the model on the Replogle-Nadig dataset [12, 52]",
          "section_heading": "2.3 X-Cell generalizes perturbation effects across cellular contexts",
          "supporting_figure_or_table": "Figure 3C",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            19,
            20,
            41
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/1208",
            "#/texts/1209",
            "#/texts/1210",
            "#/texts/1211",
            "#/texts/1212",
            "#/texts/1213",
            "#/texts/1595",
            "#/texts/1596",
            "#/texts/1597",
            "#/texts/1598",
            "#/texts/333"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_010"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0012",
            "dense::full_2026-07-06__rec_003517::0116"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a038366bc0a1",
          "configuration_id": "config_5b99f9b9b132",
          "route_label": "Parse-1M fine-tuning route",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "fine-tuned on Parse-1M to benchmark performance on highly variable genes",
          "source_object_verbatim": "Parse-1M subset from ParsePBMC donor 1",
          "source_object_normalized": "Parse-1M subset from ParsePBMC donor 1",
          "source_modality_normalized": "single-cell perturbation data",
          "transformation_chain_verbatim": [
            "1,267,690 cells from donor 1",
            "2,000 highly variable genes from 18,308 feature space",
            "cytokine perturbations mapped to protein-coding gene names",
            "59 overlapping cytokines held out in CD4 Memory cells"
          ],
          "model_visible_form_verbatim": "control and perturbed PBMC cell sets with cytokine perturbation labels",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "set-level diffusion transformer with stacked self-attention blocks and interleaved cross-attention",
          "fusion_topology": "cross_attention",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Parse-1M. We constructed the Parse-1M subset using 1,267,690 cells from donor 1 in the ParsePBMC dataset",
          "section_heading": "4.1.7 Evaluation Datasets",
          "supporting_figure_or_table": "Figure 3A",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            20,
            41
          ],
          "doc_item_refs": [
            "#/texts/1238",
            "#/texts/1239",
            "#/texts/1240",
            "#/texts/1241",
            "#/texts/1595",
            "#/texts/1596",
            "#/texts/1597",
            "#/texts/1598"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_011"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0115"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_d2c0253519a7",
          "configuration_id": "config_5415060bbfb1",
          "route_label": "HEK293T continual pretraining input",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "continually pre-trained on the X-Atlas/Pisces perturbation corpus spanning four cell types",
          "source_object_verbatim": "HEK293T",
          "source_object_normalized": "HEK293T cell line",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "cell-line transcriptome",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "continually pre-trained on the X-Atlas/Pisces perturbation corpus",
          "fusion_topology": "unclear",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "X-Cell was continually pre-trained on the X-Atlas/Pisces perturbation corpus spanning four cell types (HCT116, HEK293T, HepG2, and iPSC),",
          "section_heading": "2.3 X-Cell generalizes perturbation effects across cellular contexts",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7
          ],
          "doc_item_refs": [
            "#/texts/330"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0009"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f5d7c77286d0",
          "configuration_id": "config_6ed4f4895086",
          "route_label": "HepG2 evaluation input",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "held-out genes from the HepG2 cell line",
          "source_object_verbatim": "HepG2",
          "source_object_normalized": "HepG2 cell line",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "cell-line transcriptome",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "Comparing model performance on 380 held-out genes from the HepG2 cell line",
          "fusion_topology": "unclear",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Comparing model performance on 380 held-out genes from the HepG2 cell line, XCell ranks first",
          "section_heading": "2.3 X-Cell generalizes perturbation effects across cellular contexts",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The exact input serialization is inferred from the perturbation-prediction description.",
          "pages": [
            7
          ],
          "doc_item_refs": [
            "#/texts/333"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0010"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_35251c451da2",
          "configuration_id": "config_91179db657f9",
          "route_label": "iPSC evaluation input",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "On the held-out iPSC-200 perturbations",
          "source_object_verbatim": "iPSC",
          "source_object_normalized": "iPSC cell line",
          "source_modality_normalized": "single-cell transcriptomics",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "cell-line transcriptome",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "X-Cell was applied directly without additional tuning",
          "fusion_topology": "unclear",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "X-Cell was applied directly without additional tuning",
          "section_heading": "2.3 X-Cell generalizes perturbation effects across cellular contexts",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The exact input serialization is inferred from the perturbation-prediction description.",
          "pages": [
            18
          ],
          "doc_item_refs": [
            "#/texts/1203"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0011"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4ce86e93e133",
          "configuration_id": "config_d3c63587fdeb",
          "route_label": "JUMP Cell Painting batch-corrected embedding route",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "incorporated morphological embeddings derived from the JUMP Cell Painting dataset",
          "source_object_verbatim": "batchcorrected embeddings",
          "source_object_normalized": "JUMP Cell Painting batch-corrected morphological embeddings",
          "source_modality_normalized": "morphological phenotype embeddings",
          "transformation_chain_verbatim": [
            "batchcorrected embeddings from a CRISPRi screen performed in the U2OS bone cancer cell line"
          ],
          "model_visible_form_verbatim": "batchcorrected embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "Through cross-attention (Section 4.1.3), these priors dynamically condition perturbation predictions on a unified representation of gene function across biological modalities",
          "fusion_topology": "cross_attention",
          "text_role": "metadata_or_context",
          "input_status": "actual_model_input",
          "evidence_quote": "batchcorrected embeddings from a CRISPRi screen performed in the U2OS bone cancer cell line",
          "section_heading": "4.1 X-Cell Methodology / 4.1.2 Prior Knowledge Sources",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            15,
            46
          ],
          "doc_item_refs": [
            "#/texts/1158",
            "#/texts/1159",
            "#/texts/1160",
            "#/texts/1161",
            "#/texts/1162",
            "#/texts/1163",
            "#/texts/1164",
            "#/texts/1689"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0035"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b894cfca2ae9",
          "configuration_id": "config_3830c0dbaadd",
          "route_label": "iPSC/HepG2-200 evaluation route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "few-shot setup where perturbations are available in multiple cell types, and a proportion of perturbations in one cell type is held out for testing",
          "source_object_verbatim": "the 200 validation perturbations",
          "source_object_normalized": "held-out perturbations from iPSC and HepG2 screens in X-Atlas/Pisces",
          "source_modality_normalized": "single-cell transcriptomic perturbation profiles",
          "transformation_chain_verbatim": [],
          "model_visible_form_verbatim": "the 200 validation perturbations",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "X-Cell was applied directly without additional tuning",
          "fusion_topology": "unclear",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "the 200 validation perturbations were randomly selected",
          "section_heading": "4.1.7 Evaluation Datasets",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            8,
            18,
            19,
            20
          ],
          "doc_item_refs": [
            "#/pictures/2",
            "#/texts/1202",
            "#/texts/1203",
            "#/texts/1204",
            "#/texts/1208",
            "#/texts/1209",
            "#/texts/1210",
            "#/texts/1211"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0036"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_4e7ca98b2c81",
          "configuration_id": "config_42bb192bb897",
          "route_label": "Melanocyte progenitor holdout route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "holding out an entire cell type (Melanocyte Progenitors)",
          "source_object_verbatim": "Melanocyte Progenitors",
          "source_object_normalized": "melanocyte progenitors",
          "source_modality_normalized": "single-cell perturbation data",
          "transformation_chain_verbatim": [
            "held out entirely based on cell type annotations from the iPSC Multi-Diff screen"
          ],
          "model_visible_form_verbatim": "held-out melanocytes from the iPSC Multi-Diff screen",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "Test-time adaptation (TTA)",
          "fusion_topology": "cross_attention",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Perturbations targeting the melanocyte cell type were held out entirely based on cell type annotations from the iPSC Multi-Diff screen",
          "section_heading": "4.1.7 Evaluation Datasets",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The exact per-sample serialization is not named verbatim, but the shared X-Cell continuous-expression interface is explicit.",
          "pages": [
            21
          ],
          "doc_item_refs": [
            "#/texts/1248"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0038"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_610bc0b9c6bb",
          "configuration_id": "config_d2acf290f10d",
          "route_label": "Human primary T cells evaluation route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "extrapolating to primary cells (Human Primary T Cells)",
          "source_object_verbatim": "Human Primary T Cells",
          "source_object_normalized": "primary human CD4+ T cells from four donors",
          "source_modality_normalized": "single-cell perturbation data",
          "transformation_chain_verbatim": [
            "consists of 22 million primary human CD4+ T cells from four donors across three experimental conditions: resting, stimulated for 8 hours, and stimulated for 48 hours",
            "We selected donors D2 and D3 for evaluation",
            "we evaluated a subset of 291 key T cell activation regulators that show differential responses across the three time points"
          ],
          "model_visible_form_verbatim": "a subset of 291 key T cell activation regulators",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "the model receives a learned gene identity embedding",
          "fusion_topology": "retrieval_or_tool_context",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "For each donor, we evaluated a subset of 291 key T cell activation regulators",
          "section_heading": "4.1.7 Evaluation Datasets",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "Carrier/fusion coding is inferred from the shared set-level input encoding in Section 4.1.3.",
          "pages": [
            21
          ],
          "doc_item_refs": [
            "#/texts/1249"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0039"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f5893401e356",
          "configuration_id": "config_f19dfc5333f7",
          "route_label": "Tahoe subset evaluation route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "evaluating on distinct cell lines with chemical perturbations (Tahoe Subset) without observing perturbations in the test context",
          "source_object_verbatim": "Tahoe Subset",
          "source_object_normalized": "Tahoe-100M subset of chemical drug perturbation data",
          "source_modality_normalized": "single-cell chemical perturbation data",
          "transformation_chain_verbatim": [
            "We filtered the drug conditions to include all the drugs (1) having one single target MOA gene, (2) with mid-range dosage, and (3) being an inhibitor of the MOA gene",
            "We also use the data subset of the second plate",
            "The remaining final subset has 50 cell lines and 12 drug perturbations"
          ],
          "model_visible_form_verbatim": "MOA gene as the perturbed target and the control cell-line transcriptome as input",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "X-Cell encodes the MOA gene as the perturbed target and the control cell-line transcriptome as input to predict drug-induced expression responses",
          "fusion_topology": "cross_attention",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "X-Cell encodes the MOA gene as the perturbed target and the control cell-line transcriptome as input",
          "section_heading": "4.1.7 Evaluation Datasets",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            7,
            8,
            9
          ],
          "doc_item_refs": [
            "#/texts/334"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0040"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_d445170f41eb"
    },
    {
      "model_id": "model_3c62ea9d84a1",
      "model_name": "X-Cell-Ultra",
      "record_id": "full_2026-07-06__rec_003517",
      "collection_batch_id": "full_2026-07-06",
      "collection_date": "2026-07-06",
      "review_iteration": "2026-07-06",
      "study_id": "study_bcfca4a74dd0",
      "paper_title": "X-Cell: Scaling Causal Perturbation Prediction Across Diverse Cellular Contexts via Diffusion Language Models",
      "doi": "10.64898/2026.03.18.712807",
      "paper_url": "https://doi.org/10.64898/2026.03.18.712807",
      "route_count": 9,
      "configuration_count": 5,
      "family_counts": {
        "dense_continuous_carrier": 9
      },
      "subtype_counts": {
        "pooled_or_aggregated_embedding": 2,
        "direct_projected_embedding": 7
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "pooled_or_aggregated_embedding"
      ],
      "primary_subtype": "direct_projected_embedding",
      "modalities": [
        "single-cell perturbation data",
        "single-cell transcriptomic expression profiles"
      ],
      "lifecycle_phases": [
        "evaluation",
        "pretraining"
      ],
      "fusion_topologies": [
        "cross_attention",
        "shared_latent_alignment",
        "side_or_generative_conditioning"
      ],
      "text_roles": [
        "biological_payload",
        "no_text_on_this_route",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/full_2026_07_06_rec_003517_figure_005.png",
        "source_path": "data/living_catalog/taxonomy_rerun_preflight_2026-08-12/baseline_vlm_profiles/figures/full_2026_07_06_rec_003517_bdcbcbd932f3/figure_005.png",
        "figure_index": 5,
        "caption": "Fig. 5 X-Cell-Ultra generalizes to melanocyte progenitors and primary human T cells in a zero-shot manner. (A) Schematic shows pre-training of X-Cell-Ultra on X-Atlas/Pisces datasets, excluding the melanocyte progenitor population from the iPSC Multi-Diff screen. We evaluated X-Cell-Ultra in a zero-shot manner on the melanocyte progenitor population as well as primary human T cell datasets from Zhu et al. [57]. We used test-time adaptation to calibrate the models on these datasets using self-supervised objectives. (B) Test loss follows a power law with trainable parameters ( L ( N ) ∝ N -0 . 03 , R 2 = 0 . 97; Table A4). Five models (83M-3.1B) were trained for 20 epochs on the Replogle-Nadig dataset under identical hyperparameters. Circle size is proportional to parameter count; color indicates DE Pearson r . Line plots (right) show how X-Cell-Architecture (blue) and X-Cell-Ultra-Architecture (pink) compare with respect to collapse metrics δ -norm ratio (top) and Pearson correlation of differentially expressed genes (bottom) on test set. (C) Performance of the control mean baseline, perturbation mean baseline, STATE, X-Cell, and X-Cell-Ultra with respect to Pearson ∆ (left), differential expression direction match (middle), and mean absolute error (right). Histogram (top) shows the metrics within 1,341 perturbations and the barplots (bottom) show mean ± 95% C.I. (D) Boxplots show the distribution of Pearson ∆ (left), differential expression direction match (middle), and mean absolute error (right) for 291 perturbations in resting and stimulated (2 time points) primary human T cells for 2 different donors. (E) Heatmaps show the log 2 fold change in expression of APPL2 in D2 (left) and D3 (right) 48 hours post-stimulation as observed by the data and predicted by STATE, X-Cell, and X-Cell-Ultra.",
        "description": "SCIENTIFIC_FIGURE\n\nMulti-panel scientific figure about **X-Cell-Ultra**, a model for genome-scale zero-shot inference in human T cells.\n\nPanel A:\n- Schematic workflow labeled **X-Cell-Ultra**, **Pre-train**, and **Genome-scale Zero-shot Inference**.\n- Biological source objects include human cell lines / cell contexts labeled **HCT116**, **HEK293T**, **HepG2**, and **iPSC**, with perturbation cartoons.\n- Pretraining section shows **Jurkat resting**, **Jurkat active**, and **iPSC multi-diff** cell states.\n- The model appears to combine cell-state representations and perturbation/context information.\n- Inference example is for **Primary human T cell**, with donors/contexts labeled **D2** and **D3**.\n- Outputs include predicted responses for perturbations such as **8hr**, **Stim 48hr**, and NTC-related conditions.\n- A **Test-time Adaptation** module is shown with frozen cross-attention and reduced learning rate, using a query/key/value-style interface.\n\nPanel B:\n- Model scaling plots labeled **Model Scaling Dynamics (Replogle-Nadig)**.\n- Main scatter/line plot shows **Test Loss** decreasing as **Trainable Parameters** increase, with model sizes labeled approximately **83M**, **216M**, **543M**, **1.6B**, and **3.1B**.\n- Pearson correlation color scale is shown.\n- Side plots compare **X-Cell (55M)** and **X-Cell-Ultra (4.9B)** over epochs.\n- Metrics include **Pearson ρ** and **DE Pearson ρ**.\n- X-Cell-Ultra improves over epochs, with annotated gains around **2.3x** and **2.7x**.\n\nPanel C:\n- Evaluation labeled **Melanocyte Progenitor: 1,341 Test Perturbations**.\n- Histograms and bar charts compare methods on **Pearson Δ**, **DE Direction Match**, and **MAE**.\n- Methods include **Control mean**, **Pert. mean**, **STATE**, **X-Cell**, and **X-Cell-Ultra**.\n- X-Cell-Ultra appears to achieve higher Pearson Δ and DE direction match, and lower MAE than X-Cell.\n\nPanel D:\n- Evaluation labeled **291 Test Perturbations, 2 Donors, 3 Contexts**.\n- Boxplots compare **Control mean**, **Pert. mean**, **STATE**, **X-Cell**, and **X-Cell-Ultra**.\n- Metrics shown are **Pearson Δ**, **DE Direction Match**, and **MAE**.\n- Legend includes contexts such as **D2 Rest**, **D3 Rest**, **D2 Stim8hr**, **D3 Stim8hr**, **D2 Stim48hr**, and **D3 Stim48hr**.\n- X-Cell-Ultra generally shows stronger Pearson Δ and DE direction match and lower MAE.\n\nPanel E:\n- Heatmaps labeled **APPL2 D2 Stim 48h log₂FC** and **APPL2 D3 Stim 48h log₂FC**.\n- Rows compare **STATE**, donor/context measurements such as **D2 48h Stim** or **D3 48h Stim**, **X-Cell-Ultra**, and **X-Cell**.\n- Color scale ranges from blue to red around zero, representing log₂ fold-change values.\n- Dendrograms show clustering; X-Cell-Ultra appears closer to observed stimulated donor profiles than X-Cell in these examples.",
        "page_no": 11,
        "sha256": "5357081b6bd101bd9be2dd3bd20a4d59b7c89edcc29b086eacc9ae16bd822793",
        "pixel_width": 891,
        "pixel_height": 618,
        "crop_box": {
          "x": 0,
          "y": 0,
          "width": 0.57,
          "height": 0.5
        },
        "panel_label": "A",
        "visible_input_object": "melanocyte progenitor control cell sets from the iPSC Multi-Diff screen",
        "visible_model_interface": "Test-time Adaptation module with Q x KV, frozen cross-attention, and reduced learning rate",
        "suitability": "suitable",
        "confidence": "medium",
        "rationale": "Crop panel A only, keeping the source-cell labels, the melanocyte progenitor reference, and the test-time adaptation interface; this excludes the output-only panels B-E while preserving the grounded input route and its immediate fusion/adaptation mechanism.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_a11110cf2aec",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "the substantially larger X-Atlas/Pisces compendium",
          "actual_model_visible_form": "X-Atlas/Pisces perturbation examples"
        },
        {
          "subtype_id": "pooled_or_aggregated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_466f3e6b0aa2",
          "example_input": "{gene/cell/patch embeddings}",
          "example_carrier": "mean/attention pool = one compact vector",
          "example_interface": "aggregator → generator",
          "actual_source": "melanocyte progenitor control cell sets from the iPSC Multi-Diff screen",
          "actual_model_visible_form": "control cell sets from the unseen melanocyte progenitor domain"
        }
      ],
      "routes": [
        {
          "route_id": "route_466f3e6b0aa2",
          "configuration_id": "config_d58a50cbf3a9",
          "route_label": "Melanocyte progenitor zero-shot TTA route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "held-out entire cell type with Test Time Adaptation on unseen melanocyte progenitors",
          "source_object_verbatim": "melanocyte progenitor control cell sets from the iPSC Multi-Diff screen",
          "source_object_normalized": "melanocyte progenitor control cell sets",
          "source_modality_normalized": "single-cell perturbation data",
          "transformation_chain_verbatim": [
            "held-out cell type from iPSC Multi-Diff",
            "test-time adaptation on control-to-control pairs",
            "MMD loss only",
            "self-attention adapted while cross-attention frozen/skipped"
          ],
          "model_visible_form_verbatim": "control cell sets from the unseen melanocyte progenitor domain",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "test-time adaptation",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "In the zero-shot setting, we tested a more challenging setup by holding out an entire cell type ( Melanocyte Progenitors ), extrapolating to primary cells ( Human Primary T Cells ), or evaluating on distinct cell lines with chemical perturbations ( Tahoe Subset ) without observing perturbations in the test context (Section 4.1.7). In this setting, X-Cell and X-Cell-Ultra were lightly tuned on test control cells using Test Time Adaptation (TTA) (Methods 4.1.3). We report a STATE model pre-trained on the original Replogle-Nadig datasets [12][52] with full transcriptomic readout and the H1 screen from the Virtual Cell Challenge (VCC) release [55] (Section 4.1.5), representing a version of STATE trained on large-scale public data. Because the Melanocyte Progenitors test set was generated on the FLEX platform, as was the H1 screen, this combination helps align the STATE model with both the X-Cell-Ultra training corpus and the test data distribution. In the zero-shot setting, we also report two baselines, Control Mean and Pert Mean , which represent data-only predictions constructed using test control cells and log-fold changes from X-Atlas/Pisces (Section 4.1.6).",
          "section_heading": "2.5 X-Cell-Ultra follows universal neural scaling laws, charting a path to further performance gains by increasing compute and data",
          "supporting_figure_or_table": "Figure 5A",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper describes the TTA mechanism generically; this route specializes it to the melanocyte-progenitor holdout.",
          "pages": [
            14,
            15,
            17,
            18,
            41,
            42
          ],
          "doc_item_refs": [
            "#/texts/1152",
            "#/texts/1153",
            "#/texts/1154",
            "#/texts/1156",
            "#/texts/1197",
            "#/texts/1199",
            "#/texts/1200",
            "#/texts/1202",
            "#/texts/1203",
            "#/texts/1204",
            "#/texts/1595"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_012"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0038"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8abe04d840b4",
          "configuration_id": "config_7350e5ff691d",
          "route_label": "Primary CD4+ T cell zero-shot TTA route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "primary human CD4 + T cells across donors and stimulation states with Test Time Adaptation",
          "source_object_verbatim": "primary human CD4 + T cell control cell sets",
          "source_object_normalized": "primary human CD4 + T cell control cell sets",
          "source_modality_normalized": "single-cell perturbation data",
          "transformation_chain_verbatim": [
            "two donors (D2 and D3)",
            "resting, 8-hour stimulated, and 48-hour stimulated states",
            "test-time adaptation on unlabeled control cells",
            "MMD loss only"
          ],
          "model_visible_form_verbatim": "control cell sets from primary human CD4 + T cells",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "test-time adaptation",
          "fusion_topology": "shared_latent_alignment",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "we evaluated on 291 perturbations of T cell regulators from primary CD4 + T cells across two donors in resting and stimulated states (8 and 48 hours post CD3/CD28 stimulation)",
          "section_heading": "4.1.7 Evaluation Datasets",
          "supporting_figure_or_table": "Figure 5D",
          "evidence_status": "explicit_text",
          "uncertainty": "The paper applies the same TTA procedure across unseen domains; this route isolates the primary T-cell evaluation setting.",
          "pages": [
            11,
            18,
            20,
            21
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/1128",
            "#/texts/1202",
            "#/texts/1203",
            "#/texts/1204",
            "#/texts/1242",
            "#/texts/1243",
            "#/texts/1244",
            "#/texts/1245",
            "#/texts/1247"
          ],
          "source_candidate_refs": [
            "full_2026-07-06__rec_003517::route_013"
          ],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0039"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_a11110cf2aec",
          "configuration_id": "config_4b359f05620b",
          "route_label": "X-Cell-Ultra pretraining on X-Atlas/Pisces",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "X-Cell scales from 55M parameters to 4.9B parameters",
          "source_object_verbatim": "the substantially larger X-Atlas/Pisces compendium",
          "source_object_normalized": "X-Atlas/Pisces perturbation compendium",
          "source_modality_normalized": "single-cell perturbation data",
          "transformation_chain_verbatim": [
            "diffusion LM training",
            "cross-attention to prior knowledge",
            "iterative diffusion via remasking"
          ],
          "model_visible_form_verbatim": "X-Atlas/Pisces perturbation examples",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "through cross-attention",
          "fusion_topology": "cross_attention",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "Input consists of X-Atlas/Pisces perturbation examples.",
          "section_heading": "2.3 X-Cell generalizes perturbation effects across cellular contexts",
          "supporting_figure_or_table": "Figure 1D",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            4
          ],
          "doc_item_refs": [
            "#/pictures/0"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0003"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_cb00fd160f0d",
          "configuration_id": "config_1f75484a3514",
          "route_label": "D2 Rest context route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "291 Test Perturbations, 2 Donors, 3 Contexts",
          "source_object_verbatim": "D2 Rest",
          "source_object_normalized": "donor D2 resting-state primary human CD4+ T cells",
          "source_modality_normalized": "single-cell transcriptomic expression profiles",
          "transformation_chain_verbatim": [
            "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations"
          ],
          "model_visible_form_verbatim": "32 fixed control cells",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "D2 Rest",
          "section_heading": "2.5 X-Cell-Ultra follows universal neural scaling laws, charting a path to further performance gains by increasing compute and data",
          "supporting_figure_or_table": "Figure 5D",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            11
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/1033"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0029"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_8a01f73645e7",
          "configuration_id": "config_1f75484a3514",
          "route_label": "D3 Rest context route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "291 Test Perturbations, 2 Donors, 3 Contexts",
          "source_object_verbatim": "D3 Rest",
          "source_object_normalized": "donor D3 resting-state primary human CD4+ T cells",
          "source_modality_normalized": "single-cell transcriptomic expression profiles",
          "transformation_chain_verbatim": [
            "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations"
          ],
          "model_visible_form_verbatim": "32 fixed control cells",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "D3 Rest",
          "section_heading": "2.5 X-Cell-Ultra follows universal neural scaling laws, charting a path to further performance gains by increasing compute and data",
          "supporting_figure_or_table": "Figure 5D",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            11
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/1030"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0030"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2b4d0b79d9e9",
          "configuration_id": "config_1f75484a3514",
          "route_label": "D2 Stim8hr context route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "291 Test Perturbations, 2 Donors, 3 Contexts",
          "source_object_verbatim": "D2 Stim8hr",
          "source_object_normalized": "donor D2 8-hour stimulated primary human CD4+ T cells",
          "source_modality_normalized": "single-cell transcriptomic expression profiles",
          "transformation_chain_verbatim": [
            "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations"
          ],
          "model_visible_form_verbatim": "32 fixed control cells",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "D2 Stim8hr",
          "section_heading": "2.5 X-Cell-Ultra follows universal neural scaling laws, charting a path to further performance gains by increasing compute and data",
          "supporting_figure_or_table": "Figure 5D",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            11
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/1035"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0031"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f278942a47d6",
          "configuration_id": "config_1f75484a3514",
          "route_label": "D3 Stim8hr context route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "291 Test Perturbations, 2 Donors, 3 Contexts",
          "source_object_verbatim": "D3 Stim8hr",
          "source_object_normalized": "donor D3 8-hour stimulated primary human CD4+ T cells",
          "source_modality_normalized": "single-cell transcriptomic expression profiles",
          "transformation_chain_verbatim": [
            "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations"
          ],
          "model_visible_form_verbatim": "32 fixed control cells",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "D3 Stim8hr",
          "section_heading": "2.5 X-Cell-Ultra follows universal neural scaling laws, charting a path to further performance gains by increasing compute and data",
          "supporting_figure_or_table": "Figure 5D",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            11
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/1032"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0032"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e0fbfe069c31",
          "configuration_id": "config_21f1208a7750",
          "route_label": "D2 Stim48hr context route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "generalization beyond cell lines to primary human cells; evaluated on primary CD4+ T cells across two donors in resting and stimulated states (8 and 48 hours post CD3/CD28 stimulation)",
          "source_object_verbatim": "D2 Stim48hr",
          "source_object_normalized": "donor D2 48-hour stimulated primary human CD4+ T cells",
          "source_modality_normalized": "single-cell transcriptomic expression profiles",
          "transformation_chain_verbatim": [
            "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations"
          ],
          "model_visible_form_verbatim": "32 fixed control cells",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "D2 Stim48hr",
          "section_heading": "2.5 X-Cell-Ultra follows universal neural scaling laws, charting a path to further performance gains by increasing compute and data",
          "supporting_figure_or_table": "Figure 5D-E",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            11
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/1034"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0033"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_2f2691a4449a",
          "configuration_id": "config_21f1208a7750",
          "route_label": "D3 Stim48hr context route",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "generalization beyond cell lines to primary human cells; evaluated on primary CD4+ T cells across two donors in resting and stimulated states (8 and 48 hours post CD3/CD28 stimulation)",
          "source_object_verbatim": "D3 Stim48hr",
          "source_object_normalized": "donor D3 48-hour stimulated primary human CD4+ T cells",
          "source_modality_normalized": "single-cell transcriptomic expression profiles",
          "transformation_chain_verbatim": [
            "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations"
          ],
          "model_visible_form_verbatim": "32 fixed control cells",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "used 32 fixed control cells randomly selected from NTCs in the test set to generate predictions for all test perturbations",
          "fusion_topology": "side_or_generative_conditioning",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "D3 Stim48hr",
          "section_heading": "2.5 X-Cell-Ultra follows universal neural scaling laws, charting a path to further performance gains by increasing compute and data",
          "supporting_figure_or_table": "Figure 5D-E",
          "evidence_status": "text_plus_figure",
          "uncertainty": null,
          "pages": [
            11
          ],
          "doc_item_refs": [
            "#/pictures/4",
            "#/texts/1031"
          ],
          "source_candidate_refs": [],
          "dense_candidate_refs": [
            "dense::full_2026-07-06__rec_003517::0034"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_d7b71bdfa7dd"
    },
    {
      "model_id": "model_d3b03caa65b1",
      "model_name": "XunZi-M",
      "record_id": "update_2026-08-09__manual_recall_xunzi",
      "collection_batch_id": "update_2026-08-09",
      "collection_date": "2026-08-09",
      "review_iteration": "2026-08-09",
      "study_id": "study_db50018753cc",
      "paper_title": "XunZi, an AI biologist, reveals disease-modifying targets",
      "doi": "10.1038/s41551-026-01769-6",
      "paper_url": "https://doi.org/10.1038/s41551-026-01769-6",
      "route_count": 5,
      "configuration_count": 5,
      "family_counts": {
        "dense_continuous_carrier": 5
      },
      "subtype_counts": {
        "direct_projected_embedding": 1,
        "pooled_or_aggregated_embedding": 4
      },
      "families": [
        "dense_continuous_carrier"
      ],
      "subtypes": [
        "direct_projected_embedding",
        "pooled_or_aggregated_embedding"
      ],
      "primary_subtype": "pooled_or_aggregated_embedding",
      "modalities": [
        "graph/network",
        "multi-omics"
      ],
      "lifecycle_phases": [
        "fine_tuning",
        "pretraining"
      ],
      "fusion_topologies": [
        "concatenation",
        "other_explicit"
      ],
      "text_roles": [
        "no_text_on_this_route"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/update_2026_08_09_manual_recall_xunzi_figure_002.png",
        "source_path": "data/living_catalog_updates/update_2026-08-09/11_docling_vlm_manual_recall_xunzi_2026-08-11/profiles/figures/update_2026_08_09_manual_recall_xunzi_c3ac5dec9fcb/figure_002.png",
        "figure_index": 2,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific figure labeled **a-g** describing **XunZi**, an AI biologist framework combining reasoning and multimodal biological data fusion.\n\nPanel **a** shows the overall architecture:\n- **XunZi-R reasoning module**:\n  - Inputs include **literature**, **gene and disease knowledge**, **Gene Ontology**, and **pathway annotations**.\n  - These feed into **biological knowledge continual pretraining** of a **Mistral 7B** model, producing a **vertical LLM for biology**.\n  - Additional sources include **public databases** and **literature review**, producing a **360K curated gene-disease mechanism dataset**.\n  - A logical reasoning interface is shown with prompt text: “Is gene XX functional in disease YY?” and response fields for **mechanism summary**, **impacted genes**, and **impacted pathways**.\n- **XunZi-M multimodal data fusion**:\n  - Inputs include **protein interactions**, **biological processes**, **knowledge graph**, and **613 TB multi-omics**.\n  - These are transformed through a stacked **fusion** module.\n  - The fused representation feeds a **multimodal foundation model**, **graph convolutional network (GCN)**, and **disease-specific fine-tuning**.\n- Outputs from XunZi-R and XunZi-M feed into **XunZi AI biologist**, which ranks gene-disease-output associations and includes **feedback** and **validation** by human users/clinicians.\n\nPanel **b** is a horizontal bar chart titled **Top 6 disease classes**, showing counts of gene-disease associations, with the largest category **Neoplasms**, followed by **Congenital abnormality**, **Digestive system diseases**, **Nervous system diseases**, **Cardiovascular disease**, and **Urologic and male genital diseases**.\n\nPanel **c** compares classification performance using confusion-matrix-style heatmaps:\n- **XunZi-R**: positive class 90.2%, negative class 94.0%, with smaller off-diagonal errors.\n- **GPT-4o**: positive class 54.1%, negative class 93.5%, with a much larger positive-to-negative error of 45.9%.\n\nPanel **d** shows two circular/radar-style performance plots for **Neoplasms** and **Nervous system diseases**, comparing **XunZi-R** and **GPT-4o**. XunZi-R is shown with a larger blue performance contour than the orange GPT-4o contour.\n\nPanel **e** shows multi-omics dataset scale for **Pan-cancer** and **Neurodegenerative diseases**, comparing **sample count** and **data size (TB)**. Visible labels include **14,380 samples / 579.3 TB** for pan-cancer and **9,897 samples / 33.9 TB** for neurodegenerative diseases.\n\nPanel **f** is an ROC-like curve for **Pan-cancer**, plotting **Sn** versus **1-Sp**, comparing models including **XunZi**, **XunZi-R**, **XunZi-M**, **GPT-4o**, **DNN**, and **SVM**. XunZi has the highest reported AUC, **0.85**.\n\nPanel **g** is a similar ROC-like curve for **Neurodegenerative diseases**, with the same model types. XunZi again performs best, with reported AUC **0.80**.",
        "page_no": 3,
        "sha256": "2c425dedf484873b78cc84e1de40ae94575339ffe53a6c61a8b541e583ba0f68",
        "pixel_width": 946,
        "pixel_height": 1289,
        "crop_box": {
          "x": 0.05,
          "y": 0.29,
          "width": 0.64,
          "height": 0.27
        },
        "panel_label": "a",
        "visible_input_object": "Protein interactions, biological processes, knowledge graph, and 613 TB multi-omics",
        "visible_model_interface": "Fusion module feeding a multimodal foundation model and graph convolutional network with disease-specific fine-tuning",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop isolates the XunZi-M multimodal data-fusion block in panel a, keeping the source objects, arrows, fusion stack, and GCN interface readable while excluding the output-only AI biologist panel and the unrelated lower panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "direct_projected_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_0e824ef8ec84",
          "example_input": "RNA vector x ∈ ℝᵍ",
          "example_carrier": "Wproj x = [0.14, −0.82, 0.31, ...]",
          "example_interface": "projected vectors → generative backbone",
          "actual_source": "protein interactions and biological process annotations",
          "actual_model_visible_form": "heterogeneous graph nodes and edges with learned node embeddings"
        },
        {
          "subtype_id": "pooled_or_aggregated_embedding",
          "family_id": "dense_continuous_carrier",
          "route_id": "route_eb28b1c638ed",
          "example_input": "{gene/cell/patch embeddings}",
          "example_carrier": "mean/attention pool = one compact vector",
          "example_interface": "aggregator → generator",
          "actual_source": "pan-cancer multi-omics datasets",
          "actual_model_visible_form": "gene-node feature vectors derived from expression profiles across 33 cancer types"
        }
      ],
      "routes": [
        {
          "route_id": "route_0e824ef8ec84",
          "configuration_id": "config_0b7af8503a56",
          "route_label": "Protein interaction and GO knowledge graph",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "construction of a knowledge graph for protein interactions and their involved biological processes",
          "source_object_verbatim": "protein interactions and biological process annotations",
          "source_object_normalized": "protein-function knowledge graph",
          "source_modality_normalized": "graph/network",
          "transformation_chain_verbatim": [
            "map protein entries to standardized gene identifiers",
            "compile PPI edges and GO-term associations",
            "construct a heterogeneous knowledge graph",
            "embed nodes in 4,096-dimensional vectors",
            "train graph convolutional layers"
          ],
          "model_visible_form_verbatim": "heterogeneous graph nodes and edges with learned node embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "direct_projected_embedding",
          "insertion_or_fusion_verbatim": "graph-based representation learning with a GCN",
          "fusion_topology": "other_explicit",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "constructed a large-scale protein-function knowledge graph",
          "section_heading": "Construction of a knowledge graph for protein interactions and their involved biological processes",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": "The frozen taxonomy has no graph/network family, so this knowledge-graph route is represented by the learned node-embedding carrier used in the GCN.",
          "pages": [
            11
          ],
          "doc_item_refs": [
            "#/texts/1375"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__manual_recall_xunzi::route_005"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__manual_recall_xunzi::0012"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_eb28b1c638ed",
          "configuration_id": "config_faa9d51cedf6",
          "route_label": "Pan-cancer multi-omics model input",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "pan-cancer model",
          "source_object_verbatim": "pan-cancer multi-omics datasets",
          "source_object_normalized": "pan-cancer multi-omics datasets",
          "source_modality_normalized": "multi-omics",
          "transformation_chain_verbatim": [
            "derive gene-node expression profiles across 33 cancer types",
            "standardize values within study",
            "average across samples under the same condition",
            "train the pan-cancer GCN"
          ],
          "model_visible_form_verbatim": "gene-node feature vectors derived from expression profiles across 33 cancer types",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "feature-level input to a graph convolutional network",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "For pan-cancer, each gene node was embedded with a feature vector derived from its expression profiles across 33 cancer types.",
          "section_heading": "Implementation of XunZi-M",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13
          ],
          "doc_item_refs": [
            "#/texts/1415",
            "#/texts/1416",
            "#/texts/1417",
            "#/texts/1418",
            "#/texts/1419",
            "#/texts/1420",
            "#/texts/1421",
            "#/texts/1422"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__manual_recall_xunzi::route_006"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__manual_recall_xunzi::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_cc7391bead65",
          "configuration_id": "config_a108568db46c",
          "route_label": "NSCLC-specific multi-omics fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "NSCLC-specific model",
          "source_object_verbatim": "NSCLC-specific multi-omics expression vectors",
          "source_object_normalized": "NSCLC-specific multi-omics expression vectors",
          "source_modality_normalized": "multi-omics",
          "transformation_chain_verbatim": [
            "extract the final-layer embedding from the pretrained pan-cancer GCN model",
            "concatenate it with NSCLC-specific multi-omics expression vectors",
            "train a two-layer GCN for NSCLC"
          ],
          "model_visible_form_verbatim": "concatenated molecular representation for NSCLC genes",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "concatenation with pretrained pan-cancer node embeddings",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "These were then concatenated with NSCLC-specific multi-omics expression vectors.",
          "section_heading": "Implementation of XunZi-M",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13
          ],
          "doc_item_refs": [
            "#/texts/1433",
            "#/texts/1434",
            "#/texts/1435",
            "#/texts/1436",
            "#/texts/1437",
            "#/texts/1438",
            "#/texts/1439"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__manual_recall_xunzi::route_007"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_f07cba1a7db0",
          "configuration_id": "config_00bb54889975",
          "route_label": "Neurodegenerative multi-omics model input",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "neurodegenerative disease model",
          "source_object_verbatim": "neurodegenerative multi-omics datasets",
          "source_object_normalized": "neurodegenerative multi-omics datasets",
          "source_modality_normalized": "multi-omics",
          "transformation_chain_verbatim": [
            "derive gene-node expression across distinct brain regions under case and control conditions",
            "concatenate region-specific vectors",
            "train the neurodegenerative GCN"
          ],
          "model_visible_form_verbatim": "gene feature vectors derived from expression across distinct brain regions",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "feature-level input to a graph convolutional network",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "where e case ij and e control ik represent the normalized expression of gene i in case and control samples within region r , respectively. The full feature vector xi for each gene was constructed by concatenating these region-specific vectors across multiple brain regions.",
          "section_heading": "Implementation of XunZi-M",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13
          ],
          "doc_item_refs": [
            "#/texts/1415",
            "#/texts/1416",
            "#/texts/1417",
            "#/texts/1418",
            "#/texts/1419",
            "#/texts/1420",
            "#/texts/1421",
            "#/texts/1422",
            "#/texts/1433",
            "#/texts/1434",
            "#/texts/1435",
            "#/texts/1436",
            "#/texts/1437",
            "#/texts/1438",
            "#/texts/1439"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__manual_recall_xunzi::route_008"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__manual_recall_xunzi::0017"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_ad3fc46f8b34",
          "configuration_id": "config_faa1e7b480f4",
          "route_label": "PD-specific multi-omics fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "PD-specific subgraph classification",
          "source_object_verbatim": "PD-specific omics features",
          "source_object_normalized": "PD-specific omics features",
          "source_modality_normalized": "multi-omics",
          "transformation_chain_verbatim": [
            "combine pretrained node embeddings with PD-specific omics features",
            "construct a PD-specific subgraph",
            "train a two-layer GCN to classify PD-relevant genes and kinases"
          ],
          "model_visible_form_verbatim": "combined PD subgraph features with pretrained node embeddings",
          "carrier_family": "dense_continuous_carrier",
          "carrier_subtype": "pooled_or_aggregated_embedding",
          "insertion_or_fusion_verbatim": "fusion of pretrained node embeddings with PD-specific omics features",
          "fusion_topology": "concatenation",
          "text_role": "no_text_on_this_route",
          "input_status": "actual_model_input",
          "evidence_quote": "we constructed a PD-specific subgraph by combining the pretrained node embeddings with PD-specific omics features.",
          "section_heading": "Implementation of XunZi-M",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            13
          ],
          "doc_item_refs": [
            "#/texts/1433",
            "#/texts/1434",
            "#/texts/1435",
            "#/texts/1436",
            "#/texts/1437",
            "#/texts/1438",
            "#/texts/1439"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__manual_recall_xunzi::route_009"
          ],
          "dense_candidate_refs": [],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_d7b71bdfa7dd"
    },
    {
      "model_id": "model_0b0332ce2b16",
      "model_name": "XunZi-R",
      "record_id": "update_2026-08-09__manual_recall_xunzi",
      "collection_batch_id": "update_2026-08-09",
      "collection_date": "2026-08-09",
      "review_iteration": "2026-08-09",
      "study_id": "study_db50018753cc",
      "paper_title": "XunZi, an AI biologist, reveals disease-modifying targets",
      "doi": "10.1038/s41551-026-01769-6",
      "paper_url": "https://doi.org/10.1038/s41551-026-01769-6",
      "route_count": 5,
      "configuration_count": 4,
      "family_counts": {
        "text_native_token_stream": 5
      },
      "subtype_counts": {
        "serialized_biological_context_or_ordered_profile": 2,
        "structured_biological_prompt_or_task_scaffold": 2,
        "plain_language_prompt_or_question": 1
      },
      "families": [
        "text_native_token_stream"
      ],
      "subtypes": [
        "plain_language_prompt_or_question",
        "serialized_biological_context_or_ordered_profile",
        "structured_biological_prompt_or_task_scaffold"
      ],
      "primary_subtype": "structured_biological_prompt_or_task_scaffold",
      "modalities": [
        "text"
      ],
      "lifecycle_phases": [
        "evaluation",
        "fine_tuning",
        "pretraining"
      ],
      "fusion_topologies": [
        "tokenizer_sequence"
      ],
      "text_roles": [
        "biological_payload",
        "instruction_or_query",
        "paired_alignment_supervision"
      ],
      "figure": {
        "status": "cropped_source_figure",
        "asset": "assets/figures/update_2026_08_09_manual_recall_xunzi_figure_002.png",
        "source_path": "data/living_catalog_updates/update_2026-08-09/11_docling_vlm_manual_recall_xunzi_2026-08-11/profiles/figures/update_2026_08_09_manual_recall_xunzi_c3ac5dec9fcb/figure_002.png",
        "figure_index": 2,
        "caption": "",
        "description": "SCIENTIFIC_FIGURE\n\nThe image is a multi-panel scientific figure labeled **a-g** describing **XunZi**, an AI biologist framework combining reasoning and multimodal biological data fusion.\n\nPanel **a** shows the overall architecture:\n- **XunZi-R reasoning module**:\n  - Inputs include **literature**, **gene and disease knowledge**, **Gene Ontology**, and **pathway annotations**.\n  - These feed into **biological knowledge continual pretraining** of a **Mistral 7B** model, producing a **vertical LLM for biology**.\n  - Additional sources include **public databases** and **literature review**, producing a **360K curated gene-disease mechanism dataset**.\n  - A logical reasoning interface is shown with prompt text: “Is gene XX functional in disease YY?” and response fields for **mechanism summary**, **impacted genes**, and **impacted pathways**.\n- **XunZi-M multimodal data fusion**:\n  - Inputs include **protein interactions**, **biological processes**, **knowledge graph**, and **613 TB multi-omics**.\n  - These are transformed through a stacked **fusion** module.\n  - The fused representation feeds a **multimodal foundation model**, **graph convolutional network (GCN)**, and **disease-specific fine-tuning**.\n- Outputs from XunZi-R and XunZi-M feed into **XunZi AI biologist**, which ranks gene-disease-output associations and includes **feedback** and **validation** by human users/clinicians.\n\nPanel **b** is a horizontal bar chart titled **Top 6 disease classes**, showing counts of gene-disease associations, with the largest category **Neoplasms**, followed by **Congenital abnormality**, **Digestive system diseases**, **Nervous system diseases**, **Cardiovascular disease**, and **Urologic and male genital diseases**.\n\nPanel **c** compares classification performance using confusion-matrix-style heatmaps:\n- **XunZi-R**: positive class 90.2%, negative class 94.0%, with smaller off-diagonal errors.\n- **GPT-4o**: positive class 54.1%, negative class 93.5%, with a much larger positive-to-negative error of 45.9%.\n\nPanel **d** shows two circular/radar-style performance plots for **Neoplasms** and **Nervous system diseases**, comparing **XunZi-R** and **GPT-4o**. XunZi-R is shown with a larger blue performance contour than the orange GPT-4o contour.\n\nPanel **e** shows multi-omics dataset scale for **Pan-cancer** and **Neurodegenerative diseases**, comparing **sample count** and **data size (TB)**. Visible labels include **14,380 samples / 579.3 TB** for pan-cancer and **9,897 samples / 33.9 TB** for neurodegenerative diseases.\n\nPanel **f** is an ROC-like curve for **Pan-cancer**, plotting **Sn** versus **1-Sp**, comparing models including **XunZi**, **XunZi-R**, **XunZi-M**, **GPT-4o**, **DNN**, and **SVM**. XunZi has the highest reported AUC, **0.85**.\n\nPanel **g** is a similar ROC-like curve for **Neurodegenerative diseases**, with the same model types. XunZi again performs best, with reported AUC **0.80**.",
        "page_no": 3,
        "sha256": "2c425dedf484873b78cc84e1de40ae94575339ffe53a6c61a8b541e583ba0f68",
        "pixel_width": 946,
        "pixel_height": 1289,
        "crop_box": {
          "x": 0.07,
          "y": 0.02,
          "width": 0.46,
          "height": 0.29
        },
        "panel_label": "a",
        "visible_input_object": "Literature, gene and disease knowledge, Gene Ontology, and pathway annotations feeding biological knowledge continual pretraining into Mistral 7B.",
        "visible_model_interface": "XunZi-R reasoning module pretraining block with the biological knowledge module and vertical LLM for biology.",
        "suitability": "suitable",
        "confidence": "high",
        "rationale": "This crop keeps the left XunZi-R input-to-pretraining pathway: source biological text/knowledge, the continual-pretraining transformation, and the resulting biology LLM. It excludes the right-side reasoning/output branch and all lower evaluation panels.",
        "annotation_pass": "two_blind_selectors_plus_adjudicator_and_cropper"
      },
      "figure_status": "cropped_source_figure",
      "no_figure_rationale": "",
      "illustrative_examples": [
        {
          "subtype_id": "plain_language_prompt_or_question",
          "family_id": "text_native_token_stream",
          "route_id": "route_499acf7a1163",
          "example_input": "What phenotype does this cell exhibit?",
          "example_carrier": "[What] [phenotype] [does] [this] [cell] ...",
          "example_interface": "ordinary tokenizer → LLM",
          "actual_source": "gene-disease relationship query",
          "actual_model_visible_form": "natural language instruction prompt"
        },
        {
          "subtype_id": "serialized_biological_context_or_ordered_profile",
          "family_id": "text_native_token_stream",
          "route_id": "route_c8d792c85a6b",
          "example_input": "GAPDH|8.2; ACTB|7.9; IL7R|6.4; ...",
          "example_carrier": "ordered textual cell/profile sentence",
          "example_interface": "serializer → tokenizer → LLM",
          "actual_source": "curated biomedical publications from PubMed",
          "actual_model_visible_form": "tokenized biomedical publication text sequences"
        },
        {
          "subtype_id": "structured_biological_prompt_or_task_scaffold",
          "family_id": "text_native_token_stream",
          "route_id": "route_b1ff21bb7a67",
          "example_input": "<TASK:cell_type> genes: IL7R, LTB, MALAT1",
          "example_carrier": "task tag + labeled biological fields",
          "example_interface": "template tokenizer → LLM",
          "actual_source": "curated mechanistic interpretations corpus",
          "actual_model_visible_form": "structured input-output pairs with instruction prompts, binary labels and reasoning traces"
        }
      ],
      "routes": [
        {
          "route_id": "route_c8d792c85a6b",
          "configuration_id": "config_869b8b6ea097",
          "route_label": "PubMed biomedical publication pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "biological knowledge continual pretraining",
          "source_object_verbatim": "curated biomedical publications from PubMed",
          "source_object_normalized": "biomedical publication corpus",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "retrieve abstracts from PubMed before 2025",
            "tokenize publication text",
            "continual pretraining"
          ],
          "model_visible_form_verbatim": "tokenized biomedical publication text sequences",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "causal language modeling on the Mistral-7B backbone",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "24,411,924 curated biomedical publications",
          "section_heading": "Preparation of the biomedical knowledge data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            10
          ],
          "doc_item_refs": [
            "#/texts/1349",
            "#/texts/1350",
            "#/texts/22",
            "#/texts/23"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__manual_recall_xunzi::route_001"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__manual_recall_xunzi::0001",
            "dense::update_2026-08-09__manual_recall_xunzi::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_e70933541045",
          "configuration_id": "config_869b8b6ea097",
          "route_label": "Structured biomedical corpus pretraining",
          "lifecycle_phase": "pretraining",
          "task_or_configuration_verbatim": "biological knowledge continual pretraining",
          "source_object_verbatim": "structured biological corpora, including gene functional descriptions, Disease Ontology definitions and Gene Ontology definitions",
          "source_object_normalized": "structured biological corpus",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "collect structured biological corpora",
            "normalize textual definitions and descriptions",
            "continual pretraining"
          ],
          "model_visible_form_verbatim": "tokenized structured biological text",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "serialized_biological_context_or_ordered_profile",
          "insertion_or_fusion_verbatim": "causal language modeling on the Mistral-7B backbone",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "biological_payload",
          "input_status": "actual_model_input",
          "evidence_quote": "2,054,130 structured biological corpus data",
          "section_heading": "Preparation of the biomedical knowledge data",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            2,
            10
          ],
          "doc_item_refs": [
            "#/texts/1349",
            "#/texts/1350",
            "#/texts/22",
            "#/texts/23"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__manual_recall_xunzi::route_002"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__manual_recall_xunzi::0001",
            "dense::update_2026-08-09__manual_recall_xunzi::0007"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_b1ff21bb7a67",
          "configuration_id": "config_30f586c7b02e",
          "route_label": "Gene-disease mechanistic corpus fine-tuning",
          "lifecycle_phase": "fine_tuning",
          "task_or_configuration_verbatim": "CoT-style gene-disease mechanistic corpus",
          "source_object_verbatim": "curated mechanistic interpretations corpus",
          "source_object_normalized": "curated mechanistic interpretations corpus",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "generate mechanistic interpretations from gene-disease associations and literature evidence",
            "manually verify and correct errors",
            "format as instruction, label, and reasoning trace",
            "fine-tune the reasoning model"
          ],
          "model_visible_form_verbatim": "structured input-output pairs with instruction prompts, binary labels and reasoning traces",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "supervised fine-tuning on the reasoning module",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "paired_alignment_supervision",
          "input_status": "paired_alignment_input",
          "evidence_quote": "Each training instance was formatted as a structured input-output pair",
          "section_heading": "Construction and curation of the gene-disease mechanistic corpus",
          "supporting_figure_or_table": "Supplementary Table 2a",
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            11,
            12
          ],
          "doc_item_refs": [
            "#/texts/1352",
            "#/texts/1353",
            "#/texts/1354",
            "#/texts/1355",
            "#/texts/1356",
            "#/texts/1357",
            "#/texts/1358",
            "#/texts/1359",
            "#/texts/1361",
            "#/texts/1362",
            "#/texts/1366",
            "#/texts/1367",
            "#/texts/1368",
            "#/texts/1369",
            "#/texts/1370",
            "#/texts/1371",
            "#/texts/1377",
            "#/texts/1378",
            "#/texts/1382",
            "#/texts/1383",
            "#/texts/1384",
            "#/texts/1385",
            "#/texts/1386",
            "#/texts/1387"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__manual_recall_xunzi::route_003"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__manual_recall_xunzi::0008",
            "dense::update_2026-08-09__manual_recall_xunzi::0010",
            "dense::update_2026-08-09__manual_recall_xunzi::0013"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_499acf7a1163",
          "configuration_id": "config_733649742d5d",
          "route_label": "Gene-disease query prompting",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "classification of functional gene-disease associations",
          "source_object_verbatim": "gene-disease relationship query",
          "source_object_normalized": "gene-disease relationship query",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "apply standardized yes/no prompt"
          ],
          "model_visible_form_verbatim": "natural language instruction prompt",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "plain_language_prompt_or_question",
          "insertion_or_fusion_verbatim": "prompting through the logical reasoning interface",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Is gene [X] involved in disease [Y] in a functional way? Start with Yes or No.",
          "section_heading": "Benchmarking",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            12
          ],
          "doc_item_refs": [
            "#/texts/1389",
            "#/texts/1390",
            "#/texts/1391",
            "#/texts/1392",
            "#/texts/1393",
            "#/texts/1394",
            "#/texts/1395",
            "#/texts/1396"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__manual_recall_xunzi::route_004"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__manual_recall_xunzi::0015"
          ],
          "final_grounding_valid": true
        },
        {
          "route_id": "route_cf144d764767",
          "configuration_id": "config_5ef5d484cf7f",
          "route_label": "Abstract-conditioned mechanistic interpretation generation",
          "lifecycle_phase": "evaluation",
          "task_or_configuration_verbatim": "mechanistic interpretation generation task",
          "source_object_verbatim": "provided abstract for a gene-disease pair",
          "source_object_normalized": "provided abstract for a gene-disease pair",
          "source_modality_normalized": "text",
          "transformation_chain_verbatim": [
            "retrieve the provided abstract and gene-disease context",
            "apply the standardized mechanistic reasoning prompt"
          ],
          "model_visible_form_verbatim": "structured natural-language prompt containing an abstract, gene name, and disease name",
          "carrier_family": "text_native_token_stream",
          "carrier_subtype": "structured_biological_prompt_or_task_scaffold",
          "insertion_or_fusion_verbatim": "prompting through the mechanistic reasoning interface",
          "fusion_topology": "tokenizer_sequence",
          "text_role": "instruction_or_query",
          "input_status": "actual_model_input",
          "evidence_quote": "Based on the provided abstract below and existing biological knowledge, analyse the functional role of a specific gene in a given disease.",
          "section_heading": "Benchmarking",
          "supporting_figure_or_table": null,
          "evidence_status": "explicit_text",
          "uncertainty": null,
          "pages": [
            10,
            12
          ],
          "doc_item_refs": [
            "#/texts/1352",
            "#/texts/1353",
            "#/texts/1354",
            "#/texts/1355",
            "#/texts/1356",
            "#/texts/1357",
            "#/texts/1358",
            "#/texts/1359",
            "#/texts/1389",
            "#/texts/1390",
            "#/texts/1391",
            "#/texts/1392",
            "#/texts/1393",
            "#/texts/1394",
            "#/texts/1395",
            "#/texts/1396"
          ],
          "source_candidate_refs": [
            "update_2026-08-09__manual_recall_xunzi::route_010"
          ],
          "dense_candidate_refs": [
            "dense::update_2026-08-09__manual_recall_xunzi::0016"
          ],
          "final_grounding_valid": true
        }
      ],
      "membership_group_id": "membership_67d200027193"
    }
  ],
  "graph": {
    "nodes": [
      {
        "id": "taxonomy_root",
        "type": "root",
        "label": "Model-visible input representation"
      },
      {
        "id": "family::text_native_token_stream",
        "type": "family",
        "label": "Text-native token streams",
        "family_id": "text_native_token_stream",
        "code": "F1"
      },
      {
        "id": "subtype::plain_language_prompt_or_question",
        "type": "subtype",
        "label": "Plain language prompts and questions",
        "subtype_id": "plain_language_prompt_or_question",
        "family_id": "text_native_token_stream",
        "leaf_id": "F1.L1"
      },
      {
        "id": "subtype::structured_biological_prompt_or_task_scaffold",
        "type": "subtype",
        "label": "Structured biological prompts and task scaffolds",
        "subtype_id": "structured_biological_prompt_or_task_scaffold",
        "family_id": "text_native_token_stream",
        "leaf_id": "F1.L2"
      },
      {
        "id": "subtype::serialized_biological_context_or_ordered_profile",
        "type": "subtype",
        "label": "Serialized biological context and ordered profiles",
        "subtype_id": "serialized_biological_context_or_ordered_profile",
        "family_id": "text_native_token_stream",
        "leaf_id": "F1.L3"
      },
      {
        "id": "family::discrete_biological_symbol_stream",
        "type": "family",
        "label": "Discrete biological symbol streams",
        "family_id": "discrete_biological_symbol_stream",
        "code": "F2"
      },
      {
        "id": "subtype::native_biological_token_stream",
        "type": "subtype",
        "label": "Native biological token streams",
        "subtype_id": "native_biological_token_stream",
        "family_id": "discrete_biological_symbol_stream",
        "leaf_id": "F2.L1"
      },
      {
        "id": "subtype::multi_track_structural_symbol_stream",
        "type": "subtype",
        "label": "Multi-track structural symbol streams",
        "subtype_id": "multi_track_structural_symbol_stream",
        "family_id": "discrete_biological_symbol_stream",
        "leaf_id": "F2.L2"
      },
      {
        "id": "subtype::learned_quantized_id_or_codebook_token",
        "type": "subtype",
        "label": "Learned quantized IDs and codebook tokens",
        "subtype_id": "learned_quantized_id_or_codebook_token",
        "family_id": "discrete_biological_symbol_stream",
        "leaf_id": "F2.L3"
      },
      {
        "id": "family::dense_continuous_carrier",
        "type": "family",
        "label": "Dense continuous carriers",
        "family_id": "dense_continuous_carrier",
        "code": "F3"
      },
      {
        "id": "subtype::direct_projected_embedding",
        "type": "subtype",
        "label": "Direct projected embeddings",
        "subtype_id": "direct_projected_embedding",
        "family_id": "dense_continuous_carrier",
        "leaf_id": "F3.L1"
      },
      {
        "id": "subtype::virtual_token_prefix",
        "type": "subtype",
        "label": "Virtual-token prefixes",
        "subtype_id": "virtual_token_prefix",
        "family_id": "dense_continuous_carrier",
        "leaf_id": "F3.L2"
      },
      {
        "id": "subtype::connector_mediated_embedding",
        "type": "subtype",
        "label": "Connector-mediated embeddings",
        "subtype_id": "connector_mediated_embedding",
        "family_id": "dense_continuous_carrier",
        "leaf_id": "F3.L3"
      },
      {
        "id": "subtype::pooled_or_aggregated_embedding",
        "type": "subtype",
        "label": "Pooled or aggregated embeddings",
        "subtype_id": "pooled_or_aggregated_embedding",
        "family_id": "dense_continuous_carrier",
        "leaf_id": "F3.L4"
      },
      {
        "id": "family::visual_raster_carrier",
        "type": "family",
        "label": "Visual raster carriers",
        "family_id": "visual_raster_carrier",
        "code": "F4"
      },
      {
        "id": "subtype::raw_slide_or_patch_input",
        "type": "subtype",
        "label": "Raw slide or patch input",
        "subtype_id": "raw_slide_or_patch_input",
        "family_id": "visual_raster_carrier",
        "leaf_id": "F4.L1"
      },
      {
        "id": "subtype::patch_context_or_case_level_visual_reasoning",
        "type": "subtype",
        "label": "Patch-context or case-level visual reasoning",
        "subtype_id": "patch_context_or_case_level_visual_reasoning",
        "family_id": "visual_raster_carrier",
        "leaf_id": "F4.L2"
      },
      {
        "id": "family::geometric_or_diffusion_state_carrier",
        "type": "family",
        "label": "Geometric and diffusion-state carriers",
        "family_id": "geometric_or_diffusion_state_carrier",
        "code": "F5"
      },
      {
        "id": "subtype::noisy_diffusion_state",
        "type": "subtype",
        "label": "Noisy diffusion state",
        "subtype_id": "noisy_diffusion_state",
        "family_id": "geometric_or_diffusion_state_carrier",
        "leaf_id": "F5.L1"
      },
      {
        "id": "subtype::coordinate_backbone_or_shape_conditioning",
        "type": "subtype",
        "label": "Coordinate, backbone, or shape conditioning",
        "subtype_id": "coordinate_backbone_or_shape_conditioning",
        "family_id": "geometric_or_diffusion_state_carrier",
        "leaf_id": "F5.L2"
      },
      {
        "id": "subtype::symbolic_structural_constraint",
        "type": "subtype",
        "label": "Symbolic structural constraints",
        "subtype_id": "symbolic_structural_constraint",
        "family_id": "geometric_or_diffusion_state_carrier",
        "leaf_id": "F5.L3"
      },
      {
        "id": "group::membership_e367b465c0b9",
        "type": "membership_group",
        "label": "F3.L3",
        "group_id": "membership_e367b465c0b9",
        "subtype_ids": [
          "connector_mediated_embedding"
        ],
        "family_ids": [
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_6f7f8cee95d1",
          "model_16a912645955"
        ],
        "model_count": 2
      },
      {
        "id": "model::model_6f7f8cee95d1",
        "type": "model",
        "label": "ChatNT",
        "model_id": "model_6f7f8cee95d1",
        "membership_group_id": "membership_e367b465c0b9"
      },
      {
        "id": "model::model_16a912645955",
        "type": "model",
        "label": "Mistral 7B",
        "model_id": "model_16a912645955",
        "membership_group_id": "membership_e367b465c0b9"
      },
      {
        "id": "group::membership_20ed14556e30",
        "type": "membership_group",
        "label": "F3.L3 + F3.L1",
        "group_id": "membership_20ed14556e30",
        "subtype_ids": [
          "connector_mediated_embedding",
          "direct_projected_embedding"
        ],
        "family_ids": [
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_2a4d0a37a0ad"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_2a4d0a37a0ad",
        "type": "model",
        "label": "scDiff",
        "model_id": "model_2a4d0a37a0ad",
        "membership_group_id": "membership_20ed14556e30"
      },
      {
        "id": "group::membership_686e999550fc",
        "type": "membership_group",
        "label": "F3.L3 + F3.L1 + F5.L1 + F1.L1 + F3.L4 + F4.L1 + F1.L2",
        "group_id": "membership_686e999550fc",
        "subtype_ids": [
          "connector_mediated_embedding",
          "direct_projected_embedding",
          "noisy_diffusion_state",
          "plain_language_prompt_or_question",
          "pooled_or_aggregated_embedding",
          "raw_slide_or_patch_input",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier",
          "visual_raster_carrier",
          "geometric_or_diffusion_state_carrier"
        ],
        "model_ids": [
          "model_59fd1ac5f56e"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_59fd1ac5f56e",
        "type": "model",
        "label": "MUPAD",
        "model_id": "model_59fd1ac5f56e",
        "membership_group_id": "membership_686e999550fc"
      },
      {
        "id": "group::membership_aab758294f91",
        "type": "membership_group",
        "label": "F3.L3 + F3.L1 + F1.L1 + F1.L2",
        "group_id": "membership_aab758294f91",
        "subtype_ids": [
          "connector_mediated_embedding",
          "direct_projected_embedding",
          "plain_language_prompt_or_question",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_f8e766fddf66"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_f8e766fddf66",
        "type": "model",
        "label": "PROCYON",
        "model_id": "model_f8e766fddf66",
        "membership_group_id": "membership_aab758294f91"
      },
      {
        "id": "group::membership_6742df78bf4f",
        "type": "membership_group",
        "label": "F3.L3 + F3.L1 + F3.L4 + F1.L3 + F1.L2",
        "group_id": "membership_6742df78bf4f",
        "subtype_ids": [
          "connector_mediated_embedding",
          "direct_projected_embedding",
          "pooled_or_aggregated_embedding",
          "serialized_biological_context_or_ordered_profile",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_860d9511b312"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_860d9511b312",
        "type": "model",
        "label": "P3GPT",
        "model_id": "model_860d9511b312",
        "membership_group_id": "membership_6742df78bf4f"
      },
      {
        "id": "group::membership_433657cba128",
        "type": "membership_group",
        "label": "F3.L3 + F3.L1 + F1.L3",
        "group_id": "membership_433657cba128",
        "subtype_ids": [
          "connector_mediated_embedding",
          "direct_projected_embedding",
          "serialized_biological_context_or_ordered_profile"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_ffaa8b52273c"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_ffaa8b52273c",
        "type": "model",
        "label": "scMMGPT",
        "model_id": "model_ffaa8b52273c",
        "membership_group_id": "membership_433657cba128"
      },
      {
        "id": "group::membership_47e8053eae22",
        "type": "membership_group",
        "label": "F3.L3 + F1.L1 + F1.L3",
        "group_id": "membership_47e8053eae22",
        "subtype_ids": [
          "connector_mediated_embedding",
          "plain_language_prompt_or_question",
          "serialized_biological_context_or_ordered_profile"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_f1bf6497dcf0"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_f1bf6497dcf0",
        "type": "model",
        "label": "Mistral 7B",
        "model_id": "model_f1bf6497dcf0",
        "membership_group_id": "membership_47e8053eae22"
      },
      {
        "id": "group::membership_dbca3a16da8f",
        "type": "membership_group",
        "label": "F3.L3 + F1.L2",
        "group_id": "membership_dbca3a16da8f",
        "subtype_ids": [
          "connector_mediated_embedding",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_867ffa020512"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_867ffa020512",
        "type": "model",
        "label": "InstructCell",
        "model_id": "model_867ffa020512",
        "membership_group_id": "membership_dbca3a16da8f"
      },
      {
        "id": "group::membership_ee721887da1b",
        "type": "membership_group",
        "label": "F3.L3 + F3.L2",
        "group_id": "membership_ee721887da1b",
        "subtype_ids": [
          "connector_mediated_embedding",
          "virtual_token_prefix"
        ],
        "family_ids": [
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_699b7d92d927"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_699b7d92d927",
        "type": "model",
        "label": "Bio-BLIP",
        "model_id": "model_699b7d92d927",
        "membership_group_id": "membership_ee721887da1b"
      },
      {
        "id": "group::membership_8ec5e76d5e6a",
        "type": "membership_group",
        "label": "F5.L2 + F3.L1 + F5.L1 + F1.L1 + F1.L2",
        "group_id": "membership_8ec5e76d5e6a",
        "subtype_ids": [
          "coordinate_backbone_or_shape_conditioning",
          "direct_projected_embedding",
          "noisy_diffusion_state",
          "plain_language_prompt_or_question",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier",
          "geometric_or_diffusion_state_carrier"
        ],
        "model_ids": [
          "model_fcda05dbbcd5"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_fcda05dbbcd5",
        "type": "model",
        "label": "CaLMFlow",
        "model_id": "model_fcda05dbbcd5",
        "membership_group_id": "membership_8ec5e76d5e6a"
      },
      {
        "id": "group::membership_af3baba835a0",
        "type": "membership_group",
        "label": "F5.L2 + F2.L1 + F1.L1",
        "group_id": "membership_af3baba835a0",
        "subtype_ids": [
          "coordinate_backbone_or_shape_conditioning",
          "native_biological_token_stream",
          "plain_language_prompt_or_question"
        ],
        "family_ids": [
          "text_native_token_stream",
          "discrete_biological_symbol_stream",
          "geometric_or_diffusion_state_carrier"
        ],
        "model_ids": [
          "model_b1f724b128da"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_b1f724b128da",
        "type": "model",
        "label": "PROCYON-SPLIT",
        "model_id": "model_b1f724b128da",
        "membership_group_id": "membership_af3baba835a0"
      },
      {
        "id": "group::membership_04e6eca108d0",
        "type": "membership_group",
        "label": "F5.L2 + F5.L1 + F1.L1 + F5.L3",
        "group_id": "membership_04e6eca108d0",
        "subtype_ids": [
          "coordinate_backbone_or_shape_conditioning",
          "noisy_diffusion_state",
          "plain_language_prompt_or_question",
          "symbolic_structural_constraint"
        ],
        "family_ids": [
          "text_native_token_stream",
          "geometric_or_diffusion_state_carrier"
        ],
        "model_ids": [
          "model_efc702564a41"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_efc702564a41",
        "type": "model",
        "label": "Chroma",
        "model_id": "model_efc702564a41",
        "membership_group_id": "membership_04e6eca108d0"
      },
      {
        "id": "group::membership_000c4dd8661a",
        "type": "membership_group",
        "label": "F3.L1",
        "group_id": "membership_000c4dd8661a",
        "subtype_ids": [
          "direct_projected_embedding"
        ],
        "family_ids": [
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_9be63ad52f90",
          "model_35105bbde921",
          "model_99366b543213",
          "model_9c283f71363e",
          "model_422e09e03da5",
          "model_b4a92e5e540d"
        ],
        "model_count": 6
      },
      {
        "id": "model::model_9be63ad52f90",
        "type": "model",
        "label": "BIOVERSE",
        "model_id": "model_9be63ad52f90",
        "membership_group_id": "membership_000c4dd8661a"
      },
      {
        "id": "model::model_35105bbde921",
        "type": "model",
        "label": "Cell2Text-Gemma-4B",
        "model_id": "model_35105bbde921",
        "membership_group_id": "membership_000c4dd8661a"
      },
      {
        "id": "model::model_99366b543213",
        "type": "model",
        "label": "Cell2Text-Llama-1B",
        "model_id": "model_99366b543213",
        "membership_group_id": "membership_000c4dd8661a"
      },
      {
        "id": "model::model_9c283f71363e",
        "type": "model",
        "label": "Cell2Text-Llama-1B-LoRA",
        "model_id": "model_9c283f71363e",
        "membership_group_id": "membership_000c4dd8661a"
      },
      {
        "id": "model::model_422e09e03da5",
        "type": "model",
        "label": "Clinical-LongFormer",
        "model_id": "model_422e09e03da5",
        "membership_group_id": "membership_000c4dd8661a"
      },
      {
        "id": "model::model_b4a92e5e540d",
        "type": "model",
        "label": "text LLM",
        "model_id": "model_b4a92e5e540d",
        "membership_group_id": "membership_000c4dd8661a"
      },
      {
        "id": "group::membership_22936f4d27e9",
        "type": "membership_group",
        "label": "F3.L1 + F2.L1",
        "group_id": "membership_22936f4d27e9",
        "subtype_ids": [
          "direct_projected_embedding",
          "native_biological_token_stream"
        ],
        "family_ids": [
          "discrete_biological_symbol_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_0f8f776fb068",
          "model_9b1025af9fe7"
        ],
        "model_count": 2
      },
      {
        "id": "model::model_0f8f776fb068",
        "type": "model",
        "label": "LEONINE",
        "model_id": "model_0f8f776fb068",
        "membership_group_id": "membership_22936f4d27e9"
      },
      {
        "id": "model::model_9b1025af9fe7",
        "type": "model",
        "label": "LEONINE-GPT-2",
        "model_id": "model_9b1025af9fe7",
        "membership_group_id": "membership_22936f4d27e9"
      },
      {
        "id": "group::membership_cab7fec0fcf3",
        "type": "membership_group",
        "label": "F3.L1 + F1.L1 + F1.L3 + F1.L2",
        "group_id": "membership_cab7fec0fcf3",
        "subtype_ids": [
          "direct_projected_embedding",
          "plain_language_prompt_or_question",
          "serialized_biological_context_or_ordered_profile",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_064ca265f9bb"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_064ca265f9bb",
        "type": "model",
        "label": "OmicsLM",
        "model_id": "model_064ca265f9bb",
        "membership_group_id": "membership_cab7fec0fcf3"
      },
      {
        "id": "group::membership_db6432d4d322",
        "type": "membership_group",
        "label": "F3.L1 + F1.L1 + F1.L2",
        "group_id": "membership_db6432d4d322",
        "subtype_ids": [
          "direct_projected_embedding",
          "plain_language_prompt_or_question",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_b8eba3c26c46"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_b8eba3c26c46",
        "type": "model",
        "label": "Qwen3-1.7B and Qwen3-4B",
        "model_id": "model_b8eba3c26c46",
        "membership_group_id": "membership_db6432d4d322"
      },
      {
        "id": "group::membership_d7b71bdfa7dd",
        "type": "membership_group",
        "label": "F3.L1 + F3.L4",
        "group_id": "membership_d7b71bdfa7dd",
        "subtype_ids": [
          "direct_projected_embedding",
          "pooled_or_aggregated_embedding"
        ],
        "family_ids": [
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_3c62ea9d84a1",
          "model_d3b03caa65b1"
        ],
        "model_count": 2
      },
      {
        "id": "model::model_3c62ea9d84a1",
        "type": "model",
        "label": "X-Cell-Ultra",
        "model_id": "model_3c62ea9d84a1",
        "membership_group_id": "membership_d7b71bdfa7dd"
      },
      {
        "id": "model::model_d3b03caa65b1",
        "type": "model",
        "label": "XunZi-M",
        "model_id": "model_d3b03caa65b1",
        "membership_group_id": "membership_d7b71bdfa7dd"
      },
      {
        "id": "group::membership_d445170f41eb",
        "type": "membership_group",
        "label": "F3.L1 + F3.L4 + F1.L2",
        "group_id": "membership_d445170f41eb",
        "subtype_ids": [
          "direct_projected_embedding",
          "pooled_or_aggregated_embedding",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_4361ad5c9fbb"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_4361ad5c9fbb",
        "type": "model",
        "label": "X-Cell",
        "model_id": "model_4361ad5c9fbb",
        "membership_group_id": "membership_d445170f41eb"
      },
      {
        "id": "group::membership_cf2af8e28dd7",
        "type": "membership_group",
        "label": "F3.L1 + F1.L3",
        "group_id": "membership_cf2af8e28dd7",
        "subtype_ids": [
          "direct_projected_embedding",
          "serialized_biological_context_or_ordered_profile"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_5f2f0ba1126a"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_5f2f0ba1126a",
        "type": "model",
        "label": "OKR-CELL",
        "model_id": "model_5f2f0ba1126a",
        "membership_group_id": "membership_cf2af8e28dd7"
      },
      {
        "id": "group::membership_c17be9263882",
        "type": "membership_group",
        "label": "F3.L1 + F1.L2",
        "group_id": "membership_c17be9263882",
        "subtype_ids": [
          "direct_projected_embedding",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_77f5da416a80",
          "model_285e25ae2bfb",
          "model_9d05ddb97916",
          "model_ac7794e51425",
          "model_a493b5670a18",
          "model_ab432decc9aa"
        ],
        "model_count": 6
      },
      {
        "id": "model::model_77f5da416a80",
        "type": "model",
        "label": "BioMedGPT-10B",
        "model_id": "model_77f5da416a80",
        "membership_group_id": "membership_c17be9263882"
      },
      {
        "id": "model::model_285e25ae2bfb",
        "type": "model",
        "label": "Cell2Text",
        "model_id": "model_285e25ae2bfb",
        "membership_group_id": "membership_c17be9263882"
      },
      {
        "id": "model::model_9d05ddb97916",
        "type": "model",
        "label": "Qwen3",
        "model_id": "model_9d05ddb97916",
        "membership_group_id": "membership_c17be9263882"
      },
      {
        "id": "model::model_ac7794e51425",
        "type": "model",
        "label": "Qwen3-1.7B",
        "model_id": "model_ac7794e51425",
        "membership_group_id": "membership_c17be9263882"
      },
      {
        "id": "model::model_a493b5670a18",
        "type": "model",
        "label": "Qwen3-1.7B and Qwen3-4B-Thinking",
        "model_id": "model_a493b5670a18",
        "membership_group_id": "membership_c17be9263882"
      },
      {
        "id": "model::model_ab432decc9aa",
        "type": "model",
        "label": "Qwen3-1.7B, Qwen3-4B, and Gemma 4 E2B",
        "model_id": "model_ab432decc9aa",
        "membership_group_id": "membership_c17be9263882"
      },
      {
        "id": "group::membership_2e20f0c5603e",
        "type": "membership_group",
        "label": "F3.L1 + F5.L3",
        "group_id": "membership_2e20f0c5603e",
        "subtype_ids": [
          "direct_projected_embedding",
          "symbolic_structural_constraint"
        ],
        "family_ids": [
          "dense_continuous_carrier",
          "geometric_or_diffusion_state_carrier"
        ],
        "model_ids": [
          "model_d3d6352d5bcb"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_d3d6352d5bcb",
        "type": "model",
        "label": "Shusi",
        "model_id": "model_d3d6352d5bcb",
        "membership_group_id": "membership_2e20f0c5603e"
      },
      {
        "id": "group::membership_fbfad76cabfd",
        "type": "membership_group",
        "label": "F2.L3 + F2.L2 + F1.L1 + F1.L2",
        "group_id": "membership_fbfad76cabfd",
        "subtype_ids": [
          "learned_quantized_id_or_codebook_token",
          "multi_track_structural_symbol_stream",
          "plain_language_prompt_or_question",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "discrete_biological_symbol_stream"
        ],
        "model_ids": [
          "model_7ccbcc7f850b"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_7ccbcc7f850b",
        "type": "model",
        "label": "RVQ-Alpha",
        "model_id": "model_7ccbcc7f850b",
        "membership_group_id": "membership_fbfad76cabfd"
      },
      {
        "id": "group::membership_951dfcefa907",
        "type": "membership_group",
        "label": "F2.L2",
        "group_id": "membership_951dfcefa907",
        "subtype_ids": [
          "multi_track_structural_symbol_stream"
        ],
        "family_ids": [
          "discrete_biological_symbol_stream"
        ],
        "model_ids": [
          "model_58c5ce11b66a"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_58c5ce11b66a",
        "type": "model",
        "label": "scGPT",
        "model_id": "model_58c5ce11b66a",
        "membership_group_id": "membership_951dfcefa907"
      },
      {
        "id": "group::membership_493db55fa177",
        "type": "membership_group",
        "label": "F2.L2 + F2.L1 + F1.L1 + F1.L2",
        "group_id": "membership_493db55fa177",
        "subtype_ids": [
          "multi_track_structural_symbol_stream",
          "native_biological_token_stream",
          "plain_language_prompt_or_question",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "discrete_biological_symbol_stream"
        ],
        "model_ids": [
          "model_74e685d85fa8"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_74e685d85fa8",
        "type": "model",
        "label": "OmniGene-4",
        "model_id": "model_74e685d85fa8",
        "membership_group_id": "membership_493db55fa177"
      },
      {
        "id": "group::membership_de94be356588",
        "type": "membership_group",
        "label": "F2.L1",
        "group_id": "membership_de94be356588",
        "subtype_ids": [
          "native_biological_token_stream"
        ],
        "family_ids": [
          "discrete_biological_symbol_stream"
        ],
        "model_ids": [
          "model_eb69df38d690",
          "model_498b709f811c"
        ],
        "model_count": 2
      },
      {
        "id": "model::model_eb69df38d690",
        "type": "model",
        "label": "OmniNA-1.7B",
        "model_id": "model_eb69df38d690",
        "membership_group_id": "membership_de94be356588"
      },
      {
        "id": "model::model_498b709f811c",
        "type": "model",
        "label": "scGPT-FT",
        "model_id": "model_498b709f811c",
        "membership_group_id": "membership_de94be356588"
      },
      {
        "id": "group::membership_fdbd3ba264a9",
        "type": "membership_group",
        "label": "F2.L1 + F1.L1",
        "group_id": "membership_fdbd3ba264a9",
        "subtype_ids": [
          "native_biological_token_stream",
          "plain_language_prompt_or_question"
        ],
        "family_ids": [
          "text_native_token_stream",
          "discrete_biological_symbol_stream"
        ],
        "model_ids": [
          "model_5421186d4cbe",
          "model_3841ec0733e7",
          "model_d2833fcd27aa",
          "model_5fd880eafa1a"
        ],
        "model_count": 4
      },
      {
        "id": "model::model_5421186d4cbe",
        "type": "model",
        "label": "BioGPT",
        "model_id": "model_5421186d4cbe",
        "membership_group_id": "membership_fdbd3ba264a9"
      },
      {
        "id": "model::model_3841ec0733e7",
        "type": "model",
        "label": "gene_eng_gpt2_para_seg",
        "model_id": "model_3841ec0733e7",
        "membership_group_id": "membership_fdbd3ba264a9"
      },
      {
        "id": "model::model_d2833fcd27aa",
        "type": "model",
        "label": "gpt2-gene-eng",
        "model_id": "model_d2833fcd27aa",
        "membership_group_id": "membership_fdbd3ba264a9"
      },
      {
        "id": "model::model_5fd880eafa1a",
        "type": "model",
        "label": "scMOBA",
        "model_id": "model_5fd880eafa1a",
        "membership_group_id": "membership_fdbd3ba264a9"
      },
      {
        "id": "group::membership_ad9cb3ec7dd7",
        "type": "membership_group",
        "label": "F2.L1 + F1.L1 + F1.L3",
        "group_id": "membership_ad9cb3ec7dd7",
        "subtype_ids": [
          "native_biological_token_stream",
          "plain_language_prompt_or_question",
          "serialized_biological_context_or_ordered_profile"
        ],
        "family_ids": [
          "text_native_token_stream",
          "discrete_biological_symbol_stream"
        ],
        "model_ids": [
          "model_74758c50e662"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_74758c50e662",
        "type": "model",
        "label": "OmniNA",
        "model_id": "model_74758c50e662",
        "membership_group_id": "membership_ad9cb3ec7dd7"
      },
      {
        "id": "group::membership_8d5343810816",
        "type": "membership_group",
        "label": "F2.L1 + F1.L1 + F1.L2",
        "group_id": "membership_8d5343810816",
        "subtype_ids": [
          "native_biological_token_stream",
          "plain_language_prompt_or_question",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "discrete_biological_symbol_stream"
        ],
        "model_ids": [
          "model_5e9bb9e443dc",
          "model_b4d1be45d2bb"
        ],
        "model_count": 2
      },
      {
        "id": "model::model_5e9bb9e443dc",
        "type": "model",
        "label": "gpt2-gene-eng-ft",
        "model_id": "model_5e9bb9e443dc",
        "membership_group_id": "membership_8d5343810816"
      },
      {
        "id": "model::model_b4d1be45d2bb",
        "type": "model",
        "label": "Omni-DNA",
        "model_id": "model_b4d1be45d2bb",
        "membership_group_id": "membership_8d5343810816"
      },
      {
        "id": "group::membership_aaacf24f224c",
        "type": "membership_group",
        "label": "F2.L1 + F1.L3 + F1.L2",
        "group_id": "membership_aaacf24f224c",
        "subtype_ids": [
          "native_biological_token_stream",
          "serialized_biological_context_or_ordered_profile",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "discrete_biological_symbol_stream"
        ],
        "model_ids": [
          "model_69a439438d24",
          "model_958497e0e4e2"
        ],
        "model_count": 2
      },
      {
        "id": "model::model_69a439438d24",
        "type": "model",
        "label": "GenNA",
        "model_id": "model_69a439438d24",
        "membership_group_id": "membership_aaacf24f224c"
      },
      {
        "id": "model::model_958497e0e4e2",
        "type": "model",
        "label": "OmniNA",
        "model_id": "model_958497e0e4e2",
        "membership_group_id": "membership_aaacf24f224c"
      },
      {
        "id": "group::membership_cbdfdd5c56c0",
        "type": "membership_group",
        "label": "F4.L2 + F1.L1 + F4.L1 + F1.L3 + F1.L2",
        "group_id": "membership_cbdfdd5c56c0",
        "subtype_ids": [
          "patch_context_or_case_level_visual_reasoning",
          "plain_language_prompt_or_question",
          "raw_slide_or_patch_input",
          "serialized_biological_context_or_ordered_profile",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "visual_raster_carrier"
        ],
        "model_ids": [
          "model_23ffab302d6b"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_23ffab302d6b",
        "type": "model",
        "label": "Med-PaLM M",
        "model_id": "model_23ffab302d6b",
        "membership_group_id": "membership_cbdfdd5c56c0"
      },
      {
        "id": "group::membership_10855a0f50f0",
        "type": "membership_group",
        "label": "F4.L2 + F4.L1",
        "group_id": "membership_10855a0f50f0",
        "subtype_ids": [
          "patch_context_or_case_level_visual_reasoning",
          "raw_slide_or_patch_input"
        ],
        "family_ids": [
          "visual_raster_carrier"
        ],
        "model_ids": [
          "model_3827a28edefc"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_3827a28edefc",
        "type": "model",
        "label": "H2O",
        "model_id": "model_3827a28edefc",
        "membership_group_id": "membership_10855a0f50f0"
      },
      {
        "id": "group::membership_887038a8a19a",
        "type": "membership_group",
        "label": "F4.L2 + F4.L1 + F3.L2",
        "group_id": "membership_887038a8a19a",
        "subtype_ids": [
          "patch_context_or_case_level_visual_reasoning",
          "raw_slide_or_patch_input",
          "virtual_token_prefix"
        ],
        "family_ids": [
          "dense_continuous_carrier",
          "visual_raster_carrier"
        ],
        "model_ids": [
          "model_f67272a54ac0"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_f67272a54ac0",
        "type": "model",
        "label": "SciCore-Omics",
        "model_id": "model_f67272a54ac0",
        "membership_group_id": "membership_887038a8a19a"
      },
      {
        "id": "group::membership_357f1bf88b0b",
        "type": "membership_group",
        "label": "F1.L1",
        "group_id": "membership_357f1bf88b0b",
        "subtype_ids": [
          "plain_language_prompt_or_question"
        ],
        "family_ids": [
          "text_native_token_stream"
        ],
        "model_ids": [
          "model_0e0c28b60ee7",
          "model_7ed6448bcdeb",
          "model_c7150172725e",
          "model_f12e33a1e764"
        ],
        "model_count": 4
      },
      {
        "id": "model::model_0e0c28b60ee7",
        "type": "model",
        "label": "BIOREASON",
        "model_id": "model_0e0c28b60ee7",
        "membership_group_id": "membership_357f1bf88b0b"
      },
      {
        "id": "model::model_7ed6448bcdeb",
        "type": "model",
        "label": "DeepSeek",
        "model_id": "model_7ed6448bcdeb",
        "membership_group_id": "membership_357f1bf88b0b"
      },
      {
        "id": "model::model_c7150172725e",
        "type": "model",
        "label": "DeepSeek-R1-Distill (1.5B)",
        "model_id": "model_c7150172725e",
        "membership_group_id": "membership_357f1bf88b0b"
      },
      {
        "id": "model::model_f12e33a1e764",
        "type": "model",
        "label": "OCellus-Agent",
        "model_id": "model_f12e33a1e764",
        "membership_group_id": "membership_357f1bf88b0b"
      },
      {
        "id": "group::membership_087f460aac0a",
        "type": "membership_group",
        "label": "F1.L1 + F3.L4",
        "group_id": "membership_087f460aac0a",
        "subtype_ids": [
          "plain_language_prompt_or_question",
          "pooled_or_aggregated_embedding"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_0e34c1fd11f6"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_0e34c1fd11f6",
        "type": "model",
        "label": "Genolator",
        "model_id": "model_0e34c1fd11f6",
        "membership_group_id": "membership_087f460aac0a"
      },
      {
        "id": "group::membership_22149166d2ab",
        "type": "membership_group",
        "label": "F1.L1 + F3.L4 + F4.L1",
        "group_id": "membership_22149166d2ab",
        "subtype_ids": [
          "plain_language_prompt_or_question",
          "pooled_or_aggregated_embedding",
          "raw_slide_or_patch_input"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier",
          "visual_raster_carrier"
        ],
        "model_ids": [
          "model_a45de7f446e9"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_a45de7f446e9",
        "type": "model",
        "label": "ALTER",
        "model_id": "model_a45de7f446e9",
        "membership_group_id": "membership_22149166d2ab"
      },
      {
        "id": "group::membership_9337e69362f7",
        "type": "membership_group",
        "label": "F1.L1 + F3.L4 + F1.L3 + F1.L2",
        "group_id": "membership_9337e69362f7",
        "subtype_ids": [
          "plain_language_prompt_or_question",
          "pooled_or_aggregated_embedding",
          "serialized_biological_context_or_ordered_profile",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_c0a097206f2c"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_c0a097206f2c",
        "type": "model",
        "label": "CellHermes",
        "model_id": "model_c0a097206f2c",
        "membership_group_id": "membership_9337e69362f7"
      },
      {
        "id": "group::membership_7ec27bf438d4",
        "type": "membership_group",
        "label": "F1.L1 + F4.L1 + F1.L2",
        "group_id": "membership_7ec27bf438d4",
        "subtype_ids": [
          "plain_language_prompt_or_question",
          "raw_slide_or_patch_input",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "visual_raster_carrier"
        ],
        "model_ids": [
          "model_6a1d79c2e81c",
          "model_ff0ac12f6b2a"
        ],
        "model_count": 2
      },
      {
        "id": "model::model_6a1d79c2e81c",
        "type": "model",
        "label": "TeamPath-7B",
        "model_id": "model_6a1d79c2e81c",
        "membership_group_id": "membership_7ec27bf438d4"
      },
      {
        "id": "model::model_ff0ac12f6b2a",
        "type": "model",
        "label": "TissueCraftAI",
        "model_id": "model_ff0ac12f6b2a",
        "membership_group_id": "membership_7ec27bf438d4"
      },
      {
        "id": "group::membership_d8f17c473542",
        "type": "membership_group",
        "label": "F1.L1 + F1.L3",
        "group_id": "membership_d8f17c473542",
        "subtype_ids": [
          "plain_language_prompt_or_question",
          "serialized_biological_context_or_ordered_profile"
        ],
        "family_ids": [
          "text_native_token_stream"
        ],
        "model_ids": [
          "model_0b491a7a5d54",
          "model_3d02d9393c92",
          "model_36c6451d1b55"
        ],
        "model_count": 3
      },
      {
        "id": "model::model_0b491a7a5d54",
        "type": "model",
        "label": "DeepSeek-R1-Distill 70B",
        "model_id": "model_0b491a7a5d54",
        "membership_group_id": "membership_d8f17c473542"
      },
      {
        "id": "model::model_3d02d9393c92",
        "type": "model",
        "label": "DeepSeek-R1-Distill series (14B, 32B, and 70B)",
        "model_id": "model_3d02d9393c92",
        "membership_group_id": "membership_d8f17c473542"
      },
      {
        "id": "model::model_36c6451d1b55",
        "type": "model",
        "label": "GPT-4o",
        "model_id": "model_36c6451d1b55",
        "membership_group_id": "membership_d8f17c473542"
      },
      {
        "id": "group::membership_67d200027193",
        "type": "membership_group",
        "label": "F1.L1 + F1.L3 + F1.L2",
        "group_id": "membership_67d200027193",
        "subtype_ids": [
          "plain_language_prompt_or_question",
          "serialized_biological_context_or_ordered_profile",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream"
        ],
        "model_ids": [
          "model_68562cc69016",
          "model_72e70c356b7a",
          "model_6694aa9f2590",
          "model_0b0332ce2b16"
        ],
        "model_count": 4
      },
      {
        "id": "model::model_68562cc69016",
        "type": "model",
        "label": "CHATCELL",
        "model_id": "model_68562cc69016",
        "membership_group_id": "membership_67d200027193"
      },
      {
        "id": "model::model_72e70c356b7a",
        "type": "model",
        "label": "GPT-2 Medium",
        "model_id": "model_72e70c356b7a",
        "membership_group_id": "membership_67d200027193"
      },
      {
        "id": "model::model_6694aa9f2590",
        "type": "model",
        "label": "GPT-2 Small",
        "model_id": "model_6694aa9f2590",
        "membership_group_id": "membership_67d200027193"
      },
      {
        "id": "model::model_0b0332ce2b16",
        "type": "model",
        "label": "XunZi-R",
        "model_id": "model_0b0332ce2b16",
        "membership_group_id": "membership_67d200027193"
      },
      {
        "id": "group::membership_acce00dbb33d",
        "type": "membership_group",
        "label": "F1.L1 + F1.L3 + F1.L2 + F3.L2",
        "group_id": "membership_acce00dbb33d",
        "subtype_ids": [
          "plain_language_prompt_or_question",
          "serialized_biological_context_or_ordered_profile",
          "structured_biological_prompt_or_task_scaffold",
          "virtual_token_prefix"
        ],
        "family_ids": [
          "text_native_token_stream",
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_8b27781d1156"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_8b27781d1156",
        "type": "model",
        "label": "CellTosg2Sequence",
        "model_id": "model_8b27781d1156",
        "membership_group_id": "membership_acce00dbb33d"
      },
      {
        "id": "group::membership_85e0c6746826",
        "type": "membership_group",
        "label": "F1.L1 + F1.L2",
        "group_id": "membership_85e0c6746826",
        "subtype_ids": [
          "plain_language_prompt_or_question",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream"
        ],
        "model_ids": [
          "model_6b93a9600bf1",
          "model_fa039bd90d50"
        ],
        "model_count": 2
      },
      {
        "id": "model::model_6b93a9600bf1",
        "type": "model",
        "label": "GP-GPT",
        "model_id": "model_6b93a9600bf1",
        "membership_group_id": "membership_85e0c6746826"
      },
      {
        "id": "model::model_fa039bd90d50",
        "type": "model",
        "label": "OmniGene-4 v3",
        "model_id": "model_fa039bd90d50",
        "membership_group_id": "membership_85e0c6746826"
      },
      {
        "id": "group::membership_998d40c02fca",
        "type": "membership_group",
        "label": "F3.L4",
        "group_id": "membership_998d40c02fca",
        "subtype_ids": [
          "pooled_or_aggregated_embedding"
        ],
        "family_ids": [
          "dense_continuous_carrier"
        ],
        "model_ids": [
          "model_0dc740f11cc3",
          "model_1bd0fe9134d7"
        ],
        "model_count": 2
      },
      {
        "id": "model::model_0dc740f11cc3",
        "type": "model",
        "label": "CatBoost",
        "model_id": "model_0dc740f11cc3",
        "membership_group_id": "membership_998d40c02fca"
      },
      {
        "id": "model::model_1bd0fe9134d7",
        "type": "model",
        "label": "OCellus-GNN",
        "model_id": "model_1bd0fe9134d7",
        "membership_group_id": "membership_998d40c02fca"
      },
      {
        "id": "group::membership_cb06ed43c31a",
        "type": "membership_group",
        "label": "F4.L1",
        "group_id": "membership_cb06ed43c31a",
        "subtype_ids": [
          "raw_slide_or_patch_input"
        ],
        "family_ids": [
          "visual_raster_carrier"
        ],
        "model_ids": [
          "model_5588f179e503",
          "model_f35a32a6436b",
          "model_1bf0ccb8259f"
        ],
        "model_count": 3
      },
      {
        "id": "model::model_5588f179e503",
        "type": "model",
        "label": "in-house DINO-V2-based histopathology FM",
        "model_id": "model_5588f179e503",
        "membership_group_id": "membership_cb06ed43c31a"
      },
      {
        "id": "model::model_f35a32a6436b",
        "type": "model",
        "label": "LLaVA-7B (LoRA)",
        "model_id": "model_f35a32a6436b",
        "membership_group_id": "membership_cb06ed43c31a"
      },
      {
        "id": "model::model_1bf0ccb8259f",
        "type": "model",
        "label": "OpticalDNA",
        "model_id": "model_1bf0ccb8259f",
        "membership_group_id": "membership_cb06ed43c31a"
      },
      {
        "id": "group::membership_d5e749b180fc",
        "type": "membership_group",
        "label": "F4.L1 + F1.L2",
        "group_id": "membership_d5e749b180fc",
        "subtype_ids": [
          "raw_slide_or_patch_input",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream",
          "visual_raster_carrier"
        ],
        "model_ids": [
          "model_e0f70ec41e97"
        ],
        "model_count": 1
      },
      {
        "id": "model::model_e0f70ec41e97",
        "type": "model",
        "label": "scMIR",
        "model_id": "model_e0f70ec41e97",
        "membership_group_id": "membership_d5e749b180fc"
      },
      {
        "id": "group::membership_0cc72bd82901",
        "type": "membership_group",
        "label": "F1.L3",
        "group_id": "membership_0cc72bd82901",
        "subtype_ids": [
          "serialized_biological_context_or_ordered_profile"
        ],
        "family_ids": [
          "text_native_token_stream"
        ],
        "model_ids": [
          "model_fba98a35871c",
          "model_9ead93555f87",
          "model_14bc733e1633",
          "model_0ea0b64c5788"
        ],
        "model_count": 4
      },
      {
        "id": "model::model_fba98a35871c",
        "type": "model",
        "label": "BioMedGPT-LM-7B",
        "model_id": "model_fba98a35871c",
        "membership_group_id": "membership_0cc72bd82901"
      },
      {
        "id": "model::model_9ead93555f87",
        "type": "model",
        "label": "C2S-Scale perturbation model",
        "model_id": "model_9ead93555f87",
        "membership_group_id": "membership_0cc72bd82901"
      },
      {
        "id": "model::model_14bc733e1633",
        "type": "model",
        "label": "compound-phenotype scoring model",
        "model_id": "model_14bc733e1633",
        "membership_group_id": "membership_0cc72bd82901"
      },
      {
        "id": "model::model_0ea0b64c5788",
        "type": "model",
        "label": "LLaMA-2 7B",
        "model_id": "model_0ea0b64c5788",
        "membership_group_id": "membership_0cc72bd82901"
      },
      {
        "id": "group::membership_2bfb9536cf22",
        "type": "membership_group",
        "label": "F1.L3 + F1.L2",
        "group_id": "membership_2bfb9536cf22",
        "subtype_ids": [
          "serialized_biological_context_or_ordered_profile",
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream"
        ],
        "model_ids": [
          "model_79e042109d33",
          "model_a2ae3a8f5414",
          "model_58435d2084b0",
          "model_2b474a3814ef",
          "model_4ae9868de1c9",
          "model_f924e02d892f",
          "model_df8ed6559ceb"
        ],
        "model_count": 7
      },
      {
        "id": "model::model_79e042109d33",
        "type": "model",
        "label": "C2S-Scale",
        "model_id": "model_79e042109d33",
        "membership_group_id": "membership_2bfb9536cf22"
      },
      {
        "id": "model::model_a2ae3a8f5414",
        "type": "model",
        "label": "GEMGen",
        "model_id": "model_a2ae3a8f5414",
        "membership_group_id": "membership_2bfb9536cf22"
      },
      {
        "id": "model::model_58435d2084b0",
        "type": "model",
        "label": "gene_eng_gpt2_summary",
        "model_id": "model_58435d2084b0",
        "membership_group_id": "membership_2bfb9536cf22"
      },
      {
        "id": "model::model_2b474a3814ef",
        "type": "model",
        "label": "GPT-4",
        "model_id": "model_2b474a3814ef",
        "membership_group_id": "membership_2bfb9536cf22"
      },
      {
        "id": "model::model_4ae9868de1c9",
        "type": "model",
        "label": "Longevity-LLM v0.1",
        "model_id": "model_4ae9868de1c9",
        "membership_group_id": "membership_2bfb9536cf22"
      },
      {
        "id": "model::model_f924e02d892f",
        "type": "model",
        "label": "OCellus",
        "model_id": "model_f924e02d892f",
        "membership_group_id": "membership_2bfb9536cf22"
      },
      {
        "id": "model::model_df8ed6559ceb",
        "type": "model",
        "label": "TISSUENARRATOR",
        "model_id": "model_df8ed6559ceb",
        "membership_group_id": "membership_2bfb9536cf22"
      },
      {
        "id": "group::membership_1b67d9407546",
        "type": "membership_group",
        "label": "F1.L2",
        "group_id": "membership_1b67d9407546",
        "subtype_ids": [
          "structured_biological_prompt_or_task_scaffold"
        ],
        "family_ids": [
          "text_native_token_stream"
        ],
        "model_ids": [
          "model_a0f67b1880c7",
          "model_cc36a8d0a233",
          "model_6cd07bd7db7c",
          "model_fa94e391fe47",
          "model_1fb19463c978",
          "model_cabb39a8d967",
          "model_a62125696693",
          "model_8553f59d67fa",
          "model_4b363c370943",
          "model_c443713ee155",
          "model_78760ba29751",
          "model_61aa6fc79089",
          "model_af471cc4e529",
          "model_364e087c07ba",
          "model_073c1a9ed6a3",
          "model_7edc725404f8",
          "model_a95b39ba55bc",
          "model_c47c8bf166d6",
          "model_382b827ca741",
          "model_b47a3fca0d58",
          "model_ca02cecd45be",
          "model_4ba985125332"
        ],
        "model_count": 22
      },
      {
        "id": "model::model_a0f67b1880c7",
        "type": "model",
        "label": "Cell-o1",
        "model_id": "model_a0f67b1880c7",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_cc36a8d0a233",
        "type": "model",
        "label": "Deepseek-V3",
        "model_id": "model_cc36a8d0a233",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_6cd07bd7db7c",
        "type": "model",
        "label": "Gemma-7B",
        "model_id": "model_6cd07bd7db7c",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_fa94e391fe47",
        "type": "model",
        "label": "Geneverse",
        "model_id": "model_fa94e391fe47",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_1fb19463c978",
        "type": "model",
        "label": "GPT-4",
        "model_id": "model_1fb19463c978",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_cabb39a8d967",
        "type": "model",
        "label": "GPT-4o",
        "model_id": "model_cabb39a8d967",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_a62125696693",
        "type": "model",
        "label": "GPT-4o-mini",
        "model_id": "model_a62125696693",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_8553f59d67fa",
        "type": "model",
        "label": "GPT-OSS-120B",
        "model_id": "model_8553f59d67fa",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_4b363c370943",
        "type": "model",
        "label": "LLaMA2-13B",
        "model_id": "model_4b363c370943",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_c443713ee155",
        "type": "model",
        "label": "LLaMA2-7B",
        "model_id": "model_c443713ee155",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_78760ba29751",
        "type": "model",
        "label": "LLaMA3-8B",
        "model_id": "model_78760ba29751",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_61aa6fc79089",
        "type": "model",
        "label": "Llama3.1-8B-Instruct",
        "model_id": "model_61aa6fc79089",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_af471cc4e529",
        "type": "model",
        "label": "LLaMAPro-8B",
        "model_id": "model_af471cc4e529",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_364e087c07ba",
        "type": "model",
        "label": "Mistral-7B",
        "model_id": "model_364e087c07ba",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_073c1a9ed6a3",
        "type": "model",
        "label": "Mixtral 8x7b",
        "model_id": "model_073c1a9ed6a3",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_7edc725404f8",
        "type": "model",
        "label": "Mixtral 8x7b",
        "model_id": "model_7edc725404f8",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_a95b39ba55bc",
        "type": "model",
        "label": "o1",
        "model_id": "model_a95b39ba55bc",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_c47c8bf166d6",
        "type": "model",
        "label": "o3-mini",
        "model_id": "model_c47c8bf166d6",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_382b827ca741",
        "type": "model",
        "label": "o4-mini",
        "model_id": "model_382b827ca741",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_b47a3fca0d58",
        "type": "model",
        "label": "OpenAI's o1",
        "model_id": "model_b47a3fca0d58",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_ca02cecd45be",
        "type": "model",
        "label": "Qwen2.5-7B-Instruct",
        "model_id": "model_ca02cecd45be",
        "membership_group_id": "membership_1b67d9407546"
      },
      {
        "id": "model::model_4ba985125332",
        "type": "model",
        "label": "replaceable LLM backend",
        "model_id": "model_4ba985125332",
        "membership_group_id": "membership_1b67d9407546"
      }
    ],
    "edges": [
      {
        "id": "edge::root::text_native_token_stream",
        "source": "taxonomy_root",
        "target": "family::text_native_token_stream",
        "type": "contains_family"
      },
      {
        "id": "edge::text_native_token_stream::plain_language_prompt_or_question",
        "source": "family::text_native_token_stream",
        "target": "subtype::plain_language_prompt_or_question",
        "type": "contains_subtype"
      },
      {
        "id": "edge::text_native_token_stream::structured_biological_prompt_or_task_scaffold",
        "source": "family::text_native_token_stream",
        "target": "subtype::structured_biological_prompt_or_task_scaffold",
        "type": "contains_subtype"
      },
      {
        "id": "edge::text_native_token_stream::serialized_biological_context_or_ordered_profile",
        "source": "family::text_native_token_stream",
        "target": "subtype::serialized_biological_context_or_ordered_profile",
        "type": "contains_subtype"
      },
      {
        "id": "edge::root::discrete_biological_symbol_stream",
        "source": "taxonomy_root",
        "target": "family::discrete_biological_symbol_stream",
        "type": "contains_family"
      },
      {
        "id": "edge::discrete_biological_symbol_stream::native_biological_token_stream",
        "source": "family::discrete_biological_symbol_stream",
        "target": "subtype::native_biological_token_stream",
        "type": "contains_subtype"
      },
      {
        "id": "edge::discrete_biological_symbol_stream::multi_track_structural_symbol_stream",
        "source": "family::discrete_biological_symbol_stream",
        "target": "subtype::multi_track_structural_symbol_stream",
        "type": "contains_subtype"
      },
      {
        "id": "edge::discrete_biological_symbol_stream::learned_quantized_id_or_codebook_token",
        "source": "family::discrete_biological_symbol_stream",
        "target": "subtype::learned_quantized_id_or_codebook_token",
        "type": "contains_subtype"
      },
      {
        "id": "edge::root::dense_continuous_carrier",
        "source": "taxonomy_root",
        "target": "family::dense_continuous_carrier",
        "type": "contains_family"
      },
      {
        "id": "edge::dense_continuous_carrier::direct_projected_embedding",
        "source": "family::dense_continuous_carrier",
        "target": "subtype::direct_projected_embedding",
        "type": "contains_subtype"
      },
      {
        "id": "edge::dense_continuous_carrier::virtual_token_prefix",
        "source": "family::dense_continuous_carrier",
        "target": "subtype::virtual_token_prefix",
        "type": "contains_subtype"
      },
      {
        "id": "edge::dense_continuous_carrier::connector_mediated_embedding",
        "source": "family::dense_continuous_carrier",
        "target": "subtype::connector_mediated_embedding",
        "type": "contains_subtype"
      },
      {
        "id": "edge::dense_continuous_carrier::pooled_or_aggregated_embedding",
        "source": "family::dense_continuous_carrier",
        "target": "subtype::pooled_or_aggregated_embedding",
        "type": "contains_subtype"
      },
      {
        "id": "edge::root::visual_raster_carrier",
        "source": "taxonomy_root",
        "target": "family::visual_raster_carrier",
        "type": "contains_family"
      },
      {
        "id": "edge::visual_raster_carrier::raw_slide_or_patch_input",
        "source": "family::visual_raster_carrier",
        "target": "subtype::raw_slide_or_patch_input",
        "type": "contains_subtype"
      },
      {
        "id": "edge::visual_raster_carrier::patch_context_or_case_level_visual_reasoning",
        "source": "family::visual_raster_carrier",
        "target": "subtype::patch_context_or_case_level_visual_reasoning",
        "type": "contains_subtype"
      },
      {
        "id": "edge::root::geometric_or_diffusion_state_carrier",
        "source": "taxonomy_root",
        "target": "family::geometric_or_diffusion_state_carrier",
        "type": "contains_family"
      },
      {
        "id": "edge::geometric_or_diffusion_state_carrier::noisy_diffusion_state",
        "source": "family::geometric_or_diffusion_state_carrier",
        "target": "subtype::noisy_diffusion_state",
        "type": "contains_subtype"
      },
      {
        "id": "edge::geometric_or_diffusion_state_carrier::coordinate_backbone_or_shape_conditioning",
        "source": "family::geometric_or_diffusion_state_carrier",
        "target": "subtype::coordinate_backbone_or_shape_conditioning",
        "type": "contains_subtype"
      },
      {
        "id": "edge::geometric_or_diffusion_state_carrier::symbolic_structural_constraint",
        "source": "family::geometric_or_diffusion_state_carrier",
        "target": "subtype::symbolic_structural_constraint",
        "type": "contains_subtype"
      },
      {
        "id": "edge::connector_mediated_embedding::membership_e367b465c0b9",
        "source": "subtype::connector_mediated_embedding",
        "target": "group::membership_e367b465c0b9",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_e367b465c0b9::model_6f7f8cee95d1",
        "source": "group::membership_e367b465c0b9",
        "target": "model::model_6f7f8cee95d1",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_e367b465c0b9::model_16a912645955",
        "source": "group::membership_e367b465c0b9",
        "target": "model::model_16a912645955",
        "type": "contains_model"
      },
      {
        "id": "edge::connector_mediated_embedding::membership_20ed14556e30",
        "source": "subtype::connector_mediated_embedding",
        "target": "group::membership_20ed14556e30",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::direct_projected_embedding::membership_20ed14556e30",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_20ed14556e30",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_20ed14556e30::model_2a4d0a37a0ad",
        "source": "group::membership_20ed14556e30",
        "target": "model::model_2a4d0a37a0ad",
        "type": "contains_model"
      },
      {
        "id": "edge::connector_mediated_embedding::membership_686e999550fc",
        "source": "subtype::connector_mediated_embedding",
        "target": "group::membership_686e999550fc",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::direct_projected_embedding::membership_686e999550fc",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_686e999550fc",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::noisy_diffusion_state::membership_686e999550fc",
        "source": "subtype::noisy_diffusion_state",
        "target": "group::membership_686e999550fc",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_686e999550fc",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_686e999550fc",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::pooled_or_aggregated_embedding::membership_686e999550fc",
        "source": "subtype::pooled_or_aggregated_embedding",
        "target": "group::membership_686e999550fc",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::raw_slide_or_patch_input::membership_686e999550fc",
        "source": "subtype::raw_slide_or_patch_input",
        "target": "group::membership_686e999550fc",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_686e999550fc",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_686e999550fc",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_686e999550fc::model_59fd1ac5f56e",
        "source": "group::membership_686e999550fc",
        "target": "model::model_59fd1ac5f56e",
        "type": "contains_model"
      },
      {
        "id": "edge::connector_mediated_embedding::membership_aab758294f91",
        "source": "subtype::connector_mediated_embedding",
        "target": "group::membership_aab758294f91",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::direct_projected_embedding::membership_aab758294f91",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_aab758294f91",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_aab758294f91",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_aab758294f91",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_aab758294f91",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_aab758294f91",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_aab758294f91::model_f8e766fddf66",
        "source": "group::membership_aab758294f91",
        "target": "model::model_f8e766fddf66",
        "type": "contains_model"
      },
      {
        "id": "edge::connector_mediated_embedding::membership_6742df78bf4f",
        "source": "subtype::connector_mediated_embedding",
        "target": "group::membership_6742df78bf4f",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::direct_projected_embedding::membership_6742df78bf4f",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_6742df78bf4f",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::pooled_or_aggregated_embedding::membership_6742df78bf4f",
        "source": "subtype::pooled_or_aggregated_embedding",
        "target": "group::membership_6742df78bf4f",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_6742df78bf4f",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_6742df78bf4f",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_6742df78bf4f",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_6742df78bf4f",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_6742df78bf4f::model_860d9511b312",
        "source": "group::membership_6742df78bf4f",
        "target": "model::model_860d9511b312",
        "type": "contains_model"
      },
      {
        "id": "edge::connector_mediated_embedding::membership_433657cba128",
        "source": "subtype::connector_mediated_embedding",
        "target": "group::membership_433657cba128",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::direct_projected_embedding::membership_433657cba128",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_433657cba128",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_433657cba128",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_433657cba128",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_433657cba128::model_ffaa8b52273c",
        "source": "group::membership_433657cba128",
        "target": "model::model_ffaa8b52273c",
        "type": "contains_model"
      },
      {
        "id": "edge::connector_mediated_embedding::membership_47e8053eae22",
        "source": "subtype::connector_mediated_embedding",
        "target": "group::membership_47e8053eae22",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_47e8053eae22",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_47e8053eae22",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_47e8053eae22",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_47e8053eae22",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_47e8053eae22::model_f1bf6497dcf0",
        "source": "group::membership_47e8053eae22",
        "target": "model::model_f1bf6497dcf0",
        "type": "contains_model"
      },
      {
        "id": "edge::connector_mediated_embedding::membership_dbca3a16da8f",
        "source": "subtype::connector_mediated_embedding",
        "target": "group::membership_dbca3a16da8f",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_dbca3a16da8f",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_dbca3a16da8f",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_dbca3a16da8f::model_867ffa020512",
        "source": "group::membership_dbca3a16da8f",
        "target": "model::model_867ffa020512",
        "type": "contains_model"
      },
      {
        "id": "edge::connector_mediated_embedding::membership_ee721887da1b",
        "source": "subtype::connector_mediated_embedding",
        "target": "group::membership_ee721887da1b",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::virtual_token_prefix::membership_ee721887da1b",
        "source": "subtype::virtual_token_prefix",
        "target": "group::membership_ee721887da1b",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_ee721887da1b::model_699b7d92d927",
        "source": "group::membership_ee721887da1b",
        "target": "model::model_699b7d92d927",
        "type": "contains_model"
      },
      {
        "id": "edge::coordinate_backbone_or_shape_conditioning::membership_8ec5e76d5e6a",
        "source": "subtype::coordinate_backbone_or_shape_conditioning",
        "target": "group::membership_8ec5e76d5e6a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::direct_projected_embedding::membership_8ec5e76d5e6a",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_8ec5e76d5e6a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::noisy_diffusion_state::membership_8ec5e76d5e6a",
        "source": "subtype::noisy_diffusion_state",
        "target": "group::membership_8ec5e76d5e6a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_8ec5e76d5e6a",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_8ec5e76d5e6a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_8ec5e76d5e6a",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_8ec5e76d5e6a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_8ec5e76d5e6a::model_fcda05dbbcd5",
        "source": "group::membership_8ec5e76d5e6a",
        "target": "model::model_fcda05dbbcd5",
        "type": "contains_model"
      },
      {
        "id": "edge::coordinate_backbone_or_shape_conditioning::membership_af3baba835a0",
        "source": "subtype::coordinate_backbone_or_shape_conditioning",
        "target": "group::membership_af3baba835a0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::native_biological_token_stream::membership_af3baba835a0",
        "source": "subtype::native_biological_token_stream",
        "target": "group::membership_af3baba835a0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_af3baba835a0",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_af3baba835a0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_af3baba835a0::model_b1f724b128da",
        "source": "group::membership_af3baba835a0",
        "target": "model::model_b1f724b128da",
        "type": "contains_model"
      },
      {
        "id": "edge::coordinate_backbone_or_shape_conditioning::membership_04e6eca108d0",
        "source": "subtype::coordinate_backbone_or_shape_conditioning",
        "target": "group::membership_04e6eca108d0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::noisy_diffusion_state::membership_04e6eca108d0",
        "source": "subtype::noisy_diffusion_state",
        "target": "group::membership_04e6eca108d0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_04e6eca108d0",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_04e6eca108d0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::symbolic_structural_constraint::membership_04e6eca108d0",
        "source": "subtype::symbolic_structural_constraint",
        "target": "group::membership_04e6eca108d0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_04e6eca108d0::model_efc702564a41",
        "source": "group::membership_04e6eca108d0",
        "target": "model::model_efc702564a41",
        "type": "contains_model"
      },
      {
        "id": "edge::direct_projected_embedding::membership_000c4dd8661a",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_000c4dd8661a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_000c4dd8661a::model_9be63ad52f90",
        "source": "group::membership_000c4dd8661a",
        "target": "model::model_9be63ad52f90",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_000c4dd8661a::model_35105bbde921",
        "source": "group::membership_000c4dd8661a",
        "target": "model::model_35105bbde921",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_000c4dd8661a::model_99366b543213",
        "source": "group::membership_000c4dd8661a",
        "target": "model::model_99366b543213",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_000c4dd8661a::model_9c283f71363e",
        "source": "group::membership_000c4dd8661a",
        "target": "model::model_9c283f71363e",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_000c4dd8661a::model_422e09e03da5",
        "source": "group::membership_000c4dd8661a",
        "target": "model::model_422e09e03da5",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_000c4dd8661a::model_b4a92e5e540d",
        "source": "group::membership_000c4dd8661a",
        "target": "model::model_b4a92e5e540d",
        "type": "contains_model"
      },
      {
        "id": "edge::direct_projected_embedding::membership_22936f4d27e9",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_22936f4d27e9",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::native_biological_token_stream::membership_22936f4d27e9",
        "source": "subtype::native_biological_token_stream",
        "target": "group::membership_22936f4d27e9",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_22936f4d27e9::model_0f8f776fb068",
        "source": "group::membership_22936f4d27e9",
        "target": "model::model_0f8f776fb068",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_22936f4d27e9::model_9b1025af9fe7",
        "source": "group::membership_22936f4d27e9",
        "target": "model::model_9b1025af9fe7",
        "type": "contains_model"
      },
      {
        "id": "edge::direct_projected_embedding::membership_cab7fec0fcf3",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_cab7fec0fcf3",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_cab7fec0fcf3",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_cab7fec0fcf3",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_cab7fec0fcf3",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_cab7fec0fcf3",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_cab7fec0fcf3",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_cab7fec0fcf3",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_cab7fec0fcf3::model_064ca265f9bb",
        "source": "group::membership_cab7fec0fcf3",
        "target": "model::model_064ca265f9bb",
        "type": "contains_model"
      },
      {
        "id": "edge::direct_projected_embedding::membership_db6432d4d322",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_db6432d4d322",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_db6432d4d322",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_db6432d4d322",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_db6432d4d322",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_db6432d4d322",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_db6432d4d322::model_b8eba3c26c46",
        "source": "group::membership_db6432d4d322",
        "target": "model::model_b8eba3c26c46",
        "type": "contains_model"
      },
      {
        "id": "edge::direct_projected_embedding::membership_d7b71bdfa7dd",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_d7b71bdfa7dd",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::pooled_or_aggregated_embedding::membership_d7b71bdfa7dd",
        "source": "subtype::pooled_or_aggregated_embedding",
        "target": "group::membership_d7b71bdfa7dd",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_d7b71bdfa7dd::model_3c62ea9d84a1",
        "source": "group::membership_d7b71bdfa7dd",
        "target": "model::model_3c62ea9d84a1",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_d7b71bdfa7dd::model_d3b03caa65b1",
        "source": "group::membership_d7b71bdfa7dd",
        "target": "model::model_d3b03caa65b1",
        "type": "contains_model"
      },
      {
        "id": "edge::direct_projected_embedding::membership_d445170f41eb",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_d445170f41eb",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::pooled_or_aggregated_embedding::membership_d445170f41eb",
        "source": "subtype::pooled_or_aggregated_embedding",
        "target": "group::membership_d445170f41eb",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_d445170f41eb",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_d445170f41eb",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_d445170f41eb::model_4361ad5c9fbb",
        "source": "group::membership_d445170f41eb",
        "target": "model::model_4361ad5c9fbb",
        "type": "contains_model"
      },
      {
        "id": "edge::direct_projected_embedding::membership_cf2af8e28dd7",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_cf2af8e28dd7",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_cf2af8e28dd7",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_cf2af8e28dd7",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_cf2af8e28dd7::model_5f2f0ba1126a",
        "source": "group::membership_cf2af8e28dd7",
        "target": "model::model_5f2f0ba1126a",
        "type": "contains_model"
      },
      {
        "id": "edge::direct_projected_embedding::membership_c17be9263882",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_c17be9263882",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_c17be9263882",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_c17be9263882",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_c17be9263882::model_77f5da416a80",
        "source": "group::membership_c17be9263882",
        "target": "model::model_77f5da416a80",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_c17be9263882::model_285e25ae2bfb",
        "source": "group::membership_c17be9263882",
        "target": "model::model_285e25ae2bfb",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_c17be9263882::model_9d05ddb97916",
        "source": "group::membership_c17be9263882",
        "target": "model::model_9d05ddb97916",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_c17be9263882::model_ac7794e51425",
        "source": "group::membership_c17be9263882",
        "target": "model::model_ac7794e51425",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_c17be9263882::model_a493b5670a18",
        "source": "group::membership_c17be9263882",
        "target": "model::model_a493b5670a18",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_c17be9263882::model_ab432decc9aa",
        "source": "group::membership_c17be9263882",
        "target": "model::model_ab432decc9aa",
        "type": "contains_model"
      },
      {
        "id": "edge::direct_projected_embedding::membership_2e20f0c5603e",
        "source": "subtype::direct_projected_embedding",
        "target": "group::membership_2e20f0c5603e",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::symbolic_structural_constraint::membership_2e20f0c5603e",
        "source": "subtype::symbolic_structural_constraint",
        "target": "group::membership_2e20f0c5603e",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_2e20f0c5603e::model_d3d6352d5bcb",
        "source": "group::membership_2e20f0c5603e",
        "target": "model::model_d3d6352d5bcb",
        "type": "contains_model"
      },
      {
        "id": "edge::learned_quantized_id_or_codebook_token::membership_fbfad76cabfd",
        "source": "subtype::learned_quantized_id_or_codebook_token",
        "target": "group::membership_fbfad76cabfd",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::multi_track_structural_symbol_stream::membership_fbfad76cabfd",
        "source": "subtype::multi_track_structural_symbol_stream",
        "target": "group::membership_fbfad76cabfd",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_fbfad76cabfd",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_fbfad76cabfd",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_fbfad76cabfd",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_fbfad76cabfd",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_fbfad76cabfd::model_7ccbcc7f850b",
        "source": "group::membership_fbfad76cabfd",
        "target": "model::model_7ccbcc7f850b",
        "type": "contains_model"
      },
      {
        "id": "edge::multi_track_structural_symbol_stream::membership_951dfcefa907",
        "source": "subtype::multi_track_structural_symbol_stream",
        "target": "group::membership_951dfcefa907",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_951dfcefa907::model_58c5ce11b66a",
        "source": "group::membership_951dfcefa907",
        "target": "model::model_58c5ce11b66a",
        "type": "contains_model"
      },
      {
        "id": "edge::multi_track_structural_symbol_stream::membership_493db55fa177",
        "source": "subtype::multi_track_structural_symbol_stream",
        "target": "group::membership_493db55fa177",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::native_biological_token_stream::membership_493db55fa177",
        "source": "subtype::native_biological_token_stream",
        "target": "group::membership_493db55fa177",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_493db55fa177",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_493db55fa177",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_493db55fa177",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_493db55fa177",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_493db55fa177::model_74e685d85fa8",
        "source": "group::membership_493db55fa177",
        "target": "model::model_74e685d85fa8",
        "type": "contains_model"
      },
      {
        "id": "edge::native_biological_token_stream::membership_de94be356588",
        "source": "subtype::native_biological_token_stream",
        "target": "group::membership_de94be356588",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_de94be356588::model_eb69df38d690",
        "source": "group::membership_de94be356588",
        "target": "model::model_eb69df38d690",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_de94be356588::model_498b709f811c",
        "source": "group::membership_de94be356588",
        "target": "model::model_498b709f811c",
        "type": "contains_model"
      },
      {
        "id": "edge::native_biological_token_stream::membership_fdbd3ba264a9",
        "source": "subtype::native_biological_token_stream",
        "target": "group::membership_fdbd3ba264a9",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_fdbd3ba264a9",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_fdbd3ba264a9",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_fdbd3ba264a9::model_5421186d4cbe",
        "source": "group::membership_fdbd3ba264a9",
        "target": "model::model_5421186d4cbe",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_fdbd3ba264a9::model_3841ec0733e7",
        "source": "group::membership_fdbd3ba264a9",
        "target": "model::model_3841ec0733e7",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_fdbd3ba264a9::model_d2833fcd27aa",
        "source": "group::membership_fdbd3ba264a9",
        "target": "model::model_d2833fcd27aa",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_fdbd3ba264a9::model_5fd880eafa1a",
        "source": "group::membership_fdbd3ba264a9",
        "target": "model::model_5fd880eafa1a",
        "type": "contains_model"
      },
      {
        "id": "edge::native_biological_token_stream::membership_ad9cb3ec7dd7",
        "source": "subtype::native_biological_token_stream",
        "target": "group::membership_ad9cb3ec7dd7",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_ad9cb3ec7dd7",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_ad9cb3ec7dd7",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_ad9cb3ec7dd7",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_ad9cb3ec7dd7",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_ad9cb3ec7dd7::model_74758c50e662",
        "source": "group::membership_ad9cb3ec7dd7",
        "target": "model::model_74758c50e662",
        "type": "contains_model"
      },
      {
        "id": "edge::native_biological_token_stream::membership_8d5343810816",
        "source": "subtype::native_biological_token_stream",
        "target": "group::membership_8d5343810816",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_8d5343810816",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_8d5343810816",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_8d5343810816",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_8d5343810816",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_8d5343810816::model_5e9bb9e443dc",
        "source": "group::membership_8d5343810816",
        "target": "model::model_5e9bb9e443dc",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_8d5343810816::model_b4d1be45d2bb",
        "source": "group::membership_8d5343810816",
        "target": "model::model_b4d1be45d2bb",
        "type": "contains_model"
      },
      {
        "id": "edge::native_biological_token_stream::membership_aaacf24f224c",
        "source": "subtype::native_biological_token_stream",
        "target": "group::membership_aaacf24f224c",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_aaacf24f224c",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_aaacf24f224c",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_aaacf24f224c",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_aaacf24f224c",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_aaacf24f224c::model_69a439438d24",
        "source": "group::membership_aaacf24f224c",
        "target": "model::model_69a439438d24",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_aaacf24f224c::model_958497e0e4e2",
        "source": "group::membership_aaacf24f224c",
        "target": "model::model_958497e0e4e2",
        "type": "contains_model"
      },
      {
        "id": "edge::patch_context_or_case_level_visual_reasoning::membership_cbdfdd5c56c0",
        "source": "subtype::patch_context_or_case_level_visual_reasoning",
        "target": "group::membership_cbdfdd5c56c0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_cbdfdd5c56c0",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_cbdfdd5c56c0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::raw_slide_or_patch_input::membership_cbdfdd5c56c0",
        "source": "subtype::raw_slide_or_patch_input",
        "target": "group::membership_cbdfdd5c56c0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_cbdfdd5c56c0",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_cbdfdd5c56c0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_cbdfdd5c56c0",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_cbdfdd5c56c0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_cbdfdd5c56c0::model_23ffab302d6b",
        "source": "group::membership_cbdfdd5c56c0",
        "target": "model::model_23ffab302d6b",
        "type": "contains_model"
      },
      {
        "id": "edge::patch_context_or_case_level_visual_reasoning::membership_10855a0f50f0",
        "source": "subtype::patch_context_or_case_level_visual_reasoning",
        "target": "group::membership_10855a0f50f0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::raw_slide_or_patch_input::membership_10855a0f50f0",
        "source": "subtype::raw_slide_or_patch_input",
        "target": "group::membership_10855a0f50f0",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_10855a0f50f0::model_3827a28edefc",
        "source": "group::membership_10855a0f50f0",
        "target": "model::model_3827a28edefc",
        "type": "contains_model"
      },
      {
        "id": "edge::patch_context_or_case_level_visual_reasoning::membership_887038a8a19a",
        "source": "subtype::patch_context_or_case_level_visual_reasoning",
        "target": "group::membership_887038a8a19a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::raw_slide_or_patch_input::membership_887038a8a19a",
        "source": "subtype::raw_slide_or_patch_input",
        "target": "group::membership_887038a8a19a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::virtual_token_prefix::membership_887038a8a19a",
        "source": "subtype::virtual_token_prefix",
        "target": "group::membership_887038a8a19a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_887038a8a19a::model_f67272a54ac0",
        "source": "group::membership_887038a8a19a",
        "target": "model::model_f67272a54ac0",
        "type": "contains_model"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_357f1bf88b0b",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_357f1bf88b0b",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_357f1bf88b0b::model_0e0c28b60ee7",
        "source": "group::membership_357f1bf88b0b",
        "target": "model::model_0e0c28b60ee7",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_357f1bf88b0b::model_7ed6448bcdeb",
        "source": "group::membership_357f1bf88b0b",
        "target": "model::model_7ed6448bcdeb",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_357f1bf88b0b::model_c7150172725e",
        "source": "group::membership_357f1bf88b0b",
        "target": "model::model_c7150172725e",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_357f1bf88b0b::model_f12e33a1e764",
        "source": "group::membership_357f1bf88b0b",
        "target": "model::model_f12e33a1e764",
        "type": "contains_model"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_087f460aac0a",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_087f460aac0a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::pooled_or_aggregated_embedding::membership_087f460aac0a",
        "source": "subtype::pooled_or_aggregated_embedding",
        "target": "group::membership_087f460aac0a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_087f460aac0a::model_0e34c1fd11f6",
        "source": "group::membership_087f460aac0a",
        "target": "model::model_0e34c1fd11f6",
        "type": "contains_model"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_22149166d2ab",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_22149166d2ab",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::pooled_or_aggregated_embedding::membership_22149166d2ab",
        "source": "subtype::pooled_or_aggregated_embedding",
        "target": "group::membership_22149166d2ab",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::raw_slide_or_patch_input::membership_22149166d2ab",
        "source": "subtype::raw_slide_or_patch_input",
        "target": "group::membership_22149166d2ab",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_22149166d2ab::model_a45de7f446e9",
        "source": "group::membership_22149166d2ab",
        "target": "model::model_a45de7f446e9",
        "type": "contains_model"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_9337e69362f7",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_9337e69362f7",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::pooled_or_aggregated_embedding::membership_9337e69362f7",
        "source": "subtype::pooled_or_aggregated_embedding",
        "target": "group::membership_9337e69362f7",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_9337e69362f7",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_9337e69362f7",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_9337e69362f7",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_9337e69362f7",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_9337e69362f7::model_c0a097206f2c",
        "source": "group::membership_9337e69362f7",
        "target": "model::model_c0a097206f2c",
        "type": "contains_model"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_7ec27bf438d4",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_7ec27bf438d4",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::raw_slide_or_patch_input::membership_7ec27bf438d4",
        "source": "subtype::raw_slide_or_patch_input",
        "target": "group::membership_7ec27bf438d4",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_7ec27bf438d4",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_7ec27bf438d4",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_7ec27bf438d4::model_6a1d79c2e81c",
        "source": "group::membership_7ec27bf438d4",
        "target": "model::model_6a1d79c2e81c",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_7ec27bf438d4::model_ff0ac12f6b2a",
        "source": "group::membership_7ec27bf438d4",
        "target": "model::model_ff0ac12f6b2a",
        "type": "contains_model"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_d8f17c473542",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_d8f17c473542",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_d8f17c473542",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_d8f17c473542",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_d8f17c473542::model_0b491a7a5d54",
        "source": "group::membership_d8f17c473542",
        "target": "model::model_0b491a7a5d54",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_d8f17c473542::model_3d02d9393c92",
        "source": "group::membership_d8f17c473542",
        "target": "model::model_3d02d9393c92",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_d8f17c473542::model_36c6451d1b55",
        "source": "group::membership_d8f17c473542",
        "target": "model::model_36c6451d1b55",
        "type": "contains_model"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_67d200027193",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_67d200027193",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_67d200027193",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_67d200027193",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_67d200027193",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_67d200027193",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_67d200027193::model_68562cc69016",
        "source": "group::membership_67d200027193",
        "target": "model::model_68562cc69016",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_67d200027193::model_72e70c356b7a",
        "source": "group::membership_67d200027193",
        "target": "model::model_72e70c356b7a",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_67d200027193::model_6694aa9f2590",
        "source": "group::membership_67d200027193",
        "target": "model::model_6694aa9f2590",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_67d200027193::model_0b0332ce2b16",
        "source": "group::membership_67d200027193",
        "target": "model::model_0b0332ce2b16",
        "type": "contains_model"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_acce00dbb33d",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_acce00dbb33d",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_acce00dbb33d",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_acce00dbb33d",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_acce00dbb33d",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_acce00dbb33d",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::virtual_token_prefix::membership_acce00dbb33d",
        "source": "subtype::virtual_token_prefix",
        "target": "group::membership_acce00dbb33d",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_acce00dbb33d::model_8b27781d1156",
        "source": "group::membership_acce00dbb33d",
        "target": "model::model_8b27781d1156",
        "type": "contains_model"
      },
      {
        "id": "edge::plain_language_prompt_or_question::membership_85e0c6746826",
        "source": "subtype::plain_language_prompt_or_question",
        "target": "group::membership_85e0c6746826",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_85e0c6746826",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_85e0c6746826",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_85e0c6746826::model_6b93a9600bf1",
        "source": "group::membership_85e0c6746826",
        "target": "model::model_6b93a9600bf1",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_85e0c6746826::model_fa039bd90d50",
        "source": "group::membership_85e0c6746826",
        "target": "model::model_fa039bd90d50",
        "type": "contains_model"
      },
      {
        "id": "edge::pooled_or_aggregated_embedding::membership_998d40c02fca",
        "source": "subtype::pooled_or_aggregated_embedding",
        "target": "group::membership_998d40c02fca",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_998d40c02fca::model_0dc740f11cc3",
        "source": "group::membership_998d40c02fca",
        "target": "model::model_0dc740f11cc3",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_998d40c02fca::model_1bd0fe9134d7",
        "source": "group::membership_998d40c02fca",
        "target": "model::model_1bd0fe9134d7",
        "type": "contains_model"
      },
      {
        "id": "edge::raw_slide_or_patch_input::membership_cb06ed43c31a",
        "source": "subtype::raw_slide_or_patch_input",
        "target": "group::membership_cb06ed43c31a",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_cb06ed43c31a::model_5588f179e503",
        "source": "group::membership_cb06ed43c31a",
        "target": "model::model_5588f179e503",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_cb06ed43c31a::model_f35a32a6436b",
        "source": "group::membership_cb06ed43c31a",
        "target": "model::model_f35a32a6436b",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_cb06ed43c31a::model_1bf0ccb8259f",
        "source": "group::membership_cb06ed43c31a",
        "target": "model::model_1bf0ccb8259f",
        "type": "contains_model"
      },
      {
        "id": "edge::raw_slide_or_patch_input::membership_d5e749b180fc",
        "source": "subtype::raw_slide_or_patch_input",
        "target": "group::membership_d5e749b180fc",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_d5e749b180fc",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_d5e749b180fc",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_d5e749b180fc::model_e0f70ec41e97",
        "source": "group::membership_d5e749b180fc",
        "target": "model::model_e0f70ec41e97",
        "type": "contains_model"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_0cc72bd82901",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_0cc72bd82901",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_0cc72bd82901::model_fba98a35871c",
        "source": "group::membership_0cc72bd82901",
        "target": "model::model_fba98a35871c",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_0cc72bd82901::model_9ead93555f87",
        "source": "group::membership_0cc72bd82901",
        "target": "model::model_9ead93555f87",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_0cc72bd82901::model_14bc733e1633",
        "source": "group::membership_0cc72bd82901",
        "target": "model::model_14bc733e1633",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_0cc72bd82901::model_0ea0b64c5788",
        "source": "group::membership_0cc72bd82901",
        "target": "model::model_0ea0b64c5788",
        "type": "contains_model"
      },
      {
        "id": "edge::serialized_biological_context_or_ordered_profile::membership_2bfb9536cf22",
        "source": "subtype::serialized_biological_context_or_ordered_profile",
        "target": "group::membership_2bfb9536cf22",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_2bfb9536cf22",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_2bfb9536cf22",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_2bfb9536cf22::model_79e042109d33",
        "source": "group::membership_2bfb9536cf22",
        "target": "model::model_79e042109d33",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_2bfb9536cf22::model_a2ae3a8f5414",
        "source": "group::membership_2bfb9536cf22",
        "target": "model::model_a2ae3a8f5414",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_2bfb9536cf22::model_58435d2084b0",
        "source": "group::membership_2bfb9536cf22",
        "target": "model::model_58435d2084b0",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_2bfb9536cf22::model_2b474a3814ef",
        "source": "group::membership_2bfb9536cf22",
        "target": "model::model_2b474a3814ef",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_2bfb9536cf22::model_4ae9868de1c9",
        "source": "group::membership_2bfb9536cf22",
        "target": "model::model_4ae9868de1c9",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_2bfb9536cf22::model_f924e02d892f",
        "source": "group::membership_2bfb9536cf22",
        "target": "model::model_f924e02d892f",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_2bfb9536cf22::model_df8ed6559ceb",
        "source": "group::membership_2bfb9536cf22",
        "target": "model::model_df8ed6559ceb",
        "type": "contains_model"
      },
      {
        "id": "edge::structured_biological_prompt_or_task_scaffold::membership_1b67d9407546",
        "source": "subtype::structured_biological_prompt_or_task_scaffold",
        "target": "group::membership_1b67d9407546",
        "type": "defines_membership_group"
      },
      {
        "id": "edge::membership_1b67d9407546::model_a0f67b1880c7",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_a0f67b1880c7",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_cc36a8d0a233",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_cc36a8d0a233",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_6cd07bd7db7c",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_6cd07bd7db7c",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_fa94e391fe47",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_fa94e391fe47",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_1fb19463c978",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_1fb19463c978",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_cabb39a8d967",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_cabb39a8d967",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_a62125696693",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_a62125696693",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_8553f59d67fa",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_8553f59d67fa",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_4b363c370943",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_4b363c370943",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_c443713ee155",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_c443713ee155",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_78760ba29751",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_78760ba29751",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_61aa6fc79089",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_61aa6fc79089",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_af471cc4e529",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_af471cc4e529",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_364e087c07ba",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_364e087c07ba",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_073c1a9ed6a3",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_073c1a9ed6a3",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_7edc725404f8",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_7edc725404f8",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_a95b39ba55bc",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_a95b39ba55bc",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_c47c8bf166d6",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_c47c8bf166d6",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_382b827ca741",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_382b827ca741",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_b47a3fca0d58",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_b47a3fca0d58",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_ca02cecd45be",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_ca02cecd45be",
        "type": "contains_model"
      },
      {
        "id": "edge::membership_1b67d9407546::model_4ba985125332",
        "source": "group::membership_1b67d9407546",
        "target": "model::model_4ba985125332",
        "type": "contains_model"
      }
    ],
    "counts": {
      "root": 1,
      "families": 5,
      "subtypes": 15,
      "membership_groups": 47,
      "models": 109,
      "edges": 254
    }
  },
  "filter_values": {
    "review_iterations": [
      {
        "iteration_id": "review_iteration_2026-08-09",
        "date": "2026-08-09",
        "record_count": 3,
        "record_ids": [
          "update_2026-08-09__manual_recall_xunzi",
          "update_2026-08-09__rec_000106",
          "update_2026-08-09__rec_000138"
        ],
        "model_count": 6,
        "route_count": 37
      },
      {
        "iteration_id": "review_iteration_2026-07-06",
        "date": "2026-07-06",
        "record_count": 44,
        "record_ids": [
          "full_2026-07-06__rec_000060",
          "full_2026-07-06__rec_000063",
          "full_2026-07-06__rec_000086",
          "full_2026-07-06__rec_000090",
          "full_2026-07-06__rec_000771",
          "full_2026-07-06__rec_000827",
          "full_2026-07-06__rec_000950",
          "full_2026-07-06__rec_001074",
          "full_2026-07-06__rec_001126",
          "full_2026-07-06__rec_001187",
          "full_2026-07-06__rec_001200",
          "full_2026-07-06__rec_001218",
          "full_2026-07-06__rec_001277",
          "full_2026-07-06__rec_001319",
          "full_2026-07-06__rec_001343",
          "full_2026-07-06__rec_001352",
          "full_2026-07-06__rec_001381",
          "full_2026-07-06__rec_001519",
          "full_2026-07-06__rec_001617",
          "full_2026-07-06__rec_001687",
          "full_2026-07-06__rec_001773",
          "full_2026-07-06__rec_001830",
          "full_2026-07-06__rec_001838",
          "full_2026-07-06__rec_001889",
          "full_2026-07-06__rec_002049",
          "full_2026-07-06__rec_002105",
          "full_2026-07-06__rec_002243",
          "full_2026-07-06__rec_002304",
          "full_2026-07-06__rec_002327",
          "full_2026-07-06__rec_002517",
          "full_2026-07-06__rec_003008",
          "full_2026-07-06__rec_003043",
          "full_2026-07-06__rec_003131",
          "full_2026-07-06__rec_003188",
          "full_2026-07-06__rec_003206",
          "full_2026-07-06__rec_003214",
          "full_2026-07-06__rec_003323",
          "full_2026-07-06__rec_003394",
          "full_2026-07-06__rec_003434",
          "full_2026-07-06__rec_003517",
          "full_2026-07-06__rec_003629",
          "full_2026-07-06__rec_003852",
          "july_update_2026-07-06__rec_000050",
          "july_update_2026-07-06__rec_000060"
        ],
        "model_count": 91,
        "route_count": 473
      },
      {
        "iteration_id": "review_iteration_2026-06-10",
        "date": "2026-06-10",
        "record_count": 8,
        "record_ids": [
          "june_update_2026-06-10__rec_000121",
          "june_update_2026-06-10__rec_000133",
          "june_update_2026-06-10__rec_000148",
          "june_update_2026-06-10__rec_000152",
          "june_update_2026-06-10__rec_000194",
          "june_update_2026-06-10__rec_000246",
          "june_update_2026-06-10__rec_000248",
          "june_update_2026-06-10__rec_000350"
        ],
        "model_count": 12,
        "route_count": 75
      }
    ],
    "collection_batches": [
      {
        "iteration_id": "review_iteration_2026-08-09",
        "date": "2026-08-09",
        "record_count": 3,
        "record_ids": [
          "update_2026-08-09__manual_recall_xunzi",
          "update_2026-08-09__rec_000106",
          "update_2026-08-09__rec_000138"
        ],
        "model_count": 6,
        "route_count": 37
      },
      {
        "iteration_id": "review_iteration_2026-07-06",
        "date": "2026-07-06",
        "record_count": 44,
        "record_ids": [
          "full_2026-07-06__rec_000060",
          "full_2026-07-06__rec_000063",
          "full_2026-07-06__rec_000086",
          "full_2026-07-06__rec_000090",
          "full_2026-07-06__rec_000771",
          "full_2026-07-06__rec_000827",
          "full_2026-07-06__rec_000950",
          "full_2026-07-06__rec_001074",
          "full_2026-07-06__rec_001126",
          "full_2026-07-06__rec_001187",
          "full_2026-07-06__rec_001200",
          "full_2026-07-06__rec_001218",
          "full_2026-07-06__rec_001277",
          "full_2026-07-06__rec_001319",
          "full_2026-07-06__rec_001343",
          "full_2026-07-06__rec_001352",
          "full_2026-07-06__rec_001381",
          "full_2026-07-06__rec_001519",
          "full_2026-07-06__rec_001617",
          "full_2026-07-06__rec_001687",
          "full_2026-07-06__rec_001773",
          "full_2026-07-06__rec_001830",
          "full_2026-07-06__rec_001838",
          "full_2026-07-06__rec_001889",
          "full_2026-07-06__rec_002049",
          "full_2026-07-06__rec_002105",
          "full_2026-07-06__rec_002243",
          "full_2026-07-06__rec_002304",
          "full_2026-07-06__rec_002327",
          "full_2026-07-06__rec_002517",
          "full_2026-07-06__rec_003008",
          "full_2026-07-06__rec_003043",
          "full_2026-07-06__rec_003131",
          "full_2026-07-06__rec_003188",
          "full_2026-07-06__rec_003206",
          "full_2026-07-06__rec_003214",
          "full_2026-07-06__rec_003323",
          "full_2026-07-06__rec_003394",
          "full_2026-07-06__rec_003434",
          "full_2026-07-06__rec_003517",
          "full_2026-07-06__rec_003629",
          "full_2026-07-06__rec_003852",
          "july_update_2026-07-06__rec_000050",
          "july_update_2026-07-06__rec_000060"
        ],
        "model_count": 91,
        "route_count": 473
      },
      {
        "iteration_id": "review_iteration_2026-06-10",
        "date": "2026-06-10",
        "record_count": 8,
        "record_ids": [
          "june_update_2026-06-10__rec_000121",
          "june_update_2026-06-10__rec_000133",
          "june_update_2026-06-10__rec_000148",
          "june_update_2026-06-10__rec_000152",
          "june_update_2026-06-10__rec_000194",
          "june_update_2026-06-10__rec_000246",
          "june_update_2026-06-10__rec_000248",
          "june_update_2026-06-10__rec_000350"
        ],
        "model_count": 12,
        "route_count": 75
      }
    ],
    "modalities": [
      "DNA",
      "DNA methylation",
      "DNA methylation and dense embeddings",
      "DNA methylation-derived coefficients",
      "DNA sequence",
      "DNA sequence pairs",
      "RNA",
      "RNA expression profile",
      "RNA sequence",
      "RNA-seq",
      "batch metadata",
      "biological network knowledge",
      "biological sequence",
      "bright-field single-cell microscopy images",
      "bulk and multi-sample expression profiles",
      "categorical cellular metadata",
      "chemical perturbation and transcriptomics",
      "chemical perturbation transcriptomics",
      "chemically induced expression perturbations",
      "chest X-ray image",
      "chromatin accessibility",
      "clinical biomarker tabular data",
      "clinical blood biomarkers",
      "condition metadata",
      "confocal fluorescence microscopy images",
      "continuous numerical vectors",
      "dense continuous embedding",
      "dense embeddings",
      "differential interference contrast microscopy images",
      "disease indication",
      "expression profiles",
      "functional text",
      "gene dependency profiles",
      "gene embeddings",
      "gene expression",
      "gene lists",
      "gene neighborhood",
      "gene ontology graph",
      "gene symbol conditioning",
      "gene symbol text",
      "genetic/phenotypic tabular data",
      "genomic DNA",
      "genomic sequence",
      "genomics sequencing data",
      "geometric protein structure",
      "geometric shape specification",
      "geometric symmetry specification",
      "graph/network",
      "histology image",
      "histology section",
      "histology/slide image",
      "histopathology image",
      "instruction-formatted biological text",
      "ligand embeddings",
      "mammography image",
      "mixed",
      "mixed transcriptomics",
      "molecule descriptors / prompt text",
      "morphological phenotype embeddings",
      "multi-omics",
      "multimodal",
      "natural language",
      "natural language / protein mutation text",
      "natural language / protein question-answer text",
      "natural language / protein task text",
      "natural language / structural task text",
      "natural-language query",
      "nucleotide sequence",
      "omics",
      "oncology survival tabular data",
      "pathology image",
      "phase-contrast microscopy images",
      "protein",
      "protein knowledgebase",
      "protein sequence",
      "protein sequence pair",
      "protein sequence pairs",
      "protein sequence text",
      "protein structure",
      "protein structure constraints",
      "protein structure semantics",
      "protein structure symbols",
      "protein-protein interaction network",
      "protein/peptide",
      "proteomics",
      "quantitative phase imaging",
      "radiology image",
      "single-cell RNA sequencing",
      "single-cell RNA sequencing batch with donor metadata",
      "single-cell RNA sequencing cell with donor metadata",
      "single-cell RNA-seq",
      "single-cell and bulk transcriptomic profiles",
      "single-cell chemical perturbation data",
      "single-cell gene expression",
      "single-cell light microscopy images",
      "single-cell microscopy images",
      "single-cell perturbation data",
      "single-cell perturbation transcriptomics",
      "single-cell transcriptomic expression profiles",
      "single-cell transcriptomic perturbation profiles",
      "single-cell transcriptomics",
      "single-cell transcriptomics / prompt text",
      "skin lesion image",
      "small molecule",
      "source modality expression measures",
      "spatial transcriptomics",
      "structured biological database",
      "structured biological database record",
      "tabular",
      "text",
      "text-serialized nucleotide sequence plus natural-language annotation",
      "text-serialized nucleotide sequence plus species label",
      "transcriptomic signature",
      "transcriptomics",
      "transcriptomics and small-molecule compound",
      "visual raster",
      "widefield fluorescence microscopy images"
    ],
    "lifecycle_phases": [
      "evaluation",
      "fine_tuning",
      "inference",
      "pretraining",
      "unclear"
    ],
    "fusion_topologies": [
      "concatenation",
      "cross_attention",
      "encoder_decoder",
      "interleaving",
      "other_explicit",
      "placeholder_replacement",
      "prefix",
      "query_bottleneck",
      "retrieval_or_tool_context",
      "shared_latent_alignment",
      "side_or_generative_conditioning",
      "tokenizer_sequence",
      "unclear"
    ],
    "text_roles": [
      "biological_payload",
      "generated_output",
      "instruction_or_query",
      "metadata_or_context",
      "modality_or_task_selector",
      "no_text_on_this_route",
      "paired_alignment_supervision",
      "semantic_annotation"
    ]
  }
}
