id: https://ai4curation.io/ai-gene-review
name: gene_review
description: Schema for gene curation
  Top level entity is a GeneReview, which is about a single gene (and its equivalent swiss-prot entry).
  It contains a high level summary of the gene, plus a review of all existing annotations.
  It also contains a list of core functions, which are GO-CAM-like annotons describing the core evolved
  functions of the gene.
  

prefixes:
  linkml: https://w3id.org/linkml/
  gene_review: https://w3id.org/ai4curation/gene_review/
  xsd: http://www.w3.org/2001/XMLSchema#
  rdf: http://www.w3.org/1999/02/22-rdf-syntax-ns#
  rdfs: http://www.w3.org/2000/01/rdf-schema#
  owl: http://www.w3.org/2002/07/owl#
  oa: http://www.w3.org/ns/oa#
  skos: http://www.w3.org/2004/02/skos/core#
  dcat: http://www.w3.org/ns/dcat#
  dcterms: http://purl.org/dc/terms/
  ECO: http://purl.obolibrary.org/obo/ECO_
  IAO: http://purl.obolibrary.org/obo/IAO_
  SO: http://purl.obolibrary.org/obo/SO_

imports:
  - linkml:types

default_prefix: gene_review
default_range: string

slots:
  id:
    identifier: true
  label:
    description: Human readable name of the entity
    required: true
    slot_uri: rdfs:label
  gene_symbol:
    range: string
    description: Symbol of the gene
    required: true
  product_type:
    range: ProductTypeEnum
    description: Type of gene product (protein, ncRNA, etc.)
    required: false  ## TODO: change to required now we have added this everywhere
    comments:
      - currently not required, assumed PROTEIN by default, but this may be explicit in future
  title:
    range: string
    description: Title of the entity
    slot_uri: dcterms:title
    required: true
  aliases:
    range: string
    multivalued: true
  name:
    range: string
    description: Name of the entity (e.g., isoform name like Bcl-xL)
  sequence_note:
    range: string
    description: Brief note about sequence characteristics or differences
  tags:
    range: string
    multivalued: true
    description: Tags associated with the gene for categorization and organization
  description:
    description: Description of the entity
    slot_uri: dcterms:description
  statement:
    range: string
    description: Concise statement describing an aspect of the gene
    recommended: true ## TODO: change to required?
  references:
    range: Reference
    multivalued: true
    inlined_as_list: true
  findings:
    range: Finding
    multivalued: true
    recommended: true
  supported_by:
    range: SupportingTextInReference
    multivalued: true
    recommended: true
  summary:
    range: string
    description: Summary of the review
    recommended: true
  supporting_text:
    range: string
    description: Supporting text from the publication.
      This should be exact substrings. Different substrings can be broken up by '...'s. These substrings will be checked against the actual
      text of the paper. If editorialization is necessary, put this in square brackets (this is not checked). For example, you can say
      '...[CFAP300 shows] transport within cilia is IFT dependent...'
    implements:
      - oa:exact
    recommended: true
  supporting_text_fulltext:
    range: string
    description: Supporting text from the full-text PDF when the full text cannot be committed to the repository.
      This is an interim solution for cases where we have access to full text but cannot share it publicly.
      Unlike supporting_text, this field is not validated against cached publication text.
  full_text_unavailable:
    range: boolean
    description: Whether the full text is unavailable
  evidence_type:
    range: EvidenceType
    description: Evidence code (e.g., IDA, IBA, ISS, TAS)
    required: true
  term:
    inlined: true
    range: Term
    description: Term to be annotated
  predicate:
    range: string
    description: Predicate of the extension
    slot_uri: rdf:predicate
    required: true
  taxon:
    range: Term
    inlined: true
    required: true
  alternative_products:
    range: AlternativeProduct
    multivalued: true
    inlined: true
    inlined_as_list: true
    description: >-
      Alternative splicing products (isoforms) of the gene. Seeded from UniProt
      ALTERNATIVE PRODUCTS section. Only populated if there are multiple isoforms.
      Use this to document isoform-specific functions and biology.
      DEPRECATED: Use functional_isoforms instead for curated functional classes.
  functional_isoforms:
    range: FunctionalIsoform
    multivalued: true
    inlined: true
    inlined_as_list: true
    description: >-
      Curated functional isoform classes for the gene. Unlike alternative_products
      (which is seeded from UniProt), this field is purely curator/agent-defined
      to capture FUNCTIONALLY RELEVANT distinctions. Examples:
      - Splice classes that group multiple UniProt isoforms (e.g., WT1 +KTS vs -KTS)
      - Cleavage products from polyproteins (e.g., POMC peptides)
      - Modification states with distinct functions
      Only populate when there ARE functionally distinct forms worth documenting.
  existing_annotations:
    range: ExistingAnnotation
    multivalued: true
  core_functions:
    range: CoreFunction
    multivalued: true
    # TODO: add a rule such that this is required IF the review status is READY
  action:
    range: ActionEnum
    description: Action to be taken
    required: true
  reason:
    range: string
    description: Reason for the action
    recommended: true
  proposed_replacement_terms:
    range: Term
    description: If the action is MODIFY, then this is a list of proposed replacement terms
    multivalued: true
    inlined_as_list: true
    comments:
      - note there is a separate rule that this is required IF the action is MODIFY
  extensions:
    range: AnnotationExtension
    multivalued: true
  qualifier:
    range: AnnotationQualifierEnum
    description: >-
      The GO annotation qualifier specifying the relationship between the gene product
      and the term. For MF annotations, distinguishes 'enables' (gene product has the
      activity independently) from 'contributes_to' (gene product contributes to a
      complex's activity but does not have the activity alone). For BP, distinguishes
      'involved_in', 'acts_upstream_of', etc. For CC, distinguishes 'located_in',
      'part_of', 'is_active_in', 'colocalizes_with'.
  negated:
    range: boolean
    description: Whether the term is negated
  reference_id:
    range: Reference
    inlined: false
    implements:
      - dcterms:references
    required: true
  original_reference_id:
    range: Reference
    inlined: false
    description: ID of the original source reference, preserved as supplied by GOA/UniProt.
      Curate identifier replacements in references.reference_review.replacement and cite
      the actual evidence source in review.supported_by; do not rewrite this provenance.
  retired:
    range: boolean
    description: Whether the annotation is retired or replaced
  isoform:
    range: string
    description: >-
      UniProt isoform identifier (e.g., "P19544-1" for WT1 isoform 1).
      Only populated when the annotation is specific to a particular isoform
      rather than the canonical protein sequence. Note that just because an
      experiment used a particular isoform doesn't mean the annotation is
      isoform-specific - it may apply to all isoforms. Use this field only
      when there is clear evidence the annotation is isoform-specific.
  review:
    recommended: true
    range: Review
    description: Review of the gene
  reference_review:
    range: ReferenceReview
    description: Manual reviewer assessment of this reference (relevance, and citation
      correctness / scientific soundness). Reviewer-supplied, distinct from the
      machine-fetched id/title.
  replacement:
    range: ReferenceReplacement
    inlined: true
    description: Manually established replacement for this reference identifier. This records
      identifier provenance, not a biological contradiction or retraction, and does not
      implicitly redirect supporting_text to another publication.
  relevance:
    range: ReferenceRelevanceEnum
    description: Reviewer judgment of how relevant the reference is to the gene's function
      and this review.
  correctness:
    range: ReferenceCorrectnessEnum
    description: Reviewer's overall assessment of a reference's trustworthiness - both citation
      correctness (the identifier resolves to the intended paper that supports its use) and
      scientific soundness of that paper's claim.
  review_notes:
    range: string
    description: Free-text note explaining the relevance/correctness judgment (e.g. what was
      verified, or why a citation is wrong, disputed, or low quality).
  finding_review:
    range: FindingReview
    description: Manual reviewer assessment of this specific finding - in particular whether the
      finding remains current, is disputed, or has been overturned/superseded by later evidence.
      Reviewer-supplied; distinct from the statement/supporting_text that describe the finding itself.
  finding_status:
    range: FindingReviewStatusEnum
    description: Reviewer's assessment of the empirical standing of a specific finding in light of
      other evidence (e.g. whether it has been disputed or overturned).
  superseded_by:
    range: Reference
    multivalued: true
    inlined: false
    description: Reference(s) that dispute, correct, or overturn this finding. Used together with
      finding_status DISPUTED or OVERTURNED.
  ontology:
    range: string
    description: Ontology of the term. E.g `go`, `cl`, `hp`
  supporting_entities:
    range: string
    multivalued: true
    description: IDs of the supporting entities
  additional_reference_ids:
    range: Reference
    multivalued: true
    inlined: false
    description: IDs of the references
  is_invalid:
    range: boolean
    description: Whether the reference is invalid (e.g., retracted or replaced)
  publication_type:
    range: PublicationTypeEnum
    description: >-
      The kind of publication or source this reference is (e.g. primary research article,
      review, meta-analysis, database record, AI deep-research report). For PMIDs this is
      normally inferred from the PubMed publication-type ('PT') metadata rather than set by
      hand; for non-PMID references (GO_REF, Reactome, file:) it is inferred from the
      identifier. Lets analyses ask, e.g., whether review articles or abstracts alone are
      sufficient to support a given annotation action.
  reference_section_type:
    range: ManuscriptSection
    description: Type of section in the reference (e.g., 'ABSTRACT', 'METHODS', 'RESULTS', 'DISCUSSION')
    recommended: true
  proposed_new_terms:
    range: ProposedOntologyTerm
    multivalued: true
    description: Proposed new ontology terms that should exist but don't
  suggested_questions:
    range: Question
    multivalued: true
    description: Suggested questions to ask experts about the gene. Only include if not obvious from the literature.
    recommended: true
  suggested_experiments:
    range: Experiment
    multivalued: true
    recommended: true
  knowledge_gaps:
    range: KnowledgeGap
    multivalued: true
    inlined_as_list: true
    description: >-
      Curated, literature-grounded statements of what is NOT known — applicable at
      the level of the whole gene, a single existing annotation, a core function, a
      whole module, or a single module step/node. The inverse of core_functions:
      everywhere else the schema records what IS known; here it records, with the
      same evidentiary discipline, what is not. See the Function Knowledge Gaps
      project (projects/FUNCTION_KNOWLEDGE_GAPS.md).
  propagation_review:
    range: PropagationReview
    description: >-
      Mechanical review metadata for annotations whose evidence depends on
      propagation or inference from source genes, family nodes, orthogroups, or
      other non-target evidence. Use this to classify source-side vs
      propagation-side failure modes without duplicating the prose rationale in
      review.reason.
  status:
    range: GeneReviewStatusEnum
    description: Overall status of the gene review
    recommended: true


classes:

  GeneReview:
    description: Complete review for a gene
    tree_root: true
    slots:
     - id
     - gene_symbol
     - product_type
     - aliases
     - tags
     - status
     - description
     - taxon
     - alternative_products
     - functional_isoforms
     - references
     - existing_annotations
     - core_functions
     - proposed_new_terms
     - suggested_questions
     - suggested_experiments
     - knowledge_gaps
    slot_usage:
      description:
        recommended: true


  AlternativeProduct:
    description: >-
      An alternative splicing product (isoform) of the gene. Corresponds to UniProt
      isoform entries. Use this to document isoform-specific functions where different
      isoforms have distinct or even antagonistic biological activities.
      DEPRECATED: Use FunctionalIsoform instead for curated functional classes.
    slots:
      - id
      - name
      - sequence_note
      - description
    slot_usage:
      id:
        description: UniProt isoform ID (e.g., Q07817-1, Q07817-2)
        required: true
      name:
        description: Common name of the isoform (e.g., Bcl-xL, Bcl-xS)
        required: false
      sequence_note:
        description: Brief note about sequence differences (e.g., "lacks exon 2", "shorter C-terminus")
        required: false
      description:
        description: >-
          Agent-populated description of the isoform's function. Document any
          isoform-specific functions, expression patterns, or biological activities
          that differ from other isoforms.
        recommended: true

  FunctionalIsoform:
    description: >-
      A curated functional isoform class. Unlike AlternativeProduct (which maps 1:1
      to UniProt isoforms), this captures FUNCTIONALLY RELEVANT distinctions that may:
      - Group multiple UniProt isoforms into a functional class (e.g., WT1 +KTS isoforms)
      - Represent cleavage products from polyproteins (e.g., POMC peptides)
      - Describe modification states or conformational variants
      Only create entries when there ARE functionally distinct forms worth documenting.
    attributes:
      id:
        range: string
        required: true
        identifier: true
        description: >-
          Curator-defined identifier for this functional class. Use a descriptive
          format like GENE_CLASS (e.g., WT1_PLUS_KTS, POMC_ACTH, BCL2L1_XL).
      name:
        range: string
        required: true
        description: >-
          Human-readable name for this functional class (e.g., "+KTS isoforms",
          "ACTH/Corticotropin", "Bcl-xL").
      type:
        range: FunctionalIsoformTypeEnum
        required: true
        description: >-
          Type of functional distinction (SPLICE_VARIANT, SPLICE_CLASS, CLEAVAGE_PRODUCT,
          MODIFICATION_STATE, CONFORMATIONAL_STATE).
      maps_to:
        range: FunctionalIsoformMapping
        multivalued: true
        inlined_as_list: true
        description: >-
          Mappings to underlying UniProt identifiers. Optional - some functional classes
          may not map cleanly to specific UniProt IDs.
      description:
        range: string
        required: true
        description: >-
          Detailed description of this functional class. Document the specific functions,
          how they differ from other classes, tissue specificity, and any antagonistic
          relationships (e.g., "OREXIGENIC - opposite to alpha-MSH").
      isoform_specific_terms:
        range: Term
        multivalued: true
        inlined_as_list: true
        description: >-
          GO terms that are specific to this functional class. These are terms
          that should NOT be annotated to the gene as a whole, only to this specific
          form. Using Term objects enables linkml-term-validator checking.

  FunctionalIsoformMapping:
    description: >-
      A mapping from a functional isoform class to underlying UniProt identifiers.
      Allows grouping multiple UniProt isoforms or chains into a single functional class.
    attributes:
      type:
        range: FunctionalIsoformMappingTypeEnum
        required: true
        description: Type of identifier (UNIPROT_ISOFORM or UNIPROT_CHAIN)
      ids:
        range: string
        multivalued: true
        required: true
        description: >-
          UniProt identifiers belonging to this functional class.
          For UNIPROT_ISOFORM: P19544-1, P19544-2, etc.
          For UNIPROT_CHAIN: PRO_0000024969, PRO_0000024970, etc.
      residues:
        range: string
        description: >-
          Residue range for cleavage products (e.g., "138-176" for ACTH).
          Only applicable for UNIPROT_CHAIN type.

  Term:
    description: A term in a specific ontology
    slots:
      - id
      - label
      - description
      - ontology
    slot_usage:
      id:
        description: A CURIE for a term or database object in GO, CL, CHEBI, UniProtKB, PANTHER, etc.
      label:
        description: the term name
        required: true

  Reference:
    description: A reference is a published text  that describes a finding or a method. References
      might be formal publications (where the ID is a PMID), or for methods, a GO_REF. Additionally,
      a reference to a local ad-hoc analysis or review can be made by using the `file:` prefix.
    slots:
      - id
      - title
      - findings
      - is_invalid
      - publication_type
      - full_text_unavailable
      - reference_review
    slot_usage:
      id:
        implements:
          - dcterms:references

  ReferenceReview:
    description: Manual reviewer assessment of a reference - how relevant it is to the gene's
      function, and whether it is correctly cited and scientifically sound. Distinct from the
      machine-fetched id/title fields; all fields are optional and reviewer-supplied.
    slots:
      - relevance
      - correctness
      - review_notes
      - replacement

  ReferenceReplacement:
    description: A curated identifier remapping. Keep the original Reference.id and
      original_reference_id; declare the target in references and explicitly cite it in
      supported_by when quoting its text. Record how the mapping was established in
      review_notes. A duplicate-record deletion is not a retraction of the paper.
    slots:
      - reference_id
      - review_notes
    attributes:
      reason:
        range: ReferenceReplacementReasonEnum
        required: true
        description: Why the source identifier should resolve to the target reference.

  Finding:
    description: A finding is a statement about a gene, which is supported by a reference.
      Similar to "comments" in uniprot
    slots:
      - statement
      - supporting_text
      - full_text_unavailable
      - reference_section_type
      - finding_review

  FindingReview:
    description: Manual reviewer assessment of a specific finding within a reference - in
      particular whether it remains current, is disputed, or has been overturned/superseded by
      later evidence. This is finer-grained than reference_review (which assesses the whole
      reference); a paper may contain some findings that stand and others that are overturned.
      All fields optional and reviewer-supplied.
    slots:
      - finding_status
      - superseded_by
      - review_notes
      - supported_by
    slot_usage:
      supported_by:
        description: Evidence for this assessment, including exact snippets from the papers
          that contradict, overturn, or corroborate the original finding. Each snippet is
          attributed to its explicit reference_id, not the parent finding's publication.

  SupportingTextInReference:
    description: A supporting text in a reference.
    slots:
      - reference_id
      - supporting_text
      - supporting_text_fulltext
      - full_text_unavailable
      - reference_section_type

  # ============== Module KB Classes ==============
  #
  # Modules are reusable, recursively decomposable biological sketches. They
  # intentionally do not try to be GO-CAMs or exports of gene reviews. They
  # provide a place to bag together functions, processes, locations, variants,
  # and causal/temporal connections with lightweight provenance.

  EvidenceItem:
    description: >-
      A lightweight citable source for module-level assertions. The source may
      be a PMID, DOI, database record, local file, pathway record, issue, or any
      other citable artifact. When a literature source_id (PMID/DOI) is paired
      with a supporting_text, that quote is validated verbatim (normalized
      substring) against the cached publication by the project's module
      supporting-text check (ai_gene_review.validation.module_validator), using
      the same matcher as gene reviews; non-literature source_ids (GO, file:,
      Reactome, PANTHER, ...) carry no supporting_text quote and are not fetched.
    attributes:
      source_id:
        range: string
        required: true
        description: Identifier for the evidence source, e.g. PMID:123456, DOI:..., Reactome:R-HSA-..., MetaCyc:..., file:...
      title:
        range: string
      statement:
        range: string
        description: The assertion this evidence supports in this module.
      supporting_text:
        range: string
        description: Optional quote or excerpt from the evidence source.
      url:
        range: string
      notes:
        range: string

  Descriptor:
    description: >-
      A human-friendly descriptor with optional ontology/database grounding.
      The preferred_term may be more nuanced than the term label, and the term
      may be absent when no good identifier exists yet.
    attributes:
      preferred_term:
        range: string
        required: true
      description:
        range: string
      term:
        range: Term
        inlined: true
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  ChemicalEntityDescriptor:
    is_a: Descriptor
    description: A descriptor for a chemical entity, metabolite, cofactor, ion, or small molecule.

  GeneDescriptor:
    is_a: Descriptor
    description: A descriptor for a gene or locus.

  GeneProductDescriptor:
    is_a: Descriptor
    description: A descriptor for a gene product, protein, isoform, or gene-product form.

  FamilyDescriptor:
    is_a: Descriptor
    description: A descriptor for a protein family, orthogroup, or other evolutionary grouping.
    attributes:
      family_terms:
        range: Term
        multivalued: true
        inlined_as_list: true
        description: >-
          Multiple family-level ontology/database groundings when this descriptor
          intentionally abstracts over more than one family, such as a conserved
          role split across multiple PANTHER PTHR families. Use the inherited
          term slot for the ordinary single-family case.
      representative_members:
        range: GeneProductDescriptor
        multivalued: true
        inlined_as_list: true
        description: >-
          Representative concrete members used to orient the family. These are
          examples, not an exhaustive member list and not a claim that the
          module is limited to these proteins.
      ancestral_nodes:
        range: AncestralNodeDescriptor
        multivalued: true
        inlined_as_list: true
        description: >-
          PANTHER/PAINT ancestral node(s) at which the associated function is
          inferred to have arisen (or to have been present in the last common
          ancestor). Unlike representative_members, which only give orienting
          examples, an ancestral node makes a clade-level evolutionary claim:
          extant descendants are inferred to retain the function unless there is
          evidence of divergence, neofunctionalization, or loss of key residues.

  AncestralNodeDescriptor:
    is_a: Descriptor
    description: >-
      A PANTHER/PAINT ancestral node (a PTN identifier, e.g.
      PANTHER:PTN000299444) used to ground an evolutionary inference about a
      function. Asserting an ancestral node states that the function associated
      with the enclosing annoton is inferred to have arisen at, or been present
      in, the last common ancestor represented by this node, and is therefore
      inferred to be retained in extant descendant proteins barring divergence,
      neofunctionalization, or loss of key residues. This is a stronger,
      clade-level claim than a representative member. Such nodes can be resolved
      from the IBA WITH/FROM column (GO_REF:0000033) of a representative
      member's GOA record rather than guessed.

  DomainDescriptor:
    is_a: Descriptor
    description: A descriptor for a protein domain, motif, site, or architectural feature.

  CellularComponentDescriptor:
    is_a: Descriptor
    description: A descriptor for a cellular component, organelle, compartment, or complex location.

  ProteinComplexDescriptor:
    is_a: CellularComponentDescriptor
    description: A descriptor for a protein-containing complex or subcomplex.
    attributes:
      active_units:
        range: ComplexUnit
        multivalued: true
        inlined_as_list: true
        description: >-
          Active units or role-bearing components of the complex. This is used
          to avoid leaving functionally important complex structure as prose.

  ComplexUnit:
    description: A role-bearing unit within a protein complex descriptor.
    attributes:
      id:
        range: string
      label:
        range: string
      participant:
        range: ParticipantSelector
        inlined: true
      role:
        range: string
      stoichiometry:
        range: string
        description: Optional stoichiometry or copy-number statement when known.
      function:
        range: MolecularFunctionDescriptor
        inlined: true
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  CellTypeDescriptor:
    is_a: Descriptor
    description: A descriptor for a cell type or cell state.

  AnatomicalEntityDescriptor:
    is_a: Descriptor
    description: A descriptor for an anatomical entity, tissue, organismal region, or structure.

  DevelopmentalStageDescriptor:
    is_a: Descriptor
    description: A descriptor for a developmental stage, life-cycle stage, or temporal window.

  TaxonDescriptor:
    is_a: Descriptor
    description: A descriptor for a taxon or taxonomic scope.

  MolecularFunctionDescriptor:
    is_a: Descriptor
    description: >-
      A descriptor for a molecular function. Extra slots capture common
      functional nuance without requiring formal post-composition.
    attributes:
      substrates:
        range: Descriptor
        multivalued: true
        inlined_as_list: true
      products:
        range: Descriptor
        multivalued: true
        inlined_as_list: true
      cofactors:
        range: Descriptor
        multivalued: true
        inlined_as_list: true
      targets:
        range: Descriptor
        multivalued: true
        inlined_as_list: true
      cargo:
        range: Descriptor
        multivalued: true
        inlined_as_list: true
      source_location:
        range: CellularComponentDescriptor
        inlined: true
      destination_location:
        range: CellularComponentDescriptor
        inlined: true

  BiologicalProcessDescriptor:
    is_a: Descriptor
    description: >-
      A descriptor for a biological process, pathway, reaction, or process-like
      module grounding.
    attributes:
      inputs:
        range: Descriptor
        multivalued: true
        inlined_as_list: true
      outputs:
        range: Descriptor
        multivalued: true
        inlined_as_list: true
      occurs_in:
        range: Descriptor
        multivalued: true
        inlined_as_list: true
      starts_with:
        range: Descriptor
        inlined: true
      ends_with:
        range: Descriptor
        inlined: true

  RelationDescriptor:
    is_a: Descriptor
    description: A descriptor for a relation or connection predicate.

  ModuleReview:
    description: >-
      Review or curation record for a recursively decomposable biological
      module. This can describe a pathway, organelle lifecycle, protein complex,
      molecular function, developmental process, or abstract/evolutionary
      functional plan.
    tree_root: true
    slots:
      - id
      - title
      - description
      - references
      - knowledge_gaps
    attributes:
      status:
        range: string
      scope:
        range: ModuleScopeEnum
        description: >-
          Whether this module is a concrete biological realization or an
          abstract reusable motif/template. ABSTRACT modules are intentionally
          gene-free and are not expected to declare representative protein
          members for every leaf node.
      module:
        range: ModuleNode
        inlined: true
        required: true
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  ModuleNode:
    description: >-
      A node in a module. Nodes can be recursively decomposed using parts and
      variant_sets, and may also carry leaf annotons and connections.
    slots:
      - knowledge_gaps
    attributes:
      id:
        range: string
        required: true
        identifier: true
      label:
        range: string
        required: true
      module_type:
        range: ModuleTypeEnum
      description:
        range: string
      concepts:
        range: Descriptor
        multivalued: true
        inlined_as_list: true
        description: Optional ontology/database grounding for this module node.
      context:
        range: ModuleContext
        inlined: true
      annotons:
        range: ModuleAnnoton
        multivalued: true
        inlined_as_list: true
      parts:
        range: ModulePart
        multivalued: true
        inlined_as_list: true
      variant_sets:
        range: ModuleVariantSet
        multivalued: true
        inlined_as_list: true
      connections:
        range: ModuleConnection
        multivalued: true
        inlined_as_list: true
      conforms_to:
        range: Conformance
        multivalued: true
        inlined_as_list: true
        description: >-
          Reusable template motifs that this node (together with its parts and
          connections) is an instance of. Conformance is a compositional,
          bundle-scoped consistency check: a concrete cascade may freely extend
          its start and end, while an inner sub-bundle node declares that its
          parts match a generic motif (e.g. the three-tier MAP kinase relay).
      gocam_associations:
        range: GoCamAssociation
        multivalued: true
        inlined_as_list: true
        description: >-
          References to production GO-CAM models (or specific activities) that
          realize this module node as a whole.
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  Conformance:
    description: >-
      Assertion that a module node (together with its parts and connections) is
      an instance of a reusable template module or motif, optionally recording
      how it deviates from that template. Modeled after the dismech
      conforms_to pattern: conformance is a consistency check, not inheritance.
    attributes:
      template:
        range: string
        required: true
        description: >-
          Reference to the template, as a module path relative to modules/ with
          an optional node id after a hash (e.g. "mapk_relay" or
          "mapk_relay#map2k"). The referenced template defines the required
          steps, function terms, and connection topology this node must contain.
      status:
        range: ConformanceStatusEnum
        description: >-
          Whether the node matches the template exactly, matches with the noted
          deviations, or matches the core motif while extending it.
      deviations:
        range: string
        multivalued: true
        description: >-
          Specific differences from the template (e.g. a missing or merged tier,
          a substituted function term). Listed deviations are treated as
          informational rather than errors during conformance QC.
      notes:
        range: string
        description: Free-text rationale or context for the conformance and any deviations.

  ModulePart:
    description: A conjunctive part or step of a module node.
    attributes:
      order:
        range: integer
        description: Optional display or temporal order. Equal or absent values imply partial ordering only.
      role:
        range: string
        description: Curator-supplied role of this part within the parent module.
      optional:
        range: boolean
        description: Whether this part is optional in the parent module.
      node:
        range: ModuleNode
        inlined: true
        required: true
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  ModuleVariantSet:
    description: >-
      A set of alternative implementations for a module node or part. Variants
      may themselves contain parts, annotons, connections, and nested variant sets.
    attributes:
      id:
        range: string
        required: true
      label:
        range: string
      axis:
        range: string
        description: The dimension along which these variants differ, e.g. taxon, cell type, compartment, route, enzyme family.
      selection:
        range: VariantSelectionEnum
        description: How many variants may be selected in a concrete realization.
      variants:
        range: ModuleNode
        multivalued: true
        inlined_as_list: true
        required: true
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  ModuleAnnoton:
    description: >-
      A leaf role assertion in a module. The participant may be a concrete gene
      or an abstract selector, and the function/process/location fields are
      descriptor holders rather than direct GO annotation exports.
    attributes:
      id:
        range: string
        required: true
      label:
        range: string
      participant:
        range: ParticipantSelector
        inlined: true
      function:
        range: MolecularFunctionDescriptor
        inlined: true
      processes:
        range: BiologicalProcessDescriptor
        multivalued: true
        inlined_as_list: true
      locations:
        range: CellularComponentDescriptor
        multivalued: true
        inlined_as_list: true
      role_description:
        range: string
      gocam_associations:
        range: GoCamAssociation
        multivalued: true
        inlined_as_list: true
        description: >-
          References to production GO-CAM model activities (annotons) that
          realize this module annoton. Used to ground an abstract/non-grounded
          module role in concrete curated causal activity models.
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  GoCamAssociation:
    description: >-
      A reference from a module element to a production GO-CAM (Gene Ontology
      Causal Activity Model), optionally pinned to a specific activity (annoton)
      within that model. The referenced model is expected to be cached under
      gocams/<model_id>/<model_id>-src.yaml.
    attributes:
      model:
        range: string
        required: true
        description: >-
          GO-CAM model id, e.g. gomodel:568b0f9600000284 (or the bare local id
          568b0f9600000284). Matches the cached gocams/<model_id>/ folder.
      activity:
        range: string
        description: >-
          Optional activity/annoton id within the model that this element
          corresponds to, e.g. gomodel:568b0f9600000284/57ec3a7e00000079.
      title:
        range: string
        description: Cached model title, recorded for human readability.
      description:
        range: string
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  GoCamReview:
    description: >-
      A reviewer's assessment of a cached production GO-CAM model
      (gocams/<model_id>/<model_id>-src.yaml), recorded alongside it as
      gocams/<model_id>/<model_id>-review.yaml. Captures a standalone reading of
      the model, per-activity (annoton) QC against GO-CAM best practice, and the
      consistency of each activity with the corresponding gene annotation review.
      Validate standalone with `-C GoCamReview`.
    tree_root: true
    slots:
      - title
      - description
      - references
    attributes:
      model:
        range: string
        required: true
        identifier: true
        description: >-
          GO-CAM model id (gomodel:... or the bare local id) matching the cached
          gocams/<model_id>/ folder.
      taxon:
        range: string
        description: Primary taxon of the model, e.g. NCBITaxon:6239.
      summary:
        range: string
        description: >-
          Reviewer's standalone reading of what the model asserts (the causal
          story), independent of the curation project.
      status:
        range: GoCamReviewStatusEnum
      activity_reviews:
        range: GoCamActivityReview
        multivalued: true
        inlined_as_list: true
        description: Per-activity (annoton) reviews.
      notes:
        range: string

  GoCamActivityReview:
    description: >-
      Review of a single GO-CAM activity (annoton): the cached gene
      product / molecular function / process / location, a best-practice QC
      verdict, and how the activity relates to the gene's annotation review.
    attributes:
      activity_id:
        range: string
        required: true
        description: >-
          Activity individual id within the model, e.g.
          gomodel:568b0f9600000284/57ec3a7e00000079.
      gene_product:
        range: string
        description: enabled_by gene product id as cached (e.g. UniProtKB:..., WB:...).
      molecular_function:
        range: string
        description: Molecular function GO id of the activity as cached.
      biological_process:
        range: string
        description: part_of biological process GO id as cached, if any.
      cellular_component:
        range: string
        description: occurs_in cellular component GO id as cached, if any.
      verdict:
        range: GoCamClaimVerdictEnum
        description: >-
          Overall best-practice/evidence verdict for this activity (mirrors the
          OK / UNCERTAIN / WRONG forensic-review scale).
      qc_flags:
        range: GoCamQcFlagEnum
        multivalued: true
        description: Specific GO-CAM best-practice issues observed for this activity.
      consistency:
        range: GoCamConsistencyEnum
        description: >-
          How this activity relates to the gene's annotation review
          (genes/**/<gene>-ai-review.yaml).
      gene_review:
        range: string
        description: >-
          Reference to the gene review compared against, e.g.
          file:human/TP53/TP53-ai-review.yaml.
      supporting_text:
        range: string
        description: Verbatim supporting text from a cited reference, where applicable.
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  ParticipantSelector:
    description: >-
      A selector for a concrete or abstract participant in a module annoton.
      This can ground to a gene, gene product, complex, family, domain, ortholog,
      homolog, or any entity satisfying a functional/domain constraint.
    attributes:
      selector_type:
        range: ParticipantSelectorTypeEnum
        required: true
      gene:
        range: GeneDescriptor
        inlined: true
      gene_product:
        range: GeneProductDescriptor
        inlined: true
      protein_complex:
        range: ProteinComplexDescriptor
        inlined: true
      family:
        range: FamilyDescriptor
        inlined: true
      domain:
        range: DomainDescriptor
        inlined: true
      homolog_of:
        range: GeneDescriptor
        inlined: true
      ortholog_of:
        range: GeneDescriptor
        inlined: true
      required_function:
        range: MolecularFunctionDescriptor
        inlined: true
      required_domain:
        range: DomainDescriptor
        inlined: true
      taxon:
        range: TaxonDescriptor
        inlined: true
      description:
        range: string
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  ModuleContext:
    description: Context that applies to a module node, variant, annoton, or connection.
    attributes:
      taxa:
        range: TaxonDescriptor
        multivalued: true
        inlined_as_list: true
      cell_types:
        range: CellTypeDescriptor
        multivalued: true
        inlined_as_list: true
      anatomical_locations:
        range: AnatomicalEntityDescriptor
        multivalued: true
        inlined_as_list: true
      developmental_stages:
        range: DevelopmentalStageDescriptor
        multivalued: true
        inlined_as_list: true
      cellular_components:
        range: CellularComponentDescriptor
        multivalued: true
        inlined_as_list: true
      conditions:
        range: Descriptor
        multivalued: true
        inlined_as_list: true
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      notes:
        range: string

  ModuleConnection:
    description: >-
      A connection between module nodes, annotons, or other named elements.
      Source and target are IDs scoped to the module document.
    attributes:
      source:
        range: string
        required: true
      target:
        range: string
        required: true
      connection_type:
        range: ModuleConnectionTypeEnum
      predicate:
        range: RelationDescriptor
        inlined: true
      description:
        range: string
      context:
        range: ModuleContext
        inlined: true
      evidence:
        range: EvidenceItem
        multivalued: true
        inlined_as_list: true
      chaining_status:
        range: ChainingStatusEnum
        description: >-
          Curator adjudication of reaction continuity across this connection:
          whether the upstream reaction's product is the downstream reaction's
          substrate. This is an explicit override for the (advisory, non-blocking)
          automated chaining check, so a known gap can be acknowledged rather than
          re-reported each run. Leave unset to let the automated check report its
          finding.
      chaining_note:
        range: string
        description: >-
          Free-text explanation for the chaining_status, e.g. why a break is a
          genuine knowledge gap, or which GO/RHEA mapping is missing.
      notes:
        range: string


  ExistingAnnotation:
    description: An existing annotation from the GO database, plus a review of the annotation.
    slots:
      - term
      - qualifier
      - extensions
      - negated
      - evidence_type
      - original_reference_id
      - retired
      - isoform
      - supporting_entities
      - review
    # Note: we deliberately do NOT add bindings (GOTermEnum or label
    # validation) to term.id here.  ExistingAnnotation terms come from GOA
    # and may be out of sync with the current ontology — obsolete terms
    # (for REMOVE/MODIFY actions), synonym labels, or terms not yet in the
    # local OAK cache.  A GOTermEnum binding would reject these since it
    # uses reachable_from (current subClassOf descendants only).
    # As a consequence, linkml-term-validator does not check term IDs or
    # labels in existing_annotations.  Label correctness is instead
    # covered by the GOA comparison in check_best_practices_rules().
    # These IDs are machine-sourced from GOA (not author-supplied), so there is
    # no hallucination risk in leaving them unbound.  By contrast, the
    # author-supplied term IDs in core_functions ARE bound to dynamic enums
    # (GOMolecularActivityEnum / GOCellularLocationEnum /
    # GOProteinContainingComplexEnum, ...) and a wrong-branch term there is a
    # blocking validation error.

  Review:
    description: A review of an existing annotation.
    slots:
      - summary
      - action
      - reason
      - proposed_replacement_terms
      - additional_reference_ids
      - supported_by
      - knowledge_gaps
      - propagation_review
    rules:
      - preconditions:
          slot_conditions:
            action:
              equals_string: MODIFY
        postconditions:
          slot_conditions:
            proposed_replacement_terms:
              required: true

  PropagationReview:
    description: >-
      Structured, mechanical assessment of a propagated or inferred annotation.
      The detailed biological rationale remains in review.reason; this object
      records the reusable taxonomy and optional per-source-gene/node comments.
    attributes:
      root_cause:
        range: PropagationRootCauseEnum
        required: true
        description: >-
          Where the issue lies, or that no issue was found: source annotation,
          propagation decision, term scoping, circular/redundant evidence, or no
          propagation failure.
      failure_modes:
        range: PropagationFailureModeEnum
        multivalued: true
        description: >-
          Biological shape of the propagation issue. Multiple values are allowed
          when a case combines, for example, paralog transfer and functional
          divergence.
      source_entities:
        range: PropagationSource
        multivalued: true
        inlined_as_list: true
        description: >-
          Source genes, gene products, PANTHER nodes, family nodes, or other
          source entities inspected for the propagated annotation.
      residue_claims:
        range: ResidueClaim
        multivalued: true
        inlined_as_list: true
        description: >-
          Machine-checkable form of a residue-level argument, e.g. "this protein
          lost the catalytic cysteine". Optional and additive: the vast majority of
          existing reviews state such arguments only in review.reason prose, and
          those are grandfathered, never invalidated. Supply this whenever a NEW
          review rests on residue gain, loss, or retention, so a validator can
          resolve the claim against the actual sequences and contradict it.

          This exists because the argument is cheap to assert and hard to check. A
          survey of 17 "lacks the catalytic residue" claims in this repo found 4
          where the site was fully intact -- CASP12 (truncated instead), LPA
          (activation junction), AZIN1 (lost substrate contacts, kept the catalytic
          Cys) and HSPA13 (a different domain). Each conclusion happened to survive
          for another reason, so nothing caught the faulty mechanism.
      residue_claims_not_applicable:
        range: string
        description: >-
          Why this annotation's argument cannot be expressed as a residue claim, for
          cases that are genuinely not point-residue questions -- capping versus
          severing (CAPG), holdase versus foldase (CRYAA), or a whole domain absent
          rather than residues substituted (MEFV, RAD51C). Recording the reason keeps
          the escape valve honest rather than silent.

  ResidueClaim:
    description: >-
      A claim that a specific position in THIS gene's protein does or does not carry
      the residue an anchor protein has at the corresponding position.

      Both sides carry an explicit position and residue, which is what makes the claim
      checkable without an alignment: the anchor and target residues are each resolved
      directly against their own sequences. The alignment is only needed to confirm the
      two positions genuinely correspond, so a claim remains partially verifiable even
      when no alignment is pinned.

      Positions are always in each protein's own native numbering, never an alignment
      column -- columns shift when PANTHER re-releases a family, and a stale column
      still resolves, to the wrong residue.
    attributes:
      claim_type:
        range: ResidueClaimEnum
        required: true
        description: Whether the target lost, retained, or substituted the residue.
      site_ref:
        range: string
        pattern: "^PANTHER:PTHR[0-9]{5}#[a-z0-9_]+$"
        description: >-
          Optional reference to a curated family-level residue site, as
          ``<family_id>#<site_id>`` (e.g. PANTHER:PTHR11022#zn_triad). When given, the
          validator additionally checks that the site exists in that family review and
          that the anchor position is one the site declares -- which is what stops a
          gene review and a family review drifting apart.
      anchor:
        range: ResiduePosition
        inlined: true
        required: true
        description: >-
          The comparator: a protein known to have the functional residue, and where.
      target:
        range: ResiduePosition
        inlined: true
        description: >-
          The corresponding position in this gene's own protein. Omit only when the
          region is unalignable, in which case say so in comment; a missing target is
          weaker evidence than a stated substitution.
      role:
        range: string
        description: Mechanistic role of the anchor residue (e.g. "metal ligand", "nucleophile").
      method:
        range: ResidueClaimMethodEnum
        required: true
        description: How the correspondence between anchor and target positions was established.
      alignment_release:
        range: string
        description: >-
          Version of the alignment *resource* the correspondence was taken from, when
          one exists -- e.g. "PANTHER 19.0" or "Pfam 37.0". A tool name or a script
          path is not a release and does not belong here; put that in comment.

          Omit it for an alignment computed ad hoc. That is not a gap: the claim states
          both positions and both residues, so the alignment is only how the
          correspondence was *discovered*, while the evidence is the two sequence
          lookups the validator performs. Pinning matters for a shared resource whose
          coordinates can shift under you, not for a one-off computation.
      comment:
        range: string
        description: Short note, e.g. why the target position is absent rather than substituted.

  ResiduePosition:
    description: A single residue in a named protein, in that protein's own numbering.
    attributes:
      accession:
        range: string
        required: true
        description: UniProt CURIE, e.g. UniProtKB:Q96PD5.
      position:
        range: integer
        required: true
        minimum_value: 1
        description: 1-based position in this protein's own sequence.
      residue:
        range: string
        required: true
        pattern: "^[ACDEFGHIKLMNPQRSTVWYUO]$"
        description: >-
          Single-letter residue expected at that position. U and O are permitted for
          selenocysteine and pyrrolysine -- a catalytic position can legitimately be
          selenocysteine, as in SEPHS2.
      sequence_version: 
        range: integer
        minimum_value: 1
        description: >-
          UniProt sequence version (the SV in a P49903.2 style citation) the position
          was read against. Record it: an amino-acid sequence is not immutable. SEPHS2
          is already on sequence version 3, and a corrected or re-chosen canonical
          sequence shifts every downstream position, so an unversioned claim can go
          silently wrong -- or worse, keep passing against a different residue that
          happens to match.

          The validator compares this against the current record and, when they differ,
          says so explicitly rather than letting a stale claim look verified.

  PropagationSource:
    description: >-
      A source entity considered while reviewing an inferred annotation. This is
      intentionally compact: use comment for source-specific caveats, not for
      restating the whole review rationale.
    attributes:
      source_id:
        range: string
        required: true
        description: >-
          Identifier for the source entity, e.g. UniProtKB:P03950,
          MGI:MGI:104579, PANTHER:PTN002745520, or a GO_REF/source label when
          no gene product identifier is available.
      source_label:
        range: string
        description: Human-readable source label, such as a gene symbol or node label.
      source_status:
        range: PropagationSourceStatusEnum
        description: Mechanical status of this source with respect to the target annotation.
      comment:
        range: string
        description: >-
          Short source-specific comment, e.g. "human ANG supports angiogenesis,
          but mouse Ang2/Angrp is non-angiogenic" or "seed is inferred-only".
        
  CoreFunction:
    description: A core function is a GO-CAM-like annotation of the core evolved functions of a gene.
      This is a synthesis of the reviewed core annotations, brought together into a unified GO-CAM-like
      representation.
    slots:
      - knowledge_gaps
    attributes:
      description:
        range: string
        description: Description of the core function
      supported_by:
        range: SupportingTextInReference
        multivalued: true
      molecular_function:
        range: Term
        inlined: true
        description: >-
          The molecular function this gene product enables (i.e., has the activity
          independently). For complex subunits that contribute to but don't
          independently have a complex-level activity, use contributes_to_molecular_function
          instead and put a subunit-specific MF here (e.g., structural constituent of
          ribosome, electron transfer activity).
        bindings:
          - binds_value_of: id
            range: GOMolecularActivityEnum
            obligation_level: REQUIRED
      proposed_molecular_function:
        range: string
        description: >-
          The core activity when GO has no term for it yet: the proposed_name of an
          entry in this review's top-level proposed_new_terms. Use it instead of
          putting an obsolete or ill-fitting GO id in molecular_function (e.g. in-situ
          holdase chaperones after GO:0051082 was obsoleted with no replacement).
          When GO creates the term, move its id to molecular_function and drop this.
      contributes_to_molecular_function:
        range: Term
        inlined: true
        description: >-
          A molecular function that this gene product contributes to as part of a
          complex, but does not independently enable. Used for accessory/structural
          subunits of multi-protein complexes (e.g., an accessory subunit of Complex I
          contributes_to NADH dehydrogenase activity but does not have that activity
          on its own). The molecular_function slot should then contain the
          subunit-specific activity (e.g., structural molecule activity).
        bindings:
          - binds_value_of: id
            range: GOMolecularActivityEnum
            obligation_level: REQUIRED
      directly_involved_in:
        range: Term
        multivalued: true
        inlined_as_list: true
        bindings:
          - binds_value_of: id
            range: GOBiologicalProcessEnum
            obligation_level: REQUIRED
      locations :
        range: Term
        multivalued: true
        inlined_as_list: true
        description: >-
          Cellular anatomical entities (e.g. membranes, nucleus, cytosol, organelle
          parts) where the gene product functions. Do NOT use this for protein-containing
          complexes (GO:0032991 and its descendants) — record complex membership in
          in_complex instead.
        bindings:
          - binds_value_of: id
            range: GOCellularLocationEnum
            obligation_level: REQUIRED
      anatomical_locations:
        range: Term
        multivalued: true
        inlined_as_list: true
      substrates: 
        range: Term
        multivalued: true
        inlined_as_list: true
      in_complex:
        range: Term
        inlined: true
        description: >-
          The protein-containing complex (GO:0032991 descendant) that this gene product is
          an active unit of. Use this — not locations — for complex membership (e.g.
          ribosome, spliceosome, EMC, signal peptidase complex).
        bindings:
          - binds_value_of: id
            range: GOProteinContainingComplexEnum
            obligation_level: REQUIRED
    slot_usage:
      description:
        recommended: true
  AnnotationExtension:
    slots:
      - predicate
      - term
    slot_usage:
      predicate:
        bindings:
          - binds_value_of: id
            range: ROTermEnum
            obligation_level: REQUIRED

  TermMapping:
    description: A mapping between the proposed term and an equivalent term in another ontology
    attributes:
      predicate:
        range: string
        required: true
        description: Mapping predicate (e.g., 'skos:exactMatch', 'skos:closeMatch', 'skos:broadMatch', 'skos:narrowMatch')
      target_term:
        range: Term
        required: true
        inlined: true
        description: The target term in another ontology

  ProposedOntologyTerm:
    description: A proposed new ontology term that should exist but doesn't currently
    attributes:
      proposed_name:
        range: string
        required: true
        description: Proposed name for the new term
      proposed_definition:
        range: string
        required: true
        description: Proposed definition for the new term
      justification:
        range: string
        description: Justification for why this term is needed
      proposed_parent:
        range: Term
        inlined: true
        description: Proposed parent term in the ontology hierarchy
      proposed_mappings:
        range: TermMapping
        multivalued: true
        inlined_as_list: true
        description: Proposed mappings to equivalent terms in other ontologies
      supported_by:
        range: SupportingTextInReference
        multivalued: true

  KnowledgeGap:
    description: >-
      A curated, literature-grounded statement of what is NOT known about a gene
      product, core function, module, or step — its molecular activity, mechanism,
      partner(s), localization, or biological role. A knowledge gap is a reviewer
      judgment reached by reading the primary literature, NOT a pattern in the
      annotations: a heavily annotated gene can hide a gaping mechanistic hole, and
      a sparsely annotated one can be perfectly understood and merely under-curated.
      Each gap is a small, defensible scholarly object with the same evidentiary
      discipline used for positive claims. See projects/FUNCTION_KNOWLEDGE_GAPS.md
      for the methodology and worked exemplars.
    attributes:
      gap_statement:
        range: string
        required: true
        description: >-
          The specific unknown, stated precisely. Not "role unclear" but e.g.
          "the direct substrate / the catalytic activity / the essential partner is
          undetermined".
      boundary:
        range: string
        recommended: true
        description: >-
          What IS firmly established, so the gap is sharply delimited. The edge of
          current knowledge against which the gap is defined.
      gap_kind:
        range: KnowledgeGapKindEnum
        multivalued: true
        recommended: true
        description: >-
          The kind(s) of ignorance — biology, curation, and/or ontology — which
          determines who can resolve it. Multiple values denote a blend (e.g. a
          biology gap with an ontology shadow).
      dark_aspect:
        range: KnowledgeGapAspectEnum
        description: >-
          Which GO aspect (or pattern) is dark for this gap. Most "dark" genes are
          not uniformly dark.
      status:
        range: KnowledgeGapStatusEnum
        description: Lifecycle status of the gap, tracking progress toward resolution.
      significance:
        range: string
        description: Why closing this gap matters.
      resolution:
        range: string
        description: >-
          What would resolve the gap — the experiment, ontology term, or curation
          action. For ONTOLOGY gaps, pair with proposed_terms (or the gene's
          top-level proposed_new_terms).
      provenance:
        range: SupportingTextInReference
        multivalued: true
        inlined_as_list: true
        recommended: true
        description: >-
          Evidence that the unknown is REAL and not merely uncurated — ideally the
          field's own admissions of ignorance ("remains to be determined", "the
          precise role is unknown"). The supporting_text is a verbatim substring of
          the cited reference and is checked by the reference validator exactly like
          supported_by. When the only available source is a DOI-only paper or a local
          analysis, anchor provenance to that reference_id (e.g. a file: path).
      proposed_terms:
        range: ProposedOntologyTerm
        multivalued: true
        inlined_as_list: true
        description: >-
          For ONTOLOGY gaps, the new ontology term(s) that would let the knowledge be
          expressed (e.g. "structural constituent of complex X"). May elaborate the
          gene's top-level proposed_new_terms.

  Experiment:
    description: A suggested experiment to answer a question about the gene
    attributes:
      hypothesis:
        range: string
        recommended: true
        description: Hypothesis to be investigated
      description:
        range: string
        required: true
        description: Detailed description of the experiment to be performed
      experiment_type:
        range: string
        required: false
        description: Type of experiment or assay to answer the question

  Question:
    description: A question to be answered about the gene
    attributes:
      question:
        range: string
        required: true
        description: Question to be answered
      experts:
        range: string
        multivalued: true
        required: false
        description: Experts to answer the question. These should be drawn from the authors of relevant publications already referenced.
          If no suitable experts are available, it's OK to leave this as an empty list!


  # ============== Rule Review Classes ==============

  RuleReview:
    description: >-
      A review of a UniProt annotation rule (ARBA or UniRule).
      Each review covers ONE rule and assesses its quality, literature support,
      and biological appropriateness.
    slots:
      - id
      - description
      - references
    slot_usage:
      id:
        description: The rule ID (e.g., ARBA00026249, UR000000070)
    attributes:
      status:
        range: RuleReviewStatusEnum
        description: Status of the rule review
      rule_type:
        range: RuleTypeEnum
        required: true
        description: Type of rule (ARBA or UniRule)
      rule:
        range: EmbeddedRule
        required: true
        inlined: true
        description: The embedded rule being reviewed
      review_summary:
        range: string
        description: Overall summary of the review findings
      action:
        range: RuleActionEnum
        required: true
        description: Recommended action for this rule
      action_rationale:
        range: string
        description: Rationale for the recommended action
      suggested_modifications:
        range: string
        multivalued: true
        description: Specific modifications suggested if action is MODIFY
      parsimony:
        range: ParsimonyAssessment
        inlined: true
        description: Assessment of rule parsimony (simplicity vs complexity)
      literature_support:
        range: LiteratureSupportAssessment
        inlined: true
        description: Assessment of literature support for the rule
      condition_overlap:
        range: ConditionOverlapAssessment
        inlined: true
        description: Assessment of overlap between rule conditions
      go_specificity:
        range: GOSpecificityAssessment
        inlined: true
        description: Assessment of GO term specificity
      taxonomic_scope:
        range: TaxonomicScopeAssessment
        inlined: true
        description: Assessment of taxonomic restriction appropriateness
      confidence:
        range: float
        minimum_value: 0.0
        maximum_value: 1.0
        description: Overall confidence in the rule (0.0 to 1.0)
      supported_by:
        range: SupportingTextInReference
        multivalued: true
        description: Supporting text from literature for this review

  EmbeddedRule:
    description: >-
      An embedded representation of an ARBA or UniRule for storage in YAML.
      Captures the essential structure: conditions (antecedent) and GO annotations (consequent).
    attributes:
      rule_id:
        range: string
        required: true
        description: Original rule ID (e.g., ARBA00026249, UR000000070)
      condition_sets:
        range: RuleConditionSet
        multivalued: true
        inlined_as_list: true
        required: true
        description: >-
          List of condition sets (OR-ed together). Each condition set is a conjunction
          (AND) of conditions. The rule fires if ANY condition set matches.
      go_annotations:
        range: RuleGOAnnotation
        multivalued: true
        inlined_as_list: true
        description: GO terms assigned by this rule
      ipr2go_redundancy:
        range: InterPro2GORedundancy
        inlined: true
        description: Analysis of redundancy with InterPro2GO mappings
      entries:
        range: RuleReviewEntry
        multivalued: true
        inlined_as_list: true
        required: true
        description: >-
          Entry-centric view of all entities in the rule (domain conditions and GO terms).
          Each entry tracks its relationships (PREDICTS, PREDICTED_BY, EQUIV) to other entries.
      reviewed_protein_count:
        range: integer
        description: Number of reviewed (Swiss-Prot) proteins annotated by this rule
      unreviewed_protein_count:
        range: integer
        description: Number of unreviewed (TrEMBL) proteins annotated by this rule
      created_date:
        range: string
        description: Date the rule was created
      modified_date:
        range: string
        description: Date the rule was last modified

  RuleConditionSet:
    description: >-
      A set of conditions that must ALL be true (conjunction/AND).
      Multiple condition sets in a rule are OR-ed together (disjunction).
    attributes:
      number:
        range: integer
        required: true
        minimum_value: 1
        description: 1-based condition set number (CS1, CS2, CS3, etc.)
      conditions:
        range: RuleCondition
        multivalued: true
        inlined_as_list: true
        required: true
        description: Conditions in this set (all must match)
      notes:
        range: string
        description: Reviewer notes on this specific condition set
      pairwise_overlap:
        range: PairwiseOverlap
        multivalued: true
        inlined_as_list: true
        description: >-
          Pairwise overlap statistics for domain conditions in this set.
          Only computed for InterPro, FunFam, PANTHER conditions.
          Provides set difference metrics (uniqueness) and Jaccard similarity.

  RuleCondition:
    description: A single condition in a rule antecedent
    attributes:
      condition_type:
        range: ConditionTypeEnum
        required: true
        description: Type of condition
      value:
        range: string
        required: true
        description: The condition value (e.g., IPR000001, NCBITaxon:4751)
      curie:
        range: string
        description: Normalized CURIE form (e.g., InterPro:IPR000001, NCBITaxon:4751)
      label:
        range: string
        description: Human-readable label
      interpro_type:
        range: InterProTypeEnum
        description: >-
          InterPro entry type (family, domain, active_site, etc.).
          Only populated for InterPro conditions (condition_type = INTERPRO).
          Extracted from InterPro metadata or API.
      negated:
        range: boolean
        description: Whether this is a negative condition (NOT)
      protein_count:
        range: integer
        minimum_value: 0
        description: >-
          Number of proteins matching this condition in specified database.
          Only populated for domain/family conditions (InterPro, FunFam, PANTHER).
          Null for taxon and other condition types.
      protein_database:
        range: ProteinDatabaseEnum
        description: >-
          Which protein database was queried (e.g., SWISSPROT, TREMBL).
          Defaults to SWISSPROT (reviewed proteins only).
          Important to specify since counts differ dramatically between databases.
      uniqueness_score:
        range: float
        minimum_value: 0.0
        maximum_value: 1.0
        description: >-
          Measure of domain uniqueness (0.0 to 1.0).
          Calculated as 1.0 - mean(containment in other domains in same condition set).
          High score = more unique/specific domain.
          Low score = broad domain that commonly co-occurs.
      sample_proteins:
        range: string
        multivalued: true
        description: >-
          Sample UniProt IDs matching this condition.
          Only included when protein_count < 20 to avoid bloating files.
          Limited to max 10 examples.

  RuleGOAnnotation:
    description: A GO annotation produced by the rule
    attributes:
      go_id:
        range: string
        required: true
        description: GO term ID (e.g., GO:0004791)
      go_label:
        range: string
        description: GO term name
      aspect:
        range: string
        description: GO aspect (F, P, or C)

  # ============== Rule Analysis Classes ==============

  PairwiseOverlap:
    description: >-
      Overlap statistics between two domain conditions (InterPro, FunFam, etc.)
      in the same condition set. Provides set difference metrics to measure
      uniqueness and redundancy.
    attributes:
      condition_a:
        range: string
        required: true
        description: First condition value (e.g., IPR005982)
      condition_b:
        range: string
        required: true
        description: Second condition value (e.g., IPR008255)
      condition_a_label:
        range: string
        description: Human-readable label for condition A
      condition_b_label:
        range: string
        description: Human-readable label for condition B
      protein_database:
        range: ProteinDatabaseEnum
        required: true
        description: Which protein database was queried (SWISSPROT or TREMBL)
      count_a:
        range: integer
        required: true
        minimum_value: 0
        description: Number of proteins matching condition A in specified database
      count_b:
        range: integer
        required: true
        minimum_value: 0
        description: Number of proteins matching condition B in specified database
      intersection_count:
        range: integer
        required: true
        minimum_value: 0
        description: Number of proteins matching BOTH A AND B (|A ∩ B|) in specified database
      a_minus_b_count:
        range: integer
        required: true
        minimum_value: 0
        description: >-
          Number of proteins in A but not in B (|A - B|).
          Represents the uniqueness of A with respect to B.
          High value = A adds unique coverage beyond B.
          Zero value = A is completely contained in B (A ⊆ B).
      b_minus_a_count:
        range: integer
        required: true
        minimum_value: 0
        description: >-
          Number of proteins in B but not in A (|B - A|).
          Represents the uniqueness of B with respect to A.
          High value = B adds unique coverage beyond A.
          Zero value = B is completely contained in A (B ⊆ A).
      jaccard_similarity:
        range: float
        required: true
        minimum_value: 0.0
        maximum_value: 1.0
        description: >-
          Jaccard similarity coefficient: |A ∩ B| / |A ∪ B|
          = intersection / (count_a + count_b - intersection).
          0.0 = no overlap, 1.0 = complete overlap.
      containment_a_in_b:
        range: float
        required: true
        minimum_value: 0.0
        maximum_value: 1.0
        description: >-
          Proportion of A contained in B: |A ∩ B| / |A|.
          1.0 means A is completely contained in B (A ⊆ B).
      containment_b_in_a:
        range: float
        required: true
        minimum_value: 0.0
        maximum_value: 1.0
        description: >-
          Proportion of B contained in A: |A ∩ B| / |B|.
          1.0 means B is completely contained in A (B ⊆ A).
      interpretation:
        range: OverlapInterpretationEnum
        description: Automated interpretation of overlap pattern
      condition_a_in_sets:
        range: integer
        multivalued: true
        minimum_value: 1
        description: List of 1-based condition set indices where condition A appears
      condition_b_in_sets:
        range: integer
        multivalued: true
        minimum_value: 1
        description: List of 1-based condition set indices where condition B appears

  RuleReviewEntry:
    description: >-
      An entity in the rule - either a domain/family condition or a GO term target.
      Each entry tracks its relationships (predictions, predicted-by, equivalence) to other entries.
    attributes:
      id:
        range: string
        required: true
        identifier: true
        description: Identifier (IPR005982, GO:0004791, 3.50.50.60:FF:000064, etc.)
      label:
        range: string
        description: Human-readable name
      type:
        range: EntryTypeEnum
        required: true
        description: Type of entry (INTERPRO, FUNFAM, PANTHER, GO_TERM, etc.)
      appears_in_condition_sets:
        range: integer
        multivalued: true
        minimum_value: 1
        description: Which condition sets (1-based) contain this entry (for domain conditions only)
      protein_count:
        range: integer
        description: Number of proteins matching this condition (from SwissProt)
      source:
        range: string
        description: >-
          Source of this entry if external to the rule (e.g., 'ipr2go' for InterPro entries
          that map to the same GO term via InterPro2GO but are not part of any condition set)
      asserted_predicted_go_terms:
        range: string
        multivalued: true
        description: >-
          GO terms that this entry maps to via external mappings (e.g., ipr2go).
          Only populated for external entries not in the rule's condition sets.
      related_entries:
        range: RelatedEntry
        multivalued: true
        inlined_as_list: true
        description: Relationships to other entries in the rule

  RelatedEntry:
    description: >-
      A relationship from this entry to another entry in the rule.
      Categorized as PREDICTS (this → other), PREDICTED_BY (other → this), or EQUIV (bidirectional).
    attributes:
      relationship:
        range: EntryRelationshipEnum
        required: true
        description: Type of relationship
      target_id:
        range: string
        required: true
        description: ID of the related entry
      containment:
        range: float
        minimum_value: 0.0
        maximum_value: 1.0
        description: >-
          Containment score (0-1) for the directional relationship.
          For PREDICTS: this_in_target (how much of this is contained in target).
          For PREDICTED_BY: target_in_this (how much of target is contained in this).
          For EQUIV: max of both directions.
      jaccard_similarity:
        range: float
        minimum_value: 0.0
        maximum_value: 1.0
        description: Jaccard similarity coefficient (0-1)
      intersection_count:
        range: integer
        description: Number of proteins in both this and target
      exclusive_count:
        range: integer
        description: >-
          Number of proteins exclusive to the "source" of the relationship.
          For PREDICTS: proteins in this but not target.
          For PREDICTED_BY: proteins in target but not this.
          For EQUIV: proteins in this but not target (A - B).

  InterPro2GORedundancy:
    description: >-
      Analysis of whether rule GO annotations are redundant with
      existing InterPro2GO mappings from the GO Consortium.
    attributes:
      redundant_annotations:
        range: RedundantAnnotation
        multivalued: true
        inlined_as_list: true
        description: GO annotations that already exist in InterPro2GO
      novel_annotations:
        range: string
        multivalued: true
        description: GO IDs not found in InterPro2GO for any rule condition
      summary:
        range: string
        description: Human-readable summary of redundancy analysis

  RedundantAnnotation:
    description: A GO annotation that is redundant with an existing InterPro2GO mapping
    attributes:
      go_id:
        range: string
        required: true
        description: GO term ID (e.g., GO:0004791)
      go_label:
        range: string
        description: GO term label
      interpro_source:
        range: string
        required: true
        description: InterPro ID that already maps to this GO term in ipr2go
      interpro_label:
        range: string
        description: InterPro domain label

  # ============== Rule Assessment Classes ==============

  ParsimonyAssessment:
    description: Assessment of rule parsimony (simplicity vs complexity)
    attributes:
      assessment:
        range: ParsimonyEnum
        required: true
        description: Parsimony assessment value
      notes:
        range: string
        description: Notes on parsimony - e.g., which conditions are redundant
      supported_by:
        range: SupportingTextInReference
        multivalued: true
        description: Supporting text from literature for this assessment

  LiteratureSupportAssessment:
    description: Assessment of literature support for the rule
    attributes:
      assessment:
        range: LiteratureSupportEnum
        required: true
        description: Level of literature support
      notes:
        range: string
        description: Notes on literature support - key papers, gaps in evidence
      supported_by:
        range: SupportingTextInReference
        multivalued: true
        description: Supporting text from literature for this assessment

  ConditionOverlapAssessment:
    description: Assessment of overlap between rule conditions
    attributes:
      assessment:
        range: OverlapEnum
        required: true
        description: Overlap assessment value
      notes:
        range: string
        description: >-
          Notes on condition overlap - e.g., "IPR000001 and IPR000002 both represent
          the same structural domain" or "FunFam subsumes the InterPro entry"
      supported_by:
        range: SupportingTextInReference
        multivalued: true
        description: Supporting text from literature for this assessment

  GOSpecificityAssessment:
    description: Assessment of GO term specificity
    attributes:
      assessment:
        range: SpecificityEnum
        required: true
        description: Specificity assessment value
      notes:
        range: string
        description: Notes on specificity - suggested alternative terms if too broad/narrow
      supported_by:
        range: SupportingTextInReference
        multivalued: true
        description: Supporting text from literature for this assessment

  TaxonomicScopeAssessment:
    description: Assessment of taxonomic restriction appropriateness
    attributes:
      assessment:
        range: TaxonomicScopeEnum
        required: true
        description: Taxonomic scope assessment value
      notes:
        range: string
        description: Notes on taxonomic scope - suggested changes to taxon constraints
      supported_by:
        range: SupportingTextInReference
        multivalued: true
        description: Supporting text from literature for this assessment

  # ============== Prediction Review Classes ==============

  PredictionReview:
    description: >-
      Review of computational/ML predictions for a gene that are NOT in the curated
      GOA/UniProt annotations. This captures predictions from methods like DeepECTF,
      PANTHER/IBA, InterPro2GO, CLEAN, GloEC, MAPred, etc. and evaluates them against
      literature and bioinformatic evidence. Inspired by the systematic evaluation in
      de Crécy-Lagard et al. 2025 (PMID:40703034).
    slots:
      - id
      - gene_symbol
      - taxon
      - description
      - references
      - status
    attributes:
      locus_tag:
        range: string
        description: Locus tag for the gene (e.g., b1267 for E. coli)
      source_documents:
        range: string
        multivalued: true
        description: Paths to supporting source documents (e.g., reasoning traces, raw model outputs) for provenance
      predictions:
        range: PredictedAnnotation
        multivalued: true
        inlined_as_list: true
        description: List of predictions to review
    slot_usage:
      id:
        description: UniProt accession for the gene product
      description:
        description: Summary of the prediction review findings
        recommended: true

  PredictedAnnotation:
    description: >-
      A single computational prediction and its review. Each prediction comes from
      a specific method and predicts a term (EC number, GO term, etc.) for the gene.
    attributes:
      source_method:
        range: string
        required: true
        description: >-
          Name of the prediction method (e.g., DeepECTF, PANTHER_IBA, InterPro2GO,
          CLEAN, GloEC, MAPred, ProteinInfer)
      source_version:
        range: string
        description: Version or date of the prediction method
      source_reference_id:
        range: string
        description: >-
          Reference for the prediction method or study (e.g., PMID:37820725 for
          the Kim et al. 2023 DeepECTF study)
      predicted_term:
        range: Term
        inlined: true
        required: true
        description: The predicted term (GO term, EC number, etc.)
      predicted_term_type:
        range: PredictedTermTypeEnum
        required: true
        description: Type of predicted term (EC, GO_MF, GO_BP, GO_CC)
      review:
        range: PredictionAssessment
        inlined: true
        required: true
        description: Assessment of this prediction

  PredictionAssessment:
    description: >-
      Assessment of a single computational prediction. Uses categories from
      de Crécy-Lagard et al. 2025 (PMID:40703034) plus extensions for GO predictions.
    attributes:
      assessment:
        range: PredictionAssessmentEnum
        required: true
        description: Assessment category for this prediction
      confidence_score:
        range: integer
        minimum_value: 0
        maximum_value: 2
        required: true
        description: >-
          Confidence score following de Crécy-Lagard et al. 2025:
          2 = concordant with evidence, 1 = uncertain, 0 = discordant with evidence
      error_type:
        range: PredictionErrorTypeEnum
        description: >-
          Type of error that led to the incorrect prediction, following
          Table 1 of de Crécy-Lagard et al. 2025
      summary:
        range: string
        required: true
        description: Summary of the assessment rationale
      supported_by:
        range: SupportingTextInReference
        multivalued: true
        description: Supporting evidence for the assessment


enums:
  ModuleScopeEnum:
    description: How concrete the module document is expected to be.
    permissible_values:
      CONCRETE:
        description: >-
          A module representing a specific pathway, complex, process, or
          taxon-scoped realization where terminal steps should generally ground
          to representative members.
      ABSTRACT:
        description: >-
          A reusable motif or template that intentionally uses abstract
          participant selectors and is not expected to ground each terminal node
          to concrete representative proteins.

  ModuleTypeEnum:
    description: Broad type of biological module node.
    permissible_values:
      MODULE:
        description: Generic or unspecified module.
      BIOLOGICAL_PROCESS:
        description: Biological-process-like module.
      MOLECULAR_FUNCTION:
        description: Molecular-function-like module.
      METABOLIC_PATHWAY:
        description: Metabolic pathway or pathway segment.
      SIGNALING_PATHWAY:
        description: Signaling pathway or pathway segment.
      DEVELOPMENTAL_PROCESS:
        description: Developmental process, stage, or program.
      CELLULAR_COMPONENT:
        description: Cellular component or structure viewed as a module.
      ORGANELLE_LIFECYCLE:
        description: Assembly, maintenance, operation, and turnover of an organelle.
      PROTEIN_COMPLEX:
        description: Protein complex or complex lifecycle.
      REACTION:
        description: Reaction-like module step.
      TRANSPORT_STEP:
        description: Transport or translocation step.
      REGULATORY_STEP:
        description: Regulatory or control step.

  VariantSelectionEnum:
    description: How variants in a variant set may be selected in a realization.
    permissible_values:
      EXACTLY_ONE:
        description: Exactly one variant applies.
      ONE_OR_MORE:
        description: One or more variants may apply.
      ZERO_OR_MORE:
        description: Variants are optional and any number may apply.

  ChainingStatusEnum:
    description: >-
      Curator adjudication of reaction continuity across a module connection
      (whether the upstream reaction's product is consumed as the downstream
      reaction's substrate). Used as an explicit override for the advisory,
      non-blocking automated chaining check.
    permissible_values:
      VERIFIED:
        description: >-
          The upstream product is confirmed to be the downstream substrate
          (curator-confirmed continuity).
      KNOWLEDGE_GAP:
        description: >-
          The connecting intermediate or enzyme is genuinely not known; the break
          reflects missing biological knowledge, not a modelling error.
      MAPPING_GAP:
        description: >-
          The chemistry is known but the GO/RHEA (or GO/ChEBI) mapping does not
          yet capture the link, so the automated check cannot see the continuity.
      NOT_APPLICABLE:
        description: >-
          Chaining does not apply to this connection (e.g. a regulatory or
          non-metabolic edge, or a spiral re-entry handled elsewhere).
      UNVERIFIED:
        description: >-
          Continuity has not been assessed (explicitly, as opposed to simply
          leaving the field unset).

  ConformanceStatusEnum:
    description: How closely a module node matches a template motif it conforms to.
    permissible_values:
      EXACT:
        description: >-
          The node matches the template motif exactly: same steps, function
          terms, and connection topology, with no recorded deviations.
      WITH_DEVIATIONS:
        description: >-
          The node matches the template motif apart from the differences listed
          in deviations (e.g. a merged or missing tier, a substituted term).
      EXTENDS:
        description: >-
          The node contains the full template motif and adds further steps or
          structure beyond it.

  ParticipantSelectorTypeEnum:
    description: How a module annoton participant is selected.
    permissible_values:
      GENE:
        description: A concrete gene.
      GENE_PRODUCT:
        description: A concrete gene product, protein, isoform, or form.
      PROTEIN_COMPLEX:
        description: A concrete protein-containing complex or subcomplex.
      FAMILY:
        description: Any member of a specified family or orthogroup.
      DOMAIN:
        description: Any entity with a specified domain, motif, or site.
      ORTHOLOG_OF:
        description: Any ortholog of a specified gene.
      HOMOLOG_OF:
        description: Any homolog of a specified gene.
      ANY_WITH_FUNCTION:
        description: Any participant satisfying the specified molecular function descriptor.
      ANY_WITH_DOMAIN:
        description: Any participant satisfying the specified domain descriptor.
      ANY_PARTICIPANT:
        description: An unspecified participant.

  ModuleConnectionTypeEnum:
    description: Common connection types between module elements.
    permissible_values:
      PRECEDES:
        description: The source occurs before the target.
      CAUSES:
        description: The source causally promotes or produces the target.
      POSITIVELY_REGULATES:
        description: The source positively regulates the target.
      NEGATIVELY_REGULATES:
        description: The source negatively regulates the target.
      PROVIDES_INPUT_FOR:
        description: The source provides material, signal, or context used by the target.
      HAS_INPUT:
        description: The target has the source as input.
      HAS_OUTPUT:
        description: The source has the target as output.
      PART_OF:
        description: The source is part of the target.

  EvidenceType:
    description: |-
      Gene Ontology evidence codes mapped to Evidence and Conclusion Ontology (ECO) terms
    permissible_values:
      
      # Experimental Evidence Codes
      EXP:
        description: Inferred from Experiment
        meaning: ECO:0000269
        
      IDA:
        description: Inferred from Direct Assay
        meaning: ECO:0000314
        
      IPI:
        description: Inferred from Physical Interaction
        meaning: ECO:0000353
        
      IMP:
        description: Inferred from Mutant Phenotype
        meaning: ECO:0000315
        
      IGI:
        description: Inferred from Genetic Interaction
        meaning: ECO:0000316
        
      IEP:
        description: Inferred from Expression Pattern
        meaning: ECO:0000270

      # High Throughput Experimental Evidence Codes
      HTP:
        description: Inferred from High Throughput Experiment
        meaning: ECO:0006056
        
      HDA:
        description: Inferred from High Throughput Direct Assay
        meaning: ECO:0007005
        
      HMP:
        description: Inferred from High Throughput Mutant Phenotype
        meaning: ECO:0007001
        
      HGI:
        description: Inferred from High Throughput Genetic Interaction
        meaning: ECO:0007003
        
      HEP:
        description: Inferred from High Throughput Expression Pattern
        meaning: ECO:0007007

      # Phylogenetically-inferred annotations
      IBA:
        description: Inferred from Biological aspect of Ancestor
        meaning: ECO:0000318
        
      IBD:
        description: Inferred from Biological aspect of Descendant
        meaning: ECO:0000319
        
      IKR:
        description: Inferred from Key Residues
        meaning: ECO:0000320
        
      IRD:
        description: Inferred from Rapid Divergence
        meaning: ECO:0000321

      # Computational Analysis Evidence Codes
      ISS:
        description: Inferred from Sequence or structural Similarity
        meaning: ECO:0000250
        
      ISO:
        description: Inferred from Sequence Orthology
        meaning: ECO:0000266
        
      ISA:
        description: Inferred from Sequence Alignment
        meaning: ECO:0000247
        
      ISM:
        description: Inferred from Sequence Model
        meaning: ECO:0000255
        
      IGC:
        description: Inferred from Genomic Context
        meaning: ECO:0000317
        
      RCA:
        description: Inferred from Reviewed Computational Analysis
        meaning: ECO:0000245

      # Author Statement Evidence Codes
      TAS:
        description: Traceable Author Statement
        meaning: ECO:0000304
        
      NAS:
        description: Non-traceable Author Statement
        meaning: ECO:0000303

      # Curator Statement Evidence Codes
      IC:
        description: Inferred by Curator
        meaning: ECO:0000305
        
      ND:
        description: No biological Data available
        meaning: ECO:0000307

      # Electronic Annotation Evidence Code
      IEA:
        description: Inferred from Electronic Annotation
        meaning: ECO:0000501

  PropagationRootCauseEnum:
    description: >-
      Mechanical root-cause classification for propagated or inferred
      annotations. This distinguishes bad source annotations from bad
      propagation decisions and term-scoping issues.
    permissible_values:
      NO_FAILURE_CORE:
        description: The propagated annotation is correct and core for the target.
      NO_FAILURE_NON_CORE:
        description: The propagated annotation is biologically defensible but contextual, secondary, or generic.
      SOURCE_BAD:
        description: The source annotation is wrong, miscited, homonym-confused, or contradicted.
      SOURCE_STALE_OR_MISSING:
        description: The transferred term no longer appears on the current source record, or donor tracing cannot recover it.
      SOURCE_WEAK_OR_INFERRED:
        description: The source exists but is only inferred, statement-level, or otherwise weak for propagation.
      EVIDENCE_CIRCULAR_OR_REDUNDANT:
        description: The propagation chain transfers from another transfer, or the target already has stronger direct evidence.
      PROPAGATION_BAD:
        description: The source annotation is sound, but the term should not propagate to this target.
      TERM_SCOPING_PROBLEM:
        description: The biology is related, but the GO term is too broad, too specific, or has the wrong role or qualifier.
      UNRESOLVED:
        description: The propagation issue was investigated but could not be classified confidently.

  PropagationFailureModeEnum:
    description: Biological subtype for a propagation or inference issue.
    permissible_values:
      WRONG_ORTHOLOG_OR_PARALOG:
        description: Donor/source is a paralog, expanded family member, or wrong subfamily.
      FUNCTIONAL_DIVERGENCE:
        description: Target retained fold or orthology but changed substrate, product, activity, or pathway role.
      PSEUDO_OR_SUBACTIVITY_LOSS:
        description: Catalytic residues or a specific sub-activity are lost even though the domain remains.
      CONTEXT_OR_TISSUE_MISMATCH:
        description: Donor evidence is tissue, developmental, organismal, or disease-context specific.
      LINEAGE_OR_TAXON_MISMATCH:
        description: Process does not occur in the target lineage or organelle system.
      COMPARTMENT_OR_COMPLEX_MISMATCH:
        description: Localization, complex membership, or pathway compartment does not transfer.
      REGULATORY_SIGN_INVERSION:
        description: Family contains activators and inhibitors, and a positive/negative regulatory term leaks across members.
      ROLE_CONFLATION:
        description: Substrate, regulator, effector, or specificity subunit is annotated as the agent or core machinery.
      GRANULARITY_MISMATCH:
        description: Parent term is true but uninformative, or child term overstates specificity.
      SOURCE_MISCITATION:
        description: Source evidence points to the wrong gene, organism, publication, or homonym.
      SOURCE_EVIDENCE_WEAK:
        description: Source evidence is inferred, statement-level, stale, or otherwise too weak for confident propagation.
      CIRCULAR_PROPAGATION:
        description: Propagation chain depends on another propagated annotation rather than independent source evidence.

  ResidueClaimEnum:
    description: What a residue claim asserts about the target protein.
    permissible_values:
      LOST:
        description: >-
          The target does not carry the anchor's functional residue -- it is
          substituted, or no residue aligns to that position at all.
      RETAINED:
        description: >-
          The target does carry the residue. Worth recording explicitly: a retained
          site refutes a "lacks the catalytic residue" argument, which is how the
          CASP12 and LPA errors would have been caught.
      SUBSTITUTED:
        description: >-
          The residue differs but the substitution is conservative or otherwise
          argued not to abolish function. Distinct from LOST, which asserts the
          function is gone.

  ResidueClaimMethodEnum:
    description: How the anchor-to-target position correspondence was established.
    permissible_values:
      MSA:
        description: Multiple sequence alignment; record the alignment_release.
      STRUCTURE:
        description: Structural superposition or a solved complex.
      UNIPROT_FEATURE:
        description: >-
          Both positions read from UniProt feature tables on their own records, with
          no alignment step.
      LITERATURE:
        description: >-
          A publication states the correspondence directly; cite it in
          review.supported_by.

  PropagationSourceStatusEnum:
    description: Mechanical status of a source entity with respect to a propagated target annotation.
    permissible_values:
      SUPPORTS_TRANSFER:
        description: Source evidence supports the term and the transfer to the target.
      SUPPORTS_SOURCE_BUT_NOT_TARGET:
        description: Source evidence supports the source annotation, but propagation to the target is unsafe.
      SOURCE_BAD:
        description: Source annotation or source citation is itself wrong.
      SOURCE_STALE_OR_MISSING:
        description: Current source record no longer carries the transferred term, or tracing cannot recover it.
      SOURCE_WEAK_OR_INFERRED:
        description: Source exists but is only inferred, statement-level, or otherwise weak.
      CIRCULAR_OR_REDUNDANT:
        description: Source participates in a circular transfer chain or adds no independent support.
      NOT_RELEVANT:
        description: Source was inspected but is not relevant to the target annotation.
      UNRESOLVED:
        description: Source could not be classified confidently.

  ActionEnum:
    permissible_values:
      ACCEPT:
        description: Accept the existing annotation as-is, no modifications, and retain as representing the core function of the gene
      KEEP_AS_NON_CORE:
        description: Keep the existing annotation as-is, but mark it as non-core. For pleiotropic genes, this may be the developmental
          processes, or other processes that are not the core function of the gene.
      REMOVE:
        description: Remove the existing annotation, as it is unlikely to be correct based on combined evidence
      MODIFY:
        description: The essence of the annotation is sound, but there are better terms to use (use in combination with proposed_replacement_terms).
          if the term is too general, then MODIFY should be used, with a proposed replacement term for the correct specific function.
          sometimes terms can also be overly specific and contorted, so in some cases you might want to generalize
      MARK_AS_OVER_ANNOTATED:
        description: The term is not entirely wrong, but likely represents an over-annotation of the gene
        see_also:
          - https://github.com/ai4curation/ai-gene-review/blob/main/.claude/skills/annotation-reviewer/SKILL.md
      UNDECIDED:
        description: The annotation is not clear, and the reviewer is not sure what to do with it. ALWAYS USE THIS IF YOU ARE UNABLE TO ACCESS
          RELEVANT PUBLICATIONS
      PENDING:
        description: The review entry is a stub, and the review has not been completed yet.
      NEW:
        description: This is a proposed annotation, not one that exists in the existing GO annotations. Use this to propose a new annotation
          not covered by the existing GO annotations. Use this conservatively, do not over-annotate, especially for biological process.
          Do not use for indirect or pleiotropic effects. Be sure you have good evidence, this can be from multiple sources.

  AnnotationQualifierEnum:
    description: >-
      GO annotation qualifiers specifying the relationship between a gene product and a GO term.
      These correspond to the QUALIFIER column in GAF/GPAD files and the QuickGO API.
    permissible_values:
      enables:
        description: >-
          The gene product has the molecular function activity (MF qualifier).
          The gene product independently possesses this catalytic or binding activity.
      contributes_to:
        description: >-
          The gene product contributes to a molecular function as part of a complex (MF qualifier).
          The gene product does not independently have this activity but is required for
          the complex to have it. Typical for accessory/structural subunits of multi-protein
          complexes (e.g., accessory subunit of NADH dehydrogenase).
      involved_in:
        description: >-
          The gene product is involved in a biological process (BP qualifier).
      acts_upstream_of:
        description: >-
          The gene product acts upstream of or within a biological process (BP qualifier).
      acts_upstream_of_positive_effect:
        description: >-
          Acts upstream of or within, positive effect (BP qualifier).
      acts_upstream_of_negative_effect:
        description: >-
          Acts upstream of or within, negative effect (BP qualifier).
      acts_upstream_of_or_within:
        description: >-
          Acts upstream of or within a biological process (BP qualifier).
      acts_upstream_of_or_within_positive_effect:
        description: >-
          Acts upstream of or within, positive effect (BP qualifier).
      acts_upstream_of_or_within_negative_effect:
        description: >-
          Acts upstream of or within, negative effect (BP qualifier).
      located_in:
        description: >-
          The gene product is located in a cellular component (CC qualifier).
      is_active_in:
        description: >-
          The gene product is active in a cellular component (CC qualifier).
          Stronger than located_in -- implies the gene product carries out its
          molecular function in this location.
      part_of:
        description: >-
          The gene product is part of a cellular component, typically a complex (CC qualifier).
      colocalizes_with:
        description: >-
          The gene product colocalizes with a cellular component (CC qualifier).
          Weaker than located_in or part_of.

  GOTermEnum:
    description: A term in the GO ontology (including roots, to allow ND annotations)
    reachable_from:
      source_nodes:
        - GO:0003674
        - GO:0008150
        - GO:0005575
      is_direct: false
      include_self: true
      relationship_types:
        - rdfs:subClassOf
  
  GOMolecularActivityEnum:
    description: A molecular activity term in the GO ontology
    reachable_from:
      source_nodes:
        - GO:0003674
      is_direct: false
      relationship_types:
        - rdfs:subClassOf

  GOBiologicalProcessEnum:
    description: A biological process term in the GO ontology
    reachable_from:
      source_nodes:
        - GO:0008150
      is_direct: false
      relationship_types:
        - rdfs:subClassOf

  GOCellularLocationEnum:
    description: A cellular location term in the GO ontology (excludes protein-containing complexes)
    reachable_from:
      source_nodes:
        - GO:0110165
      is_direct: false
      relationship_types:
        - rdfs:subClassOf

  GOProteinContainingComplexEnum:
    description: A protein-containing complex term in the GO ontology
    reachable_from:
      source_nodes:
        - GO:0032991
      is_direct: false
      # Accept the root GO:0032991 ("protein-containing complex") itself as a
      # valid in_complex value (used when the specific complex is unspecified).
      # linkml-term-validator >= 0.4.1 defaults include_self to false, so this
      # must be explicit; GOTermEnum sets it for the same reason.
      include_self: true
      relationship_types:
        - rdfs:subClassOf

  ROTermEnum:
    description: A term in the relation ontology

  ProductTypeEnum:
    description: Type of gene product
    permissible_values:
      PROTEIN:
        description: Protein-coding gene
        meaning: SO:0001217  # protein_coding_gene
      MIRNA:
        description: microRNA
        meaning: SO:0000276  # miRNA
      LNCRNA:
        description: Long non-coding RNA
        meaning: SO:0001877  # lncRNA
      SNORNA:
        description: Small nucleolar RNA
        meaning: SO:0000275  # snoRNA
      SNRNA:
        description: Small nuclear RNA
        meaning: SO:0000274  # snRNA
      TRNA:
        description: Transfer RNA
        meaning: SO:0000253  # tRNA
      RRNA:
        description: Ribosomal RNA
        meaning: SO:0000252  # rRNA
      PIRNA:
        description: PIWI-interacting RNA
        meaning: SO:0001035  # piRNA
      ANTISENSE_RNA:
        description: Antisense RNA
        meaning: SO:0000644  # antisense_RNA
      PSEUDOGENE:
        description: Pseudogene
        meaning: SO:0000336  # pseudogene
      OTHER_NCRNA:
        description: Other non-coding RNA
        meaning: SO:0000655  # ncRNA

  GeneReviewStatusEnum:
    description: Status of the gene review process
    permissible_values:
      INITIALIZED:
        description: All annotations have action PENDING - review has been initialized but not started
      IN_PROGRESS:
        description: At least one annotation is PENDING and at least one annotation is not PENDING - review is underway
      DRAFT:
        description: No PENDING annotations, but may have validation warnings - review is complete but needs refinement
      COMPLETE:
        description: No PENDING annotations and no validation warnings - review is fully complete and validated

  ManuscriptSection:
    description: |-
      Sections of a scientific manuscript or publication
    permissible_values:

      TITLE:
        description: Title section
        meaning: IAO:0000305  # document title
        annotations:
          order: 1

      ABSTRACT:
        description: Abstract
        meaning: IAO:0000315  # abstract
        annotations:
          order: 3
          typical_length: 150-300 words

      KEYWORDS:
        description: Keywords
        meaning: IAO:0000630  # keywords section
        annotations:
          order: 4

      # Main Text
      INTRODUCTION:
        description: Introduction/Background
        meaning: IAO:0000316  # introduction to a publication about an investigation
        annotations:
          order: 5
          aliases: Background

      LITERATURE_REVIEW:
        description: Literature review
        meaning: IAO:0000639  # related work section
        annotations:
          order: 6
          optional: true

      METHODS:
        description: Methods/Materials and Methods
        meaning: IAO:0000317  # methods section
        annotations:
          order: 7
          aliases: Materials and Methods, Methodology, Experimental

      RESULTS:
        description: Results
        meaning: IAO:0000318  # results section
        annotations:
          order: 8
          aliases: Findings

      DISCUSSION:
        description: Discussion
        meaning: IAO:0000319  # discussion section of a publication about an investigation
        annotations:
          order: 9

      CONCLUSIONS:
        description: Conclusions
        meaning: IAO:0000615  # conclusion section
        annotations:
          order: 10
          aliases: Conclusion

      APPENDICES:
        description: Appendices
        meaning: IAO:0000326  # supplementary material to a document
        annotations:
          order: 13
          aliases: Appendix

      SUPPLEMENTARY_MATERIAL:
        description: Supplementary material
        meaning: IAO:0000326  # supplementary material to a document
        annotations:
          order: 14
          aliases: Supporting Information, Supplemental Data
          location: often online-only

      DATABASE_ENTRY:
        description: Database entry
        annotations:
          order: 15

      OTHER:
        description: Other main text section
        annotations:
          catch_all: true


  # ============== Rule Review Enums ==============

  RuleTypeEnum:
    description: Type of UniProt annotation rule
    permissible_values:
      ARBA:
        description: Association-Rule-Based Annotator rule (automatically mined)
      UNIRULE:
        description: Expert-curated UniRule

  RuleReviewStatusEnum:
    description: Status of the rule review
    permissible_values:
      PENDING:
        description: Review has not been started
      IN_PROGRESS:
        description: Review is underway
      COMPLETE:
        description: Review is complete

  RuleActionEnum:
    description: Recommended action for the rule
    permissible_values:
      ACCEPT:
        description: Rule is correct and should be kept as-is
      MODIFY:
        description: Rule needs modification (see suggested_modifications)
      DEPRECATE:
        description: Rule should be removed or retired
      SPLIT:
        description: Rule should be split into multiple more specific rules
      MERGE:
        description: Rule should be merged with another related rule
      UNDECIDED:
        description: Unable to determine appropriate action

  ParsimonyEnum:
    description: Assessment of rule parsimony (simplicity vs complexity)
    permissible_values:
      PARSIMONIOUS:
        description: Rule is appropriately simple - conditions are necessary and sufficient
      ACCEPTABLE:
        description: Rule complexity is reasonable given the biological context
      REDUNDANT:
        description: Some conditions are redundant and could be removed
      OVERLY_COMPLEX:
        description: Rule has unnecessary complexity that should be simplified

  LiteratureSupportEnum:
    description: Level of literature support for the rule
    permissible_values:
      STRONG:
        description: Multiple high-quality papers directly support the domain-function relationship
      MODERATE:
        description: Some supporting evidence exists but not comprehensive
      WEAK:
        description: Limited evidence, mostly indirect or from computational studies
      NONE:
        description: No literature support found
      CONTRADICTED:
        description: Literature contradicts the rule's predicted function

  OverlapEnum:
    description: Assessment of condition overlap/redundancy
    permissible_values:
      NONE:
        description: Conditions are independent and non-overlapping
      MINOR:
        description: Slight overlap but conditions add meaningful specificity
      SIGNIFICANT:
        description: Substantial overlap - conditions may be capturing the same thing
      COMPLETE:
        description: Conditions are essentially equivalent/redundant

  SpecificityEnum:
    description: Assessment of GO term specificity
    permissible_values:
      TOO_BROAD:
        description: GO term is too general - a more specific term should be used
      APPROPRIATE:
        description: GO term specificity matches the evidence
      TOO_NARROW:
        description: GO term is overly specific for what the domains predict
      MISMATCHED:
        description: GO term is in wrong branch or aspect

  TaxonomicScopeEnum:
    description: Assessment of taxonomic restriction appropriateness
    permissible_values:
      TOO_BROAD:
        description: Taxon is too inclusive - should be restricted further
      APPROPRIATE:
        description: Taxonomic scope matches the domain's evolutionary distribution
      TOO_NARROW:
        description: Taxon is overly restrictive - function applies more broadly
      MISSING:
        description: Rule lacks necessary taxonomic restriction
      UNNECESSARY:
        description: Taxonomic restriction is not needed for this domain

  ConditionTypeEnum:
    description: Types of conditions in rule antecedents
    permissible_values:
      INTERPRO:
        description: InterPro domain/family
      FUNFAM:
        description: CATH FunFam functional family
      PANTHER:
        description: PANTHER family
      PFAM:
        description: Pfam domain
      TAXON:
        description: Taxonomic constraint
      SEQUENCE_LENGTH:
        description: Sequence length constraint
      OTHER:
        description: Other condition type

  ProteinDatabaseEnum:
    description: Protein database types for rule analysis
    permissible_values:
      SWISSPROT:
        description: Swiss-Prot (reviewed, manually curated proteins)
      TREMBL:
        description: TrEMBL (unreviewed, automatically annotated proteins)
      UNIPROT:
        description: Full UniProtKB (Swiss-Prot + TrEMBL)

  InterProTypeEnum:
    description: InterPro entry types categorizing protein signatures
    permissible_values:
      FAMILY:
        description: Protein family (groups of proteins sharing similar sequence and function)
      DOMAIN:
        description: Protein domain (distinct functional or structural unit)
      ACTIVE_SITE:
        description: Active site (residues directly involved in catalysis)
      BINDING_SITE:
        description: Binding site (residues involved in binding substrates/ligands)
      CONSERVED_SITE:
        description: Conserved site (conserved residues with functional significance)
      REPEAT:
        description: Repeat (short sequence motif that occurs multiple times)
      HOMOLOGOUS_SUPERFAMILY:
        description: Homologous superfamily (proteins with distant evolutionary relationships)
      PTM:
        description: Post-translational modification site

  OverlapInterpretationEnum:
    description: Automated interpretation of domain overlap patterns
    permissible_values:
      REDUNDANT:
        description: Very high overlap (Jaccard > 0.9), conditions are nearly identical
      SUBSET:
        description: One condition is a subset of the other (containment > 0.95)
      HIGH_OVERLAP:
        description: High overlap (Jaccard > 0.5), conditions are similar
      MODERATE:
        description: Moderate overlap (0.2 < Jaccard <= 0.5)
      LOW:
        description: Low overlap (Jaccard <= 0.2), conditions are mostly distinct
      DISJOINT:
        description: No overlap (intersection = 0), conditions are completely distinct

  EntryTypeEnum:
    description: Type of entry in a rule review (domain/family condition or GO term target)
    permissible_values:
      INTERPRO:
        description: InterPro entry (domain, family, repeat, etc.)
      FUNFAM:
        description: CATH FunFam (functional family from CATH database)
      PANTHER:
        description: PANTHER family or subfamily
      GO_TERM:
        description: Gene Ontology term (annotation target)

  EntryRelationshipEnum:
    description: Type of relationship between entries in a rule
    permissible_values:
      PREDICTS:
        description: >-
          This entry predicts the target (this ⊆ target).
          Selected when containment_a_in_b is highest among {jaccard+0.05, containment_a_in_b, containment_b_in_a}.
      PREDICTED_BY:
        description: >-
          This entry is predicted by the target (target ⊆ this).
          Selected when containment_b_in_a is highest among {jaccard+0.05, containment_a_in_b, containment_b_in_a}.
      EQUIV:
        description: >-
          This entry is equivalent to the target (bidirectional high similarity).
          Selected when jaccard_boosted (jaccard + 0.05) is highest among {jaccard+0.05, containment_a_in_b, containment_b_in_a}.

  # ============== Functional Isoform Enums ==============

  FunctionalIsoformTypeEnum:
    description: >-
      Type of functional isoform or product. Distinguishes between different mechanisms
      that produce functionally distinct forms of a gene product.
    permissible_values:
      SPLICE_VARIANT:
        description: >-
          Alternative splicing produces functionally distinct isoforms.
          Maps to one or more UniProt isoform IDs (e.g., P19544-1, P19544-2).
      SPLICE_CLASS:
        description: >-
          A class of splice variants that share functional properties.
          Groups multiple UniProt isoform IDs that have similar functions.
          Example: WT1 +KTS isoforms (multiple UniProt IDs) vs -KTS isoforms.
      CLEAVAGE_PRODUCT:
        description: >-
          Post-translational proteolytic cleavage produces distinct peptides.
          Maps to UniProt chain IDs (PRO_NNNNNNN from FT PEPTIDE lines).
          Example: POMC cleavage into ACTH, alpha-MSH, beta-endorphin.
      MODIFICATION_STATE:
        description: >-
          Post-translational modification creates functionally distinct forms.
          Example: Phosphorylated vs unphosphorylated forms with different activities.
      CONFORMATIONAL_STATE:
        description: >-
          Different conformational states with distinct functions.
          Example: GTP-bound vs GDP-bound forms of GTPases.

  FunctionalIsoformMappingTypeEnum:
    description: Type of identifier that a functional isoform maps to
    permissible_values:
      UNIPROT_ISOFORM:
        description: UniProt isoform ID (e.g., P19544-1, Q07817-2)
      UNIPROT_CHAIN:
        description: UniProt chain/peptide ID from FT PEPTIDE (e.g., PRO_0000024969)

  # ============== Prediction Review Enums ==============

  PredictedTermTypeEnum:
    description: Type of predicted annotation term
    permissible_values:
      EC:
        description: Enzyme Commission number (e.g., EC:2.7.7.87)
      GO_MF:
        description: GO Molecular Function term
      GO_BP:
        description: GO Biological Process term
      GO_CC:
        description: GO Cellular Component term

  PredictionAssessmentEnum:
    description: >-
      Assessment categories for computational predictions, based on
      de Crécy-Lagard et al. 2025 (PMID:40703034) Fig. 4.
    permissible_values:
      COR:
        description: >-
          Correct prediction - validated by literature/bioinformatic evidence as a
          genuinely novel correct prediction (CS=2)
      CNN:
        description: >-
          Correct but Not Novel - the prediction matches an annotation already present
          in UniProt or in the training data. Not a novel discovery (CS=2)
      LSP:
        description: >-
          Less Precise - the prediction is more generic than the existing annotation.
          Correct at a higher level but not informative (CS=2)
      UNC:
        description: >-
          Uncertain - the prediction cannot be validated or refuted with available
          evidence. Requires additional experiments or data (CS=1)
      PLI:
        description: >-
          Paralog Incorrect - wrong prediction due to failure to distinguish
          nonisofunctional paralogs within a protein superfamily (CS=0)
      NPI:
        description: >-
          Nonparalog Incorrect - wrong prediction refuted by evidence:
          pathway absent in organism, activity belongs to a different gene,
          or published data contradict the prediction (CS=0)
      REP:
        description: >-
          Repetition - frequency-biased duplication of a common EC/GO term.
          The model assigns a high-frequency term (e.g., histidine kinase) to
          proteins with no sequence similarity to that family (CS=0)

  PredictionErrorTypeEnum:
    description: >-
      Types of errors that lead to incorrect functional predictions. The first
      block is based on Table 1 of de Crécy-Lagard et al. 2025 (PMID:40703034);
      the trailing values capture additional, recurrent failure patterns observed
      when evaluating sequence- and LLM-based function predictors (e.g. ProtNLM2,
      BioReason-Pro) that Table 1 does not name explicitly.
    permissible_values:
      FAILURE_TO_CAPTURE_LITERATURE:
        description: >-
          Type 1: Function is known and published but not captured in the database
          used for training. The protein is falsely labeled as unknown.
      NAMING_INCONSISTENCY:
        description: >-
          Type 2: Inconsistent naming of the same entity across databases leads
          to missed or incorrect propagation.
      MULTIPLE_FUNCTIONS:
        description: >-
          Type 3: Protein has multiple functions (fusion, moonlighting, promiscuity)
          and only one function is captured or a non-primary function is predicted.
      CURATION_MISTAKE:
        description: >-
          Type 4: The training data contain an incorrect annotation from a biocuration
          error or an outdated annotation that has since been corrected.
      EXPERIMENTAL_MISTAKE:
        description: >-
          Type 5: The training data are based on experimental findings that have been
          refuted or are inconclusive.
      PARALOG_OVERANNOTATION:
        description: >-
          Type 6: Annotation wrongly propagated to a nonisofunctional paralog.
          The model fails to distinguish between paralogs with different substrate
          specificities or functions within the same superfamily.
      FREQUENCY_BIAS:
        description: >-
          The model makes predictions biased toward high-frequency labels in the
          training data, regardless of sequence features. Common with EC numbers
          like histidine kinase (2.7.13.3) or PTS transporter (2.7.1.69).
      TRAINING_DATA_CONTAMINATION:
        description: >-
          The prediction appears novel but the annotation was already present in
          the version of the database used to build the training set.
      PATHWAY_CONTEXT_IGNORED:
        description: >-
          The model ignores metabolic/pathway context. The predicted activity requires
          a pathway that is absent from the organism's genome.
      IN_VITRO_NOT_IN_VIVO:
        description: >-
          The predicted activity can be demonstrated in vitro but does not represent
          the in vivo biological function (e.g., promiscuous activity at orders of
          magnitude lower rate than the dedicated enzyme).
      PSEUDOENZYME_OVERANNOTATION:
        description: >-
          The model assigns the ancestral catalytic activity of a domain family to a
          member that retains the fold but has lost or degraded the catalytic residues
          (a pseudoenzyme), failing to detect substituted/missing active-site residues.
          E.g. predicting demethylase activity for a JmjC protein with a degenerate
          active site, chitinase activity for a member lacking the catalytic glutamate,
          or peroxidase activity for a peroxiredoxin that has lost its resolving cysteine
          and instead acts as a chaperone. A special case of MULTIPLE_FUNCTIONS /
          neofunctionalization where the divergence is specifically loss of catalysis.
      LOCALIZATION_DEFAULT:
        description: >-
          The model defaults to a cytosolic/cytoplasmic subcellular localization when no
          transmembrane or signal-sequence features are detected, mislocalizing secreted,
          periplasmic, organellar (mitochondrial, ER, vacuolar), or membrane proteins.
          Tends to succeed only when a domain/family name explicitly encodes the
          compartment (e.g. BiP/KAR2 -> ER).
      TAXON_CONSTRAINT_VIOLATION:
        description: >-
          The predicted term is valid only in a lineage/kingdom different from the
          organism (e.g. animal-specific 'neuronal cell body' or adaptive-immune terms
          predicted for a plant or bacterial protein), reflecting homology transfer from a
          better-studied taxon. Distinct from PATHWAY_CONTEXT_IGNORED (a missing metabolic
          pathway) in that the violated constraint is taxonomic rather than pathway-level.
      WRONG_INPUT_SEQUENCE:
        description: >-
          A data-pipeline error rather than a model-reasoning error: the predictor was
          supplied the wrong input (e.g. the sequence of a different gene), so every output
          describes the wrong protein. Recorded to distinguish upstream pipeline mistakes
          from genuine model mispredictions.
      DOMAIN_ARCHITECTURE_MISMATCH:
        description: >-
          The predicted activity requires a domain, catalytic region, or complete
          architecture absent from the selected protein. Includes transfer of a
          multidomain donor's activity through a shared noncatalytic domain.
          Does not establish that the target is a fold-retaining pseudoenzyme,
          that its gene model is wrong, or that the historical model input differed.
      COMPLEX_ACTIVITY_TRANSFER:
        description: >-
          Intrinsic catalytic activity of a molecular complex is assigned to a
          noncatalytic accessory subunit. Participation in the complex or its
          biological process does not establish that the subunit catalyzes the reaction.


  ReferenceReplacementReasonEnum:
    description: Reason for a manually verified identifier replacement, distinct from
      the empirical standing of a publication's findings.
    permissible_values:
      DUPLICATE_RECORD:
        description: Source record was deleted or merged as a duplicate of the target record.
      REPLACED_RECORD:
        description: The source authority explicitly replaced this record with the target record.
      WRONG_IDENTIFIER:
        description: The original citation used the wrong identifier; the target identifies the intended paper.

  ReferenceRelevanceEnum:
    description: Reviewer's assessment of how relevant a reference is to the gene's function and review.
    permissible_values:
      HIGH:
        description: Directly establishes or strongly informs the gene's function, mechanism, process, or localization
      MEDIUM:
        description: Provides supporting or corroborating evidence for a function or annotation
      LOW:
        description: Background or contextual only (e.g. family/pathway reviews, methods, or a passing mention)
      NONE:
        description: Not relevant to this gene's function (a candidate for removal from the references)

  ReferenceCorrectnessEnum:
    description: >-
      Reviewer's overall manual assessment of a reference's trustworthiness, spanning both citation
      correctness (does the identifier point to the intended paper that supports its use) and
      scientific soundness (is that paper's claim reliable). Single-valued: record the most salient
      issue and elaborate in review_notes. Complements is_invalid (retracted/replaced) and
      full_text_unavailable.
    permissible_values:
      VERIFIED:
        description: Identifier resolves to the intended paper, which supports how it is used, with no soundness concerns
      UNVERIFIED:
        description: Not yet manually checked (the default state)
      WRONG_IDENTIFIER:
        description: Identifier resolves to a DIFFERENT paper than intended (e.g. a PMID pointing to an unrelated article)
      MISCITED:
        description: Paper is correctly identified but does not actually support the claim it is cited for
      DISPUTED:
        description: Correctly cited, but the paper's central claim is contradicted or contested by other evidence
      LOW_QUALITY:
        description: Correctly cited, but methodologically weak or preliminary; treat its conclusions with caution

  FindingReviewStatusEnum:
    description: >-
      Reviewer's assessment of the empirical standing of a specific finding (a statement extracted
      from a reference) in light of other evidence. Unlike ReferenceCorrectnessEnum, which judges a
      whole reference, this applies per finding: a paper may have some findings that stand and
      others that have been overturned. Use superseded_by to point to the reference(s) responsible.
    permissible_values:
      CURRENT:
        description: The finding is consistent with the body of evidence and can be curated from
      CORROBORATED:
        description: The finding is independently supported by additional evidence
      DISPUTED:
        description: The finding is contested or contradicted by other evidence, but not definitively refuted; curate with caution
      OVERTURNED:
        description: The finding has been refuted or superseded by later, stronger evidence and should NOT be curated from; record the overturning reference(s) in superseded_by
      UNVERIFIED:
        description: The finding has not yet been manually assessed (default)

  KnowledgeGapKindEnum:
    description: >-
      The kind of ignorance a knowledge gap represents, which determines who can
      resolve it. A single gap may carry several values when it is a blend.
    permissible_values:
      BIOLOGY:
        description: >-
          Nobody knows; resolvable only by new experiments. This is the unknome and
          the primary target of the Function Knowledge Gaps project.
      CURATION:
        description: >-
          The knowledge exists in the literature but is not yet annotated, or is
          annotated too generically. Resolvable by curation.
      ONTOLOGY:
        description: >-
          The knowledge exists but no GO/ontology term can express it (e.g.
          "structural subunit of complex X", or a novel activity). Resolvable by
          ontology development; usually paired with proposed_new_terms.

  KnowledgeGapAspectEnum:
    description: >-
      Which GO aspect (or pattern) is dark for a knowledge gap. Most "dark" genes
      are not uniformly dark.
    permissible_values:
      MF_DARK:
        description: >-
          Process/location known, molecular mechanism unknown. The most common and
          most insidious case — rich BP/CC make the gene look known. The
          "protein binding" smell lives here.
      BP_DARK:
        description: An activity is known but not what it is for (common in microbial/plant metabolism).
      CC_DARK:
        description: Function known, but where/when unknown.
      WHOLLY_DARK:
        description: Only root terms / IEA / "protein binding" survive review (the deep unknome).
      RESIDUAL_SUBGAP:
        description: >-
          The core function is textbook-solid, but one sharp, load-bearing
          mechanistic hole remains (e.g. an unidentified GEF/GAP, a
          catalysis-independent scaffolding mechanism, an unidentified recruit).
          Easy to miss because the gene looks finished.

  KnowledgeGapStatusEnum:
    description: Lifecycle status of a knowledge gap, tracking progress toward resolution.
    permissible_values:
      OPEN:
        description: Unresolved and not under active, evidence-producing investigation.
      NARROWING:
        description: Under active investigation with emerging but still incomplete evidence.
      CLOSING:
        description: >-
          Function-defining evidence exists (e.g. recent preprints) but has not yet
          propagated to peer-reviewed, GO-curated form.
      RESOLVED:
        description: The gap has been closed; retained for provenance/history.

  PublicationTypeEnum:
    description: >-
      The kind of publication or source a reference is. For PMIDs this is inferred from the
      PubMed publication-type ('PT') metadata; for non-literature references it is inferred
      from the identifier scheme. Used to test hypotheses about which evidence sources
      (primary papers, reviews, abstracts, deep research) suffice for GO annotation review.
    permissible_values:
      PRIMARY_RESEARCH:
        description: An original/primary research article reporting new experimental or observational
          results (PubMed 'Journal Article' without a more specific review/secondary type).
      REVIEW:
        description: A narrative review article that synthesizes prior literature (PubMed PT 'Review').
          Reviews often carry phylogenetic/comparative reasoning and broad functional context.
      SYSTEMATIC_REVIEW:
        description: A systematic review (PubMed PT 'Systematic Review').
      META_ANALYSIS:
        description: A meta-analysis combining results across studies (PubMed PT 'Meta-Analysis').
      COMMENT_EDITORIAL:
        description: A comment, editorial, letter, or news item (PubMed PT 'Comment', 'Editorial',
          'Letter', 'News').
      CASE_REPORT:
        description: A clinical case report (PubMed PT 'Case Reports').
      PREPRINT:
        description: A preprint or other not-yet-peer-reviewed manuscript (PubMed PT 'Preprint').
      DATABASE:
        description: A database record or curated method reference rather than a narrative publication
          (e.g. a GO_REF, Reactome pathway, or UniProt entry).
      BIOINFORMATICS:
        description: A local ad-hoc bioinformatics analysis carried out for this review, referenced
          via a 'file:' identifier (e.g. a bioinformatics RESULTS.md).
      DEEP_RESEARCH:
        description: An AI/LLM deep-research report generated for this review, referenced via a
          'file:' identifier (e.g. GENE-deep-research-PROVIDER.md).
      OTHER:
        description: A publication or source that does not fit the other categories.
      UNKNOWN:
        description: The publication type could not be determined (e.g. PubMed metadata unavailable).

  GoCamReviewStatusEnum:
    description: Progress state of a GO-CAM model review.
    permissible_values:
      DRAFT:
        description: Review started; activities not yet fully assessed.
      IN_PROGRESS:
        description: Some activities reviewed; review ongoing.
      COMPLETE:
        description: All activities reviewed.

  GoCamClaimVerdictEnum:
    description: >-
      Forensic-review verdict for a GO-CAM activity, mirroring the
      OK / UNCERTAIN / WRONG scale used for claim validation: does the asserted
      activity hold up against the cited evidence and GO-CAM best practice?
    permissible_values:
      OK:
        description: >-
          The activity is well supported and follows best practice (correct MF
          specificity, correct causal/has-input usage, adequate evidence).
      UNCERTAIN:
        description: >-
          Defensible but imprecise or under-supported; a minor best-practice or
          evidence issue that needs qualification.
      WRONG:
        description: >-
          The activity contradicts the evidence or violates a hard best-practice
          rule (e.g. binding-as-function, wrong causal directionality).

  GoCamConsistencyEnum:
    description: >-
      How a GO-CAM activity relates to the corresponding gene's annotation review
      (genes/**/<gene>-ai-review.yaml).
    permissible_values:
      CONSISTENT:
        description: The activity's function matches an accepted core function in the gene review.
      MORE_SPECIFIC:
        description: The activity asserts a more specific function than the gene review.
      MORE_GENERAL:
        description: The activity asserts a more general function than the gene review.
      RELATED:
        description: Same general area but neither a clean subsumption nor a match.
      CONFLICT:
        description: >-
          The activity asserts a function the gene review removed, negated, or
          marked as over-annotated.
      NOT_IN_REVIEW:
        description: The function is not represented in the gene review (candidate gap).
      NO_GENE_REVIEW:
        description: No gene review exists yet for this gene product.

  GoCamQcFlagEnum:
    description: >-
      Specific GO-CAM best-practice issues observed for an activity. Derived from
      the GO-CAM annotation best-practice checklist (see gocams/BEST_PRACTICE.md).
    permissible_values:
      BINDING_AS_FUNCTION:
        description: >-
          Molecular function is a bare 'binding' term with no functional
          consequence specified (use catalytic/receptor/adaptor/sequestering MF).
      GENERIC_MF:
        description: An overly generic MF term is used where a specific child term applies.
      HAS_INPUT_MISUSE:
        description: >-
          'has input' used incorrectly, e.g. a receptor's ligand or a TF's DNA
          instead of the substrate/target gene/downstream effector.
      DIRECT_VS_INDIRECT_CAUSAL:
        description: >-
          Direct regulation asserted for a multi-step (indirect) mechanism, or
          vice versa.
      INCORRECT_DIRECTIONALITY:
        description: Causal edge directionality (subject -> object) appears reversed.
      MISSING_LOCATION:
        description: Activity lacks an 'occurs in' cellular component.
      MISSING_PROCESS:
        description: Activity is not connected to a biological process via 'part of'.
      ORPHAN_ACTIVITY:
        description: Activity has no causal connections to the rest of the model.
      MISSING_EVIDENCE:
        description: An activity or relationship lacks an evidence code / reference.
      COMPLEX_SUBUNIT_REPRESENTATION:
        description: >-
          Complex represented with a complex term where a specific active subunit
          is known (or vice versa).
