@prefix ex: <https://sparql.uniprot.org/.well-known/sparql-examples/> .
@prefix rdfs: <http://www.w3.org/2000/01/rdf-schema#> .
@prefix schema: <https://schema.org/> .
@prefix sh: <http://www.w3.org/ns/shacl#> .

ex:219_human_gene_count_vs_isoform_count a sh:SPARQLExecutable,
        sh:SPARQLSelectExecutable ;
    rdfs:comment "Compare the number of reviewed human UniProtKB entries (one canonical accession per gene) to the number of distinct sequence resources those entries carry (the canonical sequence plus every annotated splice isoform), showing directly why searching human in UniProt returns far more sequences than there are protein-coding genes — the full picture also includes unreviewed (TrEMBL) entries on top of this. This is also due to there being more than one 'genome' source (proteome)."@en ;
    sh:prefixes _:sparql_examples_prefixes ;
    sh:select """PREFIX up: <http://purl.uniprot.org/core/>
PREFIX taxon: <http://purl.uniprot.org/taxonomy/>

SELECT
  (COUNT(DISTINCT ?gene) AS ?reviewedGenes)
  (COUNT(DISTINCT ?isoformSequence) AS ?distinctSequences)
WHERE {
  ?protein a up:Protein ;
    up:organism taxon:9606 ;
    up:reviewed true ;
    up:sequence ?isoformSequence .
  OPTIONAL { ?protein up:encodedBy ?gene }
}""" ;
    schema:keywords "isoform" , "human genome" , "gene count" ;
    schema:target <https://sparql.uniprot.org/sparql/> .
