218_protein_amino_acid_composition
- amino acid
- sequence
- composition
Count how many times each of the 20 standard amino acids occurs in human TP53's (P04637) canonical sequence, using STRLEN(sequence) minus STRLEN(sequence with that residue letter removed) as an amino-acid-counting trick rather than a per-position scan. Note: for performance over a larger set rewrite this as a huge union.
Use at
PREFIX up: <http://purl.uniprot.org/core/>
PREFIX rdf: <http://www.w3.org/1999/02/22-rdf-syntax-ns#>
SELECT
?residue
?count
WHERE {
<http://purl.uniprot.org/uniprot/P04637> up:sequence ?sequence .
?sequence a up:Simple_Sequence ;
rdf:value ?seq .
VALUES ?residue { "A" "R" "N" "D" "C" "Q" "E" "G" "H" "I" "L" "K" "M" "F" "P" "S" "T" "W" "Y" "V" }
BIND(STRLEN(?seq) - STRLEN(REPLACE(?seq, ?residue, "")) AS ?count)
}
ORDER BY DESC(?count)
graph TD
classDef projected fill:lightgreen;
classDef literal fill:orange;
classDef iri fill:yellow;
v5("?count"):::projected
v4("?residue"):::projected
v3("?seq")
v2("?sequence")
c4(["up:Simple_Sequence"]):::iri
c1([http://purl.uniprot.org/uniprot/P04637]):::iri
c1 --"up:sequence"--> v2
v2 --"a"--> c4
v2 --"rdf:value"--> v3
bind0[/VALUES ?residue/]
bind0-->v4
bind00(["A"])
bind00 --> bind0
bind01(["R"])
bind01 --> bind0
bind02(["N"])
bind02 --> bind0
bind03(["D"])
bind03 --> bind0
bind04(["C"])
bind04 --> bind0
bind05(["Q"])
bind05 --> bind0
bind06(["E"])
bind06 --> bind0
bind07(["G"])
bind07 --> bind0
bind08(["H"])
bind08 --> bind0
bind09(["I"])
bind09 --> bind0
bind010(["L"])
bind010 --> bind0
bind011(["K"])
bind011 --> bind0
bind012(["M"])
bind012 --> bind0
bind013(["F"])
bind013 --> bind0
bind014(["P"])
bind014 --> bind0
bind015(["S"])
bind015 --> bind0
bind016(["T"])
bind016 --> bind0
bind017(["W"])
bind017 --> bind0
bind018(["Y"])
bind018 --> bind0
bind019(["V"])
bind019 --> bind0
bind1[/"string-length(?seq) - string-length(replace(?seq,?residue,''))"/]
v3 --o bind1
v4 --o bind1
bind1 --as--o v5