CoolFace
Datasetpublic

LiteFold/UniProtKB

UniProtKB Processed The aim of the UniProt Knowledgebase (UniProtKB; https://www.uniprot.org/) is to provide users with a comprehensive, high-quality and freely accessible set of protein sequences annotated with functional information. In this publication, we describe ongoing changes to our production pipeline to limit the sequences available in UniProtKB to high-quality, non-redundant reference proteomes. We continue to manually curate the scientific literature to add the… See the full description on the dataset page: https://huggingface.co/datasets/LiteFold/UniProtKB.

sourceHugging Facecc-by-4.0updated 4mo agoView on Hugging Face
0likes60downloads
dataset_summary.json72 linesDownload Raw Back to root
1{2  "source": "LiteFold/UniProtKB",3  "source_sha": "ea4b6633410d5e2158cdb0b97fdcabd70bce3c27",4  "viewer_table_scope": "file/table shard index",5  "data_format": "jsonl.gz",6  "dataset_id": "uniprotkb",7  "source_count": 3,8  "records_total": 203172274,9  "residues_total": 75747523712,10  "total_sequence_shards": 205,11  "protein_entry_table_shards": 615,12  "index_rows": 830,13  "sequence_shard_rows": 205,14  "sequence_shard_bytes": 46504287641,15  "metadata_records_bytes": 74373082266,16  "protein_entry_table_bytes": 18549213567,17  "protein_entry_split_counts": {18    "test": 20314776,19    "train": 162548965,20    "validation": 2030853321  },22  "splits": {23    "train": 733,24    "test": 9725  },26  "split_strategy": "default index uses deterministic sha256(file_id) % 10; bucket 0 is test, buckets 1-9 are train",27  "protein_entry_split_strategy": "uniprotkb_exact_sequence_hash_v1",28  "role_counts": {29    "git_attributes": 1,30    "readme": 1,31    "aggregate_manifest": 1,32    "postprocess_manifest": 1,33    "source_manifest": 3,34    "metadata_records": 3,35    "sequence_shard": 205,36    "protein_entry_table_shard": 61537  },38  "source_set_index_counts": {39    "sprot": 6,40    "sprot_varsplic": 6,41    "trembl": 81442  },43  "columns": [44    "file_id",45    "repo_id",46    "source_sha",47    "dataset_id",48    "source_set",49    "source_slug",50    "source_file",51    "path",52    "role",53    "table_split",54    "shard_index",55    "size_bytes",56    "compression",57    "records_in_source",58    "residues_in_source",59    "shards_in_source",60    "records_in_table_split",61    "records_total",62    "residues_total",63    "total_sequence_shards",64    "is_sequence_shard",65    "is_table_shard",66    "is_metadata_records",67    "download_pattern",68    "access_note",69    "split_bucket"70  ]71}72