{"database": "metadata", "table": "run_metadata", "rows": [[19228, "ERR14935021", "ERX14339263", "ERS24360064", "ERP166068", "PRJEB82358", "Human gene duplications", "53f6d331-4985-4b69-8ac3-a57d010118dd", "Other", "Duplicated genes expanded in the human lineage likely contributed to brain evolution  yet challenges exist in their discovery due to sequence assembly errors. We used a complete telomere to telomere genome sequence to identify 213 human specific gene families. From these  362 paralogs were found in all modern human genomes tested and brain transcriptomes  making them top candidates contributing to human universal brain features. Choosing a subset of paralogs  we used long read DNA sequencing of hundreds of modern humans to reveal previously hidden signatures of selection. To understand their roles in brain development  we generated zebrafish CRISPR \u201cknockout\u201d models of nine orthologs and introduced mRNA encoding paralogs  effectively \u201chumanizing\u201d larvae. Our findings implicate two new genes in possibly contributing to hallmark features of the human brain: GPR89B in dosage mediated brain expansion and FRMPD2B in altered synapse signaling. Our holistic approach provides new insights and a comprehensive resource for studying gene expansion drivers of human brain evolution.", "ENA STATUS ID:4|ENA FIRST PUBLIC:2025 05 23|ENA LAST UPDATE:2025 05 23", null, "Single cell RNA sequencing of zebrafish heads from well E07", "Humanized Plate 1   Well E07", "SAMEA118224987", "UNIVERSITY OF CALIFORNIA - DAVIS", "ENA first public:2025 05 23|INSDC center name:UNIVERSITY OF CALIFORNIA   DAVIS|INSDC status:public|Submitter Id:Sample HUM1 E07|collection date:2022 11 14|common name:zebrafish|dev stage:72 hpf|geographic location country and/or sea:USA|sample name:Sample HUM1 E07|scientific name:Danio rerio", null, null, null, null, null, null, null, null, "Raw reads: Sample HUM1 E07", "webin reads Sample HUM1 E07", null, "unspecified", null, "ENA STATUS ID:4", "RNA-Seq", "TRANSCRIPTOMIC SINGLE CELL", "cDNA", "PAIRED", "ILLUMINA", "Illumina NovaSeq 6000", null, "ERP166068", "Raw reads: Sample HUM1 E07", "ENA STATUS ID:4|ENA FIRST PUBLIC:2025 05 23|ENA LAST UPDATE:2025 05 23", "E07.R1.fastq.gz E07.R2.fastq.gz", "fastq fastq", 274863600.0, 916212.0, "webin reads Sample HUM1 E07", "0:150 1:150", "A:89634361;C:31803602;G:65721376;T:87702420;N:1841", 150, 150, null, null, 89634361, 31803602, 65721376, 87702420, 1841, "ERX14339263", "ERS24360064", "ERA33112659", "UNIVERSITY OF CALIFORNIA - DAVIS|European Nucleotide Archive", "UNIVERSITY OF CALIFORNIA - DAVIS", null, null, null, null, null, null, null, null, null, null, null, "B", "B", "mate1-mate2 similar by mapping diff", "illumina", "novaseq_era", "unknown", "poly_a", "unknown", "sc", "single_cell_generic", "generic-scrnaseq-only", null, "United States", "2025-05-23", "Larval", "Larval", "Head", "Nervous System"]], "columns": ["rowid", "run.accession", "experiment.accession", "sample.accession", "study.accession", "bioproject", "study.title", "study.alias", "study.type", "study.abstract", "study.attributes", "study.PMIDs", "sample.description", "sample.title", "sample.alias", "sample.centername", "sample.attributes", "GEOsample.title", "GEOsample.dataprocessing", "GEOsample.source", "GEOsample.treatmentprotocol", "GEOsample.extractprotocol", "GEOsample.growthprotocol", "GEOsample.characteristics", "GEOsample.accession", "experiment.title", "experiment.alias", "experiment.library_name", "experiment.design_description", "experiment.library_construction_protocol", "experiment.attributes", "experiment.library_strategy", "experiment.library_source", "experiment.library_selection", "experiment.library_layout", "experiment.platform", "experiment.instrument_model", "experiment.spot_descriptor", "experiment.study_ref", "run.title", "run.attributes", "run.filename", "run.semantic_name", "run.total_bases", "run.total_spots", "run.alias", "run.read_lengths", "run.base_counts", "run.r1_length", "run.r2_length", "run.r3_length", "run.r4_length", "run.Acount", "run.Ccount", "run.Gcount", "run.Tcount", "run.Ncount", "run.experiment", "run.pool_member", "submission.accession", "submission.srasource", "submission.bioprojectsource", "seqdetective.n_mates", "seqdetective.mapping_rate.mate1", "seqdetective.mapping_rate.mate2", "seqdetective.nofeature_rate.mate1", "seqdetective.nofeature_rate.mate2", "seqdetective.sparsity.mate1", "seqdetective.sparsity.mate2", "seqdetective.pos_strand_rate.mate1", "seqdetective.pos_strand_rate.mate2", "seqdetective.readlen.mate1", "seqdetective.readlen.mate2", "seqdetective.judgement.mate1", "seqdetective.judgement.mate2", "seqdetective.judgement.reason", "platform_family", "instrument_generation", "read_bias", "selection_class", "prep_kit", "sc_or_bulk", "tech_class", "technology", "tech_variant", "submission.bioprojectsource.country", "earliest_date", "devstage_curation", "devstage_curation_coarse", "tissue_curation", "tissue_curation_coarse"], "primary_keys": ["rowid"], "primary_key_values": ["19228"], "units": {}, "query_ms": 12.23570100046345}