diff --git a/vignettes/get_projects.R b/vignettes/get_projects.R new file mode 100644 index 0000000..d2cf7f4 --- /dev/null +++ b/vignettes/get_projects.R @@ -0,0 +1,39 @@ +library(GenomicDataCommons) + + +#more than one possibility to get availble projects on the gdc portal +#perhaps useful for anyone searching for gdc data to combine with existing laboratory samples +gdc_proj<-projects() + +#get available keys to build queries +#available_fields(gdc_proj) +#facet(gdc_proj) + +####get available gdc projects +res_1<-(gdc_proj |> facet('project_id') |> aggregations() ) +#save to file +write.csv(res$project_id$key,file="project_id.txt",quote=FALSE,row.names=FALSE) + +#get gdc project names +res_2<-(gdc_proj |> facet('program.name') |> aggregations() ) +#Save to file +write.csv(res$program.name$key,file=='program.name.txt',quote=FALSE,row.names=FALSE) + +#addtional exp details +res_3<-(gdc_proj |> facet('name') |> aggregations() ) +#get list components +#names(res) +write.csv(res$name["key"],file=='project_name.txt',quote=FALSE,row.names=FALSE) + +#get biospecimen, tissue site +res_4<-(gdc_proj |> facet('primary_site') |> aggregations() ) + +write.csv(res$primary_site['key'],file=='primary_site.txt',quote=FALSE,row.names=FALSE) + +####use multiple fields +res_5<-(gdc_proj |> facet(gdc_proj$fields[c(4:5,7)]) |> aggregations() ) + +x<-list(N=res$name["key"],P=res$project_id["key"],T=res$primary_site["key"]) + +#save to file , not a 1-1 map +#cat(capture.output(print(x), file="test.txt")) diff --git a/vignettes/mmrf_commpass_wgs.R b/vignettes/mmrf_commpass_wgs.R new file mode 100644 index 0000000..8624de2 --- /dev/null +++ b/vignettes/mmrf_commpass_wgs.R @@ -0,0 +1,34 @@ + + +#building queries + +#get available info for project MMRF-COMMPASS +q_1 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'MMRF-COMMPASS') |> manifest() + +#save info to file for future use +write.csv(q_1,file="out.txt",quote=FALE,row.names=FALSE) + +#list only open access entries +q_2 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'MMRF-COMMPASS') |> filter(access == "open")|> manifest() + +# restrict data to WGS + + +q_3=files() |> GenomicDataCommons::filter(cases.project.project_id == 'MMRF-COMMPASS') |> filter(access == "open") |> filter(experimental_strategy == "WGS") |> manifest() + + +#restrict data to WGS Copy Number Segment +q_4=files() |> GenomicDataCommons::filter(cases.project.project_id == 'MMRF-COMMPASS') |> filter(access == "open") |> filter(experimental_strategy == "WGS" & data_type == 'Copy Number Segment') |> manifest() + + +#get files less than size 32000 + +q_5=files() |> GenomicDataCommons::filter(cases.project.project_id == 'MMRF-COMMPASS') |> filter(access == "open") |> filter(experimental_strategy == "WGS" & data_type == 'Copy Number Segment' & file_size < 32000) |> manifest() + +#get the files +for(i in seq(1,dim(q_5)[1])) +{ +uid<-0; +uid<-as.character(q_6[i,"id"]) +gdcdata(uid) +} diff --git a/vignettes/mmrf_commpass_wxs.R b/vignettes/mmrf_commpass_wxs.R new file mode 100644 index 0000000..b04e128 --- /dev/null +++ b/vignettes/mmrf_commpass_wxs.R @@ -0,0 +1,34 @@ + + +#building queries + +#get available info for project MMRF-COMMPASS +q_1 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'MMRF-COMMPASS') |> manifest() + +#save info to file for future use +write.csv(q_3,file="out.txt",quote=FALE,row.names=FALSE) + +#list only open access entries +q_2 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'MMRF-COMMPASS') |> filter(access == "open")|> manifest() + +# restrict data to WXS + + +q_3=files() |> GenomicDataCommons::filter(cases.project.project_id == 'MMRF-COMMPASS') |> filter(access == "open") |> filter(experimental_strategy == "WXS") |> manifest() + + +#restrict data to WXS Masked Somatic Mutation +q_4=files() |> GenomicDataCommons::filter(cases.project.project_id == 'MMRF-COMMPASS') |> filter(access == "open") |> filter(experimental_strategy == "WXS" & data_type == 'Masked Somatic Mutation') |> manifest() + + +#list files by timestamp + +q_5=files() |> GenomicDataCommons::filter(cases.project.project_id == 'MMRF-COMMPASS') |> filter(access == "open") |> filter(experimental_strategy == "WXS" & data_type == 'Masked Somatic Mutation' & created_datetime < "2022-08-22T10:48:06.705961-05:00" ) |> manifest() + +#get the files +for(i in seq(1,dim(q_5)[1])) +{ +uid<-0; +uid<-as.character(q_5[i,"id"]) +gdcdata(uid) +} diff --git a/vignettes/overview.Rmd b/vignettes/overview.Rmd index 17a1114..4269cda 100644 --- a/vignettes/overview.Rmd +++ b/vignettes/overview.Rmd @@ -118,6 +118,12 @@ stopifnot(GenomicDataCommons::status()$status=="OK") ## Find data +Please use a fresh R session to minimise package conflicts arising from identical function names. +For example the functions filter(), results() are available with R packages dplyr, GenomicDataCommons . Instead consider using the R package conflicted. +Example usage +conflicts_prefer(dplyr::filter()) +conflicts_prefer(GenomicDataCommons::filter) +conflicts_prefer(stats :: filter) The following code builds a `manifest` that can be used to guide the download of raw data. Here, filtering finds gene expression files @@ -164,8 +170,8 @@ the `gdc_clinical()` function will return a list of four `tibble`s. - main ```{r gdc_clinical} -case_ids = cases() |> results(size=10) |> ids() -clindat = gdc_clinical(case_ids) +case_ids <- cases() |> results(size=10) |> ids() +clindat <- gdc_clinical(case_ids) names(clindat) ``` @@ -183,9 +189,9 @@ be all that is needed, but the API and `r Biocpkg("GenomicDataCommons")` package make much flexibility if fine-tuning is required. ```{r metadataQS} -expands = c("diagnoses","annotations", +expands <- c("diagnoses","annotations", "demographic","exposures") -clinResults = cases() |> +clinResults <- cases() |> GenomicDataCommons::select(NULL) |> GenomicDataCommons::expand(expands) |> results(size=50) @@ -287,7 +293,7 @@ functions that each create `GDCQuery` objects (actually, specific subclasses of - `annotations()` ```{r projectquery} -pquery = projects() +pquery <- projects() ``` The `pquery` object is now an object of (S3) class, `GDCQuery` (and @@ -326,16 +332,16 @@ Note that we have not set any filters, so a `count()` here will represent all the project records publicly available at the GDC in the "default" archive" ```{r pquerycount} -pcount = count(pquery) +pcount <- count(pquery) # or -pcount = pquery |> count() +pcount <- pquery |> count() pcount ``` The `results()` method will fetch actual results. ```{r pqueryresults} -presults = pquery |> results() +presults <- pquery |> results() ``` These results are returned from the GDC in [JSON](http://www.json.org/) format and @@ -356,7 +362,7 @@ size is probably warranted before calling `results_all()` ```{r presultsall} length(ids(presults)) -presults = pquery |> results_all() +presults <- pquery |> results_all() length(ids(presults)) # includes all records length(ids(presults)) == count(pquery) @@ -401,11 +407,11 @@ the dplyr `select()` verb that limits from already-present fields. We ```{r selectexample} # Default fields here -qcases = cases() +qcases <- cases() qcases$fields # set up query to use ALL available fields # Note that checking of fields is done by select() -qcases = cases() |> GenomicDataCommons::select(available_fields('cases')) +qcases <- cases() |> GenomicDataCommons::select(available_fields('cases')) head(qcases$fields) ``` @@ -428,7 +434,7 @@ tibbles). ```{r aggexample} # total number of files of a specific type -res = files() |> facet(c('type','data_type')) |> aggregations() +res <- files() |> facet(c('type','data_type')) |> aggregations() res$type ``` @@ -454,7 +460,7 @@ For the user, these details will not be too important except to note that a filter expression must begin with a "~". ```{r allfilesunfiltered} -qfiles = files() +qfiles <- files() qfiles |> count() # all files ``` To limit the file type, we can refer back to the @@ -463,7 +469,7 @@ the file field "type". For example, to filter file results to only "gene_expression" files, we simply specify a filter. ```{r onlyGeneExpression} -qfiles = files() |> filter( type == 'gene_expression') +qfiles <- files() |> filter( type == 'gene_expression') # here is what the filter looks like after translation str(get_filter(qfiles)) ``` @@ -495,7 +501,7 @@ note that `TCGA-OV` is the correct project_id, not `TCGA-OVCA`. Note that filter and does not build on any previous filters*. ```{r filtfinal} -qfiles = files() |> +qfiles <- files() |> filter( cases.project.project_id == 'TCGA-OV' & type == 'gene_expression') str(get_filter(qfiles)) qfiles |> count() @@ -507,7 +513,7 @@ accomplish the same effect as multiple `&` conditionals. The `count()` below is equivalent to the `&` filtering done above. ```{r filtChain} -qfiles2 = files() |> +qfiles2 <- files() |> filter( cases.project.project_id == 'TCGA-OV') |> filter( type == 'gene_expression') qfiles2 |> count() @@ -520,7 +526,7 @@ Generating a manifest for bulk downloads is as simple as asking for the manifest from the current query. ```{r filtAndManifest} -manifest_df = qfiles |> manifest() +manifest_df <- qfiles |> manifest() head(manifest_df) ``` @@ -531,11 +537,11 @@ that the field "analysis.workflow_type" has the appropriate filter criteria. ```{r filterForSTARCounts} -qfiles = files() |> filter( ~ cases.project.project_id == 'TCGA-OV' & +qfiles <- files() |> filter( ~ cases.project.project_id == 'TCGA-OV' & type == 'gene_expression' & access == "open" & analysis.workflow_type == 'STAR - Counts') -manifest_df = qfiles |> manifest() +manifest_df <- qfiles |> manifest() nrow(manifest_df) ``` @@ -577,7 +583,7 @@ be stored in one of three ways (resolved in this order): As a concrete example: ```{r authenNoRun, eval=FALSE} -token = gdc_token() +token <- gdc_token() transfer(...,token=token) # or transfer(...,token=get_token()) @@ -594,7 +600,7 @@ ids. A simple way of producing such a vector is to produce a contain file ids. ```{r singlefileDL} -fnames = gdcdata(manifest_df$id[1:2],progress=FALSE) +fnames <- gdcdata(manifest_df$id[1:2],progress=FALSE) ``` @@ -614,7 +620,7 @@ in parallel. ```{r bulkDL, eval=FALSE} # Requires gcd_client command-line utility to be isntalled # separately. -fnames = gdcdata(manifest_df$id[3:10], access_method = 'client') +fnames <- gdcdata(manifest_df$id[3:10], access_method = 'client') ``` @@ -627,7 +633,7 @@ fnames = gdcdata(manifest_df$id[3:10], access_method = 'client') ### How many cases are there per project_id? ```{r casesPerProject} -res = cases() |> facet("project.project_id") |> aggregations() +res <- cases() |> facet("project.project_id") |> aggregations() head(res) library(ggplot2) ggplot(res$project.project_id,aes(x = key, y = doc_count)) + @@ -653,7 +659,7 @@ cases() |> filter(~ project.program.name=='TCGA') |> count() # The need to do the "&" here is a requirement of the # current version of the GDC API. I have filed a feature # request to remove this requirement. -resp = cases() |> filter(~ project.project_id=='TCGA-BRCA' & +resp <- cases() |> filter(~ project.project_id=='TCGA-BRCA' & project.project_id=='TCGA-BRCA' ) |> facet('samples.sample_type') |> aggregations() resp$samples.sample_type @@ -665,12 +671,12 @@ resp$samples.sample_type # The need to do the "&" here is a requirement of the # current version of the GDC API. I have filed a feature # request to remove this requirement. -resp = cases() |> filter(~ project.project_id=='TCGA-BRCA' & +resp <- cases() |> filter(~ project.project_id=='TCGA-BRCA' & samples.sample_type=='Solid Tissue Normal') |> GenomicDataCommons::select(c(default_fields(cases()),'samples.sample_type')) |> response_all() count(resp) -res = resp |> results() +res <- resp |> results() str(res[1],list.len=6) head(ids(resp)) ``` @@ -721,7 +727,7 @@ cases() |> ### How many of each type of file are available? ```{r filesVCFCount} -res = files() |> facet('type') |> aggregations() +res <- files() |> facet('type') |> aggregations() res$type ggplot(res$type,aes(x = key,y = doc_count)) + geom_bar(stat='identity') + theme(axis.text.x = element_text(angle = 45, hjust = 1)) @@ -730,13 +736,13 @@ ggplot(res$type,aes(x = key,y = doc_count)) + geom_bar(stat='identity') + ### Find gene-level RNA-seq quantification files for GBM ```{r filesRNAseqGeneGBM} -q = files() |> +q <- files() |> GenomicDataCommons::select(available_fields('files')) |> filter(~ cases.project.project_id=='TCGA-GBM' & data_type=='Gene Expression Quantification') q |> facet('analysis.workflow_type') |> aggregations() # so need to add another filter -file_ids = q |> filter(~ cases.project.project_id=='TCGA-GBM' & +file_ids <- q |> filter(~ cases.project.project_id=='TCGA-GBM' & data_type=='Gene Expression Quantification' & analysis.workflow_type == 'STAR - Counts') |> GenomicDataCommons::select('file_id') |> @@ -752,20 +758,20 @@ file_ids = q |> filter(~ cases.project.project_id=='TCGA-GBM' & and for vignette building**. ```{r filesRNAseqGeneGBMforBAM} -q = files() |> +q <- files() |> GenomicDataCommons::select(available_fields('files')) |> filter(~ cases.project.project_id == 'TCGA-GBM' & data_type == 'Aligned Reads' & experimental_strategy == 'RNA-Seq' & data_format == 'BAM') -file_ids = q |> response_all() |> ids() +file_ids <- q |> response_all() |> ids() ``` ```{r slicing10, eval=FALSE} -bamfile = slicing(file_ids[1],regions="chr12:6534405-6538375",token=gdc_token()) +bamfile <- slicing(file_ids[1],regions="chr12:6534405-6538375",token=gdc_token()) library(GenomicAlignments) -aligns = readGAlignments(bamfile) +aligns <- readGAlignments(bamfile) ``` # Troubleshooting diff --git a/vignettes/questions-and-answers.Rmd b/vignettes/questions-and-answers.Rmd index 67fcf74..d8649d6 100644 --- a/vignettes/questions-and-answers.Rmd +++ b/vignettes/questions-and-answers.Rmd @@ -29,7 +29,7 @@ changes to lazy evaluation best practices. There is now no need to include the `~` in the filter expression. So: ```{r} -q = files() |> +q <- files() |> GenomicDataCommons::filter( cases.project.project_id == 'TCGA-COAD' & data_type == 'Aligned Reads' & @@ -51,7 +51,7 @@ manifest(q) Your question about race and ethnicity is a good one. ```{r} -all_fields = available_fields(files()) +all_fields <- available_fields(files()) ``` And we can grep for `race` or `ethnic` to get potential matching fields @@ -72,7 +72,7 @@ available_values('files',"cases.demographic.race") We can complete our filter expression now to limit to `white` race only. ```{r} -q_white_only = q |> +q_white_only <- q |> GenomicDataCommons::filter(cases.demographic.race=='white') count(q_white_only) manifest(q_white_only) diff --git a/vignettes/somatic_mutations.Rmd b/vignettes/somatic_mutations.Rmd index 4268c16..1c6c975 100644 --- a/vignettes/somatic_mutations.Rmd +++ b/vignettes/somatic_mutations.Rmd @@ -38,7 +38,7 @@ head(available_values('genes','symbol')) ```{r} -tp53 = genes() |> +tp53 <- genes() |> GenomicDataCommons::filter(symbol=='TP53') |> results(size=10000) |> as_tibble() @@ -67,7 +67,7 @@ ssms() |> ```{r warning=FALSE,message=FALSE} library(VariantAnnotation) -vars = ssms() |> +vars <- ssms() |> GenomicDataCommons::filter( consequence.transcript.gene.symbol %in% c('TP53')) |> GenomicDataCommons::results_all() |> @@ -75,7 +75,7 @@ vars = ssms() |> ``` ```{r} -vr = VRanges(seqnames = vars$chromosome, +vr <- VRanges(seqnames = vars$chromosome, ranges = IRanges(start=vars$start_position, width=1), ref = vars$reference_allele, alt = vars$tumor_allele) @@ -89,7 +89,7 @@ ssm_occurrences() |> ``` ```{r} -var_samples = ssm_occurrences() |> +var_samples <- ssm_occurrences() |> GenomicDataCommons::filter( ssm.consequence.transcript.gene.symbol %in% c('TP53')) |> GenomicDataCommons::expand(c('case', 'ssm', 'case.project')) |> @@ -119,7 +119,7 @@ fnames <- files() |> ```{r cache=TRUE} library(maftools) -melanoma = read.maf(maf = fnames) +melanoma <- read.maf(maf = fnames) ``` ```{r} diff --git a/vignettes/ssm_examples.R b/vignettes/ssm_examples.R new file mode 100644 index 0000000..0d6f225 --- /dev/null +++ b/vignettes/ssm_examples.R @@ -0,0 +1,51 @@ +library(dplyr) +library(tibble) +library(GenomicDataCommons) + +#build queries to get simple somatic mutations +q_primsite_1 = ssm_occurrences() |> GenomicDataCommons::filter(case.primary_site== 'Esophagus') |> GenomicDataCommons::expand(c('ssm','case')) |> GenomicDataCommons::results_all() |> as_tibble() + +#add search criteria +q_primsite_2 = ssm_occurrences() |> GenomicDataCommons::filter(case.primary_site== 'Esophagus' & case.lost_to_followup == 'No' ) |> GenomicDataCommons::expand(c('ssm','case')) |> GenomicDataCommons::results_all() |> as_tibble() + +################ +q_consent = ssm_occurrences() |> GenomicDataCommons::filter(case.consent_type== 'Consent by Death') |> GenomicDataCommons::expand(c('ssm','case')) |> GenomicDataCommons::results_all() |> as_tibble() + +write.csv(q_consent,file="ssm_consentbyDeath.txt",row.names=FALSE,quote=FALSE) + +q_del = ssm_occurrences() |> GenomicDataCommons::filter(ssm.mutation_subtype == 'Small deletion') |> GenomicDataCommons::expand(c('ssm')) |> GenomicDataCommons::results_all() |> as_tibble() + +write.csv(q_del,file="ssm_subtypeDeletions.txt",row.names=FALSE,quote=FALSE) + +#get simple somatic mutations on chr one +q_chrone = ssm_occurrences() |> GenomicDataCommons::filter(ssm.chromosome == 'chr1') |> GenomicDataCommons::expand(c('ssm')) |> GenomicDataCommons::results_all() |> as_tibble() + +write.csv(q_chrone,file="simplesomaticmutschr1.txt",row.names=FALSE,quote=FALSE) + +######################------------------------------################################ +####get all simple somatic mutations on chr 17 +q_1 = ssm_occurrences() |> GenomicDataCommons::filter(ssm.chromosome == 'chr17') |> GenomicDataCommons::expand(c('ssm')) |> GenomicDataCommons::results_all() |> as_tibble() + +write.csv(q_1,file="simplesomaticmutschr17.txt",row.names=FALSE,quote=FALSE) + + +#pull mutations on chr17 for TP53 + +q_2 = ssm_occurrences() |> GenomicDataCommons::filter(ssm.consequence.transcript.gene.symbol == 'TP53' & ssm.chromosome == 'chr17') |> GenomicDataCommons::expand(c('case','ssm')) |> GenomicDataCommons::results_all() |> as_tibble() + +write.csv(q_2,file="TP53var_samples17.txt",row.names=FALSE,quote=FALSE) +#sanity check: TP53 gene located on chr17 , so same results +q_3 = ssm_occurrences() |> GenomicDataCommons::filter(ssm.consequence.transcript.gene.symbol == 'TP53') |> GenomicDataCommons::expand(c('case','ssm')) |> GenomicDataCommons::results_all() |> as_tibble() + +write.csv(q_3,file="TP53var_samples.txt",row.names=FALSE,quote=FALSE) + + +#pull mutations on chr17 for NF1 +q_4 = ssm_occurrences() |> GenomicDataCommons::filter(ssm.consequence.transcript.gene.symbol == 'NF1' & ssm.chromosome == 'chr17') |> GenomicDataCommons::expand(c('case','ssm')) |> GenomicDataCommons::results_all() |> as_tibble() + +write.csv(q_4,file="NF1var_samples17.txt",row.names=FALSE,quote=FALSE) + +#sanity check : NF1 gene located on chr17, so same results +q_5 = ssm_occurrences() |> GenomicDataCommons::filter(ssm.consequence.transcript.gene.symbol == 'NF1') |> GenomicDataCommons::expand(c('case','ssm')) |> GenomicDataCommons::results_all() |> as_tibble() +write.csv(q_5,file="NF1var_samples.txt",row.names=FALSE,quote=FALSE) + diff --git a/vignettes/target_cn.R b/vignettes/target_cn.R new file mode 100644 index 0000000..547a681 --- /dev/null +++ b/vignettes/target_cn.R @@ -0,0 +1,21 @@ +library(GenomicDataCommons) +library(dplyr) +library(tibble) +#Clear Cell Sarcoma of the Kidney. Disease Type. Complex Mixed and Stromal Neoplasms. Primary Site. Kidney. Program. TARGET + + +cn_manifest_CCSK <- files() |> GenomicDataCommons::filter( cases.project.project_id == 'TARGET-CCSK') |> GenomicDataCommons::filter( analysis.workflow_type == 'ASCAT2') |> manifest() +n<-which(cn_manifest_CCSK$access=='open') + +#save dat to file +write.csv(cn_manifest_CCSK[n,],file='out.txt',quote=FALSE,row.names=FALSE) +##########----------####################### +#select files +for( i in seq(1,length(n))) +{ +# +id<-0; + id<-as.character(cn_manifest_CCSK[n[i],1]) +###### +gdcdata(id) +} diff --git a/vignettes/target_methylation.R b/vignettes/target_methylation.R new file mode 100644 index 0000000..1f54692 --- /dev/null +++ b/vignettes/target_methylation.R @@ -0,0 +1,21 @@ +library(GenomicDataCommons) +library(dplyr) +library(tibble) +#Clear Cell Sarcoma of the Kidney. Disease Type. Complex Mixed and Stromal Neoplasms. Primary Site. Kidney. Program. TARGET + +#get familiar with aggregations +q_2 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TARGET-CCSK') |> facet(c("data_type","data_format","analysis.workflow_type","file_id","file_name")) |> aggregations() + + + +q_1 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TARGET-CCSK') |> facet(c("data_type","data_format","analysis.workflow_type")) |> aggregations() + +idat_manifest_CCSK <- files() |> GenomicDataCommons::filter( cases.project.project_id == 'TARGET-CCSK') |> GenomicDataCommons::filter( data_type == 'Methylation Beta Value' ) |> GenomicDataCommons::filter( analysis.workflow_type == 'SeSAMe Methylation Beta Estimation') |> manifest() +n<-which(idat_manifest_CCSK$access=='open') + +#save data to file +write.csv(idat_manifest_CCSK[n,],file='manifest.txt',quote=FALSE,row.names=FALSE) + + +###### test single file +gdcdata(idat_manifest_CCSK[n[1],1])) diff --git a/vignettes/target_methylation_1.R b/vignettes/target_methylation_1.R new file mode 100644 index 0000000..2c091c4 --- /dev/null +++ b/vignettes/target_methylation_1.R @@ -0,0 +1,17 @@ +library(GenomicDataCommons) +library(dplyr) +library(tibble) + + +idat_manifest_CCSK <- files() |> GenomicDataCommons::filter( cases.project.project_id == 'TARGET-CCSK') |> GenomicDataCommons::filter( data_type == 'Methylation Beta Value' ) |> GenomicDataCommons::filter( analysis.workflow_type == 'SeSAMe Methylation Beta Estimation') |> manifest() +n<-which(idat_manifest_CCSK$access=='open') + +##########----------####################### +for( i in seq(1,length(n))) +{ +#use cast +id<-0; + id<-as.character(idat_manifest_CCSK[n[i],1]) +######only test with single file +gdcdata(id) +} diff --git a/vignettes/tcgaTHCA_clinical.R b/vignettes/tcgaTHCA_clinical.R new file mode 100644 index 0000000..6a34221 --- /dev/null +++ b/vignettes/tcgaTHCA_clinical.R @@ -0,0 +1,14 @@ +library(dplyr) +library(tibble) +library(GenomicDataCommons) + +#get files for project TCGA-THCA +#build a manifest file +#use col names with function calls +p_2 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-THCA' & data_category == 'Clinical' & data_type == 'Clinical Supplement' & data_format == 'BCR XML') |> filter(access == "open") |> manifest() +#get select data +for(i in seq(20,24)) +{ +uid<-0;uid<-as.character(p_2[i,1]) +gdcdata(uid) +} diff --git a/vignettes/tcgaTHCA_pathology.R b/vignettes/tcgaTHCA_pathology.R new file mode 100644 index 0000000..6288dd9 --- /dev/null +++ b/vignettes/tcgaTHCA_pathology.R @@ -0,0 +1,33 @@ +library(dplyr) +library(tibble) +library(GenomicDataCommons) + + +# get pathology reports +q_2 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-THCA' & data_category == 'Clinical' & data_type == 'Pathology Report') |> manifest() + +for(i in seq(20,24)) +{ +uid<-0;uid<-as.character(q_2[i,1]) +gdcdata(uid) +} + +#use updated datetime to filter data +q_3 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-THCA' & data_category == 'Clinical' & data_type == 'Pathology Report' & (updated_datetime == "2022-12-06T15:32:24.960627-06:00")) |> manifest() + +#or + +q_4 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-THCA' & data_category == 'Clinical' & data_type == 'Pathology Report' & (updated_datetime > "2022-12-06T15:32:24.960627-06:00")) |> manifest() + +for(i in seq(20,24)) +{ +uid<-0;uid<-as.character(q_4[i,1]) +gdcdata(uid) +} + +#use file size to filter and updated datetime +#single hit +q_5 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-THCA' & data_category == 'Clinical' & data_type == 'Pathology Report' & (updated_datetime == "2022-12-06T15:32:24.960627-06:00") & file_size < 2350000) |> manifest() +#get one file that fits above criteria +gdcdata(q_5[1,"id"]) + diff --git a/vignettes/tcga_CEL.R b/vignettes/tcga_CEL.R new file mode 100644 index 0000000..5cceb52 --- /dev/null +++ b/vignettes/tcga_CEL.R @@ -0,0 +1,28 @@ +library(dplyr) +library(tibble) +library(GenomicDataCommons) + +# + +q_2 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-GBM' & data_category == 'Transcriptome Profiling' & data_format == 'CEL') |> manifest() +#OR use filter fn +q_3 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-GBM' & data_category == 'Transcriptome Profiling' & data_format == 'CEL') |> GenomicDataCommons::filter(access=='OPEN') |> manifest() + +for(i in seq(10,12)) +{ +uid<-0; +uid<-as.character(q_3[i,1]) +gdcdata(uid) +} + + +# +n<-which(q_2["access"] == 'open') +for(j in seq(2,3)) +{ +Id<-0 +Id<-as.character(q_3[n[j],1]) +gdcdata(Id) +} + + diff --git a/vignettes/tcga_proteome.R b/vignettes/tcga_proteome.R new file mode 100644 index 0000000..d4dd688 --- /dev/null +++ b/vignettes/tcga_proteome.R @@ -0,0 +1,29 @@ +library(dplyr) +library(tibble) +library(GenomicDataCommons) + +p_2 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-THCA' & data_type == 'Protein Expression Quantification' & data_format == 'tsv') |> manifest() +#locate open data +open_access<-which(p_2$access=="open") +#381 files +#download select files +#for(i in length(open_access)) +#for(i in seq(3,5)) +#for(i in seq(20,24)) + +for(i in seq(200,202)) +{ +uid<-0;uid<-as.character(p_2[open_access[i],1]) +gdcdata(uid) +} + +#sanity check , use the filter function + + +p_3 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-THCA' & data_type == 'Protein Expression Quantification' & data_format == 'tsv') |> filter(access == 'open') |> manifest() + +for(i in seq(20,24)) +{ +uid<-0;uid<-as.character(p_3[i,1]) +gdcdata(uid) +} diff --git a/vignettes/tcga_slides.R b/vignettes/tcga_slides.R new file mode 100644 index 0000000..f260f3d --- /dev/null +++ b/vignettes/tcga_slides.R @@ -0,0 +1,25 @@ +library(dplyr) +library(tibble) +library(GenomicDataCommons) + +#specify project name +#q_2 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-KIRC' & data_type == 'Slide Image' & data_format == 'svs') |> manifest() +q_2 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-BRCA' & data_type == 'Slide Image' & data_format == 'svs') |> manifest() +#locate open data +open_access<-which(q_2$access=="open") + +#select files +#for(i in length(open_access)) +#for(i in seq(3,5)) +#for(i in seq(20,24)) +for(i in seq(200,202)) +{ +uid<-0;uid<-as.character(q_2[open_access[i],1]) +gdcdata(uid) +} + + +######alternatively, use filter +#both approaches should list same files +q_3 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-BRCA' & data_type == 'Slide Image' & data_format == 'svs') |> filter(access == 'open') |> manifest() + diff --git a/vignettes/tcga_slides_1.R b/vignettes/tcga_slides_1.R new file mode 100644 index 0000000..057547b --- /dev/null +++ b/vignettes/tcga_slides_1.R @@ -0,0 +1,27 @@ +library(dplyr) +library(tibble) +library(GenomicDataCommons) + +q_2 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-COAD' & data_type == 'Slide Image' & data_format == 'svs') |> manifest() +#locate open data +open_access<-which(q_2$access=="open") + +#for(i in length(open_access)) +#for(i in seq(3,5)) +#for(i in seq(20,24)) +for(i in seq(200,202)) +{ +uid<-0;uid<-as.character(q_2[open_access[i],1]) +gdcdata(uid) +} + + +#or sanity check + +q_3 = files() |> GenomicDataCommons::filter(cases.project.project_id == 'TCGA-COAD' & data_type == 'Slide Image' & data_format == 'svs') |> filter(access == 'open') |> manifest() + +for(i in seq(2,4)) +{ +uid<-0;uid<-as.character(q_3[i,1]) +gdcdata(uid) +}