#http://www.htslib.org/workflow/#mapping_to_variant set -xeu FQ1=y1.fastq FQ2=y2.fastq REF=yeast.fasta BNM=yeastD RUNINDOCKER=1 SAMTOOLS=samtools BWA=bwa TABIX=tabix BCFTOOLS=bcftools PLOTVCFSTATS=plot-vcfstats if [[ "$RUNINDOCKER" -eq "1" ]]; then echo "RUNNING IN DOCKER" DRUN="docker run --rm -v $PWD:/data --workdir /data -i" #--user=biodocker SAMTOOLS_IMAGE=biodckr/samtools BWA_IMAGE=biodckr/bwa TABIX_IMAGE=biodckrdev/htslib:1.2.1 BCFTOOLS_IMAGE=biodckr/bcftools docker pull $SAMTOOLS_IMAGE docker pull $BWA_IMAGE docker pull $TABIX_IMAGE docker pull $BCFTOOLS_IMAGE SAMTOOLS="$DRUN $SAMTOOLS_IMAGE $SAMTOOLS" BWA="$DRUN $BWA_IMAGE $BWA" TABIX="$DRUN $TABIX_IMAGE $TABIX" BCFTOOLS="$DRUN $BCFTOOLS_IMAGE $BCFTOOLS" PLOTVCFSTATS="$DRUN $BCFTOOLS_IMAGE $PLOTVCFSTATS" else echo "RUNNING LOCAL" fi HEADLEN=100000 if [[ ! -f "$FQ1" ]]; then curl ftp://ftp.sra.ebi.ac.uk/vol1/fastq/SRR507/SRR507778/SRR507778_1.fastq.gz| gzip -d | head -$HEADLEN > $FQ1.tmp && mv $FQ1.tmp $FQ1 fi if [[ ! -f "$FQ2" ]]; then curl ftp://ftp.sra.ebi.ac.uk/vol1/fastq/SRR507/SRR507778/SRR507778_2.fastq.gz| gzip -d | head -$HEADLEN > $FQ2.tmp && mv $FQ2.tmp $FQ2 fi if [[ ! -f "$REF" ]]; then curl ftp://ftp.ensembl.org/pub/current_fasta/saccharomyces_cerevisiae/dna/Saccharomyces_cerevisiae.R64-1-1.dna_sm.toplevel.fa.gz | gunzip -c > $REF.tmp && mv $REF.tmp $REF fi if [[ ! -f "$REF.fai" ]]; then $SAMTOOLS faidx $REF fi if [[ ! -f "$REF.bwt" ]]; then $BWA index $REF fi if [[ ! -f "$BNM.sam" ]]; then $BWA mem -R '@RG\tID:foo\tSM:bar\tLB:library1' $REF $FQ1 $FQ2 > $BNM.sam.tmp && mv $BNM.sam.tmp $BNM.sam fi if [[ ! -f "$BNM.bam" ]]; then #$SAMTOOLS sort -O bam -T /tmp -l 0 --input-fmt-option SAM -o $BNM.tmp.bam $BNM.sam && mv $BNM.tmp.bam $BNM.bam $SAMTOOLS sort -O bam -T /tmp -l 0 -o $BNM.tmp.bam $BNM.sam && mv $BNM.tmp.bam $BNM.bam fi if [[ ! -f "$BNM.cram" ]]; then $SAMTOOLS view -T $REF -C -o $BNM.tmp.cram $BNM.bam && mv $BNM.tmp.cram $BNM.cram fi if [[ ! -f "$BNM.P.cram" ]]; then $BWA mem $REF $FQ1 $FQ2 | \ $SAMTOOLS sort -O bam -l 0 -T /tmp - | \ $SAMTOOLS view -T $REF -C -o $BNM.P.tmp.cram - && mv $BNM.P.tmp.cram $BNM.P.cram fi #if [[ ! -f "" ]]; then #$SAMTOOLS view $BNM.cram #fi #if [[ ! -f "" ]]; then #$SAMTOOLS mpileup -f $REF $BNM.cram #fi if [[ ! -f "$BNM.vcf.gz" ]]; then $SAMTOOLS mpileup -ugf $REF $BNM.bam | $BCFTOOLS call -vmO z -o $BNM.vcf.gz.tmp && mv $BNM.vcf.gz.tmp $BNM.vcf.gz fi if [[ ! -f "$BNM.vcf.gz.tbi" ]]; then $TABIX -p vcf $BNM.vcf.gz fi if [[ ! -f "$BNM.vcf.gz.stats" ]]; then $BCFTOOLS stats -F $REF -s - $BNM.vcf.gz > $BNM.vcf.gz.stats.tmp && mv $BNM.vcf.gz.stats.tmp $BNM.vcf.gz.stats fi mkdir plots &>/dev/null || true #if [[ ! -f "plots/tstv_by_sample.0.png" ]]; then #$PLOTVCFSTATS -p plots/ $BNM.vcf.gz.stats #fi if [[ ! -f "$BNM.vcf.filtered.gz" ]]; then $BCFTOOLS filter -O z -o $BNM.vcf.filtered.gz -s LOWQUAL -i'%QUAL>10' $BNM.vcf.gz fi
Sunday, 13 March 2016
Genome Mapping and SNP Calling with BioDocker
Wednesday, 21 October 2015
Installing MESOS in your Mac
1- Homebrew is an open source package management system for the Mac that simplifies installation of packages from source.
ruby -e "$(curl -fsSL https://raw.github.com/Homebrew/homebrew/go/install)"
2- Once you have Homebrew installed, you can install Mesos on your laptop with these two commands:
brew update
brew install mesos
You will need to wait while the most current, stable version of Mesos is downloaded, compiled, and installed on your machine. Homebrew will let you know when it’s finished by displaying a beer emoji in your terminal and a message like the following:
/usr/local/Cellar/mesos/0.19.0: 83 files, 24M, built in 17.4 minutes
Start Your Mesos Cluster
3- Running Mesos in your machine: Now that you have Mesos installed on your laptop, it’s easy to start your Mesos cluster. To see Mesos in action, spin up an in-memory master with the following command:
/usr/local/sbin/mesos-master --registry=in_memory --ip=127.0.0.1
A Mesos cluster needs at least one Mesos Master to coordinate and dispatch tasks onto Mesos Slaves. When experimenting on your laptop, a single master is all you need. Once your Mesos Master has started, you can visit its management console: http://localhost:5050
Since a Mesos Master needs slaves onto which it will dispatch jobs, you might also want to run some of those. Mesos Slaves can be started by running the following command for each slave you wish to launch:
sudo /usr/local/sbin/mesos-slave --master=127.0.0.1:5050
ruby -e "$(curl -fsSL https://raw.github.com/Homebrew/homebrew/go/install)"
2- Once you have Homebrew installed, you can install Mesos on your laptop with these two commands:
brew update
brew install mesos
You will need to wait while the most current, stable version of Mesos is downloaded, compiled, and installed on your machine. Homebrew will let you know when it’s finished by displaying a beer emoji in your terminal and a message like the following:
/usr/local/Cellar/mesos/0.19.0: 83 files, 24M, built in 17.4 minutes
Start Your Mesos Cluster
3- Running Mesos in your machine: Now that you have Mesos installed on your laptop, it’s easy to start your Mesos cluster. To see Mesos in action, spin up an in-memory master with the following command:
/usr/local/sbin/mesos-master --registry=in_memory --ip=127.0.0.1
A Mesos cluster needs at least one Mesos Master to coordinate and dispatch tasks onto Mesos Slaves. When experimenting on your laptop, a single master is all you need. Once your Mesos Master has started, you can visit its management console: http://localhost:5050
Since a Mesos Master needs slaves onto which it will dispatch jobs, you might also want to run some of those. Mesos Slaves can be started by running the following command for each slave you wish to launch:
sudo /usr/local/sbin/mesos-slave --master=127.0.0.1:5050
Etiquetas:
Bioinformatics,
Cluster,
Spark
Wednesday, 30 September 2015
First Scrum Board
Team members update the task board continuously each sprint; if someone thinks of a new task (“test a new machine learning algorithm”), she writes a new card and puts it on the wall. Either during or before the daily scrum, estimates are changed (up or down), and cards are moved around the board.
Each row on the Scrum board is a user story, which is the unit of work we encourage teams to use for their product backlog.
During the sprint planning meeting, the team selects the product backlog items they can complete during the next Spring. Each product backlog item is turned into multiple sprint backlog items. Each of these is represented by one task card that is placed on the Scrumboard.
- Story (User Story): The story description (“As a user we want to…”) shown on that row.
- Ongoing: Any card being worked on goes here. The programmer who chooses to work on it moves it over when she's ready to start the task. Often, this happens during the daily scrum when someone says, “I'm going to work on the boojum today.”
- Testing: A lot of tasks have corresponding test task cards. So, if there's a “Code the boojum class” card, there is likely one or more task cards related to testing: “Test the boojum”, “Write FitNesse tests for the boojum,” “Write FitNesse fixture for the boojum,”
- Done: Cards pile up over here when they're done. They're removed at the end of the sprint. Sometimes we remove some or all during a sprint if there are a lot of cards.
Optionally, depending on the team, the culture, the project and other considerations:
- Notes: Just a place to jot a note or two.
- Tests Specified: We like to do “Story Test-Driven Development,” or “Acceptance Test-Driven Development,” which means the tests are written before the story is coded. Many teams find that it helps to have acceptance tests identified before coding begins on a particular story. This column just contains a checkmark to indicate the tests are specified.
Friday, 11 September 2015
An API for all MS-based File formats
We recently released and published our first Java API (Application Programming Interface) for the most common file formats in proteomics, not only ms files but also identification files such as mzIdentML and mztab.
ms-data-core-api (https://github.com/PRIDE-Utilities/ms-data-core-api)
The library allow the end-users and the developers to use a common data structure for proteomics independently of the file types, and .. But first lets try to understand what is a API.
What is an API?
Imagine you are a builder or civil engineering and your are building your bridge, different components, blocks and different teams needs to be coordinated and plugged for the final results. Wrong communications between the members of the teams, different block sizes or building plans only produced strange results.
In the simplest terms, APIs are sets of requirements, data structures, objects that govern how applications and software components can talk each other. An API, is a set of routines and protocols that provide building blocks for computer programmers and web developers to build software applications. In the past, APIs were largely associated with computer operating systems and desktop applications. In recent years though, we have seen the emergence of Web APIs (Web Services).
What is ms-data-core-api?
The ms-data-core-api is a free, open-source library for developing computational proteomics tools and pipelines. The Application Programming Interface, written in Java, enables rapid tool creation by providing a robust, pluggable programming interface and common data model. The data model is based on controlled vocabularies/ontologies and captures the whole range of data types included in common proteomics experimental workflows, going from spectra to peptide/protein identifications to quantitative results.
The library contains readers for three of the most used Proteomics Standards Initiative standard file formats: mzML, mzIdentML, and mzTab. In addition to mzML, it also supports other common mass spectra data formats: dta, ms2, mgf, pkl, apl (text-based), mzXML and mzData (XML-based). Also, it can be used to read PRIDE XML, the original format used by the PRIDE database, one of the world-leading proteomics resources. Finally, we present a set of algorithms and tools whose implementation illustrates the simplicity of developing applications using the library.
Etiquetas:
API,
Bioinformatic,
computational proteomics,
GitHub Tips,
java,
ms proteomics,
PSI
Thursday, 13 August 2015
The future of Proteomics: The Consensus

After the Big Nature papers about the Human Proteome [1][2] the proteomics community has been divided by the same well-known topics than genomics had before: same reasons, same discussions [3-7]. No one discusses about the technical issues, the instrument settings, nothing about the samples processing, even anything about the analytical method (Most of both projects are "common" bottom-up experiments). Main issues are data-analysis problems and still Computational Proteomics Challenges.
Etiquetas:
Andromeda,
big data,
bioconductor,
Bioinformatic,
biological databases,
computational proteomics,
data analysis,
false discovery rates
Monday, 27 July 2015
one big lesson I just learn
I'm coming from small country with no resources, no big industries or capitals (Cuba); but with a big tradition in friendship and solidarity. In my previous institute (surprisingly, a big biotech company) we share openly all of our ideas, we discuss openly our results, thoughts, etc.. without thinking in competition, plagiarism, or someone from collaborator group can take your ideas and results to sell them to others or take them as his owns ideas.
The picture completely changed, after one year abroad, the only big think I learned is that outside my farm and my small country: time, ideas, contacts are gold. In science you have people with you can work and collaborate, because they are open by nature (not only because they source code is in github) but also because they share, they help, they support, and they give their ideas without concern. People, that like to talk about science, they encourage young researchers without fear of others, without fear of being open.
But you have other people, people that always looks for competition, they stealing what is not theirs, looking for ideas to be recognised, looking for contacts, looking for papers, to get citations. The good thing is that I learn, and I can recognise them. I can give them my ideas, my time, because they need it more than me. At the end, the friendly ones, the collaborative ones, the ones that share, open, help, support; we are more and not only the ones that have their code in github.
Subscribe to:
Posts (Atom)




