<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<ANALYSIS_SET>
    <ANALYSIS accession="DRZ002863" center_name="KYOTO_AG" alias="DRZ002863" analysis_date="2012-03-07T00:00:00+09:00" analysis_center="KYOTO_AG">
        <TITLE>creating consensus sequences by clustering sequence reads</TITLE>
        <STUDY_REF accession="PRJDB2367" refname="PRJDB2367">
            <IDENTIFIERS>
                <PRIMARY_ID label="BioProject ID">PRJDB2367</PRIMARY_ID>
            </IDENTIFIERS>
        </STUDY_REF>
        <DESCRIPTION>After adaptor trimming with the cutadapt program (http://code.google.com/p/cutadapt/) and discarding reads containing N bases, filter-passed sequence reads from the founder RRLs were divided into 3 groups by their 5???-terminal sequences (both-end HaeIII, both-end MboI, and others). Sequence reads within a group were clustered by the clustering program ???SEED??? (http://manuals.bioinformatics.ucr.edu/home/seed) and consensus sequences of 300 bp (read-pair) were generated. These consisted of forward and reverse 101-bp reads with internal 98 bases of N. Consensus sequences with depth ???10 were used as reference sequences for mapping of read-pairs from each founder.</DESCRIPTION>
        <ANALYSIS_TYPE>
            <DE_NOVO_ASSEMBLY>
                <PROCESSING>
                    <PIPELINE>
                        <PIPE_SECTION>
                            <STEP_INDEX>0</STEP_INDEX>
                            <PREV_STEP_INDEX>NIL</PREV_STEP_INDEX>
                            <PROGRAM>SEED</PROGRAM>
                            <VERSION>1.5.1</VERSION>
                        </PIPE_SECTION>
                    </PIPELINE>
                </PROCESSING>
            </DE_NOVO_ASSEMBLY>
        </ANALYSIS_TYPE>
        <DATA_BLOCK>
            <FILES>
                <FILE checksum="E60CA420EA623745B71D4E3AD59D566C" checksum_method="MD5" filetype="fasta" filename="ALL_Cluster_num10_readPair.fasta"/>
            </FILES>
        </DATA_BLOCK>
    </ANALYSIS>
</ANALYSIS_SET>
