diff --git a/CHANGELOG.md b/CHANGELOG.md
index 2e612a11..3fc960f4 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -71,6 +71,9 @@
 
 * `falco`: A C++ drop-in replacement of FastQC to assess the quality of sequence read data (PR #43).
 
+* `umitools`:
+    - `umitools_dedup`: Deduplicate reads based on the mapping co-ordinate and the UMI attached to the read (PR #54).
+
 * `bedtools`:
     - `bedtools_getfasta`: extract sequences from a FASTA file for each of the
                            intervals defined in a BED/GFF/VCF file (PR #59).
diff --git a/src/umi_tools/umi_tools_dedup/config.vsh.yaml b/src/umi_tools/umi_tools_dedup/config.vsh.yaml
new file mode 100644
index 00000000..a02e70a1
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/config.vsh.yaml
@@ -0,0 +1,303 @@
+name: umi_tools_dedup
+namespace: umi_tools
+description: |
+  Deduplicate reads based on the mapping co-ordinate and the UMI attached to the read.
+keywords: [umi_tools, deduplication, dedup]
+links:
+  homepage: https://umi-tools.readthedocs.io/en/latest/
+  documentation: https://umi-tools.readthedocs.io/en/latest/reference/dedup.html
+  repository: https://github.com/CGATOxford/UMI-tools
+references: 
+  doi: 10.1101/gr.209601.116
+license: MIT
+
+argument_groups:
+  - name: Inputs
+    arguments:
+      - name: --input
+        alternatives: --stdin
+        type: file
+        description: Input BAM or SAM file. Use --in_sam to specify SAM format.
+        required: true
+      - name: --in_sam
+        type: boolean_true
+        description: |
+          By default, inputs are assumed to be in BAM format. Use this options to specify the use of SAM
+          format for input.
+      - name: --bai
+        type: file
+        description: BAM index
+      - name: --random_seed
+        type: integer
+        description: Random seed to initialize number generator with.
+
+  - name: Outputs
+    arguments:
+      - name: --output
+        alternatives: --stdout
+        type: file
+        description: Deduplicated BAM file.
+        required: true
+        direction: output
+      - name: --out_sam
+        type: boolean_true
+        description: |
+          By default, outputa are written in BAM format. Use this options to specify the use of SAM format
+          for output.
+      - name: --paired
+        type: boolean_true
+        description: |
+          BAM is paired end - output both read pairs. This will also force the use of the template length
+          to determine reads with the same mapping coordinates.
+      - name: --output_stats
+        type: string
+        description: |
+          Generate files containing UMI based deduplication statistics files with this prefix in the file names.
+      - name: --extract_umi_method
+        type: string
+        choices: [read_id, tag, umis]
+        description: |
+          Specify the method by which the barcodes were encoded in the read.
+          The options are:
+            * read_id (default) 
+            * tag
+            * umis
+        example: "read_id"
+      - name: --umi_tag
+        type: string
+        description: |
+          The tag containing the UMI sequence. This is only required if the extract_umi_method is set to tag.
+      - name: --umi_separator
+        type: string
+        description: |
+          The separator used to separate the UMI from the read sequence. This is only required if the
+          extract_umi_method is set to id_read. Default: `_`.
+        example: '_'
+      - name: --umi_tag_split
+        type: string
+        description: Separate the UMI in tag by <SPLIT> and take the first element.
+      - name: --umi_tag_delimiter
+        type: string
+        description: Separate the UMI in by <DELIMITER> and concatenate the elements.
+      - name: --cell_tag
+        type: string
+        description: |
+          The tag containing the cell barcode sequence. This is only required if the extract_umi_method
+          is set to tag.
+      - name: --cell_tag_split
+        type: string
+        description: Separate the cell barcode in tag by <SPLIT> and take the first element.
+      - name: --cell_tag_delimiter
+        type: string
+        description: Separate the cell barcode in by <DELIMITER> and concatenate the elements.
+  
+  - name: Grouping Options
+    arguments:    
+      - name: --method
+        type: string
+        choices: [unique, percentile, cluster, adjacency, directional]
+        description: |
+          The method to use for grouping reads. 
+          The options are: 
+            * unique
+            * percentile
+            * cluster
+            * adjacency
+            * directional (default)
+        example: "directional"
+      - name: --edit_distance_threshold
+        type: integer
+        description: |
+          For the adjacency and cluster methods the threshold for the edit distance to connect two
+          UMIs in the network can be increased. The default value of 1 works best unless the UMI is
+          very long (>14bp). Default: `1`.
+        example: 1
+      - name: --spliced_is_unique
+        type: boolean_true
+        description: |
+          Causes two reads that start in the same position on the same strand and having the same UMI
+          to be considered unique if one is spliced and the other is not. (Uses the 'N' cigar operation
+          to test for splicing).
+      - name: --soft_clip_threshold
+        type: integer
+        description: |
+          Mappers that soft clip will sometimes do so rather than mapping a spliced read if there is only
+          a small overhang over the exon junction. By setting this option, you can treat reads with at
+          least this many bases soft-clipped at the 3' end as spliced. Default: `4`.
+        example: 4
+      - name: --multimapping_detection_method
+        type: string
+        description: |
+          If the sam/bam contains tags to identify multimapping reads, you can specify for use when selecting
+          the best read at a given loci. Supported tags are `NH`, `X0` and `XT`. If not specified, the read
+          with the highest mapping quality will be selected.
+      - name: --read_length
+        type: boolean_true
+        description: Use the read length as a criteria when deduping, for e.g. sRNA-Seq.
+  
+  - name: Single-cell RNA-Seq Options
+    arguments:
+      - name: --per_gene
+        type: boolean_true
+        description: |
+          Reads will be grouped together if they have the same gene. This is useful if your library prep
+          generates PCR duplicates with non identical alignment positions such as CEL-Seq. Note this option
+          is hardcoded to be on with the count command. I.e. counting is always performed per-gene. Must be
+          combined with either --gene_tag or --per_contig option.
+      - name: --gene_tag
+        type: string
+        description: |
+          Deduplicate per gene. The gene information is encoded in the bam read tag specified.
+      - name: --assigned_status_tag
+        type: string
+        description: |
+          BAM tag which describes whether a read is assigned to a gene. Defaults to the same value as given
+          for --gene_tag.
+      - name: --skip_tags_regex
+        type: string
+        description: |
+          Use in conjunction with the --assigned_status_tag option to skip any reads where the tag matches
+          this regex. Default ("^[__|Unassigned]") matches anything which starts with "__" or "Unassigned".
+      - name: --per_contig
+        type: boolean_true
+        description: |
+          Deduplicate per contig (field 3 in BAM; RNAME). All reads with the sam contig will be considered to
+          have the same alignment position. This is useful if you have aligned to a reference transcriptome
+          with one transcript per gene. If you have aligned to a transcriptome with more than one transcript
+          per gene, you can supply a map between transcripts and gene using the --gene_transcript_map option.
+      - name: --gene_transcript_map
+        type: file
+        description: |
+          A file containing a mapping between gene names and transcript names. The file should be tab
+          separated with the gene name in the first column and the transcript name in the second column.
+      - name: --per_cell
+        type: boolean_true
+        description: |
+          Reads will only be grouped together if they have the same cell barcode. Can be combined with
+          --per_gene.
+  
+  - name: SAM/BAM Options
+    arguments:
+      - name: --mapping_quality
+        type: integer
+        description: |
+          Minimium mapping quality (MAPQ) for a read to be retained. Default: `0`.
+        example: 0
+      - name: --unmapped_reads
+        type: string
+        description: |
+          How unmapped reads should be handled. 
+          The options are:
+            * "discard": Discard all unmapped reads. (default)
+            * "use":     If read2 is unmapped, deduplicate using read1 only. Requires --paired.
+            * "output":  Output unmapped reads/read pairs without UMI grouping/deduplication. Only available in umi_tools group.
+        example: "discard"
+      - name: --chimeric_pairs
+        type: string
+        choices: [discard, use, output]
+        description: |
+          How chimeric pairs should be handled. 
+          The options are:
+            * "discard": Discard all chimeric read pairs.
+            * "use":     Deduplicate using read1 only. (default)
+            * "output":  Output chimeric pairs without UMI grouping/deduplication. Only available in
+                         umi_tools group.
+        example: "use"
+      - name: --unpaired_reads
+        type: string
+        choices: [discard, use, output]
+        description: |
+          How unpaired reads should be handled. 
+          The options are: 
+            * "discard": Discard all unmapped reads.
+            * "use": If read2 is unmapped, deduplicate using read1 only. Requires --paired. (default)
+            * "output":  Output unmapped reads/read pairs without UMI grouping/deduplication. Only available
+                         in umi_tools group.
+        example: "use"
+      - name: --ignore_umi
+        type: boolean_true
+        description: Ignore the UMI and group reads using mapping coordinates only.
+      - name: --subset
+        type: double
+        description: |
+          Only consider a fraction of the reads, chosen at random. This is useful for doing saturation
+          analyses.
+      - name: --chrom
+        type: string
+        description: Only consider a single chromosome. This is useful for debugging/testing purposes.
+  
+  - name: Group/Dedup Options
+    arguments:
+      - name: --no_sort_output
+        type: boolean_true
+        description: |
+          By default, output is sorted. This involves the use of a temporary unsorted file (saved in
+          --temp_dir). Use this option to turn off sorting.
+      - name: --buffer_whole_contig
+        type: boolean_true
+        description: |
+          Forces dedup to parse an entire contig before yielding any reads for deduplication. This is the
+          only way to absolutely guarantee that all reads with the same start position are grouped together
+          for deduplication since dedup uses the start position of the read, not the alignment coordinate on
+          which the reads are sorted. However, by default, dedup reads for another 1000bp before outputting
+          read groups which will avoid any reads being missed with short read sequencing (<1000bp).
+  
+  - name: Common Options
+    arguments:
+      - name: --log
+        alternatives: -L
+        type: file
+        description: File with logging information.
+      - name: --log2stderr
+        type: boolean_true
+        description: Send logging information to stderr.
+      - name: --verbose
+        alternatives: -v
+        type: integer
+        description: |
+          Log level. The higher, the more output. Default: `0`.
+        example: 0
+      - name: --error
+        alternatives: -E
+        type: file
+        description: File with error information.
+      - name: --temp_dir
+        type: string
+        description: |
+          Directory for temporary files. If not set, the bash environmental variable TMPDIR is used.
+      - name: --compresslevel
+        type: integer
+        description: |
+          Level of Gzip compression to use. Default=6 matches GNU gzip rather than python gzip default.
+          Default: `6`.
+        example: 6
+      - name: --timeit
+        type: file
+        description: Store timing information in file.
+      - name: --timeit_name
+        type: string
+        description: |
+          Name in timing file for this class of jobs. Default: `all`.
+        example: "all"
+      - name: --timeit_header
+        type: string
+        description: Add header for timing information.
+
+resources:
+  - type: bash_script
+    path: script.sh
+test_resources:
+  - type: bash_script
+    path: test.sh
+  - type: file
+    path: test_data
+engines:
+  - type: docker
+    image: quay.io/biocontainers/umi_tools:1.1.5--py39hf95cd2a_1
+    setup:
+      - type: docker
+        run: |
+            umi_tools -v | sed 's/ version//g' > /var/software_versions.txt
+runners:
+- type: executable
+- type: nextflow
\ No newline at end of file
diff --git a/src/umi_tools/umi_tools_dedup/help.txt b/src/umi_tools/umi_tools_dedup/help.txt
new file mode 100644
index 00000000..87baf322
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/help.txt
@@ -0,0 +1,113 @@
+'''
+Generated from the following UMI-tools documentation:
+      https://umi-tools.readthedocs.io/en/latest/common_options.html#common-options
+      https://umi-tools.readthedocs.io/en/latest/reference/dedup.html
+'''
+
+
+dedup - Deduplicate reads using UMI and mapping coordinates
+
+Usage: umi_tools dedup [OPTIONS] [--stdin=IN_BAM] [--stdout=OUT_BAM]
+
+       note: If --stdout is ommited, standard out is output. To
+             generate a valid BAM file on standard out, please
+             redirect log with --log=LOGFILE or --log2stderr 
+
+Common UMI-tools Options:
+
+      -S, --stdout                  File where output is to go [default = stdout].
+      -L, --log                     File with logging information [default = stdout].
+      --log2stderr                  Send logging information to stderr [default = False].
+      -v, --verbose                 Log level. The higher, the more output [default = 1].
+      -E, --error                   File with error information [default = stderr].
+      --temp-dir                    Directory for temporary files. If not set, the bash environmental variable TMPDIR is used[default = None].
+      --compresslevel               Level of Gzip compression to use. Default=6 matches GNU gzip rather than python gzip default (which is 9)
+
+      profiling and debugging options:
+      --timeit                      Store timing information in file [default=none].
+      --timeit-name                 Name in timing file for this class of jobs [default=all].
+      --timeit-header               Add header for timing information [default=none].
+      --random-seed                 Random seed to initialize number generator with [default=none].
+
+Dedup Options:
+      --output-stats=<prefix>             One can use the edit distance between UMIs at the same position as an quality control for the 
+                                          deduplication process by comparing with a null expectation of random sampling. For the random
+                                          sampling, the observed frequency of UMIs is used to more reasonably model the null expectation.
+                                          Use this option to generate a stats outfiles called: 
+                                                [PREFIX]_stats_edit_distance.tsv   
+                                                      Reports the (binned) average edit distance between the UMIs at each position.
+                                          In addition, this option will trigger reporting of further summary statistics for the UMIs which
+                                          may be informative for selecting the optimal deduplication method or debugging.
+                                          Each unique UMI sequence may be observed [0-many] times at multiple positions in the BAM. The
+                                          following files report the distribution for the frequencies of each UMI.
+                                                [PREFIX]_stats_per_umi_per_position.tsv
+                                                      Tabulates the counts for unique combinations of UMI and position.
+                                                [PREFIX]_stats_per_umi_per.tsv
+                                                      The _stats_per_umi_per.tsv table provides UMI-level summary statistics. 
+      --extract-umi-method=<method>       How are the barcodes encoded in the read?
+                                          Options are: read_id (default), tag, umis
+      --umi-separator=<separator>         Separator between read id and UMI. See --extract-umi-method above. Default=_
+      --umi-tag=<tag>                     Tag which contains UMI. See --extract-umi-method above
+      --umi-tag-split=<split>             Separate the UMI in tag by SPLIT and take the first element
+      --umi-tag-delimiter=<delimiter>     Separate the UMI in by DELIMITER and concatenate the elements
+      --cell-tag=<tag>                    Tag which contains cell barcode. See --extract-umi-method above
+      --cell-tag-split=<split>            Separate the cell barcode in tag by SPLIT and take the first element
+      --cell-tag-delimiter=<delimiter>    Separate the cell barcode in by DELIMITER and concatenate the elements
+      --method=<method>                   What method to use to identify group of reads with the same (or similar) UMI(s)?
+                                          All methods start by identifying the reads with the same mapping position.
+                                          The simplest methods, unique and percentile, group reads with the exact same UMI.
+                                          The network-based methods, cluster, adjacency and directional, build networks where
+                                          nodes are UMIs and edges connect UMIs with an edit distance <= threshold (usually 1).
+                                          The groups of reads are then defined from the network in a method-specific manner.
+                                          For all the network-based methods, each read group is equivalent to one read count for the gene.
+      --edit-distance-threshold=<threshold>     For the adjacency and cluster methods the threshold for the edit distance to connect
+                                                two UMIs in the network can be increased. The default value of 1 works best unless
+                                                the UMI is very long (>14bp).
+      --spliced-is-unique           Causes two reads that start in the same position on the same strand and having the
+                                    same UMI to be considered unique if one is spliced and the other is not.
+                                    (Uses the 'N' cigar operation to test for splicing).
+      --soft-clip-threshold=<threshold>    Mappers that soft clip will sometimes do so rather than mapping a spliced read if
+                                          there is only a small overhang over the exon junction. By setting this option, you
+                                          can treat reads with at least this many bases soft-clipped at the 3' end as spliced.
+                                          Default=4.
+      --multimapping-detection-method=<method>  If the sam/bam contains tags to identify multimapping reads, you can specify
+                                                for use when selecting the best read at a given loci. Supported tags are "NH",
+                                                "X0" and "XT". If not specified, the read with the highest mapping quality will be selected.
+      --read-length                              Use the read length as a criteria when deduping, for e.g sRNA-Seq.
+      --per-gene                    Reads will be grouped together if they have the same gene. This is useful if your
+                                    library prep generates PCR duplicates with non identical alignment positions such as CEL-Seq.
+                                    Note this option is hardcoded to be on with the count command. I.e counting is always
+                                    performed per-gene. Must be combined with either --gene-tag or --per-contig option.
+      --gene-tag=<tag>              Deduplicate per gene. The gene information is encoded in the bam read tag specified
+      --assigned-status-tag=<tag>   BAM tag which describes whether a read is assigned to a gene. Defaults to the same value
+                                    as given for --gene-tag
+      --skip-tags-regex=<regex>     Use in conjunction with the --assigned-status-tag option to skip any reads where the
+                                    tag matches this regex. Default ("^[__|Unassigned]") matches anything which starts with "__"
+                                    or "Unassigned":
+      --per-contig                  Deduplicate per contig (field 3 in BAM; RNAME). All reads with the same contig will be
+                                    considered to have the same alignment position. This is useful if you have aligned to a
+                                    reference transcriptome with one transcript per gene. If you have aligned to a transcriptome
+                                    with more than one transcript per gene, you can supply a map between transcripts and gene
+                                    using the --gene-transcript-map option
+      --gene-transcript-map=<file>  File mapping genes to transcripts (tab separated)
+      --per-cell                    Reads will only be grouped together if they have the same cell barcode. Can be combined with --per-gene.
+      --mapping-quality=<quality>   Minimium mapping quality (MAPQ) for a read to be retained. Default is 0.
+      --unmapped-reads=<option>     How should unmapped reads be handled.
+      --chimeric-pairs=<option>     How should chimeric read pairs be handled.
+      --unpaired-reads=<option>     How should unpaired reads be handled.
+      --ignore-umi                  Ignore the UMI and group reads using mapping coordinates only
+      --subset=<fraction>           Only consider a fraction of the reads, chosen at random. This is useful for doing saturation analyses.
+      --chrom=<chromosome>          Only consider a single chromosome. This is useful for debugging/testing purposes
+      --in-sam                      Input is in SAM format
+      --out-sam                     Output is in SAM format
+      --paired                      BAM is paired end - output both read pairs. This will also force the use of the template
+                                    length to determine reads with the same mapping coordinates.
+      --no-sort-output              By default, output is sorted. This involves the use of a temporary unsorted file since
+                                    reads are considered in the order of their start position which may not be the same as
+                                    their alignment coordinate due to soft-clipping and reverse alignments. The temp file
+                                    will be saved (in --temp-dir) and deleted when it has been sorted to the outfile. Use
+                                    this option to turn off sorting.
+      --buffer-whole-contig         Forces dedup to parse an entire contig before yielding any reads for deduplication.
+                                    This is the only way to absolutely guarantee that all reads with the same start position
+                                    are grouped together for deduplication since dedup uses the start position of the read,
+                                    not the alignment coordinate on which the reads are
\ No newline at end of file
diff --git a/src/umi_tools/umi_tools_dedup/script.sh b/src/umi_tools/umi_tools_dedup/script.sh
new file mode 100644
index 00000000..d57a5e76
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/script.sh
@@ -0,0 +1,72 @@
+#!/bin/bash
+
+## VIASH START
+## VIASH END
+
+set -e
+
+test_dir="${metal_executable}/test_data"
+
+[[ "$par_paired" == "false" ]] && unset par_paired
+[[ "$par_in_sam" == "false" ]] && unset par_in_sam
+[[ "$par_out_sam" == "false" ]] && unset par_out_sam
+[[ "$par_spliced_is_unique" == "false" ]] && unset par_spliced_is_unique
+[[ "$par_per_gene" == "false" ]] && unset par_per_gene
+[[ "$par_per_contig" == "false" ]] && unset par_per_contig
+[[ "$par_per_cell" == "false" ]] && unset par_per_cell
+[[ "$par_no_sort_output" == "false" ]] && unset par_no_sort_output
+[[ "$par_buffer_whole_contig" == "false" ]] && unset par_buffer_whole_contig
+[[ "$par_ignore_umi" == "false" ]] && unset par_ignore_umi
+[[ "$par_subset" == "false" ]] && unset par_subset
+[[ "$par_log2stderr" == "false" ]] && unset par_log2stderr
+[[ "$par_read_length" == "false" ]] && unset par_read_length
+
+umi_tools dedup \
+    --stdin "$par_input" \
+    ${par_in_sam:+--in-sam} \
+    --stdout "$par_output" \
+    ${par_out_sam:+--out-sam} \
+    ${par_paired:+--paired} \
+    ${par_output_stats:+--output-stats "$par_output_stats"} \
+    ${par_extract_umi_method:+--extract-umi-method "$par_extract_umi_method"} \
+    ${par_umi_tag:+--umi-tag "$par_umi_tag"} \
+    ${par_umi_separator:+--umi-separator "$par_umi_separator"} \
+    ${par_umi_tag_split:+--umi-tag-split "$par_umi_tag_split"} \
+    ${par_umi_tag_delimiter:+--umi-tag-delimiter "$par_umi_tag_delimiter"} \
+    ${par_cell_tag:+--cell-tag "$par_cell_tag"} \
+    ${par_cell_tag_split:+--cell-tag-split "$par_cell_tag_split"} \
+    ${par_cell_tag_delimiter:+--cell-tag-delimiter "$par_cell_tag_delimiter"} \
+    ${par_method:+--method "$par_method"} \
+    ${par_edit_distance_threshold:+--edit-distance-threshold "$par_edit_distance_threshold"} \
+    ${par_spliced_is_unique:+--spliced-is-unique} \
+    ${par_soft_clip_threshold:+--soft-clip-threshold "$par_soft_clip_threshold"} \
+    ${par_multimapping_detection_method:+--multimapping-detection-method "$par_multimapping_detection_method"} \
+    ${par_read_length:+--read-length} \
+    ${par_per_gene:+--per-gene} \
+    ${par_gene_tag:+--gene-tag "$par_gene_tag"} \
+    ${par_assigned_status_tag:+--assigned-status-tag "$par_assigned_status_tag"} \
+    ${par_skip_tags_regex:+--skip-tags-regex "$par_skip_tags_regex"} \
+    ${par_per_contig:+--per-contig} \
+    ${par_gene_transcript_map:+--gene-transcript-map "$par_gene_transcript_map"} \
+    ${par_per_cell:+--per-cell} \
+    ${par_mapping_quality:+--mapping-quality "$par_mapping_quality"} \
+    ${par_unmapped_reads:+--unmapped-reads "$par_unmapped_reads"} \
+    ${par_chimeric_pairs:+--chimeric-pairs "$par_chimeric_pairs"} \
+    ${par_unpaired_reads:+--unpaired-reads "$par_unpaired_reads"} \
+    ${par_ignore_umi:+--ignore-umi} \
+    ${par_subset:+--subset "$par_subset"} \
+    ${par_chrom:+--chrom "$par_chrom"} \
+    ${par_no_sort_output:+--no-sort-output} \
+    ${par_buffer_whole_contig:+--buffer-whole-contig} \
+    ${par_log:+-L "$par_log"} \
+    ${par_log2stderr:+--log2stderr} \
+    ${par_verbose:+-v "$par_verbose"} \
+    ${par_error:+-E "$par_error"} \
+    ${par_temp_dir:+--temp-dir "$par_temp_dir"} \
+    ${par_compresslevel:+--compresslevel "$par_compresslevel"} \
+    ${par_timeit:+--timeit "$par_timeit"} \
+    ${par_timeit_name:+--timeit-name "$par_timeit_name"} \
+    ${par_timeit_header:+--timeit-header "$par_timeit_header"} \
+    ${par_random_seed:+--random-seed "$par_random_seed"}
+
+exit 0
diff --git a/src/umi_tools/umi_tools_dedup/test.sh b/src/umi_tools/umi_tools_dedup/test.sh
new file mode 100644
index 00000000..adadb410
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/test.sh
@@ -0,0 +1,87 @@
+#!/bin/bash
+
+test_dir="${meta_resources_dir}/test_data"
+out_dir="${meta_resources_dir}/out"
+
+mkdir -p "$out_dir"
+
+############################################################################################
+
+echo ">>> Test 1: Basic usage of $meta_functionality_name with statistics output"
+
+"$meta_executable" \
+  --paired \
+  --input "$test_dir/sample.bam" \
+  --bai "$test_dir/sample.bam.bai" \
+  --output "$out_dir/deduped.sam" \
+  --out_sam \
+  --output_stats "$out_dir/dedup" \
+  --random_seed 1
+
+echo ">>> Checking whether output exists"
+[ ! -f "$out_dir/deduped.sam" ] && echo "File 'deduped.sam' does not exist!" && exit 1
+[ ! -f "$out_dir/dedup_edit_distance.tsv" ] && echo "File 'dedup_edit_distance.tsv' does not exist!" && exit 1
+
+echo ">>> Checking whether output is non-empty"
+[ ! -s "$out_dir/deduped.sam" ] && echo "File 'deduped.sam' is empty!" && exit 1
+[ ! -s "$out_dir/dedup_edit_distance.tsv" ] && echo "File 'dedup_edit_distance.tsv' is empty!" && exit 1
+
+echo ">>> Checking whether output is correct"
+diff "$out_dir/deduped.sam" "$test_dir/deduped.sam" || \
+    (echo "Output file deduped.sam does not match expected output" && exit 1)
+diff "$out_dir/dedup_edit_distance.tsv" "$test_dir/dedup_edit_distance.tsv" || \
+    (echo "Output file dedup_edit_distance.tsv does not match expected output" && exit 1)
+
+############################################################################################
+
+echo ">>> Test 2: $meta_functionality_name with random subset selection"
+
+"$meta_executable" \
+  --paired \
+  --input "$test_dir/sample.bam" \
+  --bai "$test_dir/sample.bam.bai" \
+  --output "$out_dir/deduped_fraction.sam" \
+  --out_sam \
+  --subset 0.5 \
+  --random_seed 1
+
+
+echo ">>> Checking whether output exists"
+[ ! -f "$out_dir/deduped_fraction.sam" ] && echo "File 'deduped_fraction.sam' does not exist!" && exit 1
+
+echo ">>> Checking whether output is non-empty"
+[ ! -s "$out_dir/deduped_fraction.sam" ] && echo "File 'deduped_fraction.sam' is empty!" && exit 1
+
+echo ">>> Checking whether output is correct"
+diff "$out_dir/deduped_fraction.sam" "$test_dir/deduped_fraction.sam" || \
+    (echo "Output file deduped_fraction.sam does not match expected output" && exit 1)
+
+############################################################################################
+
+echo ">>> Test 3: $meta_functionality_name with --method unique"
+
+"$meta_executable" \
+  --paired \
+  --input "$test_dir/sample.bam" \
+  --bai "$test_dir/sample.bam.bai" \
+  --output "$out_dir/deduped_unique.sam" \
+  --out_sam \
+  --method "unique" \
+  --random_seed 1
+
+echo ">>> Checking whether output exists"
+[ ! -f "$out_dir/deduped_unique.sam" ] && echo "File 'deduped_unique.sam' does not exist!" && exit 1
+
+echo ">>> Checking whether output is non-empty"
+[ ! -s "$out_dir/deduped_unique.sam" ] && echo "File 'deduped_unique.sam' is empty!" && exit 1
+
+echo ">>> Checking whether output is correct"
+diff "$out_dir/deduped_unique.sam" "$test_dir/deduped_unique.sam" || \
+    (echo "Output file deduped_unique.sam does not match expected output" && exit 1)
+
+############################################################################################
+
+rm -rf "$out_dir"
+
+echo "All tests succeeded!"
+exit 0
\ No newline at end of file
diff --git a/src/umi_tools/umi_tools_dedup/test_data/dedup_edit_distance.tsv b/src/umi_tools/umi_tools_dedup/test_data/dedup_edit_distance.tsv
new file mode 100644
index 00000000..89684b04
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/test_data/dedup_edit_distance.tsv
@@ -0,0 +1,5 @@
+unique	unique_null	directional	directional_null	edit_distance
+3	3	4	4	Single_UMI
+0	1	0	0	0
+1	0	0	0	1
+0	0	0	0	2
diff --git a/src/umi_tools/umi_tools_dedup/test_data/dedup_per_umi.tsv b/src/umi_tools/umi_tools_dedup/test_data/dedup_per_umi.tsv
new file mode 100644
index 00000000..a1d364e2
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/test_data/dedup_per_umi.tsv
@@ -0,0 +1,6 @@
+UMI	median_counts_pre	times_observed_pre	total_counts_pre	median_counts_post	times_observed_post	total_counts_post
+ACCGGTTTA	74	1	74	74	1	74
+ACTGGTTTC	48	1	48	49	1	49
+AGCGGTTAC	1	1	1	1	1	1
+CCAGGTTCT	1	1	1	1	1	1
+TCTGGTTTC	1	1	1	0	0	0
diff --git a/src/umi_tools/umi_tools_dedup/test_data/dedup_per_umi_per_position.tsv b/src/umi_tools/umi_tools_dedup/test_data/dedup_per_umi_per_position.tsv
new file mode 100644
index 00000000..d9211d0a
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/test_data/dedup_per_umi_per_position.tsv
@@ -0,0 +1,5 @@
+counts	instances_pre	instances_post
+1	3	2
+48	1	0
+49	0	1
+74	1	1
diff --git a/src/umi_tools/umi_tools_dedup/test_data/deduped.sam b/src/umi_tools/umi_tools_dedup/test_data/deduped.sam
new file mode 100644
index 00000000..cce2efb4
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/test_data/deduped.sam
@@ -0,0 +1,30 @@
+@HD	VN:1.0	SO:coordinate
+@SQ	SN:chr1	LN:197195432
+@SQ	SN:chr10	LN:129993255
+@SQ	SN:chr11	LN:121843856
+@SQ	SN:chr12	LN:121257530
+@SQ	SN:chr13	LN:120284312
+@SQ	SN:chr14	LN:125194864
+@SQ	SN:chr15	LN:103494974
+@SQ	SN:chr16	LN:98319150
+@SQ	SN:chr17	LN:95272651
+@SQ	SN:chr18	LN:90772031
+@SQ	SN:chr19	LN:61342430
+@SQ	SN:chr2	LN:181748087
+@SQ	SN:chr3	LN:159599783
+@SQ	SN:chr4	LN:155630120
+@SQ	SN:chr5	LN:152537259
+@SQ	SN:chr6	LN:149517037
+@SQ	SN:chr7	LN:152524553
+@SQ	SN:chr8	LN:131738871
+@SQ	SN:chr9	LN:124076172
+@SQ	SN:chrM	LN:16299
+@SQ	SN:chrX	LN:166650296
+@SQ	SN:chrY	LN:15902555
+@PG	ID:Bowtie	VN:1.1.2	CL:"bowtie --wrapper basic-0 --threads 4 -v 2 -m 10 -k 1 /ifs/mirror/genomes/bowtie/mm9 /dev/fd/63 --sam"
+@PG	ID:samtools	PN:samtools	PP:Bowtie	VN:1.19.2	CL:samtools view -h example.bam
+@PG	ID:samtools.1	PN:samtools	PP:samtools	VN:1.19.2	CL:samtools view -bS -
+SRR2057595.5052066_ACCGGTTTA	16	chr1	3812794	255	51M	*	0	0	*	*	XA:i:2	MD:Z:42T2T5	NM:i:2
+SRR2057595.13520751_CCAGGTTCT	16	chr1	3967622	255	20M	*	0	0	*	*	XA:i:2	MD:Z:12A0C6	NM:i:2
+SRR2057595.8901432_AGCGGTTAC	0	chr1	4369756	255	20M	*	0	0	*	*	XA:i:2	MD:Z:1T4A13	NM:i:2
+SRR2057595.1210348_ACTGGTTTC	0	chr1	4762503	255	45M	*	0	0	*	*	XA:i:2	MD:Z:0C7A36	NM:i:2
diff --git a/src/umi_tools/umi_tools_dedup/test_data/deduped_fraction.sam b/src/umi_tools/umi_tools_dedup/test_data/deduped_fraction.sam
new file mode 100644
index 00000000..cf9e651a
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/test_data/deduped_fraction.sam
@@ -0,0 +1,29 @@
+@HD	VN:1.0	SO:coordinate
+@SQ	SN:chr1	LN:197195432
+@SQ	SN:chr10	LN:129993255
+@SQ	SN:chr11	LN:121843856
+@SQ	SN:chr12	LN:121257530
+@SQ	SN:chr13	LN:120284312
+@SQ	SN:chr14	LN:125194864
+@SQ	SN:chr15	LN:103494974
+@SQ	SN:chr16	LN:98319150
+@SQ	SN:chr17	LN:95272651
+@SQ	SN:chr18	LN:90772031
+@SQ	SN:chr19	LN:61342430
+@SQ	SN:chr2	LN:181748087
+@SQ	SN:chr3	LN:159599783
+@SQ	SN:chr4	LN:155630120
+@SQ	SN:chr5	LN:152537259
+@SQ	SN:chr6	LN:149517037
+@SQ	SN:chr7	LN:152524553
+@SQ	SN:chr8	LN:131738871
+@SQ	SN:chr9	LN:124076172
+@SQ	SN:chrM	LN:16299
+@SQ	SN:chrX	LN:166650296
+@SQ	SN:chrY	LN:15902555
+@PG	ID:Bowtie	VN:1.1.2	CL:"bowtie --wrapper basic-0 --threads 4 -v 2 -m 10 -k 1 /ifs/mirror/genomes/bowtie/mm9 /dev/fd/63 --sam"
+@PG	ID:samtools	PN:samtools	PP:Bowtie	VN:1.19.2	CL:samtools view -h example.bam
+@PG	ID:samtools.1	PN:samtools	PP:samtools	VN:1.19.2	CL:samtools view -bS -
+SRR2057595.4062788_ACCGGTTTA	16	chr1	3812793	255	52M	*	0	0	*	*	XA:i:2	MD:Z:43T2T5	NM:i:2
+SRR2057595.8901432_AGCGGTTAC	0	chr1	4369756	255	20M	*	0	0	*	*	XA:i:2	MD:Z:1T4A13	NM:i:2
+SRR2057595.1999468_ACTGGTTTC	0	chr1	4762503	255	45M	*	0	0	*	*	XA:i:2	MD:Z:0C7A36	NM:i:2
diff --git a/src/umi_tools/umi_tools_dedup/test_data/deduped_unique.sam b/src/umi_tools/umi_tools_dedup/test_data/deduped_unique.sam
new file mode 100644
index 00000000..570ea153
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/test_data/deduped_unique.sam
@@ -0,0 +1,31 @@
+@HD	VN:1.0	SO:coordinate
+@SQ	SN:chr1	LN:197195432
+@SQ	SN:chr10	LN:129993255
+@SQ	SN:chr11	LN:121843856
+@SQ	SN:chr12	LN:121257530
+@SQ	SN:chr13	LN:120284312
+@SQ	SN:chr14	LN:125194864
+@SQ	SN:chr15	LN:103494974
+@SQ	SN:chr16	LN:98319150
+@SQ	SN:chr17	LN:95272651
+@SQ	SN:chr18	LN:90772031
+@SQ	SN:chr19	LN:61342430
+@SQ	SN:chr2	LN:181748087
+@SQ	SN:chr3	LN:159599783
+@SQ	SN:chr4	LN:155630120
+@SQ	SN:chr5	LN:152537259
+@SQ	SN:chr6	LN:149517037
+@SQ	SN:chr7	LN:152524553
+@SQ	SN:chr8	LN:131738871
+@SQ	SN:chr9	LN:124076172
+@SQ	SN:chrM	LN:16299
+@SQ	SN:chrX	LN:166650296
+@SQ	SN:chrY	LN:15902555
+@PG	ID:Bowtie	VN:1.1.2	CL:"bowtie --wrapper basic-0 --threads 4 -v 2 -m 10 -k 1 /ifs/mirror/genomes/bowtie/mm9 /dev/fd/63 --sam"
+@PG	ID:samtools	PN:samtools	PP:Bowtie	VN:1.19.2	CL:samtools view -h example.bam
+@PG	ID:samtools.1	PN:samtools	PP:samtools	VN:1.19.2	CL:samtools view -bS -
+SRR2057595.5052066_ACCGGTTTA	16	chr1	3812794	255	51M	*	0	0	*	*	XA:i:2	MD:Z:42T2T5	NM:i:2
+SRR2057595.13520751_CCAGGTTCT	16	chr1	3967622	255	20M	*	0	0	*	*	XA:i:2	MD:Z:12A0C6	NM:i:2
+SRR2057595.8901432_AGCGGTTAC	0	chr1	4369756	255	20M	*	0	0	*	*	XA:i:2	MD:Z:1T4A13	NM:i:2
+SRR2057595.1210348_ACTGGTTTC	0	chr1	4762503	255	45M	*	0	0	*	*	XA:i:2	MD:Z:0C7A36	NM:i:2
+SRR2057595.1169423_TCTGGTTTC	0	chr1	4762503	255	45M	*	0	0	*	*	XA:i:2	MD:Z:0C7A36	NM:i:2
diff --git a/src/umi_tools/umi_tools_dedup/test_data/sample.bam b/src/umi_tools/umi_tools_dedup/test_data/sample.bam
new file mode 100644
index 00000000..32192fc6
Binary files /dev/null and b/src/umi_tools/umi_tools_dedup/test_data/sample.bam differ
diff --git a/src/umi_tools/umi_tools_dedup/test_data/sample.bam.bai b/src/umi_tools/umi_tools_dedup/test_data/sample.bam.bai
new file mode 100644
index 00000000..e9e2eee1
Binary files /dev/null and b/src/umi_tools/umi_tools_dedup/test_data/sample.bam.bai differ
diff --git a/src/umi_tools/umi_tools_dedup/test_data/script.sh b/src/umi_tools/umi_tools_dedup/test_data/script.sh
new file mode 100755
index 00000000..2253a0d1
--- /dev/null
+++ b/src/umi_tools/umi_tools_dedup/test_data/script.sh
@@ -0,0 +1,8 @@
+#!/bin/bash
+
+# Download test data
+wget https://github.com/CGATOxford/UMI-tools/releases/download/v0.2.3/example.bam
+# extract 150 reads with a maximum of two reads having the same start position
+samtools view -h example.bam | head -n 150 | samtools view -bS - > sample.bam
+samtools index sample.bam
+rm example.bam
\ No newline at end of file