Last active
September 14, 2026 12:26
-
-
Save nylander/287d1f47c669a350c2e7b97a3da58df5 to your computer and use it in GitHub Desktop.
Create a sample sheet from NGI (SciLifeLab) sequencing delivery
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/usr/bin/env perl | |
| =pod | |
| =encoding utf8 | |
| =head1 NAME | |
| create_sample_sheet.pl - Create sample sheet from NGI delivery | |
| =head1 SYNOPSIS | |
| $ create_sample_sheet.pl 00-Reports/S.OmeName_22_01_sample_info.txt | |
| =head1 DESCRIPTION | |
| Change directory to main delivery directory (e.g. "cd P27213"). Assuming there | |
| is a column with user defined IDs in the file "00-Reports/*_sample_info.txt" | |
| (tab-separated columns), the script will read that file (and column nrs 1 and | |
| 2), and then locate corresponding .fastq.gz files. | |
| Prints tab separated entries to STDOUT. | |
| Note: Currently assumes PE libraries, and a file name format similar to this | |
| "P27213_116_S68_L002_R1_001.fastq.gz". The "library" part ("L002") is to be | |
| extracted as well as the "NGI ID" part ("P27213_116"). | |
| =head1 OPTIONS | |
| =over 4 | |
| =item B<-m,--mapandcall> Output table for the map_and_call workflow (L<https://github.com/axeljen/map_and_call>). | |
| =item B<-a,--atlas> Output table for the ATLAS workflow (L<https://github.com/metagenome-atlas/atlas>). | |
| =item B<-e,--eager> Output table for the nf-core/eager workflow (L<https://nf-co.re/eager/2.4.7/usage#tsv-input-method>). | |
| =item B<-h,--help> Display help. | |
| =item B<-v,--version> Display version. | |
| =back | |
| =head1 AUTHOR | |
| Johan Nylander <johan.nylander@nrm.se> | |
| =head1 VERSION | |
| 0.3 | |
| =head1 COPYRIGHT AND LICENSE | |
| Copyright 2026 Johan Nylander. | |
| Distributed under the MIT License. | |
| =cut | |
| use strict; | |
| use warnings; | |
| use File::Find; | |
| use File::Basename; | |
| use Cwd; | |
| use Getopt::Long; | |
| Getopt::Long::Configure("no_ignore_case", "no_auto_abbrev"); | |
| my $version = '0.3'; | |
| # For eager tsv format settings | |
| my $eager_colour_chemistry = '2'; | |
| my $eager_seqtype = 'PE'; | |
| my $eager_organism = 'NA'; | |
| my $eager_strandedness = 'double'; | |
| my $eager_udg_treatment = 'full'; | |
| my $eager_bam = 'NA'; | |
| # For map_and_call format | |
| my $mapandcall_data_type = "1"; # 1: modern. 2: historical | |
| my $atlas = q{}; | |
| my $eager = q{}; | |
| my $mapandcall = q{}; | |
| my %samples; # $samples{$ngiid}{uid}, and $samples{$ngiid}{libs}{$library}{R1/R2} | |
| my @ngiid_order; # preserve sample sheet order | |
| my @file_paths; | |
| my $cwd = getcwd(); | |
| GetOptions( | |
| 'a|atlas' => \$atlas, | |
| 'e|eager' => \$eager, | |
| 'm|mapandcall' => \$mapandcall, | |
| 'v|version' => sub { print "$version\n"; exit(0); }, | |
| 'h' => sub { print "Usage: $0 [OPTIONS][--help] sample_info_file\n"; exit(0); }, | |
| 'help' => sub { exec("perldoc", $0); exit(0); }, | |
| ) or die("$0 Error in command line arguments\nUsage: $0 [OPTIONS][--help] sample_info_file\n"); | |
| if (@ARGV == 0 && -t STDIN && -t STDERR) { | |
| print "Usage: $0 [OPTIONS][--help] sample_info_file\n"; | |
| exit(1); | |
| } | |
| my $sample_info_file = shift or die "Error: need sample file as argument\n"; | |
| sub find_files { | |
| return unless -f; | |
| return unless /\.fastq\.gz\z/; | |
| push @file_paths, $File::Find::name; | |
| } | |
| # Read sample sheet | |
| open(my $SAMPLE, "<", $sample_info_file) or die "Error: could not open sample file: $sample_info_file\n"; | |
| while (<$SAMPLE>) { | |
| chomp; | |
| next if /^NGI ID\tUser ID\t/ || /^NGI ID\tUser ID$/ || /^NGI ID\b/; | |
| next if /^\s*$/; | |
| my ($ngiid, $userid, @rest) = split /\t/; | |
| next unless defined $ngiid && defined $userid; | |
| push @ngiid_order, $ngiid; | |
| $samples{$ngiid}{uid} = $userid; | |
| $samples{$ngiid}{ngiid} = $ngiid; | |
| } | |
| close $SAMPLE; | |
| # Find FASTQ files recursively from cwd | |
| find(\&find_files, $cwd); | |
| # Expected filename pattern includes "..._<LIB>_R1_001.fastq.gz" etc, where <LIB> = L\d+ | |
| for my $file_path (@file_paths) { | |
| my $matched_any = 0; | |
| for my $ngiid (@ngiid_order) { | |
| next unless $file_path =~ /\Q$ngiid\E/; | |
| $matched_any = 1; | |
| my ($library, $read); | |
| if ($file_path =~ /_(L\d{3})_(R[12])_001\.fastq\.gz\z/) { | |
| ($library, $read) = ($1, $2); | |
| } | |
| else { | |
| warn "Warning: could not parse library/read from file: $file_path\n"; | |
| next; | |
| } | |
| if (exists $samples{$ngiid}{libs}{$library}{$read}) { | |
| warn "Warning: duplicate $read for $ngiid $library:\n", | |
| " existing: $samples{$ngiid}{libs}{$library}{$read}\n", | |
| " new: $file_path\n"; | |
| } | |
| $samples{$ngiid}{libs}{$library}{$read} = $file_path; | |
| } | |
| warn "Warning: no NGI ID match for file: $file_path\n" unless $matched_any; | |
| } | |
| if ($mapandcall) { | |
| print STDOUT "sample_id\tlibrary\tdata_type\tread_1\tread_2\n"; | |
| for my $ngiid (@ngiid_order) { | |
| next unless exists $samples{$ngiid}{libs}; | |
| for my $library (sort keys %{ $samples{$ngiid}{libs} }) { | |
| my $r1 = $samples{$ngiid}{libs}{$library}{R1}; | |
| my $r2 = $samples{$ngiid}{libs}{$library}{R2}; | |
| unless (defined $r1 && defined $r2) { | |
| warn "Warning: incomplete pair for $ngiid $library\n"; | |
| next; | |
| } | |
| print STDOUT join("\t", | |
| $samples{$ngiid}{uid}, | |
| $library, | |
| $mapandcall_data_type, | |
| $r1, | |
| $r2 | |
| ), "\n"; | |
| } | |
| } | |
| } | |
| elsif ($eager) { | |
| print STDOUT "Sample_Name\tLibrary_ID\tLane\tColour_Chemistry\tSeqType\tOrganism\tStrandedness\tUDG_Treatment\tR1\tR2\tBAM\n"; | |
| for my $ngiid (@ngiid_order) { | |
| next unless exists $samples{$ngiid}{libs}; | |
| for my $library (sort keys %{ $samples{$ngiid}{libs} }) { | |
| my $r1 = $samples{$ngiid}{libs}{$library}{R1}; | |
| my $r2 = $samples{$ngiid}{libs}{$library}{R2}; | |
| unless (defined $r1 && defined $r2) { | |
| warn "Warning: incomplete pair for $ngiid $library\n"; | |
| next; | |
| } | |
| my $lane = $library; | |
| $lane =~ s/^L//; | |
| print STDOUT join("\t", | |
| $samples{$ngiid}{uid}, | |
| $ngiid, | |
| $lane, | |
| $eager_colour_chemistry, | |
| $eager_seqtype, | |
| $eager_organism, | |
| $eager_strandedness, | |
| $eager_udg_treatment, | |
| $r1, | |
| $r2, | |
| $eager_bam | |
| ), "\n"; | |
| } | |
| } | |
| } | |
| elsif ($atlas) { | |
| print STDOUT "Sample\tLib\tFile\tRead\tPath\n"; | |
| for my $ngiid (@ngiid_order) { | |
| next unless exists $samples{$ngiid}{libs}; | |
| for my $library (sort keys %{ $samples{$ngiid}{libs} }) { | |
| for my $read (qw/R1 R2/) { | |
| my $fq = $samples{$ngiid}{libs}{$library}{$read}; | |
| unless (defined $fq) { | |
| warn "Warning: missing $read for $ngiid $library\n"; | |
| next; | |
| } | |
| my ($filename, $dirs, $suffix) = fileparse($fq, ".fastq.gz"); | |
| print STDOUT join("\t", | |
| $samples{$ngiid}{uid}, | |
| $library, | |
| $filename, | |
| $read, | |
| $dirs | |
| ), "\n"; | |
| } | |
| } | |
| } | |
| } | |
| else { | |
| # default output | |
| for my $ngiid (@ngiid_order) { | |
| next unless exists $samples{$ngiid}{libs}; | |
| for my $library (sort keys %{ $samples{$ngiid}{libs} }) { | |
| my $r1 = $samples{$ngiid}{libs}{$library}{R1}; | |
| my $r2 = $samples{$ngiid}{libs}{$library}{R2}; | |
| unless (defined $r1 && defined $r2) { | |
| warn "Warning: incomplete pair for $ngiid $library\n"; | |
| next; | |
| } | |
| print STDOUT join("\t", | |
| $ngiid, | |
| $samples{$ngiid}{uid}, | |
| $library, | |
| $r1, | |
| $r2 | |
| ), "\n"; | |
| } | |
| } | |
| } | |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment