Skip to content

Instantly share code, notes, and snippets.

@nylander
Last active September 14, 2026 12:26
Show Gist options
  • Select an option

  • Save nylander/287d1f47c669a350c2e7b97a3da58df5 to your computer and use it in GitHub Desktop.

Select an option

Save nylander/287d1f47c669a350c2e7b97a3da58df5 to your computer and use it in GitHub Desktop.
Create a sample sheet from NGI (SciLifeLab) sequencing delivery
#!/usr/bin/env perl
=pod
=encoding utf8
=head1 NAME
create_sample_sheet.pl - Create sample sheet from NGI delivery
=head1 SYNOPSIS
$ create_sample_sheet.pl 00-Reports/S.OmeName_22_01_sample_info.txt
=head1 DESCRIPTION
Change directory to main delivery directory (e.g. "cd P27213"). Assuming there
is a column with user defined IDs in the file "00-Reports/*_sample_info.txt"
(tab-separated columns), the script will read that file (and column nrs 1 and
2), and then locate corresponding .fastq.gz files.
Prints tab separated entries to STDOUT.
Note: Currently assumes PE libraries, and a file name format similar to this
"P27213_116_S68_L002_R1_001.fastq.gz". The "library" part ("L002") is to be
extracted as well as the "NGI ID" part ("P27213_116").
=head1 OPTIONS
=over 4
=item B<-m,--mapandcall> Output table for the map_and_call workflow (L<https://github.com/axeljen/map_and_call>).
=item B<-a,--atlas> Output table for the ATLAS workflow (L<https://github.com/metagenome-atlas/atlas>).
=item B<-e,--eager> Output table for the nf-core/eager workflow (L<https://nf-co.re/eager/2.4.7/usage#tsv-input-method>).
=item B<-h,--help> Display help.
=item B<-v,--version> Display version.
=back
=head1 AUTHOR
Johan Nylander <johan.nylander@nrm.se>
=head1 VERSION
0.3
=head1 COPYRIGHT AND LICENSE
Copyright 2026 Johan Nylander.
Distributed under the MIT License.
=cut
use strict;
use warnings;
use File::Find;
use File::Basename;
use Cwd;
use Getopt::Long;
Getopt::Long::Configure("no_ignore_case", "no_auto_abbrev");
my $version = '0.3';
# For eager tsv format settings
my $eager_colour_chemistry = '2';
my $eager_seqtype = 'PE';
my $eager_organism = 'NA';
my $eager_strandedness = 'double';
my $eager_udg_treatment = 'full';
my $eager_bam = 'NA';
# For map_and_call format
my $mapandcall_data_type = "1"; # 1: modern. 2: historical
my $atlas = q{};
my $eager = q{};
my $mapandcall = q{};
my %samples; # $samples{$ngiid}{uid}, and $samples{$ngiid}{libs}{$library}{R1/R2}
my @ngiid_order; # preserve sample sheet order
my @file_paths;
my $cwd = getcwd();
GetOptions(
'a|atlas' => \$atlas,
'e|eager' => \$eager,
'm|mapandcall' => \$mapandcall,
'v|version' => sub { print "$version\n"; exit(0); },
'h' => sub { print "Usage: $0 [OPTIONS][--help] sample_info_file\n"; exit(0); },
'help' => sub { exec("perldoc", $0); exit(0); },
) or die("$0 Error in command line arguments\nUsage: $0 [OPTIONS][--help] sample_info_file\n");
if (@ARGV == 0 && -t STDIN && -t STDERR) {
print "Usage: $0 [OPTIONS][--help] sample_info_file\n";
exit(1);
}
my $sample_info_file = shift or die "Error: need sample file as argument\n";
sub find_files {
return unless -f;
return unless /\.fastq\.gz\z/;
push @file_paths, $File::Find::name;
}
# Read sample sheet
open(my $SAMPLE, "<", $sample_info_file) or die "Error: could not open sample file: $sample_info_file\n";
while (<$SAMPLE>) {
chomp;
next if /^NGI ID\tUser ID\t/ || /^NGI ID\tUser ID$/ || /^NGI ID\b/;
next if /^\s*$/;
my ($ngiid, $userid, @rest) = split /\t/;
next unless defined $ngiid && defined $userid;
push @ngiid_order, $ngiid;
$samples{$ngiid}{uid} = $userid;
$samples{$ngiid}{ngiid} = $ngiid;
}
close $SAMPLE;
# Find FASTQ files recursively from cwd
find(\&find_files, $cwd);
# Expected filename pattern includes "..._<LIB>_R1_001.fastq.gz" etc, where <LIB> = L\d+
for my $file_path (@file_paths) {
my $matched_any = 0;
for my $ngiid (@ngiid_order) {
next unless $file_path =~ /\Q$ngiid\E/;
$matched_any = 1;
my ($library, $read);
if ($file_path =~ /_(L\d{3})_(R[12])_001\.fastq\.gz\z/) {
($library, $read) = ($1, $2);
}
else {
warn "Warning: could not parse library/read from file: $file_path\n";
next;
}
if (exists $samples{$ngiid}{libs}{$library}{$read}) {
warn "Warning: duplicate $read for $ngiid $library:\n",
" existing: $samples{$ngiid}{libs}{$library}{$read}\n",
" new: $file_path\n";
}
$samples{$ngiid}{libs}{$library}{$read} = $file_path;
}
warn "Warning: no NGI ID match for file: $file_path\n" unless $matched_any;
}
if ($mapandcall) {
print STDOUT "sample_id\tlibrary\tdata_type\tread_1\tread_2\n";
for my $ngiid (@ngiid_order) {
next unless exists $samples{$ngiid}{libs};
for my $library (sort keys %{ $samples{$ngiid}{libs} }) {
my $r1 = $samples{$ngiid}{libs}{$library}{R1};
my $r2 = $samples{$ngiid}{libs}{$library}{R2};
unless (defined $r1 && defined $r2) {
warn "Warning: incomplete pair for $ngiid $library\n";
next;
}
print STDOUT join("\t",
$samples{$ngiid}{uid},
$library,
$mapandcall_data_type,
$r1,
$r2
), "\n";
}
}
}
elsif ($eager) {
print STDOUT "Sample_Name\tLibrary_ID\tLane\tColour_Chemistry\tSeqType\tOrganism\tStrandedness\tUDG_Treatment\tR1\tR2\tBAM\n";
for my $ngiid (@ngiid_order) {
next unless exists $samples{$ngiid}{libs};
for my $library (sort keys %{ $samples{$ngiid}{libs} }) {
my $r1 = $samples{$ngiid}{libs}{$library}{R1};
my $r2 = $samples{$ngiid}{libs}{$library}{R2};
unless (defined $r1 && defined $r2) {
warn "Warning: incomplete pair for $ngiid $library\n";
next;
}
my $lane = $library;
$lane =~ s/^L//;
print STDOUT join("\t",
$samples{$ngiid}{uid},
$ngiid,
$lane,
$eager_colour_chemistry,
$eager_seqtype,
$eager_organism,
$eager_strandedness,
$eager_udg_treatment,
$r1,
$r2,
$eager_bam
), "\n";
}
}
}
elsif ($atlas) {
print STDOUT "Sample\tLib\tFile\tRead\tPath\n";
for my $ngiid (@ngiid_order) {
next unless exists $samples{$ngiid}{libs};
for my $library (sort keys %{ $samples{$ngiid}{libs} }) {
for my $read (qw/R1 R2/) {
my $fq = $samples{$ngiid}{libs}{$library}{$read};
unless (defined $fq) {
warn "Warning: missing $read for $ngiid $library\n";
next;
}
my ($filename, $dirs, $suffix) = fileparse($fq, ".fastq.gz");
print STDOUT join("\t",
$samples{$ngiid}{uid},
$library,
$filename,
$read,
$dirs
), "\n";
}
}
}
}
else {
# default output
for my $ngiid (@ngiid_order) {
next unless exists $samples{$ngiid}{libs};
for my $library (sort keys %{ $samples{$ngiid}{libs} }) {
my $r1 = $samples{$ngiid}{libs}{$library}{R1};
my $r2 = $samples{$ngiid}{libs}{$library}{R2};
unless (defined $r1 && defined $r2) {
warn "Warning: incomplete pair for $ngiid $library\n";
next;
}
print STDOUT join("\t",
$ngiid,
$samples{$ngiid}{uid},
$library,
$r1,
$r2
), "\n";
}
}
}
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment