#!/bin/bash
# ============================================================
# sample_chr22_blocks.sh
#
# Randomly selects N haploblocks from chr22 and copies each
# block's three files (rep_seq.fasta, all_seqs.fasta,
# cluster.tsv) into a small dataset directory.
#
# Usage:
#   ./sample_chr22_blocks.sh block_stats.tsv results_base_dir dest_dir [n_blocks] [seed]
#
#   block_stats.tsv     used only to get the list of distinct chr22 block names
#   results_base_dir     directory containing chr22/clusters/<block>_*.{fasta,tsv}
#   dest_dir              where the sampled blocks' files are copied
#   n_blocks               how many blocks to sample (default 12)
#   seed                    optional integer for reproducible sampling
#                           (omit for a different random sample each run)
#
# Requires only: bash, awk, shuf, cp, wc. No python, no jq.
# ============================================================

set -euo pipefail

BLOCK_STATS="${1:?Usage: $0 block_stats.tsv results_base_dir dest_dir [n_blocks] [seed]}"
RESULTS_BASE="${2:?Usage: $0 block_stats.tsv results_base_dir dest_dir [n_blocks] [seed]}"
DEST="${3:?Usage: $0 block_stats.tsv results_base_dir dest_dir [n_blocks] [seed]}"
N_BLOCKS="${4:-12}"
SEED="${5:-}"

if [ ! -f "$BLOCK_STATS" ]; then
    echo "Error: file not found: $BLOCK_STATS" >&2
    exit 1
fi

SRC="$RESULTS_BASE/chr22/clusters"
if [ ! -d "$SRC" ]; then
    echo "Error: directory not found: $SRC" >&2
    exit 1
fi

mkdir -p "$DEST"

WORK=$(mktemp -d)
trap 'rm -rf "$WORK"' EXIT

# distinct chr22 block names, in file order
awk -F'\t' 'NR>1 && $1=="chr22" {print $2}' "$BLOCK_STATS" | sort -u > "$WORK/all_chr22_blocks.txt"

TOTAL_BLOCKS=$(wc -l < "$WORK/all_chr22_blocks.txt")
if [ "$TOTAL_BLOCKS" -eq 0 ]; then
    echo "Error: no chr22 blocks found in $BLOCK_STATS" >&2
    exit 1
fi

if [ -n "$SEED" ]; then
    # reproducible sample: seed a repeatable pseudo-random byte stream for shuf
    shuf --random-source=<(awk -v seed="$SEED" 'BEGIN{srand(seed); while(1) print srand()}') \
        -n "$N_BLOCKS" "$WORK/all_chr22_blocks.txt" > "$WORK/sampled_blocks.txt"
else
    shuf -n "$N_BLOCKS" "$WORK/all_chr22_blocks.txt" > "$WORK/sampled_blocks.txt"
fi

N_SAMPLED=$(wc -l < "$WORK/sampled_blocks.txt")
echo "chr22 has $TOTAL_BLOCKS distinct haploblocks; sampled $N_SAMPLED."
echo ""

while read -r block; do
    echo "Copying $block ..."
    for suffix in rep_seq.fasta all_seqs.fasta cluster.tsv; do
        src_file="$SRC/${block}_${suffix}"
        if [ -f "$src_file" ]; then
            cp "$src_file" "$DEST/"
        else
            echo "  WARNING: missing $src_file" >&2
        fi
    done
done < "$WORK/sampled_blocks.txt"

echo ""
echo "============================================================"
echo "Done. Sampled blocks copied to: $DEST"
echo ""
echo "Blocks sampled:"
cat "$WORK/sampled_blocks.txt"
echo "============================================================"
