#!/usr/bin/env bash
# Fetch the authors' H3Nx artifacts for reproduction (run from REPRODUCTION/).
# Source of truth: github.com/moncla-lab/h3nx-paper (+ treesort-pipeline for reassortment outputs).
# We reuse THEIR per-host subtree alignments/trees so we test their inference, not a re-subsampled dataset.
set -euo pipefail
cd "$(dirname "$0")/.."   # -> REPRODUCTION/

mkdir -p data/{ha,na,pb1,treesort} _repos
cd _repos

# 1. Paper repo (alignments, trees, host labels, analysis code)
[ -d h3nx-paper ] || git clone --depth 1 https://github.com/moncla-lab/h3nx-paper.git
# 2. TreeSort replicate pipeline (summary reassortment trees + support JSON, reproduction harness)
[ -d treesort-pipeline ] || git clone --depth 1 https://github.com/moncla-lab/treesort-pipeline.git

cd ..
echo
echo "== Cloned. Inspect layout, then wire the per-host HA/NA/PB1 alignments+trees into data/ =="
echo "   (paths inside h3nx-paper are not yet pinned here — confirm dir names before symlinking):"
find _repos/h3nx-paper -maxdepth 3 -type d | sed 's/^/   /' | head -40
echo
echo "TODO (manual, once layout confirmed):"
echo "  ln -s \$PWD/_repos/h3nx-paper/<align_dir>/HA/<host>.fasta data/ha/<host>.fasta   # etc."
echo "  cp    _repos/h3nx-paper/<treesort_dir>/*.json data/treesort/"
echo
echo "GISAID isolates (Supp Table 2) are acknowledgment-only; sequence pulls need a GISAID login —"
echo "the paper's public NCBI+repo artifacts should cover the reproducible alignments."
