diff --git a/SOAP/compression/make_virtual_snapshot.py b/SOAP/compression/make_virtual_snapshot.py index ea30ea1c..6f2b673a 100644 --- a/SOAP/compression/make_virtual_snapshot.py +++ b/SOAP/compression/make_virtual_snapshot.py @@ -79,11 +79,43 @@ def make_virtual_snapshot( absolute_paths: If True, use absolute paths; if False, use relative paths """ - # Copy the input virtual snapshot to the output - shutil.copyfile(snapshot, output_file) + # Check whether the input snapshot is a virtual file (distributed output) + # or a single file containing all the particle data (serial output) + with h5py.File(snapshot, "r") as infile: + is_virtual = infile["Header"].attrs.get("Virtual", [1])[0] == 1 - # Open the output file - outfile = h5py.File(output_file, "r+") + if is_virtual: + # Copy the input virtual snapshot to the output + shutil.copyfile(snapshot, output_file) + + # Open the output file + outfile = h5py.File(output_file, "r+") + else: + # Avoid copying the particle data by creating virtual datasets + # which link to the input snapshot + outfile = h5py.File(output_file, "w") + with h5py.File(snapshot, "r") as infile: + for k, v in infile.attrs.items(): + outfile.attrs[k] = v + for name in infile: + # Keep soft links (e.g. GasParticles -> PartType0) as links + link = infile.get(name, getlink=True) + if isinstance(link, h5py.SoftLink): + outfile[name] = h5py.SoftLink(link.path) + continue + # Copy groups with metadata + if not name.startswith("PartType"): + infile.copy(infile[name], outfile, name=name) + continue + group = outfile.create_group(name) + for k, v in infile[name].attrs.items(): + group.attrs[k] = v + for dset_name, dset in infile[name].items(): + layout = h5py.VirtualLayout(shape=dset.shape, dtype=dset.dtype) + layout[...] = h5py.VirtualSource(dset) + vdset = group.create_virtual_dataset(dset_name, layout) + for k, v in dset.attrs.items(): + vdset.attrs[k] = v # Calculate directories for path updates abs_snapshot_dir = os.path.abspath(os.path.dirname(snapshot)) @@ -198,6 +230,9 @@ def replace_path(old_path): else: break file_nr += 1 + # Serial output, so there is only a single auxiliary file + if "{file_nr}" not in auxiliary: + break if file_nr == 0: raise IOError(f"Failed to find files matching: {auxiliary}") diff --git a/scripts/COLIBRE/compress_group_membership.sh b/scripts/COLIBRE/compress_group_membership.sh index f04ec4c8..6a3c9eb9 100644 --- a/scripts/COLIBRE/compress_group_membership.sh +++ b/scripts/COLIBRE/compress_group_membership.sh @@ -55,7 +55,7 @@ sim="${SLURM_JOB_NAME}" inbase="${scratch_dir}/${sim}/SOAP_uncompressed/" # Location of the compressed output -outbase="${output_dir}/${sim}/SOAP/" +outbase="${output_dir}/${sim}/SOAP-HBT/" # Create the output folder if it does not exist outdir="${outbase}/membership_${snapnum}" @@ -67,28 +67,39 @@ input_filename="${inbase}/membership_${snapnum}/membership_${snapnum}" # Compressed membership file basename output_filename="${outbase}/membership_${snapnum}/membership_${snapnum}" -# Determine how many files we have -nr_files=`ls -1 ${input_filename}.*.hdf5 | wc -l` -nr_files_minus_one=$(( ${nr_files} - 1 )) +# Determine how many chunk files we have +if [[ -f ${input_filename}.0.hdf5 ]] ; then + nr_files=`ls -1 ${input_filename}.*.hdf5 | wc -l` +else + nr_files=0 +fi # run h5repack in parallel using 32 processes on files 0 to 63 # we could use more processes, but that causes a larger strain for the file # system and is therefore not really more efficient # make sure to update the 'seq' arguments when there are more/less membership # files -echo Compressing ${nr_files} group membership files -echo Source : ${input_filename} -echo Destination: ${output_filename} - -seq 0 ${nr_files_minus_one} | xargs -I {} -P 32 bash -c \ - "h5repack -i ${input_filename}.{}.hdf5 -o ${output_filename}.{}.hdf5 -l CHUNK=10000 -f GZIP=4" +echo "Source : ${input_filename}" +echo "Destination: ${output_filename}" + +if [[ ${nr_files} -eq 0 ]] ; then + # Serial output, so there is a single membership file + echo "Compressing single group membership file" + h5repack -i ${input_filename}.hdf5 -o ${output_filename}.hdf5 -l CHUNK=10000 -f GZIP=4 + membership="${output_filename}.hdf5" +else + echo "Compressing ${nr_files} group membership files" + nr_files_minus_one=$(( ${nr_files} - 1 )) + seq 0 ${nr_files_minus_one} | xargs -I {} -P 32 bash -c \ + "h5repack -i ${input_filename}.{}.hdf5 -o ${output_filename}.{}.hdf5 -l CHUNK=10000 -f GZIP=4" + membership="${output_filename}.{file_nr}.hdf5" +fi echo "Setting files to be read-only" chmod a=r "${output_filename}"* echo "Creating virtual snapshot" snapshot="${output_dir}/${sim}/snapshots/colibre_${snapnum}/colibre_${snapnum}.hdf5" -membership="${output_filename}.{file_nr}.hdf5" virtual="${outbase}/colibre_with_SOAP_membership_${snapnum}.hdf5" python SOAP/compression/make_virtual_snapshot.py \ --virtual-snapshot $snapshot \