#!/usr/bin/env bash # Builds a *filtered* ICU 68 data package with only the converters libmspub # asks for, and turns it into a C source file that the binding registers with # udata_setCommonData(). # # Why: the emscripten ICU port links libicu_stubdata (no data at all), so # ucnv_open("windows-1252") fails with U_FILE_ACCESS_ERROR. libmspub needs # codepage converters for (a) Publisher 97 text (MSPUBParser97, encoding # guessed with ucsdet) and (b) the OLE SummaryInformation metadata # (always "windows-1252"). UTF-16LE (Publisher 2000+ text) is algorithmic and # works without data. # # Converters come from libmspub_utils.cpp windowsCharsetNameByOriginalCharset() # and MSPUBMetaData.cpp. Canonical ICU names (convrtrs.txt, ICU 68): # windows-1250 -> ibm-5346_P100-1998 windows-1251 -> ibm-5347_P100-1998 # windows-1252 -> ibm-5348_P100-1997 windows-1256 -> ibm-9448_X100-2005 # windows-932 -> ibm-943_P15A-2003 windows-936 -> windows-936-2000 # windows-950 -> windows-950-2000 # # make-icudata.sh set -euo pipefail OUT="${1:?outdir}" SRC="$(em-config CACHE)/ports/icu/icu/source/data/in/icudt68l.dat" WORK="$(mktemp -d)" mkdir -p "${OUT}" ITEMS=( cnvalias.icu ibm-5346_P100-1998.cnv ibm-5347_P100-1998.cnv ibm-5348_P100-1997.cnv ibm-9448_X100-2005.cnv ibm-943_P15A-2003.cnv windows-936-2000.cnv windows-950-2000.cnv # base tables the two extension converters above depend on ibm-1386_P100-2001.cnv ibm-1373_P100-2002.cnv ) cd "${WORK}" printf '%s\n' "${ITEMS[@]}" > items.txt # Extract the items from the full package... mkdir -p extracted icupkg -x items.txt -d extracted "${SRC}" # ...and pack them into a new package. The package name must be "icudt68l" # because ICU looks items up as "icudt68l/". icupkg -tl -a items.txt -s extracted new icudt68l.dat icupkg -l icudt68l.dat python3 - icudt68l.dat "${OUT}/icudata_subset.c" <<'EOF' import sys data = open(sys.argv[1], 'rb').read() with open(sys.argv[2], 'w') as f: f.write('/* Generated by icu/make-icudata.sh: filtered ICU 68.2 data (%d bytes). */\n' % len(data)) f.write('/* ICU data is Unicode-DFS-2016 licensed (see README). */\n') f.write('#include \n') f.write('alignas(16) const unsigned char oi_icudata_subset[%d] = {\n' % len(data)) for i in range(0, len(data), 16): f.write(' ' + ','.join(str(b) for b in data[i:i + 16]) + ',\n') f.write('};\n') f.write('const unsigned long oi_icudata_subset_size = %d;\n' % len(data)) print("icudata_subset.c:", len(data), "bytes of ICU data") EOF ls -l "${OUT}"