# Diagnostic only: find the correct way to expand a .cwl.gz word list. # # Two attempts have now produced empty dictionaries: # word-list-compress d -> 0 words (that is the aspell 0.50 tool) # prezip-bin d -> 4 words (wrong invocation) # Guessing a third time is silly. Try every candidate against one known file and # print the line count for each, so the right one is chosen on evidence. # # The reference is en_GB-ise-wo_accents-only.cwl.gz, which should expand to tens # of thousands of words. set -x rm -rf /build/out; mkdir -p /build/out cd /build/aspell-en-wordlists || { echo NO_WORDLISTS; exit 1; } F=en_GB-ise-wo_accents-only.cwl.gz echo "=== the file:" ls -l "$F" gunzip -c "$F" | wc -c | sed 's/^/gz_expands_to_bytes=/' echo "=== first bytes after gunzip:" gunzip -c "$F" | od -c | head -3 echo "=== what aspell installed that could decompress it:" ls -l /usr/bin/prezip* /usr/bin/precat /usr/bin/preunzip /usr/bin/word-list-compress 2>/dev/null echo "=== candidate 1: precat on the .gz directly (what Debian's autobuildhash uses)" /usr/bin/precat "$F" 2>/dev/null | wc -l | sed 's/^/words=/' echo "=== candidate 2: preunzip" gunzip -c "$F" > /tmp/t.cwl /usr/bin/preunzip < /tmp/t.cwl 2>/dev/null | wc -l | sed 's/^/words=/' echo "=== candidate 3: prezip-bin d (already tried, for comparison)" /usr/bin/prezip-bin d < /tmp/t.cwl 2>/dev/null | wc -l | sed 's/^/words=/' echo "=== candidate 4: prezip d" /usr/bin/prezip d < /tmp/t.cwl 2>/dev/null | wc -l | sed 's/^/words=/' echo "=== candidate 5: precat on the uncompressed .cwl" /usr/bin/precat /tmp/t.cwl 2>/dev/null | wc -l | sed 's/^/words=/' echo "=== candidate 6: word-list-compress d (the 0.50 tool, for comparison)" /usr/bin/word-list-compress d < /tmp/t.cwl 2>/dev/null | wc -l | sed 's/^/words=/' echo "=== whichever gave a large count, show its first lines:" /usr/bin/precat "$F" 2>/dev/null | head -5 echo "---" /usr/bin/preunzip < /tmp/t.cwl 2>/dev/null | head -5 echo BUILD_SCRIPT_DONE