/* libimap library. * Copyright (C) 2003-2004 Pawel Salek. * * This program is free software; you can redistribute it and/or modify * it under the terms of the GNU General Public License as published by * the Free Software Foundation; either version 2, or (at your option) * any later version. * * This program is distributed in the hope that it will be useful, * but WITHOUT ANY WARRANTY; without even the implied warranty of * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the * GNU General Public License for more details. * * You should have received a copy of the GNU General Public License * along with this program; if not, write to the Free Software * Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA * 02111-1307, USA. */ /* imap_quote_string: quote string according to IMAP rules: * surround string with quotes, escape " and \ with \ */ #include "config.h" #include #include #include #include #include "util.h" #define SKIPWS(c) while (*(c) && isspace ((unsigned char) *(c))) c++; void imap_quote_string(char *dest, size_t dlen, const char *src) { char quote[] = "\"\\", *pt; const char *s; pt = dest; s = src; *pt++ = '"'; /* save room for trailing quote-char */ dlen -= 2; for (; *s && dlen; s++) { if (strchr (quote, *s)) { dlen -= 2; if (!dlen) break; *pt++ = '\\'; *pt++ = *s; } else { *pt++ = *s; dlen--; } } *pt++ = '"'; *pt = 0; } /* imap_unquote_string: * modify the argument removing quote characters */ void imap_unquote_string (char *s) { char *d = s; if (*s == '\"') s++; else return; while (*s) { if (*s == '\"') { *d = '\0'; return; } if (*s == '\\') { s++; } if (*s) { *d = *s; d++; s++; } } *d = '\0'; } /* imap_skip_atom: return index into string where next IMAP atom begins. * see RFC for a definition of the atom. */ #define LIST_SPECIALS "%*" #define QUOTED_SPECIALS "\"\\" #define RESP_SPECIALS "]" char* imap_skip_atom(char *s) { static const char ATOM_SPECIALS[] = "(){ " LIST_SPECIALS QUOTED_SPECIALS RESP_SPECIALS; while (*s && !strchr(ATOM_SPECIALS,*s)) s++; SKIPWS (s); return s; } char* imap_next_word(char *s) { int quoted = 0; while (*s) { if (*s == '\\') { s++; if (*s) s++; continue; } if (*s == '\"') quoted = quoted ? 0 : 1; if (!quoted && isspace(*s)) break; s++; } SKIPWS (s); return s; } #define BAD -1 #define base64val(c) Index_64[(unsigned int)(c)] static char B64Chars[64] = { 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '+', '/' }; static int Index_64[128] = { -1,-1,-1,-1, -1,-1,-1,-1, -1,-1,-1,-1, -1,-1,-1,-1, -1,-1,-1,-1, -1,-1,-1,-1, -1,-1,-1,-1, -1,-1,-1,-1, -1,-1,-1,-1, -1,-1,-1,-1, -1,-1,-1,62, -1,-1,-1,63, 52,53,54,55, 56,57,58,59, 60,61,-1,-1, -1,-1,-1,-1, -1, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9,10, 11,12,13,14, 15,16,17,18, 19,20,21,22, 23,24,25,-1, -1,-1,-1,-1, -1,26,27,28, 29,30,31,32, 33,34,35,36, 37,38,39,40, 41,42,43,44, 45,46,47,48, 49,50,51,-1, -1,-1,-1,-1 }; /* raw bytes to null-terminated base 64 string */ void lit_conv_to_base64(char *out, const char *sin, size_t len, size_t olen) { const unsigned char *in = sin; while (len >= 3 && olen > 10) { *out++ = B64Chars[in[0] >> 2]; *out++ = B64Chars[((in[0] << 4) & 0x30) | (in[1] >> 4)]; *out++ = B64Chars[((in[1] << 2) & 0x3c) | (in[2] >> 6)]; *out++ = B64Chars[in[2] & 0x3f]; olen -= 4; len -= 3; in += 3; } /* clean up remainder */ if (len > 0 && olen > 4) { unsigned char fragment; *out++ = B64Chars[in[0] >> 2]; fragment = (in[0] << 4) & 0x30; if (len > 1) fragment |= in[1] >> 4; *out++ = B64Chars[fragment]; *out++ = (len < 2) ? '=' : B64Chars[(in[1] << 2) & 0x3c]; *out++ = '='; } *out = '\0'; } /* Convert '\0'-terminated base 64 string to raw bytes. * Returns length of returned buffer, or -1 on error */ int lit_conv_from_base64(char *out, const char *in) { int len = 0, idx=0; register unsigned char digit1, digit2, digit3, digit4; do { digit1 = in[idx]; if (digit1 > 127 || base64val (digit1) == BAD) return -1-idx; digit2 = in[idx+1]; if (digit2 > 127 || base64val (digit2) == BAD) return -2-idx; digit3 = in[idx+2]; if (digit3 > 127 || ((digit3 != '=') && (base64val (digit3) == BAD))) return -3-idx; digit4 = in[idx+3]; if (digit4 > 127 || ((digit4 != '=') && (base64val (digit4) == BAD))) return -4-idx; idx += 4; /* digits are already sanity-checked */ *out++ = (base64val(digit1) << 2) | (base64val(digit2) >> 4); len++; if (digit3 != '=') { *out++ = ((base64val(digit2) << 4) & 0xf0) | (base64val(digit3) >> 2); len++; if (digit4 != '=') { *out++ = ((base64val(digit3) << 6) & 0xc0) | base64val(digit4); len++; } } } while (in[idx] && in[idx] != 13 && digit4 != '='); return len; } /* =================================================================== * UTF-7 conversion routines as in RFC 2192 * =================================================================== */ /* UTF7 modified base64 alphabet */ static char base64chars[] = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+,"; #define UNDEFINED 64 /* UTF16 definitions */ #define UTF16MASK 0x03FFUL #define UTF16SHIFT 10 #define UTF16BASE 0x10000UL #define UTF16HIGHSTART 0xD800UL #define UTF16HIGHEND 0xDBFFUL #define UTF16LOSTART 0xDC00UL #define UTF16LOEND 0xDFFFUL /* Convert an IMAP mailbox to a UTF-8 string. * dst needs to have roughly 4 times the storage space of src * Hex encoding can triple the size of the input * UTF-7 can be slightly denser than UTF-8 * (worst case: 8 octets UTF-7 becomes 9 octets UTF-8) */ char* imap_mailbox_to_utf8(const unsigned char *mbox) { unsigned c, i, bitcount; unsigned long ucs4, utf16, bitbuf; unsigned char base64[256]; const unsigned char *src; char *dst, *res = malloc(2*strlen(mbox)+1); bitbuf = 0; dst = res; src = mbox; if(!dst) return NULL; /* initialize modified base64 decoding table */ memset(base64, UNDEFINED, sizeof (base64)); for (i = 0; i < sizeof (base64chars); ++i) { base64[(unsigned)base64chars[i]] = i; } /* loop until end of string */ while (*src != '\0') { c = *src++; /* deal with literal characters and &- */ if (c != '&' || *src == '-') { /* encode literally */ *dst++ = c; /* skip over the '-' if this is an &- sequence */ if (c == '&') ++src; } else { /* convert modified UTF-7 -> UTF-16 -> UCS-4 -> UTF-8 -> HEX */ bitbuf = 0; bitcount = 0; ucs4 = 0; while ((c = base64[(unsigned char) *src]) != UNDEFINED) { ++src; bitbuf = (bitbuf << 6) | c; bitcount += 6; /* enough bits for a UTF-16 character? */ if (bitcount >= 16) { bitcount -= 16; utf16 = (bitcount ? bitbuf >> bitcount : bitbuf) & 0xffff; /* convert UTF16 to UCS4 */ if (utf16 >= UTF16HIGHSTART && utf16 <= UTF16HIGHEND) { ucs4 = (utf16 - UTF16HIGHSTART) << UTF16SHIFT; continue; } else if (utf16 >= UTF16LOSTART && utf16 <= UTF16LOEND) { ucs4 += utf16 - UTF16LOSTART + UTF16BASE; } else { ucs4 = utf16; } /* convert UTF-16 range of UCS4 to UTF-8 */ if (ucs4 <= 0x7fUL) { dst[0] = ucs4; dst += 1; } else if (ucs4 <= 0x7ffUL) { dst[0] = 0xc0 | (ucs4 >> 6); dst[1] = 0x80 | (ucs4 & 0x3f); dst += 2; } else if (ucs4 <= 0xffffUL) { dst[0] = 0xe0 | (ucs4 >> 12); dst[1] = 0x80 | ((ucs4 >> 6) & 0x3f); dst[2] = 0x80 | (ucs4 & 0x3f); dst += 3; } else { dst[0] = 0xf0 | (ucs4 >> 18); dst[1] = 0x80 | ((ucs4 >> 12) & 0x3f); dst[2] = 0x80 | ((ucs4 >> 6) & 0x3f); dst[3] = 0x80 | (ucs4 & 0x3f); dst += 4; } } } /* skip over trailing '-' in modified UTF-7 encoding */ if (*src == '-') ++src; } } /* terminate destination string */ *dst = '\0'; return res; } /* Convert hex coded UTF-8 string to modified UTF-7 IMAP mailbox * dst should be about twice the length of src to deal with non-hex * coded URLs */ char* imap_utf8_to_mailbox(const unsigned char *src) { unsigned int utf8pos, utf8total, c, utf7mode, bitstogo, utf16flag; unsigned long ucs4 = 0, bitbuf = 0; /* initialize hex lookup table */ unsigned char *dst, *res = malloc(2*strlen(src)+1); dst = res; if(!dst) return NULL; utf7mode = 0; utf8total = 0; bitstogo = 0; utf8pos = 0; while ((c = *src) != '\0') { ++src; /* normal character? */ if (c >= ' ' && c <= '~') { /* switch out of UTF-7 mode */ if (utf7mode) { if (bitstogo) { *dst++ = base64chars[(bitbuf << (6 - bitstogo)) & 0x3F]; } *dst++ = '-'; utf7mode = 0; utf8pos = 0; bitstogo = 0; utf8total= 0; } *dst++ = c; /* encode '&' as '&-' */ if (c == '&') { *dst++ = '-'; } continue; } /* switch to UTF-7 mode */ if (!utf7mode) { *dst++ = '&'; utf7mode = 1; } /* Encode US-ASCII characters as themselves */ if (c < 0x80) { ucs4 = c; utf8total = 1; } else if (utf8total) { /* save UTF8 bits into UCS4 */ ucs4 = (ucs4 << 6) | (c & 0x3FUL); if (++utf8pos < utf8total) { continue; } } else { utf8pos = 1; if (c < 0xE0) { utf8total = 2; ucs4 = c & 0x1F; } else if (c < 0xF0) { utf8total = 3; ucs4 = c & 0x0F; } else { /* NOTE: can't convert UTF8 sequences longer than 4 */ utf8total = 4; ucs4 = c & 0x03; } continue; } /* loop to split ucs4 into two utf16 chars if necessary */ utf8total = 0; do { if (ucs4 >= UTF16BASE) { ucs4 -= UTF16BASE; bitbuf = (bitbuf << 16) | ((ucs4 >> UTF16SHIFT) + UTF16HIGHSTART); ucs4 = (ucs4 & UTF16MASK) + UTF16LOSTART; utf16flag = 1; } else { bitbuf = (bitbuf << 16) | ucs4; utf16flag = 0; } bitstogo += 16; /* spew out base64 */ while (bitstogo >= 6) { bitstogo -= 6; *dst++ = base64chars[(bitstogo ? (bitbuf >> bitstogo) : bitbuf) & 0x3F]; } } while (utf16flag); } /* if in UTF-7 mode, finish in ASCII */ if (utf7mode) { if (bitstogo) { *dst++ = base64chars[(bitbuf << (6 - bitstogo)) & 0x3F]; } *dst++ = '-'; } /* tie off string */ *dst = '\0'; return res; } #if 0 int main(int argc, char *argv[]) { int i; for(i=1; i