【问题标题】:Use this char encoding function with only one parameter仅使用一个参数使用此 char 编码函数
【发布时间】:2016-11-30 02:01:57
【问题描述】:

经过多次搜索,我找到了适合我需要的完美功能here

代码如下:

/* UTF-8 to ISO-8859-1/ISO-8859-15 mapper.
 * Return 0..255 for valid ISO-8859-15 code points, 256 otherwise.
*/
static inline unsigned int to_latin9(const unsigned int code)
{
    //printf("\ncode = %d", code);
    /* Code points 0 to U+00FF are the same in both. */
    if (code < 256U) {
        return code;
    } 
    switch (code) {
    case 0x0152U: return 188U; /* U+0152 = 0xBC: OE ligature */
    case 0x0153U: return 189U; /* U+0153 = 0xBD: oe ligature */
    case 0x0160U: return 166U; /* U+0160 = 0xA6: S with caron */
    case 0x0161U: return 168U; /* U+0161 = 0xA8: s with caron */
    case 0x0178U: return 190U; /* U+0178 = 0xBE: Y with diaresis */
    case 0x017DU: return 180U; /* U+017D = 0xB4: Z with caron */
    case 0x017EU: return 184U; /* U+017E = 0xB8: z with caron */
    case 0x20ACU: return 164U; /* U+20AC = 0xA4: Euro */
    default:      return 256U;
    }
}

/* Convert an UTF-8 string to ISO-8859-15.
 * All invalid sequences are ignored.
 * Note: output == input is allowed,
 * but   input < output < input + length
 * is not.
 * Output has to have room for (length+1) chars, including the trailing NUL byte.
 */
size_t utf8_to_latin9(char *const output, const char *const input, const size_t length)
{
    unsigned char             *out = (unsigned char *)output;
    const unsigned char       *in  = (const unsigned char *)input;
    const unsigned char *const end = (const unsigned char *)input + length;
    unsigned int               c;

    while (in < end)
        if (*in < 128)
            *(out++) = *(in++); /* Valid codepoint */
        else
        if (*in < 192)
            in++;               /* 10000000 .. 10111111 are invalid */
        else
        if (*in < 224) {        /* 110xxxxx 10xxxxxx */
            if (in + 1 >= end)
                break;
            if ((in[1] & 192U) == 128U) {
                c = to_latin9( (((unsigned int)(in[0] & 0x1FU)) << 6U)
                             |  ((unsigned int)(in[1] & 0x3FU)) );
                if (c < 256)
                    *(out++) = c;
            }
            in += 2;

        } else
        if (*in < 240) {        /* 1110xxxx 10xxxxxx 10xxxxxx */
            if (in + 2 >= end)
                break;
            if ((in[1] & 192U) == 128U &&
                (in[2] & 192U) == 128U) {
                c = to_latin9( (((unsigned int)(in[0] & 0x0FU)) << 12U)
                             | (((unsigned int)(in[1] & 0x3FU)) << 6U)
                             |  ((unsigned int)(in[2] & 0x3FU)) );
                if (c < 256)
                    *(out++) = c;
            }
            in += 3;

        } else
        if (*in < 248) {        /* 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx */
            if (in + 3 >= end)
                break;
            if ((in[1] & 192U) == 128U &&
                (in[2] & 192U) == 128U &&
                (in[3] & 192U) == 128U) {
                c = to_latin9( (((unsigned int)(in[0] & 0x07U)) << 18U)
                             | (((unsigned int)(in[1] & 0x3FU)) << 12U)
                             | (((unsigned int)(in[2] & 0x3FU)) << 6U)
                             |  ((unsigned int)(in[3] & 0x3FU)) );
                if (c < 256)
                    *(out++) = c;
            }
            in += 4;

        } else
        if (*in < 252) {        /* 111110xx 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx */
            if (in + 4 >= end)
                break;
            if ((in[1] & 192U) == 128U &&
                (in[2] & 192U) == 128U &&
                (in[3] & 192U) == 128U &&
                (in[4] & 192U) == 128U) {
                c = to_latin9( (((unsigned int)(in[0] & 0x03U)) << 24U)
                             | (((unsigned int)(in[1] & 0x3FU)) << 18U)
                             | (((unsigned int)(in[2] & 0x3FU)) << 12U)
                             | (((unsigned int)(in[3] & 0x3FU)) << 6U)
                             |  ((unsigned int)(in[4] & 0x3FU)) );
                if (c < 256)
                    *(out++) = c;
            }
            in += 5;

        } else
        if (*in < 254) {        /* 1111110x 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx 10xxxxxx */
            if (in + 5 >= end)
                break;
            if ((in[1] & 192U) == 128U &&
                (in[2] & 192U) == 128U &&
                (in[3] & 192U) == 128U &&
                (in[4] & 192U) == 128U &&
                (in[5] & 192U) == 128U) {
                c = to_latin9( (((unsigned int)(in[0] & 0x01U)) << 30U)
                             | (((unsigned int)(in[1] & 0x3FU)) << 24U)
                             | (((unsigned int)(in[2] & 0x3FU)) << 18U)
                             | (((unsigned int)(in[3] & 0x3FU)) << 12U)
                             | (((unsigned int)(in[4] & 0x3FU)) << 6U)
                             |  ((unsigned int)(in[5] & 0x3FU)) );
                if (c < 256)
                    *(out++) = c;
            }
            in += 6;

        } else
            in++;               /* 11111110 and 11111111 are invalid */

    /* Terminate the output string. */
    *out = '\0';

    return (size_t)(out - (unsigned char *)output);
}

这工作完美。

但是这个函数需要 1 个 buffer_input 并返回我第二个 buffer_output。 为了优化它,我只想使用一个buffer_input,函数返回给我修改后的buffer_input

我不知道这是否可能。我自己没有足够的知识来做这件事(我很难理解这段代码)。

很难为我想要实现的目的编辑此代码?

目前我是这样使用的:

utf8_to_latin9(buffer_output, buffer_input, length)

编辑后我想这样使用:

utf8_to_latin9(buffer_input, buffer_input, length)

utf8_to_latin9(buffer_input, length)

【问题讨论】:

  • 注:输出==输入是允许的
  • 是的,我看到了,但尝试后它不起作用:/
  • 我好累。没有看到utf8_to_latin9 函数。删除了我的答案。对不起。
  • 删除了我以前的 cmets 以将手指放在代码中断的点上:如果 UTF-8 输入包含有效的 Latin1 字符,但也不是有效的 Latin9 字符(¤¦¨´¸¼½¾ 中的任何一个) ,它们将被“转换”为具有相同值的 Latin9 对应项(而不是给出 256 转换错误)。 \u00a4\u20ac 都将产生相同的输出,0xa4,这对于后者是正确的,但对于前者是错误的。 -- 我推荐 ICU library 来满足你所有的 Unicode 需求,而不是尝试自己动手。
  • @n.m. 1.因为正如你所说,它需要一个图书馆。 2.我从事嵌入式计算,使用1个缓冲区而不是2个是不可忽略的。

标签: c encoding


【解决方案1】:

没有更正(或检查)您的代码是否存在任何问题,我只是将其修改为就地工作。请注意我对utf8_to_latin9() 中错误转换的评论。

转换为就地操作真的很简单:

切换到一个缓冲区参数,使out指向输入缓冲区的开头(而不是单独的输出缓冲区):

< size_t utf8_to_latin9(char *const output, const char *const input, const size_t length)
< {
<     unsigned char             *out = (unsigned char *)output;
----
> size_t utf8_to_latin9(char *const input, const size_t length)
> {
>     unsigned char             *out = (unsigned char *)input;

在转换结束时,根据输入缓冲区(而不是单独的输出缓冲区)计算返回值:

< return (size_t)(out - (unsigned char *)output);
----
> return (size_t)(out - (unsigned char *)input);

由于 UTF-8 输入缓冲区的消耗速度不可能比 Latin9 输出缓冲区的填充速度,所以这是非常安全的。


我的测试main()

int main()
{
    char input[256] = {0};
                    /*  ¤ *//*    €   *//* some kanji */
    strcpy( input, "\xc2\xa4\xe2\x82\xac\xf0\xaf\xa4\xa9" );
    utf8_to_latin9( input, 9 );
    printf( "%hhx ", input[0] );
    printf( "%hhx ", input[1] );
    printf( "%hhx ", input[2] );
    printf( "%hhx ", input[3] );
    printf( "%hhx ", input[4] );
    printf( "%hhx ", input[5] );
    printf( "%hhx ", input[6] );
    printf( "%hhx ", input[7] );
    printf( "%hhx\n", input[8] );
    return 0;
}

输出:

a4 a4 0 82 ac f0 af a4 a9

这正是我所期望的(展示了我评论过的转换问题)。

【讨论】:

  • 这样写utf8_to_latin9(char *const input, const char *const input, const size_t)还是不行,不知道为什么
  • @A.Rossi:“它不起作用”不是,从来没有,也永远不会是正确的故障分析。您的代码缺少 main() 向函数提供测试输入。我写了一个,它成功地达到了预期的输出,所以我认为这个案例已经结束了。如果您还有其他问题,请提供 MCVE、观察到的输出和预期输出。
  • @A.Rossi:添加了我的测试代码。请说明您的结果和/或期望有何不同。
  • 它已按原样运行,请参阅评论Note: output == input is allowed
猜你喜欢
  • 1970-01-01
  • 1970-01-01
  • 1970-01-01
  • 2021-08-13
  • 2019-12-09
  • 1970-01-01
  • 2018-02-15
  • 2018-11-13
  • 1970-01-01
相关资源
最近更新 更多