Версия для печати темы
Нажмите сюда для просмотра этой темы в оригинальном формате
Форум программистов > C/C++: Для новичков > преобразование ascii в unicode


Автор: nitrexin 10.12.2008, 22:17
Пишу курсовик, и не знаю что делать дальше. Задание: открыть текстовый файл, и преобразовать из dos кодировки в unicode. При этом есть условие, не использовать стороних библиотек.
У меня не получается открыть текстовый файл, и прочесть его в массив, а если использовать строку в самом массиве, то на вывод она идет без слова в dos кодировке. Одним словом нужна помощь smile
Код

#include <stdio.h>
#include <stdlib.h>
unsigned char* ASCIItoUNICODE (unsigned char ch);
unsigned int* ConvertString (unsigned char *string);
void UnicodePrint(unsigned int* Message);
//прототипы

int main()
{

FILE *stream;
//пытаемся открыть текстовый файл
if((stream = fopen("text.txt","rb")) == NULL){ printf("! ");return;}
fread(&c,2,1,stream);
//и прочесть его в массив text
fread(text,2,c,stream); 
unsigned char text[100];
//unsigned char text[] = "привет,world" ;
//привет написано в dos кодировке
unsigned int *UnMess;
printf("Content-type: text/plain\n\n\n");

UnMess = ConvertString(text);
UnicodePrint(UnMess);
fclose(stream);

}

unsigned char* ASCIItoUNICODE (unsigned char ch)
{
unsigned char Val[2];
if ((ch < 192)&&(ch != 168)&&(ch != 184)) {Val[0] = 0; Val[1] = ch; return Val;}
if (ch == 168) {Val[0] = 208; Val[1] = 129; return Val;}
if (ch == 184) {Val[0] = 209; Val[1] = 145; return Val;}
if (ch < 240) {Val[0] = 208; Val[1] = ch-48; return Val;}
if (ch < 249) {Val[0] = 209; Val[1] = ch-112; return Val;}
}
unsigned int* ConvertString (unsigned char *string)
{
unsigned int size=0, *NewString;
unsigned char* Uni;
while (string[size++]!=0);
NewString = (unsigned int*)malloc(sizeof(unsigned int)*2*size-1);
NewString[0]=2*size-1;
size=0;
while (string[size]!=0)
{
Uni = ASCIItoUNICODE(string[size]);
NewString[2*size+1]=Uni[0];
NewString[2*size+2]=Uni[1];
size++;
}
return NewString;
}

void UnicodePrint(unsigned int* Message)
{
int size, i;

size=Message[0];
if (Message[1] == 0) i=2;
else i=1;

while (i<size)
{
printf("%C",Message[i]);
i++;
}
}






Автор: Alca 10.12.2008, 23:55
Код

//---------------------------------------------------------------------------
std::wstring wsASCIIToUnicode(const std::string &csSrc) {
    std::wstring wsRes;
    for (size_t i = 0; i < csSrc.length(); ++ i) {
        wsRes += static_cast<wchar_t>(csSrc[i]);
    }
        
    return wsRes;
}
//---------------------------------------------------------------------------
std::string sUnicodeToASCII(const std::wstring &cwsSrc) {
    std::string sRes;
    for (size_t i = 0; i < cwsSrc.length(); ++ i)
        sRes += static_cast<char>(cwsSrc[i]);
        
    return sRes;
}
//---------------------------------------------------------------------------

Автор: vinter 11.12.2008, 07:08
Alca, жжошь  smile

Добавлено через 1 минуту и 38 секунд
Цитата(nitrexin @  10.12.2008,  23:17 Найти цитируемый пост)
При этом есть условие, не использовать стороних библиотек.

winapi тоже сторонней считаем?

Автор: bsa 11.12.2008, 13:16
Цитата(nitrexin @ 10.12.2008,  22:17)
У меня не получается открыть текстовый файл, и прочесть его в массив

На С++ это выглядит так:
Код
#include <fstream>
#include <vector>
#include <algorithm>
#include <iterator>
#include <iostream>

int main()
{
   std::vector<char> array;   //создаем пустой массив
   std::ifstream file("file.txt"); //открываем файл
   file.unsetf(std::ios::skipws); //не пропускать пробельные символы
   //и загружаем его в массив
   std::copy(std::istream_iterator<char>(file), std::istream_iterator<char>(), std::back_inserter(array));
   //а теперь выводим содержимое массива на экран
   std::copy(array.begin(), array.end(), std::ostream_iterator<char>(std::cout));
   return 0;
}
А на Си так:
Код
#include <stdio.h>
#include <stdlib.h>
#include <fcntl.h>
#include <sys/stat.h>
#include <sys/types.h>

int main()
{
   char *array;
   size_t size;
   int f = open("file.txt", O_RDONLY);   //открываем файл
   if (f < 0)
      return 1; //ошибка
   size = lseek(f, 0, SEEK_END); //перемещаем позицию в конец файла и определяем текущую позицию в файле (размер файла, так как она в конце его)
   array = (char*)malloc(size + 1); //выделяем память
   lseek(f, 0, SEEK_SET); //возвращаем позицию файла в начало
   size = read(f, array, size); //читаем файл в массив
   array[size] = '\0'; //ставим символ конца строки
   printf("%s\n", array);  //выводим на экран
   free(array); //освобождаем память
   close(f);
   return 0;
}


Автор: Alca 11.12.2008, 13:22
Цитата

жжошь

Не понял.

Автор: Fazil6 11.12.2008, 13:54
Цитата(Alca @  11.12.2008,  13:22 Найти цитируемый пост)
Не понял. 

бред ты написал. Что тут непонятного?

Автор: bsa 11.12.2008, 14:26
Цитата(Fazil6 @ 11.12.2008,  13:54)
Цитата(Alca @  11.12.2008,  13:22 Найти цитируемый пост)
Не понял. 

бред ты написал. Что тут непонятного?

подтверждаю. Так как в юникоде старший байт русских букв точно отличается от нулевого. На счет английских не скажу.
http://ru.wikipedia.org/wiki/Юникод

Автор: Alca 11.12.2008, 14:40
А как правильно?

Автор: bsa 11.12.2008, 14:45
вот здесь дана таблица соответствия: http://ru.wikipedia.org/wiki/Альтернативная_кодировка

Автор: Alca 11.12.2008, 14:53
А кодом?

Автор: bsa 11.12.2008, 18:27
Цитата(Alca @ 11.12.2008,  14:53)
А кодом?

http://ftp.gnu.org/pub/gnu/libiconv/libiconv-1.12.tar.gz

Автор: Alca 19.12.2008, 12:45
Q: How to convert between ANSI and UNICODE strings?

A: The quick and dirty way

This way of working is correct for codepages that are single-byte and Unicode strings that are UCS2. This applies to most cases, but if your program should run correctly on Japanese, Chinese, Taiwanese and other systems which have DBCS codepages then use the "correct way" described further below.

ANSI to UNICODE:

The conversion is done using the 'MultiByteToWideChar()' function:

Код

char *ansistr = "Hello";
int a = lstrlenA(ansistr);
BSTR unicodestr = SysAllocStringLen(NULL, a);
::MultiByteToWideChar(CP_ACP, 0, ansistr, a, unicodestr, a);
//... when done, free the BSTR
::SysFreeString(unicodestr);
UNICODE to ANSI:


The UNICODE string mostly is returned by some COM function, like this one:

Код

HRESULT SomeCOMFunction(BSTR *bstr)
{
   *bstr = ::SysAllocString(L"Hello");
   return S_OK;
}


The conversion is done using the 'WideCharToMultiByte()' function:

Код

BSTR unicodestr = 0;
SomeCOMFunction(&unicodestr);
int a = SysStringLen(unicodestr)+1;
char *ansistr = new char[a];
::WideCharToMultiByte(CP_ACP, 
                        0, 
                        unicodestr, 
                        -1, 
                        ansistr, 
                        a, 
                        NULL, 
                        NULL);
//...use the strings, then free their memory:
delete[] ansistr;
::SysFreeString(unicodestr);


The correct way

If you want to handle DBCS codepages and UTF-16 Unicode strings then you should do things this way. The idea is to call 'MultiByteToWideChar()' resp. 'WideCharToMultiByte()' twice. First you get the length of the result, then you allocate the resulting string and call it again to convert.

ANSI to Unicode
Код

char *ansistr = "Hello"
int lenA = lstrlenA(ansistr);
int lenW;
BSTR unicodestr;

lenW = ::MultiByteToWideChar(CP_ACP, 0, ansistr, lenA, 0, 0);
if (lenW > 0)
{
  // Check whether conversion was successful
  unicodestr = ::SysAllocStringLen(0, lenW);
  ::MultiByteToWideChar(CP_ACP, 0, ansistr, lenA, unicodestr, lenW);
}
else
{
  // handle the error
}

// when done, free the BSTR
::SysFreeString(unicodestr);


Unicode to ANSI
Код

BSTR unicodestr = 0;
char *ansistr;
SomeCOMFunction(&unicodestr);
int lenW = ::SysStringLen(unicodestr);
int lenA = ::WideCharToMultiByte(CP_ACP, 0, unicodestr, lenW, 0, 0, NULL, NULL);
if (lenA > 0)
{
  ansistr = new char[lenA + 1]; // allocate a final null terminator as well
::WideCharToMultiByte(CP_ACP, 0, unicodestr, lenW, ansistr, lenA, NULL, NULL);
  ansistr[lenA] = 0; // Set the null terminator yourself
}
else
{
  // handle the error
}

//...use the strings, then free their memory:
delete[] ansistr;
::SysFreeString(unicodestr);

Автор: bsa 19.12.2008, 13:05
Alca, доктор, доктор, что мне делать?!? У меня нет функций MultiByteToWideChar и WideCharToMultiByte. Моя ОС (linux, кстати) о них ничего не знает (wine не предлагать).

Powered by Invision Power Board (http://www.invisionboard.com)
© Invision Power Services (http://www.invisionpower.com)