From de756166e058ef10d63d3d2363de26fa3a9354f1 Mon Sep 17 00:00:00 2001 From: Jang Myeongho Date: Mon, 31 Dec 2012 13:17:15 +0000 Subject: [PATCH] Added enum 'DEFAULT' to 'IfcCharacterDecoder::ConversionMode': using default converter as system default codepage. Added IfcCharacterDecoder::compatibility_mode avoid leaks (compatibility_converter) fixed wrong assertion. added break statements. Only allow character decoder compatibility mode if HAVE_ICU is defined since it relies on ICU to be available. --- src/ifcgeom/IfcGeomFunctions.cpp | 8 +++- src/ifcparse/IfcCharacterDecoder.cpp | 66 +++++++++++++++++++++++----- src/ifcparse/IfcCharacterDecoder.h | 60 ++++++++++++++----------- 3 files changed, 96 insertions(+), 38 deletions(-) diff --git a/src/ifcgeom/IfcGeomFunctions.cpp b/src/ifcgeom/IfcGeomFunctions.cpp index 0f5c264524..093bc4f094 100644 --- a/src/ifcgeom/IfcGeomFunctions.cpp +++ b/src/ifcgeom/IfcGeomFunctions.cpp @@ -435,16 +435,22 @@ void IfcGeom::SetValue(GeomValue var, double value) { switch (var) { case GV_DEFLECTION_TOLERANCE: deflection_tolerance = value; + break; case GV_WIRE_CREATION_TOLERANCE: wire_creation_tolerance = value; + break; case GV_MINIMAL_FACE_AREA: minimal_face_area = value; + break; case GV_POINT_EQUALITY_TOLERANCE: point_equality_tolerance = value; + break; case GV_MAX_FACES_TO_SEW: max_faces_to_sew = value; + break; + default: + assert(!"never reach here"); } - assert(!"never reach here"); } double IfcGeom::GetValue(GeomValue var) { diff --git a/src/ifcparse/IfcCharacterDecoder.cpp b/src/ifcparse/IfcCharacterDecoder.cpp index 3069bd9571..f15163d788 100644 --- a/src/ifcparse/IfcCharacterDecoder.cpp +++ b/src/ifcparse/IfcCharacterDecoder.cpp @@ -88,19 +88,34 @@ void IfcCharacterDecoder::addChar(std::stringstream& s,const UChar32& ch) { #endif } IfcCharacterDecoder::IfcCharacterDecoder(IfcParse::File* f) { - file = f; + file = f; #ifdef HAVE_ICU - if ( ! destination && mode == UTF8 ) { - destination = ucnv_open("utf-8", &status); - } else if ( ! destination && mode == LATIN ) { - destination = ucnv_open("iso-8859-1", &status); - } + if (destination) ucnv_close(destination); + if (compatibility_converter) ucnv_close(compatibility_converter); + destination = nullptr; + compatibility_converter = nullptr; + + if (mode == DEFAULT) { + destination = ucnv_open(nullptr, &status); + } else if (mode == UTF8) { + destination = ucnv_open("utf-8", &status); + } else if (mode == LATIN) { + destination = ucnv_open("iso-8859-1", &status); + } + if (compatibility_charset.empty()) { + compatibility_charset = ucnv_getDefaultName(); + } + compatibility_converter = ucnv_open(compatibility_charset.c_str(), &status); #endif } IfcCharacterDecoder::~IfcCharacterDecoder() { #ifdef HAVE_ICU - if ( destination ) ucnv_close(destination); - if ( converter ) ucnv_close(converter); + if ( destination ) ucnv_close(destination); + if ( converter ) ucnv_close(converter); + if ( compatibility_converter ) ucnv_close(compatibility_converter); + destination = nullptr; + converter = nullptr; + compatibility_converter = nullptr; #endif } IfcCharacterDecoder::operator std::string() { @@ -111,6 +126,9 @@ IfcCharacterDecoder::operator std::string() { int codepage = 1; unsigned int hex = 0; unsigned int hex_count = 0; +#ifdef HAVE_ICU + unsigned int old_hex = 0; // for compatibility_mode +#endif while ( current_char = file->Peek() ) { if ( EXPECTS_CHARACTER(parse_state) ) { #ifdef HAVE_ICU @@ -164,7 +182,24 @@ IfcCharacterDecoder::operator std::string() { if ( (hex_count == 2 && !(parse_state & EXTENDED2)) || (hex_count == 4 && !(parse_state & EXTENDED4)) || (hex_count == 8) ) { - addChar(s,(UChar32) hex); +#ifdef HAVE_ICU + if (compatibility_mode) { + if (old_hex == 0) { + old_hex = hex; + } else { + char characters[3] = { old_hex, hex }; + const char* char_array = &characters[0]; + UChar32 ch = ucnv_getNextUChar(compatibility_converter,&char_array,char_array+2,&status); + addChar(s,ch); + old_hex = 0; + } + } + else { +#endif + addChar(s,(UChar32) hex); +#ifdef HAVE_ICU + } +#endif if ( hex_count == 2 ) parse_state = 0; else CLEAR_HEX(parse_state); hex = hex_count = 0; @@ -177,8 +212,8 @@ IfcCharacterDecoder::operator std::string() { throw IfcException("Invalid character encountered"); } else { parse_state = hex = hex_count = 0; - // NOTE: this is in fact wrong, this ought to be the representation of the character. - // In UTF-8 this is the same, but we should not rely on that. + // NOTE: this is in fact wrong, this ought to be the representation of the character. + // In UTF-8 this is the same, but we should not rely on that. s.put(current_char); } file->Inc(); @@ -246,9 +281,18 @@ void IfcCharacterDecoder::dryRun() { #ifdef HAVE_ICU UConverter* IfcCharacterDecoder::destination = 0; UConverter* IfcCharacterDecoder::converter = 0; +UConverter* IfcCharacterDecoder::compatibility_converter = 0; int IfcCharacterDecoder::previous_codepage = -1; UErrorCode IfcCharacterDecoder::status = U_ZERO_ERROR; +#endif + +#ifdef HAVE_ICU IfcCharacterDecoder::ConversionMode IfcCharacterDecoder::mode = IfcCharacterDecoder::JSON; + +// Many BIM software (eg. Revit, ArchiCAD, ...) has wrong behavior +bool IfcCharacterDecoder::compatibility_mode = false; +std::string IfcCharacterDecoder::compatibility_charset = ""; + #else char IfcCharacterDecoder::substitution_character = '_'; #endif diff --git a/src/ifcparse/IfcCharacterDecoder.h b/src/ifcparse/IfcCharacterDecoder.h index a502926e27..92c24e11df 100644 --- a/src/ifcparse/IfcCharacterDecoder.h +++ b/src/ifcparse/IfcCharacterDecoder.h @@ -1,29 +1,29 @@ /******************************************************************************** - * * - * This file is part of IfcOpenShell. * - * * - * IfcOpenShell is free software: you can redistribute it and/or modify * - * it under the terms of the Lesser GNU General Public License as published by * - * the Free Software Foundation, either version 3.0 of the License, or * - * (at your option) any later version. * - * * - * IfcOpenShell is distributed in the hope that it will be useful, * - * but WITHOUT ANY WARRANTY; without even the implied warranty of * - * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the * - * Lesser GNU General Public License for more details. * - * * - * You should have received a copy of the Lesser GNU General Public License * - * along with this program. If not, see . * - * * - ********************************************************************************/ - - /******************************************************************************** - * * - * Implementation of character decoding as described in ISO 10303-21 table 2 and * - * table 4 * - * * - ********************************************************************************/ - +* * +* This file is part of IfcOpenShell. * +* * +* IfcOpenShell is free software: you can redistribute it and/or modify * +* it under the terms of the Lesser GNU General Public License as published by * +* the Free Software Foundation, either version 3.0 of the License, or * +* (at your option) any later version. * +* * +* IfcOpenShell is distributed in the hope that it will be useful, * +* but WITHOUT ANY WARRANTY; without even the implied warranty of * +* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the * +* Lesser GNU General Public License for more details. * +* * +* You should have received a copy of the Lesser GNU General Public License * +* along with this program. If not, see . * +* * +********************************************************************************/ + +/******************************************************************************** +* * +* Implementation of character decoding as described in ISO 10303-21 table 2 and * +* table 4 * +* * +********************************************************************************/ + #ifndef IFCCHARACTERDECODER_H #define IFCCHARACTERDECODER_H @@ -46,14 +46,22 @@ namespace IfcParse { #ifdef HAVE_ICU static UConverter* destination; static UConverter* converter; + static UConverter* compatibility_converter; static int previous_codepage; static UErrorCode status; #endif void addChar(std::stringstream& s,const UChar32& ch); public: #ifdef HAVE_ICU - enum ConversionMode {UTF8,LATIN,JSON,PYTHON}; + enum ConversionMode {DEFAULT,UTF8,LATIN,JSON,PYTHON}; static ConversionMode mode; + + // Many BIM software (eg. Revit, ArchiCAD, ...) has wrong behavior to encode characters. + // It just translate to extended string in system default code page, not unicode. + // If you want to process these strings, set true. + static bool compatibility_mode; + static std::string compatibility_charset; + #else static char substitution_character; #endif