git.saurik.com Git - apple/icu.git/blob - icuSources/tools/gensprep/gensprep.c

2 // License & terms of use: http://www.unicode.org/copyright.html

3 /*

4 *******************************************************************************

5 *

8 *

9 *******************************************************************************

10 * file name: gensprep.c

11 * encoding: UTF-8

12 * tab size: 8 (not used)

13 * indentation:4

14 *

15 * created on: 2003-02-06

16 * created by: Ram Viswanadha

17 *

18 * This program reads the Profile.txt files,

19 * parses them, and extracts the data for StringPrep profile.

20 * It then preprocesses it and writes a binary file for efficient use

21 * in various StringPrep conversion processes.

22 */

24 #define USPREP_TYPE_NAMES_ARRAY 1

26 #include <stdio.h>

27 #include <stdlib.h>

29 #include "cmemory.h"

30 #include "cstring.h"

31 #include "unewdata.h"

32 #include "uoptions.h"

33 #include "uparse.h"

34 #include "sprpimpl.h"

36 #include "unicode/uclean.h"

37 #include "unicode/udata.h"

38 #include "unicode/utypes.h"

39 #include "unicode/putil.h"

42 U_CDECL_BEGIN

43 #include "gensprep.h"

44 U_CDECL_END

46 UBool beVerbose=FALSE, haveCopyright=TRUE;

48 #define NORM_CORRECTIONS_FILE_NAME "NormalizationCorrections.txt"

50 #define NORMALIZE_DIRECTIVE "normalize"

51 #define NORMALIZE_DIRECTIVE_LEN 9

52 #define CHECK_BIDI_DIRECTIVE "check-bidi"

53 #define CHECK_BIDI_DIRECTIVE_LEN 10

55 /* prototypes --------------------------------------------------------------- */

57 static void

 parseMappings(const char *filename, UBool reportError, UErrorCode *pErrorCode);

60 static void

 parseNormalizationCorrections(const char *filename, UErrorCode *pErrorCode);

64 /* -------------------------------------------------------------------------- */

66 static UOption options[]={

67 UOPTION_HELP_H,

68 UOPTION_HELP_QUESTION_MARK,

69 UOPTION_VERBOSE,

70 UOPTION_COPYRIGHT,

71 UOPTION_DESTDIR,

72 UOPTION_SOURCEDIR,

73 UOPTION_ICUDATADIR,

74 UOPTION_BUNDLE_NAME,

     { "normalization", NULL, NULL, NULL, 'n', UOPT_REQUIRES_ARG, 0 },

     { "norm-correction", NULL, NULL, NULL, 'm', UOPT_REQUIRES_ARG, 0 },

     { "check-bidi", NULL, NULL, NULL,  'k', UOPT_NO_ARG, 0},

     { "unicode", NULL, NULL, NULL, 'u', UOPT_REQUIRES_ARG, 0 },

79 };

81 enum{

82 HELP,

83 HELP_QUESTION_MARK,

84 VERBOSE,

85 COPYRIGHT,

86 DESTDIR,

87 SOURCEDIR,

88 ICUDATADIR,

89 BUNDLE_NAME,

90 NORMALIZE,

91 NORM_CORRECTION_DIR,

92 CHECK_BIDI,

93 UNICODE_VERSION

94 };

 static int printHelp(int argc, char* argv[]){

97 /*

98 * Broken into chucks because the C89 standard says the minimum

99 * required supported string length is 509 bytes.

100 */

101 fprintf(stderr,

         "Usage: %s [-options] [file_name]\n"

103 "\n"

104 "Read the files specified and\n"

105 "create a binary file [package-name]_[bundle-name]." DATA_TYPE " with the StringPrep profile data\n"

106 "\n",

107 argv[0]);

108 fprintf(stderr,

109 "Options:\n"

         "\t-h or -? or --help       print this usage text\n"

         "\t-v or --verbose          verbose output\n"

         "\t-c or --copyright        include a copyright notice\n");

113 fprintf(stderr,

         "\t-d or --destdir          destination directory, followed by the path\n"

         "\t-s or --sourcedir        source directory of ICU data, followed by the path\n"

         "\t-b or --bundle-name      generate the output data file with the name specified\n"

         "\t-i or --icudatadir       directory for locating any needed intermediate data files,\n"

         "\t                         followed by path, defaults to %s\n",

119 u_getDataDirectory());

120 fprintf(stderr,

         "\t-n or --normalize        turn on the option for normalization and include mappings\n"

         "\t                         from NormalizationCorrections.txt from the given path,\n"

         "\t                         e.g: /test/icu/source/data/unidata\n");

124 fprintf(stderr,

         "\t-m or --norm-correction  use NormalizationCorrections.txt from the given path\n"

         "\t                         when the input file contains a normalization directive.\n"

         "\t                         unlike -n/--normalize, this option does not force the\n"

         "\t                         normalization.\n");

129 fprintf(stderr,

         "\t-k or --check-bidi       turn on the option for checking for BiDi in the profile\n"

         "\t-u or --unicode          version of Unicode to be used with this profile followed by the version\n"

132 );

     return argc<0 ? U_ILLEGAL_ARGUMENT_ERROR : U_ZERO_ERROR;

134 }

135

136

137 extern int

 main(int argc, char* argv[]) {

139 #if !UCONFIG_NO_IDNA

140 char* filename = NULL;

141 #endif

     const char *srcDir=NULL, *destDir=NULL, *icuUniDataDir=NULL;

     const char *bundleName=NULL, *inputFileName = NULL;

144 char *basename=NULL;

145 int32_t sprepOptions = 0;

146

147 UErrorCode errorCode=U_ZERO_ERROR;

148

149 U_MAIN_INIT_ARGS(argc, argv);

150

151 /* preset then read command line options */

     options[DESTDIR].value=u_getDataDirectory();

     options[SOURCEDIR].value="";

     options[UNICODE_VERSION].value="0"; /* don't assume the unicode version */

155 options[BUNDLE_NAME].value = DATA_NAME;

     options[NORMALIZE].value = "";

157

     argc=u_parseArgs(argc, argv, UPRV_LENGTHOF(options), options);

159

160 /* error handling, printing usage message */

     if(argc<0) {

162 fprintf(stderr,

             "error in command line argument \"%s\"\n",

164 argv[-argc]);

165 }

     if(argc<0 || options[HELP].doesOccur || options[HELP_QUESTION_MARK].doesOccur) {

         return printHelp(argc, argv);

168

169 }

170

171 /* get the options values */

172 beVerbose=options[VERBOSE].doesOccur;

173 haveCopyright=options[COPYRIGHT].doesOccur;

174 srcDir=options[SOURCEDIR].value;

175 destDir=options[DESTDIR].value;

176 bundleName = options[BUNDLE_NAME].value;

     if(options[NORMALIZE].doesOccur) {

178 icuUniDataDir = options[NORMALIZE].value;

179 } else {

180 icuUniDataDir = options[NORM_CORRECTION_DIR].value;

181 }

182

     if(argc<2) {

184 /* print the help message */

         return printHelp(argc, argv);

186 } else {

187 inputFileName = argv[1];

188 }

     if(!options[UNICODE_VERSION].doesOccur){

         return printHelp(argc, argv);

191 }

     if(options[ICUDATADIR].doesOccur) {

         u_setDataDirectory(options[ICUDATADIR].value);

194 }

195 #if UCONFIG_NO_IDNA

196

197 fprintf(stderr,

198 "gensprep writes dummy " U_ICUDATA_NAME "_" DATA_NAME "." DATA_TYPE

199 " because UCONFIG_NO_IDNA is set, \n"

200 "see icu/source/common/unicode/uconfig.h\n");

201 generateData(destDir, bundleName);

202

203 #else

204

     setUnicodeVersion(options[UNICODE_VERSION].value);

     filename = (char* ) uprv_malloc(uprv_strlen(srcDir) + uprv_strlen(inputFileName) + (icuUniDataDir == NULL ? 0 : uprv_strlen(icuUniDataDir)) + 40); /* hopefully this should be enough */

207

208 /* prepare the filename beginning with the source dir */

     if(uprv_strchr(srcDir,U_FILE_SEP_CHAR) == NULL && uprv_strchr(srcDir,U_FILE_ALT_SEP_CHAR) == NULL){

         filename[0] = '.';

211 filename[1] = U_FILE_SEP_CHAR;

         uprv_strcpy(filename+2,srcDir);

213 }else{

214 uprv_strcpy(filename, srcDir);

215 }

216

     basename=filename+uprv_strlen(filename);

     if(basename>filename && *(basename-1)!=U_FILE_SEP_CHAR) {

219 *basename++=U_FILE_SEP_CHAR;

220 }

221

222 /* initialize */

223 init();

224

225 /* process the file */

226 uprv_strcpy(basename,inputFileName);

     parseMappings(filename,FALSE, &errorCode);

     if(U_FAILURE(errorCode)) {

         fprintf(stderr, "Could not open file %s for reading. Error: %s \n", filename, u_errorName(errorCode));

230 return errorCode;

231 }

232

     if(options[NORMALIZE].doesOccur){ /* this option might be set by @normalize;; in the source file */

234 /* set up directory for NormalizationCorrections.txt */

235 uprv_strcpy(filename,icuUniDataDir);

         basename=filename+uprv_strlen(filename);

         if(basename>filename && *(basename-1)!=U_FILE_SEP_CHAR) {

238 *basename++=U_FILE_SEP_CHAR;

239 }

240

241 *basename++=U_FILE_SEP_CHAR;

242 uprv_strcpy(basename,NORM_CORRECTIONS_FILE_NAME);

243

244 parseNormalizationCorrections(filename,&errorCode);

         if(U_FAILURE(errorCode)){

             fprintf(stderr,"Could not open file %s for reading \n", filename);

247 return errorCode;

248 }

249 sprepOptions |= _SPREP_NORMALIZATION_ON;

250 }

251

     if(options[CHECK_BIDI].doesOccur){ /* this option might be set by @check-bidi;; in the source file */

253 sprepOptions |= _SPREP_CHECK_BIDI_ON;

254 }

255

256 setOptions(sprepOptions);

257

258 /* process parsed data */

     if(U_SUCCESS(errorCode)) {

260 /* write the data file */

261 generateData(destDir, bundleName);

262

263 cleanUpData();

264 }

265

266 uprv_free(filename);

267

268 u_cleanup();

269

270 #endif

271

272 return errorCode;

273 }

274

275 #if !UCONFIG_NO_IDNA

276

277 static void U_CALLCONV

 normalizationCorrectionsLineFn(void *context,

                     char *fields[][2], int32_t fieldCount,

280 UErrorCode *pErrorCode) {

     (void)context; // suppress compiler warnings about unused variable

     (void)fieldCount; // suppress compiler warnings about unused variable

283 uint32_t mapping[40];

284 char *end, *s;

285 uint32_t code;

286 int32_t length;

287 UVersionInfo version;

288 UVersionInfo thisVersion;

289

290 /* get the character code, field 0 */

     code=(uint32_t)uprv_strtoul(fields[0][0], &end, 16);

     if(U_FAILURE(*pErrorCode)) {

         fprintf(stderr, "gensprep: error parsing NormalizationCorrections.txt mapping at %s\n", fields[0][0]);

294 exit(*pErrorCode);

295 }

296 /* Original (erroneous) decomposition */

     s = fields[1][0];

298

299 /* parse the mapping string */

     length=u_parseCodePoints(s, mapping, sizeof(mapping)/4, pErrorCode);

301

302 /* ignore corrected decomposition */

303

     u_versionFromString(version,fields[3][0] );

     u_versionFromString(thisVersion, "3.2.0");

306

307

308

     if(U_FAILURE(*pErrorCode)) {

         fprintf(stderr, "gensprep error parsing NormalizationCorrections.txt of U+%04lx - %s\n",

                 (long)code, u_errorName(*pErrorCode));

312 exit(*pErrorCode);

313 }

314

315 /* store the mapping */

     if( version[0] > thisVersion[0] || 

         ((version[0]==thisVersion[0]) && (version[1] > thisVersion[1]))

318 ){

         storeMapping(code,mapping, length, USPREP_MAP, pErrorCode);

320 }

321 setUnicodeVersionNC(version);

322 }

323

324 static void

 parseNormalizationCorrections(const char *filename, UErrorCode *pErrorCode) {

     char *fields[4][2];

327

     if(pErrorCode==NULL || U_FAILURE(*pErrorCode)) {

329 return;

330 }

331

     u_parseDelimitedFile(filename, ';', fields, 4, normalizationCorrectionsLineFn, NULL, pErrorCode);

333

334 /* fprintf(stdout,"Number of code points that have NormalizationCorrections mapping with length >1 : %i\n",len); */

335

     if(U_FAILURE(*pErrorCode) && ( *pErrorCode!=U_FILE_ACCESS_ERROR)) {

         fprintf(stderr, "gensprep error: u_parseDelimitedFile(\"%s\") failed - %s\n", filename, u_errorName(*pErrorCode));

338 exit(*pErrorCode);

339 }

340 }

341

342 static void U_CALLCONV

 strprepProfileLineFn(void *context,

               char *fields[][2], int32_t fieldCount,

345 UErrorCode *pErrorCode) {

     (void)fieldCount; // suppress compiler warnings about unused variable  

347 uint32_t mapping[40];

348 char *end, *map;

349 uint32_t code;

350 int32_t length;

351 /*UBool* mapWithNorm = (UBool*) context;*/

352 const char* typeName;

     uint32_t rangeStart=0,rangeEnd =0;

     const char* filename = (const char*) context;

355 const char *s;

356

     s = u_skipWhitespace(fields[0][0]);

     if (*s == '@') {

359 /* special directive */

360 s++;

         length = (int32_t)(fields[0][1] - s);

362 if (length >= NORMALIZE_DIRECTIVE_LEN

             && uprv_strncmp(s, NORMALIZE_DIRECTIVE, NORMALIZE_DIRECTIVE_LEN) == 0) {

364 options[NORMALIZE].doesOccur = TRUE;

365 return;

366 }

367 else if (length >= CHECK_BIDI_DIRECTIVE_LEN

             && uprv_strncmp(s, CHECK_BIDI_DIRECTIVE, CHECK_BIDI_DIRECTIVE_LEN) == 0) {

369 options[CHECK_BIDI].doesOccur = TRUE;

370 return;

371 }

372 else {

             fprintf(stderr, "gensprep error parsing a directive %s.", fields[0][0]);

374 }

375 }

376

     typeName = fields[2][0];

     map = fields[1][0];

379

     if(uprv_strstr(typeName, usprepTypeNames[USPREP_UNASSIGNED])!=NULL){

381

         u_parseCodePointRange(s, &rangeStart,&rangeEnd, pErrorCode);

         if(U_FAILURE(*pErrorCode)){

             fprintf(stderr, "Could not parse code point range. Error: %s\n",u_errorName(*pErrorCode));

385 return;

386 }

387

388 /* store the range */

         storeRange(rangeStart,rangeEnd,USPREP_UNASSIGNED, pErrorCode);

390

     }else if(uprv_strstr(typeName, usprepTypeNames[USPREP_PROHIBITED])!=NULL){

392

         u_parseCodePointRange(s, &rangeStart,&rangeEnd, pErrorCode);

         if(U_FAILURE(*pErrorCode)){

             fprintf(stderr, "Could not parse code point range. Error: %s\n",u_errorName(*pErrorCode));

396 return;

397 }

398

399 /* store the range */

         storeRange(rangeStart,rangeEnd,USPREP_PROHIBITED, pErrorCode);

401

     }else if(uprv_strstr(typeName, usprepTypeNames[USPREP_MAP])!=NULL){

403

404 /* get the character code, field 0 */

         code=(uint32_t)uprv_strtoul(s, &end, 16);

         if(end<=s || end!=fields[0][1]) {

             fprintf(stderr, "gensprep: syntax error in field 0 at %s\n", fields[0][0]);

408 *pErrorCode=U_PARSE_ERROR;

409 exit(U_PARSE_ERROR);

410 }

411

412 /* parse the mapping string */

         length=u_parseCodePoints(map, mapping, sizeof(mapping)/4, pErrorCode);

414

415 /* store the mapping */

         storeMapping(code,mapping, length,USPREP_MAP, pErrorCode);

417

418 }else{

419 *pErrorCode = U_INVALID_FORMAT_ERROR;

420 }

421

     if(U_FAILURE(*pErrorCode)) {

         fprintf(stderr, "gensprep error parsing  %s line %s at %s. Error: %s\n",filename,

                fields[0][0],fields[2][0],u_errorName(*pErrorCode));

425 exit(*pErrorCode);

426 }

427

428 }

429

430 static void

 parseMappings(const char *filename, UBool reportError, UErrorCode *pErrorCode) {

     char *fields[3][2];

433

     if(pErrorCode==NULL || U_FAILURE(*pErrorCode)) {

435 return;

436 }

437

     u_parseDelimitedFile(filename, ';', fields, 3, strprepProfileLineFn, (void*)filename, pErrorCode);

439

440 /*fprintf(stdout,"Number of code points that have mappings with length >1 : %i\n",len);*/

441

     if(U_FAILURE(*pErrorCode) && (reportError || *pErrorCode!=U_FILE_ACCESS_ERROR)) {

         fprintf(stderr, "gensprep error: u_parseDelimitedFile(\"%s\") failed - %s\n", filename, u_errorName(*pErrorCode));

444 exit(*pErrorCode);

445 }

446 }

447

448

449 #endif /* #if !UCONFIG_NO_IDNA */

450

451 /*

452 * Hey, Emacs, please set the following:

453 *

454 * Local Variables:

455 * indent-tabs-mode: nil

456 * End:

457 *

458 */