[thirdparty/glibc.git] / localedata / unicode-gen / gen_translit_font.py

#!/usr/bin/python3
# -*- coding: utf-8 -*-
#
# Generate a translit_font file from a UnicodeData file.
# Copyright (C) 2015-2023 Free Software Foundation, Inc.
# This file is part of the GNU C Library.
#
# The GNU C Library is free software; you can redistribute it and/or
# modify it under the terms of the GNU Lesser General Public
# License as published by the Free Software Foundation; either
# version 2.1 of the License, or (at your option) any later version.
#
# The GNU C Library is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
# Lesser General Public License for more details.
#
# You should have received a copy of the GNU Lesser General Public
# License along with the GNU C Library; if not, see
# <https://www.gnu.org/licenses/>.

'''
Generate a translit_font file from UnicodeData.txt

To see how this script is used, call it with the “-h” option:

    $ ./gen_translit_font -h
    … prints usage message …
'''

import argparse
import time
import unicode_utils

def read_input_file(filename):
    '''Reads the original glibc translit_font file to get the
    original head and tail.

    We want to replace only the part of the file between
    “translit_start” and “translit_end”
    '''
    head = tail = ''
    with open(filename, mode='r') as translit_file:
        for line in translit_file:
            head = head + line
            if line.startswith('translit_start'):
                break
        for line in translit_file:
            if line.startswith('translit_end'):
                tail = line
                break
        for line in translit_file:
            tail = tail + line
    return (head, tail)

def output_head(translit_file, unicode_version, head=''):
    '''Write the header of the output file, i.e. the part of the file
    before the “translit_start” line.
    '''
    if ARGS.input_file and head:
        translit_file.write(head)
    else:
        translit_file.write('escape_char /\n')
        translit_file.write('comment_char %\n')
        translit_file.write(unicode_utils.COMMENT_HEADER)
        translit_file.write('\n')
        translit_file.write('% Transliterations of font equivalents.\n')
        translit_file.write('% Generated automatically from UnicodeData.txt '
                            + 'by gen_translit_font.py '
                            + 'on {:s} '.format(time.strftime('%Y-%m-%d'))
                            + 'for Unicode {:s}.\n'.format(unicode_version))
        translit_file.write('\n')
        translit_file.write('LC_CTYPE\n')
        translit_file.write('\n')
        translit_file.write('translit_start\n')

def output_tail(translit_file, tail=''):
    '''Write the tail of the output file'''
    if ARGS.input_file and tail:
        translit_file.write(tail)
    else:
        translit_file.write('translit_end\n')
        translit_file.write('\n')
        translit_file.write('END LC_CTYPE\n')

def output_transliteration(translit_file):
    '''Write the new transliteration to the output file'''
    translit_file.write('\n')
    for code_point in sorted(unicode_utils.UNICODE_ATTRIBUTES):
        name = unicode_utils.UNICODE_ATTRIBUTES[code_point]['name']
        decomposition = unicode_utils.UNICODE_ATTRIBUTES[
            code_point]['decomposition']
        if decomposition.startswith('<font>'):
            decomposition = decomposition[7:]
            decomposed_code_points = [[int(x, 16)
                                       for x in decomposition.split(' ')]]
            if decomposed_code_points[0]:
                translit_file.write('{:s} '.format(
                    unicode_utils.ucs_symbol(code_point)))
                for index in range(0, len(decomposed_code_points)):
                    if index > 0:
                        translit_file.write(';')
                    if len(decomposed_code_points[index]) > 1:
                        translit_file.write('"')
                    for decomposed_code_point in decomposed_code_points[index]:
                        translit_file.write('{:s}'.format(
                            unicode_utils.ucs_symbol(decomposed_code_point)))
                    if len(decomposed_code_points[index]) > 1:
                        translit_file.write('"')
                translit_file.write(' % {:s}\n'.format(name))
    translit_file.write('\n')

if __name__ == "__main__":
    PARSER = argparse.ArgumentParser(
        description='''
        Generate a translit_font file from UnicodeData.txt.
        ''')
    PARSER.add_argument(
        '-u', '--unicode_data_file',
        nargs='?',
        type=str,
        default='UnicodeData.txt',
        help=('The UnicodeData.txt file to read, '
              + 'default: %(default)s'))
    PARSER.add_argument(
        '-i', '--input_file',
        nargs='?',
        type=str,
        help=''' The original glibc/localedata/locales/translit_font
        file.''')
    PARSER.add_argument(
        '-o', '--output_file',
        nargs='?',
        type=str,
        default='translit_font.new',
        help='''The new translit_font file, default: %(default)s.  If the
        original glibc/localedata/locales/translit_font file has
        been given as an option, the header up to the
        “translit_start” line and the tail from the “translit_end”
        line to the end of the file will be copied unchanged into the
        output file.  ''')
    PARSER.add_argument(
        '--unicode_version',
        nargs='?',
        required=True,
        type=str,
        help='The Unicode version of the input files used.')
    ARGS = PARSER.parse_args()

    unicode_utils.fill_attributes(ARGS.unicode_data_file)
    HEAD = TAIL = ''
    if ARGS.input_file:
        (HEAD, TAIL) = read_input_file(ARGS.input_file)
    with open(ARGS.output_file, mode='w') as TRANSLIT_FILE:
        output_head(TRANSLIT_FILE, ARGS.unicode_version, head=HEAD)
        output_transliteration(TRANSLIT_FILE)
        output_tail(TRANSLIT_FILE, tail=TAIL)
Commit	Line	Data
dd8e8e54 CD	1	#!/usr/bin/python3
	2	# -- coding: utf-8 --
	3	#
	4	# Generate a translit_font file from a UnicodeData file.
6d7e8eda	5	# Copyright (C) 2015-2023 Free Software Foundation, Inc.
dd8e8e54 CD	6	# This file is part of the GNU C Library.
	7	#
	8	# The GNU C Library is free software; you can redistribute it and/or
	9	# modify it under the terms of the GNU Lesser General Public
	10	# License as published by the Free Software Foundation; either
	11	# version 2.1 of the License, or (at your option) any later version.
	12	#
	13	# The GNU C Library is distributed in the hope that it will be useful,
	14	# but WITHOUT ANY WARRANTY; without even the implied warranty of
	15	# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
	16	# Lesser General Public License for more details.
	17	#
	18	# You should have received a copy of the GNU Lesser General Public
	19	# License along with the GNU C Library; if not, see
5a82c748	20	# <https://www.gnu.org/licenses/>.
dd8e8e54 CD	21
	22	'''
	23	Generate a translit_font file from UnicodeData.txt
	24
	25	To see how this script is used, call it with the “-h” option:
	26
	27	$ ./gen_translit_font -h
	28	… prints usage message …
	29	'''
	30
	31	import argparse
	32	import time
	33	import unicode_utils
	34
	35	def read_input_file(filename):
	36	'''Reads the original glibc translit_font file to get the
	37	original head and tail.
	38
	39	We want to replace only the part of the file between
	40	“translit_start” and “translit_end”
	41	'''
	42	head = tail = ''
	43	with open(filename, mode='r') as translit_file:
	44	for line in translit_file:
	45	head = head + line
	46	if line.startswith('translit_start'):
	47	break
	48	for line in translit_file:
	49	if line.startswith('translit_end'):
	50	tail = line
	51	break
	52	for line in translit_file:
	53	tail = tail + line
	54	return (head, tail)
	55
	56	def output_head(translit_file, unicode_version, head=''):
	57	'''Write the header of the output file, i.e. the part of the file
	58	before the “translit_start” line.
	59	'''
	60	if ARGS.input_file and head:
	61	translit_file.write(head)
	62	else:
	63	translit_file.write('escape_char /\n')
	64	translit_file.write('comment_char %\n')
277da2ab	65	translit_file.write(unicode_utils.COMMENT_HEADER)
dd8e8e54 CD	66	translit_file.write('\n')
	67	translit_file.write('% Transliterations of font equivalents.\n')
	68	translit_file.write('% Generated automatically from UnicodeData.txt '
	69	+ 'by gen_translit_font.py '
	70	+ 'on {:s} '.format(time.strftime('%Y-%m-%d'))
	71	+ 'for Unicode {:s}.\n'.format(unicode_version))
	72	translit_file.write('\n')
	73	translit_file.write('LC_CTYPE\n')
	74	translit_file.write('\n')
	75	translit_file.write('translit_start\n')
	76
	77	def output_tail(translit_file, tail=''):
	78	'''Write the tail of the output file'''
	79	if ARGS.input_file and tail:
	80	translit_file.write(tail)
	81	else:
	82	translit_file.write('translit_end\n')
	83	translit_file.write('\n')
	84	translit_file.write('END LC_CTYPE\n')
	85
	86	def output_transliteration(translit_file):
	87	'''Write the new transliteration to the output file'''
	88	translit_file.write('\n')
	89	for code_point in sorted(unicode_utils.UNICODE_ATTRIBUTES):
	90	name = unicode_utils.UNICODE_ATTRIBUTES[code_point]['name']
	91	decomposition = unicode_utils.UNICODE_ATTRIBUTES[
	92	code_point]['decomposition']
	93	if decomposition.startswith('<font>'):
	94	decomposition = decomposition[7:]
	95	decomposed_code_points = [[int(x, 16)
	96	for x in decomposition.split(' ')]]
	97	if decomposed_code_points[0]:
	98	translit_file.write('{:s} '.format(
	99	unicode_utils.ucs_symbol(code_point)))
	100	for index in range(0, len(decomposed_code_points)):
	101	if index > 0:
	102	translit_file.write(';')
	103	if len(decomposed_code_points[index]) > 1:
	104	translit_file.write('"')
	105	for decomposed_code_point in decomposed_code_points[index]:
	106	translit_file.write('{:s}'.format(
	107	unicode_utils.ucs_symbol(decomposed_code_point)))
	108	if len(decomposed_code_points[index]) > 1:
	109	translit_file.write('"')
	110	translit_file.write(' % {:s}\n'.format(name))
	111	translit_file.write('\n')
	112
	113	if __name__ == "__main__":
	114	PARSER = argparse.ArgumentParser(
	115	description='''
	116	Generate a translit_font file from UnicodeData.txt.
	117	''')
	118	PARSER.add_argument(
	119	'-u', '--unicode_data_file',
	120	nargs='?',
	121	type=str,
	122	default='UnicodeData.txt',
	123	help=('The UnicodeData.txt file to read, '
	124	+ 'default: %(default)s'))
	125	PARSER.add_argument(
	126	'-i', '--input_file',
	127	nargs='?',
	128	type=str,
	129	help=''' The original glibc/localedata/locales/translit_font
130	file.''')
131	PARSER.add_argument(
132	'-o', '--output_file',
133	nargs='?',
134	type=str,
135	default='translit_font.new',
136	help='''The new translit_font file, default: %(default)s. If the
137	original glibc/localedata/locales/translit_font file has
138	been given as an option, the header up to the
139	“translit_start” line and the tail from the “translit_end”
140	line to the end of the file will be copied unchanged into the
141	output file. ''')
142	PARSER.add_argument(
143	'--unicode_version',
144	nargs='?',
145	required=True,
146	type=str,
147	help='The Unicode version of the input files used.')
148	ARGS = PARSER.parse_args()
149
150	unicode_utils.fill_attributes(ARGS.unicode_data_file)
151	HEAD = TAIL = ''
152	if ARGS.input_file:
153	(HEAD, TAIL) = read_input_file(ARGS.input_file)
154	with open(ARGS.output_file, mode='w') as TRANSLIT_FILE:
155	output_head(TRANSLIT_FILE, ARGS.unicode_version, head=HEAD)
156	output_transliteration(TRANSLIT_FILE)
157	output_tail(TRANSLIT_FILE, tail=TAIL)