305 lines
9.1 KiB
C++
305 lines
9.1 KiB
C++
|
|
#include <fstream>
|
||
|
|
|
||
|
|
#include "util.h"
|
||
|
|
#include "character.h"
|
||
|
|
#include "log.h"
|
||
|
|
#include "show.h"
|
||
|
|
|
||
|
|
inline
|
||
|
|
std::string klammertext_special_characters { "@|*^#" };
|
||
|
|
|
||
|
|
inline
|
||
|
|
std::string encoding_marker { "UU" };
|
||
|
|
|
||
|
|
inline
|
||
|
|
std::string diacritic_symbols = "-'`h~\"cbrdwa";
|
||
|
|
|
||
|
|
inline
|
||
|
|
std::string diacritic_symbols_order = "'`h~\"c-brdwa";
|
||
|
|
|
||
|
|
std::string utf8char(int cp)
|
||
|
|
{
|
||
|
|
char c[5]={ 0x00,0x00,0x00,0x00,0x00 };
|
||
|
|
if (cp<=0x7F) {
|
||
|
|
c[0] = cp;
|
||
|
|
} else if(cp<=0x7FF) {
|
||
|
|
c[0] = (cp>>6)+192;
|
||
|
|
c[1] = (cp&63)+128;
|
||
|
|
} else if(0xd800<=cp && cp<=0xdfff) {
|
||
|
|
return "Invalid Unicode: " + std::to_string(cp);
|
||
|
|
} else if(cp<=0xFFFF) {
|
||
|
|
c[0] = (cp>>12)+224;
|
||
|
|
c[1]= ((cp>>6)&63)+128;
|
||
|
|
c[2]=(cp&63)+128;
|
||
|
|
} else if (cp<=0x10FFFF) {
|
||
|
|
c[0] = (cp>>18)+240;
|
||
|
|
c[1] = ((cp>>12)&63)+128;
|
||
|
|
c[2] = ((cp>>6)&63)+128;
|
||
|
|
c[3]=(cp&63)+128;
|
||
|
|
}
|
||
|
|
return std::string(c);
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
std::string unicode_hex_to_char(std::string s, int width=4) //, std::string marker)
|
||
|
|
{
|
||
|
|
(void)K::log(4, s);
|
||
|
|
std::string result {s};
|
||
|
|
std::sregex_iterator end {};
|
||
|
|
std::regex re;
|
||
|
|
switch (width) {
|
||
|
|
case 2: re = hex2_re; break;
|
||
|
|
case 4: re = hex4_re; break;
|
||
|
|
case 5: re = hex5_re; break;
|
||
|
|
}
|
||
|
|
for (std::sregex_iterator p { s.begin(), s.end(), re }; p!= end; ++p) {
|
||
|
|
int codepoint = stoi((*p)[1].str(), nullptr, 16);
|
||
|
|
auto c = utf8char(codepoint);
|
||
|
|
std::regex hit_re { regex_escape((*p)[0]) };
|
||
|
|
result = std::regex_replace(result, hit_re, c);
|
||
|
|
}
|
||
|
|
return result;
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
std::string process_diacritics(std::string s)
|
||
|
|
{
|
||
|
|
(void)K::log(4);
|
||
|
|
std::regex diacritic_re("\\^([^\\s`'~@|^:*#])([" + diacritic_symbols + "])");
|
||
|
|
|
||
|
|
std::string result {s};
|
||
|
|
std::sregex_iterator end {};
|
||
|
|
for (std::sregex_iterator p { s.begin(), s.end(), diacritic_re }; p!= end; ++p) {
|
||
|
|
std::regex hit { regex_escape((*p)[0].str()) };
|
||
|
|
std::string ch = (*p)[1].str();
|
||
|
|
std::string d = (*p)[2].str();
|
||
|
|
result = std::regex_replace(result, hit, ch + unicode_hex_to_char(diacritics[d].first));
|
||
|
|
}
|
||
|
|
return result;
|
||
|
|
}
|
||
|
|
|
||
|
|
std::string extended_latin_symbol_pattern()
|
||
|
|
{
|
||
|
|
std::string result {};
|
||
|
|
std::string sep = "";
|
||
|
|
for (const auto& nr : extended_latin_symbols) {
|
||
|
|
result += sep;
|
||
|
|
result += nr;
|
||
|
|
sep = "|";
|
||
|
|
}
|
||
|
|
return result;
|
||
|
|
}
|
||
|
|
|
||
|
|
std::string process_extended_latin(std::string s)
|
||
|
|
{
|
||
|
|
(void)K::log(4);
|
||
|
|
std::regex diacritic_re("\\^(" + extended_latin_symbol_pattern() + ")\\^");
|
||
|
|
std::string result {s};
|
||
|
|
std::sregex_iterator end {};
|
||
|
|
for (std::sregex_iterator p { s.begin(), s.end(), diacritic_re }; p!= end; ++p) {
|
||
|
|
std::regex hit { regex_escape((*p)[0].str()) };
|
||
|
|
std::string ch = (*p)[1].str();
|
||
|
|
result = std::regex_replace(result, hit, unicode_hex_to_char(extended_latin[ch].first));
|
||
|
|
}
|
||
|
|
return result;
|
||
|
|
}
|
||
|
|
|
||
|
|
std::string process_pinyin(std::string s)
|
||
|
|
{
|
||
|
|
(void)K::log(4);
|
||
|
|
std::regex pinyin_re(R"(\^([aeiou])([1-4]))");
|
||
|
|
std::string result {s};
|
||
|
|
std::sregex_iterator end {};
|
||
|
|
for (std::sregex_iterator p { s.begin(), s.end(), pinyin_re }; p!= end; ++p) {
|
||
|
|
std::regex hit { regex_escape((*p)[0].str()) };
|
||
|
|
std::string vowel = (*p)[1].str();
|
||
|
|
std::string tone = (*p)[2].str();
|
||
|
|
result = std::regex_replace(result, hit,
|
||
|
|
vowel + unicode_hex_to_char(pinyin_tones[tone].first));
|
||
|
|
}
|
||
|
|
return result;
|
||
|
|
}
|
||
|
|
|
||
|
|
std::string process_unicode_codepoint(std::string s)
|
||
|
|
{
|
||
|
|
(void)K::log(4);
|
||
|
|
//return std::regex_replace(s, unicode_re, hidehat + "$1" + hidehat);
|
||
|
|
std::string result {s};
|
||
|
|
std::sregex_iterator end {};
|
||
|
|
for (std::sregex_iterator p { s.begin(), s.end(), unicode_re }; p!= end; ++p) {
|
||
|
|
std::regex hit_re { regex_escape((*p)[0].str()) };
|
||
|
|
result = std::regex_replace(result, hit_re, unicode_hex_to_char((*p)[1].str()));
|
||
|
|
}
|
||
|
|
return result;
|
||
|
|
|
||
|
|
}
|
||
|
|
|
||
|
|
// Old xhide/xrestore/hide/restore functions removed. The ^X mechanism
|
||
|
|
// is handled by the katomizer (katom_t::special) and the general KTESC
|
||
|
|
// escape mechanism in Target::escape_text/resolve_escapes.
|
||
|
|
|
||
|
|
|
||
|
|
|
||
|
|
std::string encode(const std::string& s)
|
||
|
|
{
|
||
|
|
if (s.find("^") == std::string::npos)
|
||
|
|
return s;
|
||
|
|
|
||
|
|
(void)(void)K::log(3);
|
||
|
|
std::string result = s;
|
||
|
|
bool dbg = verbose_level > 3;
|
||
|
|
|
||
|
|
std::string lit_start = "__LITSTART__";
|
||
|
|
std::string lit_end = "__LITEND__";
|
||
|
|
|
||
|
|
result = string_replace(result, "^'", lit_start);
|
||
|
|
result = string_replace(result, "'^", lit_end);
|
||
|
|
|
||
|
|
if (result.find("^") == std::string::npos)
|
||
|
|
return s;
|
||
|
|
|
||
|
|
if (dbg) std::cout << "start: " << result << "\n";
|
||
|
|
|
||
|
|
result = process_extended_latin(result);
|
||
|
|
if (dbg) std::cout << "extended_latin: " << result << "\n";
|
||
|
|
|
||
|
|
result = process_unicode_codepoint(result);
|
||
|
|
if (dbg) std::cout << "unicode: " << result << "\n";
|
||
|
|
|
||
|
|
result = process_diacritics(result);
|
||
|
|
if (dbg) std::cout << "diacrit: " << result << "\n";
|
||
|
|
|
||
|
|
result = process_pinyin(result);
|
||
|
|
if (dbg) std::cout << "pinyin: " << result << "\n";
|
||
|
|
|
||
|
|
result = string_replace(result, lit_start, "^'");
|
||
|
|
result = string_replace(result, lit_end, "'^");
|
||
|
|
|
||
|
|
return result;
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
|
||
|
|
// Decode
|
||
|
|
|
||
|
|
|
||
|
|
// Display
|
||
|
|
|
||
|
|
void write_kt_example_file(std::stringstream& kt, std::string kt_filename)
|
||
|
|
{
|
||
|
|
kt << "|| @line@\n";
|
||
|
|
kt << "@\n";
|
||
|
|
std::cout << "Writing " << kt_filename << "...";
|
||
|
|
std::ofstream out(kt_filename);
|
||
|
|
out << kt.str();
|
||
|
|
out.close();
|
||
|
|
std::cout << "done\n";
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
void diacritics_examples(const std::string& kt_filename)
|
||
|
|
{
|
||
|
|
if (diacritics.size() != diacritic_symbols_order.size()) {
|
||
|
|
//throw Internal_error("Mismatch between diacritics order list and their definitions");
|
||
|
|
std::cout << "Mismatch error\n";
|
||
|
|
}
|
||
|
|
|
||
|
|
std::stringstream kt {};
|
||
|
|
bool write_kt_file = kt_filename.size() != 0;
|
||
|
|
if (write_kt_file)
|
||
|
|
kt << "@table :caption Diacritics (with typical base characters) |\n"
|
||
|
|
<< " @i-Displayed | @i-Written | @i-Name\n";
|
||
|
|
else
|
||
|
|
std::cout << boldblack
|
||
|
|
<< "\nDiacritics (with typical base characters)\n" << black;
|
||
|
|
|
||
|
|
std::string line_sep = "|| @line@ ";
|
||
|
|
for (char symbol : diacritic_symbols_order) {
|
||
|
|
std::string sym { symbol };
|
||
|
|
auto [code, name] = diacritics[sym];
|
||
|
|
std::string letter = diacritic_example_letter[sym];
|
||
|
|
std::string written = "^" + letter + sym;
|
||
|
|
if (sym == "~") {
|
||
|
|
written = "^" + letter + "=7e=";
|
||
|
|
}
|
||
|
|
if (write_kt_file) {
|
||
|
|
std::string display = "^" + letter + sym;
|
||
|
|
kt << line_sep << display << " | @t ^" << written << " @ | " << name << "\n";
|
||
|
|
line_sep = "|| ";
|
||
|
|
}
|
||
|
|
else {
|
||
|
|
std::cout << " " << letter << unicode_hex_to_char(code)
|
||
|
|
<< " " << "^" << letter << symbol << " " << name << "\n";
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if (write_kt_file)
|
||
|
|
write_kt_example_file(kt, kt_filename);
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
void extended_latin_examples(const std::string& kt_filename)
|
||
|
|
{
|
||
|
|
std::stringstream kt {};
|
||
|
|
bool write_kt_file = kt_filename.size() != 0;
|
||
|
|
std::string title = "Extended Latin characters and ligatures";
|
||
|
|
if (write_kt_file)
|
||
|
|
kt << "@table :caption " << title << " |\n"
|
||
|
|
<< " @i-Displayed | @i-Written | @i-Name\n";
|
||
|
|
else
|
||
|
|
std::cout << "\n" << boldblack << title << black << "\n";
|
||
|
|
|
||
|
|
std::string line_sep = "|| @line@ ";
|
||
|
|
for (const std::string& symbol : extended_latin_symbols) {
|
||
|
|
auto [code, name] = extended_latin[symbol];
|
||
|
|
if (write_kt_file) {
|
||
|
|
std::string coded = "^" + symbol + "^";
|
||
|
|
std::string file_literal = "^^ #- " + symbol + " #- ^^";
|
||
|
|
kt << line_sep << coded << " | @t " << file_literal << " @ | " << name << "\n";
|
||
|
|
line_sep = "|| ";
|
||
|
|
}
|
||
|
|
else {
|
||
|
|
std::string screen_literal = "^" + symbol + "^";
|
||
|
|
std::cout << " " << unicode_hex_to_char(code)
|
||
|
|
<< " " << std::setw(4) << std::left << screen_literal << " " << name << "\n";
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if (write_kt_file)
|
||
|
|
write_kt_example_file(kt, kt_filename);
|
||
|
|
}
|
||
|
|
|
||
|
|
|
||
|
|
void pinyin_examples(const std::string& kt_filename)
|
||
|
|
{
|
||
|
|
std::stringstream kt {};
|
||
|
|
bool write_kt_file = kt_filename.size() != 0;
|
||
|
|
if (write_kt_file)
|
||
|
|
kt << "@table :caption Mandarin pinyin tones (using vowel ``a'') |\n"
|
||
|
|
<< " @i-Displayed | @i-Written | @i-Name\n";
|
||
|
|
else
|
||
|
|
std::cout << boldblack
|
||
|
|
<< "\nMandarin pinyin tones (using vowel \"a\")\n" << black;
|
||
|
|
|
||
|
|
std::string line_sep = "|| @line@ ";
|
||
|
|
for (auto [symbol, codename] : pinyin_tones) {
|
||
|
|
auto [code, name] = codename;
|
||
|
|
// std::string hat_code = "^a" + code;
|
||
|
|
std::string literal = "^a" + symbol;
|
||
|
|
if (write_kt_file) {
|
||
|
|
kt << line_sep << literal << " | @t ^" << literal << " @ | " << name << "\n";
|
||
|
|
line_sep = "|| ";
|
||
|
|
} else
|
||
|
|
std::cout << " a" << unicode_hex_to_char(code)
|
||
|
|
<< " "<< std::setw(4) << std::left << literal << " " << name << "\n";
|
||
|
|
}
|
||
|
|
if (write_kt_file)
|
||
|
|
write_kt_example_file(kt, kt_filename);
|
||
|
|
}
|
||
|
|
|
||
|
|
void show_special_characters()
|
||
|
|
{
|
||
|
|
std::cout << std::setfill(' ');
|
||
|
|
diacritics_examples();
|
||
|
|
extended_latin_examples();
|
||
|
|
pinyin_examples();
|
||
|
|
}
|