diff --git a/.debug.env b/.debug.env new file mode 100644 index 0000000..00be8e4 --- /dev/null +++ b/.debug.env @@ -0,0 +1,4 @@ +TXL_LOGLEVEL=DEBUG +TXL_APP_MODE=development +REMOTE_PDB_HOST=0.0.0.0 +REMOTE_PDB_PORT=4444 diff --git a/.dockerignore b/.dockerignore index 48f8ebd..6e4b904 100644 --- a/.dockerignore +++ b/.dockerignore @@ -1 +1,3 @@ ext/arabic_rom/data +.venv +data diff --git a/.gitignore b/.gitignore index 31b2073..63c4b67 100644 --- a/.gitignore +++ b/.gitignore @@ -136,8 +136,9 @@ tags.temp # Local -ext/arabic_rom/data +/ext/arabic_rom/data scriptshifter/data/*.db !.keep VERSION .~lock.* +/data diff --git a/Dockerfile b/Dockerfile index 5c3edfd..6d754e8 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,15 +1,13 @@ FROM lcnetdev/scriptshifter-base:latest -ARG WORKROOT "/usr/local/scriptshifter/src" +ENV WORKROOT="/usr/local/scriptshifter/src" # Copy core application files. WORKDIR ${WORKROOT} COPY VERSION entrypoint.sh sscli uwsgi.ini wsgi.py ./ COPY scriptshifter ./scriptshifter/ COPY test ./test/ -COPY requirements.txt ./ -RUN pip install --no-cache-dir -r requirements.txt -ENV HF_DATASETS_CACHE /data/hf/datasets +#ENV HF_DATASETS_CACHE="/data/hf/datasets" RUN ./sscli admin init-db RUN chmod +x ./entrypoint.sh @@ -17,4 +15,8 @@ RUN chmod +x ./entrypoint.sh EXPOSE 8000 +# For debugging WSGI sessions. +#RUN pip install --break-system-packages remote-pdb +#ENV PYTHONBREAKPOINT=remote_pdb.set_trace + ENTRYPOINT ["./entrypoint.sh"] diff --git a/deps.txt b/deps.txt deleted file mode 100644 index 8bf3f68..0000000 --- a/deps.txt +++ /dev/null @@ -1,5 +0,0 @@ -# Base image dependencies. -camel-tools>=1.5 -funcy>=1.15,<2 -pymarc>=4.0,<5 -repackage>=0.7.3 diff --git a/entrypoint.sh b/entrypoint.sh index 3ba294f..d4590ab 100644 --- a/entrypoint.sh +++ b/entrypoint.sh @@ -10,7 +10,7 @@ else fi # Preload Thai model. -python -c 'from esupar import load; load("th")' +#python -c 'from esupar import load; load("th")' host=${TXL_WEBAPP_HOST:-"0.0.0.0"} port=${TXL_WEBAPP_PORT:-"8000"} diff --git a/lang.json b/lang.json new file mode 100644 index 0000000..c5f4ad1 --- /dev/null +++ b/lang.json @@ -0,0 +1,1134 @@ +{ + "abazin_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Abazin (Cyrillic)", + "marc_code": "cau" + }, + "abkhaz_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Abkhaz (Cyrillic)", + "marc_code": "abk" + }, + "adygei_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Adygei (Cyrillic)", + "marc_code": "ady" + }, + "altai_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Altai (Cyrillic)", + "marc_code": "alt" + }, + "amharic": { + "alias_of": "ethiopic_generic", + "label": "Amharic", + "marc_code": "amh" + }, + "arabic": { + "case_sensitive": false, + "description": "Arabic-to-Roman transliterator using the ArabicTransliterator external library.\n", + "has_r2s": true, + "has_s2r": true, + "label": "Arabic", + "marc_code": "ara" + }, + "argobba_ethiopic": { + "alias_of": "ethiopic_generic", + "label": "Argobba (Ethiopic)", + "marc_code": "amh" + }, + "armenian": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Armenian", + "marc_code": "arm" + }, + "assamese": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Assamese", + "marc_code": "asm" + }, + "avaric_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Avaric (Cyrillic)", + "marc_code": "ava" + }, + "awadhi_devanagari": { + "alias_of": "devanagari_generic", + "label": "Awadhi (Devanagari)", + "marc_code": "awa" + }, + "azerbaijani_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Azerbaijani (Cyrillic)", + "marc_code": "aze" + }, + "balkar_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Balkar (Cyrillic)", + "marc_code": "krc" + }, + "bashkir_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Bashkir (Cyrillic)", + "marc_code": "bak" + }, + "belarusian": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Belarusian", + "marc_code": "bel" + }, + "bengali": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Bengali", + "marc_code": "ben" + }, + "bihari_devanagari": { + "alias_of": "devanagari_generic", + "label": "Bihari (Devanagari)", + "marc_code": "bih" + }, + "bodo_bengali": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Bodo (Bengali)", + "marc_code": "sit" + }, + "bodo_devanagari": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Bodo (Devanagari)", + "marc_code": "sit" + }, + "braj_devanagari": { + "alias_of": "devanagari_generic", + "label": "Braj (Devanagari)", + "marc_code": "bra" + }, + "bulgarian": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Bulgarian", + "marc_code": "bul" + }, + "buriat_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Buriat (Cyrillic)", + "marc_code": "bua" + }, + "buriat_mongol_bichig": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Buriat (Mongol bichig)", + "marc_code": "bua" + }, + "burmese": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Burmese (Myanmar)", + "marc_code": "bur" + }, + "chechen_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Chechen (Cyrillic)", + "marc_code": "che" + }, + "cherokee": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Cherokee", + "marc_code": "chr" + }, + "chinese": { + "case_sensitive": false, + "description": null, + "has_r2s": false, + "has_s2r": true, + "label": "Chinese (Hanzi)", + "marc_code": "chi" + }, + "chukchi_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Chukchi (Cyrillic)", + "marc_code": "mis" + }, + "church_slavonic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Church Slavonic", + "marc_code": "chu" + }, + "chuvash_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Chuvash (Cyrillic)", + "marc_code": "chv" + }, + "coptic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Coptic", + "marc_code": "cop" + }, + "cyrillic_generic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Cyrillic (Generic)", + "marc_code": "mul" + }, + "dargwa_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Dargwa (Cyrillic)", + "marc_code": "dar" + }, + "devanagari_generic": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Devanagari (Generic)", + "marc_code": "hin, san" + }, + "divehi_thaana": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Divehi (Thaana)", + "marc_code": "div" + }, + "dogri_devanagari": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Dogri (Devanagari)", + "marc_code": "doi" + }, + "dungan_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Dungan (Cyrillic)", + "marc_code": "sit" + }, + "dzongkha_tibetan": { + "alias_of": "tibetan", + "label": "Dzongkha (Tibetan)", + "marc_code": "dzo" + }, + "ethiopic_generic": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Ethiopic (Generic)", + "marc_code": "mul" + }, + "even-evenki_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Even/Evenki (Cyrillic)", + "marc_code": "tut" + }, + "fula_adlam": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Fula (ADLaM)", + "marc_code": "ful" + }, + "gagauz_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Gagauz (Cyrillic)", + "marc_code": "tut" + }, + "georgian": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Georgian", + "marc_code": "geo" + }, + "gilyak_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Gilyak (Cyrillic)", + "marc_code": "mis" + }, + "glagolitic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Glagolitic", + "marc_code": "chu" + }, + "greek_classical": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Greek (Classical)", + "marc_code": "grc" + }, + "greek_modern": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Greek (Modern)", + "marc_code": "gre" + }, + "gujarati": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Gujarati", + "marc_code": "guj" + }, + "gurage_ethiopic": { + "alias_of": "ethiopic_generic", + "label": "Gurage (Ethiopic)", + "marc_code": "sem" + }, + "gurmukhi_generic": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Gurmukhi (generic)", + "marc_code": null + }, + "hebrew": { + "case_sensitive": false, + "description": null, + "has_r2s": false, + "has_s2r": true, + "label": "Hebrew", + "marc_code": "heb" + }, + "hindi": { + "alias_of": "devanagari_generic", + "label": "Hindi (Devanagari)", + "marc_code": "hin" + }, + "ingush_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Ingush (Cyrillic)", + "marc_code": "inh" + }, + "inuit_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Inuit (Cyrillic)", + "marc_code": "ipk" + }, + "inuktitut": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Inuktitut", + "marc_code": "iku" + }, + "kabardian_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Kabardian (Cyrillic)", + "marc_code": "kbd" + }, + "kalmyk_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Kalmyk (Cyrillic)", + "marc_code": "xal" + }, + "kalmyk_tod": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Kalmyk (Tod)", + "marc_code": "xal" + }, + "kannada": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Kannada", + "marc_code": "kan" + }, + "kara-kalpak_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Kara-Kalpak (Cyrillic)", + "marc_code": "kaa" + }, + "karachay-balkar_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Karachay-Balkar (Cyrillic)", + "marc_code": "krc" + }, + "karelian_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Karelian (Cyrillic)", + "marc_code": "krl" + }, + "kazakh_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Kazakh (Cyrillic)", + "marc_code": "kaz" + }, + "khakass_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Khakass (Cyrillic)", + "marc_code": "tut" + }, + "khanty_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Khanty (Cyrillic)", + "marc_code": "fiu" + }, + "khmer": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Khmer", + "marc_code": "khm" + }, + "komi-permyak_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Komi-Permyak (Cyrillic)", + "marc_code": "kom" + }, + "komi_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Komi (Cyrillic)", + "marc_code": "kom" + }, + "konkani_devanagari": { + "alias_of": "devanagari_generic", + "label": "Konkani (Devanagari)", + "marc_code": "kok" + }, + "konkani_kannada": { + "alias_of": "kannada", + "label": "Konkani (Kannada)", + "marc_code": "kok" + }, + "korean_names": { + "case_sensitive": false, + "description": "Korean S2R for strings ONLY containing personal names formatted as last + first name. Separate multiple names with a comma or a center-dot (U+00B7).\n", + "has_r2s": false, + "has_s2r": true, + "label": "Korean (last + first names only)", + "marc_code": "kor" + }, + "korean_nonames": { + "case_sensitive": false, + "description": "Korean S2R for strings NOT containing any personal names.", + "has_r2s": false, + "has_s2r": true, + "label": "Korean", + "marc_code": "kor" + }, + "koryak_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Koryak (Cyrillic)", + "marc_code": "mis" + }, + "kumyk_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Kumyk (Cyrillic)", + "marc_code": "kum" + }, + "kurdish_arabic": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": false, + "label": "Kurdish (Arabic)", + "marc_code": "kur" + }, + "kurdish_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Kurdish (Cyrillic)", + "marc_code": "kur" + }, + "kyrgyz_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Kyrgyz (Cyrillic)", + "marc_code": "kir" + }, + "lahnda_gurmukhi": { + "alias_of": "gurmukhi_generic", + "label": "Lahnda (Gurmukhi)", + "marc_code": "lah" + }, + "lak_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Lak (Cyrillic)", + "marc_code": "cau" + }, + "lepcha": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Lepcha", + "marc_code": "sit" + }, + "lezghian_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Lezghian (Cyrillic)", + "marc_code": "lez" + }, + "limbu": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Limbu", + "marc_code": "sit" + }, + "lithuanian_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Lithuanian (Cyrillic)", + "marc_code": "lit" + }, + "macedonian": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Macedonian", + "marc_code": "mac" + }, + "maithili_devanagari": { + "alias_of": "devanagari_generic", + "label": "Maithili (Devanagari)", + "marc_code": "mai" + }, + "malayalam": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Malayalam", + "marc_code": "mal" + }, + "manchu": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Manchu", + "marc_code": "mnc" + }, + "manipuri_bengali": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Manipuri (Bengali)", + "marc_code": "mni" + }, + "manipuri_meitei": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Manipuri (Meitei)", + "marc_code": "mni" + }, + "mansi_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Mansi (Cyrillic)", + "marc_code": "fiu" + }, + "marathi_devanagari": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Marathi (Devanagari)", + "marc_code": "mar" + }, + "mari_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Mari (Cyrillic)", + "marc_code": "chm" + }, + "moldovan_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Moldovan (Cyrillic)", + "marc_code": "mol" + }, + "mongolian_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Mongolian (Cyrillic)", + "marc_code": "mon" + }, + "mongolian_mongol_bichig": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Mongolian (Mongol bichig)", + "marc_code": "mon" + }, + "montenegrin_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Montenegrin (Cyrillic)", + "marc_code": "cnr" + }, + "mordvin_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Mordvin (Cyrillic)", + "marc_code": "fiu" + }, + "moroccan_tamazight": { + "alias_of": "tifinagh_generic", + "label": "Moroccan Tamazight", + "marc_code": "ber" + }, + "nanai_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Nanai (Cyrillic)", + "marc_code": "tut" + }, + "nenets_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Nenets (Cyrillic)", + "marc_code": "mis" + }, + "nepali_devanagari": { + "alias_of": "devanagari_generic", + "label": "Nepali (Devanagari)", + "marc_code": "nep" + }, + "newari_devanagari": { + "alias_of": "devanagari_generic", + "label": "Newari (Devanagari)", + "marc_code": "new" + }, + "nivkh_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Nivkh (Cyrillic)", + "marc_code": "mis" + }, + "nogai_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Nogai (Cyrillic)", + "marc_code": "nog" + }, + "odia": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Odia", + "marc_code": "ori" + }, + "ossetic_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Ossetic (Cyrillic)", + "marc_code": "oss" + }, + "pahari_devanagari": { + "alias_of": "devanagari_generic", + "label": "Pahari (Devanagari)", + "marc_code": "him" + }, + "pali_bengali": { + "alias_of": "bengali", + "label": "Pali (Bengali)", + "marc_code": "pli" + }, + "pali_devanagari": { + "alias_of": "devanagari_generic", + "label": "Pali (Devanagari)", + "marc_code": "pli" + }, + "pali_sinhala": { + "alias_of": "sinhala", + "label": "Pali (Sinhala)", + "marc_code": "pli" + }, + "panjabi_gurmukhi": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Panjabi (Gurmukhi)", + "marc_code": "pan" + }, + "permyak_cyrillic": { + "alias_of": "komi-permyak_cyrillic", + "label": "Permyak (Cyrillic)", + "marc_code": "kom" + }, + "persian": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": false, + "label": "Persian", + "marc_code": "per" + }, + "prakrit_devanagari": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Prakrit (Devanagari)", + "marc_code": "pra" + }, + "pushto": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": false, + "label": "Pushto", + "marc_code": "pus" + }, + "rajasthani_devanagari": { + "alias_of": "devanagari_generic", + "label": "Rajasthani (Devanagari)", + "marc_code": "raj" + }, + "romani_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Romani (Cyrillic)", + "marc_code": "rom" + }, + "romanian_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Romanian (Cyrillic)", + "marc_code": "rum" + }, + "russian": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Russian", + "marc_code": "rus" + }, + "sami_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Sami (Cyrillic)", + "marc_code": "smi" + }, + "sanskrit_devanagari": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Sanskrit (Devanagari)", + "marc_code": "san" + }, + "santali_ol_chiki": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Santali (Ol chiki)", + "marc_code": "sat" + }, + "selkup_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Selkup (Cyrillic)", + "marc_code": "sel" + }, + "serbian": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Serbian", + "marc_code": "srp" + }, + "shor_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Shor (Cyrillic)", + "marc_code": "tut" + }, + "sindhi_arabic": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": false, + "label": "Sindhi (Arabic)", + "marc_code": "snd" + }, + "sindhi_gurmukhi": { + "alias_of": "gurmukhi_generic", + "label": "Sindhi (Gurmukhi)", + "marc_code": "snd" + }, + "sinhala": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Sinhala", + "marc_code": "sin" + }, + "syriac_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Syriac (Cyrillic)", + "marc_code": "syc" + }, + "tabasaran_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Tabasaran (Cyrillic)", + "marc_code": "cau" + }, + "tajik_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Tajik (Cyrillic)", + "marc_code": "tgk" + }, + "tamashek": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Tamashek", + "marc_code": "tmh" + }, + "tamazight_moroccan": { + "alias_of": "tifinagh_generic", + "label": "Tamazight (Moroccan)", + "marc_code": "ber" + }, + "tamil": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Tamil", + "marc_code": "tam" + }, + "tamil_brahmi": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Tamil Brahmi", + "marc_code": "tam" + }, + "tamil_extended": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Tamil (extended)", + "marc_code": "tam" + }, + "tat_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Tat (Cyrillic)", + "marc_code": "ira" + }, + "tatar-kryashen_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Tatar-Kryashen (Cyrillic)", + "marc_code": "tat" + }, + "tatar_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Tatar (Cyrillic)", + "marc_code": "tat" + }, + "telugu": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Telugu", + "marc_code": "tel" + }, + "thai": { + "case_sensitive": false, + "description": null, + "has_r2s": false, + "has_s2r": true, + "label": "Thai", + "marc_code": "tha" + }, + "tibetan": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Tibetan", + "marc_code": "tib" + }, + "tibetan_2015_r2r": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": false, + "label": "Tibetan (\u00f1,\u1e45,\u015b,\u017a to ny,ng,sh,zh only)", + "marc_code": "tib" + }, + "tifinagh_generic": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Tifinagh (Generic)", + "marc_code": "ber" + }, + "tigre_ethiopic": { + "alias_of": "ethiopic_generic", + "label": "Tigre (Ethiopic)", + "marc_code": "tig" + }, + "tigrinya_ethiopic": { + "alias_of": "ethiopic_generic", + "label": "Tigrinya (Ethiopic)", + "marc_code": "tir" + }, + "turkmen_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Turkmen (Cyrillic)", + "marc_code": "tuk" + }, + "tuvinian_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Tuvinian (Cyrillic)", + "marc_code": "tyv" + }, + "udekhe_cyrillic": { + "alias_of": "cyrillic_generic", + "label": "Udekhe (Cyrillic)", + "marc_code": "tut" + }, + "udmurt_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Udmurt (Cyrillic)", + "marc_code": "udm" + }, + "uighur_arabic": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Uighur (Arabic)", + "marc_code": "uig" + }, + "uighur_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Uighur (Cyrillic)", + "marc_code": "uig" + }, + "ukrainian": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Ukrainian", + "marc_code": "ukr" + }, + "urdu": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": false, + "label": "Urdu", + "marc_code": "urd" + }, + "uzbek_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Uzbek (Cyrillic)", + "marc_code": "uzb" + }, + "yakut_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Yakut (Cyrillic)", + "marc_code": "sah" + }, + "yiddish": { + "case_sensitive": false, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Yiddish", + "marc_code": "yid" + }, + "yuit_cyrillic": { + "case_sensitive": true, + "description": null, + "has_r2s": true, + "has_s2r": true, + "label": "Yuit (Cyrillic)", + "marc_code": "ypk" + } +} diff --git a/requirements.txt b/requirements.txt index 9a20050..b1545c2 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,10 +1,17 @@ -# Core application dependencies. +# Base image dependencies. aksharamukha>=2.2,<3 -esupar>=1.7.5 +camel-tools>=1.6 +#esupar>=1.9,<2 # For Thai. Needs training. Breaks whole container. flask>=2.3,<3 flask-cors>=4.0,<5 +#funcy>=1.15,<2 +#pymarc>=4.0,<5 python-dotenv>=1.0,<2 pyyaml>=6.0,<7 -regex>=2023.8.8 +#regex>=2023.8.8 +shekar>=1.6,<2 +#repackage>=0.7.3 +tokenizers>=0.22 +torch>=2.12,<3 uwsgi>=2.0,<2.1 yiddish==0.0.21 diff --git a/scriptshifter/hooks/asian_tokenizer/__init__.py b/scriptshifter/hooks/asian_tokenizer/__init__.py index 1b396d5..3016d7f 100644 --- a/scriptshifter/hooks/asian_tokenizer/__init__.py +++ b/scriptshifter/hooks/asian_tokenizer/__init__.py @@ -1,8 +1,8 @@ -from esupar import load - - -def s2r_tokenize(ctx, model): - nlp = load(model) - token_data = nlp(ctx.src) - - ctx._src = " ".join(token_data.values[1]) +#from esupar import load +# +# +#def s2r_tokenize(ctx, model): +# nlp = load(model) +# token_data = nlp(ctx.src) +# +# ctx._src = " ".join(token_data.values[1]) diff --git a/scriptshifter/hooks/general/nlp/seq2seq.py b/scriptshifter/hooks/general/nlp/seq2seq.py new file mode 100644 index 0000000..c28edbc --- /dev/null +++ b/scriptshifter/hooks/general/nlp/seq2seq.py @@ -0,0 +1,448 @@ +import torch +import torch.nn as nn +import torch.optim as optim +import random +import numpy as np +import spacy +import datasets +import torchtext +import tqdm +import evaluate + +torchtext.disable_torchtext_deprecation_warning() + +seed = 1234 + +random.seed(seed) +np.random.seed(seed) +torch.manual_seed(seed) +torch.cuda.manual_seed(seed) +torch.backends.cudnn.deterministic = True + +dataset = datasets.load_dataset("bentrevett/multi30k") + +train_data, valid_data, test_data = ( + dataset["train"], + dataset["validation"], + dataset["test"], +) + + +# !python -m spacy download en_core_web_sm +# !python -m spacy download de_core_news_sm + +en_nlp = spacy.load("en_core_web_sm") +de_nlp = spacy.load("de_core_news_sm") + + +def tokenize_example( + example, en_nlp, de_nlp, max_length, lower, sos_token, eos_token): + en_tokens = [token.text for token in en_nlp.tokenizer(example["en"])][ + :max_length] + de_tokens = [token.text for token in de_nlp.tokenizer(example["de"])][ + :max_length] + if lower: + en_tokens = [token.lower() for token in en_tokens] + de_tokens = [token.lower() for token in de_tokens] + en_tokens = [sos_token] + en_tokens + [eos_token] + de_tokens = [sos_token] + de_tokens + [eos_token] + return {"en_tokens": en_tokens, "de_tokens": de_tokens} + + +max_length = 1_000 +lower = True +sos_token = "" +eos_token = "" + +fn_kwargs = { + "en_nlp": en_nlp, + "de_nlp": de_nlp, + "max_length": max_length, + "lower": lower, + "sos_token": sos_token, + "eos_token": eos_token, +} + +train_data = train_data.map(tokenize_example, fn_kwargs=fn_kwargs) +valid_data = valid_data.map(tokenize_example, fn_kwargs=fn_kwargs) +test_data = test_data.map(tokenize_example, fn_kwargs=fn_kwargs) + +min_freq = 2 +unk_token = "" +pad_token = "" + +special_tokens = [ + unk_token, + pad_token, + sos_token, + eos_token, +] + +en_vocab = torchtext.vocab.build_vocab_from_iterator( + train_data["en_tokens"], + min_freq=min_freq, + specials=special_tokens, +) + +de_vocab = torchtext.vocab.build_vocab_from_iterator( + train_data["de_tokens"], + min_freq=min_freq, + specials=special_tokens, +) + +assert en_vocab[unk_token] == de_vocab[unk_token] +assert en_vocab[pad_token] == de_vocab[pad_token] + +unk_index = en_vocab[unk_token] +pad_index = en_vocab[pad_token] + + +en_vocab.set_default_index(unk_index) +de_vocab.set_default_index(unk_index) + + +def numericalize_example(example, en_vocab, de_vocab): + en_ids = en_vocab.lookup_indices(example["en_tokens"]) + de_ids = de_vocab.lookup_indices(example["de_tokens"]) + return {"en_ids": en_ids, "de_ids": de_ids} + + +fn_kwargs = {"en_vocab": en_vocab, "de_vocab": de_vocab} + +train_data = train_data.map(numericalize_example, fn_kwargs=fn_kwargs) +valid_data = valid_data.map(numericalize_example, fn_kwargs=fn_kwargs) +test_data = test_data.map(numericalize_example, fn_kwargs=fn_kwargs) + +data_type = "torch" +format_columns = ["en_ids", "de_ids"] + +train_data = train_data.with_format( + type=data_type, columns=format_columns, output_all_columns=True +) + +valid_data = valid_data.with_format( + type=data_type, + columns=format_columns, + output_all_columns=True, +) + +test_data = test_data.with_format( + type=data_type, + columns=format_columns, + output_all_columns=True, +) + + +def get_collate_fn(pad_index): + def collate_fn(batch): + batch_en_ids = [example["en_ids"] for example in batch] + batch_de_ids = [example["de_ids"] for example in batch] + batch_en_ids = nn.utils.rnn.pad_sequence( + batch_en_ids, padding_value=pad_index) + batch_de_ids = nn.utils.rnn.pad_sequence( + batch_de_ids, padding_value=pad_index) + batch = { + "en_ids": batch_en_ids, + "de_ids": batch_de_ids, + } + return batch + + return collate_fn + + +def get_data_loader(dataset, batch_size, pad_index, shuffle=False): + collate_fn = get_collate_fn(pad_index) + data_loader = torch.utils.data.DataLoader( + dataset=dataset, + batch_size=batch_size, + collate_fn=collate_fn, + shuffle=shuffle, + ) + return data_loader + + +batch_size = 128 +train_data_loader = get_data_loader( + train_data, batch_size, pad_index, shuffle=True) +valid_data_loader = get_data_loader(valid_data, batch_size, pad_index) +test_data_loader = get_data_loader(test_data, batch_size, pad_index) + + +class Encoder(nn.Module): + def __init__( + self, input_dim, embedding_dim, hidden_dim, n_layers, dropout): + super().__init__() + self.hidden_dim = hidden_dim + self.n_layers = n_layers + self.embedding = nn.Embedding(input_dim, embedding_dim) + self.rnn = nn.LSTM( + embedding_dim, hidden_dim, n_layers, dropout=dropout) + self.dropout = nn.Dropout(dropout) + + def forward(self, src): + # src = [src length, batch size] + embedded = self.dropout(self.embedding(src)) + # embedded = [src length, batch size, embedding dim] + outputs, (hidden, cell) = self.rnn(embedded) + # outputs = [src length, batch size, hidden dim * n directions] + # hidden = [n layers * n directions, batch size, hidden dim] + # cell = [n layers * n directions, batch size, hidden dim] + # outputs are always from the top hidden layer + return hidden, cell + + +class Decoder(nn.Module): + def __init__( + self, output_dim, embedding_dim, hidden_dim, n_layers, dropout): + super().__init__() + self.output_dim = output_dim + self.hidden_dim = hidden_dim + self.n_layers = n_layers + self.embedding = nn.Embedding(output_dim, embedding_dim) + self.rnn = nn.LSTM( + embedding_dim, hidden_dim, n_layers, dropout=dropout) + self.fc_out = nn.Linear(hidden_dim, output_dim) + self.dropout = nn.Dropout(dropout) + + def forward(self, input, hidden, cell): + # input = [batch size] + # hidden = [n layers * n directions, batch size, hidden dim] + # cell = [n layers * n directions, batch size, hidden dim] + # n directions in the decoder will both always be 1, therefore: + # hidden = [n layers, batch size, hidden dim] + # context = [n layers, batch size, hidden dim] + input = input.unsqueeze(0) + # input = [1, batch size] + embedded = self.dropout(self.embedding(input)) + # embedded = [1, batch size, embedding dim] + output, (hidden, cell) = self.rnn(embedded, (hidden, cell)) + # output = [seq length, batch size, hidden dim * n directions] + # hidden = [n layers * n directions, batch size, hidden dim] + # cell = [n layers * n directions, batch size, hidden dim] + # seq length and n directions will always be 1 in this decoder, + # therefore: + # output = [1, batch size, hidden dim] + # hidden = [n layers, batch size, hidden dim] + # cell = [n layers, batch size, hidden dim] + prediction = self.fc_out(output.squeeze(0)) + # prediction = [batch size, output dim] + return prediction, hidden, cell + + +class Seq2Seq(nn.Module): + def __init__(self, encoder, decoder, device): + super().__init__() + self.encoder = encoder + self.decoder = decoder + self.device = device + assert ( + encoder.hidden_dim == decoder.hidden_dim + ), "Hidden dimensions of encoder and decoder must be equal!" + assert ( + encoder.n_layers == decoder.n_layers + ), "Encoder and decoder must have equal number of layers!" + + def forward(self, src, trg, teacher_forcing_ratio): + # src = [src length, batch size] + # trg = [trg length, batch size] + # teacher_forcing_ratio is probability to use teacher forcing + # e.g. if teacher_forcing_ratio is 0.75 we use ground-truth inputs 75% + # of the time + batch_size = trg.shape[1] + trg_length = trg.shape[0] + trg_vocab_size = self.decoder.output_dim + # tensor to store decoder outputs + outputs = torch.zeros( + trg_length, batch_size, trg_vocab_size).to(self.device) + # last hidden state of the encoder is used as the initial hidden state + # of the decoder + hidden, cell = self.encoder(src) + # hidden = [n layers * n directions, batch size, hidden dim] + # cell = [n layers * n directions, batch size, hidden dim] + # first input to the decoder is the tokens + input = trg[0, :] + # input = [batch size] + for t in range(1, trg_length): + # insert input token embedding, previous hidden and previous cell + # states receive output tensor (predictions) and new hidden and + # cell states + output, hidden, cell = self.decoder(input, hidden, cell) + # output = [batch size, output dim] + # hidden = [n layers, batch size, hidden dim] + # cell = [n layers, batch size, hidden dim] + # place predictions in a tensor holding predictions for each token + outputs[t] = output + # decide if we are going to use teacher forcing or not + teacher_force = random.random() < teacher_forcing_ratio + # get the highest predicted token from our predictions + top1 = output.argmax(1) + # if teacher forcing, use actual next token as next input + # if not, use predicted token + input = trg[t] if teacher_force else top1 + # input = [batch size] + return outputs + + +input_dim = len(de_vocab) +output_dim = len(en_vocab) +encoder_embedding_dim = 256 +decoder_embedding_dim = 256 +hidden_dim = 512 +n_layers = 2 +encoder_dropout = 0.5 +decoder_dropout = 0.5 +device = torch.device("cuda" if torch.cuda.is_available() else "cpu") + +encoder = Encoder( + input_dim, + encoder_embedding_dim, + hidden_dim, + n_layers, + encoder_dropout, +) + +decoder = Decoder( + output_dim, + decoder_embedding_dim, + hidden_dim, + n_layers, + decoder_dropout, +) + +model = Seq2Seq(encoder, decoder, device).to(device) + + +def init_weights(m): + for name, param in m.named_parameters(): + nn.init.uniform_(param.data, -0.08, 0.08) + + +model.apply(init_weights) + + +def count_parameters(model): + return sum(p.numel() for p in model.parameters() if p.requires_grad) + + +print(f"The model has {count_parameters(model):,} trainable parameters") + +optimizer = optim.Adam(model.parameters()) + +criterion = nn.CrossEntropyLoss(ignore_index=pad_index) + + +def train_fn( + model, data_loader, optimizer, criterion, clip, teacher_forcing_ratio, + device): + model.train() + epoch_loss = 0 + for i, batch in enumerate(data_loader): + src = batch["de_ids"].to(device) + trg = batch["en_ids"].to(device) + # src = [src length, batch size] + # trg = [trg length, batch size] + optimizer.zero_grad() + output = model(src, trg, teacher_forcing_ratio) + # output = [trg length, batch size, trg vocab size] + output_dim = output.shape[-1] + output = output[1:].view(-1, output_dim) + # output = [(trg length - 1) * batch size, trg vocab size] + trg = trg[1:].view(-1) + # trg = [(trg length - 1) * batch size] + loss = criterion(output, trg) + loss.backward() + torch.nn.utils.clip_grad_norm_(model.parameters(), clip) + optimizer.step() + epoch_loss += loss.item() + return epoch_loss / len(data_loader) + + +def evaluate_fn(model, data_loader, criterion, device): + model.eval() + epoch_loss = 0 + with torch.no_grad(): + for i, batch in enumerate(data_loader): + src = batch["de_ids"].to(device) + trg = batch["en_ids"].to(device) + # src = [src length, batch size] + # trg = [trg length, batch size] + output = model(src, trg, 0) # turn off teacher forcing + # output = [trg length, batch size, trg vocab size] + output_dim = output.shape[-1] + output = output[1:].view(-1, output_dim) + # output = [(trg length - 1) * batch size, trg vocab size] + trg = trg[1:].view(-1) + # trg = [(trg length - 1) * batch size] + loss = criterion(output, trg) + epoch_loss += loss.item() + return epoch_loss / len(data_loader) + + +n_epochs = 10 +clip = 1.0 +teacher_forcing_ratio = 0.5 + +best_valid_loss = float("inf") + +for epoch in tqdm.tqdm(range(n_epochs)): + train_loss = train_fn( + model, + train_data_loader, + optimizer, + criterion, + clip, + teacher_forcing_ratio, + device, + ) + valid_loss = evaluate_fn( + model, + valid_data_loader, + criterion, + device, + ) + if valid_loss < best_valid_loss: + best_valid_loss = valid_loss + torch.save(model.state_dict(), "tut1-model.pt") + print( + f"\tTrain Loss: {train_loss:7.3f} | " + f"Train PPL: {np.exp(train_loss):7.3f}") + print( + f"\tValid Loss: {valid_loss:7.3f} | " + f"Valid PPL: {np.exp(valid_loss):7.3f}") + + +def translate_sentence( + sentence, + model, + en_nlp, + de_nlp, + en_vocab, + de_vocab, + lower, + sos_token, + eos_token, + device, + max_output_length=25, +): + model.eval() + with torch.no_grad(): + if isinstance(sentence, str): + tokens = [token.text for token in de_nlp.tokenizer(sentence)] + else: + tokens = [token for token in sentence] + if lower: + tokens = [token.lower() for token in tokens] + tokens = [sos_token] + tokens + [eos_token] + ids = de_vocab.lookup_indices(tokens) + tensor = torch.LongTensor(ids).unsqueeze(-1).to(device) + hidden, cell = model.encoder(tensor) + inputs = en_vocab.lookup_indices([sos_token]) + for _ in range(max_output_length): + inputs_tensor = torch.LongTensor([inputs[-1]]).to(device) + output, hidden, cell = model.decoder(inputs_tensor, hidden, cell) + predicted_token = output.argmax(-1).item() + inputs.append(predicted_token) + if predicted_token == en_vocab[eos_token]: + break + tokens = en_vocab.lookup_tokens(inputs) + return tokens diff --git a/scriptshifter/hooks/seq2seq/NOTES.txt b/scriptshifter/hooks/seq2seq/NOTES.txt new file mode 100644 index 0000000..9cca939 --- /dev/null +++ b/scriptshifter/hooks/seq2seq/NOTES.txt @@ -0,0 +1,71 @@ +Original hyperparameters for 275K data set: + +VOCAB_SIZE = 16000 +EMB_DIM = 256 +HIDDEN_DIM = 256 +DROPOUT = 0.2 +N_LAYERS = 1 +LR = 1e-3 +WEIGHT_DECAY = 1e-5 +GRAD_CLIP = 0.5 +N_EPOCHS = 30 # Number of epochs to train by. +BATCH_SIZE = 64 # Data loader batch size. +# Filter out outlier-length pairs to bound memory per batch. +MAX_SRC_CHARS = 300 +MAX_TGT_CHARS = MAX_SRC_CHARS * 1.33 + + +For a 10x data set: + + - N_EPOCHS = 30 → ~5–10: each epoch already does 10× more updates, so the + total gradient steps would otherwise be ~300 epochs' worth on the current + set. Watch the dev loss/CER curve and early-stop. + - DROPOUT = 0.2 → ~0.1: more data is itself the strongest regularizer; heavy + dropout under-fits when there's no longer an overfitting problem. + - HIDDEN_DIM = 256 and N_LAYERS = 1: 256/1-layer is a small model. With 10× + data it will likely under-fit. Bump to HIDDEN_DIM=384–512 and/or N_LAYERS=2 + and see if dev CER improves. (If you raise N_LAYERS, dropout becomes active + between GRU layers — see dropout if num_layers > 1 else 0.0 at line 301/409 + — so don't drop it to 0.) + - VOCAB_SIZE = 16000: only revisit if your character/morpheme inventory + genuinely grew (new dialect, new script range). For a single-language + transliteration task, 16k is already generous and probably unchanged. + - BATCH_SIZE = 64 and LR = 1e-3: leave alone first. If you want faster + wall-clock, raise batch to 128/256 and scale LR up roughly with √(batch + ratio); Adam at 1e-3 is otherwise robust. + + The single most important change is fewer epochs + dev-set early stopping — + everything else is secondary tuning. + + +When adding 50-100% more data, the model can be retrained from scratch, but: + + - Tokenizer must stay identical. The BPE tokenizer is fit on the corpus + (VOCAB_SIZE = 16000), so if you rebuild it on the larger corpus you'll get + different token IDs and the embedding/output layers won't line up. Either + reuse the saved tokenizer files as-is, or accept that you're retraining + from scratch. + - Lower the LR for the continuation run — try LR = 2e-4 to 5e-4 instead of + 1e-3. The model is already near a minimum; full LR can blow it out. + - Far fewer epochs — 5–10 over the new combined set is usually enough. Track + dev CER and stop when it plateaus. + - Mix old + new data, don't fine-tune on only the new shard, or you'll drift + toward whatever bias the new data has (catastrophic forgetting, even on the + same task). + + If after warm-starting the dev CER stalls noticeably above where you'd + expect, then retrain from scratch — but it's worth trying the cheap path + first. + +2026-06-05: checkpoint trained for ~30 epochs with old dataset + 10 epochs +after integrating with Yale, UPenn data + +Evaluate pass: + +{'n': 8196, 'exact_match': 0.40995607613469986, 'cer': 0.19848789036533696, 'wer': 0.34645313519755006} + +After re-training for 30 epochs from scratch, same dataset: + +{'n': 8196, 'exact_match': 0.41813079551000487, 'cer': 0.19394551839442425, 'wer': 0.33864651839951804} + +After 50 epochs: diff --git a/scriptshifter/hooks/seq2seq/__init__.py b/scriptshifter/hooks/seq2seq/__init__.py new file mode 100644 index 0000000..38ff017 --- /dev/null +++ b/scriptshifter/hooks/seq2seq/__init__.py @@ -0,0 +1,21 @@ +from logging import getLogger + +from scriptshifter.exceptions import BREAK +from scriptshifter.hooks.general import capitalize_post_assembly +from scriptshifter.hooks.seq2seq.model import S2S + + +logger = getLogger(__name__) +models = {} # Models cache. + +def s2r_post_config(ctx, src_script): + if src_script not in models: + logger.info(f"{src_script} model not yet cached. Loading.") + models[src_script] = S2S(src_script) + + ctx.dest = models[src_script].transliterate(ctx.src) + + if ctx.dest: + capitalize_post_assembly(ctx) + + return BREAK diff --git a/scriptshifter/hooks/seq2seq/build_splits.py b/scriptshifter/hooks/seq2seq/build_splits.py new file mode 100644 index 0000000..e0fc455 --- /dev/null +++ b/scriptshifter/hooks/seq2seq/build_splits.py @@ -0,0 +1,264 @@ +#!/usr/bin/env python +"""Build train/dev/test splits from data/raw/extracted--agg.csv. + +Raw rows are (src, rom, count). Pipeline: + 1. Normalize whitespace/punctuation on src and rom (no case folding). + 2. Aggregate exact (src, rom) duplicates by summing counts. + 3. Strict majority-wins on ambiguity: for each src with >1 distinct rom, + keep the rom with the strictly highest count; drop the src if the top + count is tied. (Diacritics preserved — "Tihrān" beats "Tihran" because + it has the higher count, not because we case/fold-normalized.) + 4. Shuffle deterministically and write train/dev/test CSVs to + data/source//. + +Run: python build_splits.py # all langs found in data/raw/ + python build_splits.py per ara # specific langs +""" + +import csv +import random +import re +import sys +from collections import defaultdict +from glob import glob +from os import makedirs, path +from string import punctuation, whitespace + +from s2s import CP_RANGE + + +RAW_DIR = "data/raw" +OUT_DIR = "data/source" +RAW_PATTERN = "extracted-{lang}-agg.csv" + +# Split ratios (must sum to 1.0). Matches the existing ~95/2.5/2.5 layout. +TRAIN_FRAC = 0.95 +DEV_FRAC = 0.025 +TEST_FRAC = 0.025 + +SEED = 42 + +# Whitespace runs (incl. tabs, NBSP, etc.) that should collapse to one space. +_WS_RE = re.compile(r"\s+") +# Whitespace adjacent to ASCII punctuation — strip it so " :" and ":" align. +_WS_PUNCT_RE = re.compile(r"\s+([,.;:!?\)\]\}])") +_PUNCT_WS_RE = re.compile(r"([\(\[\{])\s+") + + +def _foreign_seq(src, lang): + """ + Return numeric positions of foreign character sequences. + + Foreign sequences are characters not in the CP_RANGE for the given + language. + """ + cp_range = CP_RANGE.get(lang, CP_RANGE["latin"]) + + ranges = [] + eos = len(src) - 1 # Last character index. + s, e = None, None # Start and end markers for foreign sequences. + for i, ch in enumerate(src): + # Ignore whitespace and punctuation. + if ch in punctuation or ch in whitespace: + if i == eos and s is not None: + # At the end of a foreign sequence. + e = i + 1 + # print(f"Setting e to {e}") + if s is not None: + # Add range and reset markers. + ranges.append((s, e)) + s, e = None, None + else: + raise ValueError( + "Error computing sequence: missing start marker.") + continue + + in_range = False + for min_cp, max_cp in cp_range: + if ch >= min_cp and ch <= max_cp: + in_range = True + break + + if in_range: + # Character is in range. + if s is not None: + # We passed the end of a foreign sequence. Mark it. + e = i + # print(f"Setting e to {e}") + # Else: Continuation of a native sequence. + else: + # Character is not in range. + # print(f"{ch} at #{i} is not in range.") + if s is None: + # Stepping into a foreign sequence. Mark the start. + s = i + # print(f"Setting s to {s}") + if i == eos: + # At the end of a foreign sequence. + e = i + 1 + # print(f"Setting e to {e}") + # Else: continuation of a foreign sequence. + + if e is not None: + if s is not None: + # Add range and reset markers. + ranges.append((s, e)) + s, e = None, None + else: + raise ValueError( + "Error computing sequence: missing start marker.") + + # Verify that either both s and e are None, or both are set. + if (s is None) ^ (e is None): + raise ValueError("Error computing sequence: start and end don't match.") + + return ranges + + +def strip_foreign_seq(pair, lang): + """ + Strip foreign sequences from pairs. + """ + ranges = _foreign_seq(pair[0], lang) + + for s, e in ranges: + if pair[0].find(pair[0][s:e]) >= 0 and pair[0].find(pair[0][s:e]) >= 0: + pair[0] = pair[0].replace(pair[0][s:e], "") + pair[1] = pair[1].replace(pair[1][s:e], "") + + return pair + + +def normalize_text(s): + """Light normalization: strip, collapse whitespace, tighten punct spacing. + + Deliberately does NOT casefold or alter diacritics — diacritics are part + of the romanization signal we want to preserve. + """ + s = s.strip() + s = _WS_RE.sub(" ", s) + s = _WS_PUNCT_RE.sub(r"\1", s) + s = _PUNCT_WS_RE.sub(r"\1", s) + return s + + +def discover_langs(): + """Return language codes for every extracted-*-agg.csv in RAW_DIR.""" + langs = [] + for fpath in sorted(glob(path.join(RAW_DIR, "extracted-*-agg.csv"))): + base = path.basename(fpath) + lang = base[len("extracted-"):-len("-agg.csv")] + langs.append(lang) + return langs + + +def load_raw(lang): + """Read raw rows, normalize, and aggregate duplicates by summed count.""" + fpath = path.join(RAW_DIR, RAW_PATTERN.format(lang=lang)) + counts = defaultdict(int) + total = 0 + with open(fpath, newline="") as fh: + reader = csv.reader(fh) + for row in reader: + if len(row) < 2: + continue + row = strip_foreign_seq(row, lang) + src = normalize_text(row[0]) + rom = normalize_text(row[1]) + if not src or not rom: + continue + try: + c = int(row[2]) if len(row) >= 3 and row[2] else 1 + except ValueError: + c = 1 + counts[(src, rom)] += c + total += 1 + print(f"[{lang}] raw rows: {total}, unique (src,rom) post-normalize: {len(counts)}") + return counts + + +def resolve_majority(counts): + """Strict majority-wins disambiguation. + + For each src with multiple rom variants, keep the rom with the strictly + highest count. Tie at the top → drop the src entirely. + """ + by_src = defaultdict(dict) # src -> {rom: count} + for (src, rom), c in counts.items(): + by_src[src][rom] = c + + kept = [] + unambiguous = 0 + resolved = 0 + tied_drops = 0 + for src, rom_counts in by_src.items(): + if len(rom_counts) == 1: + rom = next(iter(rom_counts)) + kept.append((src, rom)) + unambiguous += 1 + continue + # Sort roms by descending count. + ranked = sorted(rom_counts.items(), key=lambda kv: kv[1], reverse=True) + top_rom, top_c = ranked[0] + runner_c = ranked[1][1] + if top_c > runner_c: + kept.append((src, top_rom)) + resolved += 1 + else: + tied_drops += 1 + + print( + f" unambiguous srcs: {unambiguous}; " + f"majority-resolved: {resolved}; " + f"tied-and-dropped: {tied_drops}; " + f"kept pairs: {len(kept)}" + ) + return kept + + +def split(pairs): + rng = random.Random(SEED) + pairs = list(pairs) + rng.shuffle(pairs) + n = len(pairs) + n_test = int(round(n * TEST_FRAC)) + n_dev = int(round(n * DEV_FRAC)) + n_train = n - n_dev - n_test + return { + "train": pairs[:n_train], + "dev": pairs[n_train:n_train + n_dev], + "test": pairs[n_train + n_dev:], + } + + +def write_split(lang, name, rows): + out_path = path.join(OUT_DIR, lang, f"{name}.csv") + makedirs(path.dirname(out_path), exist_ok=True) + with open(out_path, "w", newline="") as fh: + writer = csv.writer(fh) + writer.writerows(rows) + print(f" wrote {len(rows):>7} rows -> {out_path}") + + +def build(lang): + print(f"=== {lang} ===") + counts = load_raw(lang) + pairs = resolve_majority(counts) + splits = split(pairs) + for name, rows in splits.items(): + write_split(lang, name, rows) + + +def main(argv): + assert abs((TRAIN_FRAC + DEV_FRAC + TEST_FRAC) - 1.0) < 1e-9, \ + "split fractions must sum to 1.0" + langs = argv[1:] if len(argv) > 1 else discover_langs() + if not langs: + print(f"No raw files found in {RAW_DIR}.", file=sys.stderr) + sys.exit(1) + for lang in langs: + build(lang) + + +if __name__ == "__main__": + main(sys.argv) diff --git a/scriptshifter/hooks/seq2seq/data/normalized/.keep b/scriptshifter/hooks/seq2seq/data/normalized/.keep new file mode 100644 index 0000000..e69de29 diff --git a/scriptshifter/hooks/seq2seq/data/raw/.keep b/scriptshifter/hooks/seq2seq/data/raw/.keep new file mode 100644 index 0000000..e69de29 diff --git a/scriptshifter/hooks/seq2seq/data/source/.keep b/scriptshifter/hooks/seq2seq/data/source/.keep new file mode 100644 index 0000000..e69de29 diff --git a/scriptshifter/hooks/seq2seq/data/tokenizer/.keep b/scriptshifter/hooks/seq2seq/data/tokenizer/.keep new file mode 100644 index 0000000..e69de29 diff --git a/scriptshifter/hooks/seq2seq/data/train_state/.keep b/scriptshifter/hooks/seq2seq/data/train_state/.keep new file mode 100644 index 0000000..e69de29 diff --git a/scriptshifter/hooks/seq2seq/data/train_state/per/best.pth b/scriptshifter/hooks/seq2seq/data/train_state/per/best.pth new file mode 100644 index 0000000..f43033b Binary files /dev/null and b/scriptshifter/hooks/seq2seq/data/train_state/per/best.pth differ diff --git a/scriptshifter/hooks/seq2seq/model.py b/scriptshifter/hooks/seq2seq/model.py new file mode 100644 index 0000000..6257ae9 --- /dev/null +++ b/scriptshifter/hooks/seq2seq/model.py @@ -0,0 +1,848 @@ +#!/usr/bin/env python + +# original code: https://machinelearningmastery.com/building-a-seq2seq-model- +# with-attention-for-language-translation/ +# Heavily modified by hand & with AI assistant to support S2R transliteration. + +import csv +import random +from logging import getLogger +from os import makedirs, path +from shutil import copy +from unicodedata import normalize + +import torch +import torch.nn as nn +import torch.nn.functional as F +import torch.optim as optim +import tokenizers +import tqdm + +# Script-specific modules. +# Persian +from shekar import Normalizer + + +# Data root folder. +DATA_ROOT = path.join(path.dirname(__file__), "data") + +# Code point range for all Arabic scripts. +ARA_CP = ( + ("\u0600", "\u06FF"), # Arabic + ("\u0750", "\u077F"), # Arabic Supplement + ("\u08A0", "\u08FF"), # Arabic Extended-A + ("\u0870", "\u089F"), # Arabic Extended-B + ("\uFB50", "\uFDFF"), # Arabic Presentation Forms-A + ("\uFE70", "\uFEFF"), # Arabic Presentation Forms-B + ("\U00010EC0", "\U00010EFF"), # Arabic Extended-C + ("\U0001EE00", "\U0001EEFF"), # Arabic Mathematical Alphabetic Symbols +) + +# Valid code point ranges for each language. +# The values are 2D tuples, with the inner elements being 2-character tuples +# representing a code point range (min, max). +CP_RANGE = { + # Space (\u0020) is the lowest printable code point. + "latin": (("\u0020", "\u036F"),), + "ara": ARA_CP, + "per": ARA_CP, +} + +NORMALIZER = { + "per": Normalizer(), +} + +DEVICE = torch.device('cuda:0' if torch.cuda.is_available() else 'cpu') + +# Tokens. +SOS_TOK = "[start]" +EOS_TOK = "[end]" +PAD_TOK = "[pad]" +SEP_TOK = "[sep]" +CLS_TOK = "[cls]" +UNK_TOK = "[unk]" + +# Tokenizer parameters for script only. +VOCAB_SIZE = 16000 + +# Model parameters. These have been tuned to a 275K data set. +EMB_DIM = 256 +HIDDEN_DIM = 256 +DROPOUT = 0.2 +N_LAYERS = 1 +LR = 4e-4 +WEIGHT_DECAY = 1e-5 +GRAD_CLIP = 0.5 + +# Training parameters. +N_EPOCHS = 5 # Number of epochs to train by. +BATCH_SIZE = 64 # Data loader batch size. +# Filter out outlier-length pairs to bound memory per batch. +MAX_SRC_CHARS = 300 +MAX_TGT_CHARS = MAX_SRC_CHARS * 1.33 + +logger = getLogger(__name__) + + +# +# Read raw data +# + +def _in_range(s, lang): + """ + Whether a string is within a character range. + + Returns true or false, whether at least one character in the string is + within the code point range defined by CP_RANGE for the given language. + """ + cp_range = CP_RANGE.get(lang, CP_RANGE["latin"]) + + for ch in s: + for min_cp, max_cp in cp_range: + if ch >= min_cp and ch <= max_cp: + return True + return False + + +def _levenshtein(a, b): + """ + Edit distance between two sequences (strings or lists). + + O(len(a)*len(b)). + """ + if a == b: + return 0 + if not a: + return len(b) + if not b: + return len(a) + prev = list(range(len(b) + 1)) + for i, ai in enumerate(a, 1): + curr = [i] + [0] * len(b) + for j, bj in enumerate(b, 1): + cost = 0 if ai == bj else 1 + curr[j] = min( + curr[j - 1] + 1, # insertion + prev[j] + 1, # deletion + prev[j - 1] + cost, # substitution + ) + prev = curr + return prev[-1] + + +def read_langs(script, split="train"): + logger.info(f"Reading sources ({split})...") + src_path = path.join(DATA_ROOT, "source", script, f"{split}.csv") + norm_fpath = path.join(DATA_ROOT, "normalized", script, f"{split}.csv") + makedirs(path.dirname(norm_fpath), exist_ok=True) + + if path.isfile(norm_fpath): + logger.debug("Reusing cached token pairs.") + with open(norm_fpath, newline="") as fh: + reader = csv.reader(fh) + pairs = [row for row in reader] + else: + # Read the file and split into lines + with open(src_path, newline="") as fh: + reader = csv.reader(fh) + pairs = [ + (NORMALIZER[script](row[0]), normalize("NFKC", row[1])) + for row in reader + if _in_range(row[0], script) + ] + pre_filter = len(pairs) + pairs = [ + (s, r) for s, r in pairs + if len(s) <= MAX_SRC_CHARS and len(r) <= MAX_TGT_CHARS + ] + dropped = pre_filter - len(pairs) + if dropped: + logger.debug( + f"Filtered {dropped}/{pre_filter} ({split}) pairs " + "over length cap." + ) + + with open(norm_fpath, "w", newline="") as fh: + writer = csv.writer(fh) + for line in pairs: + writer.writerow(line) + logger.debug(f"Wrote normalized token pairs to {norm_fpath}.") + + return pairs + + +# +# Tokenization +# + +def tokenize(lang, code, vocab, level="bpe"): + """Build or load a tokenizer. + + level="bpe": byte-level BPE with a 16k vocab — used for the script side. + level="char": character-level vocab (~40 symbols) — used for the Roman + output side, where graphemes align directly to characters and a small + output vocab improves generalization for transliteration. + """ + tok_datadir = path.join(DATA_ROOT, "tokenizer", lang) + fname = path.join(tok_datadir, f"{code}_tokenizer.json") + + if path.exists(fname): + logger.debug("Reused token data.") + return tokenizers.Tokenizer.from_file(fname) + + if level == "char" or code == "rom": + # Build vocab from observed characters plus specials. + chars = sorted({c for s in vocab for c in s}) + char_vocab = {tok: i for i, tok in enumerate( + [PAD_TOK, SOS_TOK, EOS_TOK, UNK_TOK] + chars)} + tokenizer = tokenizers.Tokenizer( + tokenizers.models.WordLevel(char_vocab, unk_token=UNK_TOK)) + # Split on every character so each char becomes a token. + tokenizer.pre_tokenizer = tokenizers.pre_tokenizers.Split( + pattern=tokenizers.Regex(""), behavior="isolated") + tokenizer.decoder = tokenizers.decoders.Fuse() + logger.debug("Generated character-level token data.") + else: + tokenizer = tokenizers.Tokenizer(tokenizers.models.BPE()) + tokenizer.pre_tokenizer = tokenizers.pre_tokenizers.ByteLevel( + add_prefix_space=True) + tokenizer.decoder = tokenizers.decoders.ByteLevel() + trainer = tokenizers.trainers.BpeTrainer( + vocab_size=VOCAB_SIZE, + special_tokens=[SOS_TOK, EOS_TOK, PAD_TOK, UNK_TOK], + show_progress=True + ) + tokenizer.train_from_iterator(vocab, trainer=trainer) + logger.debug("Generated BPE token data.") + + # Auto-add SOS/EOS at encode time so the dataset doesn't have to + # string-wrap them (which would split under char-level tokenization). + sos_id = tokenizer.token_to_id(SOS_TOK) + eos_id = tokenizer.token_to_id(EOS_TOK) + tokenizer.post_processor = tokenizers.processors.TemplateProcessing( + single=f"{SOS_TOK} $A {EOS_TOK}", + special_tokens=[(SOS_TOK, sos_id), (EOS_TOK, eos_id)], + ) + tokenizer.enable_padding( + pad_id=tokenizer.token_to_id(PAD_TOK), pad_token=PAD_TOK) + makedirs(tok_datadir, exist_ok=True) + tokenizer.save(fname) + logger.debug("Saved token data cache.") + + return tokenizer + + +# Create PyTorch dataset for the BPE-encoded translation pairs +# +# Map-style dataset: +# https://docs.pytorch.org/docs/stable/data.html#map-style-datasets +class TransliterationDataset(torch.utils.data.Dataset): + def __init__(self, text_pairs): + self.text_pairs = text_pairs + + def __len__(self): + return len(self.text_pairs) + + def __getitem__(self, idx): + # SOS/EOS are added by the tokenizer's post-processor. + return self.text_pairs[idx] + + +def get_collate_fn(scr_tokenizer, rom_tokenizer): + def collate_fn(batch): + scr_str, rom_str = zip(*batch) + scr_enc = scr_tokenizer.encode_batch(scr_str, add_special_tokens=True) + rom_enc = rom_tokenizer.encode_batch(rom_str, add_special_tokens=True) + + return ( + torch.tensor([enc.ids for enc in scr_enc]), + torch.tensor([enc.ids for enc in rom_enc]) + ) + + return collate_fn + + +def get_dataloaders(lang): + train_pairs = read_langs(lang, "train") + logger.debug("Loaded pairs.") + # Tokenizers are fit on training data only + scr_tokenizer = tokenize(lang, "scr", [x[0] for x in train_pairs]) + logger.debug("Tokenized script.") + # Char-level for the Roman output: ~40-symbol vocab aligns to graphemes + # and avoids BPE merges that don't correspond to script boundaries. + rom_tokenizer = tokenize(lang, "rom", [x[1] for x in train_pairs]) + logger.debug("Tokenized Roman.") + + collate = get_collate_fn(scr_tokenizer, rom_tokenizer) + train_loader = torch.utils.data.DataLoader( + TransliterationDataset(train_pairs), + batch_size=BATCH_SIZE, shuffle=True, collate_fn=collate, + ) + logger.debug("Collated datasets.") + + dev_path = path.join(DATA_ROOT, "source", lang, "dev.csv") + if path.exists(dev_path): + dev_pairs = read_langs(lang, "dev") + dev_loader = torch.utils.data.DataLoader( + TransliterationDataset(dev_pairs), + batch_size=BATCH_SIZE, shuffle=False, collate_fn=collate, + ) + else: + dev_loader = None + logger.debug("Set up loaders.") + + return train_loader, dev_loader, scr_tokenizer, rom_tokenizer + + +# +# Seq2seq model with attention for transliteration +# + +class EncoderRNN(nn.Module): + """A bidirectional GRU encoder with an embedding layer. + + Outputs are projected from 2*hidden_dim back down to hidden_dim so the + decoder's attention can match dimensions. The forward and backward final + hidden states are combined into a single decoder-init hidden state. + """ + def __init__( + self, vocab_size, embedding_dim, hidden_dim, + num_layers=1, dropout=0.1 + ): + super().__init__() + self.vocab_size = vocab_size + self.embedding_dim = embedding_dim + self.hidden_dim = hidden_dim + self.num_layers = num_layers + + self.embedding = nn.Embedding(vocab_size, embedding_dim) + self.gru = nn.GRU( + embedding_dim, hidden_dim, + num_layers=num_layers, + batch_first=True, + bidirectional=True, + dropout=dropout if num_layers > 1 else 0.0, + ) + self.out_proj = nn.Linear(2 * hidden_dim, hidden_dim) + self.hidden_proj = nn.Linear(2 * hidden_dim, hidden_dim) + self.dropout = nn.Dropout(dropout) + + def forward(self, input_seq): + embedded = self.dropout(self.embedding(input_seq)) + # outputs: [B, S, 2H], hidden: [2*num_layers, B, H] + outputs, hidden = self.gru(embedded) + outputs = self.out_proj(outputs) # [B, S, H] + # Combine fwd/bwd final states for every layer so each decoder layer + # gets a corresponding seed. hidden is laid out as + # [layer0_fwd, layer0_bwd, layer1_fwd, layer1_bwd, ...]. + B, H = hidden.size(1), hidden.size(2) + hidden = hidden.view(self.num_layers, 2, B, H) + h_cat = torch.cat([hidden[:, 0], hidden[:, 1]], dim=-1) # [L, B, 2H] + dec_hidden = torch.tanh(self.hidden_proj(h_cat)) # [L, B, H] + return outputs, dec_hidden + + +class BahdanauAttention(nn.Module): + """Location-aware Bahdanau attention. + + Standard content-based scoring (Bahdanau 2014) augmented with features + derived from the previous timestep's attention weights, as in Chorowski + et al. 2015 (https://arxiv.org/abs/1506.07503). The location features + bias attention to advance smoothly along the source — a useful inductive + prior for monotonic tasks like transliteration, where alignment never + reorders. Unlike strict monotonic attention, this is still soft: the + model can revisit earlier positions if needed (e.g. for digraphs). + + Location features are produced by a 1D conv over the previous attention + distribution, then projected into the score-energy space. + """ + def __init__(self, hidden_size, loc_kernel_size=31, loc_features=32): + super().__init__() + self.Wa = nn.Linear(hidden_size, hidden_size) + self.Ua = nn.Linear(hidden_size, hidden_size) + self.Va = nn.Linear(hidden_size, 1) + # Conv1d expects [B, C_in, S]; previous weights are [B, 1, S]. + # Padding keeps S unchanged. + assert loc_kernel_size % 2 == 1, "kernel size must be odd" + self.loc_conv = nn.Conv1d( + 1, loc_features, + kernel_size=loc_kernel_size, + padding=loc_kernel_size // 2, + bias=False) + self.loc_proj = nn.Linear(loc_features, hidden_size, bias=False) + + def forward(self, query, keys, mask=None, prev_attn=None): + """ + Args: + query: [B, 1, H] + keys: [B, S, H] + mask: [B, S] bool — True for valid (non-pad) positions + prev_attn: [B, 1, S] — previous step's attention weights, or None + on the first step (treated as a uniform prior over valid + positions). + + Returns: + context: [B, 1, H] + weights: [B, 1, S] + """ + B, S, H = keys.shape + assert query.shape == (B, 1, H) + + if prev_attn is None: + # Uniform prior over valid positions for the first step. + if mask is not None: + valid = mask.float() + lengths = valid.sum(dim=-1, keepdim=True).clamp(min=1) + prev_attn = (valid / lengths).unsqueeze(1) # [B, 1, S] + else: + prev_attn = keys.new_full((B, 1, S), 1.0 / S) + + # Conv expects [B, 1, S]; outputs [B, F, S] -> [B, S, F] + loc_feats = self.loc_conv(prev_attn).transpose(1, 2) + loc_term = self.loc_proj(loc_feats) # [B, S, H] + + scores = self.Va(torch.tanh( + self.Wa(query) + self.Ua(keys) + loc_term)) # [B, S, 1] + scores = scores.transpose(1, 2) # [B, 1, S] + if mask is not None: + scores = scores.masked_fill(~mask.unsqueeze(1), float("-inf")) + weights = F.softmax(scores, dim=-1) + context = torch.bmm(weights, keys) + return context, weights + + +class DecoderRNN(nn.Module): + def __init__( + self, vocab_size, embedding_dim, hidden_dim, + num_layers=1, dropout=0.1, tie_weights=True, + ): + super().__init__() + self.vocab_size = vocab_size + self.embedding_dim = embedding_dim + self.hidden_dim = hidden_dim + self.num_layers = num_layers + + self.embedding = nn.Embedding(vocab_size, embedding_dim) + self.dropout = nn.Dropout(dropout) + self.attention = BahdanauAttention(hidden_dim) + self.gru = nn.GRU( + embedding_dim + hidden_dim, hidden_dim, + num_layers=num_layers, + batch_first=True, + dropout=dropout if num_layers > 1 else 0.0, + ) + self.out_proj = nn.Linear(hidden_dim, vocab_size) + if tie_weights: + # Share the input-embedding matrix with the output projection. + # Saves vocab_size * hidden_dim parameters and tends to improve + # generalization on small output vocabularies (common for + # transliteration). Requires embedding_dim == hidden_dim. + assert embedding_dim == hidden_dim, ( + "tie_weights requires embedding_dim == hidden_dim" + ) + self.out_proj.weight = self.embedding.weight + + def forward(self, input_seq, hidden, enc_out, enc_mask=None, + prev_attn=None): + """Single token input, single token output. + + Returns (output, hidden, attn_weights). The attn_weights should be + passed back as `prev_attn` on the next call to enable location-aware + attention to track its own progress along the source sequence. + """ + embedded = self.dropout(self.embedding(input_seq)) + # Use top layer's hidden state as the attention query. + query = hidden[-1:].transpose(0, 1) # [B, 1, H] + context, attn_weights = self.attention( + query, enc_out, enc_mask, prev_attn) + # Luong-style dropout on the attention context: regularizes the + # decoder's reliance on attention so the recurrent state carries + # backup signal when attention is imperfect at inference. + context = self.dropout(context) + rnn_input = torch.cat([embedded, context], dim=-1) + rnn_output, hidden = self.gru(rnn_input, hidden) + output = self.out_proj(rnn_output) + return output, hidden, attn_weights + + +class Seq2SeqRNN(nn.Module): + def __init__(self, encoder, decoder, src_pad_id): + super().__init__() + self.encoder = encoder + self.decoder = decoder + self.src_pad_id = src_pad_id + + def forward(self, input_seq, target_seq): + """Given the partial target sequence, predict the next token""" + batch_size, target_len = target_seq.shape + enc_mask = input_seq != self.src_pad_id # [B, S] + outputs = [] + enc_out, dec_hidden = self.encoder(input_seq) + prev_attn = None + for t in range(target_len - 1): + # Teacher forcing: feed the ground-truth previous token. + dec_in = target_seq[:, t].unsqueeze(1) + dec_out, dec_hidden, prev_attn = self.decoder( + dec_in, dec_hidden, enc_out, enc_mask, prev_attn) + outputs.append(dec_out) + outputs = torch.cat(outputs, dim=1) + return outputs + + +class S2S: + def __init__(self, lang, state_fpath=None): + """ + Instantiate a Seq2Seq model. + + @param lang (str) Language code. "per" and "ara" are supported. + + @param state_fpath (str) State file. Defaults to a predefined state + file path based on the language selected. If the file is not found, + the model must be retrained. + """ + self.lang = lang + + # Data loaders. + (self.train_loader, self.dev_loader, + self.scr_tokenizer, self.rom_tokenizer) = get_dataloaders(self.lang) + self.enc_dim = len(self.scr_tokenizer.get_vocab()) + self.dec_dim = len(self.rom_tokenizer.get_vocab()) + self.src_pad_id = self.scr_tokenizer.token_to_id(PAD_TOK) + self.tgt_pad_id = self.rom_tokenizer.token_to_id(PAD_TOK) + + # Encoder & decoder. + encoder = EncoderRNN( + self.enc_dim, EMB_DIM, HIDDEN_DIM, N_LAYERS, DROPOUT + ).to(DEVICE) + decoder = DecoderRNN( + self.dec_dim, EMB_DIM, HIDDEN_DIM, N_LAYERS, DROPOUT + ).to(DEVICE) + + # Seq2SeqRNN model. + self.model = Seq2SeqRNN(encoder, decoder, self.src_pad_id).to(DEVICE) + state_dir = path.join(DATA_ROOT, "train_state", self.lang) + self.state_fpath = path.join(state_dir, "checkpoint.pth") + self.best_fpath = path.join(state_dir, "best.pth") + # Prefer the best-on-dev checkpoint when both exist. + load_path = ( + state_fpath if state_fpath and path.exists(self.state_fpath) + else self.best_fpath if path.exists(self.best_fpath) + else self.state_fpath if path.exists(self.state_fpath) + else None + ) + if load_path is not None: + logger.debug(f"Loading checkpoint: {load_path}") + self.model.load_state_dict(torch.load( + load_path, map_location=DEVICE)) + self.trained = True + else: + logger.warn("Model is not trained.") + self.trained = False + + total_params = sum( + p.numel() for p in self.model.parameters() if p.requires_grad) + logger.debug(f"Seq2Seq model created for language: {lang}") + logger.debug("Parameters:") + logger.debug(f" Input vocabulary size: {self.enc_dim}") + logger.debug(f" Output vocabulary size: {self.dec_dim}") + logger.debug(f" Embedding dimension: {EMB_DIM}") + logger.debug(f" Hidden dimension: {HIDDEN_DIM}") + logger.debug(f" Dropout: {DROPOUT}") + logger.debug(f" Total parameters: {total_params}") + + def train(self, epochs=N_EPOCHS, eval_every=5, patience=5): + """Train with LR-on-plateau and best-checkpoint-on-dev-loss. + + eval_every: run dev evaluation every N epochs. + patience: stop after this many consecutive eval cycles without + improvement on dev loss. Ignored if no dev set is configured. + """ + logger.info(f"Training for up to {epochs} epochs.") + if self.trained: + logger.debug("Backing up existing state file.") + copy(self.state_fpath, self.state_fpath + ".bk") + else: + makedirs(path.dirname(self.state_fpath), exist_ok=True) + + optimizer = optim.AdamW( + self.model.parameters(), lr=LR, weight_decay=WEIGHT_DECAY) + loss_fn = nn.CrossEntropyLoss(ignore_index=self.tgt_pad_id) + # Linear warmup for the first epoch, then plateau decay on dev loss. + warmup_steps = max(1, len(self.train_loader)) + warmup = optim.lr_scheduler.LinearLR( + optimizer, start_factor=0.1, end_factor=1.0, + total_iters=warmup_steps) + scheduler = optim.lr_scheduler.ReduceLROnPlateau( + optimizer, mode="min", factor=0.5, patience=1) + + best_dev_loss = float("inf") + stale_evals = 0 + + for epoch in range(epochs): + self.model.train() + epoch_loss = 0 + for scr_ids, rom_ids in tqdm.tqdm( + self.train_loader, desc="Training"): + scr_ids = scr_ids.to(DEVICE) + rom_ids = rom_ids.to(DEVICE) + optimizer.zero_grad() + outputs = self.model(scr_ids, rom_ids) + loss = loss_fn(outputs.reshape( + -1, self.dec_dim), rom_ids[:, 1:].reshape(-1)) + loss.backward() + torch.nn.utils.clip_grad_norm_( + self.model.parameters(), GRAD_CLIP) + optimizer.step() + if warmup.last_epoch < warmup.total_iters: + warmup.step() + epoch_loss += loss.item() + logger.info( + f"Epoch {epoch+1}/{epochs}; " + f"Avg loss {epoch_loss/len(self.train_loader)}; " + f"Latest loss {loss.item()}" + ) + # Latest snapshot — overwritten every epoch. + torch.save(self.model.state_dict(), self.state_fpath) + + if (epoch + 1) % eval_every != 0 or self.dev_loader is None: + continue + self.model.eval() + eval_loss = 0 + with torch.no_grad(): + for scr_ids, rom_ids in tqdm.tqdm( + self.dev_loader, desc="Evaluating"): + scr_ids = scr_ids.to(DEVICE) + rom_ids = rom_ids.to(DEVICE) + outputs = self.model(scr_ids, rom_ids) + loss = loss_fn(outputs.reshape( + -1, self.dec_dim), rom_ids[:, 1:].reshape(-1)) + eval_loss += loss.item() + avg_dev = eval_loss / len(self.dev_loader) + current_lr = optimizer.param_groups[0]["lr"] + logger.info( + f"Eval loss (dev): {avg_dev:.4f}; lr: {current_lr:.2e}") + + scheduler.step(avg_dev) + + if avg_dev < best_dev_loss: + best_dev_loss = avg_dev + torch.save(self.model.state_dict(), self.best_fpath) + logger.info(f"New best dev loss → saved {self.best_fpath}") + stale_evals = 0 + else: + stale_evals += 1 + logger.info(f"No dev improvement ({stale_evals}/{patience}).") + if stale_evals >= patience: + logger.info(f"Early stopping at epoch {epoch+1}.") + break + + torch.save(self.model.state_dict(), self.state_fpath) + # Reload best weights so the live model reflects the best checkpoint. + if path.exists(self.best_fpath): + self.model.load_state_dict(torch.load(self.best_fpath)) + logger.info( + f"Reloaded best dev checkpoint (loss {best_dev_loss:.4f})." + ) + self.trained = True + + def _greedy_decode(self, src, max_len): + scr_ids = torch.tensor( + self.scr_tokenizer.encode(src).ids + ).unsqueeze(0).to(DEVICE) + enc_mask = scr_ids != self.src_pad_id + enc_out, hidden = self.model.encoder(scr_ids) + prev_token = torch.tensor( + [[self.rom_tokenizer.token_to_id(SOS_TOK)]] + ).to(DEVICE) + eos_id = self.rom_tokenizer.token_to_id(EOS_TOK) + pred_ids = [] + prev_attn = None + for _ in range(max_len): + output, hidden, prev_attn = self.model.decoder( + prev_token, hidden, enc_out, enc_mask, prev_attn) + output = output.argmax(dim=2) + pred_ids.append(output.item()) + prev_token = output + if pred_ids[-1] == eos_id: + break + return pred_ids + + def _beam_decode(self, src, max_len, beam_size=4, length_penalty=0.6): + """Beam search with Wu et al. length normalization. + + Returns the token-id list of the highest-scoring completed hypothesis, + or the best live beam if none completed within max_len. + """ + sos_id = self.rom_tokenizer.token_to_id(SOS_TOK) + eos_id = self.rom_tokenizer.token_to_id(EOS_TOK) + + scr_ids = torch.tensor( + self.scr_tokenizer.encode(src).ids + ).unsqueeze(0).to(DEVICE) + enc_mask = scr_ids != self.src_pad_id + enc_out, hidden = self.model.encoder(scr_ids) + + # Tile encoder state across the beam dimension. + # enc_out: [B=1, S, H] -> [K, S, H]; hidden: [L, B=1, H] -> [L, K, H] + enc_out = enc_out.expand(beam_size, -1, -1).contiguous() + enc_mask = enc_mask.expand(beam_size, -1).contiguous() + hidden = hidden.expand(-1, beam_size, -1).contiguous() + + # Per-beam state. + seqs = torch.full( + (beam_size, 1), sos_id, dtype=torch.long, device=DEVICE) + scores = torch.zeros(beam_size, device=DEVICE) + # Mark all but the first beam as -inf so step 1 only expands beam 0 + # (otherwise all beams start identical and produce K duplicate top-K). + scores[1:] = float("-inf") + + finished = [] # list of (normalized_score, token_ids) + prev_attn = None # threaded through location-aware attention + + def lp(length): + return ((5 + length) / 6) ** length_penalty + + for step in range(max_len): + prev_token = seqs[:, -1:] # [K, 1] + output, hidden, prev_attn = self.model.decoder( + prev_token, hidden, enc_out, enc_mask, prev_attn) + log_probs = F.log_softmax(output.squeeze(1), dim=-1) # [K, V] + V = log_probs.size(-1) + + # Total scores for all K*V continuations. + total = scores.unsqueeze(1) + log_probs # [K, V] + flat = total.view(-1) + top_scores, top_idx = flat.topk(beam_size) + beam_idx = top_idx // V # which parent beam + tok_idx = top_idx % V # which token + + new_seqs = torch.cat( + [seqs[beam_idx], tok_idx.unsqueeze(1)], dim=1) + # Reorder hidden state and prev_attn to match the chosen parents. + hidden = hidden[:, beam_idx, :].contiguous() + prev_attn = prev_attn[beam_idx].contiguous() + scores = top_scores + + # Move EOS-terminated beams to finished and replace with -inf so + # they no longer compete for top-K next round. + still_alive = [] + for i in range(beam_size): + if tok_idx[i].item() == eos_id: + seq = new_seqs[i].tolist() + norm = scores[i].item() / lp(len(seq) - 1) # exclude SOS + finished.append((norm, seq)) + scores[i] = float("-inf") + else: + still_alive.append(i) + + seqs = new_seqs + if len(finished) >= beam_size or not still_alive: + break + + if finished: + finished.sort(key=lambda x: x[0], reverse=True) + best = finished[0][1] + else: + # No EOS within max_len — pick best live beam, length-normalized. + best_i = max( + range(beam_size), + key=lambda i: scores[i].item() / lp(seqs.size(1) - 1)) + best = seqs[best_i].tolist() + + # Strip leading SOS; trailing EOS (if present) is fine for the decoder. + return best[1:] + + def transliterate(self, src, beam_size=4): + # Apply training-time normalization so the tokenizer sees the same + # form it was trained on (e.g. Arabic yeh → Persian yeh). + src = NORMALIZER[self.lang](src) + self.model.eval() + with torch.no_grad(): + if beam_size <= 1: + pred_ids = self._greedy_decode(src, max_len=MAX_SRC_CHARS) + else: + pred_ids = self._beam_decode( + src, max_len=MAX_SRC_CHARS, beam_size=beam_size) + eos_id = self.rom_tokenizer.token_to_id(EOS_TOK) + if pred_ids and pred_ids[-1] == eos_id: + pred_ids = pred_ids[:-1] + return self.rom_tokenizer.decode(pred_ids) + + def sample_predictions(self, ct=5, beam_size=4, split="dev"): + """Print a handful of full predictions for visual inspection.""" + self.model.eval() + pairs = read_langs(self.lang, split) + eos_id = self.rom_tokenizer.token_to_id(EOS_TOK) + with torch.no_grad(): + for scr, true_rom in random.sample(pairs, ct): + if beam_size <= 1: + pred_ids = self._greedy_decode(scr, max_len=60) + else: + pred_ids = self._beam_decode( + scr, max_len=60, beam_size=beam_size) + if pred_ids and pred_ids[-1] == eos_id: + pred_ids = pred_ids[:-1] + pred_rom = self.rom_tokenizer.decode(pred_ids) + print(f"Script: {scr}") + print(f"Roman: {true_rom}") + print(f"Predicted: {pred_rom}") + + def evaluate(self, split="test", beam_size=4, max_len=60, limit=None): + """End-to-end transliteration metrics on a held-out split. + + Runs full inference (beam or greedy) over every pair in the split + and reports: + - exact match: predicted == reference (after stripping trailing EOS) + - CER: character error rate (Levenshtein / |reference|) + - WER: word error rate (token-level Levenshtein / |reference words|) + + Args: + split: which CSV under data/source// to evaluate. + beam_size: 1 for greedy, >1 for beam search. + max_len: decoder step cap. + limit: optional cap on number of pairs to evaluate (for spot + checks during training). + """ + self.model.eval() + pairs = read_langs(self.lang, split) + if limit is not None: + pairs = pairs[:limit] + + eos_id = self.rom_tokenizer.token_to_id(EOS_TOK) + n = len(pairs) + exact = 0 + char_edits = 0 + char_total = 0 + word_edits = 0 + word_total = 0 + + with torch.no_grad(): + for scr, true_rom in tqdm.tqdm(pairs, desc=f"Evaluating {split}"): + if beam_size <= 1: + pred_ids = self._greedy_decode(scr, max_len=max_len) + else: + pred_ids = self._beam_decode( + scr, max_len=max_len, beam_size=beam_size) + # Strip a trailing EOS if present so it doesn't pollute CER. + if pred_ids and pred_ids[-1] == eos_id: + pred_ids = pred_ids[:-1] + pred = self.rom_tokenizer.decode(pred_ids).strip() + ref = true_rom.strip() + + if pred == ref: + exact += 1 + char_edits += _levenshtein(pred, ref) + char_total += max(len(ref), 1) + pred_words = pred.split() + ref_words = ref.split() + word_edits += _levenshtein(pred_words, ref_words) + word_total += max(len(ref_words), 1) + + em = exact / n if n else 0.0 + cer = char_edits / char_total if char_total else 0.0 + wer = word_edits / word_total if word_total else 0.0 + print( + f"\n[{split}] n={n} exact={em:.4f} CER={cer:.4f} WER={wer:.4f}" + f" beam={beam_size}" + ) + return {"n": n, "exact_match": em, "cer": cer, "wer": wer} diff --git a/scriptshifter/hooks/seq2seq/requirements.txt b/scriptshifter/hooks/seq2seq/requirements.txt new file mode 100644 index 0000000..e2da3de --- /dev/null +++ b/scriptshifter/hooks/seq2seq/requirements.txt @@ -0,0 +1,5 @@ +matplotlib +shekar +tokenizers +torch +tqdm diff --git a/scriptshifter/tables/data/persian.yml b/scriptshifter/tables/data/persian.yml index fa82e13..a5699de 100644 --- a/scriptshifter/tables/data/persian.yml +++ b/scriptshifter/tables/data/persian.yml @@ -2,12 +2,19 @@ general: name: Persian case_sensitive: false - description: Persian language in Arabic script (roman_to_script only). - version: 1.0.0 - date: 2025-12-23 + description: Persian language in Arabic script, using NLP. + version: 2.0.0 + date: 2026-06-19 parents: - _ignore_base +script_to_roman: + hooks: + post_config: + - + - seq2seq.s2r_post_config + - src_script: "per" + roman_to_script: map: # Punctuation marks: diff --git a/scriptshifter/tables/data/tamil_brahmi.yml b/scriptshifter/tables/data/tamil_brahmi.yml deleted file mode 100644 index 151aa78..0000000 --- a/scriptshifter/tables/data/tamil_brahmi.yml +++ /dev/null @@ -1,21 +0,0 @@ ---- -general: - name: Tamil Brahmi - case_sensitive: false - description: Tamil language in Tamil Brahmi script using third-part software. - version: 1.0.0 - date: 2025-12-23 - -script_to_roman: - hooks: - post_config: - - - - aksharamukha.romanizer.s2r_post_config - - src_script: "TamilBrahmi" - -roman_to_script: - hooks: - post_config: - - - - aksharamukha.romanizer.r2s_post_config - - dest_script: "TamilBrahmi" diff --git a/scriptshifter/tables/data/tamil_extended.yml b/scriptshifter/tables/data/tamil_extended.yml deleted file mode 100644 index 4387c06..0000000 --- a/scriptshifter/tables/data/tamil_extended.yml +++ /dev/null @@ -1,21 +0,0 @@ ---- -general: - name: Tamil (extended) - case_sensitive: false - description: Tamil language in extended Tamil script using third-party software. - version: 1.0.0 - date: 2025-12-23 - -script_to_roman: - hooks: - post_config: - - - - aksharamukha.romanizer.s2r_post_config - - src_script: "TamilExtended" - -roman_to_script: - hooks: - post_config: - - - - aksharamukha.romanizer.r2s_post_config - - dest_script: "TamilExtended" diff --git a/scriptshifter/tables/data/thai.yml b/scriptshifter/tables/data/thai.yml index c630c26..7b28eb0 100644 --- a/scriptshifter/tables/data/thai.yml +++ b/scriptshifter/tables/data/thai.yml @@ -9,11 +9,13 @@ general: - _ignore_base script_to_roman: - hooks: - post_normalize: - - - - asian_tokenizer.s2r_tokenize - - model: "th" + # DISABLING word separator (esupar). Latest version has broken deps and + # SS won't start. + #hooks: + # post_normalize: + # - + # - asian_tokenizer.s2r_tokenize + # - model: "th" map: # COMMON SPECIAL CHARACTERS diff --git a/scriptshifter/tables/index.yml b/scriptshifter/tables/index.yml index 68790d1..172222c 100644 --- a/scriptshifter/tables/index.yml +++ b/scriptshifter/tables/index.yml @@ -468,12 +468,6 @@ tamazight_moroccan: tamil: marc_code: tam name: Tamil -tamil_brahmi: - marc_code: tam - name: Tamil Brahmi -tamil_extended: - marc_code: tam - name: Tamil (extended) tat_cyrillic: marc_code: ira name: Tat (Cyrillic) diff --git a/scriptshifter/trans.py b/scriptshifter/trans.py index a907635..acc06e0 100644 --- a/scriptshifter/trans.py +++ b/scriptshifter/trans.py @@ -1,4 +1,4 @@ -import logging +from logging import getLogger from importlib import import_module from re import Pattern @@ -11,7 +11,7 @@ get_connection, get_lang_dcap, get_lang_general, get_lang_hooks, get_lang_ignore, get_lang_map, get_lang_normalize) -logger = logging.getLogger(__name__) +logger = getLogger(__name__) # Beginning-of-word pattern. BOW_PTN = compile(r"(?<=[\p{P}\p{Z}]|^)[\p{L}\p{M}\p{S}]") diff --git a/scriptshifter_base.Dockerfile b/scriptshifter_base.Dockerfile index 70b59ed..143a021 100644 --- a/scriptshifter_base.Dockerfile +++ b/scriptshifter_base.Dockerfile @@ -1,23 +1,26 @@ -FROM python:3.10-slim-bookworm +FROM python:3.13-slim-trixie -RUN apt update -RUN apt install -y build-essential cmake tzdata gfortran libopenblas-dev libboost-all-dev libpcre2-dev +ENV LC_ALL=C.UTF-8 +ENV TZ="America/New_York" +ENV WORKROOT="/usr/local/scriptshifter/src" -ENV TZ=America/New_York -ARG WORKROOT "/usr/local/scriptshifter/src" +RUN apt update +RUN apt install -y locales tzdata build-essential git +RUN locale-gen +RUN dpkg-reconfigure locales RUN addgroup --system www RUN adduser --system www RUN gpasswd -a www www -ENV HF_DATASETS_CACHE /data/hf/datasets +ENV HF_DATASETS_CACHE="/data/hf/datasets" # Copy external dependencies. WORKDIR ${WORKROOT} COPY ext ./ext/ -COPY deps.txt ./ +COPY requirements.txt ./ ENV CFLAGS="-DCMAKE_POLICY_VERSION_MINIMUM=3.5" -RUN pip install --no-cache-dir -r deps.txt +RUN pip install --no-cache-dir --break-system-packages -r requirements.txt # Remove development packages. RUN apt remove -y build-essential git diff --git a/test/data/script_samples/osage.csv b/test/data/script_samples/osage.csv index 8a84f0c..3c16e2c 100644 --- a/test/data/script_samples/osage.csv +++ b/test/data/script_samples/osage.csv @@ -1,5 +1,5 @@ -osage, 𐒰 𐒱 𐒲 𐒳 𐒴 𐒵 𐒶 𐒷 𐒸 𐒹 𐒺 𐒻 𐒼 𐒽 𐒾 𐒿 M 𐓁 𐓂 𐓃 𐓄 𐓅 𐓆 𐓇 𐓈 𐓉 𐓊 𐓋 𐓌 𐓍 𐓎 𐓏 𐓐 𐓑 𐓓 𐓒,A Ai Aį Ə Br Č Hč E Eį H Hy I K Hk Ky L M N O Oį P Hp S Š T Ht C Hc Tš Ð U W X Ɣ Ž Z, +osage,𐒰 𐒱 𐒲 𐒳 𐒴 𐒵 𐒶 𐒷 𐒸 𐒹 𐒺 𐒻 𐒼 𐒽 𐒾 𐒿 M 𐓁 𐓂 𐓃 𐓄 𐓅 𐓆 𐓇 𐓈 𐓉 𐓊 𐓋 𐓌 𐓍 𐓎 𐓏 𐓐 𐓑 𐓓 𐓒,A Ai Aį Ə Br Č Hč E Eį H Hy I K Hk Ky L M N O Oį P Hp S Š T Ht C Hc Tš Ð U W X Ɣ Ž Z, osage,𐓘 𐓙 𐓚 𐓛 𐓜 𐓝 𐓞 𐓟 𐓠 𐓡 𐓢 𐓣 𐓥 𐓦 𐓧 𐓨 𐓩 𐓪 𐓫 𐓬 𐓭 𐓮 𐓯 𐓰 𐓱 𐓲 𐓳 𐓴 𐓵 𐓶 𐓷 𐓸 𐓹 𐓻 𐓺,a ai aį ə br č hč e eį h hy i hk ky l m n o oį p hp s š t ht c hc tš ð u w x ɣ ž z, osage,𐒲 𐒲 𐒲 𐓚 𐓚 𐓚 / 𐒳 𐒳 𐓛 𐓛 / 𐒸 𐒸 𐒸 𐓠 𐓠 𐓠 / 𐓃 𐓃 𐓃 𐓫 𐓫 𐓫 / 𐓍 𐓍 𐓵 𐓵 / 𐓑 𐓑 𐓹 𐓹,Ain Aĩ Aį ain aĩ aį / Ə E̳ ə e̳ / Ein Eĩ Eį ein eĩ eį / Oin Oĩ Oį oin oĩ oį / Ð D̳ ð d̳ / Ɣ G̳ ɣ g̳ ,r2s -osage,𐓘 𐓘̄ 𐓘́ 𐓘́͘ 𐓘̋ 𐓘͘ 𐓙 𐓙̄ 𐓙́ 𐓚́ 𐓙̋ 𐓚 𐓛 𐓛̄ 𐓛́ 𐓛́͘ 𐓛̋ 𐓛͘ 𐓟 𐓟̄ 𐓟́ 𐓟ʼ 𐓟̋ 𐓟͘ 𐓟𐓣 𐓟𐓣̄ 𐓟𐓣́ 𐓠́ 𐓟𐓣̋ 𐓠 𐓣 𐓣̄ 𐓣́ 𐓣́͘ 𐓣̋ 𐓣͘ 𐓪 𐓪̄ 𐓪́ 𐓪́͘ 𐓪̋ 𐓪͘ 𐓪𐓣 𐓪𐓣̄ 𐓪𐓣́ 𐓫́ 𐓪̋𐓣 ͘𐓪𐓣 𐓶 𐓶̄ 𐓶́ 𐓶́͘ 𐓶̋ 𐓶͘,a ā á ą́ a̋ ą ai aī aí aį́ ai̋ aį ə ə̄ ə́ ə̨́ ə̋ ə̨ e ē é eʼ e̋ ę ei eī eí eį́ ei̋ eį i ī í į́ i̋ į o ō ó ǫ́ ő ǫ oi oī oí oį́ ői ̨oi u ū ú ų́ ű ų, +osage,𐓘 𐓘̄ 𐓘́ 𐓘́͘ 𐓘̋ 𐓘͘ 𐓙 𐓙̄ 𐓙́ 𐓚́ 𐓙̋ 𐓚 𐓛 𐓛̄ 𐓛́ 𐓛́͘ 𐓛̋ 𐓛͘ 𐓟 𐓟̄ 𐓟́ 𐓟ʼ 𐓟̋ 𐓟͘ 𐓟𐓣 𐓟𐓣̄ 𐓟𐓣́ 𐓠́ 𐓟𐓣̋ 𐓠 𐓣 𐓣̄ 𐓣́ 𐓣́͘ 𐓣̋ 𐓣͘ 𐓪 𐓪̄ 𐓪́ 𐓪́͘ 𐓪̋ 𐓪͘ 𐓪𐓣 𐓪𐓣̄ 𐓪𐓣́ 𐓫́ 𐓪̋𐓣 ͘𐓪𐓣 𐓶 𐓶̄ 𐓶́ 𐓶́͘ 𐓶̋ 𐓶͘,a ā á ą́ a̋ ą ai aī aí aį́ ai̋ aį ə ə̄ ə́ ə̨́ ə̋ ə̨ e ē é eʼ e̋ ę ei eī eí eį́ ei̋ eį i ī í į́ i̋ į o ō ó ǫ́ ő ǫ oi oī oí oį́ ői ̨oi u ū ú ų́ ű ų, osage,"𐓏𐓘𐓩𐓘͘𐓮𐓰𐓘𐓬𐓟 𐓘𐓡𐓘 𐓵𐓟, 𐓷𐓟𐓲’𐓘 𐓘𐓬𐓘 𐓵𐓘𐓹𐓰𐓘𐓤𐓘𐓬𐓟 𐓮𐓣𐓵𐓟𐓲𐓟. 𐓍𐓘𐓹𐓰𐓘𐓤𐓘𐓬𐓟 𐓘𐓡𐓘, 𐒻𐓲𐓣𐓤𐓫 𐓰𐓘͘𐓤𐓘 𐓘𐓬𐓘 𐓘𐓵𐓘𐓬𐓟 𐓰𐓘͘.","Wanąstape aha ðe, wec’a apa ðaɣtakape siðece. Ðaɣtakape aha, Icikoį tąka apa aðape tą.",