# SPDX-License-Identifier: GPL-2.0-or-later
# Sade adaptation, 2026-10-08: convert FreeDict TEI to compact bilingual rows.
# Dataset remains GPL-2.0-or-later. Preserve full original quotes alongside
# extracted primary equivalents; do not invent parts of speech or subject tags.
import hashlib
import json
import re
import tarfile
import unicodedata
import xml.etree.ElementTree as ET
from pathlib import Path

ROOT = Path(__file__).resolve().parents[1]
DEST = ROOT / 'public' / 'data'
ARCHIVE = DEST / 'freedict-eng-tur-0.3.src.tar.xz'
NS = {'t': 'http://www.tei-c.org/ns/1.0'}

def clean(text):
    return unicodedata.normalize('NFC', re.sub(r'\s+', ' ', text).strip())

def primary_equivalent(quote):
    # Parenthesized usage tags are context, not part of the lookup term.
    term = quote
    while re.match(r'^\([^)]{1,35}\)\s*', term):
        term = re.sub(r'^\([^)]{1,35}\)\s*', '', term, count=1)
    # Some source quotes append example phrases after a full stop. Keep them
    # in the original detail field, and index the primary equivalent only.
    depth = 0
    for i, char in enumerate(term):
        if char == '(':
            depth += 1
        elif char == ')':
            depth = max(0, depth - 1)
        elif char == '.' and depth == 0 and term[i:i + 2] == '. ':
            term = term[:i]
            break
    return term.strip(' .;,:')

with tarfile.open(ARCHIVE, 'r:xz') as archive:
    tree = ET.parse(archive.extractfile('eng-tur/eng-tur.tei'))

rows, seen = [], set()
source_entries = tree.findall('.//t:body/t:entry', NS)
for source_index, entry in enumerate(source_entries):
    english = clean(entry.findtext('t:form/t:orth', default='', namespaces=NS))
    if not english:
        continue
    for quote in entry.findall('.//t:cit[@type="trans"]/t:quote', NS):
        original = clean(''.join(quote.itertext()))
        turkish = primary_equivalent(original)
        if not turkish or not any(c.isalpha() for c in turkish):
            continue
        key = (english.casefold(), turkish.casefold())
        if key in seen:
            continue
        seen.add(key)
        # Store original wording only when extraction changed it, to keep the
        # payload small. Source-entry ordinal identifies the corresponding TEI.
        rows.append([english, turkish, original if original != turkish else '', source_index + 1])

payload = json.dumps(rows, ensure_ascii=False, separators=(',', ':'))
(DEST / 'freedict-data.js').write_text(
    '// SPDX-License-Identifier: GPL-2.0-or-later\n'
    '// Adapted from FreeDict eng-tur 0.3. Copyright (C) 1999-2017 various authors.\n'
    '// Original author Mehmet Ali Vardar; conversion Michael Bunk; FreeDict contributors.\n'
    '// Adaptation 2026-10-08: normalization, primary-equivalent extraction, deduplication.\n'
    '// Full corresponding source and GPL license are served alongside this file.\n'
    'export const freedictRows = ' + payload + ';\n', encoding='utf-8')
stats = {
    'name': 'FreeDict English–Turkish', 'version': '0.3',
    'sourceEntries': len(source_entries), 'pairs': len(rows),
    'englishTerms': len({r[0].casefold() for r in rows}),
    'turkishTerms': len({r[1].casefold() for r in rows}),
    'license': 'GPL-2.0-or-later',
    'archiveSHA256': hashlib.sha256(ARCHIVE.read_bytes()).hexdigest(),
    'upstream': 'https://download.freedict.org/dictionaries/eng-tur/0.3/freedict-eng-tur-0.3.src.tar.xz',
    'adaptedOn': '2026-10-08',
}
(DEST / 'freedict-metadata.js').write_text('export const freedictMetadata = ' + json.dumps(stats, ensure_ascii=False, indent=2) + ';\n', encoding='utf-8')
print(json.dumps(stats, ensure_ascii=False, indent=2))
