You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

130 lines
4.3 KiB

import re
import unicodedata
from collections import Counter
N = 7776
LETTERS = re.compile(r'^[a-záéíóúüñ]+$')
# Clitic pronouns glued to a verb: "démela", "dennos", "dígame".
CLITIC = re.compile(r'(me|te|se|nos|os|lo|la|le|los|las|les)$')
KEEP = {'miércoles'}
# A small seed blocklist; the author reviews the final list by hand anyway.
BLOCK = set('''
puta puto putas mierda joder coño cojones polla follar culo culos pene
vagina teta tetas zorra cabrón cabron maricón maricon marica gilipollas
imbécil imbecil idiota estúpido estupido estúpida subnormal retrasado
mamada chupar chupa pajero paja orgasmo semen sexo sexual violar violación
violador nazi negro negrata gitano moro sudaca puticlub prostituta
mear cagar caca pedo pis pito chocho concha verga pija pinche pendejo
chingar chingada cabrona bastardo perra cerdo cerda asesinar asesino
suicidio suicida matar muerte muerto cadáver cáncer droga drogas cocaína
heroína porno pornografía ramera furcia hostia hostias
manolo pal
'''.split())
def norm(w):
# §38.1: NFD, drop U+0300..U+036F, lowercase.
d = unicodedata.normalize('NFD', w)
d = ''.join(c for c in d if not (0x300 <= ord(c) <= 0x36F))
return d.lower()
# Rules of the suffix G (masculine to feminine) from the .aff.
grules = []
for line in open('es_ES.aff', encoding='utf-8'):
p = line.split()
if len(p) >= 5 and p[0] == 'SFX' and p[1] == 'G':
strip = '' if p[2] == '0' else p[2]
add = p[3].split('/')[0]
add = '' if add == '0' else add
grules.append((strip, add, re.compile(p[4] + '$')))
stems = set()
group = {} # word -> its masculine stem, so a gender pair keeps one word
unflagged = set()
for i, line in enumerate(open('es_ES.dic', encoding='utf-8')):
if i == 0:
continue
entry = line.strip().split()[0] if line.strip() else ''
if not entry:
continue
w, _, flags = entry.partition('/')
stems.add(w)
group.setdefault(w, w)
if not flags:
unflagged.add(w)
if 'G' in flags:
for strip, add, cond in grules:
if cond.search(w) and w.endswith(strip):
f = w[:len(w) - len(strip)] + add
stems.add(f)
group.setdefault(f, w)
break
freq = []
for line in open('es_50k.txt', encoding='utf-8'):
w, c = line.split()
freq.append((w, int(c)))
drop = {k: [] for k in ('not_letters', 'length', 'not_stem', 'clitic', 'blocked', 'gender_pair', 'collision')}
seen = {}
pairs = {}
kept = []
last = None
for w, c in freq:
if not LETTERS.match(w):
drop['not_letters'].append(w)
continue
if not 3 <= len(w) <= 9:
drop['length'].append(w)
continue
if w not in stems:
drop['not_stem'].append(w)
continue
# An entry without flags that ends in a clitic is a verb form, not a word to learn.
# Entries without flags are conjugations, plurals and pieces of names.
if w in unflagged and w not in KEEP:
drop['clitic'].append(w)
continue
if w in BLOCK:
drop['blocked'].append(w)
continue
g = group[w]
if norm(g)[-1] in 'oa':
g = norm(g)[:-1] + '·'
if g in pairs:
drop['gender_pair'].append(f'{w} (pareja de {pairs[g]})')
continue
k = norm(w)
if k in seen:
drop['collision'].append(f'{w} (choca con {seen[k]})')
continue
seen[k] = w
pairs[g] = w
kept.append((w, c))
if len(kept) == N:
last = (w, c)
break
print('kept', len(kept), 'last', last)
if len(kept) < N:
raise SystemExit('not enough words')
words = [w for w, _ in kept]
open('es_7776_rank.txt', 'w', encoding='utf-8', newline='\n').write(
''.join(f'{i + 1}\t{w}\t{c}\n' for i, (w, c) in enumerate(kept)))
open('es_7776.txt', 'w', encoding='utf-8', newline='\n').write(''.join(w + '\n' for w in sorted(words)))
with open('descartes.txt', 'w', encoding='utf-8', newline='\n') as f:
for k, v in drop.items():
f.write(f'== {k}: {len(v)}\n')
f.write('\n'.join(v) + '\n\n')
for k, v in drop.items():
print(k, len(v), v[:20])
lens = Counter(len(w) for w in words)
print('lengths', sorted(lens.items()))
print('with ñ', sum('ñ' in w for w in words), 'with accent', sum(norm(w) != w for w in words))
u = [w for w in words if w in unflagged]
print('unflagged kept', len(u), u[:60])
print('tail', words[-30:])

Powered by TurnKey Linux.