Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>main
parent
56a2edb970
commit
52a441a96f
@ -0,0 +1,10 @@
|
|||||||
|
# Listas de palabras
|
||||||
|
|
||||||
|
El script que hizo el borrador de la lista española de `datekeys-go/wordkey/lists/es.txt` (7-10). Lee, en su carpeta de trabajo:
|
||||||
|
|
||||||
|
- `es_50k.txt`, de [FrequencyWords](https://github.com/hermitdave/FrequencyWords), `content/2018/es/es_50k.txt` (CC BY-SA 4.0);
|
||||||
|
- `es_ES.dic` y `es_ES.aff`, del repo de diccionarios de LibreOffice, `es/` (GPL, LGPL o MPL; solo como filtro).
|
||||||
|
|
||||||
|
Escribe `es_7776.txt`, la lista ordenada; `es_7776_rank.txt`, con la posición y la frecuencia de cada palabra; y `descartes.txt`, lo que quitó cada filtro. Los ficheros de entrada y de salida del 7-10 están en `G:\tmp\wordlist-es`.
|
||||||
|
|
||||||
|
Para otro idioma o para el corpus de Leipzig, se cambian la fuente de frecuencias, el diccionario, la regla de la marca `G` (los femeninos, propia del diccionario español) y la lista de palabras ofensivas.
|
||||||
@ -0,0 +1,129 @@
|
|||||||
|
import re
|
||||||
|
import unicodedata
|
||||||
|
from collections import Counter
|
||||||
|
|
||||||
|
N = 7776
|
||||||
|
LETTERS = re.compile(r'^[a-záéíóúüñ]+$')
|
||||||
|
# Clitic pronouns glued to a verb: "démela", "dennos", "dígame".
|
||||||
|
CLITIC = re.compile(r'(me|te|se|nos|os|lo|la|le|los|las|les)$')
|
||||||
|
|
||||||
|
KEEP = {'miércoles'}
|
||||||
|
|
||||||
|
# A small seed blocklist; the author reviews the final list by hand anyway.
|
||||||
|
BLOCK = set('''
|
||||||
|
puta puto putas mierda joder coño cojones polla follar culo culos pene
|
||||||
|
vagina teta tetas zorra cabrón cabron maricón maricon marica gilipollas
|
||||||
|
imbécil imbecil idiota estúpido estupido estúpida subnormal retrasado
|
||||||
|
mamada chupar chupa pajero paja orgasmo semen sexo sexual violar violación
|
||||||
|
violador nazi negro negrata gitano moro sudaca puticlub prostituta
|
||||||
|
mear cagar caca pedo pis pito chocho concha verga pija pinche pendejo
|
||||||
|
chingar chingada cabrona bastardo perra cerdo cerda asesinar asesino
|
||||||
|
suicidio suicida matar muerte muerto cadáver cáncer droga drogas cocaína
|
||||||
|
heroína porno pornografía ramera furcia hostia hostias
|
||||||
|
manolo pal
|
||||||
|
'''.split())
|
||||||
|
|
||||||
|
|
||||||
|
def norm(w):
|
||||||
|
# §38.1: NFD, drop U+0300..U+036F, lowercase.
|
||||||
|
d = unicodedata.normalize('NFD', w)
|
||||||
|
d = ''.join(c for c in d if not (0x300 <= ord(c) <= 0x36F))
|
||||||
|
return d.lower()
|
||||||
|
|
||||||
|
|
||||||
|
# Rules of the suffix G (masculine to feminine) from the .aff.
|
||||||
|
grules = []
|
||||||
|
for line in open('es_ES.aff', encoding='utf-8'):
|
||||||
|
p = line.split()
|
||||||
|
if len(p) >= 5 and p[0] == 'SFX' and p[1] == 'G':
|
||||||
|
strip = '' if p[2] == '0' else p[2]
|
||||||
|
add = p[3].split('/')[0]
|
||||||
|
add = '' if add == '0' else add
|
||||||
|
grules.append((strip, add, re.compile(p[4] + '$')))
|
||||||
|
|
||||||
|
stems = set()
|
||||||
|
group = {} # word -> its masculine stem, so a gender pair keeps one word
|
||||||
|
unflagged = set()
|
||||||
|
for i, line in enumerate(open('es_ES.dic', encoding='utf-8')):
|
||||||
|
if i == 0:
|
||||||
|
continue
|
||||||
|
entry = line.strip().split()[0] if line.strip() else ''
|
||||||
|
if not entry:
|
||||||
|
continue
|
||||||
|
w, _, flags = entry.partition('/')
|
||||||
|
stems.add(w)
|
||||||
|
group.setdefault(w, w)
|
||||||
|
if not flags:
|
||||||
|
unflagged.add(w)
|
||||||
|
if 'G' in flags:
|
||||||
|
for strip, add, cond in grules:
|
||||||
|
if cond.search(w) and w.endswith(strip):
|
||||||
|
f = w[:len(w) - len(strip)] + add
|
||||||
|
stems.add(f)
|
||||||
|
group.setdefault(f, w)
|
||||||
|
break
|
||||||
|
|
||||||
|
freq = []
|
||||||
|
for line in open('es_50k.txt', encoding='utf-8'):
|
||||||
|
w, c = line.split()
|
||||||
|
freq.append((w, int(c)))
|
||||||
|
|
||||||
|
drop = {k: [] for k in ('not_letters', 'length', 'not_stem', 'clitic', 'blocked', 'gender_pair', 'collision')}
|
||||||
|
seen = {}
|
||||||
|
pairs = {}
|
||||||
|
kept = []
|
||||||
|
last = None
|
||||||
|
for w, c in freq:
|
||||||
|
if not LETTERS.match(w):
|
||||||
|
drop['not_letters'].append(w)
|
||||||
|
continue
|
||||||
|
if not 3 <= len(w) <= 9:
|
||||||
|
drop['length'].append(w)
|
||||||
|
continue
|
||||||
|
if w not in stems:
|
||||||
|
drop['not_stem'].append(w)
|
||||||
|
continue
|
||||||
|
# An entry without flags that ends in a clitic is a verb form, not a word to learn.
|
||||||
|
# Entries without flags are conjugations, plurals and pieces of names.
|
||||||
|
if w in unflagged and w not in KEEP:
|
||||||
|
drop['clitic'].append(w)
|
||||||
|
continue
|
||||||
|
if w in BLOCK:
|
||||||
|
drop['blocked'].append(w)
|
||||||
|
continue
|
||||||
|
g = group[w]
|
||||||
|
if norm(g)[-1] in 'oa':
|
||||||
|
g = norm(g)[:-1] + '·'
|
||||||
|
if g in pairs:
|
||||||
|
drop['gender_pair'].append(f'{w} (pareja de {pairs[g]})')
|
||||||
|
continue
|
||||||
|
k = norm(w)
|
||||||
|
if k in seen:
|
||||||
|
drop['collision'].append(f'{w} (choca con {seen[k]})')
|
||||||
|
continue
|
||||||
|
seen[k] = w
|
||||||
|
pairs[g] = w
|
||||||
|
kept.append((w, c))
|
||||||
|
if len(kept) == N:
|
||||||
|
last = (w, c)
|
||||||
|
break
|
||||||
|
|
||||||
|
print('kept', len(kept), 'last', last)
|
||||||
|
if len(kept) < N:
|
||||||
|
raise SystemExit('not enough words')
|
||||||
|
words = [w for w, _ in kept]
|
||||||
|
open('es_7776_rank.txt', 'w', encoding='utf-8', newline='\n').write(
|
||||||
|
''.join(f'{i + 1}\t{w}\t{c}\n' for i, (w, c) in enumerate(kept)))
|
||||||
|
open('es_7776.txt', 'w', encoding='utf-8', newline='\n').write(''.join(w + '\n' for w in sorted(words)))
|
||||||
|
with open('descartes.txt', 'w', encoding='utf-8', newline='\n') as f:
|
||||||
|
for k, v in drop.items():
|
||||||
|
f.write(f'== {k}: {len(v)}\n')
|
||||||
|
f.write('\n'.join(v) + '\n\n')
|
||||||
|
for k, v in drop.items():
|
||||||
|
print(k, len(v), v[:20])
|
||||||
|
lens = Counter(len(w) for w in words)
|
||||||
|
print('lengths', sorted(lens.items()))
|
||||||
|
print('with ñ', sum('ñ' in w for w in words), 'with accent', sum(norm(w) != w for w in words))
|
||||||
|
u = [w for w in words if w in unflagged]
|
||||||
|
print('unflagged kept', len(u), u[:60])
|
||||||
|
print('tail', words[-30:])
|
||||||
Loading…
Reference in new issue