# --- corpus controle : 4 themes x 4 mots + 3 mots partages (structure connue, jamais utilisee pour apprendre) ---
ksel_rng = np.random.default_rng(42)
ksel_vocab = [
"football", "equipe", "match", "competition",
"ordinateur", "logiciel", "internet", "application",
"recette", "ingredient", "cuisson", "gastronomie",
"hotel", "avion", "destination", "tourisme",
"recherche", "performance", "preparation", # 16-18 partages
]
ksel_V = len(ksel_vocab)
ksel_phi_gen = np.array([
[.24, .20, .18, .16, .01, .01, .01, .01, .01, .01, .01, .01, .01, .01, .01, .01, .05, .04, .03],
[.01, .01, .01, .01, .24, .20, .18, .16, .01, .01, .01, .01, .01, .01, .01, .01, .03, .05, .04],
[.01, .01, .01, .01, .01, .01, .01, .01, .24, .20, .18, .16, .01, .01, .01, .01, .04, .03, .06],
[.01, .01, .01, .01, .01, .01, .01, .01, .01, .01, .01, .01, .24, .20, .18, .16, .03, .04, .05],
])
ksel_phi_gen = ksel_phi_gen / ksel_phi_gen.sum(axis=1, keepdims=True)
ksel_thetas = []
for t in range(4):
for _ in range(3):
v = np.ones(4) * 0.1; v[t] = 0.7
ksel_thetas.append(v)
ksel_thetas += [np.array([.4, .4, .1, .1]), np.array([.1, .4, .4, .1]),
np.array([.4, .1, .4, .1]), np.array([.25, .25, .25, .25])]
ksel_thetas = np.array(ksel_thetas) # 16 train
ksel_held_thetas = np.array([[.7, .1, .1, .1], [.1, .7, .1, .1], [.1, .1, .7, .1], [.4, .1, .1, .4]])
ksel_L = 24
def ksel_gen(theta):
zs = ksel_rng.choice(4, size=ksel_L, p=theta)
return np.array([ksel_rng.choice(ksel_V, p=ksel_phi_gen[z]) for z in zs])
ksel_docs = [ksel_gen(t) for t in ksel_thetas]
ksel_held = [ksel_gen(t) for t in ksel_held_thetas]
ksel_bow = np.array([np.bincount(d, minlength=ksel_V) for d in ksel_docs])
print(f"train docs: {len(ksel_docs)}, held: {len(ksel_held)}, V={ksel_V}, L={ksel_L}")