Files
John c1a6a105ef Add memory coach, transformer recall model, and training FAB.
Introduce Q/K/V memory dialogue with coach APIs, a lightweight NumPy
transformer for per-word forgetting prediction, and a floating training
menu linking daily quiz, spell, and coach flows.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-04 18:11:49 -07:00

167 lines
5.5 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""轻量 NumPy Transformer:单条序列 → 回忆成功概率 logit。"""
from __future__ import annotations
import json
import math
from pathlib import Path
from typing import Any
import numpy as np
from services.memory_transformer.encoding import FEATURE_DIM, MAX_SEQ_LEN
def _gelu(x: np.ndarray) -> np.ndarray:
return 0.5 * x * (1.0 + np.tanh(math.sqrt(2.0 / math.pi) * (x + 0.044715 * x**3)))
def _softmax(x: np.ndarray, axis: int = -1) -> np.ndarray:
e = np.exp(x - np.max(x, axis=axis, keepdims=True))
return e / np.sum(e, axis=axis, keepdims=True)
def _layer_norm(x: np.ndarray, gamma: np.ndarray, beta: np.ndarray) -> np.ndarray:
mean = x.mean(axis=-1, keepdims=True)
var = x.var(axis=-1, keepdims=True) + 1e-6
return gamma * (x - mean) / np.sqrt(var) + beta
class MiniTransformer:
"""
结构:Linear(F→D) + 2×(MHA + FFN) + CLS 读出。
仅推理;训练在 train_memory_transformer.py 中用 PyTorch 导出权重。
"""
def __init__(
self,
d_model: int = 48,
n_heads: int = 2,
d_ff: int = 96,
n_layers: int = 2,
):
self.d_model = d_model
self.n_heads = n_heads
self.d_k = d_model // n_heads
self.d_ff = d_ff
self.n_layers = n_layers
self.weights: dict[str, np.ndarray] = {}
def load_numpy_dict(self, state: dict[str, Any]) -> None:
self.d_model = int(state["d_model"])
self.n_heads = int(state["n_heads"])
self.d_ff = int(state["d_ff"])
self.n_layers = int(state["n_layers"])
self.d_k = self.d_model // self.n_heads
self.weights = {k: np.array(v, dtype=np.float64) for k, v in state["weights"].items()}
def save_json(self, path: Path) -> None:
payload = {
"d_model": self.d_model,
"n_heads": self.n_heads,
"d_ff": self.d_ff,
"n_layers": self.n_layers,
"weights": {k: v.tolist() for k, v in self.weights.items()},
}
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload), encoding="utf-8")
@classmethod
def load_json(cls, path: Path) -> MiniTransformer:
state = json.loads(path.read_text(encoding="utf-8"))
m = cls()
m.load_numpy_dict(state)
return m
def _mha(self, x: np.ndarray, li: int) -> np.ndarray:
w = self.weights
Wq, Wk, Wv = w[f"L{li}.Wq"], w[f"L{li}.Wk"], w[f"L{li}.Wv"]
Wo = w[f"L{li}.Wo"]
Bq, Bk, Bv = w[f"L{li}.Bq"], w[f"L{li}.Bk"], w[f"L{li}.Bv"]
seq, d = x.shape
Q = x @ Wq + Bq
K = x @ Wk + Bk
V = x @ Wv + Bv
heads = []
for h in range(self.n_heads):
sl = slice(h * self.d_k, (h + 1) * self.d_k)
q, k, v = Q[:, sl], K[:, sl], V[:, sl]
scores = (q @ k.T) / math.sqrt(self.d_k)
attn = _softmax(scores, axis=-1)
heads.append(attn @ v)
concat = np.concatenate(heads, axis=-1)
return concat @ Wo + w[f"L{li}.Bo"]
def _ffn(self, x: np.ndarray, li: int) -> np.ndarray:
w = self.weights
h = _gelu(x @ w[f"L{li}.W1"] + w[f"L{li}.b1"])
return h @ w[f"L{li}.W2"] + w[f"L{li}.b2"]
def forward_logits(self, seq_features: list[list[float]], valid_len: int) -> float:
x = np.array(seq_features[:MAX_SEQ_LEN], dtype=np.float64)
mask = np.zeros(MAX_SEQ_LEN, dtype=np.float64)
mask[:valid_len] = 1.0
w = self.weights
x = x @ w["in_proj"] + w["in_bias"]
x = _layer_norm(x, w["ln_in_g"], w["ln_in_b"])
for li in range(self.n_layers):
attn_out = self._mha(x, li)
x = _layer_norm(x + attn_out, w[f"L{li}.ln1_g"], w[f"L{li}.ln1_b"])
ff = self._ffn(x, li)
x = _layer_norm(x + ff, w[f"L{li}.ln2_g"], w[f"L{li}.ln2_b"])
cls = x[0]
return float(cls @ w["head_w"] + w["head_b"])
def predict_proba(self, seq_features: list[list[float]], valid_len: int) -> float:
logit = self.forward_logits(seq_features, valid_len)
return float(1.0 / (1.0 + np.exp(-logit)))
def init_random_weights(
d_model: int = 48,
n_heads: int = 2,
d_ff: int = 96,
n_layers: int = 2,
seed: int = 42,
) -> dict[str, np.ndarray]:
rng = np.random.default_rng(seed)
d_k = d_model // n_heads
w: dict[str, np.ndarray] = {}
def glorot(shape):
fan_in, fan_out = shape[0], shape[1] if len(shape) > 1 else shape[0]
limit = math.sqrt(6.0 / (fan_in + fan_out))
return rng.uniform(-limit, limit, shape)
w["in_proj"] = glorot((FEATURE_DIM, d_model))
w["in_bias"] = np.zeros(d_model)
w["ln_in_g"] = np.ones(d_model)
w["ln_in_b"] = np.zeros(d_model)
for li in range(n_layers):
w[f"L{li}.Wq"] = glorot((d_model, d_model))
w[f"L{li}.Wk"] = glorot((d_model, d_model))
w[f"L{li}.Wv"] = glorot((d_model, d_model))
w[f"L{li}.Wo"] = glorot((d_model, d_model))
w[f"L{li}.Bq"] = np.zeros(d_model)
w[f"L{li}.Bk"] = np.zeros(d_model)
w[f"L{li}.Bv"] = np.zeros(d_model)
w[f"L{li}.Bo"] = np.zeros(d_model)
w[f"L{li}.ln1_g"] = np.ones(d_model)
w[f"L{li}.ln1_b"] = np.zeros(d_model)
w[f"L{li}.W1"] = glorot((d_model, d_ff))
w[f"L{li}.b1"] = np.zeros(d_ff)
w[f"L{li}.W2"] = glorot((d_ff, d_model))
w[f"L{li}.b2"] = np.zeros(d_model)
w[f"L{li}.ln2_g"] = np.ones(d_model)
w[f"L{li}.ln2_b"] = np.zeros(d_model)
w["head_w"] = glorot((d_model,)) * 0.1
w["head_b"] = np.array(0.0)
return w