import torchfrom huggingface_hub import hf_hub_downloadfrom tokenizers import Tokenizer # Descargar modelo y tokenizermodel_path = hf_hub_download("raulgdp/gpt-acredita-350m", "gpt_acredita.pt")tokenizer_path = hf_hub_download("raulgdp/gpt-acredita-350m", "tokenizer.json") # Cargar tokenizertokenizer = Tokenizer.from_file(tokenizer_path) # Cargar modelocheckpoint = torch.load(model_path, map_location="cpu", weights_only=False)config = checkpoint["config"] # --- Definir la arquitectura GPT-Acredita ---import torch.nn as nn class CausalSelfAttention(nn.Module): def __init__(self, config): super().__init__() self.n_head = config["n_head"] self.n_embd = config["n_embd"] self.c_attn = nn.Linear(config["n_embd"], 3 * config["n_embd"]) self.c_proj = nn.Linear(config["n_embd"], config["n_embd"]) self.register_buffer("bias", torch.tril( torch.ones(config["block_size"], config["block_size"]) ).view(1, 1, config["block_size"], config["block_size"])) def forward(self, x): B, T, C = x.size() q, k, v = self.c_attn(x).split(self.n_embd, dim=2) k = k.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) q = q.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) v = v.view(B, T, self.n_head, C // self.n_head).transpose(1, 2) att = (q @ k.transpose(-2, -1)) * (1.0 / (k.size(-1) ** 0.5)) att = att.masked_fill(self.bias[:, :, :T, :T] == 0, float("-inf")) att = torch.softmax(att, dim=-1) return (att @ v).transpose(1, 2).contiguous().view(B, T, C) class MLP(nn.Module): def __init__(self, config): super().__init__() self.c_fc = nn.Linear(config["n_embd"], 4 * config["n_embd"]) self.c_proj = nn.Linear(4 * config["n_embd"], config["n_embd"]) self.act = nn.GELU() def forward(self, x): return self.c_proj(self.act(self.c_fc(x))) class Block(nn.Module): def __init__(self, config): super().__init__() self.ln_1 = nn.LayerNorm(config["n_embd"]) self.attn = CausalSelfAttention(config) self.ln_2 = nn.LayerNorm(config["n_embd"]) self.mlp = MLP(config) def forward(self, x): x = x + self.attn(self.ln_1(x)) x = x + self.mlp(self.ln_2(x)) return x class GPTAcredita(nn.Module): def __init__(self, config): super().__init__() self.transformer = nn.ModuleDict({ "wte": nn.Embedding(config["vocab_size"], config["n_embd"]), "wpe": nn.Embedding(config["block_size"], config["n_embd"]), "h": nn.ModuleList([Block(config) for _ in range(config["n_layer"])]), "ln_f": nn.LayerNorm(config["n_embd"]), }) self.lm_head = nn.Linear(config["n_embd"], config["vocab_size"], bias=False) def forward(self, idx): B, T = idx.size() pos = torch.arange(T, device=idx.device) x = self.transformer.wte(idx) + self.transformer.wpe(pos) for block in self.transformer.h: x = block(x) return self.lm_head(self.transformer.ln_f(x)) @torch.no_grad() def generate(self, idx, max_new_tokens=200, temperature=0.7, top_k=50): for _ in range(max_new_tokens): idx_cond = idx[:, -config["block_size"]:] logits = self(idx_cond)[:, -1, :] logits = logits / temperature v, _ = torch.topk(logits, min(top_k, logits.size(-1))) logits[logits < v[:, [-1]]] = float("-inf") idx = torch.cat([idx, torch.multinomial( torch.softmax(logits, dim=-1), num_samples=1 )], dim=1) # Detener en <|eos|> if tokenizer.token_to_id("<|eos|>") in idx[0, -3:].tolist(): break return idx # Cargar pesosdevice = "cuda" if torch.cuda.is_available() else "cpu"model = GPTAcredita(config).to(device)model.load_state_dict(checkpoint["model_state_dict"])model.eval() # Hacer una preguntadef preguntar(pregunta: str, max_tokens: int = 200) -> str: prompt = f"### Pregunta: {pregunta}\n### Respuesta:" encoded = tokenizer.encode(prompt) ids = torch.tensor([encoded.ids], dtype=torch.long, device=device) out = model.generate(ids, max_new_tokens=max_tokens, temperature=0.7) decoded = tokenizer.decode(out[0].tolist()) if "### Respuesta:" in decoded: resp = decoded.split("### Respuesta:")[-1] return resp.split("<|eos|>")[0].strip() return decoded # Ejemploprint(preguntar("¿Cuáles son los 10 factores del CNA?"))print(preguntar("¿Qué establece el Decreto 1330 de 2019?"))