JARVIS/nucleo/saber/awesome.py
sito 43335d329f saber: catalogo de awesome lists en el RAG
- awesome.py parsea awesome lists (markdown y RST): limpia enlaces y badges,
  agrupa cada categoria en un fragmento, y etiqueta fuente + licencia.
- indexa.py las incluye; en el indice PUBLICO solo entran las redistribuibles:
  awesome (sindresorhus, CC0) y awesome-selfhosted (CC BY-SA 3.0). awesome-hacking
  (sin licencia) se indexa solo en el indice personal si se clona en CODERS.
- saber-publico.jsonl pasa a 699 entradas (glosario + conocimiento + catalogo).
- ATRIBUCION.md acredita fuentes y licencias; docs/rag.md lo documenta.
2026-08-16 18:55:35 +02:00

168 lines
6.7 KiB
Python

#!/usr/bin/env python3
"""Indexa las 'awesome lists' clonadas en ~/COFRE/CODERS como catalogo del RAG.
./awesome.py di cuantas entradas saldrian, por fuente
./awesome.py --volcado imprime unas cuantas entradas de muestra
Las awesome lists son catalogos curados: "herramienta -> que hace + categoria".
No son comandos ni metodologia; son el "que existe para hacer X". Aqui se parsean
—markdown y reStructuredText— a entradas limpias, agrupadas por categoria, para
que busca.py las recupere igual que todo lo demas.
Cada fuente lleva su LICENCIA. Solo las redistribuibles (CC0, CC-BY-SA) entran en
el indice PUBLICO que se sube al repo; el resto se indexa solo en la maquina del
usuario (indexado personal = uso legitimo). indexa.py respeta esa marca.
"""
import argparse
import os
import re
import sys
CODERS = os.path.expanduser("~/COFRE/CODERS")
MAX_BYTES = 4 * 1024 * 1024 # una lista no pasa de esto; guarda la swap chica
MAX_FRAGMENTO = 1100 # como el resto del indice
# Secciones que no son catalogo, se saltan vengan como vengan.
RUIDO = re.compile(r"^(table of contents|contents|contributing|license|"
r"list of licenses|anti.?features|external links|related|"
r"footnotes|backers|sponsors|thanks|credits|acknowledge|"
r"see also|index|legend|notes?)\b", re.I)
# (carpeta en CODERS, fichero, formato, etiqueta de fuente, perfil, licencia, publico)
CATALOGOS = [
("awesome-selfhosted", "README.md", "md", "awesome-selfhosted",
"AUTOHOSPEDADO", "CC-BY-SA-3.0", True),
("awesome", "readme.md", "md", "awesome (sindresorhus)",
"LISTAS", "CC0-1.0", True),
("awesome-hacking", os.path.join("_pages", "index.rst"), "rst",
"awesome-hacking (jekil)", "PENTESTERS", "", False),
]
# markdown: - [Nombre](http...) - Descripcion ... `Lic` `Lenguaje`
IT_MD = re.compile(r"^\s*[-*]\s+\[([^\]]+)\]\((https?://[^)]+)\)\s*[-–—:]?\s*(.*)$")
# rst: - `Nombre <http...>`_ - Descripcion
IT_RST = re.compile(r"^\s*[-*]\s+`([^<]+?)\s*<([^>]+)>`_+\s*[-–—:]?\s*(.*)$")
# parentesis de enlaces auxiliares: ([Source Code](...), [Demo](...))
PARENS = re.compile(r"\s*\((?:\[[^\]]+\]\([^)]*\)[,\s]*)+\)")
TAGS = re.compile(r"`([^`]+)`") # `MIT` `Docker` -> MIT, Docker
def _limpia_desc(desc):
desc = PARENS.sub("", desc) # fuera ([Source Code]...) etc.
tags = TAGS.findall(desc) # licencia/lenguaje entre backticks
desc = TAGS.sub("", desc).strip(" .-–—")
if tags: # se conservan como pista (lenguaje, lic)
desc = f"{desc} [{', '.join(t.strip() for t in tags)}]"
return re.sub(r"\s+", " ", desc).strip()
def _lee(ruta):
if not os.path.exists(ruta) or os.path.getsize(ruta) > MAX_BYTES:
return None
with open(ruta, encoding="utf-8", errors="ignore") as f:
return f.read()
def _secciones_md(texto):
"""Cede (categoria, [(nombre, desc)...]) por cada seccion ## / ###."""
categoria, items = "", []
for linea in texto.splitlines():
h = re.match(r"^#{2,4}\s+(.+?)\s*$", linea)
if h:
if items:
yield categoria, items
categoria, items = re.sub(r"[#*`\[\]]", "", h.group(1)).strip(), []
continue
m = IT_MD.match(linea)
if m and m.group(2).startswith("http"): # ignora enlaces de TOC (#ancla)
desc = _limpia_desc(m.group(3))
items.append((m.group(1).strip(), desc))
if items:
yield categoria, items
def _secciones_rst(texto):
"""En RST el titulo es una linea con otra de === o --- debajo."""
lineas = texto.splitlines()
categoria, items = "", []
for i, linea in enumerate(lineas):
sig = lineas[i + 1] if i + 1 < len(lineas) else ""
if linea.strip() and re.match(r"^[=~^\"'`#*+.-]{3,}\s*$", sig) \
and len(sig.strip()) >= len(linea.strip()) - 2 \
and not linea.startswith(("-", "*", " ")):
if items:
yield categoria, items
categoria, items = linea.strip(), []
continue
m = IT_RST.match(linea)
if m:
items.append((m.group(1).strip(), _limpia_desc(m.group(3))))
if items:
yield categoria, items
def _fragmenta(fuente, perfil, licencia, publico, secciones):
"""Agrupa los items de cada categoria en fragmentos con tope de tamaño."""
for categoria, items in secciones:
if not categoria or RUIDO.match(categoria):
continue
cabecera = f"Catalogo {fuente} / {categoria}:"
buffer, n = cabecera, 0
for nombre, desc in items:
linea = f" {nombre}" + (f"{desc}." if desc else ".")
if len(buffer) + len(linea) > MAX_FRAGMENTO and n:
yield _entrada(fuente, perfil, licencia, publico, categoria, buffer)
buffer, n = cabecera, 0
buffer += linea
n += 1
if n:
yield _entrada(fuente, perfil, licencia, publico, categoria, buffer)
def _entrada(fuente, perfil, licencia, publico, categoria, texto):
return {
"tipo": "catalogo", "herramienta": categoria, "perfil": perfil,
"fuente": fuente, "licencia": licencia, "publico": publico,
"tema": categoria.lower(),
"fichero": f"awesome: {fuente} / {categoria}",
"texto": texto,
}
def entradas(solo_publico=False):
"""Todas las entradas de catalogo de las awesome lists disponibles."""
for carpeta, fichero, formato, fuente, perfil, lic, publico in CATALOGOS:
if solo_publico and not publico:
continue
texto = _lee(os.path.join(CODERS, carpeta, fichero))
if not texto:
continue
secciones = _secciones_rst(texto) if formato == "rst" else _secciones_md(texto)
yield from _fragmenta(fuente, perfil, lic, publico, secciones)
def main():
p = argparse.ArgumentParser()
p.add_argument("--volcado", action="store_true", help="imprime entradas de muestra")
a = p.parse_args()
porfuente = {}
muestras = []
for e in entradas():
porfuente[e["fuente"]] = porfuente.get(e["fuente"], 0) + 1
if len(muestras) < 6:
muestras.append(e)
total = sum(porfuente.values())
for fuente, n in porfuente.items():
pub = next(c[5] or "sin licencia" for c in CATALOGOS if c[3] == fuente)
print(f" {fuente:26s} {n:5d} fragmentos [{pub}]")
print(f" total: {total} fragmentos de catalogo")
if a.volcado:
print("\n--- muestras ---")
for e in muestras:
print(f"\n[{e['fuente']} / {e['herramienta']}]\n {e['texto'][:300]}")
return 0
if __name__ == "__main__":
sys.exit(main())