From 275aff14307dc934716fe5d36102ac096518a97a Mon Sep 17 00:00:00 2001 From: rafa Date: Fri, 31 Jul 2026 14:04:31 -0400 Subject: [PATCH] Mirror del Joomla antiguo: versionar los scripts y reponer los assets que faltaban Los scripts del mirror (00-90) vivian solo en el disco. Van al repo; los datos que generan no (16 GB entre crawl, snapshot del origen y Joomla restaurado) -> .gitignore. Nuevo 91-repone-assets404.sh: repone los ficheros que el crawl no capturo porque se referencian SOLO desde CSS y el crawler seguia enlaces HTML (system.css, los fondos de fe_adulta_1, ratingstars.gif de K2). Salian como 404 en los logs de nginx del Hetzner. Descarga por HTTP desde el Joomla local aislado, nunca del filesystem -- mismo principio que el crawl, para no arrastrar los .php comprometidos del #183 -- y escanea PHP embebido antes de copiar a site/. Resultado sobre las 286 rutas unicas con 404 del log: 196 repuestas y verificadas en produccion (196/196 en 200 tras el rsync), 82 que dan 301->404 tambien en el origen (ya estaban rotas en la web original) y 8 rutas basura /%22/... de HTML mal formado. Refs #180 Co-Authored-By: Claude Opus 5 --- .gitignore | 8 + mirror-antiguo/CURRENT_RUN | 1 + mirror-antiguo/README.md | 76 +++++ mirror-antiguo/deploy/404.html | 87 ++++++ mirror-antiguo/deploy/nginx-mirror.conf | 27 ++ mirror-antiguo/inventory-explore.txt | 278 ++++++++++++++++++ mirror-antiguo/scripts/00-probe.sh | 16 + mirror-antiguo/scripts/01-diag-container.sh | 8 + mirror-antiguo/scripts/02-quien-para.sh | 20 ++ mirror-antiguo/scripts/03-probe-urls.sh | 18 ++ mirror-antiguo/scripts/10-build-inventory.sh | 23 ++ mirror-antiguo/scripts/11-validate-sample.sh | 27 ++ mirror-antiguo/scripts/12-net-check.sh | 8 + mirror-antiguo/scripts/19-smoke.sh | 28 ++ mirror-antiguo/scripts/19b-smoke-check.sh | 15 + mirror-antiguo/scripts/20-prep-crawl.sh | 49 +++ mirror-antiguo/scripts/21-crawl-html.sh | 34 +++ mirror-antiguo/scripts/22-monitor.sh | 40 +++ mirror-antiguo/scripts/23-estado.sh | 14 + mirror-antiguo/scripts/24-errores-passA.sh | 15 + mirror-antiguo/scripts/25-dirs-estaticos.sh | 16 + mirror-antiguo/scripts/26-probe-anterior.sh | 11 + mirror-antiguo/scripts/30-extract-links.py | 94 ++++++ mirror-antiguo/scripts/31-fetch-assets.sh | 30 ++ mirror-antiguo/scripts/32-analiza-missing.sh | 18 ++ .../scripts/33-missing-sin-query.sh | 14 + .../scripts/34-clasifica-missing.py | 64 ++++ mirror-antiguo/scripts/35-pase-c-huecos.sh | 51 ++++ mirror-antiguo/scripts/36-estado-passB.sh | 13 + mirror-antiguo/scripts/37-pase-d-anterior.sh | 26 ++ mirror-antiguo/scripts/38-diag-passB.sh | 16 + .../scripts/39-diag-interrogantes.sh | 25 ++ mirror-antiguo/scripts/40-manifest-scan.sh | 40 +++ mirror-antiguo/scripts/41-menu-paths.sh | 13 + mirror-antiguo/scripts/42-corta-a00.sh | 17 ++ mirror-antiguo/scripts/43-pendientes.sh | 46 +++ mirror-antiguo/scripts/44-check-plantilla.sh | 19 ++ .../scripts/44b-check-query-assets.sh | 13 + mirror-antiguo/scripts/45-normalize-links.py | 110 +++++++ .../scripts/46-analiza-404-passC.sh | 20 ++ mirror-antiguo/scripts/47-cruza-404-bd.sh | 18 ++ mirror-antiguo/scripts/48-estado-passD.sh | 14 + mirror-antiguo/scripts/50-parity.py | 83 ++++++ mirror-antiguo/scripts/50b-parity-full.py | 154 ++++++++++ mirror-antiguo/scripts/51-diff-paridad.sh | 11 + .../scripts/52-check-alias-interrogante.sh | 19 ++ .../scripts/53-pase-d2-anterior-inventario.sh | 45 +++ mirror-antiguo/scripts/54-cobertura.py | 47 +++ .../scripts/55-revisa-sospechosos.sh | 13 + .../scripts/56-verifica-limpieza.sh | 14 + mirror-antiguo/scripts/57-lista-22.sh | 14 + mirror-antiguo/scripts/58-verifica-site.sh | 20 ++ .../scripts/59-recopia-nombres-limpios.py | 30 ++ mirror-antiguo/scripts/60-quitar-ua.py | 74 +++++ .../scripts/60-restaurar-entorno.sh | 12 + mirror-antiguo/scripts/70-smoke-nginx.sh | 38 +++ mirror-antiguo/scripts/71-manifest-site.sh | 13 + mirror-antiguo/scripts/80-sync-hetzner.sh | 22 ++ .../scripts/80-wayback-inventario.sh | 17 ++ .../scripts/81-cobertura-wayback.py | 70 +++++ mirror-antiguo/scripts/82-wayback-es.py | 63 ++++ .../scripts/83-contraste-wayback.py | 77 +++++ .../scripts/84-inspecciona-terceros.sh | 27 ++ mirror-antiguo/scripts/85-marca-social.sh | 14 + mirror-antiguo/scripts/86-survey-social.py | 64 ++++ mirror-antiguo/scripts/87-quita-social.py | 108 +++++++ mirror-antiguo/scripts/88-comprueba-ga.sh | 19 ++ mirror-antiguo/scripts/89-verifica-social.sh | 17 ++ mirror-antiguo/scripts/89b-los-dos.sh | 9 + mirror-antiguo/scripts/90-cierre.sh | 23 ++ mirror-antiguo/scripts/91-repone-assets404.sh | 70 +++++ 71 files changed, 2667 insertions(+) create mode 100644 mirror-antiguo/CURRENT_RUN create mode 100644 mirror-antiguo/README.md create mode 100644 mirror-antiguo/deploy/404.html create mode 100644 mirror-antiguo/deploy/nginx-mirror.conf create mode 100644 mirror-antiguo/inventory-explore.txt create mode 100644 mirror-antiguo/scripts/00-probe.sh create mode 100644 mirror-antiguo/scripts/01-diag-container.sh create mode 100644 mirror-antiguo/scripts/02-quien-para.sh create mode 100644 mirror-antiguo/scripts/03-probe-urls.sh create mode 100644 mirror-antiguo/scripts/10-build-inventory.sh create mode 100644 mirror-antiguo/scripts/11-validate-sample.sh create mode 100644 mirror-antiguo/scripts/12-net-check.sh create mode 100644 mirror-antiguo/scripts/19-smoke.sh create mode 100644 mirror-antiguo/scripts/19b-smoke-check.sh create mode 100644 mirror-antiguo/scripts/20-prep-crawl.sh create mode 100644 mirror-antiguo/scripts/21-crawl-html.sh create mode 100644 mirror-antiguo/scripts/22-monitor.sh create mode 100644 mirror-antiguo/scripts/23-estado.sh create mode 100644 mirror-antiguo/scripts/24-errores-passA.sh create mode 100644 mirror-antiguo/scripts/25-dirs-estaticos.sh create mode 100644 mirror-antiguo/scripts/26-probe-anterior.sh create mode 100644 mirror-antiguo/scripts/30-extract-links.py create mode 100644 mirror-antiguo/scripts/31-fetch-assets.sh create mode 100644 mirror-antiguo/scripts/32-analiza-missing.sh create mode 100644 mirror-antiguo/scripts/33-missing-sin-query.sh create mode 100644 mirror-antiguo/scripts/34-clasifica-missing.py create mode 100644 mirror-antiguo/scripts/35-pase-c-huecos.sh create mode 100644 mirror-antiguo/scripts/36-estado-passB.sh create mode 100644 mirror-antiguo/scripts/37-pase-d-anterior.sh create mode 100644 mirror-antiguo/scripts/38-diag-passB.sh create mode 100644 mirror-antiguo/scripts/39-diag-interrogantes.sh create mode 100644 mirror-antiguo/scripts/40-manifest-scan.sh create mode 100644 mirror-antiguo/scripts/41-menu-paths.sh create mode 100644 mirror-antiguo/scripts/42-corta-a00.sh create mode 100644 mirror-antiguo/scripts/43-pendientes.sh create mode 100644 mirror-antiguo/scripts/44-check-plantilla.sh create mode 100644 mirror-antiguo/scripts/44b-check-query-assets.sh create mode 100644 mirror-antiguo/scripts/45-normalize-links.py create mode 100644 mirror-antiguo/scripts/46-analiza-404-passC.sh create mode 100644 mirror-antiguo/scripts/47-cruza-404-bd.sh create mode 100644 mirror-antiguo/scripts/48-estado-passD.sh create mode 100644 mirror-antiguo/scripts/50-parity.py create mode 100644 mirror-antiguo/scripts/50b-parity-full.py create mode 100644 mirror-antiguo/scripts/51-diff-paridad.sh create mode 100644 mirror-antiguo/scripts/52-check-alias-interrogante.sh create mode 100644 mirror-antiguo/scripts/53-pase-d2-anterior-inventario.sh create mode 100644 mirror-antiguo/scripts/54-cobertura.py create mode 100644 mirror-antiguo/scripts/55-revisa-sospechosos.sh create mode 100644 mirror-antiguo/scripts/56-verifica-limpieza.sh create mode 100644 mirror-antiguo/scripts/57-lista-22.sh create mode 100644 mirror-antiguo/scripts/58-verifica-site.sh create mode 100644 mirror-antiguo/scripts/59-recopia-nombres-limpios.py create mode 100644 mirror-antiguo/scripts/60-quitar-ua.py create mode 100644 mirror-antiguo/scripts/60-restaurar-entorno.sh create mode 100644 mirror-antiguo/scripts/70-smoke-nginx.sh create mode 100644 mirror-antiguo/scripts/71-manifest-site.sh create mode 100644 mirror-antiguo/scripts/80-sync-hetzner.sh create mode 100644 mirror-antiguo/scripts/80-wayback-inventario.sh create mode 100644 mirror-antiguo/scripts/81-cobertura-wayback.py create mode 100644 mirror-antiguo/scripts/82-wayback-es.py create mode 100644 mirror-antiguo/scripts/83-contraste-wayback.py create mode 100644 mirror-antiguo/scripts/84-inspecciona-terceros.sh create mode 100644 mirror-antiguo/scripts/85-marca-social.sh create mode 100644 mirror-antiguo/scripts/86-survey-social.py create mode 100644 mirror-antiguo/scripts/87-quita-social.py create mode 100644 mirror-antiguo/scripts/88-comprueba-ga.sh create mode 100644 mirror-antiguo/scripts/89-verifica-social.sh create mode 100644 mirror-antiguo/scripts/89b-los-dos.sh create mode 100644 mirror-antiguo/scripts/90-cierre.sh create mode 100644 mirror-antiguo/scripts/91-repone-assets404.sh diff --git a/.gitignore b/.gitignore index ca8a984..967c352 100644 --- a/.gitignore +++ b/.gitignore @@ -57,3 +57,11 @@ tools/e2e/out/ # Akeeba Kickstart (tool de restauración, no es código del repo) tools/akeeba-kickstart/ + +# Mirror del Joomla antiguo: los scripts SÍ van al repo, los datos NO (16 GB entre +# el crawl, el snapshot del origen y el Joomla restaurado) +mirror-antiguo/runs/ +mirror-antiguo/source/ +mirror-antiguo/restore/ +mirror-antiguo/smoke/ +mirror-antiguo/inventory/ diff --git a/mirror-antiguo/CURRENT_RUN b/mirror-antiguo/CURRENT_RUN new file mode 100644 index 0000000..7664d2c --- /dev/null +++ b/mirror-antiguo/CURRENT_RUN @@ -0,0 +1 @@ +20260729T224708Z diff --git a/mirror-antiguo/README.md b/mirror-antiguo/README.md new file mode 100644 index 0000000..627e9a5 --- /dev/null +++ b/mirror-antiguo/README.md @@ -0,0 +1,76 @@ +# mirror-antiguo — mirror estático del Joomla legacy de feadulta + +Construcción del mirror HTML read-only de `antiguo.feadulta.com` (issue +[rafa/feadulta#180](https://gitea.feadulta.com/rafa/feadulta/issues/180), remediación de #183). + +**Regla de diseño:** el mirror se genera **solo por HTTP**, nunca copiando el filesystem. Así es +imposible arrastrar los `.php` comprometidos del incidente. Lo que se captura es lo que Joomla +*renderiza*. + +## Origen del crawl + +No es producción: es el **Joomla legacy restaurado en local**. + +| | | +|---|---| +| Contenedor | `joomla-mirror-web` (`php:7.4-apache`), red `joomla-migration_joomla-net`, IP `172.20.0.5` | +| Acceso | `http://127.0.0.1:8086` o `http://antiguo.feadulta.com` (entrada en `/etc/hosts` → 172.20.0.5) | +| BD | `joomla_mirror` dentro del contenedor `joomla-mysql` | +| Snapshot fuente | `source/antiguo-20260729.tar.gz` + `source/fejoomla3-20260729.sql.gz`, hashes en `MANIFEST-source.sha256` | + +⚠️ La imagen base necesitó `mysqli`, `pdo_mysql`, `gd`, `zip` y `mod_rewrite` instalados en caliente: +**se pierden si el contenedor se recrea**, no si solo se para/arranca. Para cambiar memoria o política +de reinicio usar `docker update`, nunca `docker rm` + `docker run`. + +⚠️ WSL2 se apaga cuando no queda ninguna sesión abierta desde Windows, y al apagarse para los +contenedores. Antes de un proceso de horas, dejar un proceso ancla vivo en WSL. + +## Datos del sitio + +- SEF: `sef=1`, `sef_rewrite=1`, `sef_suffix=1` → las URLs terminan en `.html`. +- Un solo idioma de contenido publicado: `es-ES` (sef `es`); todo el contenido es `language='*'`. + El sitio hace **301 de `/x` → `/es/x`**, así que el inventario se genera ya con `/es/`. +- Solo 2 categorías K2 y un único ítem de menú de K2: `buscadoravanzado` (Itemid 138). + Las URLs de ítem son `/es/buscadoravanzado/item/-.html`. +- 16.269 ítems K2 publicados · 9.079 artículos com_content · 216 ítems de menú · 68 categorías. + +## Método + +El inventario **no se reconstruye a mano**: lo genera el propio router de Joomla. `_genurls.php` +(en la raíz del sitio restaurado) arranca el framework por HTTP y llama a `JRoute::_()` y +`K2HelperRoute::getItemRoute()`, así que las URLs son idénticas a las que el sitio imprime. + +Cuatro pases, todos acotados por listas — **nunca por recursión libre** (ver post-mortem del +2026-07-29 en #180 comment-497): + +| Pase | Qué captura | Script | +|---|---|---| +| A | Las 25.437 URLs del inventario | `21-crawl-html.sh` | +| B | Recursos (css/js/img/mp3/pdf) referenciados por el HTML capturado | `30-extract-links.py` + `31-fetch-assets.sh` | +| C | Huecos: rutas de menú alternativas, páginas de autor de K2, `/anterior`, `/ediciones` | `43-pendientes.sh` | +| D | `/anterior`, la web estática anterior a Joomla (HTML plano, recursión finita) | `37-pase-d-anterior.sh` | + +`raw/` es **inmutable**. `site/` es el derivado servible (`45-normalize-links.py`). + +## Trampas encontradas (y cómo se resuelven) + +1. **Paginación de K2**: los enlaces acumulan `&start=` en vez de reemplazarlo → espacio de URLs + infinito. Es lo que tumbó la VM el 29-jul. Se evita capturando por inventario, y el + `--reject-regex` incluye `start|limitstart|limit|print|tmpl|format|searchword|task|orderby|filter`. +2. **Alias con `?` literal**: ~198 artículos tienen el signo de interrogación dentro del alias + (`...-dios-nos-ama?.html`). Para cualquier cliente HTTP eso es el separador de query, así que la + ruta real es la parte anterior al `?` y wget guarda un fichero **sin extensión**. Es correcto para + servirlo estáticamente, pero nginx necesita `default_type text/html` en esa ubicación o el + navegador se lo descargará en vez de mostrarlo. +3. **Assets con cache-busting** (`core.js?a32fb…`): wget mete la query en el nombre del fichero. El + paso de normalización deja también una copia con el nombre limpio, que es la que pedirá el + servidor estático. +4. **Menús cuyo componente ya no existe** (`com_surveys`, `com_breezingforms`): `JRoute` no les + construye ruta y devuelven 404. Ya son 404 en producción; para el resto de ítems de menú la ruta + autoritativa es la columna `path` de `#__menu` (`41-menu-paths.sh`). + +## Scripts + +Numerados por orden de ejecución en `scripts/`. `23-estado.sh` da el estado en cualquier momento; +`22-monitor.sh` vigila el crawl y **lo aborta** si el servidor pasa de 100 respuestas 500 o si la RAM +libre baja de 800 MB. `60-restaurar-entorno.sh` rearranca los contenedores parados durante el crawl. diff --git a/mirror-antiguo/deploy/404.html b/mirror-antiguo/deploy/404.html new file mode 100644 index 0000000..305f39b --- /dev/null +++ b/mirror-antiguo/deploy/404.html @@ -0,0 +1,87 @@ + + + + + + +Página no encontrada · Page not found — Archivo de feadulta.com + + + +
+ +

Error 404

+

Esta página no está en el archivo

+ +

Estás en el archivo histórico de feadulta.com: una copia de solo lectura de + la web antigua, conservada tal y como estaba. No se actualiza y no admite búsquedas ni + formularios.

+ +

La dirección que has seguido no existe aquí. Puede que el enlace esté mal escrito, que la + página se retirase antes de hacer esta copia, o que su contenido viva ahora en la web nueva.

+ + + +
+ +
+

Error 404

+

This page is not in the archive

+ +

You have reached the historical archive of feadulta.com: a read-only copy + of the old website, kept as it was. It is not updated, and search and forms do not work.

+ +

The address you followed does not exist here. The link may be mistyped, the page may have + been removed before this copy was made, or its content may now live on the new website.

+ + +
+ +
+

Archivo estático · Static archive — feadulta.com

+ +
+ + diff --git a/mirror-antiguo/deploy/nginx-mirror.conf b/mirror-antiguo/deploy/nginx-mirror.conf new file mode 100644 index 0000000..11d8944 --- /dev/null +++ b/mirror-antiguo/deploy/nginx-mirror.conf @@ -0,0 +1,27 @@ +server { + listen 80; + server_name _; + root /usr/share/nginx/html; + + charset utf-8; + index index.html; + + add_header X-Robots-Tag "noindex, nofollow" always; + + # Los alias de K2 que llevan un '?' literal se sirven desde ficheros SIN extension + # (el navegador pide la ruta hasta el '?'). Sin esto, nginx los manda como + # application/octet-stream y el navegador se los descarga en vez de mostrarlos. + default_type text/html; + + location / { + try_files $uri $uri.html $uri/index.html =404; + } + + # Archivo historico: aqui no hay PHP. Cualquier .php es un fichero estatico inerte. + location ~ \.php { + default_type text/html; + try_files $uri $uri/index.html =404; + } + + error_page 404 /404.html; +} diff --git a/mirror-antiguo/inventory-explore.txt b/mirror-antiguo/inventory-explore.txt new file mode 100644 index 0000000..d316f35 --- /dev/null +++ b/mirror-antiguo/inventory-explore.txt @@ -0,0 +1,278 @@ +### Idiomas de contenido (ext_languages) +1 en-GB en -2 +3 es-ES es 1 + +### K2 items por estado +0 0 1885 +0 1 9 +1 0 16269 +1 1 54 + +### K2 items por idioma (publicados) +* 16269 + +### K2 categorias +2 + +### K2 categorias raiz (parent=0) +29 Feadulta feadulta * +30 Sin categoría sincategoria * + +### com_content por estado +-2 59 +0 38 +1 9079 + +### com_content por idioma (publicados) +* 9079 + +### Menus +mainmenu Menú Principal +idioma Idioma +buscadores Buscadores +libros LIBROS +encuesta ENCUESTA +menuuser6 menu_user6 +secciones Secciones +eucamenu eucamenu +comentmenu comentmenu +art1menu art1menu +art2menu art2menu +multimenu multimenu +resumeneslibros Resúmenes Libros +colaboradores Colaboradores +art3menu art3menu +multi2menu multi2menu +proyemenu proyemenu +sec2menu sec2menu +secc1menu secc1menu +sec3menu sec3menu +sec1-examen-menu sec1examen +sec4menu sec4menu +sec5menu sec5menu +sec2-examen1-menu sec2-examen1 +contactar CONTACTAR +el-ano-de-la-biblia Año Biblia +listado A la fuente cada día (Fray Marcos) + +### Items de menu publicados +1 Menu_Item_Root * 1 0 +184 art1menu art1col1 art1col1 index.php?option=com_content&view=article&id=109 component * 1 0 +182 art1menu art1col2 art1col2 index.php?option=com_content&view=article&id=67 component * 1 0 +183 art1menu art1col3 art1col3 index.php?option=com_content&view=article&id=66 component * 1 0 +357 art1menu art1col4 art1col4 index.php?option=com_content&view=article&id=2279 component * 1 0 +185 art2menu art2col1 art2col1 index.php?option=com_content&view=article&id=69 component * 1 0 +186 art2menu art2col2 art2col2 index.php?option=com_content&view=article&id=68 component * 1 0 +221 art2menu art2col3 art2col3 index.php?option=com_content&view=article&id=125 component * 1 0 +358 art2menu art2col4 art2col4 index.php?option=com_content&view=article&id=2280 component * 1 0 +312 art3menu art3col1 art3col1 index.php?option=com_content&view=article&id=1449 component * 1 0 +313 art3menu art3col2 art3col2 index.php?option=com_content&view=article&id=1450 component * 1 0 +354 art3menu art3col3 art3col3 index.php?option=com_content&view=article&id=2131 component * 1 0 +138 buscadores Buscador avanzado buscadoravanzado index.php?option=com_k2&view=itemlist&layout=category&task=category&id=29 component * 1 0 +200 colaboradores Arregi arregi index.php?option=com_content&view=article&id=83 component * 1 0 +555 colaboradores Inma Calvo inma index.php?option=com_content&view=article&id=5016 component * 1 0 +554 colaboradores África de La Cruz africa index.php?option=com_content&view=article&id=5017 component * 1 0 +201 colaboradores Pagola pagola index.php?option=com_content&view=article&id=86 component * 1 0 +202 colaboradores Lozano lozano index.php?option=com_content&view=article&id=87 component * 1 0 +203 colaboradores Marcos marcos index.php?option=com_content&view=article&id=85 component * 1 0 +204 colaboradores Eloy eloy index.php?option=com_content&view=article&id=88 component * 1 0 +205 colaboradores Dolores dolores index.php?option=com_content&view=article&id=84 component * 1 0 +206 colaboradores Hojman hojman index.php?option=com_content&view=article&id=90 component * 1 0 +207 colaboradores Koldo koldo index.php?option=com_content&view=article&id=92 component * 1 0 +208 colaboradores Matilde Gastalver matilde-gastalver index.php?option=com_content&view=article&id=91 component * 1 0 +209 colaboradores Ulibarri ulibarri index.php?option=com_content&view=article&id=93 component * 1 0 +210 colaboradores Vicente vicente index.php?option=com_content&view=article&id=89 component * 1 0 +211 colaboradores Rafael Calvo Beca rafael-calvo-beca index.php?option=com_content&view=article&id=94 component * 1 0 +212 colaboradores Galarreta galarreta index.php?option=com_content&view=article&id=95 component * 1 0 +213 colaboradores Mellado mellado index.php?option=com_content&view=article&id=96 component * 1 0 +214 colaboradores Salazar salazar index.php?option=com_content&view=article&id=97 component * 1 0 +243 colaboradores Mari patxi mari-patxi index.php?option=com_content&view=article&id=289 component * 1 0 +245 colaboradores Mari Paz López Santos mari-paz-lopez-santos index.php?option=com_content&view=article&id=312 component * 1 0 +247 colaboradores Mariangeles mariangeles index.php?option=com_content&view=article&id=341 component * 1 0 +291 colaboradores Carmona carmona index.php?option=com_content&view=article&id=937 component * 1 0 +292 colaboradores Rogelio rogelio index.php?option=com_content&view=article&id=938 component * 1 0 +293 colaboradores Salome salome index.php?option=com_content&view=article&id=936 component * 1 0 +294 colaboradores Victor blanco victor-blanco index.php?option=com_content&view=article&id=939 component * 1 0 +306 colaboradores Viki viki index.php?option=com_content&view=article&id=1299 component * 1 0 +311 colaboradores Lenin lenin index.php?option=com_content&view=article&id=1391 component * 1 0 +315 colaboradores Vilabrille vilabrille index.php?option=com_content&view=article&id=1470 component * 1 0 +319 colaboradores Luque luque index.php?option=com_content&view=article&id=1520 component * 1 0 +360 colaboradores Yolanda yolanda index.php?option=com_content&view=article&id=2331 component * 1 0 +369 colaboradores Gonzalo Haya gonzalo-haya index.php?option=com_content&view=article&id=2416 component * 1 0 +370 colaboradores José Luis Sicre jose-luis-sicre index.php?option=com_content&view=article&id=2424 component * 1 0 +371 colaboradores Andrés Torres Queiruga andres-torres-queiruga index.php?option=com_content&view=article&id=2425 component * 1 0 +372 colaboradores Xabier Pikaza xabier-pikaza index.php?option=com_content&view=article&id=2426 component * 1 0 +373 colaboradores Leandro Sequeiros leandro-sequeiros index.php?option=com_content&view=article&id=2423 component * 1 0 +374 colaboradores José María Castillo jose-maria-castillo index.php?option=com_content&view=article&id=2427 component * 1 0 +375 colaboradores Juan Antonio Estrada juan-antonio-estrada index.php?option=com_content&view=article&id=2428 component * 1 0 +376 colaboradores Juan José Tamayo juan-jose-tamayo index.php?option=com_content&view=article&id=2429 component * 1 0 +386 colaboradores Pope Godoy pope-godoy index.php?option=com_content&view=article&id=2666 component * 1 0 +391 colaboradores anademiguel anademiguel index.php?option=com_content&view=article&id=2887 component * 1 0 +392 colaboradores Suyapa Pérez Escapini suyapa-perez-escapini index.php?option=com_content&view=article&id=3103 component * 1 0 +553 colaboradores Ramón Hernández Martín ramon index.php?option=com_content&view=article&id=5015 component * 1 0 +174 comentmenu comentcol1 comentcol1 index.php?option=com_content&view=article&id=105 component * 1 0 +179 comentmenu comentcol2 comentcol2 index.php?option=com_content&view=article&id=106 component * 1 0 +180 comentmenu comentcol3 comentcol3 index.php?option=com_content&view=article&id=107 component * 1 0 +181 comentmenu comentcol4 comentcol4 index.php?option=com_content&view=article&id=108 component * 1 0 +537 contactar Elemento Menú Contactar elemento-menu-contactar index.php?option=com_breezingforms&view=form component * 1 0 +547 contactar Para recibir carta de novedades para-recibir-carta-de-novedades index.php?option=com_breezingforms&view=form component * 1 0 +572 el-ano-de-la-biblia Año de la Biblia biblia index.php?option=com_content&view=article&id=5335 component * 1 0 +151 encuesta encuesta encuesta index.php?option=com_surveys&view=editsurvey component * 1 0 +274 encuesta Resultados resultados index.php?option=com_surveys&view=indivsurveyresult component * 1 0 +444 encuesta Ayúdanos a elegir la nueva imagen de Feadulta ayudanos-a-elegir-nueva-imagen index.php?option=com_content&view=article&id=3957 component * 1 0 +450 encuesta Resultado resultado index.php?option=com_content&view=article&id=3990 component * 1 0 +175 eucamenu eucacol1 eucol1 index.php?option=com_content&view=article&id=101 component * 1 0 +176 eucamenu eucacol2 eucol2 index.php?option=com_content&view=article&id=102 component * 1 0 +177 eucamenu eucacol3 eucol3 index.php?option=com_content&view=article&id=103 component * 1 0 +178 eucamenu eucacol4 eucol4 index.php?option=com_content&view=article&id=104 component * 1 0 +136 idioma Español espanol index.php?option=com_content&view=article&id=1 component * 1 0 +137 idioma Ingles ingles index.php?option=com_content&view=article&id=1 component * 1 0 +187 libros libroresumen1 libroresumen1 index.php?option=com_content&view=article&id=30 component * 1 0 +188 libros libroresumen2 libroresumen2 index.php?option=com_content&view=article&id=31 component * 1 0 +101 mainmenu PORTADA home index.php?option=com_content&view=featured component * 1 1 +130 mainmenu QUIÉNES SOMOS quienessomos index.php?Itemid= alias * 1 0 +131 mainmenu COLABORADORES quienessomos/colaboradores index.php?option=com_content&view=article&id=43 component * 1 0 +170 mainmenu ESTE PORTAL quienessomos/portal index.php?option=com_content&view=article&id=59 component * 1 0 +132 mainmenu PARA PONER AL DÍA LA FE quienessomos/poneraldialafe index.php?option=com_content&view=article&id=44 component * 1 0 +134 mainmenu AYUDA ayuda index.php?option=com_content&view=article&id=45 component * 1 0 +550 mainmenu ESTA SEMANA ayuda/esta-semana index.php?option=com_content&view=category&layout=blog&id=27 component * 1 0 +551 mainmenu LA SEMANA PASADA ayuda/semana-pasada index.php?option=com_content&view=category&layout=blog&id=41 component * 1 0 +552 mainmenu OTRAS SEMANAS ayuda/otras-semanas index.php?option=com_content&view=category&id=40 component * 1 0 +240 mainmenu Acceso a web anterior Feadulta ayuda/2012-05-25-09-13-26 /anterior url * 1 0 +241 mainmenu Para navegar en esta página ayuda/para-navegar-en-esta-pagina index.php?option=com_content&view=article&id=45 component * 1 0 +244 mainmenu Vídeo tutorial ayuda/video-tutoriales index.php?option=com_content&view=article&id=293 component * 1 0 +248 mainmenu CÓMO USAR EL BUSCADOR AVANZADO ayuda/como-usar-el-buscador-avanzado index.php?option=com_content&view=article&id=358 component * 1 0 +359 mainmenu Para comprar un libro ayuda/para-comprar-un-libro index.php?option=com_content&view=article&id=2291 component * 1 0 +523 mainmenu Catálogo de publicaciones 2018 ayuda/catalogo-de-publicaciones-2018 index.php?option=com_content&view=article&id=4717 component * 1 0 +507 mainmenu NUEVA POLÍTICA DE DATOS ayuda/nueva-politica-de-datos index.php?option=com_content&view=article&id=4475 component * 1 0 +536 mainmenu CONTACTAR contactar index.php?option=com_content&view=article&id=5266 component * 1 0 +568 mainmenu Para contactar con nosotros contactar/para-contactar-con-nosotros index.php?option=com_content&view=article&id=5266 component * 1 0 +567 mainmenu Para recibir la carta de novedades contactar/para-recibir-la-carta-de-novedades index.php?option=com_content&view=article&id=5265 component * 1 0 +573 mainmenu Para inscribirse en la Escuela contactar/para-inscribirse-en-la-escuela index.php?option=com_content&view=article&id=5407 component * 1 0 +362 mainmenu ESCUELA effa index.php?option=com_content&view=article&id=2408 component * 1 0 +385 mainmenu LIBRERÍA 2015-03-30-15-34-35 https://edicionesfeadulta.com url * 1 0 +569 mainmenu Buscar buscar index.php?option=com_search&view=search component * 1 0 +161 menuuser6 Catálogo catalogolibros index.php?option=com_content&view=article&id=40 component * 1 0 +324 multi2menu multi2col1 multi2col1 index.php?option=com_content&view=article&id=1657 component * 1 0 +325 multi2menu multi2col2 multi2col2 index.php?option=com_content&view=article&id=1658 component * 1 0 +326 multi2menu multi2col3 multi2col3 index.php?option=com_content&view=article&id=1659 component * 1 0 +237 multimenu multicol1 multicol1 index.php?option=com_content&view=article&id=282 component * 1 0 +238 multimenu multicol2 multicol2 index.php?option=com_content&view=article&id=283 component * 1 0 +239 multimenu multicol3 multicol3 index.php?option=com_content&view=article&id=284 component * 1 0 +363 proyemenu proyecol1 proyecol1 index.php?option=com_content&view=article&id=2410 component * 1 0 +364 proyemenu proyecol2 proyecol2 index.php?option=com_content&view=article&id=2412 component * 1 0 +365 proyemenu proyecol3 proyecol3 index.php?option=com_content&view=article&id=2411 component * 1 0 +380 proyemenu proyecolumna4 proyecolumna4 index.php?option=com_content&view=article&id=2430 component * 1 0 +190 resumeneslibros Resumen libro 1 resumenlibro1 index.php?option=com_content&view=article&id=30 component * 1 0 +191 resumeneslibros Resumen libro 2 resumenlibro2 index.php?option=com_content&view=article&id=74 component * 1 0 +192 resumeneslibros Resumen libro 10 resumenlibro10 index.php?option=com_content&view=article&id=82 component * 1 0 +193 resumeneslibros Resumen libro 3 resumenlibro3 index.php?option=com_content&view=article&id=75 component * 1 0 +194 resumeneslibros Resumen libro 4 resumenlibro4 index.php?option=com_content&view=article&id=76 component * 1 0 +195 resumeneslibros Resumen libro 5 resumenlibro5 index.php?option=com_content&view=article&id=77 component * 1 0 +196 resumeneslibros Resumen libro 6 resumenlibro6 index.php?option=com_content&view=article&id=78 component * 1 0 +197 resumeneslibros Resumen libro 7 resumenlibro7 index.php?option=com_content&view=article&id=79 component * 1 0 +198 resumeneslibros Resumen libro 8 resumenlibro8 index.php?option=com_content&view=article&id=80 component * 1 0 +199 resumeneslibros Resumen libro 9 resumenlibro9 index.php?option=com_content&view=article&id=81 component * 1 0 +215 resumeneslibros Resumen libro 11 resumenlibro11 index.php?option=com_content&view=article&id=99 component * 1 0 +219 resumeneslibros Resumen libro 12 resumenlibro12 index.php?option=com_content&view=article&id=111 component * 1 0 +249 resumeneslibros Resumen libro 13 resumen-libro-13 index.php?option=com_content&view=article&id=363 component * 1 0 +251 resumeneslibros Resumen libro 14 libroresumen14 index.php?option=com_content&view=article&id=376 component * 1 0 +275 resumeneslibros Resumen libro 15 resumen-libro-15 index.php?option=com_content&view=article&id=849 component * 1 0 +277 resumeneslibros Resumen libro 16 resumen-libro-16 index.php?option=com_content&view=article&id=863 component * 1 0 +295 resumeneslibros Resumen libro 17 resumen-libro-17 index.php?option=com_content&view=article&id=978 component * 1 0 +301 resumeneslibros Resumen libro 18 resumen-libro-18 index.php?option=com_content&view=article&id=1098 component * 1 0 +304 resumeneslibros Resumen libro 19 resumenlibro19 index.php?option=com_content&view=article&id=1197 component * 1 0 +318 resumeneslibros Resumen libro 20 resumen-libro-20 index.php?option=com_content&view=article&id=1501 component * 1 0 +321 resumeneslibros Resumen libro 21 resumen-libro-21 index.php?option=com_content&view=article&id=1611 component * 1 0 +333 resumeneslibros Resumen libro 23 resumen-libro-23 index.php?option=com_content&view=article&id=1818 component * 1 0 +334 resumeneslibros Resumen libro 24 resumen-libro-24 index.php?option=com_content&view=article&id=1862 component * 1 0 +335 resumeneslibros Resumen libro 25 resumen-libro-25 index.php?option=com_content&view=article&id=1911 component * 1 0 +350 resumeneslibros Resumen libro 26 resumen-libro-26 index.php?option=com_content&view=article&id=2019 component * 1 0 +356 resumeneslibros Resumen libro 27 resumen-libro-27 index.php?option=com_content&view=article&id=2228 component * 1 0 +384 resumeneslibros Resumen libro 30 resumen-libro-30 index.php?option=com_content&view=article&id=2514 component * 1 0 +388 sec1-examen-menu Examen de Espiritualidad examen-de-espiritualidad index.php?option=com_surveys&view=editsurvey component * 1 0 +394 sec2-examen1-menu Examen de Hermenéutica y Antiguo Testamento examen-de-hermeneutica-y-antiguo-testamento index.php?option=com_surveys&view=editsurvey component * 1 0 +381 secc1menu secc1col1 secc1col1 index.php?option=com_content&view=article&id=2432 component * 1 0 +382 secc1menu secc1col2 secc1col2 index.php?option=com_content&view=article&id=2433 component * 1 0 +272 secciones Iniciación cristiana iniciacion-cristiana index.php?option=com_content&view=article&id=2222 component * 1 0 +556 secciones Programa de las V Jornadas EFFA programa-de-las-v-jornadas-effa index.php?option=com_content&view=article&id=2222 component * 1 0 +163 secciones El rincón del Viajero rinconviajero index.php?option=com_content&view=article&id=55 component * 1 0 +164 secciones Último libro publicado ultimolibro index.php?option=com_content&view=article&id=56 component * 1 0 +165 secciones Cartas que nos llegan cartas index.php?option=com_content&view=category&layout=blog&id=42 component * 1 0 +189 secciones Catálogo de libros Feadulta catalogo-de-libros-feadulta index.php?option=com_content&view=article&id=40 component * 1 0 +216 secciones Compras en España compras-en-espana index.php?option=com_content&view=article&id=25 component * 1 0 +217 secciones Compras NO España compras-no-espana index.php?option=com_content&view=article&id=26 component * 1 0 +218 secciones Prefacio 12 prefacio-12 index.php?option=com_content&view=article&id=110 component * 1 0 +220 secciones Portales y revistas portales-y-revistas index.php?option=com_content&view=article&id=116 component * 1 0 +223 secciones Canciones-plegarias canciones-plegarias index.php?option=com_content&view=category&id=44 component * 1 0 +222 secciones Cantos de entrada cantos-de-entrada index.php?option=com_content&view=category&id=43 component * 1 0 +224 secciones Cantos acción de gracias cantos-accion-de-gracias index.php?option=com_content&view=category&id=45 component * 1 0 +225 secciones Otros cantos otros-cantos index.php?option=com_content&view=categories&id=46 component * 1 0 +231 secciones Tablón de anuncios tablon-de-anuncios index.php?option=com_content&view=category&layout=blog&id=52 component * 1 0 +232 secciones Blogs blogs index.php?option=com_content&view=article&id=157 component * 1 0 +233 secciones ONGs ongs index.php?option=com_content&view=article&id=158 component * 1 0 +234 secciones Otras comunidades otras-comunidades index.php?option=com_content&view=article&id=159 component * 1 0 +235 secciones Registro personal registro-personal index.php?option=com_content&view=category&layout=blog&id=74 component * 1 0 +236 secciones Comunidades comunidades index.php?option=com_content&view=article&id=160 component * 1 0 +242 secciones Índice multimedia indice-multimedia index.php?option=com_content&view=category&id=54 component * 1 0 +246 secciones CRISTIANISMO, MERCADO Y MOVIMIENTOS SOCIALES cristianismo-mercado-y-movimientos-sociales index.php?option=com_content&view=article&id=332 component * 1 0 +250 secciones Índice cronológico indice-cronologico index.php?option=com_content&view=category&id=55 component * 1 0 +252 secciones Evangelios y comentarios evangelios-y-comentarios index.php?option=com_content&view=category&id=56 component * 1 0 +253 secciones Enlaces enlaces index.php?option=com_content&view=category&id=57 component * 1 0 +254 secciones Cantoral cantoral index.php?option=com_content&view=categories&id=58 component * 1 0 +255 secciones PELÍCULAS peliculas index.php?option=com_content&view=article&id=417 component * 1 0 +256 secciones RESEÑAS DE LIBROS resenas-de-libros index.php?option=com_content&view=article&id=418 component * 1 0 +257 secciones Pensamientos pensamientos index.php?option=com_content&view=article&id=419 component * 1 0 +258 secciones ÍNDICE MULTIMEDIA WEB ANTERIOR indice-multimedia-web-anterior index.php?option=com_content&view=article&id=420 component * 1 0 +259 secciones PAUSAS E INSTANTES pausas-e-instantes index.php?option=com_content&view=article&id=421 component * 1 0 +260 secciones ORACIONES EUCARÍSTICAS oraciones-eucaristicas index.php?option=com_content&view=article&id=422 component * 1 0 +261 secciones A MODO DE SALMO a-modo-de-salmos index.php?option=com_content&view=article&id=423 component * 1 0 +262 secciones AUTORES autores index.php?option=com_content&view=categories&id=59 component * 1 0 +263 secciones TEMAS temas index.php?option=com_content&view=category&id=60 component * 1 0 +264 secciones Nuestros archivos nuestros-archivos index.php?option=com_content&view=article&id=439 component * 1 0 +265 secciones PRECES Y ORACIONES VARIAS preces-y-oraciones-varias index.php?option=com_content&view=article&id=471 component * 1 0 +266 secciones Donaciones donaciones index.php?option=com_content&view=article&id=494 component * 1 0 +270 secciones Lista de autores habituales lista-de-autores-habituales index.php?option=com_content&view=category&id=62 component * 1 0 +268 secciones Lista completa de autores por orden alfabético lista-completa-de-autores-por-orden-alfabetico index.php?option=com_content&view=category&id=61 component * 1 0 +271 secciones La suma de todos la-suma-de-todos index.php?option=com_content&view=article&id=547 component * 1 0 +276 secciones PLANUAL feadulta.com 2012-2013 planual-feadultacom-2012-2013 index.php?option=com_content&view=article&id=857 component * 1 0 +278 secciones AMIGOS DE FEADULTA amigos-de-feadulta index.php?option=com_content&view=article&id=876 component * 1 0 +296 secciones NOTICIAS DE ALCANCE noticias-de-alcance index.php?option=com_content&view=category&layout=blog&id=64 component * 1 0 +297 secciones Cantoral de SALOMÉ ARRICIBITA cantoral-de-salome-arricibita index.php?option=com_content&view=category&id=65 component * 1 0 +298 secciones Ambientación musical para BODAS ambientacion-musical-para-bodas index.php?option=com_content&view=category&id=66 component * 1 0 +299 secciones Cantoral para COMUNIDADES cantoral-para-comunidades index.php?option=com_content&view=article&id=1088 component * 1 0 +300 secciones OTRAS canciones otras-canciones index.php?option=com_content&view=article&id=1089 component * 1 0 +226 secciones Himnos de gloria himnos-de-gloria index.php?option=com_content&view=category&id=47 component * 1 0 +227 secciones Marianos marianos index.php?option=com_content&view=category&id=48 component * 1 0 +228 secciones Adviento y Navidad adviento-y-navidad index.php?option=com_content&view=category&id=49 component * 1 0 +229 secciones Otros cantos otroscantos index.php?option=com_content&view=category&id=50 component * 1 0 +230 secciones Canciones populares canciones-populares index.php?option=com_content&view=category&id=51 component * 1 0 +302 secciones Autores lista autores-lista index.php?option=com_content&view=article&id=1151 component * 1 0 +303 secciones Indice cantoral indice-cantoral index.php?option=com_content&view=article&id=1152 component * 1 0 +305 secciones Temas temassubtemas index.php?option=com_content&view=article&id=1213 component * 1 0 +307 secciones Feadulta en Facebook feadulta-en-facebook index.php?option=com_content&view=category&layout=blog&id=75 component * 1 0 +308 secciones PRIMERA JORNADA 'FEADULTA' primera-jornada-feadulta index.php?option=com_content&view=article&id=1373 component * 1 0 +314 secciones 1jornada_feadulta 1jornadafeadulta index.php?option=com_content&view=article&id=1459 component * 1 0 +316 secciones PLANUAL feadulta.com 2013-2014 planual-feadultacom-2013-2014 index.php?option=com_content&view=article&id=1492 component * 1 0 +317 secciones Multimedia multimedia index.php?option=com_content&view=article&id=1493 component * 1 0 +320 secciones Videos videos index.php?option=com_content&view=category&id=77 component * 1 0 +322 secciones Libros y e-books libros-y-e-books index.php?option=com_content&view=article&id=1625 component * 1 0 +323 secciones In memoriam in-memoriam index.php?option=com_content&view=article&id=1629 component * 1 0 +329 secciones Reflexiones reflexiones index.php?option=com_content&view=article&id=1690 component * 1 0 +330 secciones ENTREVISTA A JOSÉ MARÍA CASTILLO entrevista-a-jose-maria-castillo index.php?option=com_content&view=article&id=1694 component * 1 0 +331 secciones PREGUNTA APREMIANTE: ¿RELIGIÓN O EVANGELIO? pregunta-apremiante-religion-o-evangelio index.php?option=com_content&view=article&id=1707 component * 1 0 +332 secciones MÁS ALLÁ DE MÍ... mas-alla-de-mi index.php?option=com_content&view=article&id=1791 component * 1 0 +351 secciones Cancioneros diversos cancioneros-diversos index.php?option=com_content&view=article&id=2090 component * 1 0 +352 secciones Anáfora anafora index.php?option=com_content&view=category&id=80 component * 1 0 +353 secciones Comunidades cristianas comunidades-cristianas index.php?option=com_content&view=category&layout=blog&id=81 component * 1 0 +355 secciones Condolencias Conchita condolencias-conchita index.php?option=com_content&view=category&layout=blog&id=82 component * 1 0 +361 secciones Jornadas con José Antonio Pagola jornadas-con-jose-antonio-pagola index.php?option=com_content&view=article&id=2330 component * 1 0 +102 secciones CARTA DE NOVEDADES carta index.php?Itemid= alias * 1 0 +127 secciones ESTA SEMANA carta/estasemana index.php?option=com_content&view=category&layout=blog&id=27 component * 1 0 +128 secciones LA SEMANA PASADA carta/semanapasada index.php?option=com_content&view=category&layout=blog&id=41 component * 1 0 +129 secciones OTRAS SEMANAS carta/otrassemanas index.php?option=com_content&view=category&id=40 component * 1 0 +383 secciones LA INTELIGENCIA ESPIRITUAL la-inteligencia-espiritual index.php?option=com_content&view=article&id=2438 component * 1 0 +587 secciones Evangelio diario evangelio-diario index.php?option=com_content&view=article&id=6269 component * 1 0 +617 secciones A la fuente cada día alafuente index.php?option=com_content&view=category&id=97 component * 1 0 +618 secciones Otro evangelio es posible otroevangelio index.php?option=com_content&view=category&id=98 component * 1 0 + diff --git a/mirror-antiguo/scripts/00-probe.sh b/mirror-antiguo/scripts/00-probe.sh new file mode 100644 index 0000000..6c2d2b4 --- /dev/null +++ b/mirror-antiguo/scripts/00-probe.sh @@ -0,0 +1,16 @@ +#!/bin/bash +# Sondeo del entorno restaurado +set -u +docker update --restart unless-stopped joomla-mirror-web >/dev/null +docker start joomla-mirror-web >/dev/null 2>&1 +sleep 3 +echo "=== version.php ===" +docker exec joomla-mirror-web grep -E "RELEASE|DEV_LEVEL|PRODUCT" /var/www/html/libraries/cms/version/version.php | head -6 +echo "=== sef en configuration.php ===" +docker exec joomla-mirror-web grep -E 'sef|live_site|offline|dbprefix' /var/www/html/configuration.php +echo "=== componentes ===" +docker exec joomla-mirror-web ls /var/www/html/components/ | tr '\n' ' ' +echo +echo "=== plugins system (sef/redirect) ===" +docker exec joomla-mirror-web ls /var/www/html/plugins/system/ | tr '\n' ' ' +echo diff --git a/mirror-antiguo/scripts/01-diag-container.sh b/mirror-antiguo/scripts/01-diag-container.sh new file mode 100644 index 0000000..352beb6 --- /dev/null +++ b/mirror-antiguo/scripts/01-diag-container.sh @@ -0,0 +1,8 @@ +#!/bin/bash +set -u +echo "=== estado ===" +docker inspect joomla-mirror-web --format 'Status={{.State.Status}} Exit={{.State.ExitCode}} OOM={{.State.OOMKilled}} Started={{.State.StartedAt}} Finished={{.State.FinishedAt}} RestartPolicy={{.HostConfig.RestartPolicy.Name}} Mem={{.HostConfig.Memory}}' +echo "=== ultimas lineas del log (sin access log) ===" +docker logs --tail 200 joomla-mirror-web 2>&1 | grep -v 'GET /' | tail -20 +echo "=== docker events ultimos 30 min ===" +docker events --since 30m --until 0s --filter container=joomla-mirror-web --format '{{.Time}} {{.Action}}' 2>/dev/null | tail -20 diff --git a/mirror-antiguo/scripts/02-quien-para.sh b/mirror-antiguo/scripts/02-quien-para.sh new file mode 100644 index 0000000..24f5b60 --- /dev/null +++ b/mirror-antiguo/scripts/02-quien-para.sh @@ -0,0 +1,20 @@ +#!/bin/bash +set -u +echo "=== crontab rafa ===" +crontab -l 2>/dev/null | grep -iE 'docker|joomla|mirror' || echo "(nada relevante)" +echo "=== crontab root ===" +sudo -n crontab -l 2>/dev/null | grep -iE 'docker|joomla|mirror' || echo "(no accesible o nada)" +echo "=== /etc/cron.d ===" +grep -rliE 'docker (stop|kill)|joomla' /etc/cron.d /etc/cron.daily 2>/dev/null || echo "(nada)" +echo "=== procesos sospechosos ===" +ps -eo pid,etimes,cmd | grep -iE 'docker stop|joomla|mirror|watch |while ' | grep -v grep || echo "(ninguno)" +echo "=== docker context/info ===" +docker version --format 'Server={{.Server.Version}} OS={{.Server.Os}}' 2>/dev/null +echo "=== eventos: arranco el contenedor y escucho 60s ===" +timeout 65 docker events --filter container=joomla-mirror-web --format '{{.Time}} {{.Action}} from={{index .Actor.Attributes "execID"}}' & +EVPID=$! +sleep 1 +docker start joomla-mirror-web >/dev/null +wait $EVPID +echo "=== estado final ===" +docker inspect joomla-mirror-web --format 'Status={{.State.Status}} Exit={{.State.ExitCode}}' diff --git a/mirror-antiguo/scripts/03-probe-urls.sh b/mirror-antiguo/scripts/03-probe-urls.sh new file mode 100644 index 0000000..0a46996 --- /dev/null +++ b/mirror-antiguo/scripts/03-probe-urls.sh @@ -0,0 +1,18 @@ +#!/bin/bash +set -u +H='Host: antiguo.feadulta.com' +for u in \ + "/" \ + "/es/" \ + "/carta/estasemana.html" \ + "/es/carta/estasemana.html" \ + "/buscadoravanzado/item/9-experiencia-pascual.html" \ + "/es/buscadoravanzado/item/9-experiencia-pascual.html" \ + "/20-sincategoria/10-domingo.html" \ + "/es/20-sincategoria/10-domingo.html" \ + ; do + code=$(curl -s -o /dev/null -w '%{http_code}' -H "$H" "http://127.0.0.1:8086$u") + loc=$(curl -s -o /dev/null -w '%{redirect_url}' -H "$H" "http://127.0.0.1:8086$u") + size=$(curl -s -o /dev/null -w '%{size_download}' -H "$H" "http://127.0.0.1:8086$u") + printf '%-60s %s %8s %s\n' "$u" "$code" "$size" "$loc" +done diff --git a/mirror-antiguo/scripts/10-build-inventory.sh b/mirror-antiguo/scripts/10-build-inventory.sh new file mode 100644 index 0000000..a3ac1b6 --- /dev/null +++ b/mirror-antiguo/scripts/10-build-inventory.sh @@ -0,0 +1,23 @@ +#!/bin/bash +# Fase 1 - Inventario de URLs generado por el propio router de Joomla (BD local joomla_mirror) +set -euo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +INV=$BASE/inventory +mkdir -p "$INV" +H='Host: antiguo.feadulta.com' +U='http://127.0.0.1:8086/_genurls.php' + +for s in menu catcontent content k2; do + echo "-> $s" + curl -s --max-time 900 -H "$H" "$U?set=$s" > "$INV/urls-$s.txt" + wc -l < "$INV/urls-$s.txt" +done + +cat "$INV"/urls-menu.txt "$INV"/urls-catcontent.txt "$INV"/urls-content.txt "$INV"/urls-k2.txt \ + | grep -E '^http://antiguo\.feadulta\.com/' \ + | sort -u > "$INV/urls-input.txt" + +echo "=== TOTAL unico ===" +wc -l < "$INV/urls-input.txt" +echo "=== reparto por prefijo ===" +sed 's#^http://antiguo.feadulta.com/es/##' "$INV/urls-input.txt" | cut -d/ -f1 | sort | uniq -c | sort -rn | head -30 diff --git a/mirror-antiguo/scripts/11-validate-sample.sh b/mirror-antiguo/scripts/11-validate-sample.sh new file mode 100644 index 0000000..8bc2f57 --- /dev/null +++ b/mirror-antiguo/scripts/11-validate-sample.sh @@ -0,0 +1,27 @@ +#!/bin/bash +# Valida una muestra aleatoria del inventario contra el Joomla local y mide tiempos +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +INV=$BASE/inventory +N=${1:-40} +H='Host: antiguo.feadulta.com' + +tmp=$(mktemp) +shuf -n "$N" "$INV/urls-input.txt" > "$tmp" + +ok=0; bad=0; tot=0 +start=$(date +%s.%N) +while read -r u; do + path=${u#http://antiguo.feadulta.com} + read -r code t size < <(curl -s -o /dev/null -w '%{http_code} %{time_total} %{size_download}' -H "$H" "http://127.0.0.1:8086$path"; echo) + tot=$(echo "$tot + $t" | bc) + if [ "$code" = "200" ]; then ok=$((ok+1)); else bad=$((bad+1)); printf 'FALLO %s %s %s\n' "$code" "$t" "$path"; fi +done < "$tmp" +end=$(date +%s.%N) + +echo "---" +echo "muestra=$N 200=$ok no200=$bad" +echo "tiempo medio por peticion: $(echo "scale=3; $tot / $N" | bc) s" +echo "wall: $(echo "scale=1; $end - $start" | bc) s" +echo "estimacion 25437 URLs a 1 hilo: $(echo "scale=1; $tot / $N * 25437 / 3600" | bc) h" +rm -f "$tmp" diff --git a/mirror-antiguo/scripts/12-net-check.sh b/mirror-antiguo/scripts/12-net-check.sh new file mode 100644 index 0000000..0c69dae --- /dev/null +++ b/mirror-antiguo/scripts/12-net-check.sh @@ -0,0 +1,8 @@ +#!/bin/bash +set -uo pipefail +IP=$(docker inspect joomla-mirror-web -f '{{range $k,$v := .NetworkSettings.Networks}}{{$v.IPAddress}}{{end}}') +echo "IP contenedor: $IP" +echo -n "acceso directo host->contenedor:80 " +curl -s -o /dev/null -w '%{http_code}\n' -H 'Host: antiguo.feadulta.com' "http://$IP/es/" +echo -n "wget nativo: "; which wget && wget --version | head -1 +grep -q 'antiguo.feadulta.com' /etc/hosts && echo "hosts: ya presente" || echo "hosts: falta entrada" diff --git a/mirror-antiguo/scripts/19-smoke.sh b/mirror-antiguo/scripts/19-smoke.sh new file mode 100644 index 0000000..e9c6e75 --- /dev/null +++ b/mirror-antiguo/scripts/19-smoke.sh @@ -0,0 +1,28 @@ +#!/bin/bash +# Lote de prueba de 200 URLs para validar la invocacion de wget y el arbol resultante +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +SM=$BASE/smoke +rm -rf "$SM"; mkdir -p "$SM/raw" +shuf -n 200 "$BASE/inventory/urls-input.txt" > "$SM/urls.txt" +# aseguramos la portada y una carta +echo 'http://antiguo.feadulta.com/es/' >> "$SM/urls.txt" + +t0=$(date +%s) +wget --input-file="$SM/urls.txt" \ + --force-directories --directory-prefix="$SM/raw" \ + --adjust-extension --no-verbose -e robots=off \ + --user-agent='feadulta-archiver/1.0 (+incident-183; mirror local)' \ + --wait=0.15 --tries=3 --timeout=45 --waitretry=5 \ + --reject-regex='(\?|&)(start|limitstart|limit|print|tmpl|format|searchword|task|orderby|filter|catid|month|year)=' \ + --output-file="$SM/wget.log" +t1=$(date +%s) + +echo "=== tiempo: $((t1-t0))s para $(wc -l < "$SM/urls.txt") URLs ===" +echo "=== ficheros ==="; find "$SM/raw" -type f | wc -l +echo "=== arbol (muestra) ==="; find "$SM/raw" -type f | head -8 +echo "=== errores en el log ==="; grep -icE 'error|failed' "$SM/wget.log" || true +grep -iE 'error|failed' "$SM/wget.log" | head -10 +echo "=== tamano ==="; du -sh "$SM/raw" +echo "=== 500 en apache durante el lote ===" +docker logs --since "${t0}" joomla-mirror-web 2>&1 | grep -c '" 500 ' || true diff --git a/mirror-antiguo/scripts/19b-smoke-check.sh b/mirror-antiguo/scripts/19b-smoke-check.sh new file mode 100644 index 0000000..48113d7 --- /dev/null +++ b/mirror-antiguo/scripts/19b-smoke-check.sh @@ -0,0 +1,15 @@ +#!/bin/bash +set -uo pipefail +SM=/home/rafa/joomla-migration/mirror-antiguo/smoke/raw +echo "=== titulos capturados (10) ===" +find "$SM" -name '*.html' | head -10 | while read -r f; do + t=$(grep -o '[^<]*' "$f" | head -1 | sed 's/<[^>]*>//g') + printf '%-70s %s\n' "$(basename "$f")" "$t" +done +echo +echo "=== paginas sospechosas (challenge/error) ===" +grep -rli 'Attention Required\|Just a moment\|Not Acceptable\|mod_security\|Error 500' "$SM" | wc -l +echo "=== ficheros con PHP embebido ===" +grep -rl '/dev/null 2>&1 \ + || echo "AVISO: no pude escribir /etc/hosts (hazlo como root)" +fi +grep 'antiguo.feadulta.com' /etc/hosts || true + +# --- parar contenedores no implicados (autorizado por Rafa) --- +KEEP='joomla-mirror-web|joomla-mysql|gitea|hub-proxy|beszel-agent' +STOPPED=$BASE/stopped-containers.txt +if [ ! -s "$STOPPED" ]; then + docker ps --format '{{.Names}}' | grep -vE "^($KEEP)$" > "$STOPPED" + echo "--- parando ---"; cat "$STOPPED" + xargs -r -a "$STOPPED" docker stop >/dev/null +fi +echo "--- en marcha ahora ---" +docker ps --format '{{.Names}}' | tr '\n' ' '; echo + +# --- estructura de la corrida --- +RUN=$(date -u +%Y%m%dT%H%M%SZ) +DIR=$BASE/runs/$RUN +mkdir -p "$DIR/raw" "$DIR/chunks" "$DIR/logs" +echo "$RUN" > "$BASE/CURRENT_RUN" + +split -n l/$WORKERS -d --additional-suffix=.txt "$INV/urls-input.txt" "$DIR/chunks/urls-" +wc -l "$DIR/chunks"/*.txt + +cat > "$DIR/meta.json" < "$DIR/logs/passA.start" + +for c in chunks/urls-*.txt; do + n=$(basename "$c" .txt) + wget \ + --input-file="$c" \ + --force-directories --directory-prefix="$DIR/raw" \ + --adjust-extension \ + --no-verbose \ + -e robots=off \ + --user-agent="$UA" \ + --wait=0.15 --tries=3 --timeout=45 --waitretry=5 \ + --reject-regex="$REJECT" \ + --no-check-certificate \ + --output-file="$DIR/logs/wget-$n.log" & + echo "$!" >> "$DIR/logs/passA.pids" +done + +wait +date -u +%Y-%m-%dT%H:%M:%SZ > "$DIR/logs/passA.end" +echo "PASE A TERMINADO" +find "$DIR/raw" -type f | wc -l +du -sh "$DIR/raw" diff --git a/mirror-antiguo/scripts/22-monitor.sh b/mirror-antiguo/scripts/22-monitor.sh new file mode 100644 index 0000000..ba1eec9 --- /dev/null +++ b/mirror-antiguo/scripts/22-monitor.sh @@ -0,0 +1,40 @@ +#!/bin/bash +# Vigilante del crawl: ficheros, 500 del servidor, RAM. Aborta si el servidor se degrada. +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +DIR=$BASE/runs/$RUN +ST=$DIR/logs/monitor.log +MAX500=${MAX500:-100} +MINMEM_MB=${MINMEM_MB:-800} +T0=$(date +%s) + +echo "monitor arrancado $(date -u +%FT%TZ) run=$RUN max500=$MAX500" > "$ST" + +while true; do + sleep 60 + pgrep -f 'wget --input-file=' >/dev/null || { echo "$(date -u +%FT%TZ) crawl terminado, monitor sale" >> "$ST"; break; } + + files=$(find "$DIR/raw" -type f 2>/dev/null | wc -l) + size=$(du -sm "$DIR/raw" 2>/dev/null | cut -f1) + e500=$(docker logs --since "$T0" joomla-mirror-web 2>&1 | grep -c '" 500 ') + e408=$(docker logs --since "$T0" joomla-mirror-web 2>&1 | grep -c '" 40[38] ') + mem=$(free -m | awk '/^Mem:/{print $7}') + cmem=$(docker stats --no-stream --format '{{.MemUsage}}' joomla-mirror-web 2>/dev/null) + el=$(( $(date +%s) - T0 )) + pct=$(awk -v f="$files" 'BEGIN{printf "%.1f", f*100/25437}') + + printf '%s t=%ss ficheros=%s (%s%%) %sMB 500=%s 40x=%s ram_libre=%sMB cont=%s\n' \ + "$(date -u +%FT%TZ)" "$el" "$files" "$pct" "$size" "$e500" "$e408" "$mem" "$cmem" >> "$ST" + + if [ "$e500" -gt "$MAX500" ]; then + echo "!!! ABORTO: $e500 respuestas 500 (umbral $MAX500)" >> "$ST" + pkill -f 'wget --input-file=' + break + fi + if [ "$mem" -lt "$MINMEM_MB" ]; then + echo "!!! ABORTO: RAM libre ${mem}MB por debajo de ${MINMEM_MB}MB" >> "$ST" + pkill -f 'wget --input-file=' + break + fi +done diff --git a/mirror-antiguo/scripts/23-estado.sh b/mirror-antiguo/scripts/23-estado.sh new file mode 100644 index 0000000..384b302 --- /dev/null +++ b/mirror-antiguo/scripts/23-estado.sh @@ -0,0 +1,14 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +DIR=$BASE/runs/$RUN +echo "ahora: $(date -u +%FT%TZ)" +echo "wget vivos: $(pgrep -cf 'wget --input-file=')" +echo "monitor vivo: $(pgrep -cf '22-monitor.sh')" +echo "ficheros: $(find "$DIR/raw" -type f 2>/dev/null | wc -l)" +du -sh "$DIR/raw" 2>/dev/null +echo "--- monitor.log ---"; tail -6 "$DIR/logs/monitor.log" +echo "--- ultimo log de wget activo ---" +ls -t "$DIR/logs"/wget-*.log | head -1 | xargs tail -2 +free -m | head -2 diff --git a/mirror-antiguo/scripts/24-errores-passA.sh b/mirror-antiguo/scripts/24-errores-passA.sh new file mode 100644 index 0000000..a6909f6 --- /dev/null +++ b/mirror-antiguo/scripts/24-errores-passA.sh @@ -0,0 +1,15 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +DIR=$BASE/runs/$RUN +echo "=== inicio/fin ==="; cat "$DIR/logs/passA.start" "$DIR/logs/passA.end" +echo "=== codigos de error de wget ===" +grep -hoE 'ERROR [0-9]+[^.]*' "$DIR/logs"/wget-urls-*.log | sort | uniq -c +echo "=== lineas de fallo ===" +grep -hE 'ERROR [0-9]|unable to resolve|Giving up|failed:' "$DIR/logs"/wget-urls-*.log | head -20 +echo "=== descargadas segun log ===" +grep -hc '^2026' "$DIR/logs"/wget-urls-*.log | paste -sd+ | bc +echo "=== 500/40x en apache durante el pase A ===" +docker logs --since 2026-07-29T22:47:00Z joomla-mirror-web 2>&1 | grep -c '" 500 ' +docker logs --since 2026-07-29T22:47:00Z joomla-mirror-web 2>&1 | grep -c '" 40[0-9] ' diff --git a/mirror-antiguo/scripts/25-dirs-estaticos.sh b/mirror-antiguo/scripts/25-dirs-estaticos.sh new file mode 100644 index 0000000..1714bc7 --- /dev/null +++ b/mirror-antiguo/scripts/25-dirs-estaticos.sh @@ -0,0 +1,16 @@ +#!/bin/bash +set -uo pipefail +cd /home/rafa/joomla-migration/mirror-antiguo/restore/web +for d in anterior ediciones music sport docs images media templates; do + if [ -d "$d" ]; then + printf '%-12s ficheros=%-8s php=%-6s html=%-7s %s\n' "$d" \ + "$(find "$d" -type f | wc -l)" \ + "$(find "$d" -iname '*.php' | wc -l)" \ + "$(find "$d" -iname '*.htm*' | wc -l)" \ + "$(du -sh "$d" | cut -f1)" + else + printf '%-12s (no existe)\n' "$d" + fi +done +echo "--- indice de /anterior ---" +ls anterior 2>/dev/null | head -20 diff --git a/mirror-antiguo/scripts/26-probe-anterior.sh b/mirror-antiguo/scripts/26-probe-anterior.sh new file mode 100644 index 0000000..514b6d9 --- /dev/null +++ b/mirror-antiguo/scripts/26-probe-anterior.sh @@ -0,0 +1,11 @@ +#!/bin/bash +set -uo pipefail +H='Host: antiguo.feadulta.com' +for u in /anterior /anterior/ /anterior/index.html /anterior/index.htm /docs/ /music/; do + printf '%-26s ' "$u" + curl -s -o /dev/null -w 'code=%{http_code} size=%{size_download} loc=%{redirect_url}\n' -H "$H" "http://127.0.0.1:8086$u" +done +echo "--- ficheros indice en /anterior ---" +ls /home/rafa/joomla-migration/mirror-antiguo/restore/web/anterior/ | grep -iE '^(index|default|home)\.' | head +echo "--- php dentro de /anterior ---" +find /home/rafa/joomla-migration/mirror-antiguo/restore/web/anterior -iname '*.php' diff --git a/mirror-antiguo/scripts/30-extract-links.py b/mirror-antiguo/scripts/30-extract-links.py new file mode 100644 index 0000000..8350aeb --- /dev/null +++ b/mirror-antiguo/scripts/30-extract-links.py @@ -0,0 +1,94 @@ +#!/usr/bin/env python3 +"""Pase B (1/2): extrae de las paginas capturadas los enlaces internos. + +Salidas en el directorio de la corrida: + assets-input.txt URLs internas a recursos NO html (css, js, img, pdf, mp3, doc...) + missing-pages.txt paginas .html internas enlazadas que NO estan en el inventario + external-hosts.txt hosts externos referenciados, con recuento +""" +import os, re, sys, html +from collections import Counter +from urllib.parse import urljoin, urlsplit, urlunsplit + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +DIR = os.path.join(BASE, "runs", RUN) +RAW = os.path.join(DIR, "raw", "antiguo.feadulta.com") +HOST = "antiguo.feadulta.com" + +ATTR = re.compile(rb'(?:href|src|data-src|poster)\s*=\s*["\']([^"\'>]+)["\']', re.I) +CSSURL = re.compile(rb'url\(\s*["\']?([^"\')]+)["\']?\s*\)', re.I) + +SKIP_SCHEMES = ("mailto:", "javascript:", "tel:", "data:", "#", "skype:", "whatsapp:") + +inventory = set() +for line in open(os.path.join(BASE, "inventory", "urls-input.txt")): + inventory.add(line.strip()) + +assets, pages, ext = set(), set(), Counter() +nfiles = 0 + +def norm(u, page_url): + u = html.unescape(u.strip()) + if not u or u.startswith(SKIP_SCHEMES): + return None + absu = urljoin(page_url, u) + p = urlsplit(absu) + if p.scheme not in ("http", "https"): + return None + if p.netloc.split(":")[0] != HOST: + ext[p.netloc] += 1 + return None + # sin fragmento; conservamos query (rara en assets) + return urlunsplit(("http", HOST, p.path, p.query, "")) + +for root, _dirs, files in os.walk(RAW): + for fn in files: + path = os.path.join(root, fn) + rel = os.path.relpath(path, RAW) + page_url = "http://%s/%s" % (HOST, rel.replace(os.sep, "/")) + if not fn.lower().endswith((".html", ".htm")): + continue + nfiles += 1 + try: + data = open(path, "rb").read() + except OSError: + continue + found = ATTR.findall(data) + CSSURL.findall(data) + for raw in found: + try: + u = norm(raw.decode("utf-8", "replace"), page_url) + except ValueError: + continue + if not u: + continue + tail = urlsplit(u).path.lower() + if tail.endswith((".html", ".htm")) or tail.endswith("/"): + if u not in inventory: + pages.add(u) + else: + assets.add(u) + +def dump(name, it): + p = os.path.join(DIR, name) + with open(p, "w") as f: + for x in sorted(it): + f.write(x + "\n") + return p, len(it) + +print("paginas HTML analizadas:", nfiles) +for n, c in (dump("assets-input.txt", assets), dump("missing-pages.txt", pages)): + print(n, c) +with open(os.path.join(DIR, "external-hosts.txt"), "w") as f: + for h, c in ext.most_common(): + f.write("%7d %s\n" % (c, h)) +print("hosts externos distintos:", len(ext)) + +# reparto de assets por extension +c = Counter() +for u in assets: + e = os.path.splitext(urlsplit(u).path)[1].lower() or "(sin ext)" + c[e] += 1 +print("--- assets por extension ---") +for e, n in c.most_common(25): + print("%7d %s" % (n, e)) diff --git a/mirror-antiguo/scripts/31-fetch-assets.sh b/mirror-antiguo/scripts/31-fetch-assets.sh new file mode 100644 index 0000000..a4ff6ba --- /dev/null +++ b/mirror-antiguo/scripts/31-fetch-assets.sh @@ -0,0 +1,30 @@ +#!/bin/bash +# Pase B (2/2): descarga los recursos (css/js/img/pdf/mp3...) referenciados por las paginas +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +DIR=$BASE/runs/$RUN +W=${WORKERS:-3} + +[ -s "$DIR/assets-input.txt" ] || { echo "no hay assets-input.txt"; exit 1; } +mkdir -p "$DIR/chunks-assets" "$DIR/logs" +rm -f "$DIR/chunks-assets"/*.txt +split -n l/$W -d --additional-suffix=.txt "$DIR/assets-input.txt" "$DIR/chunks-assets/a-" + +date -u +%FT%TZ > "$DIR/logs/passB.start" +for c in "$DIR/chunks-assets"/a-*.txt; do + n=$(basename "$c" .txt) + wget --input-file="$c" \ + --force-directories --directory-prefix="$DIR/raw" \ + --no-verbose -e robots=off --no-clobber \ + --user-agent='feadulta-archiver/1.0 (+incident-183; mirror local)' \ + --wait=0.05 --tries=2 --timeout=45 --waitretry=3 \ + --output-file="$DIR/logs/wget-$n.log" & +done +wait +date -u +%FT%TZ > "$DIR/logs/passB.end" +echo "PASE B TERMINADO" +find "$DIR/raw" -type f | wc -l +du -sh "$DIR/raw" +echo "=== errores ===" +grep -hoE 'ERROR [0-9]+' "$DIR/logs"/wget-a-*.log | sort | uniq -c diff --git a/mirror-antiguo/scripts/32-analiza-missing.sh b/mirror-antiguo/scripts/32-analiza-missing.sh new file mode 100644 index 0000000..fbd941c --- /dev/null +++ b/mirror-antiguo/scripts/32-analiza-missing.sh @@ -0,0 +1,18 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +DIR=$BASE/runs/$RUN +M=$DIR/missing-pages.txt +echo "total: $(wc -l < "$M")" +echo +echo "=== con query string ===" +grep -c '?' "$M" +echo "=== primer segmento de ruta ===" +sed 's#^http://antiguo.feadulta.com/##' "$M" | cut -d/ -f1 | sort | uniq -c | sort -rn | head -20 +echo +echo "=== segundo segmento bajo /es/ ===" +grep '^http://antiguo.feadulta.com/es/' "$M" | sed 's#^http://antiguo.feadulta.com/es/##' | cut -d/ -f1 | sort | uniq -c | sort -rn | head -25 +echo +echo "=== muestra aleatoria de 25 ===" +shuf -n 25 "$M" diff --git a/mirror-antiguo/scripts/33-missing-sin-query.sh b/mirror-antiguo/scripts/33-missing-sin-query.sh new file mode 100644 index 0000000..96b297a --- /dev/null +++ b/mirror-antiguo/scripts/33-missing-sin-query.sh @@ -0,0 +1,14 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +DIR=$BASE/runs/$RUN +grep -v '?' "$DIR/missing-pages.txt" > "$DIR/missing-pages-sinquery.txt" +M=$DIR/missing-pages-sinquery.txt +echo "sin query: $(wc -l < "$M")" +echo +echo "=== por prefijo (2 segmentos) ===" +sed 's#^http://antiguo.feadulta.com/##' "$M" | cut -d/ -f1,2 | sort | uniq -c | sort -rn | head -25 +echo +echo "=== muestra de 30 (fuera de /anterior) ===" +grep -v '/anterior/' "$M" | shuf -n 30 diff --git a/mirror-antiguo/scripts/34-clasifica-missing.py b/mirror-antiguo/scripts/34-clasifica-missing.py new file mode 100644 index 0000000..f1fcc7c --- /dev/null +++ b/mirror-antiguo/scripts/34-clasifica-missing.py @@ -0,0 +1,64 @@ +#!/usr/bin/env python3 +"""Clasifica missing-pages-sinquery.txt: separa lo que es ruido/codificacion de los huecos reales.""" +import os, re +from collections import Counter +from urllib.parse import unquote, urlsplit + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +DIR = os.path.join(BASE, "runs", RUN) +RAW = os.path.join(DIR, "raw", "antiguo.feadulta.com") + +inv = set(l.strip() for l in open(os.path.join(BASE, "inventory", "urls-input.txt"))) +inv_dec = set(unquote(u) for u in inv) + +cats = Counter() +real = [] +for line in open(os.path.join(DIR, "missing-pages-sinquery.txt")): + u = line.strip() + d = unquote(u) + if d in inv_dec: + cats["ya_en_inventario (solo difiere la codificacion %XX)"] += 1 + continue + # ¿existe ya el fichero en disco? + p = urlsplit(d).path + fp = os.path.join(RAW, p.lstrip("/")) + if p.endswith("/"): + fp = os.path.join(fp, "index.html") + if os.path.exists(fp): + cats["ya_capturado en disco"] += 1 + continue + + if "/itemlist/user/" in d: + cats["K2 pagina de autor (itemlist/user)"] += 1; real.append(u) + elif "/itemlist/tag/" in d: + cats["K2 pagina de etiqueta (itemlist/tag)"] += 1; real.append(u) + elif "/itemlist/date/" in d or "/itemlist/category" in d: + cats["K2 listado (fecha/categoria)"] += 1; real.append(u) + elif d.startswith("http://antiguo.feadulta.com/anterior/"): + cats["/anterior (web estatica antigua)"] += 1; real.append(u) + elif d.startswith("http://antiguo.feadulta.com/ediciones/"): + cats["/ediciones"] += 1; real.append(u) + elif "/index.php/" in d: + cats["enlace no-SEF (index.php/...)"] += 1; real.append(u) + elif re.search(r"/ES/|/BUSCADORAVANZADO/", d): + cats["enlace roto por mayusculas"] += 1 + else: + cats["OTROS - revisar"] += 1; real.append(u) + +for k, v in cats.most_common(): + print("%7d %s" % (v, k)) +print() +out = os.path.join(DIR, "missing-real.txt") +with open(out, "w") as f: + for u in sorted(set(real)): + f.write(u + "\n") +print("candidatos reales ->", out, len(set(real))) + +print("\n--- muestra de OTROS ---") +n = 0 +for u in sorted(set(real)): + d = unquote(u) + if not any(s in d for s in ("/itemlist/", "/anterior/", "/ediciones/", "/index.php/")): + print(" ", u); n += 1 + if n >= 20: break diff --git a/mirror-antiguo/scripts/35-pase-c-huecos.sh b/mirror-antiguo/scripts/35-pase-c-huecos.sh new file mode 100644 index 0000000..9118954 --- /dev/null +++ b/mirror-antiguo/scripts/35-pase-c-huecos.sh @@ -0,0 +1,51 @@ +#!/bin/bash +# Pase C: captura iterativa de las paginas internas enlazadas que no estaban en el inventario +# (rutas alternativas de menu, paginas de autor de K2, /anterior, ...). +# Itera hasta que no aparezcan URLs nuevas o hasta MAXIT vueltas. +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +DIR=$BASE/runs/$RUN +MAXIT=${MAXIT:-6} +W=${WORKERS:-4} +REJECT='(\?|&)(start|limitstart|limit|print|tmpl|format|searchword|task|orderby|filter|catid|month|year)=' + +for it in $(seq 1 $MAXIT); do + IN=$DIR/passC-$it-input.txt + if [ "$it" = "1" ]; then + cp "$DIR/missing-real.txt" "$IN" + else + # recalcular huecos con lo capturado hasta ahora + python3 "$BASE/scripts/30-extract-links.py" > "$DIR/logs/extract-$it.log" 2>&1 + grep -v '?' "$DIR/missing-pages.txt" > "$DIR/missing-pages-sinquery.txt" + python3 "$BASE/scripts/34-clasifica-missing.py" > "$DIR/logs/clasifica-$it.log" 2>&1 + cp "$DIR/missing-real.txt" "$IN" + fi + + n=$(wc -l < "$IN") + echo "=== iteracion $it: $n URLs candidatas ===" + [ "$n" -eq 0 ] && { echo "no quedan huecos"; break; } + + mkdir -p "$DIR/chunks-c" + rm -f "$DIR/chunks-c"/*.txt + split -n l/$W -d --additional-suffix=.txt "$IN" "$DIR/chunks-c/c$it-" + for c in "$DIR/chunks-c"/c$it-*.txt; do + [ -s "$c" ] || continue + b=$(basename "$c" .txt) + wget --input-file="$c" \ + --force-directories --directory-prefix="$DIR/raw" \ + --adjust-extension --no-verbose --no-clobber -e robots=off \ + --user-agent='feadulta-archiver/1.0 (+incident-183; mirror local)' \ + --wait=0.1 --tries=2 --timeout=45 --waitretry=3 \ + --reject-regex="$REJECT" \ + --output-file="$DIR/logs/wget-$b.log" & + done + wait + echo " ficheros ahora: $(find "$DIR/raw" -type f | wc -l)" + echo " errores: $(grep -hoE 'ERROR [0-9]+' "$DIR/logs"/wget-c$it-*.log | sort | uniq -c | tr '\n' ' ')" +done + +date -u +%FT%TZ > "$DIR/logs/passC.end" +echo "PASE C TERMINADO" +find "$DIR/raw" -type f | wc -l +du -sh "$DIR/raw" diff --git a/mirror-antiguo/scripts/36-estado-passB.sh b/mirror-antiguo/scripts/36-estado-passB.sh new file mode 100644 index 0000000..23f3481 --- /dev/null +++ b/mirror-antiguo/scripts/36-estado-passB.sh @@ -0,0 +1,13 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +echo "inicio/fin pase B:"; cat "$D/logs/passB.start" "$D/logs/passB.end" 2>/dev/null || echo "(sin marcas)" +echo "assets pedidos: $(wc -l < "$D/assets-input.txt")" +echo "errores:"; grep -hoE 'ERROR [0-9]+' "$D/logs"/wget-a-*.log 2>/dev/null | sort | uniq -c +echo "descargados segun log: $(grep -hc '^2026' "$D/logs"/wget-a-*.log 2>/dev/null | paste -sd+ | bc)" +echo "ficheros totales: $(find "$D/raw" -type f | wc -l)" +du -sh "$D/raw" +echo "--- reparto por tipo en raw ---" +find "$D/raw" -type f | sed 's#.*\.##' | tr 'A-Z' 'a-z' | sort | uniq -c | sort -rn | head -15 diff --git a/mirror-antiguo/scripts/37-pase-d-anterior.sh b/mirror-antiguo/scripts/37-pase-d-anterior.sh new file mode 100644 index 0000000..36c63d9 --- /dev/null +++ b/mirror-antiguo/scripts/37-pase-d-anterior.sh @@ -0,0 +1,26 @@ +#!/bin/bash +# Pase D: /anterior — la web estatica anterior a Joomla, enlazada desde el menu principal. +# Es HTML plano servido por Apache (solo 1 .php en 34.037 ficheros): la recursion aqui es finita +# y no pasa por PHP, asi que no reproduce la trampa de paginacion de K2. +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +DIR=$BASE/runs/$RUN + +date -u +%FT%TZ > "$DIR/logs/passD.start" +wget \ + --recursive --level=inf --no-parent \ + --force-directories --directory-prefix="$DIR/raw" \ + --adjust-extension --no-verbose --no-clobber -e robots=off \ + --domains=antiguo.feadulta.com --span-hosts=off \ + --user-agent='feadulta-archiver/1.0 (+incident-183; mirror local)' \ + --wait=0.05 --tries=2 --timeout=45 --waitretry=3 \ + --reject-regex='(\?|&)(C|O|start|limitstart|limit|print|tmpl|format|searchword|task|orderby|filter)=' \ + --output-file="$DIR/logs/wget-anterior.log" \ + http://antiguo.feadulta.com/anterior/ +date -u +%FT%TZ > "$DIR/logs/passD.end" + +echo "PASE D TERMINADO" +find "$DIR/raw/antiguo.feadulta.com/anterior" -type f 2>/dev/null | wc -l +du -sh "$DIR/raw/antiguo.feadulta.com/anterior" 2>/dev/null +grep -hoE 'ERROR [0-9]+' "$DIR/logs/wget-anterior.log" | sort | uniq -c diff --git a/mirror-antiguo/scripts/38-diag-passB.sh b/mirror-antiguo/scripts/38-diag-passB.sh new file mode 100644 index 0000000..f4dd3ae --- /dev/null +++ b/mirror-antiguo/scripts/38-diag-passB.sh @@ -0,0 +1,16 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +echo "=== procesos wget ===" +ps -eo pid,etimes,args | grep '[w]get --input-file' | sed 's/\(.\{160\}\).*/\1/' +echo +echo "=== ultimas 3 lineas de cada log del pase B ===" +for f in "$D/logs"/wget-a-*.log; do echo "--- $f"; tail -3 "$f"; done +echo +echo "=== 24 lineas mas recientes de apache ===" +docker logs --tail 8 joomla-mirror-web 2>&1 | sed 's/\(.\{150\}\).*/\1/' +echo +echo "=== conteo por chunk ===" +for f in "$D/chunks-assets"/a-*.txt; do echo "$f: $(wc -l < "$f")"; done diff --git a/mirror-antiguo/scripts/39-diag-interrogantes.sh b/mirror-antiguo/scripts/39-diag-interrogantes.sh new file mode 100644 index 0000000..2d0f3bc --- /dev/null +++ b/mirror-antiguo/scripts/39-diag-interrogantes.sh @@ -0,0 +1,25 @@ +#!/bin/bash +# Cuantifica el problema de los alias que llevan '?' literal dentro de la URL +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +echo "=== URLs del inventario con '?' literal ===" +grep -c '?' "$BASE/inventory/urls-input.txt" +echo "=== muestra ===" +grep '?' "$BASE/inventory/urls-input.txt" | head -5 +echo +echo "=== assets-input con '?' ===" +grep -c '?' "$D/assets-input.txt" +echo "=== assets-input SIN '?' (assets de verdad) ===" +grep -vc '?' "$D/assets-input.txt" +echo +echo "=== ficheros en raw sin extension ===" +find "$D/raw" -type f ! -name '*.*' | wc -l +echo "=== ficheros en raw con '?' en el nombre ===" +find "$D/raw" -type f -name '*[?]*' | wc -l +echo "=== muestra ===" +find "$D/raw" -type f -name '*[?]*' | head -3 +echo +echo "=== items K2 con '?' en el alias (BD) ===" +grep -c 'buscadoravanzado' "$BASE/inventory/urls-k2.txt" diff --git a/mirror-antiguo/scripts/40-manifest-scan.sh b/mirror-antiguo/scripts/40-manifest-scan.sh new file mode 100644 index 0000000..c0d53f3 --- /dev/null +++ b/mirror-antiguo/scripts/40-manifest-scan.sh @@ -0,0 +1,40 @@ +#!/bin/bash +# Fase 3: manifiestos sha256 + escaneo de seguridad del propio mirror (§4.3 del plan) +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +DIR=$BASE/runs/$RUN +cd "$DIR" + +echo "=== manifiesto raw ===" +( cd raw && find . -type f -print0 | sort -z | xargs -0 sha256sum ) > MANIFEST-raw.sha256 +wc -l < MANIFEST-raw.sha256 +du -sh raw + +echo +echo "=== 1. ficheros con PHP ejecutable ===" +grep -rl ' scan-php.txt 2>/dev/null +wc -l < scan-php.txt + +echo "=== 2. patrones tipicos de inyeccion ===" +grep -rlE 'eval\(|atob\(|document\.write\(unescape|fromCharCode' raw > scan-suspicious.txt 2>/dev/null +wc -l < scan-suspicious.txt + +echo "=== 3. paginas de challenge/error congeladas ===" +grep -rli 'Attention Required\|Just a moment\|Not Acceptable\|mod_security\|Internal Server Error' raw > scan-garbage.txt 2>/dev/null +wc -l < scan-garbage.txt + +echo "=== 4. hosts externos en script/iframe ===" +grep -rhoE '<(script|iframe)[^>]+src="https?://[^"/]+' raw \ + | grep -oE 'https?://[^"/]+' | sort | uniq -c | sort -rn > scan-external-script-hosts.txt +head -25 scan-external-script-hosts.txt + +echo +echo "=== 5. cobertura frente al inventario ===" +find raw/antiguo.feadulta.com -type f -name '*.html' \ + | sed 's#^raw/antiguo.feadulta.com#http://antiguo.feadulta.com#' | sort -u > captured-pages.txt +comm -23 <(sort -u "$BASE/inventory/urls-input.txt" | sed 's#/es/$#/es/index.html#') captured-pages.txt > coverage-missing.txt +echo "inventario: $(wc -l < "$BASE/inventory/urls-input.txt")" +echo "capturadas: $(wc -l < captured-pages.txt)" +echo "sin capturar (aprox): $(wc -l < coverage-missing.txt)" +head -20 coverage-missing.txt diff --git a/mirror-antiguo/scripts/41-menu-paths.sh b/mirror-antiguo/scripts/41-menu-paths.sh new file mode 100644 index 0000000..d5afc58 --- /dev/null +++ b/mirror-antiguo/scripts/41-menu-paths.sh @@ -0,0 +1,13 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +INV=$BASE/inventory +curl -s -H 'Host: antiguo.feadulta.com' 'http://127.0.0.1:8086/_inv/menupaths.php' | sort -u > "$INV/urls-menu-path.txt" +echo "menu por path: $(wc -l < "$INV/urls-menu-path.txt")" +echo "menu via JRoute con ?Itemid=: $(grep -c 'Itemid=' "$INV/urls-menu.txt")" +echo +echo "=== comprobacion de 12 al azar ===" +shuf -n 12 "$INV/urls-menu-path.txt" | while read -r u; do + p=${u#http://antiguo.feadulta.com} + printf '%-60s %s\n' "$p" "$(curl -s -o /dev/null -w '%{http_code}' -H 'Host: antiguo.feadulta.com' "http://127.0.0.1:8086$p")" +done diff --git a/mirror-antiguo/scripts/42-corta-a00.sh b/mirror-antiguo/scripts/42-corta-a00.sh new file mode 100644 index 0000000..dbff974 --- /dev/null +++ b/mirror-antiguo/scripts/42-corta-a00.sh @@ -0,0 +1,17 @@ +#!/bin/bash +# El lote a-00 del pase B resulto ser casi todo basura: URLs de articulo cuyo alias lleva un '?' +# literal (mal clasificadas como assets) y sus vistas de impresion. Los assets de verdad estaban en +# a-01 y a-02, que ya terminaron. Se corta a-00 y se documenta. +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +echo "=== composicion de a-00 ===" +echo "total: $(wc -l < "$D/chunks-assets/a-00.txt")" +echo "bajo /es/ (articulos, no assets): $(grep -c '/es/' "$D/chunks-assets/a-00.txt")" +echo "assets reales: $(grep -vc '/es/' "$D/chunks-assets/a-00.txt")" +pkill -f 'chunks-assets/a-00.txt' && echo "a-00 detenido" || echo "a-00 ya no corria" +sleep 2 +date -u +%FT%TZ > "$D/logs/passB.end" +echo "=== restos a limpiar (vistas de impresion) ===" +find "$D/raw" -type f -name '*print=1*' | wc -l diff --git a/mirror-antiguo/scripts/43-pendientes.sh b/mirror-antiguo/scripts/43-pendientes.sh new file mode 100644 index 0000000..e46ffc2 --- /dev/null +++ b/mirror-antiguo/scripts/43-pendientes.sh @@ -0,0 +1,46 @@ +#!/bin/bash +# Pase C unificado: conjunto completo de URLs deseadas menos lo que ya esta en disco. +# inventario + rutas de menu por `path` + huecos detectados en el HTML + assets reales +# wget con --no-clobber se salta lo ya descargado, asi que el script es idempotente. +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +W=${WORKERS:-4} +REJECT='(\?|&)(start|limitstart|limit|print|tmpl|format|searchword|task|orderby|filter|catid|month|year)=' + +cat "$BASE/inventory/urls-input.txt" \ + "$BASE/inventory/urls-menu-path.txt" \ + "$D/missing-real.txt" \ + <(grep -v 'antiguo.feadulta.com/es/' "$D/assets-input.txt") \ + | grep -E '^http://antiguo\.feadulta\.com/' \ + | grep -v 'print=1' \ + | grep -v 'tmpl=component' \ + | grep -v 'format=opensearch' \ + | grep -v '/component/mailto/' \ + | sort -u > "$D/passC-input.txt" + +echo "conjunto deseado: $(wc -l < "$D/passC-input.txt")" +echo "ficheros en disco antes: $(find "$D/raw" -type f | wc -l)" + +mkdir -p "$D/chunks-c"; rm -f "$D/chunks-c"/*.txt +split -n l/$W -d --additional-suffix=.txt "$D/passC-input.txt" "$D/chunks-c/c-" + +date -u +%FT%TZ > "$D/logs/passC.start" +for c in "$D/chunks-c"/c-*.txt; do + b=$(basename "$c" .txt) + wget --input-file="$c" \ + --force-directories --directory-prefix="$D/raw" \ + --adjust-extension --no-verbose --no-clobber -e robots=off \ + --user-agent='feadulta-archiver/1.0 (+incident-183; mirror local)' \ + --wait=0.1 --tries=2 --timeout=45 --waitretry=3 \ + --reject-regex="$REJECT" \ + --output-file="$D/logs/wget-$b.log" & +done +wait +date -u +%FT%TZ > "$D/logs/passC.end" + +echo "PASE C TERMINADO" +echo "ficheros en disco despues: $(find "$D/raw" -type f | wc -l)" +du -sh "$D/raw" +grep -hoE 'ERROR [0-9]+' "$D/logs"/wget-c-*.log | sort | uniq -c diff --git a/mirror-antiguo/scripts/44-check-plantilla.sh b/mirror-antiguo/scripts/44-check-plantilla.sh new file mode 100644 index 0000000..363b93a --- /dev/null +++ b/mirror-antiguo/scripts/44-check-plantilla.sh @@ -0,0 +1,19 @@ +#!/bin/bash +# Comprueba que los recursos de plantilla que pide la portada existen en el mirror +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +R=$BASE/runs/$RUN/raw/antiguo.feadulta.com +P=$R/es/index.html +[ -f "$P" ] || { echo "no existe $P"; exit 1; } + +grep -oE '(href|src)="[^"]+\.(css|js|png|jpg|gif|ico)[^"]*"' "$P" \ + | sed 's/^[a-z]*="//; s/"$//' | sort -u > /tmp/portada-assets.txt +echo "recursos referenciados por la portada: $(wc -l < /tmp/portada-assets.txt)" +ok=0; miss=0 +while read -r u; do + p=$(echo "$u" | sed 's#^https\?://antiguo.feadulta.com##; s#^/##; s#?.*##') + case "$u" in http*://*) case "$u" in *antiguo.feadulta.com*) ;; *) continue;; esac;; esac + if [ -f "$R/$p" ]; then ok=$((ok+1)); else miss=$((miss+1)); echo " FALTA: $p"; fi +done < /tmp/portada-assets.txt +echo "presentes=$ok ausentes=$miss" diff --git a/mirror-antiguo/scripts/44b-check-query-assets.sh b/mirror-antiguo/scripts/44b-check-query-assets.sh new file mode 100644 index 0000000..d42db85 --- /dev/null +++ b/mirror-antiguo/scripts/44b-check-query-assets.sh @@ -0,0 +1,13 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +R=$BASE/runs/$RUN/raw/antiguo.feadulta.com +for f in components/com_k2/css/k2.css media/com_jce/site/css/content.min.css \ + media/jui/js/jquery-migrate.min.js media/jui/js/jquery-noconflict.js \ + media/jui/js/jquery.min.js media/k2/assets/js/k2.frontend.js \ + media/system/js/core.js media/system/js/html5fallback.js \ + media/system/js/mootools-core.js media/system/js/mootools-more.js; do + hit=$(ls "$R/$f"* 2>/dev/null | head -1) + printf '%-50s %s\n' "$(basename "$f")" "${hit:-NO ENCONTRADO}" +done diff --git a/mirror-antiguo/scripts/45-normalize-links.py b/mirror-antiguo/scripts/45-normalize-links.py new file mode 100644 index 0000000..a02ac5a --- /dev/null +++ b/mirror-antiguo/scripts/45-normalize-links.py @@ -0,0 +1,110 @@ +#!/usr/bin/env python3 +"""§3.5 del plan: deriva `site/` a partir de `raw/` (que queda intacto). + +- Reescribe los enlaces absolutos a `antiguo.feadulta.com` como raiz-relativos, para que el mirror + funcione bajo cualquier hostname (p.ej. legacy.rafacalvo.nyc). +- Deja intactos los enlaces externos (incluido www.feadulta.com, que ahora es WordPress). +- Neutraliza los formularios que apuntan a endpoints PHP vivos: quedan inertes y con aviso. +- Descarta las vistas de impresion (`?tmpl=component&print=1`), que duplican paginas ya capturadas. +- Deja una copia sin la query en el nombre para los ficheros que wget guardo como `app.js?hash` + o `titulo?.html` (alias con '?' literal): un servidor estatico busca el nombre sin query. + +Uso: 45-normalize-links.py [--no-copy] (--no-copy reaprovecha el site/ existente) +""" +import os, re, json, shutil, sys +from collections import Counter + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +DIR = os.path.join(BASE, "runs", RUN) +RAW, SITE = os.path.join(DIR, "raw"), os.path.join(DIR, "site") + +HOSTABS = re.compile(rb'(?:https?:)?//antiguo\.feadulta\.com', re.I) +FORM = re.compile(rb']*>', re.I) +ACTION = re.compile(rb'action\s*=\s*["\']([^"\']*)["\']', re.I) + +AVISO = (b'
Archivo hist\xc3\xb3rico: este formulario no est\xc3\xa1 ' + b'operativo.
') + +TEXTEXT = (".html", ".htm", ".css", ".js") + + +def basename_sin_query(fn): + """`app.js?hash` -> `app.js`; `titulo?.html` -> `titulo`. Sin '?' devuelve el propio nombre.""" + return fn.split("?", 1)[0] + + +def paso1_nombres_con_query(stats): + """Se ejecuta ANTES de reescribir: descarta impresiones y crea las copias de nombre limpio.""" + for root, _d, files in os.walk(SITE): + for fn in list(files): + if "?" not in fn: + continue + src = os.path.join(root, fn) + if "print=1" in fn: + os.remove(src) + stats["vistas_impresion_descartadas"] += 1 + continue + base = basename_sin_query(fn) + if not base: + continue + dst = os.path.join(root, base) + if not os.path.exists(dst): + shutil.copy2(src, dst) + stats["copias_con_nombre_limpio"] += 1 + + +def paso2_reescribe(stats): + for root, _d, files in os.walk(SITE): + for fn in files: + # la extension se mira sobre el nombre SIN query: `x.html?foo` sigue siendo HTML + base = basename_sin_query(fn).lower() + if not base.endswith(TEXTEXT): + continue + p = os.path.join(root, fn) + try: + data = open(p, "rb").read() + except OSError: + continue + orig = data + + data, n = HOSTABS.subn(b"", data) + stats["enlaces_absolutos_reescritos"] += n + + if base.endswith((".html", ".htm")): + def fix_form(m): + tag = m.group(0) + a = ACTION.search(tag) + if a and b".php" in a.group(1): + stats["formularios_neutralizados"] += 1 + return ACTION.sub(b'action="#" onsubmit="return false"', tag) + AVISO + return tag + data = FORM.sub(fix_form, data) + + if data != orig: + open(p, "wb").write(data) + stats["ficheros_modificados"] += 1 + + +def main(): + if "--no-copy" not in sys.argv: + if os.path.exists(SITE): + print("site/ ya existe, lo borro"); shutil.rmtree(SITE) + print("copiando raw/ -> site/ ...") + shutil.copytree(RAW, SITE) + else: + print("reaprovechando site/ existente") + + stats = Counter() + paso1_nombres_con_query(stats) + paso2_reescribe(stats) + + out = os.path.join(DIR, "link-rewrite.json") + json.dump(dict(stats), open(out, "w"), indent=2, ensure_ascii=False) + print(json.dumps(dict(stats), indent=2, ensure_ascii=False)) + print("informe:", out) + + +if __name__ == "__main__": + main() diff --git a/mirror-antiguo/scripts/46-analiza-404-passC.sh b/mirror-antiguo/scripts/46-analiza-404-passC.sh new file mode 100644 index 0000000..114874c --- /dev/null +++ b/mirror-antiguo/scripts/46-analiza-404-passC.sh @@ -0,0 +1,20 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +grep -hB1 'ERROR 404' "$D/logs"/wget-c-*.log | grep '^http' | sed 's/:$//' | sort -u > "$D/404-passC.txt" +echo "URLs con 404: $(wc -l < "$D/404-passC.txt")" +echo +echo "=== por tipo ===" +printf 'itemlist/user %s\n' "$(grep -c '/itemlist/user/' "$D/404-passC.txt")" +printf 'item %s\n' "$(grep -c '/item/' "$D/404-passC.txt")" +printf '/anterior %s\n' "$(grep -c '/anterior/' "$D/404-passC.txt")" +printf 'resto %s\n' "$(grep -vcE '/itemlist/user/|/item/|/anterior/' "$D/404-passC.txt")" +echo +echo "=== muestra item ===" +grep '/item/' "$D/404-passC.txt" | head -5 +echo "=== muestra itemlist/user ===" +grep '/itemlist/user/' "$D/404-passC.txt" | head -5 +echo "=== muestra resto ===" +grep -vE '/itemlist/user/|/item/|/anterior/' "$D/404-passC.txt" | head -8 diff --git a/mirror-antiguo/scripts/47-cruza-404-bd.sh b/mirror-antiguo/scripts/47-cruza-404-bd.sh new file mode 100644 index 0000000..65351cc --- /dev/null +++ b/mirror-antiguo/scripts/47-cruza-404-bd.sh @@ -0,0 +1,18 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +curl -s -H 'Host: antiguo.feadulta.com' 'http://127.0.0.1:8086/_inv/k2ids.php' > "$D/k2-ids.tsv" +echo "items K2 en BD: $(wc -l < "$D/k2-ids.tsv")" + +grep -oP '/item/\K\d+' "$D/404-passC.txt" | sort -u > "$D/404-ids.txt" +echo "ids distintos con 404: $(wc -l < "$D/404-ids.txt")" + +awk -F'\t' 'NR==FNR{want[$1]=1;next} ($1 in want){print $2"\t"$3}' "$D/404-ids.txt" "$D/k2-ids.tsv" \ + | sort | uniq -c | sed 's/^/ published,trash: /' +echo "ids que no existen en la BD: $(awk -F'\t' 'NR==FNR{have[$1]=1;next} !($1 in have)' "$D/k2-ids.tsv" "$D/404-ids.txt" | wc -l)" + +echo +echo "=== los 172 'resto' ===" +grep -vE '/itemlist/user/|/item/|/anterior/' "$D/404-passC.txt" | sed 's#^http://antiguo.feadulta.com/es/##' | cut -d/ -f1 | sort | uniq -c | sort -rn | head -15 diff --git a/mirror-antiguo/scripts/48-estado-passD.sh b/mirror-antiguo/scripts/48-estado-passD.sh new file mode 100644 index 0000000..ffb56cc --- /dev/null +++ b/mirror-antiguo/scripts/48-estado-passD.sh @@ -0,0 +1,14 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +L=$D/logs/wget-anterior.log +echo "wget vivo: $(pgrep -cf 'anterior/$')" +echo "descargados OK: $(grep -c '^2026.*URL:' "$L")" +echo "404: $(grep -c 'ERROR 404' "$L")" +echo "ficheros bajo /anterior: $(find "$D/raw/antiguo.feadulta.com/anterior" -type f 2>/dev/null | wc -l)" +echo "ficheros bajo /es/anterior (redirigidos, no deberia haber): $(find "$D/raw/antiguo.feadulta.com/es/anterior" -type f 2>/dev/null | wc -l)" +du -sh "$D/raw/antiguo.feadulta.com/anterior" 2>/dev/null +echo "--- ultimas 5 descargas OK ---" +grep '^2026.*URL:' "$L" | tail -5 | sed 's/\(.\{140\}\).*/\1/' diff --git a/mirror-antiguo/scripts/50-parity.py b/mirror-antiguo/scripts/50-parity.py new file mode 100644 index 0000000..f8e1e36 --- /dev/null +++ b/mirror-antiguo/scripts/50-parity.py @@ -0,0 +1,83 @@ +#!/usr/bin/env python3 +"""Fase 5 (local): paridad entre el fichero capturado y lo que sirve ahora el Joomla local. + +Detecta capturas truncadas, paginas de error congeladas y desfases de contenido. +No toca produccion. +""" +import os, re, sys, random, hashlib, subprocess, json +from urllib.parse import urlsplit + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +DIR = os.path.join(BASE, "runs", RUN) +RAW = os.path.join(DIR, "raw", "antiguo.feadulta.com") +N = int(sys.argv[1]) if len(sys.argv) > 1 else 200 + +TITLE = re.compile(r"]*>(.*?)", re.I | re.S) +SCRIPTS = re.compile(r"<(script|style)[^>]*>.*?", re.I | re.S) +TAGS = re.compile(r"<[^>]+>") +WS = re.compile(r"\s+") +# bloques que cambian entre peticiones. El contador de visitas de K2 ("Read N times") se +# incrementa con nuestra propia peticion, asi que dos lecturas de la MISMA pagina nunca coinciden: +# se normaliza en vez de contarlo como diferencia. +VOLATILE = re.compile(r"[0-9a-f]{32}|csrf|token", re.I) +HITS = re.compile(r"(Read|Visto|Le[ií]do)\s+\d+\s+(times|veces)", re.I) + +def texthash(s): + s = SCRIPTS.sub(" ", s) + s = TAGS.sub(" ", s) + s = WS.sub(" ", s).strip() + s = VOLATILE.sub("", s) + s = HITS.sub("HITS", s) + return hashlib.sha256(s.encode("utf-8", "replace")).hexdigest(), len(s) + +def title(s): + m = TITLE.search(s) + return WS.sub(" ", m.group(1)).strip() if m else "" + +urls = [l.strip() for l in open(os.path.join(BASE, "inventory", "urls-input.txt"))] +random.seed(20260729) +sample = random.sample(urls, min(N, len(urls))) + +res = {"muestra": len(sample), "ok_status": 0, "falta_fichero": 0, + "titulo_igual": 0, "titulo_distinto": 0, "texto_igual": 0, "texto_distinto": 0, + "diffs": []} + +for u in sample: + path = urlsplit(u).path + fp = os.path.join(RAW, path.lstrip("/")) + if path.endswith("/"): + fp = os.path.join(fp, "index.html") + if not os.path.exists(fp): + res["falta_fichero"] += 1 + res["diffs"].append({"url": u, "motivo": "fichero ausente"}) + continue + res["ok_status"] += 1 + disk = open(fp, encoding="utf-8", errors="replace").read() + live = subprocess.run( + ["curl", "-s", "--max-time", "60", "-H", "Host: antiguo.feadulta.com", + "http://127.0.0.1:8086" + path], + capture_output=True).stdout.decode("utf-8", "replace") + + td, tl = title(disk), title(live) + if td == tl: + res["titulo_igual"] += 1 + else: + res["titulo_distinto"] += 1 + res["diffs"].append({"url": u, "motivo": "titulo", "mirror": td[:120], "vivo": tl[:120]}) + + hd, ld = texthash(disk) + hl, ll = texthash(live) + if hd == hl: + res["texto_igual"] += 1 + else: + res["texto_distinto"] += 1 + res["diffs"].append({"url": u, "motivo": "texto", "len_mirror": ld, "len_vivo": ll}) + +out = os.path.join(DIR, "parity-report.json") +json.dump(res, open(out, "w"), indent=2, ensure_ascii=False) +for k in ("muestra", "falta_fichero", "titulo_igual", "titulo_distinto", "texto_igual", "texto_distinto"): + print(k, "=", res[k]) +print("informe:", out) +for d in res["diffs"][:15]: + print(" ", d) diff --git a/mirror-antiguo/scripts/50b-parity-full.py b/mirror-antiguo/scripts/50b-parity-full.py new file mode 100644 index 0000000..3748eba --- /dev/null +++ b/mirror-antiguo/scripts/50b-parity-full.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +"""Paridad COMPLETA sobre el inventario entero, a fuego lento. + +Igual que 50-parity.py pero (a) recorre las 25.437 URLs en vez de una muestra, +(b) mete una pausa entre peticiones para no ahogar al Joomla local ni a la WSL +(ver leccion del crawl que tumbo la VM), y (c) escribe JSONL incremental para +poder mirar el progreso y reanudar sin repetir trabajo. + +Uso: python3 50b-parity-full.py [pausa_segundos] [workers] (por defecto 0.35 y 1) + +Se registra ademas el codigo HTTP del Joomla local: sin eso, un 500 del contenedor se contaria +como "el mirror difiere" y ensuciaria el informe con diferencias que no lo son. +""" +import os, re, sys, time, json, hashlib, subprocess, threading, collections +import concurrent.futures +from urllib.parse import urlsplit + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +DIR = os.path.join(BASE, "runs", RUN) +RAW = os.path.join(DIR, "raw", "antiguo.feadulta.com") +PAUSA = float(sys.argv[1]) if len(sys.argv) > 1 else 0.35 +WORKERS = int(sys.argv[2]) if len(sys.argv) > 2 else 1 + +JSONL = os.path.join(DIR, "parity-full.jsonl") +PROG = os.path.join(DIR, "parity-full.progress") +OUT = os.path.join(DIR, "parity-report-full.json") + +TITLE = re.compile(r"]*>(.*?)", re.I | re.S) +SCRIPTS = re.compile(r"<(script|style)[^>]*>.*?", re.I | re.S) +TAGS = re.compile(r"<[^>]+>") +WS = re.compile(r"\s+") +VOLATILE = re.compile(r"[0-9a-f]{32}|csrf|token", re.I) +HITS = re.compile(r"(Read|Visto|Le[ií]do)\s+\d+\s+(times|veces)", re.I) + + +def texthash(s): + s = SCRIPTS.sub(" ", s) + s = TAGS.sub(" ", s) + s = WS.sub(" ", s).strip() + s = VOLATILE.sub("", s) + s = HITS.sub("HITS", s) + return hashlib.sha256(s.encode("utf-8", "replace")).hexdigest(), len(s) + + +def title(s): + m = TITLE.search(s) + return WS.sub(" ", m.group(1)).strip() if m else "" + + +urls = [l.strip() for l in open(os.path.join(BASE, "inventory", "urls-input.txt")) if l.strip()] + +# Reanudable: lo ya comprobado no se repite. +hechas = set() +if os.path.exists(JSONL): + for line in open(JSONL, encoding="utf-8"): + try: + hechas.add(json.loads(line)["url"]) + except Exception: + pass + +pendientes = [u for u in urls if u not in hechas] +print(f"inventario={len(urls)} ya_hechas={len(hechas)} pendientes={len(pendientes)} " + f"pausa={PAUSA}s workers={WORKERS}", flush=True) + +t0 = time.time() +lock = threading.Lock() +contador = {"n": 0} +codigos = collections.Counter() + + +def comprueba(u): + path = urlsplit(u).path + fp = os.path.join(RAW, path.lstrip("/")) + if path.endswith("/"): + fp = os.path.join(fp, "index.html") + + if not os.path.exists(fp): + return {"url": u, "estado": "falta_fichero"} + + disk = open(fp, encoding="utf-8", errors="replace").read() + salida = subprocess.run( + ["curl", "-s", "-w", "\n%{http_code}", "--max-time", "60", + "-H", "Host: antiguo.feadulta.com", "http://127.0.0.1:8086" + path], + capture_output=True).stdout.decode("utf-8", "replace") + live, _, code = salida.rpartition("\n") + code = code.strip() or "000" + + td, tl = title(disk), title(live) + hd, ld = texthash(disk) + hl, ll = texthash(live) + if PAUSA: + time.sleep(PAUSA) + return {"url": u, "estado": "ok", "http_vivo": code, + "titulo_igual": td == tl, "texto_igual": hd == hl, + "titulo_mirror": td[:120], "titulo_vivo": tl[:120], + "len_mirror": ld, "len_vivo": ll} + + +with open(JSONL, "a", encoding="utf-8") as fh: + with concurrent.futures.ThreadPoolExecutor(max_workers=WORKERS) as ex: + for d in ex.map(comprueba, pendientes): + with lock: + fh.write(json.dumps(d, ensure_ascii=False) + "\n") + contador["n"] += 1 + i = contador["n"] + codigos[d.get("http_vivo", "-")] += 1 + if i % 100 == 0: + fh.flush() + hechas_tot = len(hechas) + i + ritmo = i / max(time.time() - t0, 1) + queda = (len(pendientes) - i) / max(ritmo, 0.001) / 60 + open(PROG, "w").write( + f"{hechas_tot}/{len(urls)} ({100*hechas_tot/len(urls):.1f}%) " + f"ritmo={ritmo:.2f}/s ETA={queda:.0f}min " + f"http_vivo={dict(codigos)}\n") + +# Resumen final a partir del JSONL completo. +res = {"inventario": len(urls), "comprobadas": 0, "falta_fichero": 0, + "vivo_no_200": 0, "http_vivo": {}, + "titulo_igual": 0, "titulo_distinto": 0, "texto_igual": 0, "texto_distinto": 0, + "diffs": []} +for line in open(JSONL, encoding="utf-8"): + d = json.loads(line) + if d["estado"] == "falta_fichero": + res["falta_fichero"] += 1 + res["diffs"].append({"url": d["url"], "motivo": "fichero ausente"}) + continue + code = d.get("http_vivo", "?") + res["http_vivo"][code] = res["http_vivo"].get(code, 0) + 1 + if code not in ("200", "?"): + # El Joomla local fallo en esta peticion: no es una diferencia del mirror. + res["vivo_no_200"] += 1 + res["diffs"].append({"url": d["url"], "motivo": "joomla local " + code}) + continue + res["comprobadas"] += 1 + if d["titulo_igual"]: + res["titulo_igual"] += 1 + else: + res["titulo_distinto"] += 1 + res["diffs"].append({"url": d["url"], "motivo": "titulo", + "mirror": d["titulo_mirror"], "vivo": d["titulo_vivo"]}) + if d["texto_igual"]: + res["texto_igual"] += 1 + else: + res["texto_distinto"] += 1 + res["diffs"].append({"url": d["url"], "motivo": "texto", + "len_mirror": d["len_mirror"], "len_vivo": d["len_vivo"]}) + +json.dump(res, open(OUT, "w"), indent=2, ensure_ascii=False) +for k in ("inventario", "comprobadas", "falta_fichero", "vivo_no_200", "http_vivo", + "titulo_igual", "titulo_distinto", "texto_igual", "texto_distinto"): + print(k, "=", res[k]) +print("informe:", OUT) diff --git a/mirror-antiguo/scripts/51-diff-paridad.sh b/mirror-antiguo/scripts/51-diff-paridad.sh new file mode 100644 index 0000000..7367fe1 --- /dev/null +++ b/mirror-antiguo/scripts/51-diff-paridad.sh @@ -0,0 +1,11 @@ +#!/bin/bash +# Averigua QUE cambia entre el fichero capturado y lo que sirve ahora el Joomla local +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +R=$BASE/runs/$RUN/raw/antiguo.feadulta.com +P=${1:-/es/buscadoravanzado/item/16690-el-dios-de-trump.html} + +curl -s -H 'Host: antiguo.feadulta.com' "http://127.0.0.1:8086$P" > /tmp/vivo.html +diff <(sed 's/>\n\n/dev/null | head -3 diff --git a/mirror-antiguo/scripts/53-pase-d2-anterior-inventario.sh b/mirror-antiguo/scripts/53-pase-d2-anterior-inventario.sh new file mode 100644 index 0000000..592fb02 --- /dev/null +++ b/mirror-antiguo/scripts/53-pase-d2-anterior-inventario.sh @@ -0,0 +1,45 @@ +#!/bin/bash +# Pase D (corregido): /anterior por INVENTARIO, no por recursion. +# +# La recursion sobre /anterior funcionaba, pero se estaba comiendo el tiempo en 404: la web antigua +# esta llena de enlaces rotos (imagenes de los `_archivos/` de exportaciones de Word que ya no +# existen). Iban 3.050 aciertos por 2.744 fallos. Misma leccion del post-mortem: acotar por +# inventario. Aqui el inventario es el listado de ficheros del snapshot restaurado — solo la LISTA +# DE RUTAS, igual que se hace con la BD; el contenido se sigue capturando por HTTP. +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +SRC=$BASE/restore/web/anterior +W=${WORKERS:-4} + +pkill -f 'antiguo.feadulta.com/anterior/$' && echo "recursion detenida" || echo "recursion ya parada" +sleep 2 + +cd "$BASE/restore/web" +find anterior -type f ! -iname '*.php' -printf '%p\n' \ + | sed 's#^#http://antiguo.feadulta.com/#' \ + | sort -u > "$D/anterior-input.txt" +echo "inventario de /anterior: $(wc -l < "$D/anterior-input.txt") ficheros (excluidos los .php)" +echo "php excluidos: $(find anterior -type f -iname '*.php' | wc -l)" + +mkdir -p "$D/chunks-d"; rm -f "$D/chunks-d"/*.txt +split -n l/$W -d --additional-suffix=.txt "$D/anterior-input.txt" "$D/chunks-d/d-" + +date -u +%FT%TZ > "$D/logs/passD2.start" +for c in "$D/chunks-d"/d-*.txt; do + b=$(basename "$c" .txt) + wget --input-file="$c" \ + --force-directories --directory-prefix="$D/raw" \ + --no-verbose --no-clobber -e robots=off \ + --user-agent='feadulta-archiver/1.0 (+incident-183; mirror local)' \ + --wait=0.02 --tries=2 --timeout=45 --waitretry=3 \ + --output-file="$D/logs/wget-$b.log" & +done +wait +date -u +%FT%TZ > "$D/logs/passD2.end" + +echo "PASE D TERMINADO" +echo "ficheros bajo /anterior: $(find "$D/raw/antiguo.feadulta.com/anterior" -type f | wc -l)" +du -sh "$D/raw/antiguo.feadulta.com/anterior" +grep -hoE 'ERROR [0-9]+' "$D/logs"/wget-d-*.log | sort | uniq -c diff --git a/mirror-antiguo/scripts/54-cobertura.py b/mirror-antiguo/scripts/54-cobertura.py new file mode 100644 index 0000000..cd70e7b --- /dev/null +++ b/mirror-antiguo/scripts/54-cobertura.py @@ -0,0 +1,47 @@ +#!/usr/bin/env python3 +"""Cobertura real: cada URL del inventario debe tener su fichero en raw/. + +Contempla las tres formas en que wget nombra el fichero: + /es/x.html -> x.html + /es/ -> index.html + /es/x?.html (alias con '?' literal) -> "x?.html" o "x" +""" +import os, json +from urllib.parse import urlsplit, unquote + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +DIR = os.path.join(BASE, "runs", RUN) +RAW = os.path.join(DIR, "raw", "antiguo.feadulta.com") + +def candidates(url): + rest = url.split("antiguo.feadulta.com", 1)[1] + rest = unquote(rest) + yield rest.lstrip("/") # nombre literal, con '?' incluido + p = urlsplit(rest).path.lstrip("/") + yield p # truncado en el '?' + if rest.endswith("/") or p.endswith("/") or p == "": + yield (p + "index.html") + +ok, missing = 0, [] +urls = [l.strip() for l in open(os.path.join(BASE, "inventory", "urls-input.txt")) if l.strip()] +for u in urls: + if any(os.path.isfile(os.path.join(RAW, c)) for c in candidates(u) if c): + ok += 1 + else: + missing.append(u) + +print("inventario:", len(urls)) +print("con fichero en raw/:", ok) +print("sin fichero:", len(missing)) +print("cobertura: %.2f%%" % (ok * 100.0 / len(urls))) +with open(os.path.join(DIR, "coverage-missing.txt"), "w") as f: + for u in missing: + f.write(u + "\n") +for u in missing[:25]: + print(" ", u) + +total = sum(len(fs) for _r, _d, fs in os.walk(os.path.join(DIR, "raw"))) +json.dump({"inventario": len(urls), "capturadas": ok, "sin_fichero": len(missing), + "cobertura_pct": round(ok * 100.0 / len(urls), 2), "ficheros_totales_raw": total}, + open(os.path.join(DIR, "coverage-report.json"), "w"), indent=2) diff --git a/mirror-antiguo/scripts/55-revisa-sospechosos.sh b/mirror-antiguo/scripts/55-revisa-sospechosos.sh new file mode 100644 index 0000000..07e4a5a --- /dev/null +++ b/mirror-antiguo/scripts/55-revisa-sospechosos.sh @@ -0,0 +1,13 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +cd "$D" +echo "=== ficheros marcados por el escaneo (§4.3.2) ===" +while read -r f; do + echo "----- $f" + du -h "$f" 2>/dev/null | cut -f1 + head -c 200 "$f" | tr -d '\0' + echo; echo +done < "$D/scan-suspicious.txt" diff --git a/mirror-antiguo/scripts/56-verifica-limpieza.sh b/mirror-antiguo/scripts/56-verifica-limpieza.sh new file mode 100644 index 0000000..baacdd2 --- /dev/null +++ b/mirror-antiguo/scripts/56-verifica-limpieza.sh @@ -0,0 +1,14 @@ +#!/bin/bash +# Que los scripts auxiliares que metimos en la raiz del Joomla restaurado NO esten en el mirror +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +echo "=== _genurls.php / _inv/ dentro de raw ===" +find "$D/raw" \( -name '_genurls.php' -o -path '*_inv*' \) | wc -l +echo "=== cualquier .php en raw ===" +find "$D/raw" -iname '*.php' | head +echo "(total: $(find "$D/raw" -iname '*.php' | wc -l))" +echo +echo "=== auxiliares presentes en el Joomla restaurado (fuera del mirror) ===" +ls "$BASE/restore/web/_genurls.php" "$BASE/restore/web/_inv/" 2>/dev/null diff --git a/mirror-antiguo/scripts/57-lista-22.sh b/mirror-antiguo/scripts/57-lista-22.sh new file mode 100644 index 0000000..8556965 --- /dev/null +++ b/mirror-antiguo/scripts/57-lista-22.sh @@ -0,0 +1,14 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +echo "=== coincidencias de '_inv' o '_genurls' en raw ===" +find "$D/raw" \( -name '_genurls.php' -o -path '*_inv*' \) | head -25 +echo +echo "=== index.php capturado: que contiene ===" +f="$D/raw/antiguo.feadulta.com/index.php" +ls -la "$f" | sed 's/\(.\{120\}\).*/\1/' +head -c 200 "$f" +echo; echo +echo "contiene '/dev/null +echo +echo "=== assets cache-busted: existe la copia limpia? ===" +for f in media/system/js/core.js media/jui/js/jquery.min.js components/com_k2/css/k2.css; do + printf '%-45s %s\n' "$f" "$([ -f "$S/$f" ] && echo OK || echo FALTA)" +done +echo +echo "=== enlaces absolutos que queden a antiguo.feadulta.com ===" +grep -rl 'http://antiguo.feadulta.com' "$S/es" 2>/dev/null | wc -l diff --git a/mirror-antiguo/scripts/59-recopia-nombres-limpios.py b/mirror-antiguo/scripts/59-recopia-nombres-limpios.py new file mode 100644 index 0000000..156cd61 --- /dev/null +++ b/mirror-antiguo/scripts/59-recopia-nombres-limpios.py @@ -0,0 +1,30 @@ +#!/usr/bin/env python3 +"""Rehace las copias de nombre limpio DESPUES de la reescritura de enlaces. + +El orden importaba: en `45-normalize-links.py` las copias se creaban antes de reescribir, asi que +`titulo?.html` quedaba reescrito pero su copia `titulo` (sin extension, la que pedira el navegador) +conservaba los enlaces absolutos. Aqui se rehacen desde el fichero ya reescrito. +""" +import os, shutil +from collections import Counter + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +SITE = os.path.join(BASE, "runs", RUN, "site") + +st = Counter() +for root, _d, files in os.walk(SITE): + for fn in list(files): + if "?" not in fn: + continue + base = fn.split("?", 1)[0] + if not base: + continue + src, dst = os.path.join(root, fn), os.path.join(root, base) + if os.path.exists(dst) and os.path.getmtime(dst) >= os.path.getmtime(src): + st["ya_al_dia"] += 1 + continue + shutil.copy2(src, dst) + st["recopiados"] += 1 + +print(dict(st)) diff --git a/mirror-antiguo/scripts/60-quitar-ua.py b/mirror-antiguo/scripts/60-quitar-ua.py new file mode 100644 index 0000000..0433082 --- /dev/null +++ b/mirror-antiguo/scripts/60-quitar-ua.py @@ -0,0 +1,74 @@ +#!/usr/bin/env python3 +"""Quita el tag de Google Analytics clasico (UA-32008163-1) del mirror servible. + +UA dejo de procesar datos en julio de 2023: el snippet solo sirve para pedir un +ga.js muerto en cada carga. Se quita el bloque que contiene el UA. El (?:(?!).)*? impide +# que el .*? se coma varios bloques seguidos y se lleve por delante el GA4. +BLOQUE = re.compile( + r"[ \t]*]*>(?:(?!).)*?" + re.escape(UA) + + r"(?:(?!).)*?\s*", re.S) + +tocados = errores = 0 +sin_ga4 = [] +bytes_antes = bytes_despues = 0 + +for raiz, _, ficheros in os.walk(SITE): + for f in ficheros: + if not f.lower().endswith((".html", ".htm")) and "." in f: + continue + ruta = os.path.join(raiz, f) + try: + txt = open(ruta, encoding="utf-8", errors="surrogateescape").read() + except (OSError, UnicodeDecodeError): + continue + if UA not in txt: + continue + + tenia_ga4 = GA4 in txt + nuevo, n = BLOQUE.subn("\n", txt) + + if UA in nuevo: + # El bloque no casó: no dejar el fichero a medias, mejor avisar. + errores += 1 + continue + if tenia_ga4 and GA4 not in nuevo: + sin_ga4.append(ruta) + continue + + bytes_antes += len(txt) + bytes_despues += len(nuevo) + tocados += 1 + if not DRY: + with open(ruta, "w", encoding="utf-8", errors="surrogateescape") as fh: + fh.write(nuevo) + +print(f"ficheros modificados : {tocados}") +print(f"no casó el patron : {errores}") +print(f"habrian perdido GA4 : {len(sin_ga4)}") +for r in sin_ga4[:5]: + print(" ", r) +if tocados: + print(f"bytes : {bytes_antes:,} -> {bytes_despues:,} " + f"({bytes_antes - bytes_despues:,} menos)") +print("(DRY RUN, no se ha escrito nada)" if DRY else "escrito") diff --git a/mirror-antiguo/scripts/60-restaurar-entorno.sh b/mirror-antiguo/scripts/60-restaurar-entorno.sh new file mode 100644 index 0000000..30967b2 --- /dev/null +++ b/mirror-antiguo/scripts/60-restaurar-entorno.sh @@ -0,0 +1,12 @@ +#!/bin/bash +# Devuelve el entorno a como estaba: rearranca los contenedores parados durante el crawl +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +STOPPED=$BASE/stopped-containers.txt +[ -s "$STOPPED" ] || { echo "no hay lista de contenedores parados"; exit 0; } +while read -r c; do + [ -n "$c" ] && docker start "$c" >/dev/null && echo "arrancado $c" +done < "$STOPPED" +mv "$STOPPED" "$STOPPED.hecho-$(date -u +%Y%m%dT%H%M%SZ)" +sleep 5 +docker ps --format '{{.Names}} {{.Status}}' diff --git a/mirror-antiguo/scripts/70-smoke-nginx.sh b/mirror-antiguo/scripts/70-smoke-nginx.sh new file mode 100644 index 0000000..8b6f164 --- /dev/null +++ b/mirror-antiguo/scripts/70-smoke-nginx.sh @@ -0,0 +1,38 @@ +#!/bin/bash +# Prueba de humo del despliegue: sirve site/ con nginx y comprueba que las rutas criticas +# responden 200 con el Content-Type correcto. Local, en el puerto 8087, se borra al terminar. +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +S=$BASE/runs/$RUN/site/antiguo.feadulta.com + +docker rm -f mirror-nginx-test >/dev/null 2>&1 +docker run -d --name mirror-nginx-test --memory 256m \ + -p 127.0.0.1:8087:80 \ + -v "$S":/usr/share/nginx/html:ro \ + -v "$BASE/deploy/nginx-mirror.conf":/etc/nginx/conf.d/default.conf:ro \ + nginx:alpine >/dev/null +sleep 3 + +probe() { + local u="$1" desc="$2" + read -r code ctype < <(curl -s -o /dev/null -w '%{http_code} %{content_type}' "http://127.0.0.1:8087$u"; echo) + printf '%-6s %-28s %-58s %s\n' "$code" "$ctype" "$u" "$desc" +} + +echo "codigo content-type url" +probe "/es/" "portada" +probe "/es/carta/estasemana.html" "carta: esta semana" +probe "/es/buscadoravanzado/item/9-experiencia-pascual.html" "item K2" +probe "/es/buscadoravanzado/item/715-%C2%BFqui%C3%A9n-es-jes%C3%BAs" "item con '?' en el alias" +probe "/es/buscadoravanzado/itemlist/user/569-agust%C3%ADnud%C3%ADasvallina.html" "pagina de autor K2" +probe "/es/lista-completa-de-autores-por-orden-alfabetico.html" "listado de autores" +probe "/anterior/" "web anterior (indice)" +probe "/media/system/js/core.js" "js con cache-busting" +probe "/components/com_k2/css/k2.css" "css de K2" +probe "/es/no-existe-esta-pagina.html" "404 esperado" + +echo +echo "=== la portada trae contenido de verdad? ===" +curl -s http://127.0.0.1:8087/es/ | grep -o '[^<]*' | head -1 +curl -s http://127.0.0.1:8087/es/ | wc -c diff --git a/mirror-antiguo/scripts/71-manifest-site.sh b/mirror-antiguo/scripts/71-manifest-site.sh new file mode 100644 index 0000000..0fd0ac2 --- /dev/null +++ b/mirror-antiguo/scripts/71-manifest-site.sh @@ -0,0 +1,13 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +cd "$D/site" && find . -type f -print0 | sort -z | xargs -0 sha256sum > "$D/MANIFEST-site.sha256" +cd "$D" +echo "MANIFEST-raw: $(wc -l < MANIFEST-raw.sha256) ficheros" +echo "MANIFEST-site: $(wc -l < MANIFEST-site.sha256) ficheros" +du -sh raw site +echo +echo "=== contenido de la corrida ===" +ls -la "$D" | grep -vE '^d.*(raw|site|chunks|logs)$' diff --git a/mirror-antiguo/scripts/80-sync-hetzner.sh b/mirror-antiguo/scripts/80-sync-hetzner.sh new file mode 100644 index 0000000..a0ac4bf --- /dev/null +++ b/mirror-antiguo/scripts/80-sync-hetzner.sh @@ -0,0 +1,22 @@ +#!/bin/bash +# Sincroniza el arbol servible con el Hetzner. Solo lo que ha cambiado. +# +# --delete es intencionado: el servidor debe ser copia exacta de site/, ni un +# fichero de mas. Por eso se comprueba antes que el origen NO esta vacio: un +# origen vacio con --delete borraria el sitio entero. +set -euo pipefail + +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +SRC="$BASE/runs/$RUN/site/antiguo.feadulta.com/" +DST=root@188.40.120.157:/data/feadulta-antiguo/site/antiguo.feadulta.com/ + +n=$(find "$SRC" -type f | wc -l) +echo "origen: $SRC" +echo "ficheros en origen: $n" +if [ "$n" -lt 60000 ]; then + echo "ABORTADO: el origen tiene menos ficheros de los esperados. No se sincroniza." + exit 1 +fi + +rsync -a --delete --stats --human-readable "$SRC" "$DST" diff --git a/mirror-antiguo/scripts/80-wayback-inventario.sh b/mirror-antiguo/scripts/80-wayback-inventario.sh new file mode 100644 index 0000000..e38533a --- /dev/null +++ b/mirror-antiguo/scripts/80-wayback-inventario.sh @@ -0,0 +1,17 @@ +#!/bin/bash +# Fuente F3 del plan: inventario historico de URLs segun Internet Archive. +# No toca el origen ni produccion; es una consulta de solo lectura a web.archive.org. +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +INV=$BASE/inventory +mkdir -p "$INV" + +for host in antiguo.feadulta.com feadulta.com; do + out="$INV/wayback-${host%%.*}.txt" + echo "-> $host" + curl -s --max-time 300 \ + "https://web.archive.org/cdx/search/cdx?url=${host}*&output=text&fl=original&collapse=urlkey&limit=200000" \ + > "$out" + echo " $(wc -l < "$out") URLs" +done +wc -l "$INV"/wayback-*.txt diff --git a/mirror-antiguo/scripts/81-cobertura-wayback.py b/mirror-antiguo/scripts/81-cobertura-wayback.py new file mode 100644 index 0000000..5d01ab8 --- /dev/null +++ b/mirror-antiguo/scripts/81-cobertura-wayback.py @@ -0,0 +1,70 @@ +#!/usr/bin/env python3 +"""Cruza el inventario historico de Internet Archive contra el mirror. + +Responde a la pregunta de aceptacion que de verdad importa: **de las URLs legacy que el mundo +exterior tiene enlazadas, cuantas resuelven en el mirror**. Sustituto parcial de la fuente F2 (GA4), +que sigue bloqueada porque requiere que Rafa abra el OAuth a mano. +""" +import os, json +from collections import Counter +from urllib.parse import urlsplit, unquote + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +DIR = os.path.join(BASE, "runs", RUN) +SITE = os.path.join(DIR, "site", "antiguo.feadulta.com") + +def existe(path): + p = unquote(path).lstrip("/") + for c in (p, p.split("?", 1)[0], os.path.join(p, "index.html")): + if c and os.path.isfile(os.path.join(SITE, c)): + return True + return False + +paths, cats = set(), Counter() +for fn in ("wayback-antiguo.txt", "wayback-feadulta.txt"): + for line in open(os.path.join(BASE, "inventory", fn), errors="replace"): + u = line.strip() + if not u: + continue + p = urlsplit(u).path + q = urlsplit(u).query + if q: # las URLs con query no forman parte del mirror estatico + cats["con_query (fuera de alcance)"] += 1 + continue + if not p or p == "/": + cats["raiz"] += 1 + continue + paths.add(p) + +ok, missing = 0, [] +for p in sorted(paths): + if existe(p): + ok += 1 + else: + missing.append(p) + +print("URLs distintas de Wayback sin query:", len(paths)) +print("presentes en el mirror:", ok, "(%.1f%%)" % (ok * 100.0 / max(len(paths), 1))) +print("ausentes:", len(missing)) +for k, v in cats.most_common(): + print(" %s: %s" % (k, v)) + +# clasificar las ausentes para ver si importan +tipo = Counter() +for p in missing: + seg = p.strip("/").split("/")[0] if p.strip("/") else "(raiz)" + tipo[seg] += 1 +print("\n--- ausentes por primer segmento ---") +for k, v in tipo.most_common(20): + print("%7d %s" % (v, k)) + +with open(os.path.join(DIR, "wayback-missing.txt"), "w") as f: + for p in missing: + f.write(p + "\n") +json.dump({"wayback_paths": len(paths), "presentes": ok, "ausentes": len(missing), + "pct": round(ok * 100.0 / max(len(paths), 1), 2)}, + open(os.path.join(DIR, "wayback-report.json"), "w"), indent=2) +print("\n--- muestra de ausentes ---") +for p in missing[:20]: + print(" ", p) diff --git a/mirror-antiguo/scripts/82-wayback-es.py b/mirror-antiguo/scripts/82-wayback-es.py new file mode 100644 index 0000000..2d45582 --- /dev/null +++ b/mirror-antiguo/scripts/82-wayback-es.py @@ -0,0 +1,63 @@ +#!/usr/bin/env python3 +"""Afina el cruce con Wayback: solo las URLs /es/ (el Joomla legacy), que es lo que el mirror cubre. + +El 52 % global del script anterior mezcla peras con manzanas: Wayback conoce feadulta.com desde +antes de que existiera el Joomla (ficheros .htm sueltos en la raiz, que hoy viven bajo /anterior/) y +tambien el WordPress actual (/wp-content, /wp-json). Nada de eso forma parte del mirror del legacy. +""" +import os, json, re +from collections import Counter +from urllib.parse import urlsplit, unquote + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +DIR = os.path.join(BASE, "runs", RUN) +SITE = os.path.join(DIR, "site", "antiguo.feadulta.com") + +def existe(path): + p = unquote(path).lstrip("/") + for c in (p, p.split("?", 1)[0], os.path.join(p, "index.html")): + if c and os.path.isfile(os.path.join(SITE, c)): + return True + return False + +paths = set() +for fn in ("wayback-antiguo.txt", "wayback-feadulta.txt"): + for line in open(os.path.join(BASE, "inventory", fn), errors="replace"): + u = line.strip() + if not u: + continue + s = urlsplit(u) + if s.query: + continue + if s.path.startswith("/es/"): + paths.add(s.path) + +ok, missing = 0, [] +for p in sorted(paths): + if existe(p): + ok += 1 + else: + missing.append(p) + +print("URLs /es/ conocidas por Wayback:", len(paths)) +print("resuelven en el mirror:", ok, "(%.1f%%)" % (ok * 100.0 / max(len(paths), 1))) +print("no resuelven:", len(missing)) + +tipo = Counter() +for p in missing: + seg = p.split("/") + tipo["/".join(seg[:3])] += 1 +print("\n--- las que faltan, por seccion ---") +for k, v in tipo.most_common(15): + print("%7d %s" % (v, k)) + +json.dump({"wayback_es_paths": len(paths), "presentes": ok, "ausentes": len(missing), + "pct": round(ok * 100.0 / max(len(paths), 1), 2)}, + open(os.path.join(DIR, "wayback-es-report.json"), "w"), indent=2) +with open(os.path.join(DIR, "wayback-es-missing.txt"), "w") as f: + for p in missing: + f.write(p + "\n") +print("\n--- muestra ---") +for p in missing[:15]: + print(" ", p) diff --git a/mirror-antiguo/scripts/83-contraste-wayback.py b/mirror-antiguo/scripts/83-contraste-wayback.py new file mode 100644 index 0000000..1edfbea --- /dev/null +++ b/mirror-antiguo/scripts/83-contraste-wayback.py @@ -0,0 +1,77 @@ +#!/usr/bin/env python3 +"""§4.3, último punto del plan: contrastar páginas capturadas contra Internet Archive. + +No compara el texto (una instantánea de hace años difiere por fuerza: fechas, barras laterales, +bloques rotativos). Compara lo que de verdad delata una inyección: **el conjunto de hosts externos +a los que la página carga scripts o iframes**. Si nuestra captura referencia hosts que la versión +histórica no tenía, hay que mirarlo. +""" +import os, re, json, sys, urllib.request, random +from collections import Counter +from urllib.parse import urlsplit, unquote + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +DIR = os.path.join(BASE, "runs", RUN) +SITE = os.path.join(DIR, "site", "antiguo.feadulta.com") +N = int(sys.argv[1]) if len(sys.argv) > 1 else 10 + +SRC = re.compile(r'<(?:script|iframe)[^>]+src=["\']((?:https?:)?//[^"\'/]+)', re.I) +UA = {"User-Agent": "feadulta-archiver/1.0 (verificacion de integridad; incident-183)"} + +def hosts(html): + out = set() + for m in SRC.findall(html): + h = m.split("//", 1)[-1].lower() + # web.archive.org reescribe los recursos: nos quedamos con el host original + if h.startswith("web.archive.org"): + continue + out.add(h) + return out + +# candidatas: URLs /es/ que Wayback conoce Y que tenemos capturadas +cand = [] +for fn in ("wayback-antiguo.txt", "wayback-feadulta.txt"): + for line in open(os.path.join(BASE, "inventory", fn), errors="replace"): + u = line.strip() + s = urlsplit(u) + if s.query or not s.path.startswith("/es/") or not s.path.endswith(".html"): + continue + p = unquote(s.path).lstrip("/") + if os.path.isfile(os.path.join(SITE, p)): + cand.append((u, p)) + +random.seed(20260730) +sample = random.sample(cand, min(N, len(cand))) +print("candidatas:", len(cand), "- muestra:", len(sample), "\n") + +res, extra_total = [], Counter() +for url, rel in sample: + local = open(os.path.join(SITE, rel), encoding="utf-8", errors="replace").read() + hl = hosts(local) + try: + req = urllib.request.Request("https://web.archive.org/web/2id_/" + url, headers=UA) + arch = urllib.request.urlopen(req, timeout=90).read().decode("utf-8", "replace") + ha = hosts(arch) + estado = "ok" + except Exception as e: + ha, estado = set(), "sin snapshot (%s)" % type(e).__name__ + extra = hl - ha + if estado == "ok": + for h in extra: + extra_total[h] += 1 + print("%-70s %s" % (rel[-68:], estado)) + if estado == "ok" and extra: + print(" hosts solo en nuestra captura:", ", ".join(sorted(extra))) + res.append({"url": url, "estado": estado, "hosts_mirror": sorted(hl), + "hosts_wayback": sorted(ha), "solo_en_mirror": sorted(extra)}) + +print("\n--- hosts presentes solo en nuestra captura (agregado) ---") +if extra_total: + for h, c in extra_total.most_common(): + print("%4d %s" % (c, h)) +else: + print("ninguno") + +json.dump(res, open(os.path.join(DIR, "wayback-contraste.json"), "w"), indent=2, ensure_ascii=False) +print("\ninforme:", os.path.join(DIR, "wayback-contraste.json")) diff --git a/mirror-antiguo/scripts/84-inspecciona-terceros.sh b/mirror-antiguo/scripts/84-inspecciona-terceros.sh new file mode 100644 index 0000000..805c1f7 --- /dev/null +++ b/mirror-antiguo/scripts/84-inspecciona-terceros.sh @@ -0,0 +1,27 @@ +#!/bin/bash +# Que codigo de terceros lleva realmente el mirror: GTM y botones sociales +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +S=$BASE/runs/$RUN/site/antiguo.feadulta.com +P=$S/es/buscadoravanzado/item/9-experiencia-pascual.html + +echo "=== IDs de contenedor GTM/GA que aparecen en el mirror ===" +grep -rhoE 'GTM-[A-Z0-9]+|UA-[0-9]+-[0-9]+|G-[A-Z0-9]+' "$S/es" 2>/dev/null | sort | uniq -c | sort -rn | head + +echo +echo "=== bloque GTM en una pagina de ejemplo ===" +grep -o 'googletagmanager[^<]*' "$P" | head -3 +grep -B2 -A6 'googletagmanager' "$P" | head -25 + +echo +echo "=== bloques sociales en esa misma pagina ===" +grep -oE ']*(connect\.facebook\.net|platform\.twitter\.com)[^>]*>' "$P" | head +grep -oE '(fb-root|fb-like|twitter-share-button|fb:like|data-href="[^"]*")' "$P" | head -10 + +echo +echo "=== cuantas paginas llevan cada cosa ===" +printf 'googletagmanager : %s\n' "$(grep -rl 'googletagmanager' "$S" 2>/dev/null | wc -l)" +printf 'connect.facebook : %s\n' "$(grep -rl 'connect.facebook.net' "$S" 2>/dev/null | wc -l)" +printf 'platform.twitter : %s\n' "$(grep -rl 'platform.twitter.com' "$S" 2>/dev/null | wc -l)" +printf 'cdnjs.cloudflare : %s\n' "$(grep -rl 'cdnjs.cloudflare.com' "$S" 2>/dev/null | wc -l)" diff --git a/mirror-antiguo/scripts/85-marca-social.sh b/mirror-antiguo/scripts/85-marca-social.sh new file mode 100644 index 0000000..cab83d7 --- /dev/null +++ b/mirror-antiguo/scripts/85-marca-social.sh @@ -0,0 +1,14 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +S=$BASE/runs/$RUN/site/antiguo.feadulta.com +P=$S/es/buscadoravanzado/item/9-experiencia-pascual.html + +echo "=== lineas con twitter / facebook / fb- ===" +grep -n -E 'platform\.twitter|connect\.facebook|fb-root|fb-like|twitter-share-button' "$P" \ + | cut -c1-400 +echo +echo "=== 12 lineas alrededor de la primera aparicion ===" +n=$(grep -n 'twitter-share-button\|platform.twitter' "$P" | head -1 | cut -d: -f1) +sed -n "$((n-6)),$((n+14))p" "$P" | cut -c1-300 diff --git a/mirror-antiguo/scripts/86-survey-social.py b/mirror-antiguo/scripts/86-survey-social.py new file mode 100644 index 0000000..3809448 --- /dev/null +++ b/mirror-antiguo/scripts/86-survey-social.py @@ -0,0 +1,64 @@ +#!/usr/bin/env python3 +"""Antes de tocar nada: que hay realmente dentro de los bloques sociales del mirror.""" +import os, re +from collections import Counter + +BASE = "/home/rafa/joomla-migration/mirror-antiguo" +RUN = open(os.path.join(BASE, "CURRENT_RUN")).read().strip() +SITE = os.path.join(BASE, "runs", RUN, "site", "antiguo.feadulta.com") + +OPEN = re.compile(r']*class="[^"]*itemSocialSharing[^"]*"[^>]*>', re.I) +DIV = re.compile(r']*>|', re.I) +CLASS = re.compile(r']*class="([^"]+)"', re.I) +SCRIPTSRC = re.compile(r']+src="([^"]+)"', re.I) + +def bloque(html, m): + """Devuelve (inicio, fin) del div equilibrado que empieza en m.""" + depth, pos = 0, m.start() + for d in DIV.finditer(html, m.start()): + if d.group(0).lower().startswith(" 400: + break + for m in OPEN.finditer(html): + r = bloque(html, m) + if not r: + sin_bloque += 1 + continue + con_bloque += 1 + frag = html[r[0]:r[1]] + for c in CLASS.findall(frag): + clases[c.strip()] += 1 + for s in SCRIPTSRC.findall(frag): + scripts[s.split("?")[0]] += 1 + if n > 400: + break + +print("paginas inspeccionadas con itemSocialSharing:", n) +print("bloques equilibrados:", con_bloque, " sin cerrar:", sin_bloque) +print("\n--- clases de div dentro del bloque ---") +for k, v in clases.most_common(15): + print("%7d %s" % (v, k)) +print("\n--- scripts dentro del bloque ---") +for k, v in scripts.most_common(15): + print("%7d %s" % (v, k)) diff --git a/mirror-antiguo/scripts/87-quita-social.py b/mirror-antiguo/scripts/87-quita-social.py new file mode 100644 index 0000000..78d644e --- /dev/null +++ b/mirror-antiguo/scripts/87-quita-social.py @@ -0,0 +1,108 @@ +#!/usr/bin/env python3 +"""Quita los botones sociales del derivado `site/`. `raw/` no se toca. + +El bloque de K2 es uniforme en las 16.708 paginas que lo llevan: `
` +contiene el boton de Twitter, el de Facebook y un clearfix, nada mas (comprobado con +86-survey-social.py sobre 400 paginas: 0 variantes). Se elimina el bloque entero, asi no quedan +huecos ni botones rotos. + +Red de seguridad: fuera del bloque tambien se eliminan los ', + re.I) +FB_ROOT = re.compile(r'\s*
', re.I) +TW_ANCHOR = re.compile(r']*class="[^"]*twitter-share-button[^"]*"[^>]*>.*?', re.I | re.S) +FB_LIKE = re.compile(r']*class="[^"]*fb-like[^"]*"[^>]*>\s*', re.I) + +AVISO = ('') + + +def es_html(path, fn): + """Decidir por extension no basta en este mirror: conviven `x.html?tmpl=…` (la extension esta + antes de la query) y `x?.html` con su copia `x` sin extension (alias de K2 con '?' literal). + Para esos casos se mira el contenido, que es lo unico fiable.""" + base = fn.split("?", 1)[0].lower() + if base.endswith((".html", ".htm")) or fn.lower().endswith((".html", ".htm")): + return True + if "." in fn.split("/")[-1].split("?", 1)[0]: + return False # tiene otra extension (jpg, mp3, css…) + try: + with open(path, "rb") as f: + cabeza = f.read(512).lstrip().lower() + return cabeza.startswith(b"/dev/null | sort | uniq -c | sort -rn | head -5 +done +echo +echo "=== ids de Google en el repo feadulta (mu-plugins/scripts) ===" +for d in /home/rafa/feadulta /home/rafa/joomla-migration; do + [ -d "$d" ] || continue + grep -rhoE 'G-[A-Z0-9]{8,}|UA-[0-9]+-[0-9]+|GTM-[A-Z0-9]+' "$d" --include='*.php' --include='*.py' --include='*.md' 2>/dev/null | sort | uniq -c | sort -rn | head -5 +done diff --git a/mirror-antiguo/scripts/89-verifica-social.sh b/mirror-antiguo/scripts/89-verifica-social.sh new file mode 100644 index 0000000..370e916 --- /dev/null +++ b/mirror-antiguo/scripts/89-verifica-social.sh @@ -0,0 +1,17 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +S=$BASE/runs/$RUN/site/antiguo.feadulta.com +R=$BASE/runs/$RUN/raw/antiguo.feadulta.com +echo "=== site/ (derivado) ===" +printf 'connect.facebook.net : %s\n' "$(grep -rl 'connect.facebook.net' "$S" 2>/dev/null | wc -l)" +printf 'platform.twitter.com : %s\n' "$(grep -rl 'platform.twitter.com' "$S" 2>/dev/null | wc -l)" +printf 'itemSocialSharing : %s\n' "$(grep -rl 'itemSocialSharing' "$S" 2>/dev/null | wc -l)" +printf 'fb-root : %s\n' "$(grep -rl 'fb-root' "$S" 2>/dev/null | wc -l)" +echo +echo "=== raw/ (intacto, debe seguir teniendolos) ===" +printf 'connect.facebook.net : %s\n' "$(grep -rl 'connect.facebook.net' "$R" 2>/dev/null | wc -l)" +echo +echo "=== restos si los hay ===" +grep -rl 'connect.facebook.net\|platform.twitter.com' "$S" 2>/dev/null | head -5 diff --git a/mirror-antiguo/scripts/89b-los-dos.sh b/mirror-antiguo/scripts/89b-los-dos.sh new file mode 100644 index 0000000..3beb32d --- /dev/null +++ b/mirror-antiguo/scripts/89b-los-dos.sh @@ -0,0 +1,9 @@ +#!/bin/bash +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +S=$BASE/runs/$RUN/site/antiguo.feadulta.com +grep -rl 'itemSocialSharing' "$S" 2>/dev/null | while read -r f; do + echo "--- $f" + grep -o '.\{0,80\}itemSocialSharing.\{0,120\}' "$f" | head -3 +done diff --git a/mirror-antiguo/scripts/90-cierre.sh b/mirror-antiguo/scripts/90-cierre.sh new file mode 100644 index 0000000..f80ff31 --- /dev/null +++ b/mirror-antiguo/scripts/90-cierre.sh @@ -0,0 +1,23 @@ +#!/bin/bash +# Verificacion de cierre de la noche +set -uo pipefail +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +D=$BASE/runs/$RUN +echo "=== snapshot fuente (integridad) ===" +cd "$BASE/source" && sha256sum -c MANIFEST-source.sha256 +echo +echo "=== corrida ===" +echo "run: $RUN" +du -sh "$D/raw" "$D/site" +echo "raw: $(find "$D/raw" -type f | wc -l) ficheros" +echo "site: $(find "$D/site" -type f | wc -l) ficheros" +echo +echo "=== contenedores ===" +docker ps --format '{{.Names}}\t{{.Status}}' | sort +echo +echo "=== mirror servido en 8087 ===" +curl -s -o /dev/null -w 'portada: %{http_code} %{content_type}\n' http://127.0.0.1:8087/es/ +echo +echo "=== disco ===" +df -h /home | tail -1 diff --git a/mirror-antiguo/scripts/91-repone-assets404.sh b/mirror-antiguo/scripts/91-repone-assets404.sh new file mode 100644 index 0000000..ee43308 --- /dev/null +++ b/mirror-antiguo/scripts/91-repone-assets404.sh @@ -0,0 +1,70 @@ +#!/bin/bash +# Repone los assets que el crawl no capturo (404 en produccion, 200 en el origen). +# +# Entrada: lista de rutas absolutas (una por linea) sacada de los logs de nginx +# del Hetzner. Origen: el Joomla local restaurado y aislado (127.0.0.1:8086), +# por HTTP -- nunca copiando el filesystem, mismo principio que el crawl. +# +# Escribe en raw/ y en site/: raw/ es el archivo tal cual se capturo, site/ es +# el arbol servible que sincroniza 80-sync-hetzner.sh. +set -uo pipefail + +BASE=/home/rafa/joomla-migration/mirror-antiguo +RUN=$(cat "$BASE/CURRENT_RUN") +DIR=$BASE/runs/$RUN +RAW=$DIR/raw/antiguo.feadulta.com +SITE=$DIR/site/antiguo.feadulta.com +ORIGEN=http://127.0.0.1:8086 +IN=${1:-/tmp/assets404.txt} +OUT=$DIR/logs/repone-assets-$(date -u +%Y%m%dT%H%M%SZ) + +[ -s "$IN" ] || { echo "no hay lista de entrada: $IN"; exit 1; } +[ -d "$SITE" ] || { echo "no existe $SITE"; exit 1; } +mkdir -p "$OUT" + +ok=0; fail=0; skip=0; ya=0 +while IFS= read -r p; do + [ -n "$p" ] || continue + case "$p" in + /%22*|*'"'*) echo "$p" >> "$OUT/descartados.txt"; skip=$((skip+1)); continue ;; + esac + dest="$RAW$p" + if [ -f "$dest" ]; then echo "$p" >> "$OUT/ya-estaban.txt"; ya=$((ya+1)); continue; fi + mkdir -p "$(dirname "$dest")" 2>/dev/null || { echo "$p" >> "$OUT/fallidos.txt"; fail=$((fail+1)); continue; } + code=$(curl -s --path-as-is -m 30 -o "$dest.part" -w '%{http_code}' "$ORIGEN$p") + if [ "$code" = "200" ] && [ -s "$dest.part" ]; then + mv "$dest.part" "$dest" + echo "$p" >> "$OUT/repuestos.txt"; ok=$((ok+1)) + else + rm -f "$dest.part" + echo "$code $p" >> "$OUT/fallidos.txt"; fail=$((fail+1)) + fi +done < "$IN" + +echo "repuestos: $ok · fallidos: $fail · descartados: $skip · ya estaban: $ya" + +# --- escaneo de seguridad antes de copiar a site/ --- +echo "== escaneo de PHP embebido en lo descargado ==" +sospechosos=0 +if [ -s "$OUT/repuestos.txt" ]; then + while IFS= read -r p; do + if head -c 4096 "$RAW$p" 2>/dev/null | grep -qa '> "$OUT/sospechosos.txt"; sospechosos=$((sospechosos+1)) + fi + done < "$OUT/repuestos.txt" +fi +echo " sospechosos: $sospechosos" +if [ "$sospechosos" -gt 0 ]; then + echo "ABORTADO: hay ficheros con PHP embebido. No se copian a site/." + exit 1 +fi + +# --- copia a site/ --- +if [ -s "$OUT/repuestos.txt" ]; then + while IFS= read -r p; do + mkdir -p "$(dirname "$SITE$p")" + cp -p "$RAW$p" "$SITE$p" + done < "$OUT/repuestos.txt" +fi +echo "copiados a site/: $ok" +echo "detalle en: $OUT"