diff --git a/fuzz/Makefile.am b/fuzz/Makefile.am
index 6a42d477..8f77b1df 100644
--- a/fuzz/Makefile.am
+++ b/fuzz/Makefile.am
@@ -2,7 +2,7 @@
if FUZZERS
noinst_PROGRAMS = fuzz-charset fuzz-meta fuzz-idna fuzz-entities \
fuzz-unescape fuzz-filters fuzz-url fuzz-header fuzz-cachendx \
- fuzz-htsparse fuzz-singlefile
+ fuzz-htsparse fuzz-singlefile fuzz-sitemap
endif
AM_CPPFLAGS = \
@@ -28,6 +28,7 @@ fuzz_header_SOURCES = fuzz-header.c fuzz.h
fuzz_cachendx_SOURCES = fuzz-cachendx.c fuzz.h
fuzz_htsparse_SOURCES = fuzz-htsparse.c fuzz.h
fuzz_singlefile_SOURCES = fuzz-singlefile.c fuzz.h
+fuzz_sitemap_SOURCES = fuzz-sitemap.c fuzz.h
# List corpus files explicitly: automake does not expand EXTRA_DIST globs.
EXTRA_DIST = README.md run-fuzzers.sh \
@@ -52,4 +53,6 @@ EXTRA_DIST = README.md run-fuzzers.sh \
corpus/singlefile/img-src.html corpus/singlefile/link-rel.html \
corpus/singlefile/style-block.html corpus/singlefile/style-attr.html \
corpus/singlefile/srcset.html corpus/singlefile/rawtext.html \
- corpus/singlefile/malformed.html corpus/singlefile/many-attrs.html
+ corpus/singlefile/malformed.html corpus/singlefile/many-attrs.html \
+ corpus/sitemap/urlset.xml corpus/sitemap/sitemapindex.xml \
+ corpus/sitemap/truncated.xml corpus/sitemap/urlset.xml.gz
diff --git a/fuzz/corpus/sitemap/sitemapindex.xml b/fuzz/corpus/sitemap/sitemapindex.xml
new file mode 100644
index 00000000..70f470a3
--- /dev/null
+++ b/fuzz/corpus/sitemap/sitemapindex.xml
@@ -0,0 +1 @@
+http://h.test/s2.xml.gz
diff --git a/fuzz/corpus/sitemap/truncated.xml b/fuzz/corpus/sitemap/truncated.xml
new file mode 100644
index 00000000..8eefa62a
--- /dev/null
+++ b/fuzz/corpus/sitemap/truncated.xml
@@ -0,0 +1 @@
+http://h.test/x
\ No newline at end of file
diff --git a/fuzz/corpus/sitemap/urlset.xml b/fuzz/corpus/sitemap/urlset.xml
new file mode 100644
index 00000000..2c28be61
--- /dev/null
+++ b/fuzz/corpus/sitemap/urlset.xml
@@ -0,0 +1 @@
+http://h.test/a.htmlhttps://h.test/b?x=1&y=2
diff --git a/fuzz/corpus/sitemap/urlset.xml.gz b/fuzz/corpus/sitemap/urlset.xml.gz
new file mode 100644
index 00000000..6233acf0
Binary files /dev/null and b/fuzz/corpus/sitemap/urlset.xml.gz differ
diff --git a/fuzz/fuzz-sitemap.c b/fuzz/fuzz-sitemap.c
new file mode 100644
index 00000000..482e8918
--- /dev/null
+++ b/fuzz/fuzz-sitemap.c
@@ -0,0 +1,60 @@
+/* ------------------------------------------------------------ */
+/*
+HTTrack Website Copier, Offline Browser for Windows and Unix
+Copyright (C) 2026 Xavier Roche and other contributors
+
+SPDX-License-Identifier: GPL-3.0-or-later
+
+This program is free software: you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation, either version 3 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License
+along with this program. If not, see .
+
+Ethical use: we kindly ask that you NOT use this software to harvest email
+addresses or to collect any other private information about people. Doing so
+would dishonor our work and waste the many hours we have spent on it.
+
+Please visit our Website: http://www.httrack.com
+*/
+
+/* Fuzz the sitemap scanner (htssitemap.c): raw XML, gzip-framed bodies
+ and truncated streams all arrive here straight off the network. */
+#include "fuzz.h"
+#include "htssitemap.h"
+
+static hts_boolean sm_count(void *arg, const char *url) {
+ int *const n = (int *) arg;
+
+ (void) url;
+ (*n)++;
+ return HTS_TRUE;
+}
+
+int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) {
+ static const int caps[] = {0, 1, 16, HTS_SITEMAP_MAX_URLS_DOC};
+ hts_boolean is_index;
+ char *body;
+ int n = 0, cap;
+
+ if (size == 0)
+ return 0;
+ cap = caps[data[0] % (sizeof(caps) / sizeof(caps[0]))];
+ data++, size--;
+ /* A heap copy of exactly `size` bytes: the scanner must never rely on a
+ terminator, and ASan turns any overread into a report. */
+ body = malloct(size != 0 ? size : 1);
+ memcpy(body, data, size);
+
+ (void) hts_sitemap_scan(body, size, cap, &is_index, sm_count, &n);
+
+ freet(body);
+ return 0;
+}
diff --git a/html/cmdguide.html b/html/cmdguide.html
index 273dd96a..a916cf89 100644
--- a/html/cmdguide.html
+++ b/html/cmdguide.html
@@ -163,8 +163,26 @@
2. Scope: how far the crawl reaches
--near (-n)
Also fetch non-HTML files "near" a followed link, such as an image linked from a page you kept but hosted elsewhere.
--ext-depth (-%e)
How many levels of external links to follow once the crawl leaves your scope (default 0).
--test (-t)
Also HEAD-test links that fall outside the scope, which are normally refused, without downloading them: a way to see what scope is excluding.
+
--sitemap (-%m), --sitemap-url URL (-%mu)
Also take start URLs from the site's sitemap, for pages nothing links to. Off by default.
+
Link-following only finds what something links to. Anything a site publishes
+solely in its sitemap is invisible to HTTrack unless you ask for it.
+--sitemap reads the start host's robots.txt for
+Sitemap: lines and falls back to /sitemap.xml;
+--sitemap-url names one directly. Nested sitemapindex files
+and gzipped .xml.gz sitemaps are followed. The URLs found become start
+URLs with the full depth budget, but they still go through your filters and
+scope rules, so a sitemap cannot widen a crawl you deliberately narrowed. It is
+off by default because a sitemap can list thousands of pages nothing links
+to.
+
+
One surprise worth knowing: a sitemap you name with --sitemap-url,
+and one the site itself declares in robots.txt, are fetched even when
+robots.txt disallows that path, because naming or declaring a sitemap
+is an invitation to read it. Only the guessed /sitemap.xml obeys a
+Disallow. The URLs listed inside are gated normally either way.
+
The single most common surprise is "only the home page came down." That is
usually not a scope option at all: it is an off-host redirect. A start URL of
http://example.com/ that redirects to https://www.example.com/
diff --git a/html/httrack.man.html b/html/httrack.man.html
index 9b35ff5c..0dd8b13d 100644
--- a/html/httrack.man.html
+++ b/html/httrack.man.html
@@ -87,8 +87,8 @@
<file> add all scan rules located in this text
file (one scan rule per line) (--urllist <param>)
+
+
+
+
+
+
-%m
+
+
+
+
+
seed the crawl from the site’s sitemap (robots.txt
+Sitemap:, then /sitemap.xml); --sitemap-url URL names one
+explicitly. A sitemap you name, or one the site declares, is
+fetched even under robots.txt Disallow; only the guessed
+/sitemap.xml obeys it. The URLs found still pass every
+filter and scope rule (--sitemap)
KeepQueryOrder=${ztest:keepqueryorder:0:1}
StripQuery=${stripquery}
StoreAllInCache=${ztest:cache2:0:1}
+Sitemap=${ztest:sitemap:0:1}
+SitemapUrl=${sitemapurl}
Warc=${ztest:warc:0:1}
WarcFile=${warcfile}
Changes=${ztest:changes:0:1}
diff --git a/lang.def b/lang.def
index f6462ff4..9a2cb40b 100755
--- a/lang.def
+++ b/lang.def
@@ -1054,3 +1054,11 @@ LANG_SINGLEFILEMAX
Largest inlined asset (bytes):
LANG_SINGLEFILEMAXTIP
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
+LANG_SITEMAP
+Seed the crawl from the site's sitemap
+LANG_SITEMAPTIP
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+LANG_SITEMAPURL
+Sitemap address:
+LANG_SITEMAPURLTIP
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
diff --git a/lang/Bulgarian.txt b/lang/Bulgarian.txt
index 507421ab..461c5ace 100644
--- a/lang/Bulgarian.txt
+++ b/lang/Bulgarian.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
- ():
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
; 10485760 .
+Seed the crawl from the site's sitemap
+
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+ ( Sitemap: robots.txt, /sitemap.xml) URL .
+Sitemap address:
+ :
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+ , ; , robots.txt, /sitemap.xml.
diff --git a/lang/Castellano.txt b/lang/Castellano.txt
index c2949659..7c0ed58e 100644
--- a/lang/Castellano.txt
+++ b/lang/Castellano.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Tamao mximo del recurso incrustado (bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Un recurso mayor que este tamao conserva un enlace normal; djelo vaco para el valor predeterminado de 10485760 bytes.
+Seed the crawl from the site's sitemap
+Iniciar el rastreo desde el mapa del sitio
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Leer el mapa del sitio (lneas Sitemap: de robots.txt, luego /sitemap.xml) y aadir como direccin inicial cada URL que incluya.
+Sitemap address:
+Direccin del mapa del sitio:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Direccin de un mapa del sitio que leer en lugar de sondear el sitio; djelo vaco para sondear robots.txt y luego /sitemap.xml.
diff --git a/lang/Cesky.txt b/lang/Cesky.txt
index eb194a12..04da31e8 100644
--- a/lang/Cesky.txt
+++ b/lang/Cesky.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Nejvt vloen zdroj (bajty):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Zdroj vt ne tato velikost si ponech bn odkaz; ponechte przdn pro vchoz hodnotu 10485760 bajt.
+Seed the crawl from the site's sitemap
+Zahjit prochzen z mapy webu
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Nast mapu webu (dky Sitemap: v souboru robots.txt, pot /sitemap.xml) a pidat kadou uvedenou adresu URL jako vchoz.
+Sitemap address:
+Adresa mapy webu:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adresa mapy webu, kter se m nast msto zjiovn na webu; ponechte przdn pro zjitn z robots.txt a pot /sitemap.xml.
diff --git a/lang/Chinese-BIG5.txt b/lang/Chinese-BIG5.txt
index c0664649..e5015144 100644
--- a/lang/Chinese-BIG5.txt
+++ b/lang/Chinese-BIG5.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
O귽jpW]줸ա^G
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
WLjp귽|Od@sFdūhϥιw] 10485760 줸աC
+Seed the crawl from the site's sitemap
+q Sitemap }l^
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Ū Sitemap]robots.txt Sitemap: AM /sitemap.xml^AñN䤤CXCӺ}[J_l}C
+Sitemap address:
+Sitemap }G
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+nŪ Sitemap }AΨӨN۰ʱFdūh robots.txt A /sitemap.xmlC
diff --git a/lang/Chinese-Simplified.txt b/lang/Chinese-Simplified.txt
index 5ab74e2e..2c4837fb 100644
--- a/lang/Chinese-Simplified.txt
+++ b/lang/Chinese-Simplified.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
ǶԴСޣֽڣ
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
˴СԴᱣͨӣʹĬϵ 10485760 ֽڡ
+Seed the crawl from the site's sitemap
+վ Sitemap ʼץȡ
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+ȡվ Sitemaprobots.txt е Sitemap: УȻ /sitemap.xmlгÿַΪʼַ
+Sitemap address:
+Sitemap ַ
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Ҫȡ Sitemap ַԶ̽⣻̽ robots.txt ̽ /sitemap.xml
diff --git a/lang/Croatian.txt b/lang/Croatian.txt
index bf4a4397..f5b69f5f 100644
--- a/lang/Croatian.txt
+++ b/lang/Croatian.txt
@@ -978,3 +978,11 @@ Largest inlined asset (bytes):
Najvei ugraeni resurs (bajtovi):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Resurs vei od ove veliine zadrava obinu poveznicu; ostavite prazno za zadanih 10485760 bajtova.
+Seed the crawl from the site's sitemap
+Pokreni pretraivanje iz karte web-mjesta
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Proitaj kartu web-mjesta (retke Sitemap: iz robots.txt, zatim /sitemap.xml) i dodaj svaki navedeni URL kao poetnu adresu.
+Sitemap address:
+Adresa karte web-mjesta:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adresa karte web-mjesta koju treba proitati umjesto ispitivanja web-mjesta; ostavite prazno za ispitivanje robots.txt pa /sitemap.xml.
diff --git a/lang/Dansk.txt b/lang/Dansk.txt
index 96261c90..f995c72a 100644
--- a/lang/Dansk.txt
+++ b/lang/Dansk.txt
@@ -1024,3 +1024,11 @@ Largest inlined asset (bytes):
Strste indlejrede ressource (byte):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
En ressource over denne strrelse beholder et almindeligt link; lad feltet st tomt for standardvrdien p 10485760 byte.
+Seed the crawl from the site's sitemap
+Start gennemgangen fra webstedets sitemap
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Ls webstedets sitemap (Sitemap:-linjer i robots.txt, derefter /sitemap.xml) og tilfj hver angivet URL som startadresse.
+Sitemap address:
+Sitemap-adresse:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adressen p et sitemap, der skal lses i stedet for at undersge webstedet; lad feltet st tomt for at undersge robots.txt og derefter /sitemap.xml.
diff --git a/lang/Deutsch.txt b/lang/Deutsch.txt
index 141c1edf..3f14c762 100644
--- a/lang/Deutsch.txt
+++ b/lang/Deutsch.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Grte eingebettete Ressource (Bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Eine Ressource ber dieser Gre behlt einen gewhnlichen Link; leer lassen fr den Standardwert von 10485760 Bytes.
+Seed the crawl from the site's sitemap
+Erfassung mit der Sitemap der Website beginnen
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Die Sitemap der Website lesen (Sitemap:-Zeilen in robots.txt, dann /sitemap.xml) und jede dort aufgefhrte URL als Startadresse hinzufgen.
+Sitemap address:
+Sitemap-Adresse:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adresse einer Sitemap, die anstelle der Suche auf der Website gelesen wird; leer lassen, um robots.txt und dann /sitemap.xml zu prfen.
diff --git a/lang/Eesti.txt b/lang/Eesti.txt
index 31b1d377..c2da2f15 100644
--- a/lang/Eesti.txt
+++ b/lang/Eesti.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Suurim manustatud ressurss (baiti):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Sellest suurem ressurss silitab tavalise lingi; jta thjaks vaikevrtuse 10485760 baiti jaoks.
+Seed the crawl from the site's sitemap
+Alusta kogumist saidi saidikaardist
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Loe saidi saidikaarti (robots.txt-i Sitemap:-read, seejrel /sitemap.xml) ja lisa iga seal loetletud URL alguslingina.
+Sitemap address:
+Saidikaardi aadress:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Saidikaardi aadress, mida lugeda saidi sondeerimise asemel; jta thjaks, et kontrollida robots.txt-i ja seejrel /sitemap.xml-i.
diff --git a/lang/English.txt b/lang/English.txt
index 8832e1d9..db48dd67 100644
--- a/lang/English.txt
+++ b/lang/English.txt
@@ -1024,3 +1024,11 @@ Largest inlined asset (bytes):
Largest inlined asset (bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
+Seed the crawl from the site's sitemap
+Seed the crawl from the site's sitemap
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Sitemap address:
+Sitemap address:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
diff --git a/lang/Finnish.txt b/lang/Finnish.txt
index 18b6dfa8..a6b08cf3 100644
--- a/lang/Finnish.txt
+++ b/lang/Finnish.txt
@@ -978,3 +978,11 @@ Largest inlined asset (bytes):
Suurin upotettu resurssi (tavua):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Tt suurempi resurssi silytt tavallisen linkin; jt tyhjksi, jolloin kytetn oletusarvoa 10485760 tavua.
+Seed the crawl from the site's sitemap
+Aloita haku sivuston sivukartasta
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Lue sivuston sivukartta (robots.txt-tiedoston Sitemap:-rivit, sitten /sitemap.xml) ja lis jokainen siin lueteltu URL-osoite aloitusosoitteeksi.
+Sitemap address:
+Sivukartan osoite:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Luettavan sivukartan osoite sivuston luotaamisen sijaan; jt tyhjksi, jolloin tarkistetaan robots.txt ja sitten /sitemap.xml.
diff --git a/lang/Francais.txt b/lang/Francais.txt
index 9bdf3213..20d19e73 100644
--- a/lang/Francais.txt
+++ b/lang/Francais.txt
@@ -1024,3 +1024,11 @@ Largest inlined asset (bytes):
Taille maximale d'une ressource intgre (octets) :
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Au-del de cette taille, la ressource reste un lien ordinaire ; laissez vide pour la valeur par dfaut de 10485760 octets.
+Seed the crawl from the site's sitemap
+Partir du plan de site (sitemap)
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Lire le plan de site (lignes Sitemap: de robots.txt, puis /sitemap.xml) et ajouter chaque URL liste comme adresse de dpart.
+Sitemap address:
+Adresse du plan de site :
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adresse d'un plan de site lire au lieu de sonder le site ; laissez vide pour sonder robots.txt puis /sitemap.xml.
diff --git a/lang/Greek.txt b/lang/Greek.txt
index d75e3af7..9b450e57 100644
--- a/lang/Greek.txt
+++ b/lang/Greek.txt
@@ -978,3 +978,11 @@ Largest inlined asset (bytes):
(byte):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
. 10485760 byte.
+Seed the crawl from the site's sitemap
+
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+ ( Sitemap: robots.txt, /sitemap.xml) URL .
+Sitemap address:
+ :
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+ . robots.txt /sitemap.xml.
diff --git a/lang/Italiano.txt b/lang/Italiano.txt
index bbd01240..1ead9e9f 100644
--- a/lang/Italiano.txt
+++ b/lang/Italiano.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Dimensione massima della risorsa incorporata (byte):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Una risorsa oltre questa dimensione mantiene un collegamento normale; lasciare vuoto per il valore predefinito di 10485760 byte.
+Seed the crawl from the site's sitemap
+Avvia la scansione dalla mappa del sito
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Legge la mappa del sito (righe Sitemap: in robots.txt, poi /sitemap.xml) e aggiunge come indirizzo iniziale ogni URL elencato.
+Sitemap address:
+Indirizzo della mappa del sito:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Indirizzo di una mappa del sito da leggere invece di sondare il sito; lasciare vuoto per sondare robots.txt e poi /sitemap.xml.
diff --git a/lang/Japanese.txt b/lang/Japanese.txt
index c4a20d98..c318a97d 100644
--- a/lang/Japanese.txt
+++ b/lang/Japanese.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
ߍލőTCY (oCg):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
̃TCY郊\[X͒ʏ̃N̂܂܂ɂȂ܂BɂƊl 10485760 oCgɂȂ܂B
+Seed the crawl from the site's sitemap
+TCg}bv~[OJn
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+TCg}bv (robots.txt Sitemap: sA /sitemap.xml) ǂݍ݁ALڂĂ邷ׂĂ URL JnAhXƂĒlj܂B
+Sitemap address:
+TCg}bṽAhX:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+TCgTɓǂݍރTCg}bṽAhXBɂ robots.txtA /sitemap.xml T܂B
diff --git a/lang/Macedonian.txt b/lang/Macedonian.txt
index 11fa04cf..66d92da5 100644
--- a/lang/Macedonian.txt
+++ b/lang/Macedonian.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
():
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
; 10485760 .
+Seed the crawl from the site's sitemap
+
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+ ( Sitemap: robots.txt, /sitemap.xml) URL .
+Sitemap address:
+ :
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+ ; robots.txt, /sitemap.xml.
diff --git a/lang/Magyar.txt b/lang/Magyar.txt
index e0cfea3e..2b42419c 100644
--- a/lang/Magyar.txt
+++ b/lang/Magyar.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Legnagyobb begyazott erforrs (bjt):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Az ennl nagyobb erforrs kznsges hivatkozs marad; hagyja resen a 10485760 bjtos alaprtelmezshez.
+Seed the crawl from the site's sitemap
+A letlts indtsa a webhely webhelytrkprl
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+A webhely webhelytrkpnek beolvassa (a robots.txt Sitemap: sorai, majd a /sitemap.xml), s a benne felsorolt sszes URL felvtele kiindulsi cmknt.
+Sitemap address:
+Webhelytrkp cme:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+A webhely vizsglata helyett beolvasand webhelytrkp cme; hagyja resen a robots.txt, majd a /sitemap.xml vizsglathoz.
diff --git a/lang/Nederlands.txt b/lang/Nederlands.txt
index 6488f26a..2eea63c1 100644
--- a/lang/Nederlands.txt
+++ b/lang/Nederlands.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Grootste ingesloten bron (bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Een bron boven deze grootte houdt een gewone koppeling; laat leeg voor de standaardwaarde van 10485760 bytes.
+Seed the crawl from the site's sitemap
+De crawl starten vanaf de sitemap van de site
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+De sitemap van de site lezen (Sitemap:-regels in robots.txt, daarna /sitemap.xml) en elke vermelde URL als startadres toevoegen.
+Sitemap address:
+Sitemap-adres:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adres van een sitemap die gelezen moet worden in plaats van de site te onderzoeken; laat leeg om robots.txt en daarna /sitemap.xml te controleren.
diff --git a/lang/Norsk.txt b/lang/Norsk.txt
index 3e4f5bf9..f58659d1 100644
--- a/lang/Norsk.txt
+++ b/lang/Norsk.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Strste innebygde ressurs (byte):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
En ressurs over denne strrelsen beholder en vanlig lenke; la feltet st tomt for standardverdien p 10485760 byte.
+Seed the crawl from the site's sitemap
+Start gjennomgangen fra nettstedets nettstedskart
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Les nettstedets nettstedskart (Sitemap:-linjer i robots.txt, deretter /sitemap.xml) og legg til hver oppfrt URL som startadresse.
+Sitemap address:
+Adresse til nettstedskart:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adressen til et nettstedskart som skal leses i stedet for underske nettstedet; la feltet st tomt for underske robots.txt og deretter /sitemap.xml.
diff --git a/lang/Polski.txt b/lang/Polski.txt
index 6c792098..89298904 100644
--- a/lang/Polski.txt
+++ b/lang/Polski.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Najwikszy osadzony zasb (bajty):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Zasb wikszy ni ten rozmiar zachowuje zwyky odnonik; pozostaw puste, aby uy domylnych 10485760 bajtw.
+Seed the crawl from the site's sitemap
+Rozpocznij pobieranie od mapy witryny
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Odczytaj map witryny (wiersze Sitemap: w pliku robots.txt, nastpnie /sitemap.xml) i dodaj kady wymieniony adres URL jako adres pocztkowy.
+Sitemap address:
+Adres mapy witryny:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adres mapy witryny do odczytania zamiast sondowania witryny; pozostaw puste, aby sprawdzi robots.txt, a nastpnie /sitemap.xml.
diff --git a/lang/Portugues-Brasil.txt b/lang/Portugues-Brasil.txt
index fc42701f..299a45a8 100644
--- a/lang/Portugues-Brasil.txt
+++ b/lang/Portugues-Brasil.txt
@@ -1024,3 +1024,11 @@ Largest inlined asset (bytes):
Maior recurso incorporado (bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Um recurso acima desse tamanho mantm um link comum; deixe em branco para o padro de 10485760 bytes.
+Seed the crawl from the site's sitemap
+Iniciar a captura pelo mapa do site
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Ler o mapa do site (linhas Sitemap: do robots.txt, depois /sitemap.xml) e adicionar como endereo inicial cada URL nele listada.
+Sitemap address:
+Endereo do mapa do site:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Endereo de um mapa do site a ser lido em vez de sondar o site; deixe em branco para sondar robots.txt e depois /sitemap.xml.
diff --git a/lang/Portugues.txt b/lang/Portugues.txt
index 7aa597e9..65982176 100644
--- a/lang/Portugues.txt
+++ b/lang/Portugues.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Maior recurso incorporado (bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Um recurso acima deste tamanho mantm uma ligao normal; deixe em branco para o valor predefinido de 10485760 bytes.
+Seed the crawl from the site's sitemap
+Iniciar a recolha pelo mapa do site
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Ler o mapa do site (linhas Sitemap: do robots.txt, depois /sitemap.xml) e adicionar como endereo inicial cada URL nele listado.
+Sitemap address:
+Endereo do mapa do site:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Endereo de um mapa do site a ler em vez de sondar o site; deixe em branco para sondar robots.txt e depois /sitemap.xml.
diff --git a/lang/Romanian.txt b/lang/Romanian.txt
index f99359c6..f2dd7e79 100644
--- a/lang/Romanian.txt
+++ b/lang/Romanian.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Cea mai mare resursa ncorporata (octeti):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
O resursa mai mare dect aceasta dimensiune pastreaza o legatura obisnuita; lasati gol pentru valoarea implicita de 10485760 de octeti.
+Seed the crawl from the site's sitemap
+Porneste explorarea de la harta sitului
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Citeste harta sitului (liniile Sitemap: din robots.txt, apoi /sitemap.xml) si adauga fiecare URL listat ca adresa de pornire.
+Sitemap address:
+Adresa hartii sitului:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adresa unei harti a sitului care sa fie citita n loc de sondarea sitului; lasati gol pentru a sonda robots.txt, apoi /sitemap.xml.
diff --git a/lang/Russian.txt b/lang/Russian.txt
index cdab3064..59588d0e 100644
--- a/lang/Russian.txt
+++ b/lang/Russian.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
():
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
; 10485760 .
+Seed the crawl from the site's sitemap
+
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+ ( Sitemap: robots.txt, /sitemap.xml) URL .
+Sitemap address:
+ :
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+ , ; , robots.txt, /sitemap.xml.
diff --git a/lang/Slovak.txt b/lang/Slovak.txt
index 86e990f3..26aa7a96 100644
--- a/lang/Slovak.txt
+++ b/lang/Slovak.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Najv vloen zdroj (bajty):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Zdroj v ne tto vekos si ponech ben odkaz; ponechajte przdne pre predvolench 10485760 bajtov.
+Seed the crawl from the site's sitemap
+Zaa prehliadanie z mapy strnok
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Nata mapu strnok (riadky Sitemap: v sbore robots.txt, potom /sitemap.xml) a prida kad uveden adresu URL ako poiaton.
+Sitemap address:
+Adresa mapy strnok:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adresa mapy strnok, ktor sa m nata namiesto zisovania na strnke; ponechajte przdne na zistenie z robots.txt a potom /sitemap.xml.
diff --git a/lang/Slovenian.txt b/lang/Slovenian.txt
index b333b466..d38131cf 100644
--- a/lang/Slovenian.txt
+++ b/lang/Slovenian.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Najvecji vgrajeni vir (bajti):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Vir, vecji od te velikosti, ohrani obicajno povezavo; pustite prazno za privzetih 10485760 bajtov.
+Seed the crawl from the site's sitemap
+Zacni zajem z zemljevidom spletnega mesta
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Preberi zemljevid spletnega mesta (vrstice Sitemap: v robots.txt, nato /sitemap.xml) in dodaj vsak navedeni URL kot zacetni naslov.
+Sitemap address:
+Naslov zemljevida spletnega mesta:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Naslov zemljevida spletnega mesta, ki naj se prebere namesto preverjanja mesta; pustite prazno za preverjanje robots.txt in nato /sitemap.xml.
diff --git a/lang/Svenska.txt b/lang/Svenska.txt
index 0daeac28..078edeec 100644
--- a/lang/Svenska.txt
+++ b/lang/Svenska.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Strsta inbddade resurs (byte):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
En resurs ver den hr storleken behller en vanlig lnk; lmna tomt fr standardvrdet 10485760 byte.
+Seed the crawl from the site's sitemap
+Starta insamlingen frn webbplatsens webbplatskarta
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Ls webbplatsens webbplatskarta (Sitemap:-rader i robots.txt, sedan /sitemap.xml) och lgg till varje angiven URL som startadress.
+Sitemap address:
+Webbplatskartans adress:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Adress till en webbplatskarta som ska lsas i stllet fr att ska p webbplatsen; lmna tomt fr att kontrollera robots.txt och sedan /sitemap.xml.
diff --git a/lang/Turkish.txt b/lang/Turkish.txt
index 9af30624..e9c31c2b 100644
--- a/lang/Turkish.txt
+++ b/lang/Turkish.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
En byk gml kaynak (bayt):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Bu boyutun zerindeki bir kaynak sradan balantsn korur; 10485760 baytlk varsaylan iin bo brakn.
+Seed the crawl from the site's sitemap
+Taramay sitenin site haritasndan balat
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Sitenin site haritasn oku (robots.txt iindeki Sitemap: satrlar, ardndan /sitemap.xml) ve listelenen her URL'yi balang adresi olarak ekle.
+Sitemap address:
+Site haritas adresi:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Siteyi yoklamak yerine okunacak site haritasnn adresi; robots.txt ve ardndan /sitemap.xml yoklamas iin bo brakn.
diff --git a/lang/Ukrainian.txt b/lang/Ukrainian.txt
index a84c3c95..2ec7a996 100644
--- a/lang/Ukrainian.txt
+++ b/lang/Ukrainian.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
():
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
, , ; 10485760 .
+Seed the crawl from the site's sitemap
+
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+ ( Sitemap: robots.txt, /sitemap.xml) URL- .
+Sitemap address:
+ :
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+ , ; , robots.txt, /sitemap.xml.
diff --git a/lang/Uzbek.txt b/lang/Uzbek.txt
index 52cf0cf3..17a1a0ab 100644
--- a/lang/Uzbek.txt
+++ b/lang/Uzbek.txt
@@ -976,3 +976,11 @@ Largest inlined asset (bytes):
Eng katta joylangan resurs (bayt):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Bu olchamdan katta resurs oddiy havolani saqlab qoladi; standart 10485760 bayt uchun bosh qoldiring.
+Seed the crawl from the site's sitemap
+Yigishni saytning sayt xaritasidan boshlash
+Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
+Saytning sayt xaritasini oqish (robots.txt dagi Sitemap: qatorlari, songra /sitemap.xml) va unda korsatilgan har bir URL manzilni boshlangich manzil sifatida qoshish.
+Sitemap address:
+Sayt xaritasi manzili:
+Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
+Saytni tekshirish orniga oqiladigan sayt xaritasi manzili; robots.txt, songra /sitemap.xml ni tekshirish uchun bosh qoldiring.
diff --git a/man/httrack.1 b/man/httrack.1
index 60a32474..0191e0c6 100644
--- a/man/httrack.1
+++ b/man/httrack.1
@@ -36,6 +36,7 @@ httrack \- offline browser : copy websites to a local directory
[ \fB\-t, \-\-test\fR ]
[ \fB\-%L, \-\-list\fR ]
[ \fB\-%S, \-\-urllist\fR ]
+[ \fB\-%m, \-\-sitemap\fR ]
[ \fB\-NN, \-\-structure[=N]\fR ]
[ \fB\-%N, \-\-delayed\-type\-check\fR ]
[ \fB\-%D, \-\-cached\-delayed\-type\-check\fR ]
@@ -189,6 +190,8 @@ test all URLs (even forbidden ones) (\-\-test)
add all URL located in this text file (one URL per line) (\-\-list )
.IP \-%S
add all scan rules located in this text file (one scan rule per line) (\-\-urllist )
+.IP \-%m
+seed the crawl from the site's sitemap (robots.txt Sitemap:, then /sitemap.xml); \-\-sitemap\-url URL names one explicitly. A sitemap you name, or one the site declares, is fetched even under robots.txt Disallow; only the guessed /sitemap.xml obeys it. The URLs found still pass every filter and scope rule (\-\-sitemap)
.SS Build options:
.IP \-NN
structure type (0 *original structure, 1+: see below) (\-\-structure[=N])
diff --git a/src/Makefile.am b/src/Makefile.am
index c9c971c0..51e1880a 100644
--- a/src/Makefile.am
+++ b/src/Makefile.am
@@ -66,7 +66,7 @@ libhttrack_la_SOURCES = htscore.c htsparse.c htsback.c htscache.c \
htscmdline.c htshelp.c htslib.c htsurlport.c htscoremain.c \
htsname.c htsrobots.c htstools.c htswizard.c \
htsalias.c htsthread.c htsindex.c htsbauth.c \
- htsmd5.c htscodec.c htswarc.c htschanges.c htssinglefile.c htsproxy.c htszlib.c htswrap.c htsconcat.c \
+ htsmd5.c htscodec.c htswarc.c htschanges.c htssinglefile.c htssitemap.c htsproxy.c htszlib.c htswrap.c htsconcat.c \
htsmodules.c htscharset.c punycode.c htsencoding.c htssniff.c \
md5.c \
minizip/ioapi.c minizip/mztools.c minizip/unzip.c minizip/zip.c \
@@ -77,7 +77,7 @@ libhttrack_la_SOURCES = htscore.c htsparse.c htsback.c htscache.c \
htshelp.h htsindex.h htslib.h htsurlport.h htsmd5.h \
htsmodules.h htsname.h htsnet.h htssniff.h \
htsopt.h htsrobots.h htsthread.h \
- htstools.h htswizard.h htswrap.h htscodec.h htswarc.h htschanges.h htssinglefile.h htsproxy.h htszlib.h \
+ htstools.h htswizard.h htswrap.h htscodec.h htswarc.h htschanges.h htssinglefile.h htssitemap.h htsproxy.h htszlib.h \
htsstrings.h htsarrays.h httrack-library.h \
htscharset.h punycode.h htsencoding.h \
htsentities.h htsentities.sh htsbasiccharsets.sh htscodepages.h \
diff --git a/src/htsalias.c b/src/htsalias.c
index e646d83a..7b4052f5 100644
--- a/src/htsalias.c
+++ b/src/htsalias.c
@@ -116,6 +116,10 @@ const char *hts_optalias[][4] = {
"load extra cookies from a Netscape cookies.txt"},
{"changes", "-%d", "single",
"write hts-changes.json: what this crawl changed vs. the previous mirror"},
+ {"sitemap", "-%m", "single",
+ "seed the crawl from the start host's sitemap (robots.txt, then "
+ "/sitemap.xml)"},
+ {"sitemap-url", "-%mu", "param1", "seed the crawl from this sitemap URL"},
{"warc", "-%r", "single", "write an ISO-28500 WARC/1.1 archive of the crawl"},
{"warc-file", "-%rf", "param1", "write a WARC archive to the given base name"},
{"warc-max-size", "-%rs", "param1",
diff --git a/src/htscore.c b/src/htscore.c
index 25817936..a904c6fe 100644
--- a/src/htscore.c
+++ b/src/htscore.c
@@ -39,6 +39,7 @@ Please visit our Website: http://www.httrack.com
/* File defs */
#include "htscore.h"
+#include "htssitemap.h"
#include "htswarc.h"
#include "htschanges.h"
#include "htssinglefile.h"
@@ -949,6 +950,22 @@ int httpmirror(char *url1, httrackp * opt) {
heap_top()->premier = heap_top_index(); // premier lien, objet-père=objet
heap_top()->precedent = heap_top_index(); // lien précédent
+ /* --sitemap: queue the sitemap probe just after the seeds, so its URLs are
+ injected before the crawl gets far. */
+ hts_sitemap_free(opt); /* an earlier mirror may have left a doc list */
+ if (opt->sitemap || StringNotEmpty(opt->sitemap_url)) {
+ char BIGSTK first[HTS_URLMAXSIZE * 2];
+ const char *const eol = strchr(primary, '\n');
+ const size_t len = eol != NULL ? (size_t) (eol - primary) : 0;
+
+ first[0] = '\0';
+ if (len > 0 && len < sizeof(first)) {
+ memcpy(first, primary, len);
+ first[len] = '\0';
+ }
+ hts_sitemap_seed(opt, first);
+ }
+
// Initialiser cache
{
opt->state._hts_in_html_parsing = 4;
@@ -1593,11 +1610,21 @@ int httpmirror(char *url1, httrackp * opt) {
stre.maketrack_fp = maketrack_fp;
/* Parse */
- if (hts_mirror_check_moved(&str, &stre) != 0) {
- XH_uninit;
- return -1;
- }
+ {
+ const int nlinks = opt->lien_tot;
+ if (hts_mirror_check_moved(&str, &stre) != 0) {
+ XH_uninit;
+ return -1;
+ }
+ /* A redirect re-queues the target as a fresh link; without carrying
+ the marking over, a moved sitemap is fetched and then ignored. */
+ if (opt->sitemap_state != NULL && opt->lien_tot > nlinks &&
+ hts_sitemap_pending(opt, urladr(), urlfil())) {
+ hts_sitemap_redirect(opt, urladr(), urlfil(), heap_top()->adr,
+ heap_top()->fil);
+ }
+ }
}
} // if !error
@@ -1615,6 +1642,29 @@ int httpmirror(char *url1, httrackp * opt) {
/* Load file and decode if necessary, after redirect check. */
LOAD_IN_MEMORY_IF_NECESSARY();
+ /* Sitemap document: turn its URLs into top-level seeds. They go
+ through htsAddLink, so the wizard's filters and scope rules decide, and
+ this link's max depth leaves them the full budget. */
+ if (opt->sitemap_state != NULL &&
+ hts_sitemap_pending(opt, urladr(), urlfil())) {
+ htsmoduleStruct BIGSTK smstr;
+ int smptr = ptr;
+
+ memset(&smstr, 0, sizeof(smstr));
+ smstr.opt = opt;
+ smstr.sback = sback;
+ smstr.cache = &cache;
+ smstr.hashptr = hashptr;
+ smstr.numero_passe = numero_passe;
+ smstr.ptr_ = &smptr; /* scratch: the ingester retargets the wizard */
+ smstr.addLink = htsAddLink;
+ smstr.url_host = urladr();
+ smstr.url_file = urlfil();
+ smstr.mime = r.contenttype;
+ hts_sitemap_ingest(opt, &smstr, urladr(), urlfil(), r.adr,
+ r.adr != NULL && r.size > 0 ? (size_t) r.size : 0);
+ }
+
// ------------------------------------------------------
// ok, fichier chargé localement
// ------------------------------------------------------
@@ -1825,6 +1875,9 @@ int httpmirror(char *url1, httrackp * opt) {
if (strnotempty(savename()) == 0) { // pas de chemin de sauvegarde
if (strcmp(urlfil(), "/robots.txt") == 0) { // robots.txt
+ char BIGSTK sitemaps[8192];
+
+ sitemaps[0] = '\0';
if (r.adr) {
char BIGSTK infobuff[8192];
#ifdef IGNORE_RESTRICTIVE_ROBOTS
@@ -1836,7 +1889,8 @@ int httpmirror(char *url1, httrackp * opt) {
#endif
robots_parse(&robots, urladr(), r.adr, r.size, infobuff,
- sizeof(infobuff), keep_root);
+ sizeof(infobuff), keep_root, sitemaps,
+ sizeof(sitemaps));
if (strnotempty(infobuff)) {
hts_log_print(opt, LOG_INFO,
"Note: robots.txt forbidden links for %s are: %s",
@@ -1846,6 +1900,10 @@ int httpmirror(char *url1, httrackp * opt) {
urladr(), infobuff);
}
}
+ /* After robots_parse, so the rules this very body carries already
+ gate the sitemap fetch. Runs even on a failed probe, which is
+ what falls back to the well-known location. */
+ hts_sitemap_robots(opt, urladr(), sitemaps);
}
} else if (r.is_write) { // déja sauvé sur disque
/*
@@ -2286,6 +2344,7 @@ int httpmirror(char *url1, httrackp * opt) {
usercommand(opt, 0, NULL, NULL, NULL, NULL);
warc_close_opt(opt);
hts_changes_close_opt(opt);
+ hts_sitemap_free(opt);
// désallocation mémoire & buffers
XH_uninit;
@@ -3684,6 +3743,10 @@ HTSEXT_API int copy_htsopt(const httrackp * from, httrackp * to) {
to->single_file = from->single_file;
if (from->single_file_max_size > 0)
to->single_file_max_size = from->single_file_max_size;
+ if (from->sitemap)
+ to->sitemap = from->sitemap;
+ if (StringNotEmpty(from->sitemap_url))
+ StringCopyS(to->sitemap_url, from->sitemap_url);
if (from->pause_max_ms > 0) {
to->pause_min_ms = from->pause_min_ms;
diff --git a/src/htscoremain.c b/src/htscoremain.c
index 19cbcaf1..10c87f0f 100644
--- a/src/htscoremain.c
+++ b/src/htscoremain.c
@@ -1833,6 +1833,26 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
}
}
break;
+ case 'm': // sitemap / sitemap-url: seed the crawl from sitemaps
+ if (*(com + 1) == 'u') { // --sitemap-url URL: explicit sitemap
+ com++;
+ if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
+ HTS_PANIC_PRINTF(
+ "Option sitemap-url needs a blank space and a URL");
+ htsmain_free();
+ return -1;
+ }
+ na++;
+ if (strlen(argv[na]) >= HTS_URLMAXSIZE) {
+ HTS_PANIC_PRINTF("Sitemap URL too long");
+ htsmain_free();
+ return -1;
+ }
+ StringCopy(opt->sitemap_url, argv[na]);
+ } else { // --sitemap: robots.txt probe, then /sitemap.xml
+ opt->sitemap = HTS_TRUE;
+ }
+ break;
case 'Y': // why: explain the filter verdict for a URL, no crawl
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
HTS_PANIC_PRINTF("Option why needs a blank space and a URL");
diff --git a/src/htshelp.c b/src/htshelp.c
index 9d550e70..93a5ecb3 100644
--- a/src/htshelp.c
+++ b/src/htshelp.c
@@ -526,6 +526,11 @@ void help(const char *app, int more) {
(" %L add all URL located in this text file (one URL per line)");
infomsg
(" %S add all scan rules located in this text file (one scan rule per line)");
+ infomsg(" %m seed the crawl from the site's sitemap (robots.txt Sitemap:, "
+ "then /sitemap.xml); --sitemap-url URL names one explicitly. A "
+ "sitemap you name, or one the site declares, is fetched even under "
+ "robots.txt Disallow; only the guessed /sitemap.xml obeys it. The "
+ "URLs found still pass every filter and scope rule");
infomsg("");
infomsg("Build options:");
infomsg(" NN structure type (0 *original structure, 1+: see below)");
diff --git a/src/htslib.c b/src/htslib.c
index aab7dacc..a46a046c 100644
--- a/src/htslib.c
+++ b/src/htslib.c
@@ -36,6 +36,7 @@ Please visit our Website: http://www.httrack.com
// Fichier librairie .c
#include "htscore.h"
+#include "htssitemap.h"
#include "htswarc.h"
#include "htschanges.h"
#include "htssinglefile.h"
@@ -6030,6 +6031,7 @@ HTSEXT_API httrackp *hts_create_opt(void) {
StringCopy(opt->strip_query, "");
StringCopy(opt->cookies_file, "");
StringCopy(opt->warc_file, "");
+ StringCopy(opt->sitemap_url, "");
opt->warc_max_size = 0; /* no rotation unless --warc-max-size sets it */
opt->changes = HTS_FALSE;
opt->changes_state = NULL;
@@ -6187,6 +6189,8 @@ HTSEXT_API void hts_free_opt(httrackp * opt) {
StringFree(opt->cookies_file);
StringFree(opt->why_url);
StringFree(opt->warc_file);
+ StringFree(opt->sitemap_url);
+ hts_sitemap_free(opt); /* backstop: httpmirror's early-return paths */
hts_changes_free_opt(opt);
diff --git a/src/htsopt.h b/src/htsopt.h
index e3db3e43..365a77be 100644
--- a/src/htsopt.h
+++ b/src/htsopt.h
@@ -557,6 +557,13 @@ struct httrackp {
LLint single_file_max_size; /**< --single-file-max-size: per-asset cap in
bytes; a bigger asset stays a link.
Tail: ABI */
+ hts_boolean sitemap; /**< --sitemap: probe the start host's robots.txt for
+ Sitemap: lines, else /sitemap.xml. Tail: ABI */
+ String sitemap_url; /**< --sitemap-url: sitemap to ingest. Tail: ABI */
+ /* Live state, not an option: copy_htsopt must leave it alone. It sits here
+ rather than in htsoptstate because that struct is embedded by value, so
+ growing it would shift every httrackp field declared after it. */
+ void *sitemap_state; /**< hts_sitemap_state*, or NULL. Tail: ABI */
};
/* Running statistics for a mirror. */
diff --git a/src/htsrobots.c b/src/htsrobots.c
index d723c2ed..801469ea 100644
--- a/src/htsrobots.c
+++ b/src/htsrobots.c
@@ -147,7 +147,8 @@ static void robots_blob_add(char *blob, size_t blobsize, char marker,
void robots_parse(robots_wizard *robots, const char *adr, const char *body,
size_t bodysize, char *info, size_t infosize,
- hts_boolean keep_root_disallow) {
+ hts_boolean keep_root_disallow, char *sitemaps,
+ size_t sitemapsize) {
size_t bptr = 0;
int record = 0;
char BIGSTK line[1024];
@@ -156,6 +157,8 @@ void robots_parse(robots_wizard *robots, const char *adr, const char *body,
blob[0] = '\0';
if (info != NULL && infosize > 0)
info[0] = '\0';
+ if (sitemaps != NULL && sitemapsize > 0)
+ sitemaps[0] = '\0';
#if DEBUG_ROBOTS
printf("robots.txt dump:\n%s\n", body);
#endif
@@ -172,7 +175,19 @@ void robots_parse(robots_wizard *robots, const char *adr, const char *body,
line[llen - 1] = '\0';
llen--;
}
- if (strfield(line, "user-agent:")) {
+ if (sitemaps != NULL && strfield(line, "sitemap:")) {
+ // group-independent record (RFC 9309): collected whatever the group
+ char *a = line + 8;
+
+ while (is_realspace(*a))
+ a++;
+ /* A line at the buffer limit was truncated: a half URL is not one. */
+ if (strnotempty(a) && strlen(line) < sizeof(line) - 3 &&
+ strlen(a) + 2 < sitemapsize - strlen(sitemaps)) {
+ strlcatbuff(sitemaps, a, sitemapsize);
+ strlcatbuff(sitemaps, "\n", sitemapsize);
+ }
+ } else if (strfield(line, "user-agent:")) {
char *a = line + 11;
while (is_realspace(*a))
diff --git a/src/htsrobots.h b/src/htsrobots.h
index e2d1fb3e..84ef58aa 100644
--- a/src/htsrobots.h
+++ b/src/htsrobots.h
@@ -56,10 +56,12 @@ int checkrobots(robots_wizard * robots, const char *adr, const char *fil);
void checkrobots_free(robots_wizard * robots);
int checkrobots_set(robots_wizard * robots, const char *adr, const char *data);
/* Parse robots.txt `body` for `adr`, storing the HTTrack group's rules; `info`
- gets a disallow summary, `keep_root_disallow` FALSE drops "Disallow: /". */
+ gets a disallow summary, `keep_root_disallow` FALSE drops "Disallow: /", and
+ `sitemaps` (optional) collects the Sitemap: URLs, one per line. */
void robots_parse(robots_wizard *robots, const char *adr, const char *body,
size_t bodysize, char *info, size_t infosize,
- hts_boolean keep_root_disallow);
+ hts_boolean keep_root_disallow, char *sitemaps,
+ size_t sitemapsize);
#endif
#endif
diff --git a/src/htsselftest.c b/src/htsselftest.c
index 4686d39b..81c27a2e 100644
--- a/src/htsselftest.c
+++ b/src/htsselftest.c
@@ -59,6 +59,7 @@ Please visit our Website: http://www.httrack.com
#include "htssniff.h"
#include "htscodec.h"
#include "htsproxy.h"
+#include "htssitemap.h"
#include "htswarc.h"
#include "htschanges.h"
#include "htssinglefile.h"
@@ -1762,6 +1763,22 @@ static int st_copyopt(httrackp *opt, int argc, char **argv) {
if (to->single_file_max_size != 4096)
err = 1;
+ /* sitemap pair: the flag latches on, the URL takes the String deep copy */
+ from->sitemap = HTS_TRUE;
+ StringCopy(from->sitemap_url, "http://h.test/sitemap.xml");
+ to->sitemap = HTS_FALSE;
+ StringCopy(to->sitemap_url, "");
+ copy_htsopt(from, to);
+ if (!to->sitemap ||
+ strcmp(StringBuff(to->sitemap_url), "http://h.test/sitemap.xml") != 0)
+ err = 1;
+ from->sitemap = HTS_FALSE;
+ StringCopy(from->sitemap_url, "");
+ copy_htsopt(from, to);
+ if (!to->sitemap ||
+ strcmp(StringBuff(to->sitemap_url), "http://h.test/sitemap.xml") != 0)
+ err = 1;
+
/* #185 pause pair: copied when enabled (max>0), the 0 sentinel skips */
from->pause_min_ms = 5000;
from->pause_max_ms = 10000;
@@ -3686,7 +3703,7 @@ static int rb_decide(robots_wizard *r, const char *txt, const char *path) {
char host[64];
snprintf(host, sizeof(host), "h%d.example", n++);
- robots_parse(r, host, txt, strlen(txt), NULL, 0, HTS_TRUE);
+ robots_parse(r, host, txt, strlen(txt), NULL, 0, HTS_TRUE, NULL, 0);
return checkrobots(r, host, path);
}
@@ -3763,6 +3780,267 @@ static int st_robots(httrackp *opt, int argc, char **argv) {
return 0;
}
+/* Collect the URLs a sitemap scan hands out. */
+typedef struct sm_collect {
+ int n;
+ char url[8][HTS_URLMAXSIZE];
+} sm_collect;
+
+static hts_boolean sm_take(void *arg, const char *url) {
+ sm_collect *const c = (sm_collect *) arg;
+
+ if (c->n < (int) (sizeof(c->url) / sizeof(c->url[0])))
+ strcpybuff(c->url[c->n], url);
+ c->n++;
+ return HTS_TRUE;
+}
+
+/* Scan `doc` off a heap buffer with no NUL terminator, so a read past the
+ declared size is an ASan error rather than a silent pass. */
+static int sm_scan(const char *doc, int maxurls, hts_boolean *is_index,
+ sm_collect *out) {
+ const size_t len = strlen(doc);
+ char *raw = malloct(len);
+ int n;
+
+ memset(out, 0, sizeof(*out));
+ assertf(raw != NULL);
+ memcpy(raw, doc, len);
+ n = hts_sitemap_scan(raw, len, maxurls, is_index, sm_take, out);
+ freet(raw);
+ return n;
+}
+
+static int st_sitemap(httrackp *opt, int argc, char **argv) {
+ sm_collect c;
+ hts_boolean idx;
+ (void) opt;
+ (void) argc;
+ (void) argv;
+
+ /* A urlset yields its URLs, in order, unescaped. */
+ assertf(sm_scan(""
+ "http://h.test/a.html"
+ " https://h.test/b?x=1&y=2\n "
+ "",
+ 100, &idx, &c) == 2);
+ assertf(!idx);
+ assertf(strcmp(c.url[0], "http://h.test/a.html") == 0);
+ assertf(strcmp(c.url[1], "https://h.test/b?x=1&y=2") == 0);
+
+ /* A sitemapindex is flagged: its URLs are child sitemaps, not pages. */
+ assertf(sm_scan("http://h.test/s2.xml.gz"
+ "",
+ 100, &idx, &c) == 1);
+ assertf(idx);
+
+ /* Root element decides even when the other name appears later as text. */
+ assertf(sm_scan("http://h.test/a"
+ "",
+ 100, &idx, &c) == 1);
+ assertf(!idx);
+
+ /* Numeric character references, decimal and hex, decode to ASCII. */
+ assertf(sm_scan("http://h.test/a?b=c",
+ 100, &idx, &c) == 1);
+ assertf(strcmp(c.url[0], "http://h.test/a?b=c") == 0);
+
+ /* A reference decoding to a control byte is dropped: the shared decoder
+ writes the real character and the URL check refuses it. A reference the
+ decoder cannot represent () stays verbatim, like an unknown entity. */
+ assertf(sm_scan("http://h.test/a
b", 100,
+ &idx, &c) == 0);
+ assertf(sm_scan("http://h.test/a b", 100, &idx,
+ &c) == 0);
+ assertf(sm_scan("http://h.test/ab", 100, &idx,
+ &c) == 1);
+ assertf(strcmp(c.url[0], "http://h.test/ab") == 0);
+
+ /* A comment naming the other root element must not flip the verdict. */
+ assertf(sm_scan(""
+ "http://h.test/p",
+ 100, &idx, &c) == 1);
+ assertf(!idx);
+ assertf(sm_scan(""
+ "http://h.test/s",
+ 100, &idx, &c) == 1);
+ assertf(idx);
+
+ /* is not . */
+ assertf(sm_scan("http://h.test/a", 100,
+ &idx, &c) == 0);
+
+ /* Rejected: relative, non-http scheme, embedded space, empty. */
+ assertf(sm_scan("/a.htmlftp://h.test/a"
+ "javascript:alert(1)"
+ "http://h.test/a b",
+ 100, &idx, &c) == 0);
+
+ /* The URL length bound: one under fits, exactly at it is dropped rather than
+ truncated into a different URL. */
+ {
+ char BIGSTK doc[HTS_URLMAXSIZE * 2];
+ char BIGSTK url[HTS_URLMAXSIZE + 1];
+ size_t i;
+
+ strcpybuff(url, "http://h.test/");
+ for (i = strlen(url); i < HTS_URLMAXSIZE - 1; i++)
+ url[i] = 'a';
+ url[i] = '\0';
+ snprintf(doc, sizeof(doc), "%s", url);
+ assertf(sm_scan(doc, 100, &idx, &c) == 1);
+
+ url[i] = 'a';
+ url[i + 1] = '\0';
+ snprintf(doc, sizeof(doc), "%s", url);
+ assertf(sm_scan(doc, 100, &idx, &c) == 0);
+ }
+
+ /* The URL cap stops the scan. */
+ assertf(sm_scan("http://h.test/1http://h.test/2"
+ "http://h.test/3",
+ 2, &idx, &c) == 2);
+
+ /* The per-document cap at the value the engine actually uses. */
+ {
+ const int many = HTS_SITEMAP_MAX_URLS_DOC + 10;
+ const size_t cap = (size_t) many * 40 + 32;
+ char *big = malloct(cap);
+ size_t off;
+ int i;
+
+ assertf(big != NULL);
+ off = (size_t) snprintf(big, cap, "");
+ assertf(off < cap);
+ for (i = 0; i < many; i++) {
+ const int len =
+ snprintf(big + off, cap - off, "http://h.test/%d", i);
+
+ assertf(len > 0 && (size_t) len < cap - off);
+ off += (size_t) len;
+ }
+ memset(&c, 0, sizeof(c));
+ assertf(hts_sitemap_scan(big, off, HTS_SITEMAP_MAX_URLS_DOC, &idx, sm_take,
+ &c) == HTS_SITEMAP_MAX_URLS_DOC);
+ /* The handler count, not just the return: a call site hardcoding a smaller
+ cap would still return its own argument. */
+ assertf(c.n == HTS_SITEMAP_MAX_URLS_DOC);
+ freet(big);
+ }
+
+#if HTS_USEZLIB
+ /* A highly compressible document decodes without running away: the ratio
+ budget cannot bind (deflate tops out near 1032:1), so this pins the
+ decompression path itself rather than the 64 MiB ceiling. */
+ {
+ const char *const one = "http://h.test/bomb";
+ const size_t reps = 40000;
+ size_t xlen = 8 + reps * strlen(one) + 10, i;
+ char *x = malloct(xlen + 1);
+ uLongf zlen;
+ char *z;
+ z_stream zs;
+
+ assertf(x != NULL);
+ {
+ size_t w = (size_t) snprintf(x, xlen, "");
+ int len;
+
+ assertf(w < xlen);
+ for (i = 0; i < reps; i++) {
+ len = snprintf(x + w, xlen - w, "%s", one);
+ assertf(len > 0 && (size_t) len < xlen - w);
+ w += (size_t) len;
+ }
+ len = snprintf(x + w, xlen - w, "");
+ assertf(len > 0 && (size_t) len < xlen - w);
+ w += (size_t) len;
+ xlen = w;
+ }
+ zlen = compressBound((uLong) xlen) + 32;
+ z = malloct((size_t) zlen);
+ assertf(z != NULL);
+ memset(&zs, 0, sizeof(zs));
+ assertf(deflateInit2(&zs, 9, Z_DEFLATED, 16 + MAX_WBITS, 8,
+ Z_DEFAULT_STRATEGY) == Z_OK);
+ zs.next_in = (const Bytef *) x;
+ zs.avail_in = (uInt) xlen;
+ zs.next_out = (Bytef *) z;
+ zs.avail_out = (uInt) zlen;
+ assertf(deflate(&zs, Z_FINISH) == Z_STREAM_END);
+ zlen = (uLongf) zs.total_out;
+ deflateEnd(&zs);
+ /* well over the 4096:1 budget's 1 MiB floor, and far under the 64 MiB cap
+ */
+ assertf(xlen > 1024 * 1024 && (size_t) zlen < xlen / 100);
+ memset(&c, 0, sizeof(c));
+ assertf(hts_sitemap_scan(z, (size_t) zlen, 10, &idx, sm_take, &c) == 10);
+ assertf(strcmp(c.url[0], "http://h.test/bomb") == 0);
+ freet(z);
+ freet(x);
+ }
+#endif
+
+ /* An unterminated at end of buffer must not read past it. */
+ assertf(sm_scan("http://h.test/a", 100, &idx, &c) == 0);
+ assertf(sm_scan("http://h.test/gz.html";
+ uLongf zlen = compressBound((uLong) strlen(xml)) + 32;
+ char *z = malloct((size_t) zlen);
+ z_stream zs;
+
+ assertf(z != NULL);
+ memset(&zs, 0, sizeof(zs));
+ assertf(deflateInit2(&zs, 9, Z_DEFLATED, 16 + MAX_WBITS, 8,
+ Z_DEFAULT_STRATEGY) == Z_OK);
+ zs.next_in = (const Bytef *) xml;
+ zs.avail_in = (uInt) strlen(xml);
+ zs.next_out = (Bytef *) z;
+ zs.avail_out = (uInt) zlen;
+ assertf(deflate(&zs, Z_FINISH) == Z_STREAM_END);
+ zlen = (uLongf) zs.total_out;
+ deflateEnd(&zs);
+
+ memset(&c, 0, sizeof(c));
+ assertf(hts_sitemap_scan(z, (size_t) zlen, 100, &idx, sm_take, &c) == 1);
+ assertf(strcmp(c.url[0], "http://h.test/gz.html") == 0);
+
+ /* Truncated gzip: refused, not scanned as plain text. */
+ memset(&c, 0, sizeof(c));
+ assertf(hts_sitemap_scan(z, 4, 100, &idx, sm_take, &c) == -1);
+ freet(z);
+ }
+#endif
+
+ /* robots.txt: only Sitemap: records, comments stripped, case-insensitive,
+ and group-independent (no User-agent line needed). */
+ /* robots_parse collects Sitemap: whatever the user-agent group, strips the
+ comment and keeps the rules working alongside it. */
+ {
+ const char *const txt = "User-agent: *\nDisallow: /x\n"
+ "SITEMAP: http://h.test/s1.xml # first\n"
+ "Sitemapper: http://h.test/no.xml\n"
+ "Sitemap:\thttps://h.test/s2.xml\n";
+ char BIGSTK maps[1024];
+ robots_wizard rb;
+
+ memset(&rb, 0, sizeof(rb));
+ robots_parse(&rb, "h.test", txt, strlen(txt), NULL, 0, HTS_TRUE, maps,
+ sizeof(maps));
+ assertf(strcmp(maps, "http://h.test/s1.xml\nhttps://h.test/s2.xml\n") == 0);
+ assertf(checkrobots(&rb, "h.test", "/x") == -1);
+ checkrobots_free(&rb);
+ }
+
+ printf("sitemap self-test OK\n");
+ return 0;
+}
+
/* Connected stream pair over loopback; Windows has no socketpair(). */
static int st_socketpair(T_SOC sv[2]) {
struct sockaddr_in sa;
@@ -3860,21 +4138,6 @@ static int st_ftpuser(httrackp *opt, int argc, char **argv) {
return 0;
}
-/* Bounded substring search (records carry NUL bytes; strstr won't do). */
-static const char *warc_memstr(const char *hay, const char *needle,
- size_t haylen, size_t nlen) {
- if (nlen == 0 || haylen < nlen)
- return NULL;
- {
- size_t i;
- for (i = 0; i + nlen <= haylen; i++) {
- if (memcmp(hay + i, needle, nlen) == 0)
- return hay + i;
- }
- }
- return NULL;
-}
-
/* Slurp a whole file into a malloc'd buffer; sets *len. NULL on error. */
static unsigned char *warc_slurp(const char *path, size_t *len) {
FILE *f = FOPEN(path, "rb");
@@ -3952,6 +4215,13 @@ static unsigned char *warc_next_member(const unsigned char **in,
Content-Length == block length, the \r\n\r\n trailer intact, the response
body round-trips, and the hop-by-hop Transfer-Encoding is dropped (a real
Content-Encoding is kept verbatim; see warc-verbatim). */
+/* Argument order kept for the existing call sites; the search itself is the
+ shared hts_memstr. */
+static const char *warc_memstr(const char *hay, const char *needle,
+ size_t haylen, size_t nlen) {
+ return hts_memstr(hay, haylen, needle, nlen);
+}
+
static int st_warc(httrackp *opt, int argc, char **argv) {
char path[HTS_URLMAXSIZE];
warc_writer *w;
@@ -6075,6 +6345,8 @@ static const struct selftest_entry {
st_contentcodings},
{"robots", "", "robots.txt RFC 9309 Allow/Disallow precedence self-test",
st_robots},
+ {"sitemap", "",
+ "sitemap extraction, caps and robots.txt Sitemap:", st_sitemap},
{"ftp-line", "", "get_ftp_line bounds a hostile FTP reply line",
st_ftpline},
{"ftp-userpass", "", "ftp_split_userpass bounds URL userinfo", st_ftpuser},
diff --git a/src/htssitemap.c b/src/htssitemap.c
new file mode 100644
index 00000000..84bb7652
--- /dev/null
+++ b/src/htssitemap.c
@@ -0,0 +1,615 @@
+/* ------------------------------------------------------------ */
+/*
+HTTrack Website Copier, Offline Browser for Windows and Unix
+Copyright (C) 2026 Xavier Roche and other contributors
+
+SPDX-License-Identifier: GPL-3.0-or-later
+
+This program is free software: you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation, either version 3 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License
+along with this program. If not, see .
+
+Ethical use: we kindly ask that you NOT use this software to harvest email
+addresses or to collect any other private information about people. Doing so
+would dishonor our work and waste the many hours we have spent on it.
+
+Please visit our Website: http://www.httrack.com
+*/
+
+/* ------------------------------------------------------------ */
+/* File: sitemap ingestion (sitemaps.org 0.9) */
+/* Author: Xavier Roche */
+/* ------------------------------------------------------------ */
+
+#define HTS_INTERNAL_BYTECODE
+
+#include "htscore.h"
+#include "htssitemap.h"
+
+#include "htsbase.h"
+#include "htscodec.h"
+#include "htsencoding.h"
+#include "htsfilters.h"
+#include "htshash.h"
+#include "htsmodules.h"
+#include "htslib.h"
+#include "htsrobots.h"
+#include "htssafe.h"
+#include "htstools.h"
+
+#include
+#include
+
+/* One queued sitemap document awaiting ingestion. */
+typedef struct sitemap_doc {
+ char adr[HTS_URLMAXSIZE];
+ char fil[HTS_URLMAXSIZE];
+ int level;
+ hts_sitemap_source src;
+ hts_boolean done;
+ struct sitemap_doc *next;
+} sitemap_doc;
+
+struct hts_sitemap_state {
+ sitemap_doc *docs;
+ int ndocs; /* documents queued, capped by HTS_SITEMAP_MAX_DOCS */
+ int nurls; /* URLs seeded, capped by HTS_SITEMAP_MAX_URLS_TOTAL */
+ hts_boolean probe_done; /* the robots.txt probe has been answered */
+ hts_boolean fallback_done; /* the /sitemap.xml fallback was already queued */
+ /* The crawl's own start URL. Seeded URLs are judged against it, so a site
+ cannot widen a subtree crawl by putting its sitemap at the root. */
+ char anchor_adr[HTS_URLMAXSIZE];
+ char anchor_fil[HTS_URLMAXSIZE];
+};
+typedef struct hts_sitemap_state hts_sitemap_state;
+
+/* --------------------------------------------------------------------- */
+/* Document parsing (no engine state: fuzzable and self-testable) */
+/* --------------------------------------------------------------------- */
+
+/* Accept only an absolute http(s) URL with no space or control byte. */
+static hts_boolean sitemap_url_ok(const char *url) {
+ const char *p;
+
+ if (!strfield(url, "http://") && !strfield(url, "https://"))
+ return HTS_FALSE;
+ for (p = url; *p != '\0'; p++) {
+ if ((unsigned char) *p <= ' ' || (unsigned char) *p == 0x7f)
+ return HTS_FALSE;
+ }
+ return HTS_TRUE;
+}
+
+/* Skip to the character after the next '>' at or after p, or NULL. */
+static const char *sitemap_tag_end(const char *p, const char *end) {
+ while (p < end && *p != '>')
+ p++;
+ return p < end ? p + 1 : NULL;
+}
+
+/* HTS_TRUE when the document's root element is `name`. Skips the XML
+ declaration, comments and processing instructions first, so a comment
+ mentioning the other root element cannot decide the document type. */
+static hts_boolean sitemap_root_is(const char *doc, size_t size,
+ const char *name) {
+ const size_t nlen = strlen(name);
+ size_t i = 0;
+
+ if (size >= 3 && memcmp(doc, "\xef\xbb\xbf", 3) == 0)
+ i = 3; /* UTF-8 BOM */
+ while (i < size) {
+ if (isspace((unsigned char) doc[i])) {
+ i++;
+ } else if (doc[i] != '<') {
+ return HTS_FALSE; /* character data before any element: not XML */
+ } else if (i + 4 <= size && memcmp(doc + i, "", 3);
+
+ if (e == NULL)
+ return HTS_FALSE;
+ i = (size_t) (e - doc) + 3;
+ } else if (i + 2 <= size && (doc[i + 1] == '?' || doc[i + 1] == '!')) {
+ while (i < size && doc[i] != '>')
+ i++;
+ i++;
+ } else {
+ size_t j = i + 1;
+
+ /* an optional namespace prefix: is the same element */
+ while (j < size && doc[j] != ':' && doc[j] != '>' &&
+ !isspace((unsigned char) doc[j]))
+ j++;
+ if (j >= size || doc[j] != ':')
+ j = i + 1;
+ else
+ j++;
+ return j + nlen <= size && memcmp(doc + j, name, nlen) == 0 &&
+ (j + nlen == size ||
+ isspace((unsigned char) doc[j + nlen]) ||
+ doc[j + nlen] == '>' || doc[j + nlen] == '/')
+ ? HTS_TRUE
+ : HTS_FALSE;
+ }
+ }
+ return HTS_FALSE;
+}
+
+/* Decompress a gzip-framed body into a fresh buffer. The 64 MiB cap is what
+ binds in practice; deflate tops out near 1032:1, so the tree's codec budget
+ only matters as the shared policy for a coding that could go further. */
+static char *sitemap_gunzip(const char *body, size_t size, size_t *outsize) {
+ const LLint budget = hts_codec_maxout((LLint) size);
+ size_t cap = budget < (LLint) HTS_SITEMAP_MAX_BYTES
+ ? (size_t) budget
+ : (size_t) HTS_SITEMAP_MAX_BYTES;
+ char *out;
+ size_t n;
+
+ if (cap == 0)
+ return NULL;
+ out = malloct(cap + 1);
+ if (out == NULL)
+ return NULL;
+ n = hts_codec_head(HTS_CODEC_DEFLATE, body, size, out, cap);
+ if (n == 0) {
+ freet(out);
+ return NULL;
+ }
+ out[n] = '\0';
+ *outsize = n;
+ return out;
+}
+
+int hts_sitemap_scan(const char *body, size_t size, int maxurls,
+ hts_boolean *is_index, hts_sitemap_handler handler,
+ void *arg) {
+ char *unpacked = NULL;
+ const char *doc;
+ const char *end;
+ const char *p;
+ int n = 0;
+
+ if (is_index != NULL)
+ *is_index = HTS_FALSE;
+ if (body == NULL || size < 2 || handler == NULL)
+ return 0;
+
+ /* Content-Encoding gzip is undone upstream; only the container is left. */
+ if ((unsigned char) body[0] == 0x1f && (unsigned char) body[1] == 0x8b) {
+ unpacked = sitemap_gunzip(body, size, &size);
+ if (unpacked == NULL)
+ return -1;
+ doc = unpacked;
+ } else {
+ if (size > (size_t) HTS_SITEMAP_MAX_BYTES)
+ size = (size_t) HTS_SITEMAP_MAX_BYTES;
+ doc = body;
+ }
+ end = doc + size;
+
+ /* Set before the first callback: the handler reads the verdict. */
+ if (is_index != NULL)
+ *is_index = sitemap_root_is(doc, size, "sitemapindex");
+
+ for (p = doc; n < maxurls;) {
+ const char *loc = hts_memstr(p, (size_t) (end - p), "" or "", never "" */
+ if (loc + 4 >= end || (loc[4] != '>' && !isspace((unsigned char) loc[4]))) {
+ p = loc + 4;
+ continue;
+ }
+ val = sitemap_tag_end(loc + 4, end);
+ if (val == NULL)
+ break;
+ for (stop = val; stop < end && *stop != '<'; stop++)
+ ;
+ /* No closing tag: truncated document, so the value may be a partial URL. */
+ if (stop == end)
+ break;
+ p = stop;
+ while (val < stop && isspace((unsigned char) *val))
+ val++;
+ while (stop > val && isspace((unsigned char) *(stop - 1)))
+ stop--;
+ len = (size_t) (stop - val);
+ /* Overflow-safe: the untrusted length alone against the room left. */
+ if (len == 0 || len >= sizeof(url))
+ continue;
+ memcpy(url, val, len);
+ url[len] = '\0';
+ /* hts_unescapeEntities decodes in place and tolerates src == dest; a
+ reference to a control byte survives as one and sitemap_url_ok drops it.
+ */
+ if (hts_unescapeEntities(url, url, sizeof(url)) != 0 ||
+ !sitemap_url_ok(url))
+ continue;
+ n++;
+ if (!handler(arg, url))
+ break;
+ }
+
+ if (unpacked != NULL)
+ freet(unpacked);
+ return n;
+}
+
+/* --------------------------------------------------------------------- */
+/* Engine glue */
+/* --------------------------------------------------------------------- */
+
+static hts_sitemap_state *sitemap_get_state(httrackp *opt) {
+ if (opt->sitemap_state == NULL)
+ opt->sitemap_state = calloct(1, sizeof(hts_sitemap_state));
+ return (hts_sitemap_state *) opt->sitemap_state;
+}
+
+static sitemap_doc *sitemap_find(httrackp *opt, const char *adr,
+ const char *fil) {
+ hts_sitemap_state *const st = (hts_sitemap_state *) opt->sitemap_state;
+ sitemap_doc *d;
+
+ if (st == NULL)
+ return NULL;
+ for (d = st->docs; d != NULL; d = d->next) {
+ if (strfield2(d->adr, adr) && strcmp(d->fil, fil) == 0)
+ return d;
+ }
+ return NULL;
+}
+
+/* Who asked for this document decides how far it is gated. The wizard proper
+ is not usable here: it wants a referring link, and its up/down travel rules
+ would judge a child sitemap against the parent sitemap's directory. */
+static hts_boolean sitemap_fetch_allowed(httrackp *opt, const char *adr,
+ const char *fil,
+ hts_sitemap_source src) {
+ /* adr and fil are each capped just under HTS_URLMAXSIZE, and lfull prefixes
+ a scheme and a slash on top of both: 2 * HTS_URLMAXSIZE does not fit. */
+ char BIGSTK l[HTS_URLMAXSIZE * 2 + 16], lfull[HTS_URLMAXSIZE * 2 + 16];
+ int jokdepth = 0, jok;
+
+ hts_boolean refused;
+
+ /* The user naming a sitemap is the same intent as naming a start URL, which
+ the wizard admits unconditionally. */
+ if (src == HTS_SITEMAP_SRC_USER)
+ return HTS_TRUE;
+ strcpybuff(l, jump_identification_const(adr));
+ if (*fil != '/')
+ strcatbuff(l, "/");
+ strcatbuff(l, fil);
+ strcpybuff(lfull, link_has_authority(adr) ? "" : "http://");
+ strcatbuff(lfull, adr);
+ if (*fil != '/')
+ strcatbuff(lfull, "/");
+ strcatbuff(lfull, fil);
+ jok = fa_strjoker_dual(0, *opt->filters.filters, *opt->filters.filptr, lfull,
+ l, NULL, NULL, &jokdepth);
+ refused = (jok == -1) ? HTS_TRUE : HTS_FALSE;
+ if (refused) {
+ hts_log_print(opt, LOG_NOTICE, "Sitemap: filter rule #%d refuses %s%s",
+ jokdepth + 1, adr, fil);
+ return HTS_FALSE;
+ }
+ /* A Sitemap: line, or a sitemapindex entry, is the site inviting the fetch;
+ a Disallow elsewhere in the same file does not retract it. The well-known
+ location is only ever a guess, so there a Disallow wins. */
+ if (src == HTS_SITEMAP_SRC_GUESSED &&
+ hts_robots_forbids(opt, adr, fil, (jok != 0) ? HTS_TRUE : HTS_FALSE,
+ refused)) {
+ hts_log_print(opt, LOG_NOTICE, "Sitemap: robots.txt forbids %s%s", adr,
+ fil);
+ return HTS_FALSE;
+ }
+ return HTS_TRUE;
+}
+
+/* Record the link with save="" so the body stays in memory: a sitemap is
+ ingested, never mirrored. */
+static hts_boolean sitemap_queue_(httrackp *opt, const char *adr,
+ const char *fil, int level,
+ hts_sitemap_source src, hts_boolean link_it) {
+ hts_sitemap_state *const st = sitemap_get_state(opt);
+ sitemap_doc *d;
+
+ if (st == NULL)
+ return HTS_FALSE;
+ if (st->ndocs >= HTS_SITEMAP_MAX_DOCS || level > HTS_SITEMAP_MAX_LEVEL) {
+ hts_log_print(opt, LOG_WARNING, "Sitemap: cap reached, skipping %s%s", adr,
+ fil);
+ return HTS_FALSE;
+ }
+ if (strlen(adr) >= sizeof(d->adr) || strlen(fil) >= sizeof(d->fil))
+ return HTS_FALSE;
+ if (sitemap_find(opt, adr, fil) != NULL)
+ return HTS_FALSE;
+ if (!sitemap_fetch_allowed(opt, adr, fil, src))
+ return HTS_FALSE;
+ d = calloct(1, sizeof(sitemap_doc));
+ if (d == NULL)
+ return HTS_FALSE;
+ strcpybuff(d->adr, adr);
+ strcpybuff(d->fil, fil);
+ d->level = level;
+ d->src = src;
+ d->next = st->docs;
+ st->docs = d;
+ st->ndocs++;
+
+ if (!link_it)
+ return HTS_TRUE;
+ if (!hts_record_link(opt, adr, fil, "", "", "", NULL))
+ return HTS_FALSE;
+ heap_top()->testmode = 0;
+ heap_top()->link_import = 0;
+ heap_top()->depth = opt->depth + 1;
+ heap_top()->pass2 = 0;
+ heap_top()->retry = opt->retry;
+ heap_top()->premier = heap_top_index();
+ heap_top()->precedent = heap_top_index();
+ hts_log_print(opt, LOG_INFO, "Sitemap: queued %s%s", adr, fil);
+ return HTS_TRUE;
+}
+
+static hts_boolean sitemap_queue(httrackp *opt, const char *adr,
+ const char *fil, int level,
+ hts_sitemap_source src) {
+ return sitemap_queue_(opt, adr, fil, level, src, HTS_TRUE);
+}
+
+void hts_sitemap_redirect(httrackp *opt, const char *adr, const char *fil,
+ const char *newadr, const char *newfil) {
+ sitemap_doc *const d = sitemap_find(opt, adr, fil);
+
+ if (d == NULL || d->done)
+ return;
+ d->done = HTS_TRUE; /* the body lives at the target now */
+ /* The engine already queued the target link, so only the marking moves. */
+ (void) sitemap_queue_(opt, newadr, newfil, d->level, d->src, HTS_FALSE);
+ hts_log_print(opt, LOG_NOTICE, "Sitemap: %s%s redirects to %s%s", adr, fil,
+ newadr, newfil);
+}
+
+void hts_sitemap_seed(httrackp *opt, const char *starturl) {
+ char BIGSTK url[HTS_URLMAXSIZE * 2];
+ lien_adrfil af;
+
+ if (StringNotEmpty(opt->sitemap_url)) {
+ if (strlen(StringBuff(opt->sitemap_url)) >= sizeof(url)) {
+ hts_log_print(opt, LOG_ERROR, "Sitemap URL too long");
+ } else {
+ strcpybuff(url, StringBuff(opt->sitemap_url));
+ if (strstr(url, ":/") == NULL)
+ hts_log_print(opt, LOG_ERROR, "Sitemap URL must be absolute: %s", url);
+ else if (ident_url_absolute(url, &af) >= 0)
+ (void) sitemap_queue(opt, af.adr, af.fil, 0, HTS_SITEMAP_SRC_USER);
+ }
+ }
+ if (starturl == NULL || starturl[0] == '\0' ||
+ strlen(starturl) >= sizeof(url))
+ return;
+ strcpybuff(url, starturl);
+ if (ident_url_absolute(url, &af) < 0)
+ return;
+ {
+ hts_sitemap_state *const st = sitemap_get_state(opt);
+
+ if (st != NULL && strlen(af.adr) < sizeof(st->anchor_adr) &&
+ strlen(af.fil) < sizeof(st->anchor_fil)) {
+ strcpybuff(st->anchor_adr, af.adr);
+ strcpybuff(st->anchor_fil, af.fil);
+ }
+ }
+ if (!opt->sitemap)
+ return;
+ /* Answered in hts_sitemap_robots, once the parsed rules are installed. */
+ if (hts_record_link(opt, af.adr, "/robots.txt", "", "", "", NULL)) {
+ heap_top()->testmode = 0;
+ heap_top()->link_import = 0;
+ heap_top()->depth = 0;
+ heap_top()->pass2 = 0;
+ heap_top()->retry = opt->retry;
+ heap_top()->premier = heap_top_index();
+ heap_top()->precedent = heap_top_index();
+ /* Claim the host so the parser does not queue robots.txt a second time. */
+ if (opt->robotsptr != NULL)
+ (void) checkrobots_set((robots_wizard *) opt->robotsptr, af.adr, "");
+ }
+}
+
+void hts_sitemap_robots(httrackp *opt, const char *adr, const char *sitemaps) {
+ hts_sitemap_state *const st = (hts_sitemap_state *) opt->sitemap_state;
+ int queued = 0;
+
+ if (st == NULL || !opt->sitemap || st->probe_done ||
+ !strfield2(st->anchor_adr, adr))
+ return;
+ st->probe_done = HTS_TRUE;
+ if (sitemaps != NULL) {
+ const char *p = sitemaps;
+
+ while (*p != '\0') {
+ const char *const eol = strchr(p, '\n');
+ const size_t len = eol != NULL ? (size_t) (eol - p) : strlen(p);
+ char BIGSTK line[HTS_URLMAXSIZE];
+ lien_adrfil af;
+
+ if (len > 0 && len < sizeof(line)) {
+ memcpy(line, p, len);
+ line[len] = '\0';
+ /* Same host: a Sitemap: line must not aim the fetcher elsewhere. */
+ if (sitemap_url_ok(line) && ident_url_absolute(line, &af) >= 0 &&
+ strfield2(af.adr, adr) &&
+ sitemap_queue(opt, af.adr, af.fil, 0, HTS_SITEMAP_SRC_DECLARED))
+ queued++;
+ }
+ if (eol == NULL)
+ break;
+ p = eol + 1;
+ }
+ }
+ if (queued == 0 && !st->fallback_done) {
+ st->fallback_done = HTS_TRUE;
+ if (sitemap_queue(opt, adr, "/sitemap.xml", 0, HTS_SITEMAP_SRC_GUESSED))
+ queued++;
+ }
+ hts_log_print(opt, LOG_NOTICE, "Sitemap: %d sitemap(s) queued for %s", queued,
+ adr);
+}
+
+hts_boolean hts_sitemap_pending(httrackp *opt, const char *adr,
+ const char *fil) {
+ const sitemap_doc *const d = sitemap_find(opt, adr, fil);
+
+ return d != NULL && !d->done ? HTS_TRUE : HTS_FALSE;
+}
+
+/* Handler context: seeding URLs from one document. */
+typedef struct sitemap_ingest_ctx {
+ httrackp *opt;
+ htsmoduleStruct *str;
+ const char *adr; /* host of the document being ingested */
+ int level;
+ hts_boolean is_index;
+ int accepted; /* URLs seeded or documents queued, not merely parsed */
+} sitemap_ingest_ctx;
+
+/* A of a : hand it to the wizard as a top-level seed.
+ The wizard is pointed at the crawl's own start URL, not at the sitemap: the
+ site picks where its sitemap lives, so anchoring travel there would let a
+ root sitemap widen a subtree crawl to the whole host. The URL then becomes
+ its own anchor, exactly as a command-line seed does. */
+static hts_boolean sitemap_seed_url(void *arg, const char *url) {
+ sitemap_ingest_ctx *const c = (sitemap_ingest_ctx *) arg;
+ httrackp *const opt = c->opt;
+ hts_sitemap_state *const st = sitemap_get_state(opt);
+ char BIGSTK buff[HTS_URLMAXSIZE];
+ int before;
+
+ if (st == NULL || st->nurls >= HTS_SITEMAP_MAX_URLS_TOTAL) {
+ hts_log_print(opt, LOG_WARNING,
+ "Sitemap: URL cap reached, ignoring the rest");
+ return HTS_FALSE;
+ }
+ /* strcpybuff aborts rather than truncating: never feed it unchecked input. */
+ if (strlen(url) >= sizeof(buff))
+ return HTS_TRUE;
+ st->nurls++;
+ strcpybuff(buff, url);
+ before = opt->lien_tot;
+ if (htsAddLink(c->str, buff))
+ c->accepted++;
+ if (opt->lien_tot > before)
+ heap_top()->premier = heap_top_index(); /* a seed anchors on itself */
+ return HTS_TRUE;
+}
+
+/* A of a : cross-host children are dropped, so a hostile
+ sitemap cannot aim the fetcher elsewhere. */
+static hts_boolean sitemap_seed_child(void *arg, const char *url) {
+ sitemap_ingest_ctx *const c = (sitemap_ingest_ctx *) arg;
+ char BIGSTK buff[HTS_URLMAXSIZE];
+ lien_adrfil af;
+
+ if (strlen(url) >= sizeof(buff))
+ return HTS_TRUE;
+ strcpybuff(buff, url);
+ if (ident_url_absolute(buff, &af) < 0)
+ return HTS_TRUE;
+ if (!strfield2(af.adr, c->adr)) {
+ hts_log_print(c->opt, LOG_WARNING,
+ "Sitemap: ignoring off-host child sitemap %s%s", af.adr,
+ af.fil);
+ return HTS_TRUE;
+ }
+ if (sitemap_queue(c->opt, af.adr, af.fil, c->level + 1,
+ HTS_SITEMAP_SRC_DECLARED))
+ c->accepted++;
+ return HTS_TRUE;
+}
+
+/* The scan classifies the document before the first callback. */
+static hts_boolean sitemap_seed_any(void *arg, const char *url) {
+ sitemap_ingest_ctx *const c = (sitemap_ingest_ctx *) arg;
+
+ return c->is_index ? sitemap_seed_child(arg, url)
+ : sitemap_seed_url(arg, url);
+}
+
+void hts_sitemap_ingest(httrackp *opt, htsmoduleStruct *str, const char *adr,
+ const char *fil, const char *body, size_t size) {
+ sitemap_doc *const d = sitemap_find(opt, adr, fil);
+ hts_sitemap_state *const st = (hts_sitemap_state *) opt->sitemap_state;
+ sitemap_ingest_ctx ctx;
+ int n, anchor, saved_depth;
+
+ if (d == NULL || d->done)
+ return;
+ d->done = HTS_TRUE;
+ /* str->ptr_ is a scratch int owned by the caller, so nothing else moves. */
+ anchor = *str->ptr_;
+ if (st != NULL && st->anchor_adr[0] != '\0' && opt->hash != NULL) {
+ const int i = hash_read((const hash_struct *) opt->hash, st->anchor_adr,
+ st->anchor_fil, 1);
+
+ if (i >= 0)
+ anchor = i;
+ }
+ *str->ptr_ = anchor;
+ /* Borrow the anchor's position but keep a seed's full depth budget. */
+ saved_depth = heap(anchor)->depth;
+ heap(anchor)->depth = opt->depth + 1;
+ ctx.opt = opt;
+ ctx.str = str;
+ ctx.adr = adr;
+ ctx.level = d->level;
+ ctx.is_index = HTS_FALSE;
+ ctx.accepted = 0;
+
+ n = hts_sitemap_scan(body, size, HTS_SITEMAP_MAX_URLS_DOC, &ctx.is_index,
+ sitemap_seed_any, &ctx);
+ heap(anchor)->depth = saved_depth;
+ if (n < 0) {
+ hts_log_print(opt, LOG_ERROR, "Sitemap: could not decompress %s%s", adr,
+ fil);
+ return;
+ }
+ if (ctx.is_index)
+ hts_log_print(opt, LOG_NOTICE,
+ "Sitemap: %d of %d child sitemap(s) listed by %s%s",
+ ctx.accepted, n, adr, fil);
+ else
+ hts_log_print(opt, LOG_NOTICE, "Sitemap: %d of %d URL(s) added from %s%s",
+ ctx.accepted, n, adr, fil);
+}
+
+void hts_sitemap_free(httrackp *opt) {
+ hts_sitemap_state *const st = (hts_sitemap_state *) opt->sitemap_state;
+
+ if (st == NULL)
+ return;
+ while (st->docs != NULL) {
+ sitemap_doc *const next = st->docs->next;
+
+ freet(st->docs);
+ st->docs = next;
+ }
+ freet(opt->sitemap_state);
+ opt->sitemap_state = NULL;
+}
diff --git a/src/htssitemap.h b/src/htssitemap.h
new file mode 100644
index 00000000..433a40f5
--- /dev/null
+++ b/src/htssitemap.h
@@ -0,0 +1,108 @@
+/* ------------------------------------------------------------ */
+/*
+HTTrack Website Copier, Offline Browser for Windows and Unix
+Copyright (C) 2026 Xavier Roche and other contributors
+
+SPDX-License-Identifier: GPL-3.0-or-later
+
+This program is free software: you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation, either version 3 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License
+along with this program. If not, see .
+
+Ethical use: we kindly ask that you NOT use this software to harvest email
+addresses or to collect any other private information about people. Doing so
+would dishonor our work and waste the many hours we have spent on it.
+
+Please visit our Website: http://www.httrack.com
+*/
+
+/* ------------------------------------------------------------ */
+/* HTTrack sitemap ingestion (sitemaps.org 0.9). Internal, not installed.
+ Reads / documents, plain or gzip-framed, and feeds
+ their URLs to the crawl as top-level seeds. The whole input is
+ attacker-controlled, so every entry point below is capped. */
+/* ------------------------------------------------------------ */
+
+#ifndef HTS_SITEMAP_DEFH
+#define HTS_SITEMAP_DEFH
+
+#include "htsdefines.h"
+#include "htsopt.h"
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+/* Caps. sitemaps.org allows 50000 URLs and 50 MB uncompressed per document;
+ the byte cap sits above that so a conformant sitemap always fits. */
+#define HTS_SITEMAP_MAX_URLS_DOC 50000 /* per document */
+#define HTS_SITEMAP_MAX_URLS_TOTAL 200000 /* per mirror */
+#define HTS_SITEMAP_MAX_DOCS 256 /* documents per mirror */
+#define HTS_SITEMAP_MAX_LEVEL 4 /* sitemapindex nesting */
+#define HTS_SITEMAP_MAX_BYTES (64 * 1024 * 1024) /* decompressed document */
+
+/* Who asked for a sitemap document, which decides how far its fetch is gated.
+ The user naming one is the same intent as a start URL; a site declaring one
+ invites the fetch; the well-known location is only ever our guess. */
+typedef enum {
+ HTS_SITEMAP_SRC_USER, /**< --sitemap-url */
+ HTS_SITEMAP_SRC_DECLARED, /**< a Sitemap: line or a sitemapindex entry */
+ HTS_SITEMAP_SRC_GUESSED /**< the /sitemap.xml fallback */
+} hts_sitemap_source;
+
+/* Per-URL handler; returning HTS_FALSE stops the scan. */
+typedef hts_boolean (*hts_sitemap_handler)(void *arg, const char *url);
+
+/* Scan one sitemap document, plain or gzip-framed, handing every acceptable
+ absolute http(s) URL to `handler`. Stops after `maxurls` URLs, or when
+ the handler refuses. `is_index` (optional) reports a , whose
+ URLs are child sitemaps rather than pages. Returns the number of URLs handed
+ out, or -1 when the document could not be decompressed within the caps. */
+int hts_sitemap_scan(const char *body, size_t size, int maxurls,
+ hts_boolean *is_index, hts_sitemap_handler handler,
+ void *arg);
+
+/* --- Engine glue (needs a live httrackp). --- */
+
+/* Queue the first sitemap document of the mirror: the explicit --sitemap-url,
+ or the start host's /robots.txt probe for --sitemap. `starturl` is the first
+ command-line seed. No-op when neither option is set. */
+void hts_sitemap_seed(httrackp *opt, const char *starturl);
+
+/* Act on the start host's robots.txt once its rules are installed: queue the
+ Sitemap: URLs it names (newline-separated, from robots_parse), or the
+ well-known /sitemap.xml when it names none. No-op unless --sitemap. */
+void hts_sitemap_robots(httrackp *opt, const char *adr, const char *sitemaps);
+
+/* Carry the sitemap marking of (adr,fil) over to the target of a redirect the
+ engine has already queued, so a moved sitemap is still ingested. */
+void hts_sitemap_redirect(httrackp *opt, const char *adr, const char *fil,
+ const char *newadr, const char *newfil);
+
+/* HTS_TRUE when (adr,fil) is a queued sitemap document awaiting ingestion. */
+hts_boolean hts_sitemap_pending(httrackp *opt, const char *adr,
+ const char *fil);
+
+/* Ingest a fetched sitemap document (or the robots.txt probe): seed its URLs
+ through the wizard via htsAddLink, and queue nested sitemaps. `str` supplies
+ the parser context of the document being processed. */
+void hts_sitemap_ingest(httrackp *opt, htsmoduleStruct *str, const char *adr,
+ const char *fil, const char *body, size_t size);
+
+/* Release the ingestion state held in opt (NULL-safe, idempotent). */
+void hts_sitemap_free(httrackp *opt);
+
+#ifdef __cplusplus
+}
+#endif
+
+#endif
diff --git a/src/htstools.c b/src/htstools.c
index 42b2b94b..b4ea3859 100644
--- a/src/htstools.c
+++ b/src/htstools.c
@@ -276,6 +276,21 @@ int ident_url_relatif(const char *lien, const char *origin_adr,
return ok;
}
+/* Bounded substring search: bodies and archive records carry NUL bytes, so
+ strstr() would stop at the first one. */
+const char *hts_memstr(const char *hay, size_t haylen, const char *needle,
+ size_t nlen) {
+ size_t i;
+
+ if (nlen == 0 || haylen < nlen)
+ return NULL;
+ for (i = 0; i + nlen <= haylen; i++) {
+ if (hay[i] == *needle && memcmp(hay + i, needle, nlen) == 0)
+ return hay + i;
+ }
+ return NULL;
+}
+
// créer dans s, à partir du chemin courant curr_fil, le lien vers link (absolu)
// un ident_url_relatif a déja été fait avant, pour que link ne soit pas un chemin relatif
int lienrelatif(char *s, size_t ssize, const char *link, const char *curr_fil) {
diff --git a/src/htstools.h b/src/htstools.h
index c0a57d2a..7a6e1849 100644
--- a/src/htstools.h
+++ b/src/htstools.h
@@ -61,6 +61,11 @@ typedef struct lien_adrfilsave lien_adrfilsave;
int ident_url_relatif(const char *lien, const char *origin_adr,
const char *origin_fil,
lien_adrfil* const adrfil);
+/* Bounded substring search over data that may hold NUL bytes; NULL if absent.
+ */
+const char *hts_memstr(const char *hay, size_t haylen, const char *needle,
+ size_t nlen);
+
int lienrelatif(char *s, size_t ssize, const char *link, const char *curr);
int link_has_authority(const char *lien);
int link_has_authorization(const char *lien);
diff --git a/src/htswizard.c b/src/htswizard.c
index 30bb3769..2925acac 100644
--- a/src/htswizard.c
+++ b/src/htswizard.c
@@ -151,6 +151,23 @@ static hts_boolean is_embed_pair(const htspair_t *table, const char *tag,
return HTS_FALSE;
}
+/* The engine's robots.txt verdict for (adr,fil). Under HTS_ROBOTS_SOMETIMES an
+ explicit filter acceptance overrides the ban, which is why the filter outcome
+ is an input; the sitemap fetcher asks the same question outside the wizard.
+ */
+hts_boolean hts_robots_forbids(httrackp *opt, const char *adr, const char *fil,
+ hts_boolean filters_decided,
+ hts_boolean filters_refused) {
+ if (!opt->robots || opt->robotsptr == NULL)
+ return HTS_FALSE;
+ if (checkrobots((robots_wizard *) opt->robotsptr, adr, fil) != -1)
+ return HTS_FALSE;
+ if (filters_decided && !filters_refused &&
+ opt->robots == HTS_ROBOTS_SOMETIMES)
+ return HTS_FALSE;
+ return HTS_TRUE;
+}
+
static int hts_acceptlink_(httrackp * opt, int ptr,
const char *adr, const char *fil, const char *tag,
const char *attribute, int *set_prio_to,
@@ -576,30 +593,26 @@ static int hts_acceptlink_(httrackp * opt, int ptr,
}
}
// vérifier robots.txt
- if (opt->robots) {
- int r = checkrobots(_ROBOTS, adr, fil);
-
- if (r == -1) { // interdiction
+ if (opt->robots && checkrobots(_ROBOTS, adr, fil) == -1) {
#if DEBUG_ROBOTS
- printf("robots.txt forbidden: %s%s\n", adr, fil);
+ printf("robots.txt forbidden: %s%s\n", adr, fil);
#endif
- // question résolue, par les filtres, et mode robot non strict
- if ((!question) && (filters_answer) &&
- (opt->robots == HTS_ROBOTS_SOMETIMES) && (forbidden_url != 1)) {
- r = 0; // annuler interdiction des robots
- if (!forbidden_url) {
- hts_log_print(opt, LOG_DEBUG,
- "Warning link followed against robots.txt: link %s at %s%s",
- l, adr, fil);
- }
- }
- if (r == -1) { // interdire
- forbidden_url = 1;
- question = 0;
- hts_log_print(opt, LOG_DEBUG,
- "(robots.txt) forbidden link: link %s at %s%s", l, adr,
- fil);
+ if (!hts_robots_forbids(opt, adr, fil,
+ (!question && filters_answer) ? HTS_TRUE
+ : HTS_FALSE,
+ (forbidden_url == 1) ? HTS_TRUE : HTS_FALSE)) {
+ if (!forbidden_url) {
+ hts_log_print(
+ opt, LOG_DEBUG,
+ "Warning link followed against robots.txt: link %s at %s%s", l,
+ adr, fil);
}
+ } else {
+ forbidden_url = 1;
+ question = 0;
+ hts_log_print(opt, LOG_DEBUG,
+ "(robots.txt) forbidden link: link %s at %s%s", l, adr,
+ fil);
}
}
diff --git a/src/htswizard.h b/src/htswizard.h
index 5a65ef9e..0485c4c6 100644
--- a/src/htswizard.h
+++ b/src/htswizard.h
@@ -49,6 +49,13 @@ typedef struct httrackp httrackp;
typedef struct lien_url lien_url;
#endif
+/* The engine's robots.txt verdict for (adr,fil): HTS_TRUE when the fetch is
+ forbidden. `filters_decided`/`filters_refused` carry the filter outcome,
+ which overrides a ban under -s1 (HTS_ROBOTS_SOMETIMES). */
+hts_boolean hts_robots_forbids(httrackp *opt, const char *adr, const char *fil,
+ hts_boolean filters_decided,
+ hts_boolean filters_refused);
+
int hts_acceptlink(httrackp * opt, int ptr,
const char *adr, const char *fil,
const char *tag, const char *attribute,
diff --git a/src/libhttrack.vcxproj b/src/libhttrack.vcxproj
index 3f5afa7e..2e90d77e 100644
--- a/src/libhttrack.vcxproj
+++ b/src/libhttrack.vcxproj
@@ -140,6 +140,7 @@
+
diff --git a/tests/01_zlib-sitemap.test b/tests/01_zlib-sitemap.test
new file mode 100644
index 00000000..35322c05
--- /dev/null
+++ b/tests/01_zlib-sitemap.test
@@ -0,0 +1,10 @@
+#!/bin/bash
+#
+# Sitemap parser self-test: extraction, entity decoding, URL and length
+# rejections, the URL cap, gzip framing and robots.txt Sitemap: records.
+
+set -euo pipefail
+
+out=$(httrack -O /dev/null '-#test=sitemap')
+echo "$out"
+test "$out" = "sitemap self-test OK"
diff --git a/tests/90_webhttrack-checkbox-clear.test b/tests/90_webhttrack-checkbox-clear.test
index 1407a4aa..3982b429 100755
--- a/tests/90_webhttrack-checkbox-clear.test
+++ b/tests/90_webhttrack-checkbox-clear.test
@@ -149,6 +149,7 @@ BOXES = [
("keepqueryorder", "KeepQueryOrder", "--keep-query-order", None),
("toler", "TolerantRequests", "--tolerant", None),
("http10", "HTTP10", "--http-10", None),
+ ("sitemap", "Sitemap", "--sitemap", None),
("warc", "Warc", "--warc", None),
("changes", "Changes", "--changes", None),
("norecatch", "NoRecatch", "--do-not-recatch", None),
diff --git a/tests/95_local-sitemap.test b/tests/95_local-sitemap.test
new file mode 100644
index 00000000..c6461d80
--- /dev/null
+++ b/tests/95_local-sitemap.test
@@ -0,0 +1,134 @@
+#!/bin/bash
+#
+# --sitemap seeds the crawl from robots.txt -> sitemapindex -> gzipped urlset.
+# start.html links to nothing, so orphan*.html can only arrive through the
+# sitemap; deep1.html proves the seeds keep a full depth budget under -r2, and
+# the off-host page and child sitemap must both be refused.
+
+set -eu
+
+: "${top_srcdir:=..}"
+
+crawl() { bash "$top_srcdir/tests/local-crawl.sh" "$@"; }
+
+# robots.txt Sitemap: -> index -> .xml.gz, seeds behaving like -r2 seeds. The
+# log assertions pin which route was taken: the two documents are served from
+# both /sitemapdir/index.xml and the well-known /sitemap.xml, so "some sitemap
+# was read" would pass either way. --not-found pins ingestion-only: the sitemap
+# documents feed the crawl but never land in the mirror.
+# --rerun also walks the update path over a mirror holding sitemap seeds.
+crawl --errors 0 --rerun \
+ --found 'sitemapdir/orphan1.html' \
+ --found 'sitemapdir/orphan2.html' \
+ --found 'sitemapdir/deep1.html' \
+ --not-found 'sitemapdir/index.xml' \
+ --not-found 'sitemapdir/pages.xml.gz' \
+ --log-found '2 of 3 URL\(s\) added from 127\.0\.0\.1:[0-9]+/sitemapdir/pages\.xml\.gz' \
+ --log-found '1 of 2 child sitemap\(s\) listed by 127\.0\.0\.1:[0-9]+/sitemapdir/index\.xml' \
+ --log-found 'ignoring off-host child sitemap' \
+ --log-not-found '/sitemap\.xml' \
+ httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2
+
+# Negative control: without the option nothing but the start page is reached.
+crawl --errors 0 \
+ --found 'sitemapdir/start.html' \
+ --not-found 'sitemapdir/orphan1.html' \
+ --not-found 'sitemapdir/orphan2.html' \
+ httrack 'BASEURL/sitemapdir/start.html' -r2
+
+# Sitemap URLs are not a filter bypass: a -*orphan2* rule still rejects one.
+crawl --errors 0 \
+ --found 'sitemapdir/orphan1.html' \
+ --not-found 'sitemapdir/orphan2.html' \
+ httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2 '-*orphan2*'
+
+# A robots.txt naming no sitemap falls back to the well-known /sitemap.xml.
+# The test server drops its Sitemap: record for this User-Agent.
+crawl --errors 0 \
+ --found 'sitemapdir/orphan1.html' \
+ --log-found 'listed by 127\.0\.0\.1:[0-9]+/sitemap\.xml' \
+ --log-not-found 'sitemapdir/index\.xml' \
+ httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2 -F 'nositemap-agent'
+
+# --sitemap-url alone reads the named document and probes nothing.
+crawl --errors 0 \
+ --found 'sitemapdir/orphan1.html' \
+ --found 'sitemapdir/orphan2.html' \
+ --log-not-found 'sitemapdir/index\.xml' \
+ --log-not-found '/sitemap\.xml' \
+ httrack 'BASEURL/sitemapdir/start.html' -r2 \
+ --sitemap-url 'BASEURL/sitemapdir/pages.xml.gz'
+
+# The two options are additive: an unfetchable --sitemap-url is parsed as an
+# empty document and does not stop --sitemap from seeding the crawl.
+crawl --errors-content 1 \
+ --found 'sitemapdir/orphan1.html' \
+ --log-found 'Sitemap: 0 of 0 URL\(s\) added' \
+ httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2 \
+ --sitemap-url 'BASEURL/sitemapdir/missing.xml'
+
+# The sitemapindex nesting cap stops the chain before its deepest urlset. The
+# positive control is the page one level inside the cap: without it, a cap
+# mutated to 0 (nothing followed) would pass this just as well.
+crawl --errors 0 \
+ --found 'sitemapdir/cap3.html' \
+ --not-found 'sitemapdir/cap4.html' \
+ --log-found 'Sitemap: cap reached' \
+ httrack 'BASEURL/sitemapdir/start.html' -r2 \
+ --sitemap-url 'BASEURL/sitemapdir/chain0.xml'
+
+# A site cannot widen a subtree crawl by putting its sitemap at the root: the
+# page below the start directory is seeded, the one above it is refused.
+crawl --errors 0 \
+ --found 'deep/dir/below.html' \
+ --not-found 'elsewhere/updir.html' \
+ httrack 'BASEURL/deep/dir/start.html' --sitemap -r3 -F 'scopesitemap-agent'
+
+# How far a sitemap fetch is gated depends on who asked for it, so the three
+# cases below have to differ: one "robots is honoured" assertion would hide a
+# regression in either direction. The harness disables robots, hence -s2 here.
+
+# We guessed /sitemap.xml, so a Disallow on it wins.
+crawl --errors 0 \
+ --not-found 'sitemapdir/orphan1.html' \
+ --log-found 'Sitemap: robots.txt forbids' \
+ httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2 -s2 \
+ -F 'denysitemap-agent'
+
+# The site declared it through a Sitemap: line, which the same Disallow does
+# not retract.
+crawl --errors 0 \
+ --found 'sitemapdir/orphan1.html' \
+ --log-not-found 'Sitemap: robots.txt forbids' \
+ httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2 -s2 \
+ -F 'denydeclared-agent'
+
+# The user named it, which is the same intent as a start URL: not refused
+# either, even though robots.txt forbids that exact path.
+crawl --errors 0 \
+ --found 'sitemapdir/orphan1.html' \
+ --log-not-found 'Sitemap: robots.txt forbids' \
+ httrack 'BASEURL/sitemapdir/start.html' -r2 -s2 -F 'denysitemap-agent' \
+ --sitemap-url 'BASEURL/sitemap.xml'
+
+# A child sitemap is a fetch like any other: a filter rejecting it stops the
+# whole subtree, so the page only it lists is never reached.
+crawl --errors 0 \
+ --not-found 'sitemapdir/gated.html' \
+ --log-found 'Sitemap: filter rule #[0-9]+ refuses' \
+ httrack 'BASEURL/sitemapdir/start.html' -r2 '-*filtered.xml*' \
+ --sitemap-url 'BASEURL/sitemapdir/gatedindex.xml'
+
+# Control: without that filter the same chain does reach the page.
+crawl --errors 0 \
+ --found 'sitemapdir/gated.html' \
+ httrack 'BASEURL/sitemapdir/start.html' -r2 \
+ --sitemap-url 'BASEURL/sitemapdir/gatedindex.xml'
+
+# A redirected sitemap is still ingested: the marking follows the 301.
+crawl --errors 0 \
+ --found 'sitemapdir/orphan1.html' \
+ --log-found 'moved\.xml redirects to .*pages\.xml\.gz' \
+ --log-found 'URL\(s\) added from 127\.0\.0\.1:[0-9]+/sitemapdir/pages\.xml\.gz' \
+ httrack 'BASEURL/sitemapdir/start.html' -r2 \
+ --sitemap-url 'BASEURL/sitemapdir/moved.xml'
diff --git a/tests/Makefile.am b/tests/Makefile.am
index ba771ed1..0d9a1e84 100644
--- a/tests/Makefile.am
+++ b/tests/Makefile.am
@@ -91,6 +91,7 @@ TESTS = \
01_engine-xfread.test \
01_zlib-acceptencoding.test \
01_zlib-warc.test \
+ 01_zlib-sitemap.test \
01_zlib-warc-cdx.test \
01_zlib-warc-wacz.test \
01_zlib-contentcodings.test \
@@ -192,6 +193,7 @@ TESTS = \
91_webhttrack-directory.test \
92_local-proxytrack-ndx-fields.test \
93_local-changes.test \
- 94_local-single-file.test
+ 94_local-single-file.test \
+ 95_local-sitemap.test
CLEANFILES = check-network_sh.cache
diff --git a/tests/local-server.py b/tests/local-server.py
index c3de826d..72051663 100755
--- a/tests/local-server.py
+++ b/tests/local-server.py
@@ -18,6 +18,7 @@
import gzip
import hashlib
import os
+import re
import sys
import time
from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer
@@ -542,8 +543,36 @@ def route_gated_secret(self):
return self.fail_cookie(name)
self.send_html("\tThis is the secret.")
+ # A User-Agent carrying NO_SITEMAP_UA gets a robots.txt with no Sitemap:
+ # record, so a test can drive the /sitemap.xml fallback instead.
+ NO_SITEMAP_UA = "nositemap"
+ # ... and one that additionally Disallows the well-known location, so the
+ # fallback has to be refused by the rules this very body carries.
+ DENY_SITEMAP_UA = "denysitemap"
+ # ... the same Disallow, but with the sitemap also declared: the
+ # declaration is the site inviting the fetch and must win.
+ DENY_DECLARED_UA = "denydeclared"
+ # ... and one that points the sitemap at the site root, to check a subtree
+ # crawl is not widened by where the site chooses to put its sitemap.
+ SCOPE_SITEMAP_UA = "scopesitemap"
+
def route_robots(self):
- body = b"User-agent: *\nDisallow:\n"
+ # The Sitemap: record is group-independent; only --sitemap acts on it.
+ ua = self.headers.get("User-Agent") or ""
+ host = self.headers.get("Host")
+ body = "User-agent: *\nDisallow:\n"
+ if self.DENY_DECLARED_UA in ua:
+ body = (
+ "User-agent: *\nDisallow: /sitemap.xml\n"
+ f"Sitemap: http://{host}/sitemap.xml\n"
+ )
+ elif self.DENY_SITEMAP_UA in ua:
+ body = "User-agent: *\nDisallow: /sitemap.xml\n"
+ elif self.SCOPE_SITEMAP_UA in ua:
+ body += f"Sitemap: http://{host}/scopesitemap.xml\n"
+ elif self.NO_SITEMAP_UA not in ua:
+ body += f"Sitemap: http://{host}/sitemapdir/index.xml\n"
+ body = body.encode()
self.send_response(200)
self.send_header("Content-Type", "text/plain")
self.send_header("Content-Length", str(len(body)))
@@ -551,6 +580,132 @@ def route_robots(self):
if self.command != "HEAD":
self.wfile.write(body)
+ # --- sitemap ingestion (issue #712) ------------------------------------
+ # start.html links to nothing, so orphan*.html are reachable only through
+ # the sitemap. deep1.html proves the seeds keep a full depth budget; the
+ # off-host page must be dropped by the travel scope, and the off-host
+ # child sitemap by the ingester's same-host rule. The index is served both
+ # from /sitemapdir/ (named by robots.txt) and from the well-known
+ # /sitemap.xml (the fallback).
+
+ def route_sitemap_index(self):
+ host = self.headers.get("Host")
+ self.send_raw(
+ '\n'
+ ''
+ f"http://{host}/sitemapdir/pages.xml.gz"
+ "http://sitemap-offhost.invalid/s.xml"
+ "\n".encode(),
+ "application/xml",
+ )
+
+ def route_sitemap_pages(self):
+ host = self.headers.get("Host")
+ xml = (
+ '\n'
+ ''
+ f"http://{host}/sitemapdir/orphan1.html"
+ "http://sitemap-offhost.invalid/x.html"
+ f"http://{host}/sitemapdir/orphan2.html"
+ "\n"
+ ).encode()
+ self.send_raw(gzip.compress(xml), "application/x-gzip")
+
+ def route_sitemap_start(self):
+ self.send_html("\tNothing links to the sitemap pages.")
+
+ def route_sitemap_orphan1(self):
+ self.send_html('\tdeeper')
+
+ def route_sitemap_orphan2(self):
+ self.send_html("\tSecond orphan.")
+
+ def route_sitemap_deep1(self):
+ self.send_html("\tOne level below an orphan.")
+
+ # chainN is a sitemapindex at nesting level N, listing chain(N+1) and a
+ # urlset capN.xml whose single page is capN.html. Levels up to
+ # HTS_SITEMAP_MAX_LEVEL are followed, so capN.html appears for N below the
+ # cap and stops appearing at it: the pair pins the boundary, which a cap
+ # mutated either way would break.
+ def route_sitemap_chain(self):
+ host = self.headers.get("Host")
+ level = int(self.path.rsplit("/", 1)[-1][len("chain") : -len(".xml")])
+ self.send_raw(
+ (
+ '\n'
+ f"http://{host}/sitemapdir/chain{level + 1}.xml"
+ ""
+ f"http://{host}/sitemapdir/cap{level}.xml"
+ "\n"
+ ).encode(),
+ "application/xml",
+ )
+
+ def route_sitemap_capset(self):
+ host = self.headers.get("Host")
+ level = self.path.rsplit("/", 1)[-1][len("cap") : -len(".xml")]
+ self.send_raw(
+ (
+ '\n'
+ f"http://{host}/sitemapdir/cap{level}.html"
+ "\n"
+ ).encode(),
+ "application/xml",
+ )
+
+ def route_sitemap_cappage(self):
+ self.send_html("\tReached through a nested sitemapindex.")
+
+ def route_sitemap_gatedindex(self):
+ host = self.headers.get("Host")
+ self.send_raw(
+ ''
+ f"http://{host}/sitemapdir/filtered.xml"
+ "\n".encode(),
+ "application/xml",
+ )
+
+ def route_sitemap_filtered(self):
+ host = self.headers.get("Host")
+ self.send_raw(
+ ''
+ f"http://{host}/sitemapdir/gated.html"
+ "\n".encode(),
+ "application/xml",
+ )
+
+ def route_sitemap_gated(self):
+ self.send_html("\tListed only by the filtered child sitemap.")
+
+ # A moved sitemap: the marking has to follow the redirect.
+ # A root sitemap naming a page below the crawl's start directory and one
+ # above it. Only the first may be seeded when the crawl started at /deep/dir/.
+ def route_sitemap_scope(self):
+ host = self.headers.get("Host")
+ self.send_raw(
+ ''
+ f"http://{host}/deep/dir/below.html"
+ f"http://{host}/elsewhere/updir.html"
+ "\n".encode(),
+ "application/xml",
+ )
+
+ def route_sitemap_deepstart(self):
+ self.send_html("\tA start page in a subdirectory, linking nothing.")
+
+ def route_sitemap_below(self):
+ self.send_html("\tBelow the start directory.")
+
+ def route_sitemap_updir(self):
+ self.send_html("\tAbove the start directory.")
+
+ def route_sitemap_moved(self):
+ self.send_response(301)
+ self.send_header("Location", "/sitemapdir/pages.xml.gz")
+ self.send_header("Content-Length", "0")
+ self.end_headers()
+
# --- type/extension matrix (issue #267 family) -------------------------
def send_raw(self, body, content_type, extra_headers=()):
@@ -1739,6 +1894,21 @@ def route_changes2_ticker(self):
"/gated/index.php": route_gated_index,
"/gated/secret.php": route_gated_secret,
"/robots.txt": route_robots,
+ "/sitemapdir/index.xml": route_sitemap_index,
+ "/sitemap.xml": route_sitemap_index,
+ "/sitemapdir/pages.xml.gz": route_sitemap_pages,
+ "/sitemapdir/start.html": route_sitemap_start,
+ "/sitemapdir/orphan1.html": route_sitemap_orphan1,
+ "/sitemapdir/orphan2.html": route_sitemap_orphan2,
+ "/sitemapdir/deep1.html": route_sitemap_deep1,
+ "/sitemapdir/gatedindex.xml": route_sitemap_gatedindex,
+ "/sitemapdir/filtered.xml": route_sitemap_filtered,
+ "/sitemapdir/gated.html": route_sitemap_gated,
+ "/sitemapdir/moved.xml": route_sitemap_moved,
+ "/scopesitemap.xml": route_sitemap_scope,
+ "/deep/dir/start.html": route_sitemap_deepstart,
+ "/deep/dir/below.html": route_sitemap_below,
+ "/elsewhere/updir.html": route_sitemap_updir,
"/warcgz/index.html": route_warcgz_index,
"/warcgz/page.html": route_warcgz_page,
"/warcgz/data.bin": route_warcgz_data,
@@ -2084,6 +2254,13 @@ def dispatch(self):
return True
# Match percent-encoded paths (accented #157 route) by their decoded form.
handler = self.ROUTES.get(path) or self.ROUTES.get(unquote(path))
+ if handler is None:
+ if re.fullmatch(r"/sitemapdir/chain\d+\.xml", path):
+ handler = type(self).route_sitemap_chain
+ elif re.fullmatch(r"/sitemapdir/cap\d+\.xml", path):
+ handler = type(self).route_sitemap_capset
+ elif re.fullmatch(r"/sitemapdir/cap\d+\.html", path):
+ handler = type(self).route_sitemap_cappage
if handler is not None:
handler(self)
return True
diff --git a/tests/webhttrack-smoke.sh b/tests/webhttrack-smoke.sh
index 7fd86f25..821fc369 100644
--- a/tests/webhttrack-smoke.sh
+++ b/tests/webhttrack-smoke.sh
@@ -46,17 +46,20 @@ cat >"$stubdir/x-www-browser" <&2
# Also fetch an option page and require a rendered title='' tooltip: proves the
# option template expands and the \${html:} filter escapes into the attribute.
-# option9 additionally proves the WARC and change-report controls render with
-# their expanded labels, and option2 the --single-file pair: on field names plus
-# the absence of an unexpanded key, since the default locale here is French.
+# option9 additionally proves the WARC and change-report controls and
+# option2 the --single-file pair render with their expanded labels, and
+# option8 the sitemap ones. option2 also pins the absence of an
+# unexpanded key, since the default locale here is French.
opturl="\${1%/}/server/option2.html"
warcurl="\${1%/}/server/option9.html"
+smurl="\${1%/}/server/option8.html"
if body="\$(curl -fsSL --max-time 20 "\$1")" && printf '%s' "\$body" | grep -qai httrack && printf '%s' "\$body" | grep -qaF step2.html &&
opt="\$(curl -fsSL --max-time 20 "\$opturl")" && printf '%s' "\$opt" | grep -qaF "title='" &&
printf '%s' "\$opt" | grep -qaF 'name="singlefile"' && printf '%s' "\$opt" | grep -qaF 'name="singlefilemax"' &&
! printf '%s' "\$opt" | grep -qaF '\${LANG_SINGLEFILE}' &&
warc="\$(curl -fsSL --max-time 20 "\$warcurl")" && printf '%s' "\$warc" | grep -qaF 'name="warcfile"' && printf '%s' "\$warc" | grep -qaF WARC &&
- printf '%s' "\$warc" | grep -qaF 'name="changes"' && printf '%s' "\$warc" | grep -qaF hts-changes.json; then
+ printf '%s' "\$warc" | grep -qaF 'name="changes"' && printf '%s' "\$warc" | grep -qaF hts-changes.json &&
+ sm="\$(curl -fsSL --max-time 20 "\$smurl")" && printf '%s' "\$sm" | grep -qaF 'name="sitemapurl"' && printf '%s' "\$sm" | grep -qaF 'name="sitemap"'; then
echo PASS >"$marker"
else
echo "FAIL: unexpected response from \$1" >"$marker"