mirror of
https://github.com/xroche/httrack.git
synced 2026-07-27 19:12:54 +03:00
Compare commits
16 Commits
fix/failed
...
feat/chang
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6bb609ba27 | ||
|
|
1d61bf94ca | ||
|
|
25ea1d5c4c | ||
|
|
8ebe99be9d | ||
|
|
a00b95dcb4 | ||
|
|
fd20086d05 | ||
|
|
bcef3a1096 | ||
|
|
6f4b390aa5 | ||
|
|
b099e303e5 | ||
|
|
6a20dfd998 | ||
|
|
08b32f4878 | ||
|
|
a7ed9c3ca0 | ||
|
|
11c0aa59df | ||
|
|
f7f428bc93 | ||
|
|
9061f372e4 | ||
|
|
c38bf69c7d |
26
.github/workflows/ci.yml
vendored
26
.github/workflows/ci.yml
vendored
@@ -46,32 +46,10 @@ jobs:
|
||||
- name: Configure
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# Regenerate: configure and the Makefile.in's are not tracked.
|
||||
# Regenerate from configure.ac/Makefile.am to validate them; the
|
||||
# committed generated files already let a plain checkout build.
|
||||
autoreconf -fi
|
||||
# Disabling zlib must fail here rather than at link with a pile of
|
||||
# undefined minizip references (#735). Both spellings, so a rewrite
|
||||
# cannot keep one and lose the other. Probed out-of-tree to leave
|
||||
# nothing behind.
|
||||
nozlib="$RUNNER_TEMP/nozlib"
|
||||
for arg in --without-zlib --with-zlib=no; do
|
||||
rm -rf "$nozlib" && mkdir -p "$nozlib"
|
||||
if (cd "$nozlib" && "$GITHUB_WORKSPACE/configure" "$arg" >out.log 2>&1); then
|
||||
echo "::error::configure $arg succeeded; it must be rejected"
|
||||
exit 1
|
||||
fi
|
||||
# ... and for the stated reason, not an unrelated configure failure.
|
||||
grep -q "zlib cannot be disabled" "$nozlib/out.log" \
|
||||
|| { cat "$nozlib/out.log"; exit 1; }
|
||||
done
|
||||
./configure
|
||||
# Same dead end from the compile side. The bare compile is the
|
||||
# control: without it a broken probe would pass vacuously.
|
||||
hdr='#include "htsglobal.h"'
|
||||
echo "$hdr" | $CC -I. -Isrc -fsyntax-only -xc -
|
||||
if echo "$hdr" | $CC -DHTS_USEZLIB=0 -I. -Isrc -fsyntax-only -xc - 2>/dev/null; then
|
||||
echo "::error::-DHTS_USEZLIB=0 compiled; htsglobal.h must reject it"
|
||||
exit 1
|
||||
fi
|
||||
# a missing decoder would silently drop the coding from Accept-Encoding
|
||||
grep -q "define HTS_USEBROTLI 1" config.h
|
||||
grep -q "define HTS_USEZSTD 1" config.h
|
||||
|
||||
5
.github/workflows/windows-build.yml
vendored
5
.github/workflows/windows-build.yml
vendored
@@ -227,9 +227,8 @@ jobs:
|
||||
# tested nothing: pin the skips, and floor the passes in case the glob empties.
|
||||
# footer-overflow skips on Windows (needs a path past MAX_PATH); crange pending #581;
|
||||
# webdav-mime needs a reapable background listener, which MSYS cannot give it;
|
||||
# badmtime needs a filesystem that stores an mtime past gmtime's range;
|
||||
# single-file ends on a GUI half needing htsserver, which this job does not build.
|
||||
expected_skips=" 01_engine-footer-overflow.test 48_local-crange-memresume.test 71_local-crange-repaircache.test 79_local-proxytrack-webdav-mime.test 88_local-proxytrack-badmtime.test 94_local-single-file.test"
|
||||
# badmtime needs a filesystem that stores an mtime past gmtime's range.
|
||||
expected_skips=" 01_engine-footer-overflow.test 48_local-crange-memresume.test 71_local-crange-repaircache.test 79_local-proxytrack-webdav-mime.test 88_local-proxytrack-badmtime.test"
|
||||
[ "$pass" -ge 90 ] || { echo "::error::only $pass tests passed ($skip skipped)"; exit 1; }
|
||||
[ "$skipped" = "$expected_skips" ] || { echo "::error::unexpected skips:$skipped"; exit 1; }
|
||||
[ "$fail" -eq 0 ] || { echo "::error::failing:$failed"; exit 1; }
|
||||
|
||||
@@ -175,7 +175,7 @@ AX_CHECK_ALIGNED_ACCESS_REQUIRED
|
||||
# check for various headers
|
||||
AC_CHECK_HEADERS([execinfo.h sys/ioctl.h])
|
||||
|
||||
### zlib (mandatory)
|
||||
### zlib
|
||||
CHECK_ZLIB()
|
||||
|
||||
### brotli and zstd content codings (optional)
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
if FUZZERS
|
||||
noinst_PROGRAMS = fuzz-charset fuzz-meta fuzz-idna fuzz-entities \
|
||||
fuzz-unescape fuzz-filters fuzz-url fuzz-header fuzz-cachendx \
|
||||
fuzz-htsparse fuzz-singlefile fuzz-sitemap
|
||||
fuzz-htsparse
|
||||
endif
|
||||
|
||||
AM_CPPFLAGS = \
|
||||
@@ -27,8 +27,6 @@ fuzz_url_SOURCES = fuzz-url.c fuzz.h
|
||||
fuzz_header_SOURCES = fuzz-header.c fuzz.h
|
||||
fuzz_cachendx_SOURCES = fuzz-cachendx.c fuzz.h
|
||||
fuzz_htsparse_SOURCES = fuzz-htsparse.c fuzz.h
|
||||
fuzz_singlefile_SOURCES = fuzz-singlefile.c fuzz.h
|
||||
fuzz_sitemap_SOURCES = fuzz-sitemap.c fuzz.h
|
||||
|
||||
# List corpus files explicitly: automake does not expand EXTRA_DIST globs.
|
||||
EXTRA_DIST = README.md run-fuzzers.sh \
|
||||
@@ -49,10 +47,4 @@ EXTRA_DIST = README.md run-fuzzers.sh \
|
||||
corpus/cachendx/regress-overadvance.bin \
|
||||
corpus/cachendx/regress-truncated-entry.bin \
|
||||
corpus/htsparse/basic.html corpus/htsparse/script-inscript.html \
|
||||
corpus/htsparse/meta-usemap.html corpus/htsparse/malformed.html \
|
||||
corpus/singlefile/img-src.html corpus/singlefile/link-rel.html \
|
||||
corpus/singlefile/style-block.html corpus/singlefile/style-attr.html \
|
||||
corpus/singlefile/srcset.html corpus/singlefile/rawtext.html \
|
||||
corpus/singlefile/malformed.html corpus/singlefile/many-attrs.html \
|
||||
corpus/sitemap/urlset.xml corpus/sitemap/sitemapindex.xml \
|
||||
corpus/sitemap/truncated.xml corpus/sitemap/urlset.xml.gz
|
||||
corpus/htsparse/meta-usemap.html corpus/htsparse/malformed.html
|
||||
|
||||
@@ -1,3 +0,0 @@
|
||||
<img src="a.png">
|
||||
<img src=big.png alt=over-cap>
|
||||
<img src="../escape.png"><img src="/abs.png"><img src="data:,x">
|
||||
@@ -1,4 +0,0 @@
|
||||
<link rel="stylesheet" href="s.css">
|
||||
<link rel=icon href=a.png>
|
||||
<link rel="next" href="p2.html">
|
||||
<link rel="preload" href="j.js">
|
||||
@@ -1,5 +0,0 @@
|
||||
<img src="unterminated.png
|
||||
<div style="background:url(a.png">
|
||||
<style>@import url(
|
||||
<!-- unterminated comment
|
||||
<a href=
|
||||
@@ -1 +0,0 @@
|
||||
<img src="a.png" a0="v" a1="v" a2="v" a3="v" a4="v" a5="v" a6="v" a7="v" a8="v" a9="v" a10="v" a11="v" a12="v" a13="v" a14="v" a15="v" a16="v" a17="v" a18="v" a19="v" a20="v" a21="v" a22="v" a23="v" a24="v" a25="v" a26="v" a27="v" a28="v" a29="v" a30="v" a31="v" a32="v" a33="v" a34="v" a35="v" a36="v" a37="v" a38="v" a39="v" a40="v" a41="v" a42="v" a43="v" a44="v" a45="v" a46="v" a47="v" a48="v" a49="v" a50="v" a51="v" a52="v" a53="v" a54="v" a55="v" a56="v" a57="v" a58="v" a59="v" a60="v" a61="v" a62="v" a63="v" a64="v" a65="v" a66="v" a67="v" a68="v" a69="v">
|
||||
@@ -1,4 +0,0 @@
|
||||
<script>var s="</scripting>"; if(a</b) x=1;</script>
|
||||
<script src="j.js"></script>
|
||||
<textarea></textareas></textarea>
|
||||
<title></titles></title>
|
||||
@@ -1,2 +0,0 @@
|
||||
<img srcset="a.png 1x, big.png 2x, a.png 100w">
|
||||
<source srcset="a.png,, a.png 2x," src="a.png">
|
||||
@@ -1,2 +0,0 @@
|
||||
<div style="background:url(a.png);list-style:url('a.png')"></div>
|
||||
<p style='background:url("a.png")'>x</p>
|
||||
@@ -1,5 +0,0 @@
|
||||
<style>@import "s.css";
|
||||
@import url(sub/b.css);
|
||||
div{background:url(a.png)}
|
||||
/* url(a.png) */ p:after{content:"url(a.png)"}
|
||||
</style>
|
||||
@@ -1 +0,0 @@
|
||||
<sitemapindex><sitemap><loc>http://h.test/s2.xml.gz</loc></sitemap></sitemapindex>
|
||||
@@ -1 +0,0 @@
|
||||
<urlset><loc>http://h.test/x
|
||||
@@ -1 +0,0 @@
|
||||
<?xml version="1.0"?><urlset><url><loc>http://h.test/a.html</loc></url><url><loc>https://h.test/b?x=1&y=2</loc></url></urlset>
|
||||
Binary file not shown.
@@ -1,131 +0,0 @@
|
||||
/* ------------------------------------------------------------ */
|
||||
/*
|
||||
HTTrack Website Copier, Offline Browser for Windows and Unix
|
||||
Copyright (C) 2026 Xavier Roche and other contributors
|
||||
|
||||
SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License
|
||||
along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
Ethical use: we kindly ask that you NOT use this software to harvest email
|
||||
addresses or to collect any other private information about people. Doing so
|
||||
would dishonor our work and waste the many hours we have spent on it.
|
||||
|
||||
Please visit our Website: http://www.httrack.com
|
||||
*/
|
||||
|
||||
/* Fuzz the --single-file rewriter (htssinglefile.c): hostile HTML walked
|
||||
through the tag, CSS url()/@import and srcset parsers, then re-serialized.
|
||||
The resolver is aimed at a private temp tree, so the inlining half (MIME
|
||||
guess, base64, nested stylesheet) is reached and nothing else on disk is. */
|
||||
#include "fuzz.h"
|
||||
|
||||
#include "httrack-library.h"
|
||||
#include "htssinglefile.h"
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/* Between a.png and big.png, so one input reaches both the inline path and the
|
||||
over-cap fallback. */
|
||||
#define FUZZ_SF_CAP 64
|
||||
|
||||
static char sf_root[512];
|
||||
static char sf_page[600];
|
||||
|
||||
/* The asset tree, in removal order: the subdirectory comes after its file. */
|
||||
static const char *const sf_files[] = {"a.png", "big.png", "j.js", "s.css",
|
||||
"sub/b.css", "sub", NULL};
|
||||
|
||||
static void sf_cleanup(void) {
|
||||
char path[700];
|
||||
int i;
|
||||
|
||||
for (i = 0; sf_files[i] != NULL; i++) {
|
||||
snprintf(path, sizeof(path), "%s/%s", sf_root, sf_files[i]);
|
||||
(void) remove(path);
|
||||
}
|
||||
(void) remove(sf_root);
|
||||
}
|
||||
|
||||
/* A missing asset would silently reduce the target to its parser half. */
|
||||
static void sf_write(const char *name, const char *data, size_t len) {
|
||||
char path[700];
|
||||
FILE *fp;
|
||||
|
||||
snprintf(path, sizeof(path), "%s/%s", sf_root, name);
|
||||
fp = fopen(path, "wb");
|
||||
if (fp == NULL || fwrite(data, 1, len, fp) != len)
|
||||
abort();
|
||||
fclose(fp);
|
||||
}
|
||||
|
||||
static void sf_text(const char *name, const char *data) {
|
||||
sf_write(name, data, strlen(data));
|
||||
}
|
||||
|
||||
static void sf_init(void) {
|
||||
static const char png[] = "\x89PNG\r\n\x1a\n";
|
||||
static const char big[4096] = "\x89PNG";
|
||||
const char *tmp = getenv("TMPDIR");
|
||||
char path[700];
|
||||
|
||||
hts_init();
|
||||
snprintf(sf_root, sizeof(sf_root), "%s/httrack-fuzz-sf-XXXXXX",
|
||||
tmp != NULL && tmp[0] != '\0' ? tmp : "/tmp");
|
||||
if (mkdtemp(sf_root) == NULL)
|
||||
abort();
|
||||
atexit(sf_cleanup);
|
||||
snprintf(sf_page, sizeof(sf_page), "%s/page.html", sf_root);
|
||||
snprintf(path, sizeof(path), "%s/sub", sf_root);
|
||||
if (mkdir(path, 0700) != 0)
|
||||
abort();
|
||||
sf_write("a.png", png, sizeof(png) - 1);
|
||||
sf_write("big.png", big, sizeof(big));
|
||||
sf_text("j.js", "var x=1;\n");
|
||||
/* @import plus a url(), so an inlined stylesheet recurses and its own
|
||||
relative reference is rebased. */
|
||||
sf_text("s.css", "@import url(sub/b.css);\ndiv{background:url(a.png)}\n");
|
||||
sf_text("sub/b.css", "p{background:url(../a.png)}\n");
|
||||
}
|
||||
|
||||
int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) {
|
||||
static int inited = 0;
|
||||
|
||||
String out = STRING_EMPTY;
|
||||
httrackp *opt;
|
||||
/* Exact-length, unterminated: the rewriter is span-based, so ASan bounds a
|
||||
read past html_len instead of it landing on a terminator. */
|
||||
char *html = malloct(size != 0 ? size : 1);
|
||||
|
||||
if (!inited) {
|
||||
sf_init();
|
||||
inited = 1;
|
||||
}
|
||||
memcpy(html, data, size);
|
||||
|
||||
opt = hts_create_opt();
|
||||
opt->log = opt->errlog = NULL;
|
||||
opt->single_file_max_size = FUZZ_SF_CAP;
|
||||
|
||||
StringClear(out);
|
||||
(void) singlefile_rewrite_html(opt, sf_root, sf_page, html, size,
|
||||
SINGLEFILE_MAX_PAGE_SIZE, &out);
|
||||
|
||||
StringFree(out);
|
||||
freet(html);
|
||||
hts_free_opt(opt);
|
||||
return 0;
|
||||
}
|
||||
@@ -1,60 +0,0 @@
|
||||
/* ------------------------------------------------------------ */
|
||||
/*
|
||||
HTTrack Website Copier, Offline Browser for Windows and Unix
|
||||
Copyright (C) 2026 Xavier Roche and other contributors
|
||||
|
||||
SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License
|
||||
along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
Ethical use: we kindly ask that you NOT use this software to harvest email
|
||||
addresses or to collect any other private information about people. Doing so
|
||||
would dishonor our work and waste the many hours we have spent on it.
|
||||
|
||||
Please visit our Website: http://www.httrack.com
|
||||
*/
|
||||
|
||||
/* Fuzz the sitemap <loc> scanner (htssitemap.c): raw XML, gzip-framed bodies
|
||||
and truncated streams all arrive here straight off the network. */
|
||||
#include "fuzz.h"
|
||||
#include "htssitemap.h"
|
||||
|
||||
static hts_boolean sm_count(void *arg, const char *url) {
|
||||
int *const n = (int *) arg;
|
||||
|
||||
(void) url;
|
||||
(*n)++;
|
||||
return HTS_TRUE;
|
||||
}
|
||||
|
||||
int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) {
|
||||
static const int caps[] = {0, 1, 16, HTS_SITEMAP_MAX_URLS_DOC};
|
||||
hts_boolean is_index;
|
||||
char *body;
|
||||
int n = 0, cap;
|
||||
|
||||
if (size == 0)
|
||||
return 0;
|
||||
cap = caps[data[0] % (sizeof(caps) / sizeof(caps[0]))];
|
||||
data++, size--;
|
||||
/* A heap copy of exactly `size` bytes: the scanner must never rely on a
|
||||
terminator, and ASan turns any overread into a report. */
|
||||
body = malloct(size != 0 ? size : 1);
|
||||
memcpy(body, data, size);
|
||||
|
||||
(void) hts_sitemap_scan(body, size, cap, &is_index, sm_count, &n);
|
||||
|
||||
freet(body);
|
||||
return 0;
|
||||
}
|
||||
@@ -163,26 +163,8 @@ the index" problems disappear.</p>
|
||||
<tr><td><tt>--near (-n)</tt></td><td>Also fetch non-HTML files "near" a followed link, such as an image linked from a page you kept but hosted elsewhere.</td></tr>
|
||||
<tr><td><tt>--ext-depth (-%e)</tt></td><td>How many levels of external links to follow once the crawl leaves your scope (default 0).</td></tr>
|
||||
<tr><td><tt>--test (-t)</tt></td><td>Also HEAD-test links that fall outside the scope, which are normally refused, without downloading them: a way to see what scope is excluding.</td></tr>
|
||||
<tr><td><tt>--sitemap (-%m), --sitemap-url URL (-%mu)</tt></td><td>Also take start URLs from the site's sitemap, for pages nothing links to. Off by default.</td></tr>
|
||||
</table>
|
||||
|
||||
<p>Link-following only finds what something links to. Anything a site publishes
|
||||
solely in its sitemap is invisible to HTTrack unless you ask for it.
|
||||
<tt>--sitemap</tt> reads the start host's <tt>robots.txt</tt> for
|
||||
<tt>Sitemap:</tt> lines and falls back to <tt>/sitemap.xml</tt>;
|
||||
<tt>--sitemap-url</tt> names one directly. Nested <tt>sitemapindex</tt> files
|
||||
and gzipped <tt>.xml.gz</tt> sitemaps are followed. The URLs found become start
|
||||
URLs with the full depth budget, but they still go through your filters and
|
||||
scope rules, so a sitemap cannot widen a crawl you deliberately narrowed. It is
|
||||
off by default because a sitemap can list thousands of pages nothing links
|
||||
to.</p>
|
||||
|
||||
<p>One surprise worth knowing: a sitemap you name with <tt>--sitemap-url</tt>,
|
||||
and one the site itself declares in <tt>robots.txt</tt>, are fetched even when
|
||||
<tt>robots.txt</tt> disallows that path, because naming or declaring a sitemap
|
||||
is an invitation to read it. Only the guessed <tt>/sitemap.xml</tt> obeys a
|
||||
<tt>Disallow</tt>. The URLs listed inside are gated normally either way.</p>
|
||||
|
||||
<p>The single most common surprise is "only the home page came down." That is
|
||||
usually not a scope option at all: it is an off-host redirect. A start URL of
|
||||
<tt>http://example.com/</tt> that redirects to <tt>https://www.example.com/</tt>
|
||||
@@ -318,7 +300,6 @@ ones it kept so the local copy browses offline. These options tune both halves.<
|
||||
<tr><td><tt>--preserve (-%p), --disable-passwords (-%x)</tt></td><td>Leave HTML untouched (no rewriting), and strip passwords out of saved links.</td></tr>
|
||||
<tr><td><tt>--extended-parsing (-%P), --parse-java (-j)</tt></td><td>Aggressive link discovery, and how much script content is parsed for links.</td></tr>
|
||||
<tr><td><tt>--mime-html (-%M)</tt></td><td>Save the whole mirror as a single MIME-encapsulated <tt>.mht</tt> archive (<tt>index.mht</tt>).</td></tr>
|
||||
<tr><td><tt>--single-file (-%Z), --single-file-max-size N</tt></td><td>Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded as <tt>data:</tt> URIs. Assets over the cap (10 MB by default) keep their link, as do audio, video, and the links from one page to another. A sibling of <tt>-%M</tt>, not a replacement: see the recipe below for which to pick.</td></tr>
|
||||
<tr><td><tt>--index (-I), --build-top-index (-%i), --search-index (-%I)</tt></td><td>Build a per-mirror index, a top index across projects, and a searchable keyword index.</td></tr>
|
||||
</table>
|
||||
|
||||
@@ -510,25 +491,6 @@ with the same content it served last time lands in <tt>unchanged</tt>. Deletions
|
||||
are reported whether or not <tt>--purge-old</tt> is deleting them. The format is
|
||||
documented in <a href="changes.html">the change report specification</a>.</small></p>
|
||||
|
||||
<h4>Pages you can hand to someone as one file</h4>
|
||||
<p><tt>httrack https://example.com/ --single-file --path mydir</tt><br>
|
||||
<small>Rewrites each saved page after the crawl so its stylesheets, scripts,
|
||||
images and fonts are embedded as <tt>data:</tt> URIs. The mirror stays a normal
|
||||
browsable tree, with links between pages relative and the assets still on disk,
|
||||
but any single <tt>.html</tt> file can now be mailed or archived on its own.
|
||||
Raise or lower the 10 MB per-asset limit with
|
||||
<tt>--single-file-max-size N</tt>; anything over it, plus audio and video, keeps
|
||||
its link. One caveat if you raise it: an inlined stylesheet becomes a
|
||||
<tt>data:</tt> URL, whose path is opaque, so an asset it referenced that stayed
|
||||
over the cap no longer resolves from inside it. Raising the cap past that asset
|
||||
embeds it too and the question goes away.<br>
|
||||
Reach for this when the file has to open for someone you cannot make assumptions
|
||||
about: it is plain HTML and needs no add-on. Reach for <tt>--mime-html</tt>
|
||||
instead when a Chromium-family browser is a given and the mirror is large: MIME
|
||||
carries text parts without the base64 tax, keeps each resource's original URL,
|
||||
and stores a shared asset once rather than re-embedding it in every page that
|
||||
uses it.</small></p>
|
||||
|
||||
<h4>HTTrack as a fetch tool</h4>
|
||||
<p><tt>httrack --get https://host/file.bin --path tmp</tt><br>
|
||||
<small><tt>--get</tt> fetches one file with cache, index, depth, cookies and robots
|
||||
|
||||
@@ -87,13 +87,12 @@ offline browser : copy websites to a local directory</p>
|
||||
--host-control[=N]</b> ] [ <b>-%P,
|
||||
--extended-parsing[=N]</b> ] [ <b>-n, --near</b> ] [ <b>-t,
|
||||
--test</b> ] [ <b>-%L, --list</b> ] [ <b>-%S, --urllist</b>
|
||||
] [ <b>-%m, --sitemap</b> ] [ <b>-NN, --structure[=N]</b> ]
|
||||
[ <b>-%N, --delayed-type-check</b> ] [ <b>-%D,
|
||||
] [ <b>-NN, --structure[=N]</b> ] [ <b>-%N,
|
||||
--delayed-type-check</b> ] [ <b>-%D,
|
||||
--cached-delayed-type-check</b> ] [ <b>-%M, --mime-html</b>
|
||||
] [ <b>-%Z, --single-file</b> ] [ <b>-LN,
|
||||
--long-names[=N]</b> ] [ <b>-KN, --keep-links[=N]</b> ] [
|
||||
<b>-x, --replace-external</b> ] [ <b>-%x,
|
||||
--disable-passwords</b> ] [ <b>-%q,
|
||||
] [ <b>-LN, --long-names[=N]</b> ] [ <b>-KN,
|
||||
--keep-links[=N]</b> ] [ <b>-x, --replace-external</b> ] [
|
||||
<b>-%x, --disable-passwords</b> ] [ <b>-%q,
|
||||
--include-query-string</b> ] [ <b>-%g, --strip-query</b> ] [
|
||||
<b>-o, --generate-errors</b> ] [ <b>-X, --purge-old[=N]</b>
|
||||
] [ <b>-%p, --preserve</b> ] [ <b>-%T, --utf8-conversion</b>
|
||||
@@ -577,22 +576,6 @@ URL per line) (--list <param>)</p></td></tr>
|
||||
|
||||
<p><file> add all scan rules located in this text
|
||||
file (one scan rule per line) (--urllist <param>)</p></td></tr>
|
||||
<tr valign="top" align="left">
|
||||
<td width="9%"></td>
|
||||
<td width="4%">
|
||||
|
||||
|
||||
<p>-%m</p></td>
|
||||
<td width="5%"></td>
|
||||
<td width="82%">
|
||||
|
||||
|
||||
<p>seed the crawl from the site’s sitemap (robots.txt
|
||||
Sitemap:, then /sitemap.xml); --sitemap-url URL names one
|
||||
explicitly. A sitemap you name, or one the site declares, is
|
||||
fetched even under robots.txt Disallow; only the guessed
|
||||
/sitemap.xml obeys it. The URLs found still pass every
|
||||
filter and scope rule (--sitemap)</p></td></tr>
|
||||
</table>
|
||||
|
||||
<h3>Build options:
|
||||
@@ -666,24 +649,6 @@ don’t wait) (--cached-delayed-type-check)</p></td></tr>
|
||||
<td width="4%">
|
||||
|
||||
|
||||
<p>-%Z</p></td>
|
||||
<td width="5%"></td>
|
||||
<td width="82%">
|
||||
|
||||
|
||||
<p>after the mirror, rewrite each saved page with its
|
||||
stylesheets, scripts, images and fonts inlined as data:
|
||||
URIs, so any page opens by double-click anywhere (links
|
||||
between pages stay relative; audio and video stay links);
|
||||
--single-file-max-size N caps each asset (default 10485760
|
||||
bytes). %M is the better container where a Chromium-family
|
||||
browser is a given: one archive, no base64 tax on text, a
|
||||
shared asset stored once (--single-file)</p></td></tr>
|
||||
<tr valign="top" align="left">
|
||||
<td width="9%"></td>
|
||||
<td width="4%">
|
||||
|
||||
|
||||
<p>-%t</p></td>
|
||||
<td width="5%"></td>
|
||||
<td width="82%">
|
||||
|
||||
@@ -104,7 +104,6 @@ ${do:end-if}
|
||||
<input type="hidden" name="hidepwd" value="">
|
||||
<input type="hidden" name="hidequery" value="">
|
||||
<input type="hidden" name="nopurge" value="">
|
||||
<input type="hidden" name="singlefile" value="">
|
||||
|
||||
${LANG_I33}
|
||||
<br>
|
||||
@@ -144,13 +143,6 @@ ${listid:build:LISTDEF_3}
|
||||
<tr><td><input type="checkbox" name="nopurge" ${checked:nopurge}
|
||||
title='${html:LANG_I1a}' onMouseOver="info('${html:LANG_I1a}'); return true" onMouseOut="info(' '); return true"
|
||||
> ${LANG_I57}</td></tr>
|
||||
<tr><td><input type="checkbox" name="singlefile" ${checked:singlefile}
|
||||
title='${html:LANG_SINGLEFILETIP}' onMouseOver="info('${html:LANG_SINGLEFILETIP}'); return true" onMouseOut="info(' '); return true"
|
||||
> ${LANG_SINGLEFILE}</td></tr>
|
||||
<tr><td>${LANG_SINGLEFILEMAX}
|
||||
<input name="singlefilemax" value="${singlefilemax}" size="12"
|
||||
title='${html:LANG_SINGLEFILEMAXTIP}' onMouseOver="info('${html:LANG_SINGLEFILEMAXTIP}'); return true" onMouseOut="info(' '); return true"
|
||||
></td></tr>
|
||||
</table>
|
||||
|
||||
<tr><td>
|
||||
|
||||
@@ -108,7 +108,6 @@ ${do:end-if}
|
||||
<input type="hidden" name="keepqueryorder" value="">
|
||||
<input type="hidden" name="toler" value="">
|
||||
<input type="hidden" name="http10" value="">
|
||||
<input type="hidden" name="sitemap" value="">
|
||||
|
||||
<input type="checkbox" name="cookies" ${checked:cookies}
|
||||
title='${html:LANG_I1b}' onMouseOver="info('${html:LANG_I1b}'); return true" onMouseOut="info(' '); return true"
|
||||
@@ -144,17 +143,6 @@ ${listid:robots:LISTDEF_8}
|
||||
</select>
|
||||
<br><br>
|
||||
|
||||
<input type="checkbox" name="sitemap" ${checked:sitemap}
|
||||
title='${html:LANG_SITEMAPTIP}' onMouseOver="info('${html:LANG_SITEMAPTIP}'); return true" onMouseOut="info(' '); return true"
|
||||
> ${LANG_SITEMAP}
|
||||
<br><br>
|
||||
|
||||
${LANG_SITEMAPURL}
|
||||
<input name="sitemapurl" value="${sitemapurl}" size="40"
|
||||
title='${html:LANG_SITEMAPURLTIP}' onMouseOver="info('${html:LANG_SITEMAPURLTIP}'); return true" onMouseOut="info(' '); return true"
|
||||
>
|
||||
<br><br>
|
||||
|
||||
<input type="checkbox" name="updhack" ${checked:updhack}
|
||||
title='${html:LANG_I1k}' onMouseOver="info('${html:LANG_I1k}'); return true" onMouseOut="info(' '); return true"
|
||||
> ${LANG_I62b}
|
||||
|
||||
@@ -141,13 +141,9 @@ ${do:copy:KeepSlashes:keepslashes}
|
||||
${do:copy:KeepQueryOrder:keepqueryorder}
|
||||
${do:copy:StripQuery:stripquery}
|
||||
${do:copy:StoreAllInCache:cache2}
|
||||
${do:copy:Sitemap:sitemap}
|
||||
${do:copy:SitemapUrl:sitemapurl}
|
||||
${do:copy:Warc:warc}
|
||||
${do:copy:WarcFile:warcfile}
|
||||
${do:copy:Changes:changes}
|
||||
${do:copy:SingleFile:singlefile}
|
||||
${do:copy:SingleFileMaxSize:singlefilemax}
|
||||
${do:copy:LogType:logtype}
|
||||
${do:copy:UseHTTPProxyForFTP:ftpprox}
|
||||
${do:copy:ProxyType:proxytype}
|
||||
|
||||
@@ -188,13 +188,9 @@ ${/* -m<n> resets the html limit, so the bare form must precede the -m,<n> one *
|
||||
${test:toler:--tolerant}
|
||||
${test:http10:--http-10}
|
||||
${test:cache2:--store-all-in-cache}
|
||||
${test:sitemap:--sitemap}
|
||||
${test:sitemapurl:--sitemap-url "}${html:sitemapurl}${test:sitemapurl:"}
|
||||
${test:warc:--warc}
|
||||
${test:warcfile:--warc-file "}${arg:warcfile}${test:warcfile:"}
|
||||
${test:changes:--changes}
|
||||
${test:singlefile:--single-file}
|
||||
${test:singlefilemax:--single-file-max-size=}${unquoted:singlefilemax}
|
||||
${test:norecatch:--do-not-recatch}
|
||||
${test:logf:--single-log}
|
||||
${test:logtype:::--extra-log:--debug-log}
|
||||
@@ -245,13 +241,9 @@ KeepSlashes=${ztest:keepslashes:0:1}
|
||||
KeepQueryOrder=${ztest:keepqueryorder:0:1}
|
||||
StripQuery=${stripquery}
|
||||
StoreAllInCache=${ztest:cache2:0:1}
|
||||
Sitemap=${ztest:sitemap:0:1}
|
||||
SitemapUrl=${sitemapurl}
|
||||
Warc=${ztest:warc:0:1}
|
||||
WarcFile=${warcfile}
|
||||
Changes=${ztest:changes:0:1}
|
||||
SingleFile=${ztest:singlefile:0:1}
|
||||
SingleFileMaxSize=${singlefilemax}
|
||||
LogType=${logtype}
|
||||
UseHTTPProxyForFTP=${ztest:ftpprox:0:1}
|
||||
ProxyType=${proxytype}
|
||||
|
||||
16
lang.def
16
lang.def
@@ -1046,19 +1046,3 @@ LANG_CHANGES
|
||||
Report what changed since the previous mirror
|
||||
LANG_CHANGESTIP
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
LANG_SINGLEFILE
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
LANG_SINGLEFILETIP
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
LANG_SINGLEFILEMAX
|
||||
Largest inlined asset (bytes):
|
||||
LANG_SINGLEFILEMAXTIP
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
LANG_SITEMAP
|
||||
Seed the crawl from the site's sitemap
|
||||
LANG_SITEMAPTIP
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
LANG_SITEMAPURL
|
||||
Sitemap address:
|
||||
LANG_SITEMAPURLTIP
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Îò÷åò çà ïðîìåíèòå ñïðÿìî ïðåäèøíîòî îãëåäàëî
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Çàïèñâàíå è íà hts-changes.json ñúñ ñïèñúê íà íîâèòå, ïðîìåíåíèòå, íåïðîìåíåíèòå è èç÷åçíàëèòå ôàéëîâå ñïðÿìî ïðåäèøíîòî îãëåäàëî.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Âãðàæäàíå íà ðåñóðñèòå êàòî data: URI (ñàìîñòîÿòåëíè ñòðàíèöè)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Ñëåä çàâúðøâàíå íà îãëåäàëîòî âñÿêà çàïàçåíà ñòðàíèöà ñå ïðåçàïèñâà ñ âãðàäåíè ñòèëîâå, ñêðèïòîâå, èçîáðàæåíèÿ è øðèôòîâå, òàêà ÷å ñòðàíèöàòà äà ìîæå äà ñå îòâîðè è ñàìîñòîÿòåëíî.
|
||||
Largest inlined asset (bytes):
|
||||
Íàé-ãîëÿì âãðàäåí ðåñóðñ (áàéòîâå):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Ðåñóðñ íàä òîçè ðàçìåð çàïàçâà îáèêíîâåíà âðúçêà; îñòàâåòå ïðàçíî çà ñòîéíîñòòà ïî ïîäðàçáèðàíå îò 10485760 áàéòà.
|
||||
Seed the crawl from the site's sitemap
|
||||
Çàïî÷âàíå íà îáõîæäàíåòî îò êàðòàòà íà ñàéòà
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Ïðî÷èòàíå íà êàðòàòà íà ñàéòà (ðåäîâåòå Sitemap: â robots.txt, ñëåä òîâà /sitemap.xml) è äîáàâÿíå íà âñåêè ïîñî÷åí URL àäðåñ êàòî íà÷àëåí.
|
||||
Sitemap address:
|
||||
Àäðåñ íà êàðòàòà íà ñàéòà:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Àäðåñ íà êàðòà íà ñàéòà, êîÿòî äà áúäå ïðî÷åòåíà âìåñòî ñîíäèðàíå íà ñàéòà; îñòàâåòå ïðàçíî, çà äà ñå ïðîâåðè robots.txt, ñëåä òîâà /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Informar de los cambios desde la copia anterior
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Escribir también hts-changes.json con la lista de lo que esta captura deja como nuevo, modificado, sin cambios o desaparecido respecto a la copia anterior.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Incrustar los recursos como URI data: (páginas autónomas)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Una vez terminada la copia, reescribir cada página guardada con sus hojas de estilo, scripts, imágenes y fuentes incrustadas, de modo que una página también pueda abrirse por sí sola.
|
||||
Largest inlined asset (bytes):
|
||||
Tamaño máximo del recurso incrustado (bytes):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Un recurso mayor que este tamaño conserva un enlace normal; déjelo vacío para el valor predeterminado de 10485760 bytes.
|
||||
Seed the crawl from the site's sitemap
|
||||
Iniciar el rastreo desde el mapa del sitio
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Leer el mapa del sitio (líneas Sitemap: de robots.txt, luego /sitemap.xml) y añadir como dirección inicial cada URL que incluya.
|
||||
Sitemap address:
|
||||
Dirección del mapa del sitio:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Dirección de un mapa del sitio que leer en lugar de sondear el sitio; déjelo vacío para sondear robots.txt y luego /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Hlásit, co se od pøedchozího zrcadlení zmìnilo
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Zapsat také hts-changes.json se seznamem toho, co je oproti pøedchozímu zrcadlení nové, zmìnìné, nezmìnìné nebo chybìjící.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Vložit zdroje jako URI data: (samostatné stránky)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Po dokonèení zrcadlení pøepsat každou uloženou stránku s vloženými styly, skripty, obrázky a písmy, aby ji bylo možné otevøít i samostatnì.
|
||||
Largest inlined asset (bytes):
|
||||
Nejvìtší vložený zdroj (bajty):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Zdroj vìtší než tato velikost si ponechá bìžný odkaz; ponechte prázdné pro výchozí hodnotu 10485760 bajtù.
|
||||
Seed the crawl from the site's sitemap
|
||||
Zahájit procházení z mapy webu
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Naèíst mapu webu (øádky Sitemap: v souboru robots.txt, poté /sitemap.xml) a pøidat každou uvedenou adresu URL jako výchozí.
|
||||
Sitemap address:
|
||||
Adresa mapy webu:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adresa mapy webu, která se má naèíst místo zjiš<69>ování na webu; ponechte prázdné pro zjištìní z robots.txt a poté /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
回報自上次鏡射以來的變更
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
同時寫入 hts-changes.json,列出這次擷取相對於上次鏡射的新增、變更、未變更與消失的項目。
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
將資源內嵌為 data: URI(獨立網頁)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
鏡射完成後,重寫每個已儲存的網頁,將樣式表、指令碼、圖片與字型內嵌其中,讓網頁也能單獨開啟。
|
||||
Largest inlined asset (bytes):
|
||||
內嵌資源大小上限(位元組):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
超過此大小的資源會保留一般連結;留空則使用預設的 10485760 位元組。
|
||||
Seed the crawl from the site's sitemap
|
||||
從網站的 Sitemap 開始擷取
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
讀取網站的 Sitemap(robots.txt 中的 Sitemap: 行,然後 /sitemap.xml),並將其中列出的每個網址加入為起始網址。
|
||||
Sitemap address:
|
||||
Sitemap 位址:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
要讀取的 Sitemap 位址,用來取代自動探測;留空則先探測 robots.txt 再探測 /sitemap.xml。
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
报告自上次镜像以来的变更
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
同时写入 hts-changes.json,列出本次抓取相对于上次镜像的新增、更改、未更改和消失的项目。
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
将资源内嵌为 data: URI(独立网页)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
镜像完成后,重写每个已保存的网页,将样式表、脚本、图片和字体内嵌其中,使网页也能单独打开。
|
||||
Largest inlined asset (bytes):
|
||||
内嵌资源大小上限(字节):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
超过此大小的资源会保留普通链接;留空则使用默认的 10485760 字节。
|
||||
Seed the crawl from the site's sitemap
|
||||
从网站的 Sitemap 开始抓取
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
读取网站的 Sitemap(robots.txt 中的 Sitemap: 行,然后 /sitemap.xml),并将其中列出的每个网址添加为起始网址。
|
||||
Sitemap address:
|
||||
Sitemap 地址:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
要读取的 Sitemap 地址,用来代替自动探测;留空则先探测 robots.txt 再探测 /sitemap.xml。
|
||||
|
||||
@@ -970,19 +970,3 @@ Report what changed since the previous mirror
|
||||
Prijavi ¹to se promijenilo od prethodnog zrcaljenja
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Zapi¹i i hts-changes.json s popisom onoga ¹to je u odnosu na prethodno zrcaljenje novo, promijenjeno, nepromijenjeno ili nestalo.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Ugradi resurse kao data: URI-je (samostalne stranice)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Nakon dovr¹etka zrcaljenja ponovno zapi¹i svaku spremljenu stranicu s ugraðenim stilskim listovima, skriptama, slikama i fontovima, tako da se stranica mo¾e otvoriti i zasebno.
|
||||
Largest inlined asset (bytes):
|
||||
Najveæi ugraðeni resurs (bajtovi):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Resurs veæi od ove velièine zadr¾ava obiènu poveznicu; ostavite prazno za zadanih 10485760 bajtova.
|
||||
Seed the crawl from the site's sitemap
|
||||
Pokreni pretra¾ivanje iz karte web-mjesta
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Proèitaj kartu web-mjesta (retke Sitemap: iz robots.txt, zatim /sitemap.xml) i dodaj svaki navedeni URL kao poèetnu adresu.
|
||||
Sitemap address:
|
||||
Adresa karte web-mjesta:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adresa karte web-mjesta koju treba proèitati umjesto ispitivanja web-mjesta; ostavite prazno za ispitivanje robots.txt pa /sitemap.xml.
|
||||
|
||||
@@ -1016,19 +1016,3 @@ Report what changed since the previous mirror
|
||||
Rapportér hvad der er ændret siden den forrige spejling
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Skriv også hts-changes.json med en liste over, hvad denne gennemgang efterlader som nyt, ændret, uændret eller forsvundet i forhold til den forrige spejling.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Indlejr ressourcer som data:-URI'er (selvstændige sider)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Når spejlingen er færdig, omskrives hver gemt side med dens typografiark, scripts, billeder og skrifttyper indlejret, så en side også kan åbnes alene.
|
||||
Largest inlined asset (bytes):
|
||||
Største indlejrede ressource (byte):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
En ressource over denne størrelse beholder et almindeligt link; lad feltet stå tomt for standardværdien på 10485760 byte.
|
||||
Seed the crawl from the site's sitemap
|
||||
Start gennemgangen fra webstedets sitemap
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Læs webstedets sitemap (Sitemap:-linjer i robots.txt, derefter /sitemap.xml) og tilføj hver angivet URL som startadresse.
|
||||
Sitemap address:
|
||||
Sitemap-adresse:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adressen på et sitemap, der skal læses i stedet for at undersøge webstedet; lad feltet stå tomt for at undersøge robots.txt og derefter /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Melden, was sich seit der vorherigen Spiegelung geändert hat
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Zusätzlich hts-changes.json schreiben, das auflistet, was dieser Durchlauf gegenüber der vorherigen Spiegelung als neu, geändert, unverändert oder verschwunden hinterlässt.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Ressourcen als data:-URIs einbetten (eigenständige Seiten)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Nach Abschluss der Spiegelung jede gespeicherte Seite mit eingebetteten Stylesheets, Skripten, Bildern und Schriftarten neu schreiben, sodass eine Seite auch für sich allein geöffnet werden kann.
|
||||
Largest inlined asset (bytes):
|
||||
Größte eingebettete Ressource (Bytes):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Eine Ressource über dieser Größe behält einen gewöhnlichen Link; leer lassen für den Standardwert von 10485760 Bytes.
|
||||
Seed the crawl from the site's sitemap
|
||||
Erfassung mit der Sitemap der Website beginnen
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Die Sitemap der Website lesen (Sitemap:-Zeilen in robots.txt, dann /sitemap.xml) und jede dort aufgeführte URL als Startadresse hinzufügen.
|
||||
Sitemap address:
|
||||
Sitemap-Adresse:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adresse einer Sitemap, die anstelle der Suche auf der Website gelesen wird; leer lassen, um robots.txt und dann /sitemap.xml zu prüfen.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Teata, mis on eelmisest peegeldusest saadik muutunud
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Kirjuta ka hts-changes.json, mis loetleb, mis on võrreldes eelmise peegeldusega uus, muutunud, muutumatu või kadunud.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Manusta ressursid data: URI-dena (iseseisvad lehed)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Pärast peegelduse lõppu kirjuta iga salvestatud leht ümber, manustades selle laaditabelid, skriptid, pildid ja fondid, nii et lehte saab avada ka eraldi.
|
||||
Largest inlined asset (bytes):
|
||||
Suurim manustatud ressurss (baiti):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Sellest suurem ressurss säilitab tavalise lingi; jäta tühjaks vaikeväärtuse 10485760 baiti jaoks.
|
||||
Seed the crawl from the site's sitemap
|
||||
Alusta kogumist saidi saidikaardist
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Loe saidi saidikaarti (robots.txt-i Sitemap:-read, seejärel /sitemap.xml) ja lisa iga seal loetletud URL alguslingina.
|
||||
Sitemap address:
|
||||
Saidikaardi aadress:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Saidikaardi aadress, mida lugeda saidi sondeerimise asemel; jäta tühjaks, et kontrollida robots.txt-i ja seejärel /sitemap.xml-i.
|
||||
|
||||
@@ -1016,19 +1016,3 @@ Report what changed since the previous mirror
|
||||
Report what changed since the previous mirror
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Largest inlined asset (bytes):
|
||||
Largest inlined asset (bytes):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Seed the crawl from the site's sitemap
|
||||
Seed the crawl from the site's sitemap
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Sitemap address:
|
||||
Sitemap address:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
|
||||
@@ -970,19 +970,3 @@ Report what changed since the previous mirror
|
||||
Raportoi, mikä on muuttunut edellisen peilauksen jälkeen
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Kirjoita myös hts-changes.json, joka luettelee, mikä on edelliseen peilaukseen verrattuna uutta, muuttunutta, muuttumatonta tai kadonnutta.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Upota resurssit data:-URI-osoitteina (itsenäiset sivut)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Kun peilaus on valmis, kirjoita jokainen tallennettu sivu uudelleen niin, että sen tyylitiedostot, komentosarjat, kuvat ja fontit upotetaan, jolloin sivun voi avata myös yksinään.
|
||||
Largest inlined asset (bytes):
|
||||
Suurin upotettu resurssi (tavua):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Tätä suurempi resurssi säilyttää tavallisen linkin; jätä tyhjäksi, jolloin käytetään oletusarvoa 10485760 tavua.
|
||||
Seed the crawl from the site's sitemap
|
||||
Aloita haku sivuston sivukartasta
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Lue sivuston sivukartta (robots.txt-tiedoston Sitemap:-rivit, sitten /sitemap.xml) ja lisää jokainen siinä lueteltu URL-osoite aloitusosoitteeksi.
|
||||
Sitemap address:
|
||||
Sivukartan osoite:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Luettavan sivukartan osoite sivuston luotaamisen sijaan; jätä tyhjäksi, jolloin tarkistetaan robots.txt ja sitten /sitemap.xml.
|
||||
|
||||
@@ -1016,19 +1016,3 @@ Report what changed since the previous mirror
|
||||
Signaler ce qui a changé depuis le miroir précédent
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Écrire aussi hts-changes.json, qui liste ce que ce crawl laisse nouveau, modifié, inchangé ou disparu par rapport au miroir précédent.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Intégrer les ressources en URI data: (pages autonomes)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Une fois le miroir terminé, réécrire chaque page enregistrée avec ses feuilles de style, scripts, images et polices intégrés, afin qu'une page puisse aussi être ouverte seule.
|
||||
Largest inlined asset (bytes):
|
||||
Taille maximale d'une ressource intégrée (octets) :
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Au-delà de cette taille, la ressource reste un lien ordinaire ; laissez vide pour la valeur par défaut de 10485760 octets.
|
||||
Seed the crawl from the site's sitemap
|
||||
Partir du plan de site (sitemap)
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Lire le plan de site (lignes Sitemap: de robots.txt, puis /sitemap.xml) et ajouter chaque URL listée comme adresse de départ.
|
||||
Sitemap address:
|
||||
Adresse du plan de site :
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adresse d'un plan de site à lire au lieu de sonder le site ; laissez vide pour sonder robots.txt puis /sitemap.xml.
|
||||
|
||||
@@ -970,19 +970,3 @@ Report what changed since the previous mirror
|
||||
Αναφορά των αλλαγών από το προηγούμενο αντίγραφο
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Εγγραφή και του hts-changes.json, που παραθέτει τι είναι νέο, τι άλλαξε, τι έμεινε ίδιο και τι χάθηκε σε σχέση με το προηγούμενο αντίγραφο.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Ενσωμάτωση των πόρων ως URI data: (αυτόνομες σελίδες)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Μόλις ολοκληρωθεί το αντίγραφο, κάθε αποθηκευμένη σελίδα ξαναγράφεται με ενσωματωμένα τα φύλλα στυλ, τα σενάρια, τις εικόνες και τις γραμματοσειρές της, ώστε η σελίδα να μπορεί να ανοίξει και μόνη της.
|
||||
Largest inlined asset (bytes):
|
||||
Μέγιστος ενσωματωμένος πόρος (byte):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Πόρος μεγαλύτερος από αυτό το μέγεθος διατηρεί κανονικό σύνδεσμο. Αφήστε το κενό για την προεπιλογή των 10485760 byte.
|
||||
Seed the crawl from the site's sitemap
|
||||
Έναρξη της ανίχνευσης από τον χάρτη του ιστότοπου
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Ανάγνωση του χάρτη του ιστότοπου (γραμμές Sitemap: στο robots.txt, έπειτα /sitemap.xml) και προσθήκη κάθε διεύθυνσης URL που περιέχει ως αρχικής διεύθυνσης.
|
||||
Sitemap address:
|
||||
Διεύθυνση χάρτη ιστότοπου:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Διεύθυνση χάρτη ιστότοπου προς ανάγνωση αντί για αναζήτηση στον ιστότοπο. Αφήστε το κενό για έλεγχο του robots.txt και έπειτα του /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Segnala che cosa è cambiato dalla copia precedente
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Scrive anche hts-changes.json, che elenca ciò che questa scansione lascia come nuovo, modificato, invariato o scomparso rispetto alla copia precedente.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Incorpora le risorse come URI data: (pagine autonome)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Al termine della copia, riscrive ogni pagina salvata incorporando fogli di stile, script, immagini e caratteri, in modo che una pagina possa essere aperta anche da sola.
|
||||
Largest inlined asset (bytes):
|
||||
Dimensione massima della risorsa incorporata (byte):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Una risorsa oltre questa dimensione mantiene un collegamento normale; lasciare vuoto per il valore predefinito di 10485760 byte.
|
||||
Seed the crawl from the site's sitemap
|
||||
Avvia la scansione dalla mappa del sito
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Legge la mappa del sito (righe Sitemap: in robots.txt, poi /sitemap.xml) e aggiunge come indirizzo iniziale ogni URL elencato.
|
||||
Sitemap address:
|
||||
Indirizzo della mappa del sito:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Indirizzo di una mappa del sito da leggere invece di sondare il sito; lasciare vuoto per sondare robots.txt e poi /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
前回のミラーからの変更点を報告する
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
hts-changes.json も書き出し、前回のミラーと比べて新規、変更、変更なし、消滅となったものを一覧にします。
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
リソースを data: URI として埋め込む (単独で開けるページ)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
ミラーの完了後、保存された各ページを、スタイルシート、スクリプト、画像、フォントを埋め込んだ形で書き直します。ページ単体でも開けるようになります。
|
||||
Largest inlined asset (bytes):
|
||||
埋め込む最大サイズ (バイト):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
このサイズを超えるリソースは通常のリンクのままになります。空欄にすると既定値の 10485760 バイトになります。
|
||||
Seed the crawl from the site's sitemap
|
||||
サイトマップからミラーリングを開始する
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
サイトマップ (robots.txt の Sitemap: 行、次に /sitemap.xml) を読み込み、記載されているすべての URL を開始アドレスとして追加します。
|
||||
Sitemap address:
|
||||
サイトマップのアドレス:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
サイトを探索する代わりに読み込むサイトマップのアドレス。空欄にすると robots.txt、次に /sitemap.xml を探索します。
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
¸×ÒÕáâØ èâÞ Õ ßàÞÜÕÝÕâÞ ÞÔ ßàÕâåÞÔÝÞâÞ ÞÓÛÕÔÐÛÞ
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
·ÐßØèØ Ø hts-changes.json áÞ áߨáÞÚ ÝÐ âÞÐ èâÞ Õ ÝÞÒÞ, ßàÞÜÕÝÕâÞ, ÝÕßàÞÜÕÝÕâÞ ØÛØ ØáçÕ×ÝÐâÞ ÒÞ ÞÔÝÞá ÝÐ ßàÕâåÞÔÝÞâÞ ÞÓÛÕÔÐÛÞ.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
²ÓàÐÔØ ÓØ àÕáãàáØâÕ ÚÐÚÞ data: URI (áÐÜÞáâÞøÝØ áâàÐÝØæØ)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
¿Þ ×ÐÒàèãÒÐúÕ ÝÐ ÞÓÛÕÔÐÛÞâÞ, áÕÚÞøÐ ×ÐçãÒÐÝÐ áâàÐÝØæÐ áÕ ßàÕߨèãÒÐ áÞ ÒÓàÐÔÕÝØ áâØÛáÚØ ÛØáâÞÒØ, áÚàØßâØ, áÛØÚØ Ø äÞÝâÞÒØ, ×Ð ÔÐ ÜÞÖÕ áâàÐÝØæÐâÐ ÔÐ áÕ ÞâÒÞàØ Ø áÐÜÞáâÞøÝÞ.
|
||||
Largest inlined asset (bytes):
|
||||
½ÐøÓÞÛÕÜ ÒÓàÐÔÕÝ àÕáãàá (ÑÐøâØ):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
ÀÕáãàá ßÞÓÞÛÕÜ ÞÔ ÞÒÐÐ ÓÞÛÕÜØÝÐ ×ÐÔàÖãÒÐ ÞÑØçÝÐ ÒàáÚÐ; ÞáâÐÒÕâÕ ßàÐ×ÝÞ ×Ð áâÐÝÔÐàÔÝØâÕ 10485760 ÑÐøâØ.
|
||||
Seed the crawl from the site's sitemap
|
||||
·ÐßÞçÝØ ÓÞ ßàÕÑÐàãÒÐúÕâÞ ÞÔ ÚÐàâÐâÐ ÝÐ áÐøâÞâ
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
¿àÞçØâÐø øÐ ÚÐàâÐâÐ ÝÐ áÐøâÞâ (àÕÔÞÒØâÕ Sitemap: ÒÞ robots.txt, ßÞâÞÐ /sitemap.xml) Ø ÔÞÔÐø øÐ áÕÚÞøÐ ÝÐÒÕÔÕÝÐ URL ÐÔàÕáÐ ÚÐÚÞ ßÞçÕâÝÐ.
|
||||
Sitemap address:
|
||||
°ÔàÕáÐ ÝÐ ÚÐàâÐâÐ ÝÐ áÐøâÞâ:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
°ÔàÕáÐ ÝÐ ÚÐàâÐ ÝÐ áÐøâÞâ èâÞ âàÕÑÐ ÔÐ áÕ ßàÞçØâÐ ÝÐÜÕáâÞ ØáߨâãÒÐúÕ ÝÐ áÐøâÞâ; ÞáâÐÒÕâÕ ßàÐ×ÝÞ ×Ð ÔÐ áÕ ØáߨâÐ robots.txt, ßÐ /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Jelentés arról, mi változott az elõzõ tükrözés óta
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
A hts-changes.json fájl írása is, amely felsorolja, mi új, mi változott, mi maradt változatlan és mi tûnt el az elõzõ tükrözéshez képest.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Erõforrások beágyazása data: URI-ként (önálló oldalak)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
A tükrözés befejezése után minden mentett oldal újraírása a stíluslapok, parancsfájlok, képek és betûtípusok beágyazásával, így az oldal önmagában is megnyitható.
|
||||
Largest inlined asset (bytes):
|
||||
Legnagyobb beágyazott erõforrás (bájt):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Az ennél nagyobb erõforrás közönséges hivatkozás marad; hagyja üresen a 10485760 bájtos alapértelmezéshez.
|
||||
Seed the crawl from the site's sitemap
|
||||
A letöltés indítása a webhely webhelytérképérõl
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
A webhely webhelytérképének beolvasása (a robots.txt Sitemap: sorai, majd a /sitemap.xml), és a benne felsorolt összes URL felvétele kiindulási címként.
|
||||
Sitemap address:
|
||||
Webhelytérkép címe:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
A webhely vizsgálata helyett beolvasandó webhelytérkép címe; hagyja üresen a robots.txt, majd a /sitemap.xml vizsgálatához.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Rapporteren wat er sinds de vorige spiegeling is gewijzigd
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Ook hts-changes.json schrijven met een overzicht van wat deze doorloop nieuw, gewijzigd, ongewijzigd of verdwenen laat ten opzichte van de vorige spiegeling.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Bronnen insluiten als data:-URI's (op zichzelf staande pagina's)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Zodra de spiegeling klaar is, elke opgeslagen pagina herschrijven met ingesloten stijlbladen, scripts, afbeeldingen en lettertypen, zodat een pagina ook los geopend kan worden.
|
||||
Largest inlined asset (bytes):
|
||||
Grootste ingesloten bron (bytes):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Een bron boven deze grootte houdt een gewone koppeling; laat leeg voor de standaardwaarde van 10485760 bytes.
|
||||
Seed the crawl from the site's sitemap
|
||||
De crawl starten vanaf de sitemap van de site
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
De sitemap van de site lezen (Sitemap:-regels in robots.txt, daarna /sitemap.xml) en elke vermelde URL als startadres toevoegen.
|
||||
Sitemap address:
|
||||
Sitemap-adres:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adres van een sitemap die gelezen moet worden in plaats van de site te onderzoeken; laat leeg om robots.txt en daarna /sitemap.xml te controleren.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Rapporter hva som er endret siden forrige speiling
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Skriv også hts-changes.json som lister opp hva denne gjennomgangen etterlater som nytt, endret, uendret eller forsvunnet sammenlignet med forrige speiling.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Bygg inn ressurser som data:-URI-er (selvstendige sider)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Når speilingen er ferdig, skrives hver lagrede side om med stilark, skript, bilder og skrifter innebygd, slik at en side også kan åpnes alene.
|
||||
Largest inlined asset (bytes):
|
||||
Største innebygde ressurs (byte):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
En ressurs over denne størrelsen beholder en vanlig lenke; la feltet stå tomt for standardverdien på 10485760 byte.
|
||||
Seed the crawl from the site's sitemap
|
||||
Start gjennomgangen fra nettstedets nettstedskart
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Les nettstedets nettstedskart (Sitemap:-linjer i robots.txt, deretter /sitemap.xml) og legg til hver oppført URL som startadresse.
|
||||
Sitemap address:
|
||||
Adresse til nettstedskart:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adressen til et nettstedskart som skal leses i stedet for å undersøke nettstedet; la feltet stå tomt for å undersøke robots.txt og deretter /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Zg³o¶, co zmieni³o siê od poprzedniej kopii
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Zapisz równie¿ plik hts-changes.json z list± tego, co w porównaniu z poprzedni± kopi± jest nowe, zmienione, niezmienione lub usuniête.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Osad¼ zasoby jako identyfikatory data: URI (samodzielne strony)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Po zakoñczeniu tworzenia kopii przepisz ka¿d± zapisan± stronê z osadzonymi arkuszami stylów, skryptami, obrazami i czcionkami, aby stronê mo¿na by³o otworzyæ równie¿ samodzielnie.
|
||||
Largest inlined asset (bytes):
|
||||
Najwiêkszy osadzony zasób (bajty):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Zasób wiêkszy ni¿ ten rozmiar zachowuje zwyk³y odno¶nik; pozostaw puste, aby u¿yæ domy¶lnych 10485760 bajtów.
|
||||
Seed the crawl from the site's sitemap
|
||||
Rozpocznij pobieranie od mapy witryny
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Odczytaj mapê witryny (wiersze Sitemap: w pliku robots.txt, nastêpnie /sitemap.xml) i dodaj ka¿dy wymieniony adres URL jako adres pocz±tkowy.
|
||||
Sitemap address:
|
||||
Adres mapy witryny:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adres mapy witryny do odczytania zamiast sondowania witryny; pozostaw puste, aby sprawdziæ robots.txt, a nastêpnie /sitemap.xml.
|
||||
|
||||
@@ -1016,19 +1016,3 @@ Report what changed since the previous mirror
|
||||
Relatar o que mudou desde o espelhamento anterior
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Gravar também hts-changes.json listando o que esta captura deixa como novo, alterado, inalterado ou removido em relação ao espelhamento anterior.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Incorporar os recursos como URIs data: (páginas autônomas)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Ao terminar o espelhamento, reescrever cada página salva com suas folhas de estilo, scripts, imagens e fontes incorporadas, para que a página também possa ser aberta sozinha.
|
||||
Largest inlined asset (bytes):
|
||||
Maior recurso incorporado (bytes):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Um recurso acima desse tamanho mantém um link comum; deixe em branco para o padrão de 10485760 bytes.
|
||||
Seed the crawl from the site's sitemap
|
||||
Iniciar a captura pelo mapa do site
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Ler o mapa do site (linhas Sitemap: do robots.txt, depois /sitemap.xml) e adicionar como endereço inicial cada URL nele listada.
|
||||
Sitemap address:
|
||||
Endereço do mapa do site:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Endereço de um mapa do site a ser lido em vez de sondar o site; deixe em branco para sondar robots.txt e depois /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Comunicar o que mudou desde o espelho anterior
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Escrever também hts-changes.json, que lista o que esta recolha deixa como novo, alterado, inalterado ou desaparecido em relação ao espelho anterior.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Incorporar os recursos como URI data: (páginas autónomas)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Depois de terminado o espelho, reescrever cada página guardada com as suas folhas de estilo, scripts, imagens e tipos de letra incorporados, para que a página também possa ser aberta sozinha.
|
||||
Largest inlined asset (bytes):
|
||||
Maior recurso incorporado (bytes):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Um recurso acima deste tamanho mantém uma ligação normal; deixe em branco para o valor predefinido de 10485760 bytes.
|
||||
Seed the crawl from the site's sitemap
|
||||
Iniciar a recolha pelo mapa do site
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Ler o mapa do site (linhas Sitemap: do robots.txt, depois /sitemap.xml) e adicionar como endereço inicial cada URL nele listado.
|
||||
Sitemap address:
|
||||
Endereço do mapa do site:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Endereço de um mapa do site a ler em vez de sondar o site; deixe em branco para sondar robots.txt e depois /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Raporteaza ce s-a schimbat fata de copia anterioara
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Scrie si hts-changes.json, care listeaza ce este nou, modificat, nemodificat sau disparut fata de copia anterioara.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Încorporeaza resursele ca URI data: (pagini de sine statatoare)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Dupa terminarea copiei, fiecare pagina salvata este rescrisa cu foile de stil, scripturile, imaginile si fonturile încorporate, astfel încât pagina sa poata fi deschisa si singura.
|
||||
Largest inlined asset (bytes):
|
||||
Cea mai mare resursa încorporata (octeti):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
O resursa mai mare decât aceasta dimensiune pastreaza o legatura obisnuita; lasati gol pentru valoarea implicita de 10485760 de octeti.
|
||||
Seed the crawl from the site's sitemap
|
||||
Porneste explorarea de la harta sitului
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Citeste harta sitului (liniile Sitemap: din robots.txt, apoi /sitemap.xml) si adauga fiecare URL listat ca adresa de pornire.
|
||||
Sitemap address:
|
||||
Adresa hartii sitului:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adresa unei harti a sitului care sa fie citita în loc de sondarea sitului; lasati gol pentru a sonda robots.txt, apoi /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Ñîîáùàòü, ÷òî èçìåíèëîñü ñ ïðåäûäóùåãî çåðêàëà
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Òàêæå çàïèñûâàòü hts-changes.json ñî ñïèñêîì òîãî, ÷òî ïî ñðàâíåíèþ ñ ïðåäûäóùèì çåðêàëîì ñòàëî íîâûì, èçìåí¸ííûì, íåèçìåí¸ííûì èëè èñ÷åçëî.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Âñòðàèâàòü ðåñóðñû êàê data: URI (ñàìîñòîÿòåëüíûå ñòðàíèöû)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Ïîñëå çàâåðøåíèÿ çåðêàëèðîâàíèÿ êàæäàÿ ñîõðàí¸ííàÿ ñòðàíèöà ïåðåçàïèñûâàåòñÿ ñî âñòðîåííûìè òàáëèöàìè ñòèëåé, ñöåíàðèÿìè, èçîáðàæåíèÿìè è øðèôòàìè, ÷òîáû ñòðàíèöó ìîæíî áûëî îòêðûòü îòäåëüíî.
|
||||
Largest inlined asset (bytes):
|
||||
Íàèáîëüøèé âñòðàèâàåìûé ðåñóðñ (áàéòû):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Ðåñóðñ áîëüøå ýòîãî ðàçìåðà ñîõðàíÿåò îáû÷íóþ ññûëêó; îñòàâüòå ïóñòûì äëÿ çíà÷åíèÿ ïî óìîë÷àíèþ 10485760 áàéò.
|
||||
Seed the crawl from the site's sitemap
|
||||
Íà÷èíàòü îáõîä ñ êàðòû ñàéòà
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Ïðî÷èòàòü êàðòó ñàéòà (ñòðîêè Sitemap: â robots.txt, çàòåì /sitemap.xml) è äîáàâèòü êàæäûé óêàçàííûé â íåé URL êàê íà÷àëüíûé àäðåñ.
|
||||
Sitemap address:
|
||||
Àäðåñ êàðòû ñàéòà:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Àäðåñ êàðòû ñàéòà, êîòîðóþ íóæíî ïðî÷èòàòü âìåñòî îïðîñà ñàéòà; îñòàâüòå ïóñòûì, ÷òîáû ïðîâåðèòü robots.txt, çàòåì /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Oznámi», èo sa od predchádzajúceho zrkadlenia zmenilo
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Zapísa» aj hts-changes.json so zoznamom toho, èo je oproti predchádzajúcemu zrkadleniu nové, zmenené, nezmenené alebo chýbajúce.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Vlo¾i» zdroje ako URI data: (samostatné stránky)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Po dokonèení zrkadlenia prepísa» ka¾dú ulo¾enú stránku s vlo¾enými ¹týlmi, skriptmi, obrázkami a písmami, aby sa stránka dala otvori» aj samostatne.
|
||||
Largest inlined asset (bytes):
|
||||
Najväè¹í vlo¾ený zdroj (bajty):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Zdroj väè¹í ne¾ táto veµkos» si ponechá be¾ný odkaz; ponechajte prázdne pre predvolených 10485760 bajtov.
|
||||
Seed the crawl from the site's sitemap
|
||||
Zaèa» prehliadanie z mapy stránok
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Naèíta» mapu stránok (riadky Sitemap: v súbore robots.txt, potom /sitemap.xml) a prida» ka¾dú uvedenú adresu URL ako poèiatoènú.
|
||||
Sitemap address:
|
||||
Adresa mapy stránok:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adresa mapy stránok, ktorá sa má naèíta» namiesto zis»ovania na stránke; ponechajte prázdne na zistenie z robots.txt a potom /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Porocaj, kaj se je spremenilo od prejsnjega zrcaljenja
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Zapisi tudi hts-changes.json s seznamom tega, kar je v primerjavi s prejsnjim zrcaljenjem novo, spremenjeno, nespremenjeno ali izginilo.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Vgradi vire kot data: URI (samostojne strani)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Ko je zrcaljenje koncano, se vsaka shranjena stran prepise z vgrajenimi slogovnimi listi, skripti, slikami in pisavami, tako da jo je mogoce odpreti tudi samostojno.
|
||||
Largest inlined asset (bytes):
|
||||
Najvecji vgrajeni vir (bajti):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Vir, vecji od te velikosti, ohrani obicajno povezavo; pustite prazno za privzetih 10485760 bajtov.
|
||||
Seed the crawl from the site's sitemap
|
||||
Zacni zajem z zemljevidom spletnega mesta
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Preberi zemljevid spletnega mesta (vrstice Sitemap: v robots.txt, nato /sitemap.xml) in dodaj vsak navedeni URL kot zacetni naslov.
|
||||
Sitemap address:
|
||||
Naslov zemljevida spletnega mesta:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Naslov zemljevida spletnega mesta, ki naj se prebere namesto preverjanja mesta; pustite prazno za preverjanje robots.txt in nato /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Rapportera vad som ändrats sedan föregående spegling
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Skriv även hts-changes.json som listar vad den här genomgången lämnar som nytt, ändrat, oförändrat eller försvunnet jämfört med föregående spegling.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Bädda in resurser som data:-URI:er (fristående sidor)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
När speglingen är klar skrivs varje sparad sida om med sina formatmallar, skript, bilder och teckensnitt inbäddade, så att en sida också kan öppnas för sig.
|
||||
Largest inlined asset (bytes):
|
||||
Största inbäddade resurs (byte):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
En resurs över den här storleken behåller en vanlig länk; lämna tomt för standardvärdet 10485760 byte.
|
||||
Seed the crawl from the site's sitemap
|
||||
Starta insamlingen från webbplatsens webbplatskarta
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Läs webbplatsens webbplatskarta (Sitemap:-rader i robots.txt, sedan /sitemap.xml) och lägg till varje angiven URL som startadress.
|
||||
Sitemap address:
|
||||
Webbplatskartans adress:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Adress till en webbplatskarta som ska läsas i stället för att söka på webbplatsen; lämna tomt för att kontrollera robots.txt och sedan /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Önceki yansýmadan bu yana deðiþenleri bildir
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Ayrýca hts-changes.json yazarak bu taramanýn önceki yansýmaya göre neyi yeni, deðiþmiþ, deðiþmemiþ veya kaybolmuþ býraktýðýný listele.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Kaynaklarý data: URI olarak göm (kendi kendine yeten sayfalar)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Yansýlama bittiðinde, kaydedilen her sayfa stil sayfalarý, betikleri, görselleri ve yazý tipleri gömülü olarak yeniden yazýlýr; böylece sayfa tek baþýna da açýlabilir.
|
||||
Largest inlined asset (bytes):
|
||||
En büyük gömülü kaynak (bayt):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Bu boyutun üzerindeki bir kaynak sýradan baðlantýsýný korur; 10485760 baytlýk varsayýlan için boþ býrakýn.
|
||||
Seed the crawl from the site's sitemap
|
||||
Taramayý sitenin site haritasýndan baþlat
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Sitenin site haritasýný oku (robots.txt içindeki Sitemap: satýrlarý, ardýndan /sitemap.xml) ve listelenen her URL'yi baþlangýç adresi olarak ekle.
|
||||
Sitemap address:
|
||||
Site haritasý adresi:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Siteyi yoklamak yerine okunacak site haritasýnýn adresi; robots.txt ve ardýndan /sitemap.xml yoklamasý için boþ býrakýn.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Ïîâ³äîìëÿòè, ùî çì³íèëîñÿ ç ïîïåðåäíüîãî äçåðêàëà
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
Òàêîæ çàïèñóâàòè hts-changes.json ç³ ñïèñêîì òîãî, ùî ïîð³âíÿíî ç ïîïåðåäí³ì äçåðêàëîì º íîâèì, çì³íåíèì, íåçì³íåíèì àáî çíèêëèì.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Âáóäîâóâàòè ðåñóðñè ÿê data: URI (ñàìîñò³éí³ ñòîð³íêè)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
ϳñëÿ çàâåðøåííÿ äçåðêàëþâàííÿ êîæíà çáåðåæåíà ñòîð³íêà ïåðåçàïèñóºòüñÿ ç âáóäîâàíèìè òàáëèöÿìè ñòèë³â, ñöåíàð³ÿìè, çîáðàæåííÿìè òà øðèôòàìè, ùîá ñòîð³íêó ìîæíà áóëî â³äêðèòè îêðåìî.
|
||||
Largest inlined asset (bytes):
|
||||
Íàéá³ëüøèé âáóäîâàíèé ðåñóðñ (áàéòè):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Ðåñóðñ, á³ëüøèé çà öåé ðîçì³ð, çáåð³ãຠçâè÷àéíå ïîñèëàííÿ; çàëèøòå ïîðîæí³ì äëÿ òèïîâîãî çíà÷åííÿ 10485760 áàéò³â.
|
||||
Seed the crawl from the site's sitemap
|
||||
Ïî÷èíàòè îáõ³ä ç êàðòè ñàéòó
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Ïðî÷èòàòè êàðòó ñàéòó (ðÿäêè Sitemap: ó robots.txt, ïîò³ì /sitemap.xml) ³ äîäàòè êîæíó âêàçàíó â í³é URL-àäðåñó ÿê ïî÷àòêîâó.
|
||||
Sitemap address:
|
||||
Àäðåñà êàðòè ñàéòó:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Àäðåñà êàðòè ñàéòó, ÿêó ñë³ä ïðî÷èòàòè çàì³ñòü îïèòóâàííÿ ñàéòó; çàëèøòå ïîðîæí³ì, ùîá ïåðåâ³ðèòè robots.txt, à ïîò³ì /sitemap.xml.
|
||||
|
||||
@@ -968,19 +968,3 @@ Report what changed since the previous mirror
|
||||
Oldingi nusxadan beri nima o’zgarganini xabar qilish
|
||||
Also write hts-changes.json listing what this crawl left new, changed, unchanged and gone compared to the previous mirror.
|
||||
hts-changes.json ham yoziladi: unda ushbu yig’ish oldingi nusxaga nisbatan nimani yangi, o’zgargan, o’zgarmagan yoki yo’qolgan holda qoldirgani ro’yxati bo’ladi.
|
||||
Inline assets as data: URIs (self-contained pages)
|
||||
Resurslarni data: URI sifatida joylash (mustaqil sahifalar)
|
||||
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
|
||||
Nusxalash tugagach, har bir saqlangan sahifa uslublar jadvallari, skriptlar, rasmlar va shriftlar joylangan holda qayta yoziladi, shunda sahifani alohida ham ochish mumkin.
|
||||
Largest inlined asset (bytes):
|
||||
Eng katta joylangan resurs (bayt):
|
||||
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
|
||||
Bu o’lchamdan katta resurs oddiy havolani saqlab qoladi; standart 10485760 bayt uchun bo’sh qoldiring.
|
||||
Seed the crawl from the site's sitemap
|
||||
Yig’ishni saytning sayt xaritasidan boshlash
|
||||
Read the site's sitemap (robots.txt Sitemap: lines, then /sitemap.xml) and add every URL it lists as a start URL.
|
||||
Saytning sayt xaritasini o’qish (robots.txt dagi Sitemap: qatorlari, so’ngra /sitemap.xml) va unda ko’rsatilgan har bir URL manzilni boshlang’ich manzil sifatida qo’shish.
|
||||
Sitemap address:
|
||||
Sayt xaritasi manzili:
|
||||
Address of a sitemap to read instead of probing the site; leave blank to probe robots.txt then /sitemap.xml.
|
||||
Saytni tekshirish o’rniga o’qiladigan sayt xaritasi manzili; robots.txt, so’ngra /sitemap.xml ni tekshirish uchun bo’sh qoldiring.
|
||||
|
||||
103
m4/check_zlib.m4
103
m4/check_zlib.m4
@@ -1,33 +1,82 @@
|
||||
dnl @synopsis CHECK_ZLIB()
|
||||
dnl
|
||||
dnl Look for zlib. It is a hard requirement, not an option: the cache and the
|
||||
dnl WARC output are zip/gzip containers, and the bundled minizip calls zlib
|
||||
dnl directly. --with-zlib=DIR points at a non-standard prefix.
|
||||
dnl This macro searches for an installed zlib library. If nothing
|
||||
dnl was specified when calling configure, it searches first in /usr/local
|
||||
dnl and then in /usr. If the --with-zlib=DIR is specified, it will try
|
||||
dnl to find it in DIR/include/zlib.h and DIR/lib/libz.a. If --without-zlib
|
||||
dnl is specified, the library is not searched at all.
|
||||
dnl
|
||||
dnl If either the header file (zlib.h) or the library (libz) is not
|
||||
dnl found, the configuration exits on error, asking for a valid
|
||||
dnl zlib installation directory or --without-zlib.
|
||||
dnl
|
||||
dnl The macro defines the symbol HAVE_LIBZ if the library is found. You should
|
||||
dnl use autoheader to include a definition for this symbol in a config.h
|
||||
dnl file. Sample usage in a C/C++ source is as follows:
|
||||
dnl
|
||||
dnl #ifdef HAVE_LIBZ
|
||||
dnl #include <zlib.h>
|
||||
dnl #endif /* HAVE_LIBZ */
|
||||
dnl
|
||||
dnl @version $Id$
|
||||
dnl @author Loic Dachary <loic@senga.org>
|
||||
dnl
|
||||
dnl Adds -lz to LIBS and defines HAVE_LIBZ.
|
||||
|
||||
AC_DEFUN([CHECK_ZLIB], [
|
||||
AC_ARG_WITH([zlib],
|
||||
[AS_HELP_STRING([--with-zlib=DIR],[root directory of the zlib installation])],
|
||||
[zlib_want=$withval], [zlib_want=yes])
|
||||
if test "$zlib_want" = "no"; then
|
||||
AC_MSG_ERROR([zlib cannot be disabled: the cache and the WARC output are zip/gzip containers, and the bundled minizip calls zlib directly])
|
||||
AC_DEFUN([CHECK_ZLIB],
|
||||
#
|
||||
# Handle user hints
|
||||
#
|
||||
[AC_MSG_CHECKING(if zlib is wanted)
|
||||
AC_ARG_WITH(zlib,
|
||||
[ --with-zlib=DIR root directory path of zlib installation [defaults to
|
||||
/usr/local or /usr if not found in /usr/local]
|
||||
--without-zlib to disable zlib usage completely],
|
||||
[if test "$withval" != no ; then
|
||||
AC_MSG_RESULT(yes)
|
||||
ZLIB_HOME="$withval"
|
||||
else
|
||||
AC_MSG_RESULT(no)
|
||||
fi], [
|
||||
AC_MSG_RESULT(yes)
|
||||
ZLIB_HOME=/usr/local
|
||||
if test ! -f "${ZLIB_HOME}/include/zlib.h"
|
||||
then
|
||||
ZLIB_HOME=/usr
|
||||
fi
|
||||
if test "$zlib_want" != "yes"; then
|
||||
# An explicit prefix is authoritative: if the header is not under it,
|
||||
# error rather than silently pick a system copy.
|
||||
if test ! -f "$zlib_want/include/zlib.h"; then
|
||||
AC_MSG_ERROR([zlib requested at $zlib_want but $zlib_want/include/zlib.h is missing])
|
||||
fi
|
||||
CPPFLAGS="$CPPFLAGS -I$zlib_want/include"
|
||||
LDFLAGS="$LDFLAGS -L$zlib_want/lib"
|
||||
elif test -f /usr/local/include/zlib.h; then
|
||||
# Where the BSD ports tree lands zlib, and not always searched by default.
|
||||
CPPFLAGS="$CPPFLAGS -I/usr/local/include"
|
||||
LDFLAGS="$LDFLAGS -L/usr/local/lib"
|
||||
fi
|
||||
AC_CHECK_HEADER([zlib.h], [],
|
||||
[AC_MSG_ERROR([zlib.h not found; install the zlib development files or pass --with-zlib=DIR])])
|
||||
AC_CHECK_LIB([z], [inflateEnd], [],
|
||||
[AC_MSG_ERROR([libz not found; install the zlib development files or pass --with-zlib=DIR])])
|
||||
])
|
||||
|
||||
#
|
||||
# Locate zlib, if wanted
|
||||
#
|
||||
if test -n "${ZLIB_HOME}"
|
||||
then
|
||||
ZLIB_OLD_LDFLAGS=$LDFLAGS
|
||||
ZLIB_OLD_CPPFLAGS=$LDFLAGS
|
||||
LDFLAGS="$LDFLAGS -L${ZLIB_HOME}/lib"
|
||||
CPPFLAGS="$CPPFLAGS -I${ZLIB_HOME}/include"
|
||||
AC_LANG_SAVE
|
||||
AC_LANG_C
|
||||
AC_CHECK_LIB(z, inflateEnd, [zlib_cv_libz=yes], [zlib_cv_libz=no])
|
||||
AC_CHECK_HEADER(zlib.h, [zlib_cv_zlib_h=yes], [zlib_cv_zlib_h=no])
|
||||
AC_LANG_RESTORE
|
||||
if test "$zlib_cv_libz" = "yes" -a "$zlib_cv_zlib_h" = "yes"
|
||||
then
|
||||
#
|
||||
# If both library and header were found, use them
|
||||
#
|
||||
AC_CHECK_LIB(z, inflateEnd)
|
||||
AC_MSG_CHECKING(zlib in ${ZLIB_HOME})
|
||||
AC_MSG_RESULT(ok)
|
||||
else
|
||||
#
|
||||
# If either header or library was not found, revert and bomb
|
||||
#
|
||||
AC_MSG_CHECKING(zlib in ${ZLIB_HOME})
|
||||
LDFLAGS="$ZLIB_OLD_LDFLAGS"
|
||||
CPPFLAGS="$ZLIB_OLD_CPPFLAGS"
|
||||
AC_MSG_RESULT(failed)
|
||||
AC_MSG_ERROR(either specify a valid zlib installation with --with-zlib=DIR or disable zlib usage with --without-zlib)
|
||||
fi
|
||||
fi
|
||||
|
||||
])
|
||||
|
||||
@@ -36,12 +36,10 @@ httrack \- offline browser : copy websites to a local directory
|
||||
[ \fB\-t, \-\-test\fR ]
|
||||
[ \fB\-%L, \-\-list\fR ]
|
||||
[ \fB\-%S, \-\-urllist\fR ]
|
||||
[ \fB\-%m, \-\-sitemap\fR ]
|
||||
[ \fB\-NN, \-\-structure[=N]\fR ]
|
||||
[ \fB\-%N, \-\-delayed\-type\-check\fR ]
|
||||
[ \fB\-%D, \-\-cached\-delayed\-type\-check\fR ]
|
||||
[ \fB\-%M, \-\-mime\-html\fR ]
|
||||
[ \fB\-%Z, \-\-single\-file\fR ]
|
||||
[ \fB\-LN, \-\-long\-names[=N]\fR ]
|
||||
[ \fB\-KN, \-\-keep\-links[=N]\fR ]
|
||||
[ \fB\-x, \-\-replace\-external\fR ]
|
||||
@@ -190,8 +188,6 @@ test all URLs (even forbidden ones) (\-\-test)
|
||||
<file> add all URL located in this text file (one URL per line) (\-\-list <param>)
|
||||
.IP \-%S
|
||||
<file> add all scan rules located in this text file (one scan rule per line) (\-\-urllist <param>)
|
||||
.IP \-%m
|
||||
seed the crawl from the site's sitemap (robots.txt Sitemap:, then /sitemap.xml); \-\-sitemap\-url URL names one explicitly. A sitemap you name, or one the site declares, is fetched even under robots.txt Disallow; only the guessed /sitemap.xml obeys it. The URLs found still pass every filter and scope rule (\-\-sitemap)
|
||||
.SS Build options:
|
||||
.IP \-NN
|
||||
structure type (0 *original structure, 1+: see below) (\-\-structure[=N])
|
||||
@@ -203,8 +199,6 @@ delayed type check, don't make any link test but wait for files download to star
|
||||
cached delayed type check, don't wait for remote type during updates, to speedup them (%D0 wait, * %D1 don't wait) (\-\-cached\-delayed\-type\-check)
|
||||
.IP \-%M
|
||||
generate a RFC MIME\-encapsulated full\-archive (.mht) (\-\-mime\-html)
|
||||
.IP \-%Z
|
||||
after the mirror, rewrite each saved page with its stylesheets, scripts, images and fonts inlined as data: URIs, so any page opens by double\-click anywhere (links between pages stay relative; audio and video stay links); \-\-single\-file\-max\-size N caps each asset (default 10485760 bytes). %M is the better container where a Chromium\-family browser is a given: one archive, no base64 tax on text, a shared asset stored once (\-\-single\-file)
|
||||
.IP \-%t
|
||||
keep the original file extension, don't rewrite it from the MIME type (%t0 rewrite)
|
||||
.IP \-LN
|
||||
|
||||
@@ -66,7 +66,7 @@ libhttrack_la_SOURCES = htscore.c htsparse.c htsback.c htscache.c \
|
||||
htscmdline.c htshelp.c htslib.c htsurlport.c htscoremain.c \
|
||||
htsname.c htsrobots.c htstools.c htswizard.c \
|
||||
htsalias.c htsthread.c htsindex.c htsbauth.c \
|
||||
htsmd5.c htscodec.c htswarc.c htschanges.c htssinglefile.c htssitemap.c htsproxy.c htszlib.c htswrap.c htsconcat.c \
|
||||
htsmd5.c htscodec.c htswarc.c htschanges.c htsproxy.c htszlib.c htswrap.c htsconcat.c \
|
||||
htsmodules.c htscharset.c punycode.c htsencoding.c htssniff.c \
|
||||
md5.c \
|
||||
minizip/ioapi.c minizip/mztools.c minizip/unzip.c minizip/zip.c \
|
||||
@@ -77,7 +77,7 @@ libhttrack_la_SOURCES = htscore.c htsparse.c htsback.c htscache.c \
|
||||
htshelp.h htsindex.h htslib.h htsurlport.h htsmd5.h \
|
||||
htsmodules.h htsname.h htsnet.h htssniff.h \
|
||||
htsopt.h htsrobots.h htsthread.h \
|
||||
htstools.h htswizard.h htswrap.h htscodec.h htswarc.h htschanges.h htssinglefile.h htssitemap.h htsproxy.h htszlib.h \
|
||||
htstools.h htswizard.h htswrap.h htscodec.h htswarc.h htschanges.h htsproxy.h htszlib.h \
|
||||
htsstrings.h htsarrays.h httrack-library.h \
|
||||
htscharset.h punycode.h htsencoding.h \
|
||||
htsentities.h htsentities.sh htsbasiccharsets.sh htscodepages.h \
|
||||
|
||||
@@ -116,10 +116,6 @@ const char *hts_optalias[][4] = {
|
||||
"load extra cookies from a Netscape cookies.txt"},
|
||||
{"changes", "-%d", "single",
|
||||
"write hts-changes.json: what this crawl changed vs. the previous mirror"},
|
||||
{"sitemap", "-%m", "single",
|
||||
"seed the crawl from the start host's sitemap (robots.txt, then "
|
||||
"/sitemap.xml)"},
|
||||
{"sitemap-url", "-%mu", "param1", "seed the crawl from this sitemap URL"},
|
||||
{"warc", "-%r", "single", "write an ISO-28500 WARC/1.1 archive of the crawl"},
|
||||
{"warc-file", "-%rf", "param1", "write a WARC archive to the given base name"},
|
||||
{"warc-max-size", "-%rs", "param1",
|
||||
@@ -129,10 +125,6 @@ const char *hts_optalias[][4] = {
|
||||
{"warc-cdxj", "-%rc", "single", ""},
|
||||
{"wacz", "-%rz", "single",
|
||||
"package the WARC archive, CDXJ index and pages as a WACZ file"},
|
||||
{"single-file", "-%Z", "single",
|
||||
"after the mirror, inline each page's assets as data: URIs"},
|
||||
{"single-file-max-size", "-%Zs", "param1",
|
||||
"per-asset cap for --single-file, in bytes (implies it; default 10485760)"},
|
||||
{"why", "-%Y", "param1",
|
||||
"explain which filter rule accepts or rejects a URL, then exit"},
|
||||
{"pause", "-%G", "param1",
|
||||
|
||||
@@ -48,6 +48,11 @@ Please visit our Website: http://www.httrack.com
|
||||
#include "htsftp.h"
|
||||
#include "htscodec.h"
|
||||
#include "htsproxy.h"
|
||||
#if HTS_USEZLIB
|
||||
#include "htszlib.h"
|
||||
#else
|
||||
#error HTS_USEZLIB not defined
|
||||
#endif
|
||||
|
||||
#ifdef _WIN32
|
||||
#ifndef __cplusplus
|
||||
@@ -585,17 +590,12 @@ static int create_back_tmpfile(httrackp *opt, lien_back *const back,
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Did the fetch fail to produce a response, as opposed to the engine
|
||||
deliberately passing the resource over? Only the latter may be purged. */
|
||||
static hts_boolean back_transfer_failed(const int statuscode) {
|
||||
switch (statuscode) {
|
||||
case STATUSCODE_TOO_BIG:
|
||||
case STATUSCODE_EXCLUDED:
|
||||
case STATUSCODE_TEST_OK:
|
||||
return HTS_FALSE;
|
||||
default:
|
||||
return statuscode <= 0 ? HTS_TRUE : HTS_FALSE;
|
||||
}
|
||||
/* Move src onto dst; RENAME does not clobber an existing target on Windows. */
|
||||
static hts_boolean replace_file(const char *src, const char *dst) {
|
||||
if (RENAME(src, dst) == 0)
|
||||
return HTS_TRUE;
|
||||
(void) UNLINK(dst);
|
||||
return RENAME(src, dst) == 0 ? HTS_TRUE : HTS_FALSE;
|
||||
}
|
||||
|
||||
/* Commit or restore a re-fetch backup (#77 follow-up): a re-fetch over an
|
||||
@@ -615,7 +615,7 @@ static void back_finalize_backup(httrackp *opt, lien_back *const back,
|
||||
}
|
||||
/* On failure keep the backup: an orphaned temp beats losing the good copy.
|
||||
*/
|
||||
if (!hts_rename_over(back->tmpfile, back->url_sav))
|
||||
if (!replace_file(back->tmpfile, back->url_sav))
|
||||
hts_log_print(opt, LOG_WARNING | LOG_ERRNO,
|
||||
"could not restore %s; previous copy kept as %s",
|
||||
back->url_sav, back->tmpfile);
|
||||
@@ -746,7 +746,7 @@ int back_finalize(httrackp * opt, cache_back * cache, struct_back * sback,
|
||||
"Read error when decompressing");
|
||||
}
|
||||
UNLINK(unpacked);
|
||||
} else if (hts_rename_over(unpacked, back[p].url_sav)) {
|
||||
} else if (replace_file(unpacked, back[p].url_sav)) {
|
||||
/* The temp bypassed filecreate(), which is what chmods. */
|
||||
#ifndef _WIN32
|
||||
chmod(back[p].url_sav, HTS_ACCESS_FILE);
|
||||
@@ -1044,14 +1044,6 @@ int back_finalize(httrackp * opt, cache_back * cache, struct_back * sback,
|
||||
/* Aborted, error, or not ready: url_sav (if written) is broken; restore the
|
||||
previous copy from the backup. */
|
||||
back_finalize_backup(opt, &back[p], HTS_FALSE);
|
||||
/* Note the surviving copy, or the end-of-update purge drops what this run
|
||||
never managed to replace (#746). */
|
||||
if (!back[p].testmode && back_transfer_failed(back[p].r.statuscode) &&
|
||||
back[p].url_sav[0] != '\0' && fexist_utf8(back[p].url_sav)) {
|
||||
filenote(&opt->state.strc, back[p].url_sav, NULL);
|
||||
file_notify(opt, back[p].url_adr, back[p].url_fil, back[p].url_sav, 0, 0,
|
||||
back[p].r.notmodified);
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
|
||||
@@ -1297,17 +1297,12 @@ char *readfile2(const char *fil, LLint * size) {
|
||||
}
|
||||
|
||||
/* Note: utf-8 */
|
||||
char *readfile_utf8(const char *fil) { return readfile2_utf8(fil, NULL); }
|
||||
|
||||
/* Note: utf-8 */
|
||||
char *readfile2_utf8(const char *fil, LLint *size) {
|
||||
char *readfile_utf8(const char *fil) {
|
||||
char *adr = NULL;
|
||||
char catbuff[CATBUFF_SIZE];
|
||||
const LLint len = fsize_utf8(fil);
|
||||
const size_t buflen = len >= 0 ? llint_to_size_t(len) : (size_t) -1;
|
||||
|
||||
if (size != NULL)
|
||||
*size = len;
|
||||
if (buflen != (size_t) -1) { // exists, and is addressable (see readfile2)
|
||||
FILE *const fp = FOPEN(fconv(catbuff, sizeof(catbuff), fil), "rb");
|
||||
|
||||
|
||||
@@ -72,7 +72,7 @@ hts_codec hts_codec_parse(const char *encoding) {
|
||||
return HTS_CODEC_IDENTITY;
|
||||
if (strfield2(encoding, "gzip") || strfield2(encoding, "x-gzip") ||
|
||||
strfield2(encoding, "deflate") || strfield2(encoding, "x-deflate"))
|
||||
return HTS_CODEC_DEFLATE;
|
||||
return HTS_USEZLIB ? HTS_CODEC_DEFLATE : HTS_CODEC_UNSUPPORTED;
|
||||
if (strfield2(encoding, "br"))
|
||||
return HTS_USEBROTLI ? HTS_CODEC_BROTLI : HTS_CODEC_UNSUPPORTED;
|
||||
if (strfield2(encoding, "zstd"))
|
||||
@@ -98,11 +98,16 @@ hts_codec hts_codec_parse(const char *encoding) {
|
||||
const char *hts_acceptencoding(hts_boolean compressible, hts_boolean secure) {
|
||||
if (!compressible)
|
||||
return "identity";
|
||||
#if HTS_USEZLIB
|
||||
/* br and zstd over TLS only, as browsers do: a cleartext intermediary that
|
||||
rewrites a coding it can not read would corrupt the mirror. */
|
||||
if (secure)
|
||||
return "gzip, deflate" HTS_AE_BROTLI HTS_AE_ZSTD ", identity;q=0.9";
|
||||
return "gzip, deflate, identity;q=0.9";
|
||||
#else
|
||||
(void) secure;
|
||||
return "identity";
|
||||
#endif
|
||||
}
|
||||
|
||||
hts_boolean hts_codec_is_archive_ext(hts_codec codec, const char *ext) {
|
||||
@@ -295,7 +300,11 @@ int hts_codec_unpack(hts_codec codec, const char *filename,
|
||||
return -1;
|
||||
switch (codec) {
|
||||
case HTS_CODEC_DEFLATE:
|
||||
#if HTS_USEZLIB
|
||||
return hts_zunpack(filename, newfile);
|
||||
#else
|
||||
return -1;
|
||||
#endif
|
||||
case HTS_CODEC_BROTLI:
|
||||
case HTS_CODEC_ZSTD:
|
||||
break;
|
||||
@@ -339,8 +348,10 @@ size_t hts_codec_head(hts_codec codec, const void *in, size_t in_len, void *out,
|
||||
if (in == NULL || in_len == 0 || out == NULL || out_len == 0)
|
||||
return 0;
|
||||
switch (codec) {
|
||||
#if HTS_USEZLIB
|
||||
case HTS_CODEC_DEFLATE:
|
||||
return hts_zhead(in, in_len, out, out_len);
|
||||
#endif
|
||||
#if HTS_USEBROTLI
|
||||
case HTS_CODEC_BROTLI:
|
||||
return codec_head_brotli(in, in_len, out, out_len);
|
||||
|
||||
106
src/htscore.c
106
src/htscore.c
@@ -39,10 +39,8 @@ Please visit our Website: http://www.httrack.com
|
||||
|
||||
/* File defs */
|
||||
#include "htscore.h"
|
||||
#include "htssitemap.h"
|
||||
#include "htswarc.h"
|
||||
#include "htschanges.h"
|
||||
#include "htssinglefile.h"
|
||||
|
||||
/* specific definitions */
|
||||
#include "htsbase.h"
|
||||
@@ -950,22 +948,6 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
heap_top()->premier = heap_top_index(); // premier lien, objet-père=objet
|
||||
heap_top()->precedent = heap_top_index(); // lien précédent
|
||||
|
||||
/* --sitemap: queue the sitemap probe just after the seeds, so its URLs are
|
||||
injected before the crawl gets far. */
|
||||
hts_sitemap_free(opt); /* an earlier mirror may have left a doc list */
|
||||
if (opt->sitemap || StringNotEmpty(opt->sitemap_url)) {
|
||||
char BIGSTK first[HTS_URLMAXSIZE * 2];
|
||||
const char *const eol = strchr(primary, '\n');
|
||||
const size_t len = eol != NULL ? (size_t) (eol - primary) : 0;
|
||||
|
||||
first[0] = '\0';
|
||||
if (len > 0 && len < sizeof(first)) {
|
||||
memcpy(first, primary, len);
|
||||
first[len] = '\0';
|
||||
}
|
||||
hts_sitemap_seed(opt, first);
|
||||
}
|
||||
|
||||
// Initialiser cache
|
||||
{
|
||||
opt->state._hts_in_html_parsing = 4;
|
||||
@@ -1610,21 +1592,11 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
stre.maketrack_fp = maketrack_fp;
|
||||
|
||||
/* Parse */
|
||||
{
|
||||
const int nlinks = opt->lien_tot;
|
||||
|
||||
if (hts_mirror_check_moved(&str, &stre) != 0) {
|
||||
XH_uninit;
|
||||
return -1;
|
||||
}
|
||||
/* A redirect re-queues the target as a fresh link; without carrying
|
||||
the marking over, a moved sitemap is fetched and then ignored. */
|
||||
if (opt->sitemap_state != NULL && opt->lien_tot > nlinks &&
|
||||
hts_sitemap_pending(opt, urladr(), urlfil())) {
|
||||
hts_sitemap_redirect(opt, urladr(), urlfil(), heap_top()->adr,
|
||||
heap_top()->fil);
|
||||
}
|
||||
if (hts_mirror_check_moved(&str, &stre) != 0) {
|
||||
XH_uninit;
|
||||
return -1;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
} // if !error
|
||||
@@ -1642,29 +1614,6 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
/* Load file and decode if necessary, after redirect check. */
|
||||
LOAD_IN_MEMORY_IF_NECESSARY();
|
||||
|
||||
/* Sitemap document: turn its <loc> URLs into top-level seeds. They go
|
||||
through htsAddLink, so the wizard's filters and scope rules decide, and
|
||||
this link's max depth leaves them the full budget. */
|
||||
if (opt->sitemap_state != NULL &&
|
||||
hts_sitemap_pending(opt, urladr(), urlfil())) {
|
||||
htsmoduleStruct BIGSTK smstr;
|
||||
int smptr = ptr;
|
||||
|
||||
memset(&smstr, 0, sizeof(smstr));
|
||||
smstr.opt = opt;
|
||||
smstr.sback = sback;
|
||||
smstr.cache = &cache;
|
||||
smstr.hashptr = hashptr;
|
||||
smstr.numero_passe = numero_passe;
|
||||
smstr.ptr_ = &smptr; /* scratch: the ingester retargets the wizard */
|
||||
smstr.addLink = htsAddLink;
|
||||
smstr.url_host = urladr();
|
||||
smstr.url_file = urlfil();
|
||||
smstr.mime = r.contenttype;
|
||||
hts_sitemap_ingest(opt, &smstr, urladr(), urlfil(), r.adr,
|
||||
r.adr != NULL && r.size > 0 ? (size_t) r.size : 0);
|
||||
}
|
||||
|
||||
// ------------------------------------------------------
|
||||
// ok, fichier chargé localement
|
||||
// ------------------------------------------------------
|
||||
@@ -1875,9 +1824,6 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
|
||||
if (strnotempty(savename()) == 0) { // pas de chemin de sauvegarde
|
||||
if (strcmp(urlfil(), "/robots.txt") == 0) { // robots.txt
|
||||
char BIGSTK sitemaps[8192];
|
||||
|
||||
sitemaps[0] = '\0';
|
||||
if (r.adr) {
|
||||
char BIGSTK infobuff[8192];
|
||||
#ifdef IGNORE_RESTRICTIVE_ROBOTS
|
||||
@@ -1889,8 +1835,7 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
#endif
|
||||
|
||||
robots_parse(&robots, urladr(), r.adr, r.size, infobuff,
|
||||
sizeof(infobuff), keep_root, sitemaps,
|
||||
sizeof(sitemaps));
|
||||
sizeof(infobuff), keep_root);
|
||||
if (strnotempty(infobuff)) {
|
||||
hts_log_print(opt, LOG_INFO,
|
||||
"Note: robots.txt forbidden links for %s are: %s",
|
||||
@@ -1900,10 +1845,6 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
urladr(), infobuff);
|
||||
}
|
||||
}
|
||||
/* After robots_parse, so the rules this very body carries already
|
||||
gate the sitemap fetch. Runs even on a failed probe, which is
|
||||
what falls back to the well-known location. */
|
||||
hts_sitemap_robots(opt, urladr(), sitemaps);
|
||||
}
|
||||
} else if (r.is_write) { // déja sauvé sur disque
|
||||
/*
|
||||
@@ -1980,9 +1921,10 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
}
|
||||
|
||||
// ATTENTION C'EST ICI QU'ON SAUVE LE FICHIER!!
|
||||
// A failed transfer has no body: r.adr holds debris from the aborted
|
||||
// read, which would destroy the copy being re-fetched (#748).
|
||||
if (r.statuscode > 0 && (r.adr != NULL || r.size == 0)) {
|
||||
// An empty body must not overwrite the file when the transfer failed
|
||||
// (statuscode <= 0, e.g. an -M hard-stop): it would truncate a good
|
||||
// copy to 0 (#77 follow-up).
|
||||
if (r.adr != NULL || (r.size == 0 && r.statuscode > 0)) {
|
||||
file_notify(opt, urladr(), urlfil(), savename(), 1, 1, r.notmodified);
|
||||
if (filesave(opt, r.adr, (int) r.size, savename(), urladr(), urlfil()) !=
|
||||
0) {
|
||||
@@ -2264,10 +2206,6 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
}
|
||||
// fin purge!
|
||||
|
||||
/* --single-file: inline each page's assets now the tree is final, after the
|
||||
purge has deleted whatever this run dropped. */
|
||||
singlefile_process_mirror(opt);
|
||||
|
||||
// Indexation
|
||||
if (opt->kindex)
|
||||
index_finish(StringBuff(opt->path_html), opt->kindex);
|
||||
@@ -2343,7 +2281,6 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
usercommand(opt, 0, NULL, NULL, NULL, NULL);
|
||||
warc_close_opt(opt);
|
||||
hts_changes_close_opt(opt);
|
||||
hts_sitemap_free(opt);
|
||||
|
||||
// désallocation mémoire & buffers
|
||||
XH_uninit;
|
||||
@@ -2692,11 +2629,7 @@ HTSEXT_API int structcheck(const char *path) {
|
||||
if (!S_ISDIR(st.st_mode)) {
|
||||
#if HTS_REMOVE_ANNOYING_INDEX
|
||||
if (S_ISREG(st.st_mode)) { /* Regular file in place ; move it and create directory */
|
||||
/* bounded here, not by the path-length guard far above */
|
||||
if (!sprintfbuff(tmpbuf, "%s.txt", file)) {
|
||||
errno = ENAMETOOLONG;
|
||||
return -1;
|
||||
}
|
||||
sprintf(tmpbuf, "%s.txt", file);
|
||||
if (rename(file, tmpbuf) != 0) { /* Can't rename regular file */
|
||||
return -1;
|
||||
}
|
||||
@@ -2804,11 +2737,7 @@ HTSEXT_API int structcheck_utf8(const char *path) {
|
||||
if (!S_ISDIR(st.st_mode)) {
|
||||
#if HTS_REMOVE_ANNOYING_INDEX
|
||||
if (S_ISREG(st.st_mode)) { /* Regular file in place ; move it and create directory */
|
||||
/* bounded here, not by the path-length guard far above */
|
||||
if (!sprintfbuff(tmpbuf, "%s.txt", file)) {
|
||||
errno = ENAMETOOLONG;
|
||||
return -1;
|
||||
}
|
||||
sprintf(tmpbuf, "%s.txt", file);
|
||||
if (RENAME(file, tmpbuf) != 0) { /* Can't rename regular file */
|
||||
return -1;
|
||||
}
|
||||
@@ -3747,14 +3676,6 @@ HTSEXT_API int copy_htsopt(const httrackp * from, httrackp * to) {
|
||||
to->warc_wacz = from->warc_wacz;
|
||||
to->changes = from->changes;
|
||||
|
||||
to->single_file = from->single_file;
|
||||
if (from->single_file_max_size > 0)
|
||||
to->single_file_max_size = from->single_file_max_size;
|
||||
if (from->sitemap)
|
||||
to->sitemap = from->sitemap;
|
||||
if (StringNotEmpty(from->sitemap_url))
|
||||
StringCopyS(to->sitemap_url, from->sitemap_url);
|
||||
|
||||
if (from->pause_max_ms > 0) {
|
||||
to->pause_min_ms = from->pause_min_ms;
|
||||
to->pause_max_ms = from->pause_max_ms;
|
||||
@@ -3828,12 +3749,11 @@ int htsAddLink(htsmoduleStruct * str, char *link) {
|
||||
strcpybuff(codebase, heap(ptr)->fil);
|
||||
else
|
||||
strcpybuff(codebase, heap(heap(ptr)->precedent)->fil);
|
||||
// empty codebase has no last char; codebase-1 would underflow
|
||||
a = codebase[0] != '\0' ? codebase + strlen(codebase) - 1 : codebase;
|
||||
a = codebase + strlen(codebase) - 1;
|
||||
while((*a) && (*a != '/') && (a > codebase))
|
||||
a--;
|
||||
if (*a == '/')
|
||||
*(a + 1) = '\0'; // cut
|
||||
*(a + 1) = '\0'; // couper
|
||||
} else { // couper http:// éventuel
|
||||
if (strfield(codebase, "http://")) {
|
||||
char BIGSTK tempo[HTS_URLMAXSIZE * 2];
|
||||
|
||||
@@ -415,8 +415,6 @@ char *readfile2(const char *fil, LLint * size);
|
||||
|
||||
char *readfile_utf8(const char *fil);
|
||||
|
||||
char *readfile2_utf8(const char *fil, LLint *size);
|
||||
|
||||
char *readfile_or(const char *fil, const char *defaultdata);
|
||||
|
||||
/* Backing (download-slot) scheduler. Operate on the back[] ring (struct_back).
|
||||
|
||||
@@ -1803,56 +1803,6 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
StringCopy(opt->warc_file, WARC_AUTONAME);
|
||||
}
|
||||
break;
|
||||
case 'Z': // single-file: inline each page's assets as data: URIs
|
||||
if (*(com + 1) == 's') { // --single-file-max-size N
|
||||
com++;
|
||||
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
|
||||
HTS_PANIC_PRINTF(
|
||||
"Option single-file-max-size needs a blank "
|
||||
"space and a size");
|
||||
htsmain_free();
|
||||
return -1;
|
||||
}
|
||||
na++;
|
||||
{ // reject non-numeric/negative/overflow; keep the default
|
||||
char *end;
|
||||
LLint v;
|
||||
|
||||
errno = 0;
|
||||
v = strtoll(argv[na], &end, 10);
|
||||
if (isdigit((unsigned char) argv[na][0]) && *end == '\0' &&
|
||||
errno != ERANGE && v > 0)
|
||||
opt->single_file_max_size = v;
|
||||
}
|
||||
opt->single_file = HTS_TRUE;
|
||||
} else {
|
||||
opt->single_file = HTS_TRUE;
|
||||
if (*(com + 1) == '0') {
|
||||
opt->single_file = HTS_FALSE;
|
||||
com++;
|
||||
}
|
||||
}
|
||||
break;
|
||||
case 'm': // sitemap / sitemap-url: seed the crawl from sitemaps
|
||||
if (*(com + 1) == 'u') { // --sitemap-url URL: explicit sitemap
|
||||
com++;
|
||||
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
|
||||
HTS_PANIC_PRINTF(
|
||||
"Option sitemap-url needs a blank space and a URL");
|
||||
htsmain_free();
|
||||
return -1;
|
||||
}
|
||||
na++;
|
||||
if (strlen(argv[na]) >= HTS_URLMAXSIZE) {
|
||||
HTS_PANIC_PRINTF("Sitemap URL too long");
|
||||
htsmain_free();
|
||||
return -1;
|
||||
}
|
||||
StringCopy(opt->sitemap_url, argv[na]);
|
||||
} else { // --sitemap: robots.txt probe, then /sitemap.xml
|
||||
opt->sitemap = HTS_TRUE;
|
||||
}
|
||||
break;
|
||||
case 'Y': // why: explain the filter verdict for a URL, no crawl
|
||||
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
|
||||
HTS_PANIC_PRINTF("Option why needs a blank space and a URL");
|
||||
|
||||
@@ -138,11 +138,10 @@ Please visit our Website: http://www.httrack.com
|
||||
#define HTS_DOSNAME 0
|
||||
#endif
|
||||
|
||||
// zlib is mandatory: the cache is a zip and minizip calls it regardless
|
||||
// utiliser zlib?
|
||||
#ifndef HTS_USEZLIB
|
||||
// autoload
|
||||
#define HTS_USEZLIB 1
|
||||
#elif !HTS_USEZLIB
|
||||
#error HTS_USEZLIB=0 is not a supported configuration
|
||||
#endif
|
||||
|
||||
// brotli and zstd content codings; off unless the build opted in (configure,
|
||||
|
||||
@@ -526,11 +526,6 @@ void help(const char *app, int more) {
|
||||
(" %L <file> add all URL located in this text file (one URL per line)");
|
||||
infomsg
|
||||
(" %S <file> add all scan rules located in this text file (one scan rule per line)");
|
||||
infomsg(" %m seed the crawl from the site's sitemap (robots.txt Sitemap:, "
|
||||
"then /sitemap.xml); --sitemap-url URL names one explicitly. A "
|
||||
"sitemap you name, or one the site declares, is fetched even under "
|
||||
"robots.txt Disallow; only the guessed /sitemap.xml obeys it. The "
|
||||
"URLs found still pass every filter and scope rule");
|
||||
infomsg("");
|
||||
infomsg("Build options:");
|
||||
infomsg(" NN structure type (0 *original structure, 1+: see below)");
|
||||
@@ -540,13 +535,6 @@ void help(const char *app, int more) {
|
||||
infomsg
|
||||
(" %D cached delayed type check, don't wait for remote type during updates, to speedup them (%D0 wait, * %D1 don't wait)");
|
||||
infomsg(" %M generate a RFC MIME-encapsulated full-archive (.mht)");
|
||||
infomsg(" %Z after the mirror, rewrite each saved page with its "
|
||||
"stylesheets, scripts, images and fonts inlined as data: URIs, so "
|
||||
"any page opens by double-click anywhere (links between pages stay "
|
||||
"relative; audio and video stay links); --single-file-max-size N "
|
||||
"caps each asset (default 10485760 bytes). %M is the better "
|
||||
"container where a Chromium-family browser is a given: one archive, "
|
||||
"no base64 tax on text, a shared asset stored once");
|
||||
infomsg(" %t keep the original file extension, don't rewrite it from the "
|
||||
"MIME type (%t0 rewrite)");
|
||||
infomsg
|
||||
|
||||
13
src/htslib.c
13
src/htslib.c
@@ -36,10 +36,8 @@ Please visit our Website: http://www.httrack.com
|
||||
// Fichier librairie .c
|
||||
|
||||
#include "htscore.h"
|
||||
#include "htssitemap.h"
|
||||
#include "htswarc.h"
|
||||
#include "htschanges.h"
|
||||
#include "htssinglefile.h"
|
||||
|
||||
/* specific definitions */
|
||||
#include "htsbase.h"
|
||||
@@ -1131,10 +1129,12 @@ int http_sendhead(httrackp * opt, t_cookie * cookie, int mode,
|
||||
|
||||
// Compression accepted ?
|
||||
if (retour->req.http11) {
|
||||
hts_boolean compressible =
|
||||
(!retour->req.range_used && !retour->req.nocompression);
|
||||
hts_boolean compressible = HTS_FALSE;
|
||||
hts_boolean secure = HTS_FALSE;
|
||||
|
||||
#if HTS_USEZLIB
|
||||
compressible = (!retour->req.range_used && !retour->req.nocompression);
|
||||
#endif
|
||||
#if HTS_USEOPENSSL
|
||||
secure = retour->ssl ? HTS_TRUE : HTS_FALSE;
|
||||
#endif
|
||||
@@ -6029,12 +6029,9 @@ HTSEXT_API httrackp *hts_create_opt(void) {
|
||||
StringCopy(opt->strip_query, "");
|
||||
StringCopy(opt->cookies_file, "");
|
||||
StringCopy(opt->warc_file, "");
|
||||
StringCopy(opt->sitemap_url, "");
|
||||
opt->warc_max_size = 0; /* no rotation unless --warc-max-size sets it */
|
||||
opt->changes = HTS_FALSE;
|
||||
opt->changes_state = NULL;
|
||||
opt->single_file = HTS_FALSE;
|
||||
opt->single_file_max_size = SINGLEFILE_DEFAULT_MAX_SIZE;
|
||||
StringCopy(opt->why_url, "");
|
||||
opt->pause_min_ms = 0;
|
||||
opt->pause_max_ms = 0;
|
||||
@@ -6187,8 +6184,6 @@ HTSEXT_API void hts_free_opt(httrackp * opt) {
|
||||
StringFree(opt->cookies_file);
|
||||
StringFree(opt->why_url);
|
||||
StringFree(opt->warc_file);
|
||||
StringFree(opt->sitemap_url);
|
||||
hts_sitemap_free(opt); /* backstop: httpmirror's early-return paths */
|
||||
|
||||
hts_changes_free_opt(opt);
|
||||
|
||||
|
||||
@@ -43,7 +43,9 @@ Please visit our Website: http://www.httrack.com
|
||||
#include "htsencoding.h"
|
||||
#include "htssniff.h"
|
||||
#include "htscodec.h"
|
||||
#if HTS_USEZLIB
|
||||
#include "htszlib.h"
|
||||
#endif
|
||||
#include <ctype.h>
|
||||
#include <limits.h>
|
||||
|
||||
|
||||
13
src/htsopt.h
13
src/htsopt.h
@@ -551,19 +551,6 @@ struct httrackp {
|
||||
the previous mirror. Tail: ABI */
|
||||
void *changes_state; /**< live change-report accumulator (hts_changes*),
|
||||
engine-owned. Tail: ABI */
|
||||
hts_boolean single_file; /**< --single-file: once the mirror is done, rewrite
|
||||
each saved page with its assets inlined as
|
||||
data: URIs. Tail: ABI */
|
||||
LLint single_file_max_size; /**< --single-file-max-size: per-asset cap in
|
||||
bytes; a bigger asset stays a link.
|
||||
Tail: ABI */
|
||||
hts_boolean sitemap; /**< --sitemap: probe the start host's robots.txt for
|
||||
Sitemap: lines, else /sitemap.xml. Tail: ABI */
|
||||
String sitemap_url; /**< --sitemap-url: sitemap to ingest. Tail: ABI */
|
||||
/* Live state, not an option: copy_htsopt must leave it alone. It sits here
|
||||
rather than in htsoptstate because that struct is embedded by value, so
|
||||
growing it would shift every httrackp field declared after it. */
|
||||
void *sitemap_state; /**< hts_sitemap_state*, or NULL. Tail: ABI */
|
||||
};
|
||||
|
||||
/* Running statistics for a mirror. */
|
||||
|
||||
@@ -147,8 +147,7 @@ static void robots_blob_add(char *blob, size_t blobsize, char marker,
|
||||
|
||||
void robots_parse(robots_wizard *robots, const char *adr, const char *body,
|
||||
size_t bodysize, char *info, size_t infosize,
|
||||
hts_boolean keep_root_disallow, char *sitemaps,
|
||||
size_t sitemapsize) {
|
||||
hts_boolean keep_root_disallow) {
|
||||
size_t bptr = 0;
|
||||
int record = 0;
|
||||
char BIGSTK line[1024];
|
||||
@@ -157,8 +156,6 @@ void robots_parse(robots_wizard *robots, const char *adr, const char *body,
|
||||
blob[0] = '\0';
|
||||
if (info != NULL && infosize > 0)
|
||||
info[0] = '\0';
|
||||
if (sitemaps != NULL && sitemapsize > 0)
|
||||
sitemaps[0] = '\0';
|
||||
#if DEBUG_ROBOTS
|
||||
printf("robots.txt dump:\n%s\n", body);
|
||||
#endif
|
||||
@@ -175,19 +172,7 @@ void robots_parse(robots_wizard *robots, const char *adr, const char *body,
|
||||
line[llen - 1] = '\0';
|
||||
llen--;
|
||||
}
|
||||
if (sitemaps != NULL && strfield(line, "sitemap:")) {
|
||||
// group-independent record (RFC 9309): collected whatever the group
|
||||
char *a = line + 8;
|
||||
|
||||
while (is_realspace(*a))
|
||||
a++;
|
||||
/* A line at the buffer limit was truncated: a half URL is not one. */
|
||||
if (strnotempty(a) && strlen(line) < sizeof(line) - 3 &&
|
||||
strlen(a) + 2 < sitemapsize - strlen(sitemaps)) {
|
||||
strlcatbuff(sitemaps, a, sitemapsize);
|
||||
strlcatbuff(sitemaps, "\n", sitemapsize);
|
||||
}
|
||||
} else if (strfield(line, "user-agent:")) {
|
||||
if (strfield(line, "user-agent:")) {
|
||||
char *a = line + 11;
|
||||
|
||||
while (is_realspace(*a))
|
||||
|
||||
@@ -56,12 +56,10 @@ int checkrobots(robots_wizard * robots, const char *adr, const char *fil);
|
||||
void checkrobots_free(robots_wizard * robots);
|
||||
int checkrobots_set(robots_wizard * robots, const char *adr, const char *data);
|
||||
/* Parse robots.txt `body` for `adr`, storing the HTTrack group's rules; `info`
|
||||
gets a disallow summary, `keep_root_disallow` FALSE drops "Disallow: /", and
|
||||
`sitemaps` (optional) collects the Sitemap: URLs, one per line. */
|
||||
gets a disallow summary, `keep_root_disallow` FALSE drops "Disallow: /". */
|
||||
void robots_parse(robots_wizard *robots, const char *adr, const char *body,
|
||||
size_t bodysize, char *info, size_t infosize,
|
||||
hts_boolean keep_root_disallow, char *sitemaps,
|
||||
size_t sitemapsize);
|
||||
hts_boolean keep_root_disallow);
|
||||
#endif
|
||||
|
||||
#endif
|
||||
|
||||
1106
src/htsselftest.c
1106
src/htsselftest.c
File diff suppressed because it is too large
Load Diff
1155
src/htssinglefile.c
1155
src/htssinglefile.c
File diff suppressed because it is too large
Load Diff
@@ -1,84 +0,0 @@
|
||||
/* ------------------------------------------------------------ */
|
||||
/*
|
||||
HTTrack Website Copier, Offline Browser for Windows and Unix
|
||||
Copyright (C) 2026 Xavier Roche and other contributors
|
||||
|
||||
SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License
|
||||
along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
Ethical use: we kindly ask that you NOT use this software to harvest email
|
||||
addresses or to collect any other private information about people. Doing so
|
||||
would dishonor our work and waste the many hours we have spent on it.
|
||||
|
||||
Please visit our Website: http://www.httrack.com
|
||||
*/
|
||||
|
||||
/* ------------------------------------------------------------ */
|
||||
/* --single-file: post-mirror pass embedding each page's assets as data: URIs.
|
||||
Internal, not installed. Unlike -%M (one MHT archive for the whole mirror)
|
||||
it rewrites the saved pages in place and keeps page-to-page links relative,
|
||||
so the mirror stays browsable and every page also stands alone.
|
||||
Limitation: an inlined stylesheet becomes a data: URL, whose path is opaque,
|
||||
so an asset inside it that stayed a link (over the cap) no longer resolves.
|
||||
Raise --single-file-max-size past it to inline it too. */
|
||||
/* ------------------------------------------------------------ */
|
||||
|
||||
#ifndef HTS_SINGLEFILE_DEFH
|
||||
#define HTS_SINGLEFILE_DEFH
|
||||
|
||||
#include "htsopt.h"
|
||||
#include "htsstrings.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* --single-file-max-size default: a bigger asset stays an ordinary link. */
|
||||
#define SINGLEFILE_DEFAULT_MAX_SIZE (10 * 1024 * 1024)
|
||||
|
||||
/* Never base64 more than this, whatever the cap: code64() sizes in int. */
|
||||
#define SINGLEFILE_HARD_MAX_SIZE (256 * 1024 * 1024)
|
||||
|
||||
/* Total bytes one page may inline. Nested @import fans out multiplicatively,
|
||||
so a few hundred bytes of hostile CSS can otherwise ask for gigabytes. */
|
||||
#define SINGLEFILE_MAX_PAGE_SIZE (64 * 1024 * 1024)
|
||||
|
||||
/* Rewrite every HTML page the mirror produced. No-op unless opt->single_file;
|
||||
call once the tree is final, after the update purge. */
|
||||
void singlefile_process_mirror(httrackp *opt);
|
||||
|
||||
/* Rewrite one HTML document held in memory, appending the result to out.
|
||||
root is the mirror directory that references may not escape; page_path is
|
||||
the document's own path under it (both UTF-8, '/' or native separators).
|
||||
page_budget caps the total inlined bytes, since nested @import fans out
|
||||
multiplicatively; the mirror pass passes SINGLEFILE_MAX_PAGE_SIZE.
|
||||
Returns HTS_TRUE if at least one reference was replaced; out may still
|
||||
differ from the input when that is HTS_FALSE, since a style or srcset value
|
||||
is re-serialized in place. */
|
||||
hts_boolean singlefile_rewrite_html(httrackp *opt, const char *root,
|
||||
const char *page_path, const char *html,
|
||||
size_t html_len, LLint page_budget,
|
||||
String *out);
|
||||
|
||||
/* singlefile_rewrite_html() over the file at page_path, rewritten in place.
|
||||
Returns HTS_TRUE if the file changed. */
|
||||
hts_boolean singlefile_rewrite_file(httrackp *opt, const char *root,
|
||||
const char *page_path);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
615
src/htssitemap.c
615
src/htssitemap.c
@@ -1,615 +0,0 @@
|
||||
/* ------------------------------------------------------------ */
|
||||
/*
|
||||
HTTrack Website Copier, Offline Browser for Windows and Unix
|
||||
Copyright (C) 2026 Xavier Roche and other contributors
|
||||
|
||||
SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License
|
||||
along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
Ethical use: we kindly ask that you NOT use this software to harvest email
|
||||
addresses or to collect any other private information about people. Doing so
|
||||
would dishonor our work and waste the many hours we have spent on it.
|
||||
|
||||
Please visit our Website: http://www.httrack.com
|
||||
*/
|
||||
|
||||
/* ------------------------------------------------------------ */
|
||||
/* File: sitemap ingestion (sitemaps.org 0.9) */
|
||||
/* Author: Xavier Roche */
|
||||
/* ------------------------------------------------------------ */
|
||||
|
||||
#define HTS_INTERNAL_BYTECODE
|
||||
|
||||
#include "htscore.h"
|
||||
#include "htssitemap.h"
|
||||
|
||||
#include "htsbase.h"
|
||||
#include "htscodec.h"
|
||||
#include "htsencoding.h"
|
||||
#include "htsfilters.h"
|
||||
#include "htshash.h"
|
||||
#include "htsmodules.h"
|
||||
#include "htslib.h"
|
||||
#include "htsrobots.h"
|
||||
#include "htssafe.h"
|
||||
#include "htstools.h"
|
||||
|
||||
#include <ctype.h>
|
||||
#include <string.h>
|
||||
|
||||
/* One queued sitemap document awaiting ingestion. */
|
||||
typedef struct sitemap_doc {
|
||||
char adr[HTS_URLMAXSIZE];
|
||||
char fil[HTS_URLMAXSIZE];
|
||||
int level;
|
||||
hts_sitemap_source src;
|
||||
hts_boolean done;
|
||||
struct sitemap_doc *next;
|
||||
} sitemap_doc;
|
||||
|
||||
struct hts_sitemap_state {
|
||||
sitemap_doc *docs;
|
||||
int ndocs; /* documents queued, capped by HTS_SITEMAP_MAX_DOCS */
|
||||
int nurls; /* URLs seeded, capped by HTS_SITEMAP_MAX_URLS_TOTAL */
|
||||
hts_boolean probe_done; /* the robots.txt probe has been answered */
|
||||
hts_boolean fallback_done; /* the /sitemap.xml fallback was already queued */
|
||||
/* The crawl's own start URL. Seeded URLs are judged against it, so a site
|
||||
cannot widen a subtree crawl by putting its sitemap at the root. */
|
||||
char anchor_adr[HTS_URLMAXSIZE];
|
||||
char anchor_fil[HTS_URLMAXSIZE];
|
||||
};
|
||||
typedef struct hts_sitemap_state hts_sitemap_state;
|
||||
|
||||
/* --------------------------------------------------------------------- */
|
||||
/* Document parsing (no engine state: fuzzable and self-testable) */
|
||||
/* --------------------------------------------------------------------- */
|
||||
|
||||
/* Accept only an absolute http(s) URL with no space or control byte. */
|
||||
static hts_boolean sitemap_url_ok(const char *url) {
|
||||
const char *p;
|
||||
|
||||
if (!strfield(url, "http://") && !strfield(url, "https://"))
|
||||
return HTS_FALSE;
|
||||
for (p = url; *p != '\0'; p++) {
|
||||
if ((unsigned char) *p <= ' ' || (unsigned char) *p == 0x7f)
|
||||
return HTS_FALSE;
|
||||
}
|
||||
return HTS_TRUE;
|
||||
}
|
||||
|
||||
/* Skip to the character after the next '>' at or after p, or NULL. */
|
||||
static const char *sitemap_tag_end(const char *p, const char *end) {
|
||||
while (p < end && *p != '>')
|
||||
p++;
|
||||
return p < end ? p + 1 : NULL;
|
||||
}
|
||||
|
||||
/* HTS_TRUE when the document's root element is `name`. Skips the XML
|
||||
declaration, comments and processing instructions first, so a comment
|
||||
mentioning the other root element cannot decide the document type. */
|
||||
static hts_boolean sitemap_root_is(const char *doc, size_t size,
|
||||
const char *name) {
|
||||
const size_t nlen = strlen(name);
|
||||
size_t i = 0;
|
||||
|
||||
if (size >= 3 && memcmp(doc, "\xef\xbb\xbf", 3) == 0)
|
||||
i = 3; /* UTF-8 BOM */
|
||||
while (i < size) {
|
||||
if (isspace((unsigned char) doc[i])) {
|
||||
i++;
|
||||
} else if (doc[i] != '<') {
|
||||
return HTS_FALSE; /* character data before any element: not XML */
|
||||
} else if (i + 4 <= size && memcmp(doc + i, "<!--", 4) == 0) {
|
||||
const char *const e = hts_memstr(doc + i, size - i, "-->", 3);
|
||||
|
||||
if (e == NULL)
|
||||
return HTS_FALSE;
|
||||
i = (size_t) (e - doc) + 3;
|
||||
} else if (i + 2 <= size && (doc[i + 1] == '?' || doc[i + 1] == '!')) {
|
||||
while (i < size && doc[i] != '>')
|
||||
i++;
|
||||
i++;
|
||||
} else {
|
||||
size_t j = i + 1;
|
||||
|
||||
/* an optional namespace prefix: <sm:sitemapindex> is the same element */
|
||||
while (j < size && doc[j] != ':' && doc[j] != '>' &&
|
||||
!isspace((unsigned char) doc[j]))
|
||||
j++;
|
||||
if (j >= size || doc[j] != ':')
|
||||
j = i + 1;
|
||||
else
|
||||
j++;
|
||||
return j + nlen <= size && memcmp(doc + j, name, nlen) == 0 &&
|
||||
(j + nlen == size ||
|
||||
isspace((unsigned char) doc[j + nlen]) ||
|
||||
doc[j + nlen] == '>' || doc[j + nlen] == '/')
|
||||
? HTS_TRUE
|
||||
: HTS_FALSE;
|
||||
}
|
||||
}
|
||||
return HTS_FALSE;
|
||||
}
|
||||
|
||||
/* Decompress a gzip-framed body into a fresh buffer. The 64 MiB cap is what
|
||||
binds in practice; deflate tops out near 1032:1, so the tree's codec budget
|
||||
only matters as the shared policy for a coding that could go further. */
|
||||
static char *sitemap_gunzip(const char *body, size_t size, size_t *outsize) {
|
||||
const LLint budget = hts_codec_maxout((LLint) size);
|
||||
size_t cap = budget < (LLint) HTS_SITEMAP_MAX_BYTES
|
||||
? (size_t) budget
|
||||
: (size_t) HTS_SITEMAP_MAX_BYTES;
|
||||
char *out;
|
||||
size_t n;
|
||||
|
||||
if (cap == 0)
|
||||
return NULL;
|
||||
out = malloct(cap + 1);
|
||||
if (out == NULL)
|
||||
return NULL;
|
||||
n = hts_codec_head(HTS_CODEC_DEFLATE, body, size, out, cap);
|
||||
if (n == 0) {
|
||||
freet(out);
|
||||
return NULL;
|
||||
}
|
||||
out[n] = '\0';
|
||||
*outsize = n;
|
||||
return out;
|
||||
}
|
||||
|
||||
int hts_sitemap_scan(const char *body, size_t size, int maxurls,
|
||||
hts_boolean *is_index, hts_sitemap_handler handler,
|
||||
void *arg) {
|
||||
char *unpacked = NULL;
|
||||
const char *doc;
|
||||
const char *end;
|
||||
const char *p;
|
||||
int n = 0;
|
||||
|
||||
if (is_index != NULL)
|
||||
*is_index = HTS_FALSE;
|
||||
if (body == NULL || size < 2 || handler == NULL)
|
||||
return 0;
|
||||
|
||||
/* Content-Encoding gzip is undone upstream; only the container is left. */
|
||||
if ((unsigned char) body[0] == 0x1f && (unsigned char) body[1] == 0x8b) {
|
||||
unpacked = sitemap_gunzip(body, size, &size);
|
||||
if (unpacked == NULL)
|
||||
return -1;
|
||||
doc = unpacked;
|
||||
} else {
|
||||
if (size > (size_t) HTS_SITEMAP_MAX_BYTES)
|
||||
size = (size_t) HTS_SITEMAP_MAX_BYTES;
|
||||
doc = body;
|
||||
}
|
||||
end = doc + size;
|
||||
|
||||
/* Set before the first callback: the handler reads the verdict. */
|
||||
if (is_index != NULL)
|
||||
*is_index = sitemap_root_is(doc, size, "sitemapindex");
|
||||
|
||||
for (p = doc; n < maxurls;) {
|
||||
const char *loc = hts_memstr(p, (size_t) (end - p), "<loc", 4);
|
||||
const char *val;
|
||||
const char *stop;
|
||||
size_t len;
|
||||
char BIGSTK url[HTS_URLMAXSIZE];
|
||||
|
||||
if (loc == NULL)
|
||||
break;
|
||||
/* "<loc>" or "<loc xmlns:..>", never "<location>" */
|
||||
if (loc + 4 >= end || (loc[4] != '>' && !isspace((unsigned char) loc[4]))) {
|
||||
p = loc + 4;
|
||||
continue;
|
||||
}
|
||||
val = sitemap_tag_end(loc + 4, end);
|
||||
if (val == NULL)
|
||||
break;
|
||||
for (stop = val; stop < end && *stop != '<'; stop++)
|
||||
;
|
||||
/* No closing tag: truncated document, so the value may be a partial URL. */
|
||||
if (stop == end)
|
||||
break;
|
||||
p = stop;
|
||||
while (val < stop && isspace((unsigned char) *val))
|
||||
val++;
|
||||
while (stop > val && isspace((unsigned char) *(stop - 1)))
|
||||
stop--;
|
||||
len = (size_t) (stop - val);
|
||||
/* Overflow-safe: the untrusted length alone against the room left. */
|
||||
if (len == 0 || len >= sizeof(url))
|
||||
continue;
|
||||
memcpy(url, val, len);
|
||||
url[len] = '\0';
|
||||
/* hts_unescapeEntities decodes in place and tolerates src == dest; a
|
||||
reference to a control byte survives as one and sitemap_url_ok drops it.
|
||||
*/
|
||||
if (hts_unescapeEntities(url, url, sizeof(url)) != 0 ||
|
||||
!sitemap_url_ok(url))
|
||||
continue;
|
||||
n++;
|
||||
if (!handler(arg, url))
|
||||
break;
|
||||
}
|
||||
|
||||
if (unpacked != NULL)
|
||||
freet(unpacked);
|
||||
return n;
|
||||
}
|
||||
|
||||
/* --------------------------------------------------------------------- */
|
||||
/* Engine glue */
|
||||
/* --------------------------------------------------------------------- */
|
||||
|
||||
static hts_sitemap_state *sitemap_get_state(httrackp *opt) {
|
||||
if (opt->sitemap_state == NULL)
|
||||
opt->sitemap_state = calloct(1, sizeof(hts_sitemap_state));
|
||||
return (hts_sitemap_state *) opt->sitemap_state;
|
||||
}
|
||||
|
||||
static sitemap_doc *sitemap_find(httrackp *opt, const char *adr,
|
||||
const char *fil) {
|
||||
hts_sitemap_state *const st = (hts_sitemap_state *) opt->sitemap_state;
|
||||
sitemap_doc *d;
|
||||
|
||||
if (st == NULL)
|
||||
return NULL;
|
||||
for (d = st->docs; d != NULL; d = d->next) {
|
||||
if (strfield2(d->adr, adr) && strcmp(d->fil, fil) == 0)
|
||||
return d;
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Who asked for this document decides how far it is gated. The wizard proper
|
||||
is not usable here: it wants a referring link, and its up/down travel rules
|
||||
would judge a child sitemap against the parent sitemap's directory. */
|
||||
static hts_boolean sitemap_fetch_allowed(httrackp *opt, const char *adr,
|
||||
const char *fil,
|
||||
hts_sitemap_source src) {
|
||||
/* adr and fil are each capped just under HTS_URLMAXSIZE, and lfull prefixes
|
||||
a scheme and a slash on top of both: 2 * HTS_URLMAXSIZE does not fit. */
|
||||
char BIGSTK l[HTS_URLMAXSIZE * 2 + 16], lfull[HTS_URLMAXSIZE * 2 + 16];
|
||||
int jokdepth = 0, jok;
|
||||
|
||||
hts_boolean refused;
|
||||
|
||||
/* The user naming a sitemap is the same intent as naming a start URL, which
|
||||
the wizard admits unconditionally. */
|
||||
if (src == HTS_SITEMAP_SRC_USER)
|
||||
return HTS_TRUE;
|
||||
strcpybuff(l, jump_identification_const(adr));
|
||||
if (*fil != '/')
|
||||
strcatbuff(l, "/");
|
||||
strcatbuff(l, fil);
|
||||
strcpybuff(lfull, link_has_authority(adr) ? "" : "http://");
|
||||
strcatbuff(lfull, adr);
|
||||
if (*fil != '/')
|
||||
strcatbuff(lfull, "/");
|
||||
strcatbuff(lfull, fil);
|
||||
jok = fa_strjoker_dual(0, *opt->filters.filters, *opt->filters.filptr, lfull,
|
||||
l, NULL, NULL, &jokdepth);
|
||||
refused = (jok == -1) ? HTS_TRUE : HTS_FALSE;
|
||||
if (refused) {
|
||||
hts_log_print(opt, LOG_NOTICE, "Sitemap: filter rule #%d refuses %s%s",
|
||||
jokdepth + 1, adr, fil);
|
||||
return HTS_FALSE;
|
||||
}
|
||||
/* A Sitemap: line, or a sitemapindex entry, is the site inviting the fetch;
|
||||
a Disallow elsewhere in the same file does not retract it. The well-known
|
||||
location is only ever a guess, so there a Disallow wins. */
|
||||
if (src == HTS_SITEMAP_SRC_GUESSED &&
|
||||
hts_robots_forbids(opt, adr, fil, (jok != 0) ? HTS_TRUE : HTS_FALSE,
|
||||
refused)) {
|
||||
hts_log_print(opt, LOG_NOTICE, "Sitemap: robots.txt forbids %s%s", adr,
|
||||
fil);
|
||||
return HTS_FALSE;
|
||||
}
|
||||
return HTS_TRUE;
|
||||
}
|
||||
|
||||
/* Record the link with save="" so the body stays in memory: a sitemap is
|
||||
ingested, never mirrored. */
|
||||
static hts_boolean sitemap_queue_(httrackp *opt, const char *adr,
|
||||
const char *fil, int level,
|
||||
hts_sitemap_source src, hts_boolean link_it) {
|
||||
hts_sitemap_state *const st = sitemap_get_state(opt);
|
||||
sitemap_doc *d;
|
||||
|
||||
if (st == NULL)
|
||||
return HTS_FALSE;
|
||||
if (st->ndocs >= HTS_SITEMAP_MAX_DOCS || level > HTS_SITEMAP_MAX_LEVEL) {
|
||||
hts_log_print(opt, LOG_WARNING, "Sitemap: cap reached, skipping %s%s", adr,
|
||||
fil);
|
||||
return HTS_FALSE;
|
||||
}
|
||||
if (strlen(adr) >= sizeof(d->adr) || strlen(fil) >= sizeof(d->fil))
|
||||
return HTS_FALSE;
|
||||
if (sitemap_find(opt, adr, fil) != NULL)
|
||||
return HTS_FALSE;
|
||||
if (!sitemap_fetch_allowed(opt, adr, fil, src))
|
||||
return HTS_FALSE;
|
||||
d = calloct(1, sizeof(sitemap_doc));
|
||||
if (d == NULL)
|
||||
return HTS_FALSE;
|
||||
strcpybuff(d->adr, adr);
|
||||
strcpybuff(d->fil, fil);
|
||||
d->level = level;
|
||||
d->src = src;
|
||||
d->next = st->docs;
|
||||
st->docs = d;
|
||||
st->ndocs++;
|
||||
|
||||
if (!link_it)
|
||||
return HTS_TRUE;
|
||||
if (!hts_record_link(opt, adr, fil, "", "", "", NULL))
|
||||
return HTS_FALSE;
|
||||
heap_top()->testmode = 0;
|
||||
heap_top()->link_import = 0;
|
||||
heap_top()->depth = opt->depth + 1;
|
||||
heap_top()->pass2 = 0;
|
||||
heap_top()->retry = opt->retry;
|
||||
heap_top()->premier = heap_top_index();
|
||||
heap_top()->precedent = heap_top_index();
|
||||
hts_log_print(opt, LOG_INFO, "Sitemap: queued %s%s", adr, fil);
|
||||
return HTS_TRUE;
|
||||
}
|
||||
|
||||
static hts_boolean sitemap_queue(httrackp *opt, const char *adr,
|
||||
const char *fil, int level,
|
||||
hts_sitemap_source src) {
|
||||
return sitemap_queue_(opt, adr, fil, level, src, HTS_TRUE);
|
||||
}
|
||||
|
||||
void hts_sitemap_redirect(httrackp *opt, const char *adr, const char *fil,
|
||||
const char *newadr, const char *newfil) {
|
||||
sitemap_doc *const d = sitemap_find(opt, adr, fil);
|
||||
|
||||
if (d == NULL || d->done)
|
||||
return;
|
||||
d->done = HTS_TRUE; /* the body lives at the target now */
|
||||
/* The engine already queued the target link, so only the marking moves. */
|
||||
(void) sitemap_queue_(opt, newadr, newfil, d->level, d->src, HTS_FALSE);
|
||||
hts_log_print(opt, LOG_NOTICE, "Sitemap: %s%s redirects to %s%s", adr, fil,
|
||||
newadr, newfil);
|
||||
}
|
||||
|
||||
void hts_sitemap_seed(httrackp *opt, const char *starturl) {
|
||||
char BIGSTK url[HTS_URLMAXSIZE * 2];
|
||||
lien_adrfil af;
|
||||
|
||||
if (StringNotEmpty(opt->sitemap_url)) {
|
||||
if (strlen(StringBuff(opt->sitemap_url)) >= sizeof(url)) {
|
||||
hts_log_print(opt, LOG_ERROR, "Sitemap URL too long");
|
||||
} else {
|
||||
strcpybuff(url, StringBuff(opt->sitemap_url));
|
||||
if (strstr(url, ":/") == NULL)
|
||||
hts_log_print(opt, LOG_ERROR, "Sitemap URL must be absolute: %s", url);
|
||||
else if (ident_url_absolute(url, &af) >= 0)
|
||||
(void) sitemap_queue(opt, af.adr, af.fil, 0, HTS_SITEMAP_SRC_USER);
|
||||
}
|
||||
}
|
||||
if (starturl == NULL || starturl[0] == '\0' ||
|
||||
strlen(starturl) >= sizeof(url))
|
||||
return;
|
||||
strcpybuff(url, starturl);
|
||||
if (ident_url_absolute(url, &af) < 0)
|
||||
return;
|
||||
{
|
||||
hts_sitemap_state *const st = sitemap_get_state(opt);
|
||||
|
||||
if (st != NULL && strlen(af.adr) < sizeof(st->anchor_adr) &&
|
||||
strlen(af.fil) < sizeof(st->anchor_fil)) {
|
||||
strcpybuff(st->anchor_adr, af.adr);
|
||||
strcpybuff(st->anchor_fil, af.fil);
|
||||
}
|
||||
}
|
||||
if (!opt->sitemap)
|
||||
return;
|
||||
/* Answered in hts_sitemap_robots, once the parsed rules are installed. */
|
||||
if (hts_record_link(opt, af.adr, "/robots.txt", "", "", "", NULL)) {
|
||||
heap_top()->testmode = 0;
|
||||
heap_top()->link_import = 0;
|
||||
heap_top()->depth = 0;
|
||||
heap_top()->pass2 = 0;
|
||||
heap_top()->retry = opt->retry;
|
||||
heap_top()->premier = heap_top_index();
|
||||
heap_top()->precedent = heap_top_index();
|
||||
/* Claim the host so the parser does not queue robots.txt a second time. */
|
||||
if (opt->robotsptr != NULL)
|
||||
(void) checkrobots_set((robots_wizard *) opt->robotsptr, af.adr, "");
|
||||
}
|
||||
}
|
||||
|
||||
void hts_sitemap_robots(httrackp *opt, const char *adr, const char *sitemaps) {
|
||||
hts_sitemap_state *const st = (hts_sitemap_state *) opt->sitemap_state;
|
||||
int queued = 0;
|
||||
|
||||
if (st == NULL || !opt->sitemap || st->probe_done ||
|
||||
!strfield2(st->anchor_adr, adr))
|
||||
return;
|
||||
st->probe_done = HTS_TRUE;
|
||||
if (sitemaps != NULL) {
|
||||
const char *p = sitemaps;
|
||||
|
||||
while (*p != '\0') {
|
||||
const char *const eol = strchr(p, '\n');
|
||||
const size_t len = eol != NULL ? (size_t) (eol - p) : strlen(p);
|
||||
char BIGSTK line[HTS_URLMAXSIZE];
|
||||
lien_adrfil af;
|
||||
|
||||
if (len > 0 && len < sizeof(line)) {
|
||||
memcpy(line, p, len);
|
||||
line[len] = '\0';
|
||||
/* Same host: a Sitemap: line must not aim the fetcher elsewhere. */
|
||||
if (sitemap_url_ok(line) && ident_url_absolute(line, &af) >= 0 &&
|
||||
strfield2(af.adr, adr) &&
|
||||
sitemap_queue(opt, af.adr, af.fil, 0, HTS_SITEMAP_SRC_DECLARED))
|
||||
queued++;
|
||||
}
|
||||
if (eol == NULL)
|
||||
break;
|
||||
p = eol + 1;
|
||||
}
|
||||
}
|
||||
if (queued == 0 && !st->fallback_done) {
|
||||
st->fallback_done = HTS_TRUE;
|
||||
if (sitemap_queue(opt, adr, "/sitemap.xml", 0, HTS_SITEMAP_SRC_GUESSED))
|
||||
queued++;
|
||||
}
|
||||
hts_log_print(opt, LOG_NOTICE, "Sitemap: %d sitemap(s) queued for %s", queued,
|
||||
adr);
|
||||
}
|
||||
|
||||
hts_boolean hts_sitemap_pending(httrackp *opt, const char *adr,
|
||||
const char *fil) {
|
||||
const sitemap_doc *const d = sitemap_find(opt, adr, fil);
|
||||
|
||||
return d != NULL && !d->done ? HTS_TRUE : HTS_FALSE;
|
||||
}
|
||||
|
||||
/* Handler context: seeding URLs from one document. */
|
||||
typedef struct sitemap_ingest_ctx {
|
||||
httrackp *opt;
|
||||
htsmoduleStruct *str;
|
||||
const char *adr; /* host of the document being ingested */
|
||||
int level;
|
||||
hts_boolean is_index;
|
||||
int accepted; /* URLs seeded or documents queued, not merely parsed */
|
||||
} sitemap_ingest_ctx;
|
||||
|
||||
/* A <loc> of a <urlset>: hand it to the wizard as a top-level seed.
|
||||
The wizard is pointed at the crawl's own start URL, not at the sitemap: the
|
||||
site picks where its sitemap lives, so anchoring travel there would let a
|
||||
root sitemap widen a subtree crawl to the whole host. The URL then becomes
|
||||
its own anchor, exactly as a command-line seed does. */
|
||||
static hts_boolean sitemap_seed_url(void *arg, const char *url) {
|
||||
sitemap_ingest_ctx *const c = (sitemap_ingest_ctx *) arg;
|
||||
httrackp *const opt = c->opt;
|
||||
hts_sitemap_state *const st = sitemap_get_state(opt);
|
||||
char BIGSTK buff[HTS_URLMAXSIZE];
|
||||
int before;
|
||||
|
||||
if (st == NULL || st->nurls >= HTS_SITEMAP_MAX_URLS_TOTAL) {
|
||||
hts_log_print(opt, LOG_WARNING,
|
||||
"Sitemap: URL cap reached, ignoring the rest");
|
||||
return HTS_FALSE;
|
||||
}
|
||||
/* strcpybuff aborts rather than truncating: never feed it unchecked input. */
|
||||
if (strlen(url) >= sizeof(buff))
|
||||
return HTS_TRUE;
|
||||
st->nurls++;
|
||||
strcpybuff(buff, url);
|
||||
before = opt->lien_tot;
|
||||
if (htsAddLink(c->str, buff))
|
||||
c->accepted++;
|
||||
if (opt->lien_tot > before)
|
||||
heap_top()->premier = heap_top_index(); /* a seed anchors on itself */
|
||||
return HTS_TRUE;
|
||||
}
|
||||
|
||||
/* A <loc> of a <sitemapindex>: cross-host children are dropped, so a hostile
|
||||
sitemap cannot aim the fetcher elsewhere. */
|
||||
static hts_boolean sitemap_seed_child(void *arg, const char *url) {
|
||||
sitemap_ingest_ctx *const c = (sitemap_ingest_ctx *) arg;
|
||||
char BIGSTK buff[HTS_URLMAXSIZE];
|
||||
lien_adrfil af;
|
||||
|
||||
if (strlen(url) >= sizeof(buff))
|
||||
return HTS_TRUE;
|
||||
strcpybuff(buff, url);
|
||||
if (ident_url_absolute(buff, &af) < 0)
|
||||
return HTS_TRUE;
|
||||
if (!strfield2(af.adr, c->adr)) {
|
||||
hts_log_print(c->opt, LOG_WARNING,
|
||||
"Sitemap: ignoring off-host child sitemap %s%s", af.adr,
|
||||
af.fil);
|
||||
return HTS_TRUE;
|
||||
}
|
||||
if (sitemap_queue(c->opt, af.adr, af.fil, c->level + 1,
|
||||
HTS_SITEMAP_SRC_DECLARED))
|
||||
c->accepted++;
|
||||
return HTS_TRUE;
|
||||
}
|
||||
|
||||
/* The scan classifies the document before the first callback. */
|
||||
static hts_boolean sitemap_seed_any(void *arg, const char *url) {
|
||||
sitemap_ingest_ctx *const c = (sitemap_ingest_ctx *) arg;
|
||||
|
||||
return c->is_index ? sitemap_seed_child(arg, url)
|
||||
: sitemap_seed_url(arg, url);
|
||||
}
|
||||
|
||||
void hts_sitemap_ingest(httrackp *opt, htsmoduleStruct *str, const char *adr,
|
||||
const char *fil, const char *body, size_t size) {
|
||||
sitemap_doc *const d = sitemap_find(opt, adr, fil);
|
||||
hts_sitemap_state *const st = (hts_sitemap_state *) opt->sitemap_state;
|
||||
sitemap_ingest_ctx ctx;
|
||||
int n, anchor, saved_depth;
|
||||
|
||||
if (d == NULL || d->done)
|
||||
return;
|
||||
d->done = HTS_TRUE;
|
||||
/* str->ptr_ is a scratch int owned by the caller, so nothing else moves. */
|
||||
anchor = *str->ptr_;
|
||||
if (st != NULL && st->anchor_adr[0] != '\0' && opt->hash != NULL) {
|
||||
const int i = hash_read((const hash_struct *) opt->hash, st->anchor_adr,
|
||||
st->anchor_fil, 1);
|
||||
|
||||
if (i >= 0)
|
||||
anchor = i;
|
||||
}
|
||||
*str->ptr_ = anchor;
|
||||
/* Borrow the anchor's position but keep a seed's full depth budget. */
|
||||
saved_depth = heap(anchor)->depth;
|
||||
heap(anchor)->depth = opt->depth + 1;
|
||||
ctx.opt = opt;
|
||||
ctx.str = str;
|
||||
ctx.adr = adr;
|
||||
ctx.level = d->level;
|
||||
ctx.is_index = HTS_FALSE;
|
||||
ctx.accepted = 0;
|
||||
|
||||
n = hts_sitemap_scan(body, size, HTS_SITEMAP_MAX_URLS_DOC, &ctx.is_index,
|
||||
sitemap_seed_any, &ctx);
|
||||
heap(anchor)->depth = saved_depth;
|
||||
if (n < 0) {
|
||||
hts_log_print(opt, LOG_ERROR, "Sitemap: could not decompress %s%s", adr,
|
||||
fil);
|
||||
return;
|
||||
}
|
||||
if (ctx.is_index)
|
||||
hts_log_print(opt, LOG_NOTICE,
|
||||
"Sitemap: %d of %d child sitemap(s) listed by %s%s",
|
||||
ctx.accepted, n, adr, fil);
|
||||
else
|
||||
hts_log_print(opt, LOG_NOTICE, "Sitemap: %d of %d URL(s) added from %s%s",
|
||||
ctx.accepted, n, adr, fil);
|
||||
}
|
||||
|
||||
void hts_sitemap_free(httrackp *opt) {
|
||||
hts_sitemap_state *const st = (hts_sitemap_state *) opt->sitemap_state;
|
||||
|
||||
if (st == NULL)
|
||||
return;
|
||||
while (st->docs != NULL) {
|
||||
sitemap_doc *const next = st->docs->next;
|
||||
|
||||
freet(st->docs);
|
||||
st->docs = next;
|
||||
}
|
||||
freet(opt->sitemap_state);
|
||||
opt->sitemap_state = NULL;
|
||||
}
|
||||
108
src/htssitemap.h
108
src/htssitemap.h
@@ -1,108 +0,0 @@
|
||||
/* ------------------------------------------------------------ */
|
||||
/*
|
||||
HTTrack Website Copier, Offline Browser for Windows and Unix
|
||||
Copyright (C) 2026 Xavier Roche and other contributors
|
||||
|
||||
SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License
|
||||
along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
Ethical use: we kindly ask that you NOT use this software to harvest email
|
||||
addresses or to collect any other private information about people. Doing so
|
||||
would dishonor our work and waste the many hours we have spent on it.
|
||||
|
||||
Please visit our Website: http://www.httrack.com
|
||||
*/
|
||||
|
||||
/* ------------------------------------------------------------ */
|
||||
/* HTTrack sitemap ingestion (sitemaps.org 0.9). Internal, not installed.
|
||||
Reads <urlset>/<sitemapindex> documents, plain or gzip-framed, and feeds
|
||||
their <loc> URLs to the crawl as top-level seeds. The whole input is
|
||||
attacker-controlled, so every entry point below is capped. */
|
||||
/* ------------------------------------------------------------ */
|
||||
|
||||
#ifndef HTS_SITEMAP_DEFH
|
||||
#define HTS_SITEMAP_DEFH
|
||||
|
||||
#include "htsdefines.h"
|
||||
#include "htsopt.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* Caps. sitemaps.org allows 50000 URLs and 50 MB uncompressed per document;
|
||||
the byte cap sits above that so a conformant sitemap always fits. */
|
||||
#define HTS_SITEMAP_MAX_URLS_DOC 50000 /* <loc> per document */
|
||||
#define HTS_SITEMAP_MAX_URLS_TOTAL 200000 /* <loc> per mirror */
|
||||
#define HTS_SITEMAP_MAX_DOCS 256 /* documents per mirror */
|
||||
#define HTS_SITEMAP_MAX_LEVEL 4 /* sitemapindex nesting */
|
||||
#define HTS_SITEMAP_MAX_BYTES (64 * 1024 * 1024) /* decompressed document */
|
||||
|
||||
/* Who asked for a sitemap document, which decides how far its fetch is gated.
|
||||
The user naming one is the same intent as a start URL; a site declaring one
|
||||
invites the fetch; the well-known location is only ever our guess. */
|
||||
typedef enum {
|
||||
HTS_SITEMAP_SRC_USER, /**< --sitemap-url */
|
||||
HTS_SITEMAP_SRC_DECLARED, /**< a Sitemap: line or a sitemapindex entry */
|
||||
HTS_SITEMAP_SRC_GUESSED /**< the /sitemap.xml fallback */
|
||||
} hts_sitemap_source;
|
||||
|
||||
/* Per-URL handler; returning HTS_FALSE stops the scan. */
|
||||
typedef hts_boolean (*hts_sitemap_handler)(void *arg, const char *url);
|
||||
|
||||
/* Scan one sitemap document, plain or gzip-framed, handing every acceptable
|
||||
absolute http(s) <loc> URL to `handler`. Stops after `maxurls` URLs, or when
|
||||
the handler refuses. `is_index` (optional) reports a <sitemapindex>, whose
|
||||
URLs are child sitemaps rather than pages. Returns the number of URLs handed
|
||||
out, or -1 when the document could not be decompressed within the caps. */
|
||||
int hts_sitemap_scan(const char *body, size_t size, int maxurls,
|
||||
hts_boolean *is_index, hts_sitemap_handler handler,
|
||||
void *arg);
|
||||
|
||||
/* --- Engine glue (needs a live httrackp). --- */
|
||||
|
||||
/* Queue the first sitemap document of the mirror: the explicit --sitemap-url,
|
||||
or the start host's /robots.txt probe for --sitemap. `starturl` is the first
|
||||
command-line seed. No-op when neither option is set. */
|
||||
void hts_sitemap_seed(httrackp *opt, const char *starturl);
|
||||
|
||||
/* Act on the start host's robots.txt once its rules are installed: queue the
|
||||
Sitemap: URLs it names (newline-separated, from robots_parse), or the
|
||||
well-known /sitemap.xml when it names none. No-op unless --sitemap. */
|
||||
void hts_sitemap_robots(httrackp *opt, const char *adr, const char *sitemaps);
|
||||
|
||||
/* Carry the sitemap marking of (adr,fil) over to the target of a redirect the
|
||||
engine has already queued, so a moved sitemap is still ingested. */
|
||||
void hts_sitemap_redirect(httrackp *opt, const char *adr, const char *fil,
|
||||
const char *newadr, const char *newfil);
|
||||
|
||||
/* HTS_TRUE when (adr,fil) is a queued sitemap document awaiting ingestion. */
|
||||
hts_boolean hts_sitemap_pending(httrackp *opt, const char *adr,
|
||||
const char *fil);
|
||||
|
||||
/* Ingest a fetched sitemap document (or the robots.txt probe): seed its URLs
|
||||
through the wizard via htsAddLink, and queue nested sitemaps. `str` supplies
|
||||
the parser context of the document being processed. */
|
||||
void hts_sitemap_ingest(httrackp *opt, htsmoduleStruct *str, const char *adr,
|
||||
const char *fil, const char *body, size_t size);
|
||||
|
||||
/* Release the ingestion state held in opt (NULL-safe, idempotent). */
|
||||
void hts_sitemap_free(httrackp *opt);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -44,18 +44,9 @@ Please visit our Website: http://www.httrack.com
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/* Outstanding threads, counted at spawn rather than by the child at entry, so
|
||||
that a caller which spawns and immediately waits still joins them (#747). */
|
||||
static int process_chain = 0;
|
||||
static htsmutex process_chain_mutex = HTSMUTEX_INIT;
|
||||
|
||||
static void process_chain_add(int delta) {
|
||||
hts_mutexlock(&process_chain_mutex);
|
||||
process_chain += delta;
|
||||
assertf(process_chain >= 0);
|
||||
hts_mutexrelease(&process_chain_mutex);
|
||||
}
|
||||
|
||||
HTSEXT_API void htsthread_wait(void) {
|
||||
htsthread_wait_n(0);
|
||||
}
|
||||
@@ -108,12 +99,20 @@ static void *hts_entry_point(void *tharg)
|
||||
void *const arg = s_args->arg;
|
||||
void (*fun) (void *arg) = s_args->fun;
|
||||
|
||||
freet(tharg);
|
||||
free(tharg);
|
||||
|
||||
hts_mutexlock(&process_chain_mutex);
|
||||
process_chain++;
|
||||
assertf(process_chain > 0);
|
||||
hts_mutexrelease(&process_chain_mutex);
|
||||
|
||||
/* run */
|
||||
fun(arg);
|
||||
|
||||
process_chain_add(-1);
|
||||
hts_mutexlock(&process_chain_mutex);
|
||||
process_chain--;
|
||||
assertf(process_chain >= 0);
|
||||
hts_mutexrelease(&process_chain_mutex);
|
||||
#ifdef _WIN32
|
||||
return 0;
|
||||
#else
|
||||
@@ -123,20 +122,18 @@ static void *hts_entry_point(void *tharg)
|
||||
|
||||
/* create a thread */
|
||||
HTSEXT_API int hts_newthread(void (*fun) (void *arg), void *arg) {
|
||||
hts_thread_s *s_args = malloct(sizeof(hts_thread_s));
|
||||
hts_thread_s *s_args = malloc(sizeof(hts_thread_s));
|
||||
|
||||
assertf(s_args != NULL);
|
||||
s_args->arg = arg;
|
||||
s_args->fun = fun;
|
||||
process_chain_add(1);
|
||||
#ifdef _WIN32
|
||||
{
|
||||
unsigned int idt;
|
||||
HANDLE handle =
|
||||
(HANDLE) _beginthreadex(NULL, 0, hts_entry_point, s_args, 0, &idt);
|
||||
if (handle == 0) {
|
||||
process_chain_add(-1);
|
||||
freet(s_args);
|
||||
free(s_args);
|
||||
return -1;
|
||||
} else {
|
||||
/* detach the thread from the main process so that is can be independent */
|
||||
@@ -154,8 +151,7 @@ HTSEXT_API int hts_newthread(void (*fun) (void *arg), void *arg) {
|
||||
|| pthread_attr_setstacksize(&attr, stackSize) != 0
|
||||
|| (retcode =
|
||||
pthread_create(&handle, &attr, hts_entry_point, s_args)) != 0) {
|
||||
process_chain_add(-1);
|
||||
freet(s_args);
|
||||
free(s_args);
|
||||
return -1;
|
||||
} else {
|
||||
/* detach the thread from the main process so that is can be independent */
|
||||
|
||||
@@ -276,21 +276,6 @@ int ident_url_relatif(const char *lien, const char *origin_adr,
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Bounded substring search: bodies and archive records carry NUL bytes, so
|
||||
strstr() would stop at the first one. */
|
||||
const char *hts_memstr(const char *hay, size_t haylen, const char *needle,
|
||||
size_t nlen) {
|
||||
size_t i;
|
||||
|
||||
if (nlen == 0 || haylen < nlen)
|
||||
return NULL;
|
||||
for (i = 0; i + nlen <= haylen; i++) {
|
||||
if (hay[i] == *needle && memcmp(hay + i, needle, nlen) == 0)
|
||||
return hay + i;
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
// créer dans s, à partir du chemin courant curr_fil, le lien vers link (absolu)
|
||||
// un ident_url_relatif a déja été fait avant, pour que link ne soit pas un chemin relatif
|
||||
int lienrelatif(char *s, size_t ssize, const char *link, const char *curr_fil) {
|
||||
@@ -1443,15 +1428,3 @@ HTSEXT_API hts_boolean hts_findissystem(find_handle find) {
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
hts_boolean hts_rename_over(const char *src, const char *dst) {
|
||||
char csrc[CATBUFF_SIZE], cdst[CATBUFF_SIZE];
|
||||
|
||||
fconv(csrc, sizeof(csrc), src);
|
||||
fconv(cdst, sizeof(cdst), dst);
|
||||
if (RENAME(csrc, cdst) == 0)
|
||||
return HTS_TRUE;
|
||||
/* RENAME does not clobber an existing target on Windows. */
|
||||
(void) UNLINK(cdst);
|
||||
return RENAME(csrc, cdst) == 0 ? HTS_TRUE : HTS_FALSE;
|
||||
}
|
||||
|
||||
@@ -61,11 +61,6 @@ typedef struct lien_adrfilsave lien_adrfilsave;
|
||||
int ident_url_relatif(const char *lien, const char *origin_adr,
|
||||
const char *origin_fil,
|
||||
lien_adrfil* const adrfil);
|
||||
/* Bounded substring search over data that may hold NUL bytes; NULL if absent.
|
||||
*/
|
||||
const char *hts_memstr(const char *hay, size_t haylen, const char *needle,
|
||||
size_t nlen);
|
||||
|
||||
int lienrelatif(char *s, size_t ssize, const char *link, const char *curr);
|
||||
int link_has_authority(const char *lien);
|
||||
int link_has_authorization(const char *lien);
|
||||
@@ -137,10 +132,6 @@ HTSEXT_API hts_boolean hts_findisdir(find_handle find);
|
||||
HTSEXT_API hts_boolean hts_findisfile(find_handle find);
|
||||
HTSEXT_API hts_boolean hts_findissystem(find_handle find);
|
||||
|
||||
/* Move src onto dst, replacing an existing dst; HTS_TRUE on success. Both
|
||||
paths are fconv()'d. */
|
||||
hts_boolean hts_rename_over(const char *src, const char *dst);
|
||||
|
||||
#endif
|
||||
|
||||
#endif
|
||||
|
||||
@@ -964,6 +964,16 @@ static zipFile wacz_zip_open(const char *path) {
|
||||
return zipOpen2_64(path, 0 /*create*/, NULL, &ff);
|
||||
}
|
||||
|
||||
/* Move src onto dst (UTF-8/Windows-safe); RENAME won't clobber on Windows, so
|
||||
fall back to unlink+rename. Returns 0 on success. */
|
||||
static int wacz_rename_over(const char *src, const char *dst) {
|
||||
char cs[CATBUFF_SIZE], cd[CATBUFF_SIZE];
|
||||
if (RENAME(fconv(cs, sizeof(cs), src), fconv(cd, sizeof(cd), dst)) == 0)
|
||||
return 0;
|
||||
(void) UNLINK(fconv(cd, sizeof(cd), dst));
|
||||
return RENAME(fconv(cs, sizeof(cs), src), fconv(cd, sizeof(cd), dst));
|
||||
}
|
||||
|
||||
/* Package the segment(s) + .cdx + a generated pages.jsonl into <base>.wacz at
|
||||
crawl end (the archive file(s) and .cdx are already closed on disk). */
|
||||
static void warc_wacz_package(warc_writer *w) {
|
||||
@@ -1082,7 +1092,7 @@ static void warc_wacz_package(warc_writer *w) {
|
||||
hts_log_print(w->opt, LOG_WARNING,
|
||||
"WACZ: packaging failed, kept existing %s untouched",
|
||||
waczpath);
|
||||
} else if (!hts_rename_over(tmppath, waczpath)) {
|
||||
} else if (wacz_rename_over(tmppath, waczpath) != 0) {
|
||||
(void) UNLINK(fconv(catbuff, sizeof(catbuff), tmppath));
|
||||
hts_log_print(w->opt, LOG_WARNING | LOG_ERRNO,
|
||||
"WACZ: could not finalize %s", waczpath);
|
||||
|
||||
21
src/htsweb.c
21
src/htsweb.c
@@ -101,8 +101,8 @@ static void htsweb_sig_brpipe(int code) {
|
||||
/* ignore */
|
||||
}
|
||||
|
||||
/* Threads that never return; no wait may count on them draining. */
|
||||
static int nonjoinable_threads = 0;
|
||||
/* Number of background threads */
|
||||
static int background_threads = 0;
|
||||
|
||||
/* Server/client ping handling */
|
||||
static htsmutex pingMutex = HTSMUTEX_INIT;
|
||||
@@ -224,7 +224,9 @@ int main(int argc, char *argv[]) {
|
||||
#ifdef HTS_USESWF
|
||||
smallserver_setkey("USESWF", "1");
|
||||
#endif
|
||||
#ifdef HTS_USEZLIB
|
||||
smallserver_setkey("USEZLIB", "1");
|
||||
#endif
|
||||
#ifdef _WIN32
|
||||
smallserver_setkey("WIN32", "1");
|
||||
#endif
|
||||
@@ -297,19 +299,15 @@ int main(int argc, char *argv[]) {
|
||||
|
||||
/* pinger */
|
||||
if (parentPid > 0) {
|
||||
if (hts_newthread(client_ping, (void *) (uintptr_t) parentPid) == 0) {
|
||||
#ifndef _WIN32
|
||||
nonjoinable_threads++; /* client_ping() only ever leaves through exit() */
|
||||
#endif
|
||||
}
|
||||
hts_newthread(client_ping, (void *) (uintptr_t) parentPid);
|
||||
background_threads++; /* Do not wait for this thread! */
|
||||
smallserver_setpinghandler(pingHandler, NULL);
|
||||
}
|
||||
|
||||
/* launch */
|
||||
ret = help_server(argv[1], defaultPort, bindAddr);
|
||||
|
||||
/* Drain everything a mirror may still have in flight, the pinger aside. */
|
||||
htsthread_wait_n(nonjoinable_threads);
|
||||
htsthread_wait_n(background_threads - 1);
|
||||
hts_uninit();
|
||||
|
||||
#ifdef _WIN32
|
||||
@@ -384,6 +382,7 @@ void webhttrack_main(char *cmd) {
|
||||
commandRunning = 1;
|
||||
DEBUG(fprintf(stderr, "commandRunning=1\n"));
|
||||
hts_newthread(back_launch_cmd, (void *) strdup(cmd));
|
||||
background_threads++; /* Do not wait for this thread! */
|
||||
}
|
||||
|
||||
void webhttrack_lock(void) {
|
||||
@@ -424,8 +423,8 @@ static int webhttrack_runmain(httrackp * opt, int argc, char **argv) {
|
||||
/* Rock'in! */
|
||||
ret = hts_main2(argc, argv, opt);
|
||||
|
||||
/* Wait for pending threads to finish; the pinger and this thread stay. */
|
||||
htsthread_wait_n(nonjoinable_threads + 1);
|
||||
/* Wait for pending threads to finish */
|
||||
htsthread_wait_n(background_threads);
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
@@ -151,23 +151,6 @@ static hts_boolean is_embed_pair(const htspair_t *table, const char *tag,
|
||||
return HTS_FALSE;
|
||||
}
|
||||
|
||||
/* The engine's robots.txt verdict for (adr,fil). Under HTS_ROBOTS_SOMETIMES an
|
||||
explicit filter acceptance overrides the ban, which is why the filter outcome
|
||||
is an input; the sitemap fetcher asks the same question outside the wizard.
|
||||
*/
|
||||
hts_boolean hts_robots_forbids(httrackp *opt, const char *adr, const char *fil,
|
||||
hts_boolean filters_decided,
|
||||
hts_boolean filters_refused) {
|
||||
if (!opt->robots || opt->robotsptr == NULL)
|
||||
return HTS_FALSE;
|
||||
if (checkrobots((robots_wizard *) opt->robotsptr, adr, fil) != -1)
|
||||
return HTS_FALSE;
|
||||
if (filters_decided && !filters_refused &&
|
||||
opt->robots == HTS_ROBOTS_SOMETIMES)
|
||||
return HTS_FALSE;
|
||||
return HTS_TRUE;
|
||||
}
|
||||
|
||||
static int hts_acceptlink_(httrackp * opt, int ptr,
|
||||
const char *adr, const char *fil, const char *tag,
|
||||
const char *attribute, int *set_prio_to,
|
||||
@@ -593,26 +576,30 @@ static int hts_acceptlink_(httrackp * opt, int ptr,
|
||||
}
|
||||
}
|
||||
// vérifier robots.txt
|
||||
if (opt->robots && checkrobots(_ROBOTS, adr, fil) == -1) {
|
||||
if (opt->robots) {
|
||||
int r = checkrobots(_ROBOTS, adr, fil);
|
||||
|
||||
if (r == -1) { // interdiction
|
||||
#if DEBUG_ROBOTS
|
||||
printf("robots.txt forbidden: %s%s\n", adr, fil);
|
||||
printf("robots.txt forbidden: %s%s\n", adr, fil);
|
||||
#endif
|
||||
if (!hts_robots_forbids(opt, adr, fil,
|
||||
(!question && filters_answer) ? HTS_TRUE
|
||||
: HTS_FALSE,
|
||||
(forbidden_url == 1) ? HTS_TRUE : HTS_FALSE)) {
|
||||
if (!forbidden_url) {
|
||||
hts_log_print(
|
||||
opt, LOG_DEBUG,
|
||||
"Warning link followed against robots.txt: link %s at %s%s", l,
|
||||
adr, fil);
|
||||
// question résolue, par les filtres, et mode robot non strict
|
||||
if ((!question) && (filters_answer) &&
|
||||
(opt->robots == HTS_ROBOTS_SOMETIMES) && (forbidden_url != 1)) {
|
||||
r = 0; // annuler interdiction des robots
|
||||
if (!forbidden_url) {
|
||||
hts_log_print(opt, LOG_DEBUG,
|
||||
"Warning link followed against robots.txt: link %s at %s%s",
|
||||
l, adr, fil);
|
||||
}
|
||||
}
|
||||
if (r == -1) { // interdire
|
||||
forbidden_url = 1;
|
||||
question = 0;
|
||||
hts_log_print(opt, LOG_DEBUG,
|
||||
"(robots.txt) forbidden link: link %s at %s%s", l, adr,
|
||||
fil);
|
||||
}
|
||||
} else {
|
||||
forbidden_url = 1;
|
||||
question = 0;
|
||||
hts_log_print(opt, LOG_DEBUG,
|
||||
"(robots.txt) forbidden link: link %s at %s%s", l, adr,
|
||||
fil);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -49,13 +49,6 @@ typedef struct httrackp httrackp;
|
||||
typedef struct lien_url lien_url;
|
||||
#endif
|
||||
|
||||
/* The engine's robots.txt verdict for (adr,fil): HTS_TRUE when the fetch is
|
||||
forbidden. `filters_decided`/`filters_refused` carry the filter outcome,
|
||||
which overrides a ban under -s1 (HTS_ROBOTS_SOMETIMES). */
|
||||
hts_boolean hts_robots_forbids(httrackp *opt, const char *adr, const char *fil,
|
||||
hts_boolean filters_decided,
|
||||
hts_boolean filters_refused);
|
||||
|
||||
int hts_acceptlink(httrackp * opt, int ptr,
|
||||
const char *adr, const char *fil,
|
||||
const char *tag, const char *attribute,
|
||||
|
||||
@@ -40,6 +40,7 @@ Please visit our Website: http://www.httrack.com
|
||||
#include "htscodec.h"
|
||||
#include "htszlib.h"
|
||||
|
||||
#if HTS_USEZLIB
|
||||
/* zlib */
|
||||
/*
|
||||
#include <zlib.h>
|
||||
@@ -273,3 +274,4 @@ const char *hts_get_zerror(int err) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -306,8 +306,8 @@ int main(int argc, char **argv) {
|
||||
fprintf(stderr, "* %s\n", hts_errmsg(opt));
|
||||
}
|
||||
global_opt = NULL;
|
||||
htsthread_wait(); /* pending threads still read opt */
|
||||
hts_free_opt(opt);
|
||||
htsthread_wait(); /* wait for pending threads */
|
||||
hts_uninit();
|
||||
|
||||
#ifdef _WIN32
|
||||
|
||||
@@ -131,7 +131,6 @@
|
||||
<ClCompile Include="htsproxy.c" />
|
||||
<ClCompile Include="htsrobots.c" />
|
||||
<ClCompile Include="htsselftest.c" />
|
||||
<ClCompile Include="htssinglefile.c" />
|
||||
<ClCompile Include="htssniff.c" />
|
||||
<ClCompile Include="htsthread.c" />
|
||||
<ClCompile Include="htstools.c" />
|
||||
@@ -140,7 +139,6 @@
|
||||
<ClCompile Include="htszlib.c" />
|
||||
<ClCompile Include="htswarc.c" />
|
||||
<ClCompile Include="htschanges.c" />
|
||||
<ClCompile Include="htssitemap.c" />
|
||||
<ClCompile Include="md5.c" />
|
||||
<ClCompile Include="minizip\ioapi.c" />
|
||||
<ClCompile Include="minizip\iowin32.c" />
|
||||
|
||||
@@ -1,7 +0,0 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# an empty current path underflowed htsAddLink's codebase walk (#730).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
httrack -O /dev/null -#test=addlink | grep -q "addlink self-test OK"
|
||||
@@ -1,12 +0,0 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# structcheck(): the path-length guard, and the <name>.txt rename that once
|
||||
# built its target with an unbounded sprintf (#745).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
dir=$(mktemp -d)
|
||||
trap 'rm -rf "$dir"' EXIT
|
||||
|
||||
httrack -O /dev/null -#test=structcheck "$dir" |
|
||||
grep -q "structcheck self-test OK"
|
||||
@@ -1,28 +0,0 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
tmpdir=$(mktemp -d "${TMPDIR:-/tmp}/httrack_threadwait_st.XXXXXX") || exit 1
|
||||
trap 'rm -rf "$tmpdir"' EXIT HUP INT QUIT PIPE TERM
|
||||
|
||||
# No pipe into grep: SIGPIPE would mask a failing exit status.
|
||||
expect_ok() {
|
||||
local label="$1" out
|
||||
shift
|
||||
out=$("$@" 2>&1) || {
|
||||
echo "FAIL: ${label} exited non-zero: ${out}"
|
||||
exit 1
|
||||
}
|
||||
case "$out" in
|
||||
*"${label}: OK"*) ;;
|
||||
*)
|
||||
echo "FAIL: ${out}"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
# A thread is outstanding from the moment hts_newthread() returns, so a wait
|
||||
# that follows the spawn joins it, and wait_n(n) still leaves n behind (#747).
|
||||
expect_ok "threadwait self-test" httrack -O "${tmpdir}/o1" -#test=threadwait
|
||||
@@ -1,10 +0,0 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# Sitemap parser self-test: <loc> extraction, entity decoding, URL and length
|
||||
# rejections, the URL cap, gzip framing and robots.txt Sitemap: records.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
out=$(httrack -O /dev/null '-#test=sitemap')
|
||||
echo "$out"
|
||||
test "$out" = "sitemap self-test OK"
|
||||
@@ -18,7 +18,6 @@ set -euo pipefail
|
||||
bash "$top_srcdir/tests/local-crawl.sh" \
|
||||
--rerun-args '--update -M400000' \
|
||||
--log-found 'More than 400000 bytes have been transferred.. giving up' \
|
||||
--log-not-found 'could not restore' \
|
||||
--found bigtrunc/slow.bin \
|
||||
--file-min-bytes bigtrunc/slow.bin 655360 \
|
||||
--file-min-bytes bigtrunc/fast.bin 655360 \
|
||||
|
||||
@@ -79,29 +79,6 @@ done
|
||||
# 65535 is a valid port the old "< 65535" bound refused
|
||||
web_accepted 65535
|
||||
|
||||
# --- htsserver returns from main() instead of blocking at exit -------------
|
||||
# The exit wait must exclude exactly the threads that never return (#753).
|
||||
# $tmp has no lang.def, so the server fails right after announcing and main()
|
||||
# reaches the wait on its own; rc, not EXITED, carries the verdict.
|
||||
|
||||
web_exits() {
|
||||
local rc=0
|
||||
|
||||
: >"$tmp/exit.log"
|
||||
run_with_timeout 30 htsserver "$tmp" "$@" >"$tmp/exit.log" 2>&1 || rc=$?
|
||||
grep -q "^EXITED" "$tmp/exit.log" ||
|
||||
! echo "FAIL: #753: htsserver ${*:-(no options)} never finished serving" || exit 1
|
||||
# exactly 1, not merely "not 124": a crash or an assertf abort also escapes
|
||||
# the wait, and would otherwise read as a pass
|
||||
test "$rc" -eq 1 ||
|
||||
! echo "FAIL: #753: htsserver ${*:-(no options)} exited $rc, want 1" || exit 1
|
||||
}
|
||||
|
||||
# no pinger: the excluded count went negative
|
||||
web_exits
|
||||
# a pinger, which never returns and so must stay excluded
|
||||
web_exits --ppid $$
|
||||
|
||||
# --- proxytrack <proxy-addr:port> <ICP-addr:port> --------------------------
|
||||
# A bad argument falls through to the usage screen; it had no range check at
|
||||
# all, so 65616 quietly listened on port 80. A valid one binds and blocks.
|
||||
|
||||
@@ -5,16 +5,11 @@
|
||||
# real gate: STORE-mode entries, the fixed layout, recomputed sha256 digests and
|
||||
# the datapackage-digest chain (plus py-wacz/pywb when importable). The crawl
|
||||
# skips cleanly on a build without OpenSSL (no conformant SHA-256 -> no package).
|
||||
#
|
||||
# The cacheless second pass repackages over the .wacz the first one left: the
|
||||
# clobber, not the happy rename, is what breaks when a platform refuses to
|
||||
# rename onto an existing file (#726).
|
||||
|
||||
set -eu
|
||||
|
||||
: "${top_srcdir:=..}"
|
||||
|
||||
bash "$top_srcdir/tests/local-crawl.sh" --errors 0 --wacz-validate \
|
||||
--rerun-args '-C0' --log-not-found 'could not finalize' \
|
||||
--found 'mini304/index.html' --found 'mini304/page.html' \
|
||||
httrack 'BASEURL/mini304/index.html' --warc-file warc-out --wacz
|
||||
|
||||
@@ -149,7 +149,6 @@ BOXES = [
|
||||
("keepqueryorder", "KeepQueryOrder", "--keep-query-order", None),
|
||||
("toler", "TolerantRequests", "--tolerant", None),
|
||||
("http10", "HTTP10", "--http-10", None),
|
||||
("sitemap", "Sitemap", "--sitemap", None),
|
||||
("warc", "Warc", "--warc", None),
|
||||
("changes", "Changes", "--changes", None),
|
||||
("norecatch", "NoRecatch", "--do-not-recatch", None),
|
||||
@@ -157,7 +156,6 @@ BOXES = [
|
||||
("index", "Index", None, None),
|
||||
("index2", "WordIndex", "--search-index", None),
|
||||
("ftpprox", "UseHTTPProxyForFTP", "--httpproxy-ftp", None),
|
||||
("singlefile", "SingleFile", "--single-file", None),
|
||||
]
|
||||
|
||||
for field, key, when_set, when_clear in BOXES:
|
||||
|
||||
@@ -94,7 +94,10 @@ expect_counts_match() {
|
||||
echo "OK"
|
||||
}
|
||||
lines() { printf '%s\n' "$@" | sort; }
|
||||
digest() { cksum <"$1" | tr -d '[:space:]'; }
|
||||
# A connection killed before the status line surfaces differently per platform
|
||||
# (macOS truncates the file, #748; Linux leaves it), so reset.bin is asserted on
|
||||
# its own and kept out of the exact lists.
|
||||
listed_but_reset() { listed "$1" | grep -v '/reset\.bin$' | sort; }
|
||||
|
||||
# --- pass 1: nothing to compare against --------------------------------------
|
||||
httrack "${common[@]}" -O "$out" --purge-old=0 "${base}/changes/index.html" \
|
||||
@@ -117,7 +120,6 @@ grep -aq "first crawl" "${out}/hts-log.txt" || {
|
||||
}
|
||||
|
||||
host="127.0.0.1_${port}"
|
||||
resetdigest=$(digest "${out}/${host}/changes/reset.bin")
|
||||
# A leftover from some earlier failed attempt, at a name pass 2 fetches for the
|
||||
# first time: on disk, but never part of the previous mirror, so it is new.
|
||||
printf 'leftover junk' >"${out}/${host}/changes/fresh.html"
|
||||
@@ -142,19 +144,21 @@ expect "gone is doomed.html and the old redirect target" \
|
||||
expect "changed is exactly the four moved resources" \
|
||||
"$(lines "${host}/changes/coded.bin" "${host}/changes/index.html" \
|
||||
"${host}/changes/moved.bin" "${host}/changes/moved.html")" \
|
||||
"$(listed changed | sort)"
|
||||
"$(listed_but_reset changed)"
|
||||
# Re-served byte for byte, 200, no Last-Modified, no ETag. flaky.bin gets there
|
||||
# through a failed transfer and a retry, so its file is notified twice;
|
||||
# redirtarget.html arrives behind a 302. sized.html has a byte-identical
|
||||
# payload, but the rewritten link to the renamed redirect target makes the file
|
||||
# on disk change length. reset.bin never completes a transfer (#746), so its
|
||||
# previous copy stands.
|
||||
expect "unchanged is exactly the seven stable resources" \
|
||||
# on disk change length.
|
||||
expect "unchanged is exactly the six stable resources" \
|
||||
"$(lines "${host}/changes/codedstable.bin" "${host}/changes/flaky.bin" \
|
||||
"${host}/changes/redirtarget.html" "${host}/changes/reset.bin" \
|
||||
"${host}/changes/sized.html" "${host}/changes/stable.bin" \
|
||||
"${host}/changes/stable.html")" \
|
||||
"$(listed unchanged | sort)"
|
||||
"${host}/changes/redirtarget.html" "${host}/changes/sized.html" \
|
||||
"${host}/changes/stable.bin" "${host}/changes/stable.html")" \
|
||||
"$(listed_but_reset unchanged)"
|
||||
# reset.bin never completes a transfer, so it drops out of new.lst with its
|
||||
# mirrored copy still there. Whatever else it is, it is not a deletion.
|
||||
expect "a failed re-fetch is never reported gone" "" \
|
||||
"$(listed gone | grep '/reset\.bin$' || true)"
|
||||
expect_counts_match "pass 2 counts match its lists"
|
||||
|
||||
# Every mirrored file appears once and only once, across all four buckets.
|
||||
@@ -188,15 +192,19 @@ test -f "${out}/${host}/changes/doomed.html" || {
|
||||
exit 1
|
||||
}
|
||||
echo "OK"
|
||||
expect "a failed transfer keeps its bytes" "$resetdigest" \
|
||||
"$(digest "${out}/${host}/changes/reset.bin")"
|
||||
printf '[a failed transfer keeps its file] ..\t'
|
||||
test -f "${out}/${host}/changes/reset.bin" || {
|
||||
echo "FAIL: reset.bin lost its previous copy"
|
||||
exit 1
|
||||
}
|
||||
echo "OK"
|
||||
|
||||
# --- pass 3: the same run with purging on ------------------------------------
|
||||
httrack "${common[@]}" -O "$out" --update "${base}/changes/index.html" \
|
||||
>"${tmpdir}/log3" 2>&1
|
||||
expect "the report says files were purged" "true" "$(field purged)"
|
||||
expect "gone is the page that only pass 2 linked" \
|
||||
"${host}/changes/transient.html" "$(listed gone | sort)"
|
||||
"${host}/changes/transient.html" "$(listed_but_reset gone)"
|
||||
printf '[a purged file is off disk] ..\t'
|
||||
test ! -f "${out}/${host}/changes/transient.html" || {
|
||||
echo "FAIL: transient.html survived --purge-old"
|
||||
|
||||
@@ -1,379 +0,0 @@
|
||||
#!/bin/bash
|
||||
# Issue #713: --single-file inlines each page's assets as data: URIs. End to
|
||||
# end over a real crawl, decoding the payloads against what the server served.
|
||||
set -e
|
||||
|
||||
: "${top_srcdir:=..}"
|
||||
testdir=$(cd "$(dirname "$0")" && pwd)
|
||||
# shellcheck source=tests/testlib.sh
|
||||
. "${testdir}/testlib.sh"
|
||||
server=$(nativepath "${testdir}/local-server.py")
|
||||
|
||||
python=$(find_python) || ! echo "python3 not found; skipping" >&2 || exit 77
|
||||
|
||||
tmpdir=$(mktemp -d "${TMPDIR:-/tmp}/httrack_713.XXXXXX") || exit 1
|
||||
serverpid=
|
||||
htssrv=
|
||||
cleanup() {
|
||||
stop_server "$serverpid"
|
||||
# htsserver keeps SIGTERM ignored across its exec, so only -9 reaps it.
|
||||
test -z "$htssrv" || kill -9 "$htssrv" 2>/dev/null || true
|
||||
wait "$htssrv" 2>/dev/null || true # absorb bash's async "Killed" notice
|
||||
htssrv=
|
||||
rm -rf "$tmpdir"
|
||||
}
|
||||
trap cleanup EXIT HUP INT QUIT PIPE TERM
|
||||
|
||||
# --- the site ---------------------------------------------------------------
|
||||
doc="${tmpdir}/doc"
|
||||
mkdir -p "$doc"
|
||||
cat >"${doc}/index.html" <<'EOF'
|
||||
<html><head>
|
||||
<link rel="stylesheet" href="style.css">
|
||||
<link rel="canonical" href="other.html">
|
||||
</head><body>
|
||||
<img src="pixel.svg" alt="p">
|
||||
<img src="big.svg" alt="b">
|
||||
<img src="pixel.svg" data-src="lazy.svg" alt="l">
|
||||
<script src="app.js"></script>
|
||||
<a href="other.html">other</a>
|
||||
</body></html>
|
||||
EOF
|
||||
cat >"${doc}/other.html" <<'EOF'
|
||||
<html><body><a href="index.html">back</a></body></html>
|
||||
EOF
|
||||
cat >"${doc}/style.css" <<'EOF'
|
||||
@import "nested.css";
|
||||
body { background: url(pixel.svg); }
|
||||
EOF
|
||||
cat >"${doc}/nested.css" <<'EOF'
|
||||
div { background: url(pixel.svg); }
|
||||
EOF
|
||||
printf 'var app = 1;\n' >"${doc}/app.js"
|
||||
printf '<svg xmlns="http://www.w3.org/2000/svg"><rect width="1" height="1"/></svg>\n' \
|
||||
>"${doc}/pixel.svg"
|
||||
printf '<svg xmlns="http://www.w3.org/2000/svg"><circle r="1"/></svg>\n' \
|
||||
>"${doc}/lazy.svg"
|
||||
{
|
||||
printf '<svg xmlns="http://www.w3.org/2000/svg"><!--'
|
||||
head -c 20000 /dev/zero | tr '\0' 'x'
|
||||
printf '%s\n' '--><rect width="1" height="1"/></svg>'
|
||||
} >"${doc}/big.svg"
|
||||
|
||||
# --- server -----------------------------------------------------------------
|
||||
serverlog="${tmpdir}/server.log"
|
||||
: >"$serverlog"
|
||||
"$python" "$server" --root "$(nativepath "$doc")" >"$serverlog" 2>&1 &
|
||||
serverpid=$!
|
||||
port=
|
||||
for _ in $(seq 1 50); do
|
||||
line=$(head -n1 "$serverlog" 2>/dev/null)
|
||||
if test "${line%% *}" == "PORT"; then
|
||||
port="${line#PORT }"
|
||||
break
|
||||
fi
|
||||
kill -0 "$serverpid" 2>/dev/null || {
|
||||
echo "server exited early: $(cat "$serverlog")"
|
||||
exit 1
|
||||
}
|
||||
sleep 0.1
|
||||
done
|
||||
test -n "$port" || {
|
||||
echo "could not discover server port"
|
||||
exit 1
|
||||
}
|
||||
base="http://127.0.0.1:${port}"
|
||||
|
||||
which httrack >/dev/null || {
|
||||
echo "could not find httrack"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# --- crawl ------------------------------------------------------------------
|
||||
# The cap sits between pixel.svg and big.svg, so one inlines and one must not.
|
||||
out="${tmpdir}/crawl"
|
||||
mkdir "$out"
|
||||
common=(-O "$out" --quiet --disable-security-limits --robots=0 --timeout=30 --retries=0)
|
||||
httrack "${common[@]}" --single-file --single-file-max-size 4096 \
|
||||
"${base}/index.html" >"${tmpdir}/log1" 2>&1
|
||||
|
||||
page="${out}/127.0.0.1_${port}/index.html"
|
||||
test -f "$page" || {
|
||||
echo "FAIL: no mirrored page at $page"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# --- what must be inlined, byte for byte ------------------------------------
|
||||
# The decoder is python, not httrack, so a broken encoder cannot agree with
|
||||
# itself here.
|
||||
extract() {
|
||||
"$python" - "$1" "$2" "$3" <<'PY'
|
||||
import base64, re, sys
|
||||
html = open(sys.argv[1], "rb").read().decode("utf-8", "replace")
|
||||
m = re.search(r'data:%s;base64,([A-Za-z0-9+/=]*)' % re.escape(sys.argv[2]), html)
|
||||
if m is None:
|
||||
sys.exit(1)
|
||||
open(sys.argv[3], "wb").write(base64.b64decode(m.group(1)))
|
||||
PY
|
||||
}
|
||||
|
||||
printf '[image inlined and identical] ..\t'
|
||||
extract "$page" "image/svg+xml" "${tmpdir}/got.svg" || {
|
||||
echo "FAIL: no inlined image"
|
||||
exit 1
|
||||
}
|
||||
cmp -s "${doc}/pixel.svg" "${tmpdir}/got.svg" || {
|
||||
echo "FAIL: inlined image differs from the served file"
|
||||
exit 1
|
||||
}
|
||||
echo OK
|
||||
|
||||
printf '[script inlined and identical] ..\t'
|
||||
extract "$page" "application/x-javascript" "${tmpdir}/got.js" || {
|
||||
echo "FAIL: no inlined script"
|
||||
exit 1
|
||||
}
|
||||
cmp -s "${doc}/app.js" "${tmpdir}/got.js" || {
|
||||
echo "FAIL: inlined script differs from the served file"
|
||||
exit 1
|
||||
}
|
||||
echo OK
|
||||
|
||||
printf '[lazy-loaded image inlined] ..\t'
|
||||
# src is the placeholder a real page ships; the asset rides data-src.
|
||||
grep -q 'data-src="data:image/svg+xml;base64,' "$page" || {
|
||||
echo "FAIL: data-src was not inlined"
|
||||
exit 1
|
||||
}
|
||||
if grep -q 'data-src="lazy.svg"' "$page"; then
|
||||
echo "FAIL: data-src kept its link"
|
||||
exit 1
|
||||
fi
|
||||
test -f "${out}/127.0.0.1_${port}/lazy.svg" || {
|
||||
echo "FAIL: the lazy image was never downloaded, so nothing was proven"
|
||||
exit 1
|
||||
}
|
||||
echo OK
|
||||
|
||||
printf '[stylesheet inlined, recursively] ..\t'
|
||||
extract "$page" "text/css" "${tmpdir}/got.css" || {
|
||||
echo "FAIL: no inlined stylesheet"
|
||||
exit 1
|
||||
}
|
||||
grep -q 'url(data:image/svg+xml;base64,' "${tmpdir}/got.css" || {
|
||||
echo "FAIL: url() inside the inlined stylesheet was not inlined"
|
||||
exit 1
|
||||
}
|
||||
extract "${tmpdir}/got.css" "text/css" "${tmpdir}/got2.css" || {
|
||||
echo "FAIL: @import was not inlined"
|
||||
exit 1
|
||||
}
|
||||
grep -q 'url(data:image/svg+xml;base64,' "${tmpdir}/got2.css" || {
|
||||
echo "FAIL: url() inside the @import'ed stylesheet was not inlined"
|
||||
exit 1
|
||||
}
|
||||
echo OK
|
||||
|
||||
# --- what must keep its link ------------------------------------------------
|
||||
printf '[page links stay relative] ..\t'
|
||||
grep -q 'href="other.html"' "$page" || {
|
||||
echo "FAIL: the link to another page was not left relative"
|
||||
exit 1
|
||||
}
|
||||
test -f "${out}/127.0.0.1_${port}/other.html" || {
|
||||
echo "FAIL: the linked page is missing from the mirror"
|
||||
exit 1
|
||||
}
|
||||
echo OK
|
||||
|
||||
printf '[over-cap asset stays a link] ..\t'
|
||||
grep -q 'src="big.svg"' "$page" || {
|
||||
echo "FAIL: the over-cap asset did not keep its link"
|
||||
exit 1
|
||||
}
|
||||
test -f "${out}/127.0.0.1_${port}/big.svg" || {
|
||||
echo "FAIL: the over-cap asset is missing from the mirror"
|
||||
exit 1
|
||||
}
|
||||
grep -q 'over the 4096-byte cap' "${out}/hts-log.txt" || {
|
||||
echo "FAIL: the over-cap asset was not reported in the log"
|
||||
exit 1
|
||||
}
|
||||
echo OK
|
||||
|
||||
printf '[the mirror still works] ..\t'
|
||||
if grep -q 'href="style.css"' "$page"; then
|
||||
echo "FAIL: the stylesheet link survived"
|
||||
exit 1
|
||||
fi
|
||||
test -f "${out}/127.0.0.1_${port}/style.css" || {
|
||||
echo "FAIL: the mirror lost the standalone stylesheet"
|
||||
exit 1
|
||||
}
|
||||
echo OK
|
||||
|
||||
# --- a second run must not re-encode ----------------------------------------
|
||||
printf '[idempotent under --update] ..\t'
|
||||
# HTTrack stamps the time into its own footer comment, so compare without it.
|
||||
grep -v 'Mirrored from' "$page" >"${tmpdir}/page1.html"
|
||||
httrack "${common[@]}" --single-file --single-file-max-size 4096 --update \
|
||||
"${base}/index.html" >"${tmpdir}/log2" 2>&1
|
||||
grep -v 'Mirrored from' "$page" >"${tmpdir}/page2.html"
|
||||
cmp -s "${tmpdir}/page1.html" "${tmpdir}/page2.html" || {
|
||||
echo "FAIL: the second run changed the page"
|
||||
exit 1
|
||||
}
|
||||
# A double-encoded payload would decode to another data: URI, not to the SVG.
|
||||
extract "$page" "image/svg+xml" "${tmpdir}/got2.svg" || {
|
||||
echo "FAIL: no inlined image after the second run"
|
||||
exit 1
|
||||
}
|
||||
cmp -s "${doc}/pixel.svg" "${tmpdir}/got2.svg" || {
|
||||
echo "FAIL: the second run re-encoded the inlined image"
|
||||
exit 1
|
||||
}
|
||||
echo OK
|
||||
|
||||
printf '[-%%Z0 and a bad cap] ..\t'
|
||||
# -%Z0 turns it back off, so the page must come out exactly as the crawl left
|
||||
# it; a non-numeric cap is rejected and the built-in default used instead.
|
||||
off="${tmpdir}/off"
|
||||
mkdir "$off"
|
||||
httrack -O "$off" --quiet --disable-security-limits --robots=0 --timeout=30 \
|
||||
--retries=0 -%Z0 "${base}/index.html" >"${tmpdir}/log3" 2>&1
|
||||
offpage="${off}/127.0.0.1_${port}/index.html"
|
||||
if grep -q 'base64,' "$offpage"; then
|
||||
echo "FAIL: -%Z0 still inlined"
|
||||
exit 1
|
||||
fi
|
||||
bad="${tmpdir}/bad"
|
||||
mkdir "$bad"
|
||||
httrack -O "$bad" --quiet --disable-security-limits --robots=0 --timeout=30 \
|
||||
--retries=0 --single-file-max-size notanumber "${base}/index.html" \
|
||||
>"${tmpdir}/log4" 2>&1
|
||||
badpage="${bad}/127.0.0.1_${port}/index.html"
|
||||
# The default cap is 10 MB, so everything here inlines, big.svg included.
|
||||
grep -q 'data:image/svg+xml;base64,' "$badpage" || {
|
||||
echo "FAIL: a rejected cap did not fall back to the default"
|
||||
exit 1
|
||||
}
|
||||
if grep -q 'src="big.svg"' "$badpage"; then
|
||||
echo "FAIL: the default cap did not cover big.svg"
|
||||
exit 1
|
||||
fi
|
||||
echo OK
|
||||
|
||||
printf '[the rewritten page keeps the mirror file mode] ..\t'
|
||||
# The rewrite spools to a temp and renames, bypassing the engine's chmod.
|
||||
if is_windows; then
|
||||
echo "SKIP (no POSIX file modes)"
|
||||
else
|
||||
umaskdir="${tmpdir}/umask"
|
||||
mkdir "$umaskdir"
|
||||
(
|
||||
umask 077
|
||||
httrack -O "$umaskdir" --quiet --disable-security-limits --robots=0 \
|
||||
--timeout=30 --retries=0 --single-file "${base}/index.html" \
|
||||
>"${tmpdir}/log5" 2>&1
|
||||
)
|
||||
umaskroot="${umaskdir}/127.0.0.1_${port}"
|
||||
mode() { "$python" -c 'import os,sys
|
||||
print("%04o" % (os.stat(sys.argv[1]).st_mode & 0o777))' "$1"; }
|
||||
pagemode=$(mode "${umaskroot}/index.html")
|
||||
assetmode=$(mode "${umaskroot}/big.svg")
|
||||
test "$pagemode" = "$assetmode" || {
|
||||
echo "FAIL: page is $pagemode, the asset beside it is $assetmode"
|
||||
exit 1
|
||||
}
|
||||
echo OK
|
||||
fi
|
||||
|
||||
# --- the rewriter in isolation, over a hand-built mirror --------------------
|
||||
# After the crawl assertions, so a failure here cannot hide them: the two cover
|
||||
# different ground (this one reaches media, anchors and the mirror-root clamp).
|
||||
printf '[engine self-test] ..\t'
|
||||
st=$(httrack -O /dev/null "-#test=singlefile" "${tmpdir}/st/" 2>/dev/null)
|
||||
case "$st" in
|
||||
*": OK") echo OK ;;
|
||||
*)
|
||||
echo "FAIL: $st"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
# --- the web GUI must emit the same options ---------------------------------
|
||||
distdir=$(cd "${top_srcdir}" && pwd)
|
||||
printf '[project reload maps the keys back] ..\t'
|
||||
grep -q 'do:copy:SingleFile:singlefile' "${distdir}/html/server/step2.html" || {
|
||||
echo "FAIL: step2.html does not restore SingleFile"
|
||||
exit 1
|
||||
}
|
||||
grep -q 'do:copy:SingleFileMaxSize:singlefilemax' "${distdir}/html/server/step2.html" || {
|
||||
echo "FAIL: step2.html does not restore SingleFileMaxSize"
|
||||
exit 1
|
||||
}
|
||||
echo OK
|
||||
|
||||
printf '[webhttrack renders the options] ..\t'
|
||||
# The Windows job builds httrack.exe only and reaches this file through its
|
||||
# *_local-*.test glob, so report a skip there rather than a pass.
|
||||
if ! command -v htsserver >/dev/null; then
|
||||
echo "SKIP (no htsserver in PATH)"
|
||||
exit 77
|
||||
fi
|
||||
sport=$("$python" -c 'import socket
|
||||
s = socket.socket()
|
||||
s.bind(("127.0.0.1", 0))
|
||||
print(s.getsockname()[1])
|
||||
s.close()')
|
||||
srvlog="${tmpdir}/htsserver.log"
|
||||
(
|
||||
trap '' TERM TTOU
|
||||
export HOME="${tmpdir}/home"
|
||||
mkdir -p "$HOME"
|
||||
exec htsserver "${distdir}/" --port "${sport}" >"${srvlog}" 2>&1
|
||||
) &
|
||||
htssrv=$!
|
||||
url=
|
||||
for _ in $(seq 1 40); do
|
||||
url=$(sed -n 's/^URL=//p' "$srvlog" 2>/dev/null) && test -n "$url" && break
|
||||
kill -0 "$htssrv" 2>/dev/null || break
|
||||
sleep 0.25
|
||||
done
|
||||
test -n "${url:-}" || {
|
||||
echo "FAIL: htsserver did not start: $(cat "$srvlog")"
|
||||
exit 1
|
||||
}
|
||||
"$python" - "$url" <<'PY' || exit 1
|
||||
import re, sys, urllib.parse, urllib.request
|
||||
|
||||
url = sys.argv[1]
|
||||
form = urllib.request.urlopen(url + "server/index.html", timeout=20).read()
|
||||
m = re.search(rb'name="sid" value="([0-9a-f]+)"', form)
|
||||
if not m:
|
||||
sys.exit("FAIL: no session id in server/index.html")
|
||||
fields = [("sid", m.group(1).decode()), ("singlefile", "on"),
|
||||
("singlefilemax", "5000")]
|
||||
body = "&".join("%s=%s" % (k, urllib.parse.quote(v)) for k, v in fields)
|
||||
req = urllib.request.Request(url + "server/step4.html",
|
||||
data=body.encode("latin-1"), method="POST")
|
||||
page = urllib.request.urlopen(req, timeout=20).read().decode("latin-1")
|
||||
|
||||
|
||||
def textarea(name):
|
||||
m = re.search(r'<textarea name="%s".*?>(.*?)</textarea>' % name, page, re.S)
|
||||
if m is None:
|
||||
sys.exit("FAIL: no %s textarea in the rendered step4.html" % name)
|
||||
return m.group(1)
|
||||
|
||||
|
||||
cmd = textarea("command")
|
||||
for want in ("--single-file", "--single-file-max-size=5000"):
|
||||
if want not in cmd:
|
||||
sys.exit("FAIL: %s missing from the command line\n%s" % (want, cmd))
|
||||
ini = textarea("winprofile").replace("\r\n", "\n")
|
||||
for want in ("\nSingleFile=1\n", "\nSingleFileMaxSize=5000\n"):
|
||||
if want not in ini:
|
||||
sys.exit("FAIL: %r missing from winprofile.ini\n%s" % (want, ini))
|
||||
PY
|
||||
echo OK
|
||||
@@ -1,134 +0,0 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# --sitemap seeds the crawl from robots.txt -> sitemapindex -> gzipped urlset.
|
||||
# start.html links to nothing, so orphan*.html can only arrive through the
|
||||
# sitemap; deep1.html proves the seeds keep a full depth budget under -r2, and
|
||||
# the off-host page and child sitemap must both be refused.
|
||||
|
||||
set -eu
|
||||
|
||||
: "${top_srcdir:=..}"
|
||||
|
||||
crawl() { bash "$top_srcdir/tests/local-crawl.sh" "$@"; }
|
||||
|
||||
# robots.txt Sitemap: -> index -> .xml.gz, seeds behaving like -r2 seeds. The
|
||||
# log assertions pin which route was taken: the two documents are served from
|
||||
# both /sitemapdir/index.xml and the well-known /sitemap.xml, so "some sitemap
|
||||
# was read" would pass either way. --not-found pins ingestion-only: the sitemap
|
||||
# documents feed the crawl but never land in the mirror.
|
||||
# --rerun also walks the update path over a mirror holding sitemap seeds.
|
||||
crawl --errors 0 --rerun \
|
||||
--found 'sitemapdir/orphan1.html' \
|
||||
--found 'sitemapdir/orphan2.html' \
|
||||
--found 'sitemapdir/deep1.html' \
|
||||
--not-found 'sitemapdir/index.xml' \
|
||||
--not-found 'sitemapdir/pages.xml.gz' \
|
||||
--log-found '2 of 3 URL\(s\) added from 127\.0\.0\.1:[0-9]+/sitemapdir/pages\.xml\.gz' \
|
||||
--log-found '1 of 2 child sitemap\(s\) listed by 127\.0\.0\.1:[0-9]+/sitemapdir/index\.xml' \
|
||||
--log-found 'ignoring off-host child sitemap' \
|
||||
--log-not-found '/sitemap\.xml' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2
|
||||
|
||||
# Negative control: without the option nothing but the start page is reached.
|
||||
crawl --errors 0 \
|
||||
--found 'sitemapdir/start.html' \
|
||||
--not-found 'sitemapdir/orphan1.html' \
|
||||
--not-found 'sitemapdir/orphan2.html' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' -r2
|
||||
|
||||
# Sitemap URLs are not a filter bypass: a -*orphan2* rule still rejects one.
|
||||
crawl --errors 0 \
|
||||
--found 'sitemapdir/orphan1.html' \
|
||||
--not-found 'sitemapdir/orphan2.html' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2 '-*orphan2*'
|
||||
|
||||
# A robots.txt naming no sitemap falls back to the well-known /sitemap.xml.
|
||||
# The test server drops its Sitemap: record for this User-Agent.
|
||||
crawl --errors 0 \
|
||||
--found 'sitemapdir/orphan1.html' \
|
||||
--log-found 'listed by 127\.0\.0\.1:[0-9]+/sitemap\.xml' \
|
||||
--log-not-found 'sitemapdir/index\.xml' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2 -F 'nositemap-agent'
|
||||
|
||||
# --sitemap-url alone reads the named document and probes nothing.
|
||||
crawl --errors 0 \
|
||||
--found 'sitemapdir/orphan1.html' \
|
||||
--found 'sitemapdir/orphan2.html' \
|
||||
--log-not-found 'sitemapdir/index\.xml' \
|
||||
--log-not-found '/sitemap\.xml' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' -r2 \
|
||||
--sitemap-url 'BASEURL/sitemapdir/pages.xml.gz'
|
||||
|
||||
# The two options are additive: an unfetchable --sitemap-url is parsed as an
|
||||
# empty document and does not stop --sitemap from seeding the crawl.
|
||||
crawl --errors-content 1 \
|
||||
--found 'sitemapdir/orphan1.html' \
|
||||
--log-found 'Sitemap: 0 of 0 URL\(s\) added' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2 \
|
||||
--sitemap-url 'BASEURL/sitemapdir/missing.xml'
|
||||
|
||||
# The sitemapindex nesting cap stops the chain before its deepest urlset. The
|
||||
# positive control is the page one level inside the cap: without it, a cap
|
||||
# mutated to 0 (nothing followed) would pass this just as well.
|
||||
crawl --errors 0 \
|
||||
--found 'sitemapdir/cap3.html' \
|
||||
--not-found 'sitemapdir/cap4.html' \
|
||||
--log-found 'Sitemap: cap reached' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' -r2 \
|
||||
--sitemap-url 'BASEURL/sitemapdir/chain0.xml'
|
||||
|
||||
# A site cannot widen a subtree crawl by putting its sitemap at the root: the
|
||||
# page below the start directory is seeded, the one above it is refused.
|
||||
crawl --errors 0 \
|
||||
--found 'deep/dir/below.html' \
|
||||
--not-found 'elsewhere/updir.html' \
|
||||
httrack 'BASEURL/deep/dir/start.html' --sitemap -r3 -F 'scopesitemap-agent'
|
||||
|
||||
# How far a sitemap fetch is gated depends on who asked for it, so the three
|
||||
# cases below have to differ: one "robots is honoured" assertion would hide a
|
||||
# regression in either direction. The harness disables robots, hence -s2 here.
|
||||
|
||||
# We guessed /sitemap.xml, so a Disallow on it wins.
|
||||
crawl --errors 0 \
|
||||
--not-found 'sitemapdir/orphan1.html' \
|
||||
--log-found 'Sitemap: robots.txt forbids' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2 -s2 \
|
||||
-F 'denysitemap-agent'
|
||||
|
||||
# The site declared it through a Sitemap: line, which the same Disallow does
|
||||
# not retract.
|
||||
crawl --errors 0 \
|
||||
--found 'sitemapdir/orphan1.html' \
|
||||
--log-not-found 'Sitemap: robots.txt forbids' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' --sitemap -r2 -s2 \
|
||||
-F 'denydeclared-agent'
|
||||
|
||||
# The user named it, which is the same intent as a start URL: not refused
|
||||
# either, even though robots.txt forbids that exact path.
|
||||
crawl --errors 0 \
|
||||
--found 'sitemapdir/orphan1.html' \
|
||||
--log-not-found 'Sitemap: robots.txt forbids' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' -r2 -s2 -F 'denysitemap-agent' \
|
||||
--sitemap-url 'BASEURL/sitemap.xml'
|
||||
|
||||
# A child sitemap is a fetch like any other: a filter rejecting it stops the
|
||||
# whole subtree, so the page only it lists is never reached.
|
||||
crawl --errors 0 \
|
||||
--not-found 'sitemapdir/gated.html' \
|
||||
--log-found 'Sitemap: filter rule #[0-9]+ refuses' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' -r2 '-*filtered.xml*' \
|
||||
--sitemap-url 'BASEURL/sitemapdir/gatedindex.xml'
|
||||
|
||||
# Control: without that filter the same chain does reach the page.
|
||||
crawl --errors 0 \
|
||||
--found 'sitemapdir/gated.html' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' -r2 \
|
||||
--sitemap-url 'BASEURL/sitemapdir/gatedindex.xml'
|
||||
|
||||
# A redirected sitemap is still ingested: the marking follows the 301.
|
||||
crawl --errors 0 \
|
||||
--found 'sitemapdir/orphan1.html' \
|
||||
--log-found 'moved\.xml redirects to .*pages\.xml\.gz' \
|
||||
--log-found 'URL\(s\) added from 127\.0\.0\.1:[0-9]+/sitemapdir/pages\.xml\.gz' \
|
||||
httrack 'BASEURL/sitemapdir/start.html' -r2 \
|
||||
--sitemap-url 'BASEURL/sitemapdir/moved.xml'
|
||||
@@ -1,25 +0,0 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# A re-fetch cut mid-header must leave the mirrored file alone (#748) and keep
|
||||
# it out of the update purge (#746). err.bin is the control, an HTTP 500 on the
|
||||
# same resource already surviving both; stay.bin answers normally with a new
|
||||
# body, so a fix that stopped overwriting anything fails.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
: "${top_srcdir:=..}"
|
||||
|
||||
bash "$top_srcdir/tests/local-crawl.sh" \
|
||||
--rerun-args '--update' \
|
||||
--log-found 'after 2 retries at link .*keep/page\.html' \
|
||||
--log-found 'after 2 retries at link .*keep/data\.bin' \
|
||||
--file-matches 'keep/page.html' 'KEEP-PAGE-V1' \
|
||||
--file-not-matches 'keep/page.html' 'HTTP/1\.0 200' \
|
||||
--file-matches 'keep/data.bin' 'KEEP-BIN-V1' \
|
||||
--file-not-matches 'keep/data.bin' 'HTTP/1\.0 200' \
|
||||
--file-min-bytes 'keep/data.bin' 2048 \
|
||||
--file-matches 'keep/err.bin' 'KEEP-ERR-V1' \
|
||||
--file-min-bytes 'keep/err.bin' 2048 \
|
||||
--file-matches 'keep/stay.bin' 'KEEP-STAY-V2' \
|
||||
--file-min-bytes 'keep/stay.bin' 2048 \
|
||||
httrack 'BASEURL/keep/index.html' --retries=2
|
||||
@@ -30,7 +30,6 @@ TEST_EXTENSIONS = .test
|
||||
TEST_LOG_COMPILER = $(BASH)
|
||||
TESTS = \
|
||||
00_runnable.test \
|
||||
01_engine-addlink.test \
|
||||
01_engine-changes.test \
|
||||
01_engine-charset.test \
|
||||
01_engine-cmdline.test \
|
||||
@@ -82,9 +81,7 @@ TESTS = \
|
||||
01_engine-status.test \
|
||||
01_engine-stripquery.test \
|
||||
01_engine-strsafe.test \
|
||||
01_engine-structcheck.test \
|
||||
01_engine-syscharset.test \
|
||||
01_engine-threadwait.test \
|
||||
01_engine-topindex.test \
|
||||
01_engine-urlhack.test \
|
||||
01_engine-unescape-bounds.test \
|
||||
@@ -94,7 +91,6 @@ TESTS = \
|
||||
01_engine-xfread.test \
|
||||
01_zlib-acceptencoding.test \
|
||||
01_zlib-warc.test \
|
||||
01_zlib-sitemap.test \
|
||||
01_zlib-warc-cdx.test \
|
||||
01_zlib-warc-wacz.test \
|
||||
01_zlib-contentcodings.test \
|
||||
@@ -195,9 +191,6 @@ TESTS = \
|
||||
90_webhttrack-checkbox-clear.test \
|
||||
91_webhttrack-directory.test \
|
||||
92_local-proxytrack-ndx-fields.test \
|
||||
93_local-changes.test \
|
||||
94_local-single-file.test \
|
||||
95_local-sitemap.test \
|
||||
96_local-refetch-keep.test
|
||||
93_local-changes.test
|
||||
|
||||
CLEANFILES = check-network_sh.cache
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user