Compare commits

..

1 Commits

Author SHA1 Message Date
Xavier Roche
abbe806907 Windows CI: cache vcpkg binary archives across runs
vcpkg builds openssl/brotli/zlib/zstd from source on every Windows run.
Redirect its files binary cache into the workspace and persist it with
actions/cache, keyed on the manifest (which carries the builtin-baseline).

x-gha, which used to do this, was dropped from vcpkg-tool (#1662) after
GitHub changed the Actions cache API; actions/cache over the archive dir is
the endorsed replacement and stays within the GitHub-owned-actions policy.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Signed-off-by: Xavier Roche <roche@httrack.com>
2026-07-25 09:56:43 +02:00
105 changed files with 508 additions and 5836 deletions

View File

@@ -6,9 +6,3 @@ updates:
directory: /src
schedule:
interval: weekly
# Keep the workflow action pins current (they only rot manually otherwise).
- package-ecosystem: github-actions
directory: /
schedule:
interval: weekly

View File

@@ -31,7 +31,7 @@ jobs:
env:
CC: ${{ matrix.cc }}
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -75,7 +75,7 @@ jobs:
name: build (no python3, Debian buildd)
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -119,7 +119,7 @@ jobs:
name: build (macOS arm64, clang)
runs-on: macos-14
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -166,7 +166,7 @@ jobs:
name: webhttrack smoke (macOS arm64)
runs-on: macos-14
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -199,7 +199,7 @@ jobs:
name: build (linux i386, gcc -m32)
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -242,7 +242,7 @@ jobs:
name: sanitize (ASan+UBSan, gcc)
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -296,7 +296,7 @@ jobs:
name: msan (MemorySanitizer, clang)
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -344,7 +344,7 @@ jobs:
name: fuzz (libFuzzer corpus replay, clang)
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -383,7 +383,7 @@ jobs:
name: build (no openssl, --disable-https)
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -437,7 +437,7 @@ jobs:
zlib1g-dev libssl-dev libbrotli-dev libzstd-dev \
debhelper devscripts lintian fakeroot
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -468,7 +468,7 @@ jobs:
name: distcheck (release tarball)
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive
@@ -493,7 +493,7 @@ jobs:
if: github.event_name == 'pull_request'
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
fetch-depth: 0
@@ -537,7 +537,7 @@ jobs:
tests/*.test
tools/mkdeb.sh
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
- name: Install linters
run: |
@@ -572,7 +572,7 @@ jobs:
if: github.event_name == 'pull_request'
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
fetch-depth: 0
@@ -624,7 +624,7 @@ jobs:
if: github.event_name == 'pull_request'
runs-on: ubuntu-24.04
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
fetch-depth: 0

View File

@@ -26,7 +26,7 @@ jobs:
# Upload findings to the repo's code-scanning dashboard.
security-events: write
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v6
with:
submodules: recursive

View File

@@ -29,7 +29,7 @@ jobs:
env:
VCPKG_DEFAULT_BINARY_CACHE: ${{ github.workspace }}\vcpkg_cache
steps:
- uses: actions/checkout@v7
- uses: actions/checkout@v4
with:
submodules: recursive # coucal lives in src/coucal
@@ -60,7 +60,7 @@ jobs:
# carries the builtin-baseline, so a Dependabot bump busts it; restore-keys
# still seeds the unchanged ports' archives, so only the bumped one rebuilds.
- name: Cache vcpkg binary archives
uses: actions/cache@v6
uses: actions/cache@v4
with:
path: ${{ github.workspace }}\vcpkg_cache
key: vcpkg-${{ matrix.platform }}-${{ hashFiles('src/vcpkg.json') }}
@@ -225,16 +225,15 @@ jobs:
# Every gate here exits 77, so an all-skipped suite would report green having
# tested nothing: pin the skips, and floor the passes in case the glob empties.
# footer-overflow skips on Windows (needs a path past MAX_PATH); crange pending #581;
# webdav-mime needs a reapable background listener, which MSYS cannot give it.
expected_skips=" 01_engine-footer-overflow.test 48_local-crange-memresume.test 71_local-crange-repaircache.test 79_local-proxytrack-webdav-mime.test"
# footer-overflow skips on Windows (needs a path past MAX_PATH); crange pending #581.
expected_skips=" 01_engine-footer-overflow.test 48_local-crange-memresume.test 71_local-crange-repaircache.test"
[ "$pass" -ge 90 ] || { echo "::error::only $pass tests passed ($skip skipped)"; exit 1; }
[ "$skipped" = "$expected_skips" ] || { echo "::error::unexpected skips:$skipped"; exit 1; }
[ "$fail" -eq 0 ] || { echo "::error::failing:$failed"; exit 1; }
- name: Upload the test logs
if: always()
uses: actions/upload-artifact@v7
uses: actions/upload-artifact@v4
with:
name: engine-tests-${{ matrix.platform }}-${{ matrix.configuration }}
path: tests/*.log
@@ -242,7 +241,7 @@ jobs:
- name: Upload MSBuild logs
if: always()
uses: actions/upload-artifact@v7
uses: actions/upload-artifact@v4
with:
name: msbuild-${{ matrix.platform }}-${{ matrix.configuration }}
path: msbuild-*.log

View File

@@ -66,11 +66,10 @@ AC_SUBST(LT_CV_OBJDIR,$lt_cv_objdir)
AC_SUBST(VERSION_INFO)
### Default CFLAGS
# No -Wdeclaration-after-statement: nothing sets -std=, so this builds as gnu17.
DEFAULT_CFLAGS="-Wall -Wformat -Wformat-security \
-Wmultichar -Wwrite-strings -Wcast-qual -Wcast-align \
-Wstrict-prototypes -Wmissing-prototypes \
-Wmissing-declarations \
-Wmissing-declarations -Wdeclaration-after-statement \
-Wpointer-arith -Wsequence-point -Wnested-externs \
-D_REENTRANT"
AC_SUBST(DEFAULT_CFLAGS)
@@ -78,30 +77,27 @@ DEFAULT_LDFLAGS=""
AC_SUBST(DEFAULT_LDFLAGS)
### Additional flags (if supported)
# -Werror on probes: exit-status-only checks let clang's warn-on-unknown-flag through.
AX_CHECK_COMPILE_FLAG([-Wparentheses], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wparentheses"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-Winit-self], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Winit-self"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-Wunused-but-set-parameter], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wunused-but-set-parameter"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-Waddress], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Waddress"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-Wuninitialized], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wuninitialized"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-Wformat=2], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wformat=2"], [], [-Werror])
# -Wformat-nonliteral needs -Wformat in the probe or gcc rejects it as ignored.
AX_CHECK_COMPILE_FLAG([-Wformat-nonliteral], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wformat-nonliteral"], [], [-Werror -Wformat])
AX_CHECK_COMPILE_FLAG([-Wmissing-parameter-type], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wmissing-parameter-type"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-Wold-style-definition], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wold-style-definition"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-Wignored-qualifiers], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wignored-qualifiers"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-Wparentheses], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wparentheses"])
AX_CHECK_COMPILE_FLAG([-Winit-self], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Winit-self"])
AX_CHECK_COMPILE_FLAG([-Wunused-but-set-parameter], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wunused-but-set-parameter"])
AX_CHECK_COMPILE_FLAG([-Waddress], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Waddress"])
AX_CHECK_COMPILE_FLAG([-Wuninitialized], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wuninitialized"])
AX_CHECK_COMPILE_FLAG([-Wformat=2], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wformat=2"])
AX_CHECK_COMPILE_FLAG([-Wformat-nonliteral], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wformat-nonliteral"])
AX_CHECK_COMPILE_FLAG([-Wmissing-parameter-type], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wmissing-parameter-type"])
AX_CHECK_COMPILE_FLAG([-Wold-style-definition], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wold-style-definition"])
AX_CHECK_COMPILE_FLAG([-Wignored-qualifiers], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Wignored-qualifiers"])
# Make htssafe.h's pointer-dest 'warning' attribute a hard error in our build
# (migration is at zero; a new char* dest is a regression). gcc/clang each take
# only their own spelling; downstream keeps the plain warning, not a build break.
AX_CHECK_COMPILE_FLAG([-Werror=attribute-warning], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Werror=attribute-warning"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-Werror=user-defined-warnings], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Werror=user-defined-warnings"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-fstrict-aliasing -Wstrict-aliasing], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -fstrict-aliasing -Wstrict-aliasing"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-Werror=attribute-warning], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Werror=attribute-warning"])
AX_CHECK_COMPILE_FLAG([-Werror=user-defined-warnings], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -Werror=user-defined-warnings"])
AX_CHECK_COMPILE_FLAG([-fstrict-aliasing -Wstrict-aliasing], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -fstrict-aliasing -Wstrict-aliasing"])
AX_CHECK_COMPILE_FLAG([-fstack-protector-strong], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -fstack-protector-strong"],
[AX_CHECK_COMPILE_FLAG([-fstack-protector], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -fstack-protector"], [], [-Werror])], [-Werror])
AX_CHECK_COMPILE_FLAG([-fstack-clash-protection], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -fstack-clash-protection"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-fcf-protection], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -fcf-protection"], [], [-Werror])
# No --discard-all: it drops the local symbols naming every static function, so
# a trace misattributes them to the nearest surviving global. Costs 0.6% size.
[AX_CHECK_COMPILE_FLAG([-fstack-protector], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -fstack-protector"])])
AX_CHECK_COMPILE_FLAG([-fstack-clash-protection], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -fstack-clash-protection"])
AX_CHECK_COMPILE_FLAG([-fcf-protection], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -fcf-protection"])
AX_CHECK_LINK_FLAG([-Wl,--discard-all], [DEFAULT_LDFLAGS="$DEFAULT_LDFLAGS -Wl,--discard-all"])
AX_CHECK_LINK_FLAG([-Wl,--no-undefined], [DEFAULT_LDFLAGS="$DEFAULT_LDFLAGS -Wl,--no-undefined"])
AX_CHECK_LINK_FLAG([-Wl,-z,relro,-z,now], [DEFAULT_LDFLAGS="$DEFAULT_LDFLAGS -Wl,-z,relro,-z,now"])
AX_CHECK_LINK_FLAG([-Wl,-z,noexecstack], [DEFAULT_LDFLAGS="$DEFAULT_LDFLAGS -Wl,-z,noexecstack"])
@@ -123,13 +119,13 @@ AC_SUBST([LIBC_FORCE_LINK])
### PIE
CFLAGS_PIE=""
LDFLAGS_PIE=""
AX_CHECK_COMPILE_FLAG([-fpie], [CFLAGS_PIE="-fpie"], [], [-Werror])
AX_CHECK_COMPILE_FLAG([-fpie -pie], [CFLAGS_PIE="-fpie -pie"])
AX_CHECK_LINK_FLAG([-pie], [LDFLAGS_PIE="-pie"])
AC_SUBST([CFLAGS_PIE])
AC_SUBST([LDFLAGS_PIE])
# Ties a crash trace from a stripped build back to its separate debug symbols.
AX_CHECK_LINK_FLAG([-Wl,--build-id], [DEFAULT_LDFLAGS="$DEFAULT_LDFLAGS -Wl,--build-id"])
## Export all symbols for backtraces
AX_CHECK_COMPILE_FLAG([-rdynamic], [DEFAULT_CFLAGS="$DEFAULT_CFLAGS -rdynamic"])
### Check for -fvisibility=hidden support
gl_VISIBILITY
@@ -173,7 +169,7 @@ AC_CHECK_TYPE(sa_family_t, [], [AC_DEFINE([sa_family_t], [uint16_t], [sa_family_
AX_CHECK_ALIGNED_ACCESS_REQUIRED
# check for various headers
AC_CHECK_HEADERS([execinfo.h sys/ioctl.h])
AC_CHECK_HEADERS([execinfo.h])
### zlib
CHECK_ZLIB()

View File

@@ -300,7 +300,6 @@ ones it kept so the local copy browses offline. These options tune both halves.<
<tr><td><tt>--preserve (-%p), --disable-passwords (-%x)</tt></td><td>Leave HTML untouched (no rewriting), and strip passwords out of saved links.</td></tr>
<tr><td><tt>--extended-parsing (-%P), --parse-java (-j)</tt></td><td>Aggressive link discovery, and how much script content is parsed for links.</td></tr>
<tr><td><tt>--mime-html (-%M)</tt></td><td>Save the whole mirror as a single MIME-encapsulated <tt>.mht</tt> archive (<tt>index.mht</tt>).</td></tr>
<tr><td><tt>--single-file (-%Z), --single-file-max-size N</tt></td><td>Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded as <tt>data:</tt> URIs. Assets over the cap (10&nbsp;MB by default) keep their link, as do audio, video, and the links from one page to another. A sibling of <tt>-%M</tt>, not a replacement: see the recipe below for which to pick.</td></tr>
<tr><td><tt>--index (-I), --build-top-index (-%i), --search-index (-%I)</tt></td><td>Build a per-mirror index, a top index across projects, and a searchable keyword index.</td></tr>
</table>
@@ -483,22 +482,6 @@ add <tt>--warc-cdx</tt> for a sorted CDXJ index, or <tt>--wacz</tt> to bundle th
archive, index and pages into one WACZ for replay tools such as
replayweb.page.</small></p>
<h4>Pages you can hand to someone as one file</h4>
<p><tt>httrack https://example.com/ --single-file --path mydir</tt><br>
<small>Rewrites each saved page after the crawl so its stylesheets, scripts,
images and fonts are embedded as <tt>data:</tt> URIs. The mirror stays a normal
browsable tree, with links between pages relative and the assets still on disk,
but any single <tt>.html</tt> file can now be mailed or archived on its own.
Raise or lower the 10&nbsp;MB per-asset limit with
<tt>--single-file-max-size N</tt>; anything over it, plus audio and video, keeps
its link.<br>
Reach for this when the file has to open for someone you cannot make assumptions
about: it is plain HTML and needs no add-on. Reach for <tt>--mime-html</tt>
instead when a Chromium-family browser is a given and the mirror is large: MIME
carries text parts without the base64 tax, keeps each resource's original URL,
and stores a shared asset once rather than re-embedding it in every page that
uses it.</small></p>
<h4>HTTrack as a fetch tool</h4>
<p><tt>httrack --get https://host/file.bin --path tmp</tt><br>
<small><tt>--get</tt> fetches one file with cache, index, depth, cookies and robots

View File

@@ -90,10 +90,9 @@ offline browser : copy websites to a local directory</p>
] [ <b>-NN, --structure[=N]</b> ] [ <b>-%N,
--delayed-type-check</b> ] [ <b>-%D,
--cached-delayed-type-check</b> ] [ <b>-%M, --mime-html</b>
] [ <b>-%Z, --single-file</b> ] [ <b>-LN,
--long-names[=N]</b> ] [ <b>-KN, --keep-links[=N]</b> ] [
<b>-x, --replace-external</b> ] [ <b>-%x,
--disable-passwords</b> ] [ <b>-%q,
] [ <b>-LN, --long-names[=N]</b> ] [ <b>-KN,
--keep-links[=N]</b> ] [ <b>-x, --replace-external</b> ] [
<b>-%x, --disable-passwords</b> ] [ <b>-%q,
--include-query-string</b> ] [ <b>-%g, --strip-query</b> ] [
<b>-o, --generate-errors</b> ] [ <b>-X, --purge-old[=N]</b>
] [ <b>-%p, --preserve</b> ] [ <b>-%T, --utf8-conversion</b>
@@ -649,24 +648,6 @@ don&rsquo;t wait) (--cached-delayed-type-check)</p></td></tr>
<td width="4%">
<p>-%Z</p></td>
<td width="5%"></td>
<td width="82%">
<p>after the mirror, rewrite each saved page with its
stylesheets, scripts, images and fonts inlined as data:
URIs, so any page opens by double-click anywhere (links
between pages stay relative; audio and video stay links);
--single-file-max-size N caps each asset (default 10485760
bytes). %M is the better container where a Chromium-family
browser is a given: one archive, no base64 tax on text, a
shared asset stored once (--single-file)</p></td></tr>
<tr valign="top" align="left">
<td width="9%"></td>
<td width="4%">
<p>-%t</p></td>
<td width="5%"></td>
<td width="82%">

View File

@@ -97,8 +97,9 @@ ${do:end-if}
</pre>
${LANG_G8} :
${/* an http: page cannot navigate to file:, so the mirror is reached through the server */}
<a href="/website/index.html" target="_new">
${do:output-mode:html-urlescaped}
<a href="file://${path}/${projname}/" target="_new">
${do:output-mode:}
${path}/${projname}
</a></li>
<ul>

View File

@@ -136,13 +136,6 @@ ${listid:build:LISTDEF_3}
<tr><td><input type="checkbox" name="nopurge" ${checked:nopurge}
title='${html:LANG_I1a}' onMouseOver="info('${html:LANG_I1a}'); return true" onMouseOut="info('&nbsp;'); return true"
> ${LANG_I57}</td></tr>
<tr><td><input type="checkbox" name="singlefile" ${checked:singlefile}
title='${html:LANG_SINGLEFILETIP}' onMouseOver="info('${html:LANG_SINGLEFILETIP}'); return true" onMouseOut="info('&nbsp;'); return true"
> ${LANG_SINGLEFILE}</td></tr>
<tr><td>${LANG_SINGLEFILEMAX}
<input name="singlefilemax" value="${singlefilemax}" size="12"
title='${html:LANG_SINGLEFILEMAXTIP}' onMouseOver="info('${html:LANG_SINGLEFILEMAXTIP}'); return true" onMouseOut="info('&nbsp;'); return true"
></td></tr>
</table>
<tr><td>

View File

@@ -103,7 +103,7 @@ ${LANG_Q3}
<tr><td>
<table width="100%">
<tr><td align="left">
<input type="submit" value="${LANG_OK}"
<input type="submit" value="${LANG_OK]"
${do:output-mode:html-urlescaped}
onClick="if (confirm(str_replace(str_replace('${LANG_DIAL7}', '%20', ' '), '%0a', ' '))) { form.closeme.value=1; form.submit(); } return false;"
${do:output-mode:}

View File

@@ -143,8 +143,6 @@ ${do:copy:StripQuery:stripquery}
${do:copy:StoreAllInCache:cache2}
${do:copy:Warc:warc}
${do:copy:WarcFile:warcfile}
${do:copy:SingleFile:singlefile}
${do:copy:SingleFileMaxSize:singlefilemax}
${do:copy:LogType:logtype}
${do:copy:UseHTTPProxyForFTP:ftpprox}
${do:copy:ProxyType:proxytype}

View File

@@ -121,9 +121,9 @@ httrack \
--quiet \
--build-top-index \
${test:todo:--mirror:--mirror:--mirror-wizard:--get:--mirrorlinks:--testlinks:--continue:--update}
${unquoted:urls}
${test:filelist:-%L "}${arg:filelist}${test:filelist:"}
--path "${arg:path}/${arg:projname}"
${urls}
${test:filelist:-%L "}${filelist}${test:filelist:"}
--path "${html:path}/${html:projname}"
\
${test:parseall:--near}
${test:link:--test}
@@ -131,7 +131,7 @@ httrack \
${test:htmlfirst::--priority=7}
\
${do:if-not-empty:BuildString}
--structure "${arg:BuildString}"
--structure "${BuildString}"
${do:end-if}
${test:build:-N0:-N0:-N1:-N2:-N3:-N4:-N5:-N100:-N101:-N102:-N103:-N104:-N105:-N99:-N199:}
\
@@ -150,30 +150,29 @@ ${do:end-if}
${test:travel3::--keep-links=0:--keep-links:--keep-links=3:--keep-links=4}
${test:windebug:--debug-headers}
\
${test:connexion:--sockets=}${unquoted:connexion}
${test:connexion:--sockets=}${connexion}
${test:ka:--keep-alive}
${test:timeout:--timeout=}${unquoted:timeout}
${test:timeout:--timeout=}${timeout}
${test:remt:--host-control=1}
${test:retry:--retries=}${unquoted:retry}
${test:rate:--min-rate=}${unquoted:rate}
${test:retry:--retries=}${retry}
${test:rate:--min-rate=}${rate}
${test:rems:--host-control=2}
\
${test:depth:--depth=}${unquoted:depth}
${test:depth2:--ext-depth=}${unquoted:depth2}
${/* -m<n> resets the html limit, so the bare form must precede the -m,<n> one */}
${test:othermax:--max-files=}${unquoted:othermax}
${test:maxhtml:--max-files=,}${unquoted:maxhtml}
${test:sizemax:--max-size=}${unquoted:sizemax}
${test:pausebytes:--max-pause=}${unquoted:pausebytes}
${test:maxtime:--max-time=}${unquoted:maxtime}
${test:maxrate:--max-rate=}${unquoted:maxrate}
${test:maxconn:--connection-per-second=}${unquoted:maxconn}
${test:maxlinks:--advanced-maxlinks=}${unquoted:maxlinks}
${test:depth:--depth=}${depth}
${test:depth2:--ext-depth=}${depth2}
${test:maxhtml:--max-files=,}${maxhtml}
${test:othermax:--max-files=}${othermax}
${test:sizemax:--max-files=}${sizemax}
${test:pausebytes:--max-pause=}${pausebytes}
${test:maxtime:--max-time=}${maxtime}
${test:maxrate:--max-rate=}${maxrate}
${test:maxconn:--connection-per-second=}${maxconn}
${test:maxlinks:--advanced-maxlinks=}${maxlinks}
\
--user-agent "${arg:user}"
--footer "${arg:footer}"
--user-agent "${html:user}"
--footer "${html:footer}"
\
${unquoted:url2}
${url2}
\
${test:cookies:--cookies=0:}
${test:parsejava:--parse-java=0:}
@@ -182,22 +181,20 @@ ${/* -m<n> resets the html limit, so the bare form must precede the -m,<n> one *
${test:keepwww:--keep-www-prefix}
${test:keepslashes:--keep-double-slashes}
${test:keepqueryorder:--keep-query-order}
${test:cookiesfile:--cookies-file "}${arg:cookiesfile}${test:cookiesfile:"}
${test:pausefiles:--pause "}${arg:pausefiles}${test:pausefiles:"}
${test:stripquery:--strip-query "}${arg:stripquery}${test:stripquery:"}
${test:cookiesfile:--cookies-file "}${html:cookiesfile}${test:cookiesfile:"}
${test:pausefiles:--pause "}${pausefiles}${test:pausefiles:"}
${test:stripquery:--strip-query "}${html:stripquery}${test:stripquery:"}
${test:toler:--tolerant}
${test:http10:--http-10}
${test:cache2:--store-all-in-cache}
${test:warc:--warc}
${test:warcfile:--warc-file "}${arg:warcfile}${test:warcfile:"}
${test:singlefile:--single-file}
${test:singlefilemax:--single-file-max-size=}${unquoted:singlefilemax}
${test:warcfile:--warc-file "}${html:warcfile}${test:warcfile:"}
${test:norecatch:--do-not-recatch}
${test:logf:--single-log}
${test:logtype:::--extra-log:--debug-log}
${test:index:--index=0:}
${test:index2:--search-index=0:--search-index}
${test:prox:--proxy "}${do:if-not-empty:prox}${test:proxytype::socks5:connect}${test:proxytype:\3A//}${do:end-if}${do:output-mode:html}${arg:prox}${test:prox:\3A}${arg:portprox}${test:prox:"}
${test:prox:--proxy "}${do:if-not-empty:prox}${test:proxytype::socks5:connect}${test:proxytype:\3A//}${do:end-if}${do:output-mode:html}${prox}${test:prox:\3A}${portprox}${test:prox:"}
${test:ftpprox:--httpproxy-ftp=0:--httpproxy-ftp}
</textarea>
@@ -216,7 +213,7 @@ ParseAll=${ztest:parseall:0:1}
HTMLFirst=${ztest:htmlfirst:0:1}
Cache=${ztest:cache:0:1}
NoRecatch=${ztest:norecatch:0:1}
Dos=${dos}
Dos=${dos
Index=${ztest:index:0:1}
WordIndex=${ztest:index2:0:1}
Log=${ztest:logf:0:1:2}
@@ -244,8 +241,6 @@ StripQuery=${stripquery}
StoreAllInCache=${ztest:cache2:0:1}
Warc=${ztest:warc:0:1}
WarcFile=${warcfile}
SingleFile=${ztest:singlefile:0:1}
SingleFileMaxSize=${singlefilemax}
LogType=${logtype}
UseHTTPProxyForFTP=${ztest:ftpprox:0:1}
ProxyType=${proxytype}

View File

@@ -1042,11 +1042,3 @@ LANG_WARCFILE
WARC archive name:
LANG_WARCFILETIP
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
LANG_SINGLEFILE
Inline assets as data: URIs (self-contained pages)
LANG_SINGLEFILETIP
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
LANG_SINGLEFILEMAX
Largest inlined asset (bytes):
LANG_SINGLEFILEMAXTIP
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Èìå íà WARC àðõèâà:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Íåçàäúëæèòåëíî áàçîâî èìå çà WARC àðõèâà; îñòàâåòå ïðàçíî çà àâòîìàòè÷íî èìåíóâàíå â èçõîäíàòà äèðåêòîðèÿ.
Inline assets as data: URIs (self-contained pages)
Âãðàæäàíå íà ðåñóðñèòå êàòî data: URI (ñàìîñòîÿòåëíè ñòðàíèöè)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Ñëåä çàâúðøâàíå íà îãëåäàëîòî âñÿêà çàïàçåíà ñòðàíèöà ñå ïðåçàïèñâà ñ âãðàäåíè ñòèëîâå, ñêðèïòîâå, èçîáðàæåíèÿ è øðèôòîâå, òàêà ÷å ñòðàíèöàòà äà ìîæå äà ñå îòâîðè è ñàìîñòîÿòåëíî.
Largest inlined asset (bytes):
Íàé-ãîëÿì âãðàäåí ðåñóðñ (áàéòîâå):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Ðåñóðñ íàä òîçè ðàçìåð çàïàçâà îáèêíîâåíà âðúçêà; îñòàâåòå ïðàçíî çà ñòîéíîñòòà ïî ïîäðàçáèðàíå îò 10485760 áàéòà.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Nombre del archivo WARC:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Nombre base opcional para el archivo WARC; déjelo en blanco para nombrarlo automáticamente en el directorio de salida.
Inline assets as data: URIs (self-contained pages)
Incrustar los recursos como URI data: (páginas autónomas)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Una vez terminada la copia, reescribir cada página guardada con sus hojas de estilo, scripts, imágenes y fuentes incrustadas, de modo que una página también pueda abrirse por sí sola.
Largest inlined asset (bytes):
Tamaño máximo del recurso incrustado (bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Un recurso mayor que este tamaño conserva un enlace normal; déjelo vacío para el valor predeterminado de 10485760 bytes.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Název archivu WARC:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Volitelný základní název archivu WARC; ponechte prázdné pro automatické pojmenování ve výstupním adresáøi.
Inline assets as data: URIs (self-contained pages)
Vložit zdroje jako URI data: (samostatné stránky)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Po dokonèení zrcadlení pøepsat každou uloženou stránku s vloženými styly, skripty, obrázky a písmy, aby ji bylo možné otevøít i samostatnì.
Largest inlined asset (bytes):
Nejvìtší vložený zdroj (bajty):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Zdroj vìtší než tato velikost si ponechá bìžný odkaz; ponechte prázdné pro výchozí hodnotu 10485760 bajtù.

View File

@@ -964,11 +964,3 @@ WARC archive name:
WARC 封存檔名稱:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
WARC 封存檔的選用基本名稱;留空則於輸出目錄中自動命名。
Inline assets as data: URIs (self-contained pages)
將資源內嵌為 data: URI獨立網頁
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
鏡射完成後,重寫每個已儲存的網頁,將樣式表、指令碼、圖片與字型內嵌其中,讓網頁也能單獨開啟。
Largest inlined asset (bytes):
內嵌資源大小上限(位元組):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
超過此大小的資源會保留一般連結;留空則使用預設的 10485760 位元組。

View File

@@ -964,11 +964,3 @@ WARC archive name:
WARC 归档名称:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
WARC 归档的可选基本名称;留空则在输出目录中自动命名。
Inline assets as data: URIs (self-contained pages)
将资源内嵌为 data: URI独立网页
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
镜像完成后,重写每个已保存的网页,将样式表、脚本、图片和字体内嵌其中,使网页也能单独打开。
Largest inlined asset (bytes):
内嵌资源大小上限(字节):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
超过此大小的资源会保留普通链接;留空则使用默认的 10485760 字节。

View File

@@ -966,11 +966,3 @@ WARC archive name:
Naziv WARC arhive:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Neobavezni osnovni naziv WARC arhive; ostavite prazno za automatsko imenovanje u izlaznom direktoriju.
Inline assets as data: URIs (self-contained pages)
Ugradi resurse kao data: URI-je (samostalne stranice)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Nakon dovr¹etka zrcaljenja ponovno zapi¹i svaku spremljenu stranicu s ugraðenim stilskim listovima, skriptama, slikama i fontovima, tako da se stranica mo¾e otvoriti i zasebno.
Largest inlined asset (bytes):
Najveæi ugraðeni resurs (bajtovi):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Resurs veæi od ove velièine zadr¾ava obiènu poveznicu; ostavite prazno za zadanih 10485760 bajtova.

View File

@@ -1012,11 +1012,3 @@ WARC archive name:
Navn på WARC-arkiv:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Valgfrit basisnavn til WARC-arkivet; lad feltet stå tomt for automatisk navngivning i outputmappen.
Inline assets as data: URIs (self-contained pages)
Indlejr ressourcer som data:-URI'er (selvstændige sider)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Når spejlingen er færdig, omskrives hver gemt side med dens typografiark, scripts, billeder og skrifttyper indlejret, så en side også kan åbnes alene.
Largest inlined asset (bytes):
Største indlejrede ressource (byte):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
En ressource over denne størrelse beholder et almindeligt link; lad feltet stå tomt for standardværdien på 10485760 byte.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Name des WARC-Archivs:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Optionaler Basisname für das WARC-Archiv; leer lassen, um es automatisch im Ausgabeverzeichnis zu benennen.
Inline assets as data: URIs (self-contained pages)
Ressourcen als data:-URIs einbetten (eigenständige Seiten)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Nach Abschluss der Spiegelung jede gespeicherte Seite mit eingebetteten Stylesheets, Skripten, Bildern und Schriftarten neu schreiben, sodass eine Seite auch für sich allein geöffnet werden kann.
Largest inlined asset (bytes):
Größte eingebettete Ressource (Bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Eine Ressource über dieser Größe behält einen gewöhnlichen Link; leer lassen für den Standardwert von 10485760 Bytes.

View File

@@ -964,11 +964,3 @@ WARC archive name:
WARC-arhiivi nimi:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
WARC-arhiivi valikuline põhinimi; jäta tühjaks, et see väljundkataloogis automaatselt nimetada.
Inline assets as data: URIs (self-contained pages)
Manusta ressursid data: URI-dena (iseseisvad lehed)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Pärast peegelduse lõppu kirjuta iga salvestatud leht ümber, manustades selle laaditabelid, skriptid, pildid ja fondid, nii et lehte saab avada ka eraldi.
Largest inlined asset (bytes):
Suurim manustatud ressurss (baiti):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Sellest suurem ressurss säilitab tavalise lingi; jäta tühjaks vaikeväärtuse 10485760 baiti jaoks.

View File

@@ -1012,11 +1012,3 @@ WARC archive name:
WARC archive name:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Inline assets as data: URIs (self-contained pages)
Inline assets as data: URIs (self-contained pages)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Largest inlined asset (bytes):
Largest inlined asset (bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.

View File

@@ -966,11 +966,3 @@ WARC archive name:
WARC-arkiston nimi:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Valinnainen WARC-arkiston perusnimi; jätä tyhjäksi, jotta se nimetään automaattisesti tulostehakemistoon.
Inline assets as data: URIs (self-contained pages)
Upota resurssit data:-URI-osoitteina (itsenäiset sivut)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Kun peilaus on valmis, kirjoita jokainen tallennettu sivu uudelleen niin, että sen tyylitiedostot, komentosarjat, kuvat ja fontit upotetaan, jolloin sivun voi avata myös yksinään.
Largest inlined asset (bytes):
Suurin upotettu resurssi (tavua):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Tätä suurempi resurssi säilyttää tavallisen linkin; jätä tyhjäksi, jolloin käytetään oletusarvoa 10485760 tavua.

View File

@@ -1012,11 +1012,3 @@ WARC archive name:
Nom de l'archive WARC :
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Nom de base optionnel pour l'archive WARC ; laissez vide pour le générer automatiquement dans le répertoire de sortie.
Inline assets as data: URIs (self-contained pages)
Intégrer les ressources en URI data: (pages autonomes)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Une fois le miroir terminé, réécrire chaque page enregistrée avec ses feuilles de style, scripts, images et polices intégrés, afin qu'une page puisse aussi être ouverte seule.
Largest inlined asset (bytes):
Taille maximale d'une ressource intégrée (octets) :
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Au-delà de cette taille, la ressource reste un lien ordinaire ; laissez vide pour la valeur par défaut de 10485760 octets.

View File

@@ -966,11 +966,3 @@ WARC archive name:
Όνομα αρχείου WARC:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Προαιρετικό βασικό όνομα για το αρχείο WARC. Αφήστε το κενό για αυτόματη ονομασία στον κατάλογο εξόδου.
Inline assets as data: URIs (self-contained pages)
Ενσωμάτωση των πόρων ως URI data: (αυτόνομες σελίδες)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Μόλις ολοκληρωθεί το αντίγραφο, κάθε αποθηκευμένη σελίδα ξαναγράφεται με ενσωματωμένα τα φύλλα στυλ, τα σενάρια, τις εικόνες και τις γραμματοσειρές της, ώστε η σελίδα να μπορεί να ανοίξει και μόνη της.
Largest inlined asset (bytes):
Μέγιστος ενσωματωμένος πόρος (byte):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Πόρος μεγαλύτερος από αυτό το μέγεθος διατηρεί κανονικό σύνδεσμο. Αφήστε το κενό για την προεπιλογή των 10485760 byte.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Nome dell'archivio WARC:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Nome di base facoltativo per l'archivio WARC; lascia vuoto per assegnarlo automaticamente nella directory di output.
Inline assets as data: URIs (self-contained pages)
Incorpora le risorse come URI data: (pagine autonome)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Al termine della copia, riscrive ogni pagina salvata incorporando fogli di stile, script, immagini e caratteri, in modo che una pagina possa essere aperta anche da sola.
Largest inlined asset (bytes):
Dimensione massima della risorsa incorporata (byte):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Una risorsa oltre questa dimensione mantiene un collegamento normale; lasciare vuoto per il valore predefinito di 10485760 byte.

View File

@@ -964,11 +964,3 @@ WARC archive name:
WARC アーカイブ名:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
WARC アーカイブの任意のベース名。空欄にすると出力ディレクトリ内で自動的に名前が付けられます。
Inline assets as data: URIs (self-contained pages)
リソースを data: URI として埋め込む (単独で開けるページ)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
ミラーの完了後、保存された各ページを、スタイルシート、スクリプト、画像、フォントを埋め込んだ形で書き直します。ページ単体でも開けるようになります。
Largest inlined asset (bytes):
埋め込む最大サイズ (バイト):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
このサイズを超えるリソースは通常のリンクのままになります。空欄にすると既定値の 10485760 バイトになります。

View File

@@ -964,11 +964,3 @@ WARC archive name:
¸ÜÕ ÝÐ WARC ÐàåØÒÐâÐ:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
¸×ÑÞàÝÞ ÞáÝÞÒÝÞ ØÜÕ ×Ð WARC ÐàåØÒÐâÐ; ÞáâÐÒÕâÕ ßàÐ×ÝÞ ×Ð ÐÒâÞÜÐâáÚÞ ØÜÕÝãÒÐúÕ ÒÞ Ø×ÛÕ×ÝØÞâ ÔØàÕÚâÞàØãÜ.
Inline assets as data: URIs (self-contained pages)
²ÓàÐÔØ ÓØ àÕáãàáØâÕ ÚÐÚÞ data: URI (áÐÜÞáâÞøÝØ áâàÐÝØæØ)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
¿Þ ×ÐÒàèãÒÐúÕ ÝÐ ÞÓÛÕÔÐÛÞâÞ, áÕÚÞøÐ ×ÐçãÒÐÝÐ áâàÐÝØæÐ áÕ ßàÕߨèãÒÐ áÞ ÒÓàÐÔÕÝØ áâØÛáÚØ ÛØáâÞÒØ, áÚàØßâØ, áÛØÚØ Ø äÞÝâÞÒØ, ×Ð ÔÐ ÜÞÖÕ áâàÐÝØæÐâÐ ÔÐ áÕ ÞâÒÞàØ Ø áÐÜÞáâÞøÝÞ.
Largest inlined asset (bytes):
½ÐøÓÞÛÕÜ ÒÓàÐÔÕÝ àÕáãàá (ÑÐøâØ):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
ÀÕáãàá ßÞÓÞÛÕÜ ÞÔ ÞÒÐÐ ÓÞÛÕÜØÝÐ ×ÐÔàÖãÒÐ ÞÑØçÝÐ ÒàáÚÐ; ÞáâÐÒÕâÕ ßàÐ×ÝÞ ×Ð áâÐÝÔÐàÔÝØâÕ 10485760 ÑÐøâØ.

View File

@@ -964,11 +964,3 @@ WARC archive name:
WARC archívum neve:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
A WARC archívum opcionális alapneve; hagyja üresen az automatikus elnevezéshez a kimeneti könyvtárban.
Inline assets as data: URIs (self-contained pages)
Erõforrások beágyazása data: URI-ként (önálló oldalak)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
A tükrözés befejezése után minden mentett oldal újraírása a stíluslapok, parancsfájlok, képek és betûtípusok beágyazásával, így az oldal önmagában is megnyitható.
Largest inlined asset (bytes):
Legnagyobb beágyazott erõforrás (bájt):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Az ennél nagyobb erõforrás közönséges hivatkozás marad; hagyja üresen a 10485760 bájtos alapértelmezéshez.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Naam van WARC-archief:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Optionele basisnaam voor het WARC-archief; laat leeg om het automatisch een naam te geven in de uitvoermap.
Inline assets as data: URIs (self-contained pages)
Bronnen insluiten als data:-URI's (op zichzelf staande pagina's)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Zodra de spiegeling klaar is, elke opgeslagen pagina herschrijven met ingesloten stijlbladen, scripts, afbeeldingen en lettertypen, zodat een pagina ook los geopend kan worden.
Largest inlined asset (bytes):
Grootste ingesloten bron (bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Een bron boven deze grootte houdt een gewone koppeling; laat leeg voor de standaardwaarde van 10485760 bytes.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Navn på WARC-arkiv:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Valgfritt basisnavn for WARC-arkivet; la feltet stå tomt for automatisk navngivning i utdatamappen.
Inline assets as data: URIs (self-contained pages)
Bygg inn ressurser som data:-URI-er (selvstendige sider)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Når speilingen er ferdig, skrives hver lagrede side om med stilark, skript, bilder og skrifter innebygd, slik at en side også kan åpnes alene.
Largest inlined asset (bytes):
Største innebygde ressurs (byte):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
En ressurs over denne størrelsen beholder en vanlig lenke; la feltet stå tomt for standardverdien på 10485760 byte.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Nazwa archiwum WARC:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Opcjonalna nazwa bazowa archiwum WARC; pozostaw puste, aby nazwaæ je automatycznie w katalogu wyj¶ciowym.
Inline assets as data: URIs (self-contained pages)
Osad¼ zasoby jako identyfikatory data: URI (samodzielne strony)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Po zakoñczeniu tworzenia kopii przepisz ka¿d± zapisan± stronê z osadzonymi arkuszami stylów, skryptami, obrazami i czcionkami, aby stronê mo¿na by³o otworzyæ równie¿ samodzielnie.
Largest inlined asset (bytes):
Najwiêkszy osadzony zasób (bajty):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Zasób wiêkszy ni¿ ten rozmiar zachowuje zwyk³y odno¶nik; pozostaw puste, aby u¿yæ domy¶lnych 10485760 bajtów.

View File

@@ -1012,11 +1012,3 @@ WARC archive name:
Nome do arquivo WARC:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Nome base opcional para o arquivo WARC; deixe em branco para nomeá-lo automaticamente no diretório de saída.
Inline assets as data: URIs (self-contained pages)
Incorporar os recursos como URIs data: (páginas autônomas)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Ao terminar o espelhamento, reescrever cada página salva com suas folhas de estilo, scripts, imagens e fontes incorporadas, para que a página também possa ser aberta sozinha.
Largest inlined asset (bytes):
Maior recurso incorporado (bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Um recurso acima desse tamanho mantém um link comum; deixe em branco para o padrão de 10485760 bytes.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Nome do arquivo WARC:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Nome base opcional para o arquivo WARC; deixe em branco para o nomear automaticamente no diretório de saída.
Inline assets as data: URIs (self-contained pages)
Incorporar os recursos como URI data: (páginas autónomas)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Depois de terminado o espelho, reescrever cada página guardada com as suas folhas de estilo, scripts, imagens e tipos de letra incorporados, para que a página também possa ser aberta sozinha.
Largest inlined asset (bytes):
Maior recurso incorporado (bytes):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Um recurso acima deste tamanho mantém uma ligação normal; deixe em branco para o valor predefinido de 10485760 bytes.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Numele arhivei WARC:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Nume de baza optional pentru arhiva WARC; lasati gol pentru a-l denumi automat in directorul de iesire.
Inline assets as data: URIs (self-contained pages)
Încorporeaza resursele ca URI data: (pagini de sine statatoare)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Dupa terminarea copiei, fiecare pagina salvata este rescrisa cu foile de stil, scripturile, imaginile si fonturile încorporate, astfel încât pagina sa poata fi deschisa si singura.
Largest inlined asset (bytes):
Cea mai mare resursa încorporata (octeti):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
O resursa mai mare decât aceasta dimensiune pastreaza o legatura obisnuita; lasati gol pentru valoarea implicita de 10485760 de octeti.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Èìÿ WARC-àðõèâà:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Íåîáÿçàòåëüíîå áàçîâîå èìÿ WARC-àðõèâà; îñòàâüòå ïóñòûì äëÿ àâòîìàòè÷åñêîãî èìåíîâàíèÿ â âûõîäíîì êàòàëîãå.
Inline assets as data: URIs (self-contained pages)
Âñòðàèâàòü ðåñóðñû êàê data: URI (ñàìîñòîÿòåëüíûå ñòðàíèöû)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Ïîñëå çàâåðøåíèÿ çåðêàëèðîâàíèÿ êàæäàÿ ñîõðàí¸ííàÿ ñòðàíèöà ïåðåçàïèñûâàåòñÿ ñî âñòðîåííûìè òàáëèöàìè ñòèëåé, ñöåíàðèÿìè, èçîáðàæåíèÿìè è øðèôòàìè, ÷òîáû ñòðàíèöó ìîæíî áûëî îòêðûòü îòäåëüíî.
Largest inlined asset (bytes):
Íàèáîëüøèé âñòðàèâàåìûé ðåñóðñ (áàéòû):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Ðåñóðñ áîëüøå ýòîãî ðàçìåðà ñîõðàíÿåò îáû÷íóþ ññûëêó; îñòàâüòå ïóñòûì äëÿ çíà÷åíèÿ ïî óìîë÷àíèþ 10485760 áàéò.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Názov archívu WARC:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Voliteµný základný názov archívu WARC; ponechajte prázdne pre automatické pomenovanie vo výstupnom adresári.
Inline assets as data: URIs (self-contained pages)
Vlo¾i» zdroje ako URI data: (samostatné stránky)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Po dokonèení zrkadlenia prepísa» ka¾dú ulo¾enú stránku s vlo¾enými ¹týlmi, skriptmi, obrázkami a písmami, aby sa stránka dala otvori» aj samostatne.
Largest inlined asset (bytes):
Najväè¹í vlo¾ený zdroj (bajty):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Zdroj väè¹í ne¾ táto veµkos» si ponechá be¾ný odkaz; ponechajte prázdne pre predvolených 10485760 bajtov.

View File

@@ -964,11 +964,3 @@ WARC archive name:
Ime arhiva WARC:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Neobvezno osnovno ime arhiva WARC; pustite prazno za samodejno poimenovanje v izhodni mapi.
Inline assets as data: URIs (self-contained pages)
Vgradi vire kot data: URI (samostojne strani)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Ko je zrcaljenje koncano, se vsaka shranjena stran prepise z vgrajenimi slogovnimi listi, skripti, slikami in pisavami, tako da jo je mogoce odpreti tudi samostojno.
Largest inlined asset (bytes):
Najvecji vgrajeni vir (bajti):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Vir, vecji od te velikosti, ohrani obicajno povezavo; pustite prazno za privzetih 10485760 bajtov.

View File

@@ -964,11 +964,3 @@ WARC archive name:
WARC-arkivets namn:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Valfritt basnamn för WARC-arkivet; lämna tomt för att namnge det automatiskt i utdatakatalogen.
Inline assets as data: URIs (self-contained pages)
Bädda in resurser som data:-URI:er (fristående sidor)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
När speglingen är klar skrivs varje sparad sida om med sina formatmallar, skript, bilder och teckensnitt inbäddade, så att en sida också kan öppnas för sig.
Largest inlined asset (bytes):
Största inbäddade resurs (byte):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
En resurs över den här storleken behåller en vanlig länk; lämna tomt för standardvärdet 10485760 byte.

View File

@@ -964,11 +964,3 @@ WARC archive name:
WARC arþivi adý:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
WARC arþivi için isteðe baðlý temel ad; çýktý dizininde otomatik adlandýrma için boþ býrakýn.
Inline assets as data: URIs (self-contained pages)
Kaynaklarý data: URI olarak göm (kendi kendine yeten sayfalar)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Yansýlama bittiðinde, kaydedilen her sayfa stil sayfalarý, betikleri, görselleri ve yazý tipleri gömülü olarak yeniden yazýlýr; böylece sayfa tek baþýna da açýlabilir.
Largest inlined asset (bytes):
En büyük gömülü kaynak (bayt):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Bu boyutun üzerindeki bir kaynak sýradan baðlantýsýný korur; 10485760 baytlýk varsayýlan için boþ býrakýn.

View File

@@ -964,11 +964,3 @@ WARC archive name:
²ì'ÿ WARC-àðõ³âó:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
Íåîáîâ'ÿçêîâà áàçîâà íàçâà WARC-àðõ³âó; çàëèøòå ïîðîæí³ì äëÿ àâòîìàòè÷íîãî íàéìåíóâàííÿ ó âèõ³äíîìó êàòàëîç³.
Inline assets as data: URIs (self-contained pages)
Âáóäîâóâàòè ðåñóðñè ÿê data: URI (ñàìîñò³éí³ ñòîð³íêè)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
ϳñëÿ çàâåðøåííÿ äçåðêàëþâàííÿ êîæíà çáåðåæåíà ñòîð³íêà ïåðåçàïèñóºòüñÿ ç âáóäîâàíèìè òàáëèöÿìè ñòèë³â, ñöåíàð³ÿìè, çîáðàæåííÿìè òà øðèôòàìè, ùîá ñòîð³íêó ìîæíà áóëî â³äêðèòè îêðåìî.
Largest inlined asset (bytes):
Íàéá³ëüøèé âáóäîâàíèé ðåñóðñ (áàéòè):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Ðåñóðñ, á³ëüøèé çà öåé ðîçì³ð, çáåð³ãຠçâè÷àéíå ïîñèëàííÿ; çàëèøòå ïîðîæí³ì äëÿ òèïîâîãî çíà÷åííÿ 10485760 áàéò³â.

View File

@@ -964,11 +964,3 @@ WARC archive name:
WARC arxivi nomi:
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
WARC arxivi uchun ixtiyoriy asosiy nom; chiqish katalogida avtomatik nomlash uchun bo'sh qoldiring.
Inline assets as data: URIs (self-contained pages)
Resurslarni data: URI sifatida joylash (mustaqil sahifalar)
Once the mirror is finished, rewrite every saved page with its stylesheets, scripts, images and fonts embedded, so a page can also be opened on its own.
Nusxalash tugagach, har bir saqlangan sahifa uslublar jadvallari, skriptlar, rasmlar va shriftlar joylangan holda qayta yoziladi, shunda sahifani alohida ham ochish mumkin.
Largest inlined asset (bytes):
Eng katta joylangan resurs (bayt):
An asset above this size keeps an ordinary link; leave blank for the 10485760-byte default.
Bu olchamdan katta resurs oddiy havolani saqlab qoladi; standart 10485760 bayt uchun bosh qoldiring.

View File

@@ -3,7 +3,7 @@
.\"
.\" This file is generated by man/makeman.sh; do not edit by hand.
.\" SPDX-License-Identifier: GPL-3.0-or-later
.TH httrack 1 "26 July 2026" "httrack website copier"
.TH httrack 1 "23 July 2026" "httrack website copier"
.SH NAME
httrack \- offline browser : copy websites to a local directory
.SH SYNOPSIS
@@ -40,7 +40,6 @@ httrack \- offline browser : copy websites to a local directory
[ \fB\-%N, \-\-delayed\-type\-check\fR ]
[ \fB\-%D, \-\-cached\-delayed\-type\-check\fR ]
[ \fB\-%M, \-\-mime\-html\fR ]
[ \fB\-%Z, \-\-single\-file\fR ]
[ \fB\-LN, \-\-long\-names[=N]\fR ]
[ \fB\-KN, \-\-keep\-links[=N]\fR ]
[ \fB\-x, \-\-replace\-external\fR ]
@@ -199,8 +198,6 @@ delayed type check, don't make any link test but wait for files download to star
cached delayed type check, don't wait for remote type during updates, to speedup them (%D0 wait, * %D1 don't wait) (\-\-cached\-delayed\-type\-check)
.IP \-%M
generate a RFC MIME\-encapsulated full\-archive (.mht) (\-\-mime\-html)
.IP \-%Z
after the mirror, rewrite each saved page with its stylesheets, scripts, images and fonts inlined as data: URIs, so any page opens by double\-click anywhere (links between pages stay relative; audio and video stay links); \-\-single\-file\-max\-size N caps each asset (default 10485760 bytes). %M is the better container where a Chromium\-family browser is a given: one archive, no base64 tax on text, a shared asset stored once (\-\-single\-file)
.IP \-%t
keep the original file extension, don't rewrite it from the MIME type (%t0 rewrite)
.IP \-LN

View File

@@ -32,9 +32,7 @@ AM_LDFLAGS = \
bin_PROGRAMS = proxytrack httrack htsserver
httrack_SOURCES = httrack.c htsbacktrace.c htsbacktrace.h
# $(DL_LIBS): dladdr() in the crash handler, still in libdl on pre-2.34 glibc.
httrack_LDADD = $(THREADS_LIBS) $(DL_LIBS) libhttrack.la
httrack_LDADD = $(THREADS_LIBS) libhttrack.la
htsserver_LDADD = $(THREADS_LIBS) $(SOCKET_LIBS) libhttrack.la
proxytrack_LDADD = $(THREADS_LIBS) $(SOCKET_LIBS)
@@ -49,7 +47,6 @@ htsserver_LDFLAGS = $(AM_LDFLAGS) $(LDFLAGS_PIE)
lib_LTLIBRARIES = libhttrack.la
htsserver_SOURCES = htsserver.c htsserver.h htsweb.c htsweb.h \
htscmdline.c htscmdline.h \
htsurlport.c htsurlport.h
proxytrack_SOURCES = proxy/main.c \
proxy/proxytrack.c proxy/store.c \
@@ -63,21 +60,21 @@ whttrackrun_SCRIPTS = webhttrack
libhttrack_la_SOURCES = htscore.c htsparse.c htsback.c htscache.c \
htscache_selftest.c htsdns_selftest.c htsselftest.c \
htscatchurl.c htsfilters.c htsftp.c htshash.c coucal/coucal.c \
htscmdline.c htshelp.c htslib.c htsurlport.c htscoremain.c \
htshelp.c htslib.c htsurlport.c htscoremain.c \
htsname.c htsrobots.c htstools.c htswizard.c \
htsalias.c htsthread.c htsindex.c htsbauth.c \
htsmd5.c htscodec.c htswarc.c htssinglefile.c htsproxy.c htszlib.c htswrap.c htsconcat.c \
htsmd5.c htscodec.c htswarc.c htsproxy.c htszlib.c htswrap.c htsconcat.c \
htsmodules.c htscharset.c punycode.c htsencoding.c htssniff.c \
md5.c \
minizip/ioapi.c minizip/mztools.c minizip/unzip.c minizip/zip.c \
hts-indextmpl.h htsalias.h htsback.h htsbase.h htssafe.h \
htsbasenet.h htsbauth.h htscache.h htscache_selftest.h htsdns_selftest.h htsselftest.h htscatchurl.h \
htscmdline.h htsconfig.h htscore.h htsparse.h htscoremain.h htsdefines.h \
htsconfig.h htscore.h htsparse.h htscoremain.h htsdefines.h \
htsfilters.h htsftp.h htsglobal.h htshash.h coucal/coucal.h \
htshelp.h htsindex.h htslib.h htsurlport.h htsmd5.h \
htsmodules.h htsname.h htsnet.h htssniff.h \
htsopt.h htsrobots.h htsthread.h \
htstools.h htswizard.h htswrap.h htscodec.h htswarc.h htssinglefile.h htsproxy.h htszlib.h \
htstools.h htswizard.h htswrap.h htscodec.h htswarc.h htsproxy.h htszlib.h \
htsstrings.h htsarrays.h httrack-library.h \
htscharset.h punycode.h htsencoding.h \
htsentities.h htsentities.sh htsbasiccharsets.sh htscodepages.h \

View File

@@ -123,10 +123,6 @@ const char *hts_optalias[][4] = {
{"warc-cdxj", "-%rc", "single", ""},
{"wacz", "-%rz", "single",
"package the WARC archive, CDXJ index and pages as a WACZ file"},
{"single-file", "-%Z", "single",
"after the mirror, inline each page's assets as data: URIs"},
{"single-file-max-size", "-%Zs", "param1",
"per-asset cap for --single-file, in bytes (implies it; default 10485760)"},
{"why", "-%Y", "param1",
"explain which filter rule accepts or rejects a URL, then exit"},
{"pause", "-%G", "param1",

View File

@@ -125,22 +125,6 @@ void back_free(struct_back ** sback) {
above a normal handshake. The last candidate still gets the full timeout. */
#define HTS_CONNECT_FALLBACK_TIMEOUT 10
void back_read_ftp_result(FILE *fp, htsblk *r) {
size_t j = 0;
if (fscanf(fp, "%d ", &r->statuscode) != 1)
r->statuscode = STATUSCODE_INVALID;
// an external helper writes this file: stop at capacity, not at EOF
while (j + 1 < sizeof(r->msg)) {
const int c = fgetc(fp);
if (c == EOF)
break;
r->msg[j++] = (char) c;
}
r->msg[j] = '\0';
}
int back_connect_fallback_due(int addr_index, int addr_count, int elapsed,
int timeout) {
int deadline;
@@ -557,15 +541,9 @@ static int create_back_tmpfile(httrackp *opt, lien_back *const back,
const char *ext) {
// do not use tempnam() but a regular filename
back->tmpfile_buffer[0] = '\0';
if (back->url_sav[0] != '\0') {
/* same capacity as url_sav, so truncation drops the extension and aliases
the temp name onto the live file that back_finalize_backup() UNLINKs */
if (!sprintfbuff(back->tmpfile_buffer, "%s.%s", back->url_sav, ext)) {
hts_log_print(opt, LOG_WARNING, "temporary filename too long for %s",
back->url_sav);
back->tmpfile_buffer[0] = '\0';
return -1;
}
if (back->url_sav != NULL && back->url_sav[0] != '\0') {
snprintf(back->tmpfile_buffer, sizeof(back->tmpfile_buffer), "%s.%s",
back->url_sav, ext);
back->tmpfile = back->tmpfile_buffer;
if (structcheck(back->tmpfile) != 0) {
hts_log_print(opt, LOG_WARNING, "can not create directory to %s",
@@ -573,15 +551,8 @@ static int create_back_tmpfile(httrackp *opt, lien_back *const back,
return -1;
}
} else {
/* truncation here would collide distinct tmpnameid's onto one name */
if (!sprintfbuff(back->tmpfile_buffer, "%s/tmp%d.%s",
StringBuff(opt->path_html_utf8), opt->state.tmpnameid++,
ext)) {
hts_log_print(opt, LOG_WARNING, "temporary filename too long in %s",
StringBuff(opt->path_html_utf8));
back->tmpfile_buffer[0] = '\0';
return -1;
}
snprintf(back->tmpfile_buffer, sizeof(back->tmpfile_buffer), "%s/tmp%d.%s",
StringBuff(opt->path_html_utf8), opt->state.tmpnameid++, ext);
back->tmpfile = back->tmpfile_buffer;
}
/* OK */
@@ -916,7 +887,8 @@ int back_finalize(httrackp * opt, cache_back * cache, struct_back * sback,
HTS_STAT.stat_bytes += back[p].r.size;
HTS_STAT.stat_files++;
hts_log_print(opt, LOG_TRACE, "added file %s%s => %s",
back[p].url_adr, back[p].url_fil, back[p].url_sav);
back[p].url_adr, back[p].url_fil,
back[p].url_sav != NULL ? back[p].url_sav : "");
}
if ((!back[p].r.notmodified) && (opt->is_update)) {
HTS_STAT.stat_updated_files++; // page modifiée
@@ -2330,24 +2302,26 @@ int back_add(struct_back *sback, httrackp *opt, cache_back *cache,
&& slot_can_be_finalized(opt, &back[i]);
int may_serialize = slot_can_be_cached_on_disk(&back[i]);
hts_log_print(
opt, LOG_DEBUG,
"back[%03d]: may_clean=%d, may_finalize_disk=%d, "
"may_serialize=%d:" LF "\t"
"finalized(%d), status(%d), locked(%d), delayed(%d), "
"test(%d), " LF "\t"
"statuscode(%d), size(%d), is_write(%d), may_hypertext(%d), " LF
"\t"
"contenttype(%s), url(%s%s), save(%s)",
i, may_clean, may_finalize, may_serialize, back[i].finalized,
back[i].status, back[i].locked, IS_DELAYED_EXT(back[i].url_sav),
back[i].testmode, back[i].r.statuscode, (int) back[i].r.size,
back[i].r.is_write,
may_be_hypertext_mime(opt, back[i].r.contenttype,
back[i].url_fil),
/* */
back[i].r.contenttype, back[i].url_adr, back[i].url_fil,
back[i].url_sav);
hts_log_print(opt, LOG_DEBUG,
"back[%03d]: may_clean=%d, may_finalize_disk=%d, may_serialize=%d:"
LF "\t"
"finalized(%d), status(%d), locked(%d), delayed(%d), test(%d), "
LF "\t"
"statuscode(%d), size(%d), is_write(%d), may_hypertext(%d), "
LF "\t" "contenttype(%s), url(%s%s), save(%s)", i,
may_clean, may_finalize, may_serialize,
back[i].finalized, back[i].status, back[i].locked,
IS_DELAYED_EXT(back[i].url_sav), back[i].testmode,
back[i].r.statuscode, (int) back[i].r.size,
back[i].r.is_write, may_be_hypertext_mime(opt,
back[i].r.
contenttype,
back[i].
url_fil),
/* */
back[i].r.contenttype, back[i].url_adr,
back[i].url_fil,
back[i].url_sav ? back[i].url_sav : "<null>");
}
}
}
@@ -2883,8 +2857,7 @@ void back_wait(struct_back * sback, httrackp * opt, cache_back * cache,
// new session
back[i].r.ssl_con = SSL_new(openssl_ctx);
if (back[i].r.ssl_con) {
/* non-const twin: the OpenSSL macro casts the qualifier away */
char *hostname = jump_protocol(back[i].url_adr);
const char* hostname = jump_protocol_const(back[i].url_adr);
// some servers expect the hostname on the clienthello (SNI TLS extension)
SSL_set_tlsext_host_name(back[i].r.ssl_con, hostname);
SSL_clear(back[i].r.ssl_con);
@@ -2976,7 +2949,7 @@ void back_wait(struct_back * sback, httrackp * opt, cache_back * cache,
back[i].r.msg[0] = '\0';
strncatbuff(back[i].r.msg, tmp, sizeof(back[i].r.msg) - 2);
if (!strnotempty(back[i].r.msg)) {
htsblk_failf(&back[i].r, "SSL/TLS error %d", err_code);
sprintf(back[i].r.msg, "SSL/TLS error %d", err_code);
}
deletehttp(&back[i].r);
back[i].r.soc = INVALID_SOCKET;
@@ -3049,7 +3022,16 @@ void back_wait(struct_back * sback, httrackp * opt, cache_back * cache,
FOPEN(fconcat(OPT_GET_BUFF(opt), back[i].location_buffer, ".ok"),
"rb");
if (fp) {
back_read_ftp_result(fp, &back[i].r);
int j = 0;
fscanf(fp, "%d ", &(back[i].r.statuscode));
while(!feof(fp)) {
int c = fgetc(fp);
if (c != EOF)
back[i].r.msg[j++] = c;
}
back[i].r.msg[j++] = '\0';
fclose(fp);
UNLINK(fconcat(OPT_GET_BUFF(opt), back[i].location_buffer, ".ok"));
strcpybuff(fconcat
@@ -3350,11 +3332,10 @@ void back_wait(struct_back * sback, httrackp * opt, cache_back * cache,
deleteaddr(&back[i].r);
if (back[i].r.size < back[i].r.totalsize)
back[i].r.statuscode = STATUSCODE_CONNERROR; // recatch
htsblk_failf(&back[i].r,
"Incorrect length (" LLintP " Bytes, " LLintP
" expected)",
(LLint) back[i].r.size,
(LLint) back[i].r.totalsize);
sprintf(back[i].r.msg,
"Incorrect length (" LLintP " Bytes, " LLintP
" expected)", (LLint) back[i].r.size,
(LLint) back[i].r.totalsize);
} else {
// Un warning suffira..
hts_log_print(opt, LOG_WARNING,

View File

@@ -74,10 +74,6 @@ void back_free(struct_back ** sback);
// backing
#define BACK_ADD_TEST "(dummy)"
#define BACK_ADD_TEST2 "(dummy2)"
/* Parse an external FTP helper's "<statuscode> <message>" result file into r,
clipping the message to r->msg. */
void back_read_ftp_result(FILE *fp, htsblk *r);
int back_index(httrackp * opt, struct_back * sback, const char *adr, const char *fil,
const char *sav);
int back_available(const struct_back * sback);

View File

@@ -1,238 +0,0 @@
/* ------------------------------------------------------------ */
/*
HTTrack Website Copier, Offline Browser for Windows and Unix
Copyright (C) 1998 Xavier Roche and other contributors
SPDX-License-Identifier: GPL-3.0-or-later
This program is free software: you can redistribute it and/or modify
it under the terms of the GNU General Public License as published by
the Free Software Foundation, either version 3 of the License, or
(at your option) any later version.
This program is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
GNU General Public License for more details.
You should have received a copy of the GNU General Public License
along with this program. If not, see <http://www.gnu.org/licenses/>.
Ethical use: we kindly ask that you NOT use this software to harvest email
addresses or to collect any other private information about people. Doing so
would dishonor our work and waste the many hours we have spent on it.
Please visit our Website: http://www.httrack.com
*/
/* ------------------------------------------------------------ */
/* File: crash backtrace printer */
/* Author: Xavier Roche */
/* ------------------------------------------------------------ */
/* Before every header: glibc gates dladdr() on it. */
#if defined(__linux) && !defined(_GNU_SOURCE)
#define _GNU_SOURCE
#endif
#include "htsbacktrace.h"
#include "htsglobal.h"
#include <stdlib.h>
#include <string.h>
#ifdef _WIN32
#include <io.h> /* write */
#endif
#ifdef HAVE_UNISTD_H
#include <unistd.h>
#endif
#if (defined(__linux) && defined(HAVE_EXECINFO_H))
#include <dlfcn.h>
#include <errno.h>
#include <execinfo.h>
#include <signal.h>
#include <time.h>
#include <sys/wait.h>
#define USES_BACKTRACE
#endif
#ifdef USES_BACKTRACE
#define BT_MAX_FRAMES 64 /* frames we try to name */
#define BT_MAX_MODULES 8 /* distinct modules, one child each */
#define BT_HEX_SIZE 19 /* "0x" + 16 nibbles + NUL */
#define BT_PATH_SIZE 1024 /* module path; longer is skipped */
#define BT_WAIT_TICKS 300 /* 10ms ticks, shared: cap a slow child */
#define BT_NO_SYMBOLIZER 127 /* child exit: execvp() found none */
static hts_boolean symbolize_crash = HTS_TRUE;
/* "0x"-prefixed hex: the handler must stay stdio-free. */
static void print_hex(char *buffer, uintptr_t value) {
static const char digits[] = "0123456789abcdef";
size_t i = 2, a, b;
buffer[0] = '0';
buffer[1] = 'x';
do {
buffer[i++] = digits[value & 0xf];
value >>= 4;
} while (value != 0);
buffer[i] = '\0';
for (a = 2, b = i - 1; a < b; a++, b--) {
const char c = buffer[a];
buffer[a] = buffer[b];
buffer[b] = c;
}
}
/* HTS_FALSE if src does not fit: a truncated module path would point the
symbolizer at the wrong file. */
static hts_boolean copy_bounded(char *dest, size_t size, const char *src) {
size_t i;
for (i = 0; i < size - 1 && src[i] != '\0'; i++) {
dest[i] = src[i];
}
dest[i] = '\0';
return src[i] == '\0' ? HTS_TRUE : HTS_FALSE;
}
/* Run the symbolizer on argv, output on fd, within *budget ticks. HTS_FALSE
only if none could be run at all; otherwise silent, the raw trace stands. */
static hts_boolean spawn_symbolizer(char **argv, int fd, int *budget) {
const pid_t pid = fork();
int status = 0;
if (pid == -1)
return HTS_FALSE;
if (pid == 0) {
static char llvm_prog[] = "llvm-symbolizer";
static char llvm_opts[] = "-p";
dup2(fd, 1); /* both symbolizers write on stdout */
execvp(argv[0], argv);
argv[0] = llvm_prog; /* an LLVM-only install ships no addr2line */
argv[1] = llvm_opts;
execvp(argv[0], argv);
_exit(BT_NO_SYMBOLIZER);
}
for (; *budget > 0; (*budget)--) {
const struct timespec tick = {0, 10 * 1000 * 1000};
const pid_t reaped = waitpid(pid, &status, WNOHANG);
if (reaped == pid)
return WIFEXITED(status) && WEXITSTATUS(status) == BT_NO_SYMBOLIZER
? HTS_FALSE
: HTS_TRUE;
if (reaped == -1 && errno != EINTR)
return HTS_TRUE;
nanosleep(&tick, NULL);
}
kill(pid, SIGKILL);
waitpid(pid, NULL, 0);
return HTS_TRUE;
}
/* Name the frames backtrace_symbols_fd() leaves as module+offset:
-fvisibility=hidden keeps them out of .dynsym, but DWARF has them. dladdr()
is not formally async-signal-safe; accepted, this path is already fatal. */
static void symbolize_backtrace(void *const *stack, int size, int fd) {
static char prog[] = "addr2line";
static char opts[] = "-Cfipa";
static char dashe[] = "-e";
char hex[BT_MAX_FRAMES][BT_HEX_SIZE];
const void *base[BT_MAX_FRAMES];
const char *name[BT_MAX_FRAMES];
hts_boolean grouped[BT_MAX_FRAMES];
char module[BT_PATH_SIZE];
char *argv[4 + BT_MAX_FRAMES + 1];
int budget = BT_WAIT_TICKS;
int i, spawned;
if (size > BT_MAX_FRAMES)
size = BT_MAX_FRAMES;
for (i = 0; i < size; i++) {
Dl_info info;
grouped[i] = HTS_TRUE; /* skipped unless dladdr() places the frame */
if (dladdr(stack[i], &info) == 0 || info.dli_fname == NULL ||
info.dli_fname[0] == '\0')
continue;
base[i] = info.dli_fbase;
name[i] = info.dli_fname;
print_hex(hex[i], (uintptr_t) ((const char *) stack[i] -
(const char *) info.dli_fbase));
grouped[i] = HTS_FALSE;
}
/* One child per module: addr2line takes a single -e. Each frame is claimed
once, so argc cannot exceed argv[]. */
for (spawned = 0; spawned < BT_MAX_MODULES; spawned++) {
int first, j, argc = 0;
for (first = 0; first < size && grouped[first]; first++)
;
if (first >= size)
break;
argv[argc++] = prog;
argv[argc++] = opts;
argv[argc++] = dashe;
argv[argc++] = module;
for (j = first; j < size; j++) {
if (grouped[j] || base[j] != base[first])
continue;
grouped[j] = HTS_TRUE;
argv[argc++] = hex[j];
}
argv[argc] = NULL;
/* access(): skip pseudo-modules like linux-vdso, which have no file and
would draw nothing but an addr2line complaint. */
if (copy_bounded(module, sizeof(module), name[first]) &&
access(module, R_OK) == 0) {
const size_t len = strlen(module);
/* addr2line -a prints offsets only: say which module they are in. */
(void) (write(fd, module, len) == (ssize_t) len);
(void) (write(fd, ":\n", 2) == 2);
if (!spawn_symbolizer(argv, fd, &budget))
break; /* no symbolizer: stop at one header */
}
}
}
#endif
void hts_backtrace_init(void) {
#ifdef USES_BACKTRACE
symbolize_crash =
getenv("HTTRACK_NO_SYMBOLIZE") == NULL ? HTS_TRUE : HTS_FALSE;
#endif
}
void hts_print_backtrace(int fd) {
#ifdef USES_BACKTRACE
void *stack[256];
const int size = backtrace(stack, sizeof(stack) / sizeof(stack[0]));
/* A fault inside the handler lands back here: symbolizing twice interleaves
two traces on fd and spends a second budget. */
static volatile sig_atomic_t entered = 0;
if (size != 0) {
backtrace_symbols_fd(stack, size, fd);
if (symbolize_crash && entered == 0) {
entered = 1;
symbolize_backtrace(stack, size, fd);
entered = 0;
}
}
#else
const char msg[] = "No stack trace available on this OS :(\n";
if (write(fd, msg, sizeof(msg) - 1) != sizeof(msg) - 1) {
/* sorry GCC */
}
#endif
}

View File

@@ -1,45 +0,0 @@
/* ------------------------------------------------------------ */
/*
HTTrack Website Copier, Offline Browser for Windows and Unix
Copyright (C) 1998 Xavier Roche and other contributors
SPDX-License-Identifier: GPL-3.0-or-later
This program is free software: you can redistribute it and/or modify
it under the terms of the GNU General Public License as published by
the Free Software Foundation, either version 3 of the License, or
(at your option) any later version.
This program is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
GNU General Public License for more details.
You should have received a copy of the GNU General Public License
along with this program. If not, see <http://www.gnu.org/licenses/>.
Ethical use: we kindly ask that you NOT use this software to harvest email
addresses or to collect any other private information about people. Doing so
would dishonor our work and waste the many hours we have spent on it.
Please visit our Website: http://www.httrack.com
*/
/* ------------------------------------------------------------ */
/* File: crash backtrace printer */
/* Author: Xavier Roche */
/* ------------------------------------------------------------ */
#ifndef HTSBACKTRACE_DEFH
#define HTSBACKTRACE_DEFH
/* Sample HTTRACK_NO_SYMBOLIZE before any crash: getenv() is not signal-safe.
Call once, from the process that installs the fatal-signal handlers. */
void hts_backtrace_init(void);
/* Write the calling thread's stack to fd, callable from a fatal signal handler:
raw frames first, then whatever an external symbolizer can name. Allocates
nothing; prints a one-line notice where the OS has no backtrace(). */
void hts_print_backtrace(int fd);
#endif

View File

@@ -629,8 +629,8 @@ static htsblk cache_readex_new(httrackp * opt, cache_back * cache,
// File exists on disk with declared cache name (this is expected!)
if (fexist_utf8(fconv(catbuff, sizeof(catbuff), previous_save))) { // un fichier existe déja
// Expected size ?
const LLint fsize = fsize_utf8(
fconv(catbuff, sizeof(catbuff), previous_save));
const size_t fsize =
fsize_utf8(fconv(catbuff, sizeof(catbuff), previous_save));
if (fsize == r.size) {
// Target name is the previous name, and the file looks good: nothing to do!
if (strcmp(previous_save, target_save) == 0) {
@@ -666,8 +666,7 @@ static htsblk cache_readex_new(httrackp * opt, cache_back * cache,
// Suppose a broken mirror, with a file being renamed: OK
else if (fexist_utf8(fconv(catbuff, sizeof(catbuff), target_save))) {
// Expected size ?
const LLint fsize =
fsize_utf8(fconv(catbuff, sizeof(catbuff), target_save));
const size_t fsize = fsize_utf8(fconv(catbuff, sizeof(catbuff), target_save));
if (fsize == r.size) {
// So far so good
@@ -1295,17 +1294,12 @@ char *readfile2(const char *fil, LLint * size) {
}
/* Note: utf-8 */
char *readfile_utf8(const char *fil) { return readfile2_utf8(fil, NULL); }
/* Note: utf-8 */
char *readfile2_utf8(const char *fil, LLint *size) {
char *readfile_utf8(const char *fil) {
char *adr = NULL;
char catbuff[CATBUFF_SIZE];
const LLint len = fsize_utf8(fil);
const size_t buflen = len >= 0 ? llint_to_size_t(len) : (size_t) -1;
if (size != NULL)
*size = len;
if (buflen != (size_t) -1) { // exists, and is addressable (see readfile2)
FILE *const fp = FOPEN(fconv(catbuff, sizeof(catbuff), fil), "rb");
@@ -1446,7 +1440,7 @@ int cache_brstr(char *adr, char *s, size_t s_size) {
/* binput bounded to a NUL-terminated buffer: refuse to start a read at or
past `end`, so a prior over-advance can't walk a cache-index parse OOB. */
int cache_binput(const char *adr, const char *end, char *s, int max) {
int cache_binput(char *adr, const char *end, char *s, int max) {
if (adr >= end) {
s[0] = '\0';
return 0;

View File

@@ -93,7 +93,7 @@ void cache_rstr(FILE *fp, char *s, size_t s_size);
char *cache_rstr_addr(FILE * fp);
int cache_brstr(char *adr, char *s, size_t s_size);
/* binput over a NUL-terminated buffer, bounded: no read starts at/past end. */
int cache_binput(const char *adr, const char *end, char *s, int max);
int cache_binput(char *adr, const char *end, char *s, int max);
int cache_brint(char *adr, int *i);
void cache_rint(FILE * fp, int *i);
void cache_rLLint(FILE * fp, LLint * i);

View File

@@ -390,10 +390,6 @@ HTSEXT_API char *hts_convertStringSystemToUTF8(const char *s, size_t size) {
return hts_convertStringCPToUTF8(s, size, GetACP());
}
HTSEXT_API char *hts_convertStringUTF8ToSystem(const char *s, size_t size) {
return hts_convertStringCPFromUTF8(s, size, GetACP());
}
HTSEXT_API void hts_argv_utf8(int *pargc, char ***pargv) {
int wargc = 0;
LPWSTR *const wargv = CommandLineToArgvW(GetCommandLineW(), &wargc);

View File

@@ -184,12 +184,6 @@ extern LPWSTR hts_pathToUCS2(const char *path);
**/
HTSEXT_API char *hts_convertStringSystemToUTF8(const char *s, size_t size);
/**
* Convert UTF-8 to the current system codepage. Caller frees; NULL upon error.
* This function is WIN32 specific.
**/
HTSEXT_API char *hts_convertStringUTF8ToSystem(const char *s, size_t size);
/**
* Replace the CRT's ANSI argv by a UTF-8 one decoded from the real UTF-16
* command line: every char* is UTF-8 on Windows (FOPEN, STAT, ... convert at

View File

@@ -1,95 +0,0 @@
/* ------------------------------------------------------------ */
/*
HTTrack Website Copier, Offline Browser for Windows and Unix
Copyright (C) 1998 Xavier Roche and other contributors
SPDX-License-Identifier: GPL-3.0-or-later
This program is free software: you can redistribute it and/or modify
it under the terms of the GNU General Public License as published by
the Free Software Foundation, either version 3 of the License, or
(at your option) any later version.
This program is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
GNU General Public License for more details.
You should have received a copy of the GNU General Public License
along with this program. If not, see <http://www.gnu.org/licenses/>.
Ethical use: we kindly ask that you NOT use this software to harvest email
addresses or to collect any other private information about people. Doing so
would dishonor our work and waste the many hours we have spent on it.
Please visit our Website: http://www.httrack.com
*/
/* ------------------------------------------------------------ */
/* File: command line splitter, shared by the engine and */
/* htsserver */
/* Author: Xavier Roche */
/* ------------------------------------------------------------ */
#include "htscmdline.h"
#include "htssafe.h"
#include <limits.h>
#include <stdint.h>
char **hts_split_cmdline(char *cmd, int *nargs) {
size_t nsep = 0;
size_t capacity;
size_t r;
size_t w;
int argc = 0;
hts_boolean quoted = HTS_FALSE;
char **argv;
*nargs = 0;
/* fold the other separators, so counting them sizes the vector exactly */
for (r = 0; cmd[r] != '\0'; r++) {
if (cmd[r] == '\t' || cmd[r] == '\r' || cmd[r] == '\n') {
cmd[r] = ' ';
}
if (cmd[r] == ' ') {
nsep++;
}
}
/* at most one argument per separator, plus the leading one and the NULL */
if (nsep > (size_t) INT_MAX - 1 || nsep > SIZE_MAX / sizeof(char *) - 2) {
return NULL;
}
capacity = nsep + 2;
argv = (char **) malloct(capacity * sizeof(char *));
if (argv == NULL) {
return NULL;
}
argv[argc++] = cmd;
for (r = 0, w = 0; cmd[r] != '\0';) {
if (quoted && cmd[r] == '\\' &&
(cmd[r + 1] == '\\' || cmd[r + 1] == '\"')) {
r++;
cmd[w++] = cmd[r++];
} else if (cmd[r] == '\"') {
quoted = !quoted;
cmd[w++] = cmd[r++];
} else if (cmd[r] == ' ' && !quoted) {
cmd[w++] = '\0';
assertf((size_t) argc < capacity - 1); /* the last slot holds the NULL */
argv[argc++] = cmd + w;
r++;
} else {
cmd[w++] = cmd[r++];
}
}
cmd[w] = '\0';
argv[argc] = NULL; /* callers may rely on argv[argc] == NULL */
*nargs = argc;
return argv;
}

View File

@@ -1,45 +0,0 @@
/* ------------------------------------------------------------ */
/*
HTTrack Website Copier, Offline Browser for Windows and Unix
Copyright (C) 1998 Xavier Roche and other contributors
SPDX-License-Identifier: GPL-3.0-or-later
This program is free software: you can redistribute it and/or modify
it under the terms of the GNU General Public License as published by
the Free Software Foundation, either version 3 of the License, or
(at your option) any later version.
This program is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
GNU General Public License for more details.
You should have received a copy of the GNU General Public License
along with this program. If not, see <http://www.gnu.org/licenses/>.
Ethical use: we kindly ask that you NOT use this software to harvest email
addresses or to collect any other private information about people. Doing so
would dishonor our work and waste the many hours we have spent on it.
Please visit our Website: http://www.httrack.com
*/
/* ------------------------------------------------------------ */
/* File: command line splitter, shared by the engine and */
/* htsserver */
/* Author: Xavier Roche */
/* ------------------------------------------------------------ */
#ifndef HTSCMDLINE_DEFH
#define HTSCMDLINE_DEFH
#include "htsglobal.h"
/* Split "cmd" in place into a NULL-terminated argv vector of *nargs entries,
argv[0] being the program name and quotes left for the engine to strip.
Returns a malloct'ed vector of pointers into cmd (freet the vector, never its
entries), or NULL when it cannot be sized or allocated. */
char **hts_split_cmdline(char *cmd, int *nargs);
#endif

View File

@@ -40,7 +40,6 @@ Please visit our Website: http://www.httrack.com
/* File defs */
#include "htscore.h"
#include "htswarc.h"
#include "htssinglefile.h"
/* specific definitions */
#include "htsbase.h"
@@ -448,22 +447,13 @@ void hts_finish_makeindex(httrackp *opt, int *makeindex_done,
const char *fil) {
if (!*makeindex_done) {
if (*makeindex_fp) {
/* sized off link_escaped below: at the old flat 1024 a long first link
produced a redirect to a clipped URL */
char BIGSTK tempo[HTS_URLMAXSIZE * 2 + 64];
char BIGSTK tempo[1024];
if (makeindex_links == 1) {
char BIGSTK link_escaped[HTS_URLMAXSIZE * 2];
escape_uri_utf(makeindex_firstlink, link_escaped, sizeof(link_escaped));
/* no redirect beats one pointing at a clipped URL */
if (!sprintfbuff(
tempo,
"<meta HTTP-EQUIV=\"Refresh\" CONTENT=\"0; URL=%s\">" CRLF,
link_escaped)) {
hts_log_print(opt, LOG_WARNING,
"index redirect omitted: first link too long (%s)",
makeindex_firstlink);
tempo[0] = '\0';
}
snprintf(tempo, sizeof(tempo),
"<meta HTTP-EQUIV=\"Refresh\" CONTENT=\"0; URL=%s\">" CRLF,
link_escaped);
} else
tempo[0] = '\0';
hts_template_format(*makeindex_fp, template_footer,
@@ -706,19 +696,17 @@ int httpmirror(char *url1, httrackp * opt) {
// copier adresse(s) dans liste des adresses
{
char *a = url1;
const LLint list_sz = StringNotEmpty(opt->filelist)
? fsize_utf8(StringBuff(opt->filelist))
: 0;
/* two bytes reserved per list byte; -1 makes an undoublable size refused */
const LLint list_room =
list_sz > 0 ? (list_sz <= INT64_MAX / 2 ? list_sz * 2 : -1) : 0;
const size_t primary_len =
llint_grow_size_t(8192 + strlen(url1) * 2, list_room, 0);
int primary_len = 8192;
if (StringNotEmpty(opt->filelist)) {
primary_len += max(0, fsize_utf8(StringBuff(opt->filelist)) * 2);
}
primary_len += (int) strlen(url1) * 2;
// création de la première page, qui contient les liens de base à scanner
// c'est plus propre et plus logique que d'entrer à la main les liens dans la pile
// on bénéficie ainsi des vérifications et des tests du robot pour les liens "primaires"
primary = primary_len != (size_t) -1 ? (char *) malloct(primary_len) : NULL;
primary = (char *) malloct(primary_len);
if (!primary) {
printf("PANIC! : Not enough memory [%d]\n", __LINE__);
XH_extuninit;
@@ -899,7 +887,7 @@ int httpmirror(char *url1, httrackp * opt) {
}
if (filelist_buff != NULL) {
size_t filelist_ptr = 0;
int filelist_ptr = 0;
int n = 0;
char BIGSTK line[HTS_URLMAXSIZE * 2];
@@ -2183,10 +2171,6 @@ int httpmirror(char *url1, httrackp * opt) {
}
// fin purge!
/* --single-file: inline each page's assets now the tree is final, after the
purge has deleted whatever this run dropped. */
singlefile_process_mirror(opt);
// Indexation
if (opt->kindex)
index_finish(StringBuff(opt->path_html), opt->kindex);
@@ -3648,10 +3632,6 @@ HTSEXT_API int copy_htsopt(const httrackp * from, httrackp * to) {
to->warc_cdx = from->warc_cdx;
to->warc_wacz = from->warc_wacz;
to->single_file = from->single_file;
if (from->single_file_max_size > 0)
to->single_file_max_size = from->single_file_max_size;
if (from->pause_max_ms > 0) {
to->pause_min_ms = from->pause_min_ms;
to->pause_max_ms = from->pause_max_ms;

View File

@@ -409,8 +409,6 @@ char *readfile2(const char *fil, LLint * size);
char *readfile_utf8(const char *fil);
char *readfile2_utf8(const char *fil, LLint *size);
char *readfile_or(const char *fil, const char *defaultdata);
/* Backing (download-slot) scheduler. Operate on the back[] ring (struct_back).

View File

@@ -145,7 +145,7 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
int argv_url = -1; // ==0 : utiliser cache et doit.log
char *argv_firsturl = NULL; // utilisé pour nommage par défaut
char *url = NULL; // URLS séparées par un espace
size_t url_sz = 65535;
int url_sz = 65535;
// the parametres
int httrack_logmode = 3; // ONE log file
@@ -224,22 +224,21 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
/* create x_argvblk buffer for transformed command line */
{
size_t current_size = 0;
const LLint size = fsize("config");
size_t blk_size;
int current_size = 0;
int size;
int na;
for(na = 0; na < argc; na++)
current_size += strlen(argv[na]) + 1;
/* a huge file named "config" must saturate, not wrap, the capacity */
blk_size = llint_grow_size_t(current_size, size > 0 ? size : 0, 32768);
x_argvblk = blk_size != (size_t) -1 ? (char *) malloct(blk_size) : NULL;
current_size += (int) (strlen(argv[na]) + 1);
if ((size = fsize("config")) > 0)
current_size += size;
x_argvblk = (char *) malloct(current_size + 32768);
if (x_argvblk == NULL) {
HTS_PANIC_PRINTF("Error, not enough memory");
htsmain_free();
return -1;
}
x_argvblk_size = blk_size;
x_argvblk_size = (size_t) (current_size + 32768);
x_argvblk[0] = '\0';
x_ptr = 0;
@@ -1457,29 +1456,20 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
FILE *fp = FOPEN(argv[na], "rb");
if (fp != NULL) {
size_t cl = strlen(url);
const size_t fzs = llint_to_size_t(fz);
const size_t capa = llint_grow_size_t(cl, fz, 8192);
int cl = (int) strlen(url);
if (capa == (size_t) -1) {
fclose(fp);
HTS_PANIC_PRINTF("File url list too large");
htsmain_free();
return -1;
}
ensureUrlCapacity(url, url_sz, capa);
ensureUrlCapacity(url, url_sz, cl + fz + 8192);
if (cl > 0) { /* don't stick! (3.43) */
url[cl] = ' ';
cl++;
}
if (fread(url + cl, 1, fzs, fp) != fzs) {
fclose(fp);
if (fread(url + cl, 1, fz, fp) != fz) {
HTS_PANIC_PRINTF("File url list could not be read");
htsmain_free();
return -1;
}
fclose(fp);
*(url + cl + fzs) = '\0';
*(url + cl + fz) = '\0';
}
}
}
@@ -1795,36 +1785,6 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
StringCopy(opt->warc_file, WARC_AUTONAME);
}
break;
case 'Z': // single-file: inline each page's assets as data: URIs
if (*(com + 1) == 's') { // --single-file-max-size N
com++;
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
HTS_PANIC_PRINTF(
"Option single-file-max-size needs a blank "
"space and a size");
htsmain_free();
return -1;
}
na++;
{ // reject non-numeric/negative/overflow; keep the default
char *end;
LLint v;
errno = 0;
v = strtoll(argv[na], &end, 10);
if (isdigit((unsigned char) argv[na][0]) && *end == '\0' &&
errno != ERANGE && v > 0)
opt->single_file_max_size = v;
}
opt->single_file = HTS_TRUE;
} else {
opt->single_file = HTS_TRUE;
if (*(com + 1) == '0') {
opt->single_file = HTS_FALSE;
com++;
}
}
break;
case 'Y': // why: explain the filter verdict for a URL, no crawl
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
HTS_PANIC_PRINTF("Option why needs a blank space and a URL");
@@ -2336,8 +2296,8 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
} else { // URL/filters
char catbuff[CATBUFF_SIZE];
const size_t urlSize = strlen(argv[na]);
const size_t capa = strlen(url) + urlSize + 32;
const int urlSize = (int) strlen(argv[na]);
const int capa = (int) (strlen(url) + urlSize + 32);
assertf(urlSize < HTS_URLMAXSIZE);
if (urlSize < HTS_URLMAXSIZE) {

View File

@@ -251,7 +251,7 @@ int run_launch_ftp(FTPDownloadStruct * pStruct) {
// folding a nonsense port into 1..65535 fetches one the link never named;
// an empty "host:" just means the default (#614)
if (a[1] != '\0' && !hts_parse_url_port(a + 1, &port)) {
htsblk_failf(&back->r, "Invalid port: %s", a + 1);
snprintf(back->r.msg, sizeof(back->r.msg), "Invalid port: %s", a + 1);
back->r.statuscode = STATUSCODE_INVALID; // permanent, unlike a DNS miss
_HALT_FTP return 0;
}
@@ -262,7 +262,8 @@ int run_launch_ftp(FTPDownloadStruct * pStruct) {
// récupérer adresse résolue
strcpybuff(back->info, "host name");
if (hts_dns_resolve2(opt, _adr, &server, &error) == NULL) {
htsblk_failf(&back->r, "Unable to get server's address: %s", error);
snprintf(back->r.msg, sizeof(back->r.msg),
"Unable to get server's address: %s", error);
back->r.statuscode = STATUSCODE_NON_FATAL;
_HALT_FTP return 0;
}
@@ -331,15 +332,18 @@ int run_launch_ftp(FTPDownloadStruct * pStruct) {
}
} else {
htsblk_failf(&back->r, "Bad password: %s", linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg), "Bad password: %s",
linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
}
} else {
htsblk_failf(&back->r, "Bad user name: %s", linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg), "Bad user name: %s",
linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
}
} else {
htsblk_failf(&back->r, "Connection refused: %s", linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg), "Connection refused: %s",
linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
}
@@ -406,7 +410,8 @@ int run_launch_ftp(FTPDownloadStruct * pStruct) {
}
// -- fin analyse de l'adresse IP et du port --
} else {
htsblk_failf(&back->r, "PASV incorrect: %s", linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg), "PASV incorrect: %s",
linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
} // sinon on est prêts
} else {
@@ -437,11 +442,13 @@ int run_launch_ftp(FTPDownloadStruct * pStruct) {
}
}
} else {
htsblk_failf(&back->r, "EPSV incorrect: %s", linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg), "EPSV incorrect: %s",
linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
}
} else {
htsblk_failf(&back->r, "PASV/EPSV error: %s", linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg), "PASV/EPSV error: %s",
linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
} // sinon on est prêts
}
@@ -547,8 +554,8 @@ int run_launch_ftp(FTPDownloadStruct * pStruct) {
deletesoc(soc_dat);
soc_dat = INVALID_SOCKET;
//
htsblk_failf(&back->r, "RETR command error: %s",
linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg),
"RETR command error: %s", linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
} // sinon on est prêts
} else {
@@ -566,12 +573,13 @@ int run_launch_ftp(FTPDownloadStruct * pStruct) {
back->r.statuscode = STATUSCODE_INVALID;
} // sinon on est prêts
} else {
htsblk_failf(&back->r, "Unable to resolve IP %s: %s", adr_ip,
error);
snprintf(back->r.msg, sizeof(back->r.msg),
"Unable to resolve IP %s: %s", adr_ip, error);
back->r.statuscode = STATUSCODE_INVALID;
} // sinon on est prêts
} else {
htsblk_failf(&back->r, "PASV incorrect: %s", linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg), "PASV incorrect: %s",
linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
} // sinon on est prêts
#else
@@ -595,11 +603,13 @@ int run_launch_ftp(FTPDownloadStruct * pStruct) {
back->r.statuscode = STATUSCODE_INVALID;
}
} else {
htsblk_failf(&back->r, "RETR command error: %s", linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg),
"RETR command error: %s", linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
}
} else {
htsblk_failf(&back->r, "PORT command error: %s", linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg), "PORT command error: %s",
linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
}
#ifdef _WIN32
@@ -642,7 +652,8 @@ int run_launch_ftp(FTPDownloadStruct * pStruct) {
len = 0; // fin
break;
case 0:
htsblk_failf(&back->r, "Time out (%d)", timeout);
snprintf(back->r.msg, sizeof(back->r.msg), "Time out (%d)",
timeout);
back->r.statuscode = STATUSCODE_INVALID;
len = 0; // fin
break;
@@ -705,7 +716,8 @@ int run_launch_ftp(FTPDownloadStruct * pStruct) {
strcpybuff(back->r.msg, "OK");
back->r.statuscode = HTTP_OK;
} else {
htsblk_failf(&back->r, "RETR incorrect: %s", linejmp(line));
snprintf(back->r.msg, sizeof(back->r.msg), "RETR incorrect: %s",
linejmp(line));
back->r.statuscode = STATUSCODE_INVALID;
}
} else {

View File

@@ -72,8 +72,7 @@ Please visit our Website: http://www.httrack.com
HTS_UNUSED: suppress unused-symbol warnings. HTS_STATIC: an unused-safe
static. HTS_PRINTF_FUN(fmt, arg): mark a printf-like function so the
compiler type-checks the format string at argument index fmt against the
varargs starting at arg. HTS_CHECK_RESULT: the return value carries the only
error signal, so dropping it is a bug; a (void) cast does not silence it. */
varargs starting at arg. */
#ifndef HTS_UNUSED
#ifdef __GNUC__
#define HTS_UNUSED __attribute__((unused))
@@ -81,13 +80,10 @@ Please visit our Website: http://www.httrack.com
#define HTS_STATIC static __attribute__((unused))
#define HTS_PRINTF_FUN(fmt, arg) __attribute__((format(printf, fmt, arg)))
#define HTS_CHECK_RESULT __attribute__((warn_unused_result))
#else
#define HTS_UNUSED
#define HTS_STATIC static
#define HTS_PRINTF_FUN(fmt, arg)
#define HTS_CHECK_RESULT
#endif
#endif
@@ -340,20 +336,16 @@ typedef int INTsys;
#endif
/* Socket-handle type. An unsigned integer wide enough for a Windows SOCKET;
a plain int file descriptor on POSIX. T_SOCP is its printf conversion,
'%' included: unsigned __int64 on Win64 must not be printed with "%d". */
a plain int file descriptor on POSIX. */
#ifdef _WIN32
#if defined(_WIN64)
typedef unsigned __int64 T_SOC;
#define T_SOCP "%" PRIu64
#else
typedef unsigned __int32 T_SOC;
#define T_SOCP "%" PRIu32
#endif
#else
typedef int T_SOC;
#define T_SOCP "%d"
#endif
/* Buffer size for a printed network address (IPv4 or IPv6, NUL included). */

View File

@@ -124,9 +124,7 @@ typedef struct help_wizard_buffers {
char stropt[2048]; // options
char stropt2[2048]; // options longues
char strwild[2048]; // wildcards
/* holds all four of the above plus separators: at 4096 a long answer set
clipped the filters off the command line */
char cmd[HTS_URLMAXSIZE * 2 + 3 * 2048 + 4];
char cmd[4096];
char str[256];
char *argv[256];
} help_wizard_buffers;
@@ -158,7 +156,9 @@ void help_wizard(httrackp * opt) {
char *a;
//
if (buffers == NULL) {
if (urls == NULL || mainpath == NULL || projname == NULL || stropt == NULL
|| stropt2 == NULL || strwild == NULL || cmd == NULL || str == NULL
|| argv == NULL) {
fprintf(stderr, "* memory exhausted in %s, line %d\n", __FILE__, __LINE__);
return;
}
@@ -251,7 +251,6 @@ void help_wizard(httrackp * opt) {
strcatbuff(stropt2, "--update ");
break;
case 0:
freet(buffers);
return;
break;
}
@@ -310,23 +309,14 @@ void help_wizard(httrackp * opt) {
printf("\n");
if (strlen(stropt) == 1)
stropt[0] = '\0'; // aucune
/* the tail is the filter list, and cmd is split into the argv handed to
hts_main() below: a clipped line would silently widen the crawl */
if (!sprintfbuff(cmd, "%s %s %s %s", urls, stropt, stropt2, strwild)) {
printf("* command line too long (%d bytes max)\n",
(int) sizeof(cmd) - 1);
freet(buffers);
return;
}
snprintf(cmd, sizeof(cmd), "%s %s %s %s", urls, stropt, stropt2, strwild);
printf("---> Wizard command line: httrack %s\n\n", cmd);
printf("Ready to launch the mirror? (Y/n) :");
fflush(stdout);
linput(stdin, str, 250);
if (strnotempty(str)) {
if (!((str[0] == 'y') || (str[0] == 'Y'))) {
freet(buffers);
if (!((str[0] == 'y') || (str[0] == 'Y')))
return;
}
}
printf("\n");
@@ -350,7 +340,7 @@ void help_wizard(httrackp * opt) {
}
/* Free buffers */
freet(buffers);
free(buffers);
#undef urls
#undef mainpath
#undef projname
@@ -534,13 +524,6 @@ void help(const char *app, int more) {
infomsg
(" %D cached delayed type check, don't wait for remote type during updates, to speedup them (%D0 wait, * %D1 don't wait)");
infomsg(" %M generate a RFC MIME-encapsulated full-archive (.mht)");
infomsg(" %Z after the mirror, rewrite each saved page with its "
"stylesheets, scripts, images and fonts inlined as data: URIs, so "
"any page opens by double-click anywhere (links between pages stay "
"relative; audio and video stay links); --single-file-max-size N "
"caps each asset (default 10485760 bytes). %M is the better "
"container where a Chromium-family browser is a given: one archive, "
"no base64 tax on text, a shared asset stored once");
infomsg(" %t keep the original file extension, don't rewrite it from the "
"MIME type (%t0 rewrite)");
infomsg

View File

@@ -37,7 +37,6 @@ Please visit our Website: http://www.httrack.com
#include "htscore.h"
#include "htswarc.h"
#include "htssinglefile.h"
/* specific definitions */
#include "htsbase.h"
@@ -705,16 +704,18 @@ T_SOC http_xfopen(httrackp *opt, int mode, int treat, int waitconnect,
/* Check for errors */
if (soc == INVALID_SOCKET) {
if (retour) {
if (!strnotempty(retour->msg)) {
if (retour->msg) {
if (!strnotempty(retour->msg)) {
#ifdef _WIN32
int last_errno = WSAGetLastError();
int last_errno = WSAGetLastError();
htsblk_failf(retour, "Connect error: %s", strerror(last_errno));
sprintf(retour->msg, "Connect error: %s", strerror(last_errno));
#else
int last_errno = errno;
int last_errno = errno;
htsblk_failf(retour, "Connect error: %s", strerror(last_errno));
sprintf(retour->msg, "Connect error: %s", strerror(last_errno));
#endif
}
}
}
}
@@ -2206,7 +2207,7 @@ T_SOC newhttp_addr(httrackp *opt, const char *_iadr, htsblk *retour, int port,
#if DEBUG
printf("erreur gethostbyname\n");
#endif
if (retour != NULL) {
if (retour && retour->msg) {
#ifdef _WIN32
snprintf(retour->msg, sizeof(retour->msg),
"Unable to get server's address: %s", error);
@@ -2237,17 +2238,17 @@ T_SOC newhttp_addr(httrackp *opt, const char *_iadr, htsblk *retour, int port,
DEBUG_W("socket()=%d\n" _(int) soc);
#endif
if (soc == INVALID_SOCKET) {
if (retour != NULL) {
if (retour && retour->msg) {
#ifdef _WIN32
int last_errno = WSAGetLastError();
htsblk_failf(retour, "Unable to create a socket: %s",
strerror(last_errno));
sprintf(retour->msg, "Unable to create a socket: %s",
strerror(last_errno));
#else
int last_errno = errno;
htsblk_failf(retour, "Unable to create a socket: %s",
strerror(last_errno));
sprintf(retour->msg, "Unable to create a socket: %s",
strerror(last_errno));
#endif
}
return INVALID_SOCKET; // erreur création socket impossible
@@ -2261,8 +2262,17 @@ T_SOC newhttp_addr(httrackp *opt, const char *_iadr, htsblk *retour, int port,
&bind_addr, &error) == NULL
|| bind(soc, &SOCaddr_sockaddr(bind_addr),
SOCaddr_size(bind_addr)) != 0) {
snprintf(retour->msg, sizeof(retour->msg),
"Unable to bind the specificied server address: %s", error);
if (retour && retour->msg) {
#ifdef _WIN32
snprintf(retour->msg, sizeof(retour->msg),
"Unable to bind the specificied server address: %s",
error);
#else
snprintf(retour->msg, sizeof(retour->msg),
"Unable to bind the specificied server address: %s",
error);
#endif
}
deletesoc(soc);
return INVALID_SOCKET;
}
@@ -2309,17 +2319,17 @@ T_SOC newhttp_addr(httrackp *opt, const char *_iadr, htsblk *retour, int port,
#if HDEBUG
printf("unable to connect!\n");
#endif
if (retour != NULL) {
if (retour != NULL && retour->msg) {
#ifdef _WIN32
const int last_errno = WSAGetLastError();
htsblk_failf(retour, "Unable to connect to the server: %s",
strerror(last_errno));
sprintf(retour->msg, "Unable to connect to the server: %s",
strerror(last_errno));
#else
const int last_errno = errno;
htsblk_failf(retour, "Unable to connect to the server: %s",
strerror(last_errno));
sprintf(retour->msg, "Unable to connect to the server: %s",
strerror(last_errno));
#endif
}
/* Close the socket and notify the error!!! */
@@ -2538,15 +2548,6 @@ void fil_simplifie(char *f) {
}
}
void htsblk_failf(htsblk *r, const char *fmt, ...) {
va_list args;
va_start(args, fmt);
// deliberate clip: the reason is quoted from a remote reply
(void) vslprintfbuff(r->msg, sizeof(r->msg), fmt, args);
va_end(args);
}
// fermer liaison fichier ou socket
void deletehttp(htsblk * r) {
#if HTS_DEBUG_CLOSESOCK
@@ -2597,15 +2598,13 @@ void deletesoc(T_SOC soc) {
if (closesocket(soc) != 0) {
int err = WSAGetLastError();
fprintf(stderr, "* error closing socket " T_SOCP ": %s\n", soc,
strerror(err));
fprintf(stderr, "* error closing socket %d: %s\n", soc, strerror(err));
}
#else
if (close(soc) != 0) {
const int err = errno;
fprintf(stderr, "* error closing socket " T_SOCP ": %s\n", soc,
strerror(err));
fprintf(stderr, "* error closing socket %d: %s\n", soc, strerror(err));
}
#endif
#if HTS_WIDE_DEBUG
@@ -3018,7 +3017,7 @@ int finput(T_SOC fd, char *s, int max) {
}
// Like linput, but in memory (optimized)
int binput(const char *buff, char *s, int max) {
int binput(char *buff, char *s, int max) {
int count = 0;
int destCount = 0;
@@ -6021,8 +6020,6 @@ HTSEXT_API httrackp *hts_create_opt(void) {
StringCopy(opt->cookies_file, "");
StringCopy(opt->warc_file, "");
opt->warc_max_size = 0; /* no rotation unless --warc-max-size sets it */
opt->single_file = HTS_FALSE;
opt->single_file_max_size = SINGLEFILE_DEFAULT_MAX_SIZE;
StringCopy(opt->why_url, "");
opt->pause_min_ms = 0;
opt->pause_max_ms = 0;

View File

@@ -199,10 +199,6 @@ T_SOC newhttp(httrackp * opt, const char *iadr, htsblk * retour, int port,
etc.). */
T_SOC newhttp_addr(httrackp *opt, const char *iadr, htsblk *retour, int port,
int waitconnect, int addr_index, int *addr_count);
/* Clips the formatted failure reason into r->msg, which also round-trips
through the cache as X-StatusMessage. Leaves r->statuscode to the caller. */
void htsblk_failf(htsblk *r, const char *fmt, ...) HTS_PRINTF_FUN(2, 3);
HTS_INLINE void deletehttp(htsblk * r);
HTS_INLINE int deleteaddr(htsblk * r);
HTS_INLINE void deletesoc(T_SOC soc);
@@ -266,7 +262,7 @@ HTS_INLINE void time_rfc822_local(char *s, struct tm *A);
HTS_INLINE int sendc(htsblk * r, const char *s);
int finput(T_SOC fd, char *s, int max);
int binput(const char *buff, char *s, int max);
int binput(char *buff, char *s, int max);
int linput(FILE * fp, char *s, int max);
int linputsoc(T_SOC soc, char *s, int max);
int linputsoc_t(T_SOC soc, char *s, int max, int timeout);
@@ -611,21 +607,6 @@ static HTS_UNUSED size_t llint_to_size_t(LLint o) {
}
}
/* Capacity for @p used bytes plus @p extra more plus @p slack spare;
(size_t) -1 if the total exceeds (size_t) -2 or @p extra is negative
(llint_to_size_t() would map that to a huge valid-looking size). */
static HTS_UNUSED size_t llint_grow_size_t(size_t used, LLint extra,
size_t slack) {
const size_t max = (size_t) -2; /* (size_t) -1 is the error value */
const size_t e = extra >= 0 ? llint_to_size_t(extra) : (size_t) -1;
if (e == (size_t) -1 || used > max || slack > max - used ||
e > max - used - slack) {
return (size_t) -1;
}
return used + e + slack;
}
/* dirent() compatibility */
#ifdef _WIN32
/* Holds a UTF-8 d_name: MAX_PATH (260) UTF-16 units expand to <=3 bytes each.

View File

@@ -547,12 +547,6 @@ struct httrackp {
archive. Tail: ABI */
hts_boolean warc_wacz; /**< --wacz: package archive+index+pages as a WACZ zip
(implies --warc + --warc-cdx). Tail: ABI */
hts_boolean single_file; /**< --single-file: once the mirror is done, rewrite
each saved page with its assets inlined as
data: URIs. Tail: ABI */
LLint single_file_max_size; /**< --single-file-max-size: per-asset cap in
bytes; a bigger asset stays a link.
Tail: ABI */
};
/* Running statistics for a mirror. */

View File

@@ -4583,7 +4583,7 @@ int hts_wait_delayed(htsmoduleStruct * str, lien_adrfilsave *afs,
/* seen as in error */
in_error = back[b].r.statuscode;
in_error_msg[0] = 0;
strncatbuff(in_error_msg, back[b].r.msg, sizeof(in_error_msg) - 1);
strncat(in_error_msg, back[b].r.msg, sizeof(in_error_msg) - 1);
in_error_size = back[b].r.totalsize;
/* don't break, even with "don't take error pages" switch, because we need to process the slot anyway (and cache the error) */
}

View File

@@ -173,8 +173,8 @@ int http_proxy_tunnel(httrackp *opt, htsblk *retour, const char *adr,
if (sscanf(line, "HTTP/%*d.%*d %d", &code) < 1)
code = 0;
if (code < 200 || code >= 300) {
htsblk_failf(retour, "proxy CONNECT refused: %s",
strnotempty(line) ? line : "(no status)");
snprintf(retour->msg, sizeof(retour->msg), "proxy CONNECT refused: %s",
strnotempty(line) ? line : "(no status)");
return 0;
}

View File

@@ -33,7 +33,6 @@ Please visit our Website: http://www.httrack.com
#ifndef HTSSAFE_DEFH
#define HTSSAFE_DEFH
#include <stdarg.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
@@ -117,15 +116,6 @@ static HTS_UNUSED void abortf_(const char *exp, const char *file, int line) {
#endif
#define HTS_IS_NOT_CHAR_BUFFER(VAR) (!HTS_IS_CHAR_BUFFER(VAR))
/* Source capacity for the buff() family, (size_t)-1 when unknown; sizeof of the
TYPE keeps a decayed operand ("buf + 1") off -Wsizeof-array-decay. */
#if (defined(__GNUC__) && !defined(__cplusplus))
#define HTS_SIZEOF_SRC_(B) \
(HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(__typeof__(B)))
#else
#define HTS_SIZEOF_SRC_(B) (HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(B))
#endif
/* Compile-time checks. */
static HTS_UNUSED void htssafe_compile_time_check_(void) {
char array[32];
@@ -215,17 +205,19 @@ static char *strncatbuff_ptr_(char *dest, const char *src, size_t n) {
#if (defined(__GNUC__) && !defined(__cplusplus))
#define strncatbuff(A, B, N) \
__builtin_choose_expr(HTS_IS_CHAR_BUFFER(A), \
strncat_safe_(A, sizeof(A), B, HTS_SIZEOF_SRC_(B), N, \
"overflow while appending '" #B \
"' to '" #A "'", \
__FILE__, __LINE__), \
strncatbuff_ptr_((A), (B), (N)))
__builtin_choose_expr( \
HTS_IS_CHAR_BUFFER(A), \
strncat_safe_(A, sizeof(A), B, \
HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(B), N, \
"overflow while appending '" #B "' to '" #A "'", __FILE__, \
__LINE__), \
strncatbuff_ptr_((A), (B), (N)))
#else
#define strncatbuff(A, B, N) \
(HTS_IS_NOT_CHAR_BUFFER(A) \
? strncat(A, B, N) \
: strncat_safe_(A, sizeof(A), B, HTS_SIZEOF_SRC_(B), N, \
: strncat_safe_(A, sizeof(A), B, \
HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(B), N, \
"overflow while appending '" #B "' to '" #A "'", \
__FILE__, __LINE__))
#endif
@@ -240,7 +232,9 @@ static char *strncatbuff_ptr_(char *dest, const char *src, size_t n) {
#define strcatbuff(A, B) \
__builtin_choose_expr( \
HTS_IS_CHAR_BUFFER(A), \
strncat_safe_(A, sizeof(A), B, HTS_SIZEOF_SRC_(B), (size_t) -1, \
strncat_safe_(A, sizeof(A), B, \
HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(B), \
(size_t) -1, \
"overflow while appending '" #B "' to '" #A "'", __FILE__, \
__LINE__), \
strcatbuff_ptr_((A), (B)))
@@ -248,7 +242,9 @@ static char *strncatbuff_ptr_(char *dest, const char *src, size_t n) {
#define strcatbuff(A, B) \
(HTS_IS_NOT_CHAR_BUFFER(A) \
? strcat(A, B) \
: strncat_safe_(A, sizeof(A), B, HTS_SIZEOF_SRC_(B), (size_t) -1, \
: strncat_safe_(A, sizeof(A), B, \
HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(B), \
(size_t) -1, \
"overflow while appending '" #B "' to '" #A "'", \
__FILE__, __LINE__))
#endif
@@ -261,17 +257,19 @@ static char *strncatbuff_ptr_(char *dest, const char *src, size_t n) {
#if (defined(__GNUC__) && !defined(__cplusplus))
#define strcpybuff(A, B) \
__builtin_choose_expr(HTS_IS_CHAR_BUFFER(A), \
strcpy_safe_(A, sizeof(A), B, HTS_SIZEOF_SRC_(B), \
"overflow while copying '" #B "' to '" #A \
"'", \
__FILE__, __LINE__), \
strcpybuff_ptr_((A), (B)))
__builtin_choose_expr( \
HTS_IS_CHAR_BUFFER(A), \
strcpy_safe_(A, sizeof(A), B, \
HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(B), \
"overflow while copying '" #B "' to '" #A "'", __FILE__, \
__LINE__), \
strcpybuff_ptr_((A), (B)))
#else
#define strcpybuff(A, B) \
(HTS_IS_NOT_CHAR_BUFFER(A) \
? strcpy(A, B) \
: strcpy_safe_(A, sizeof(A), B, HTS_SIZEOF_SRC_(B), \
: strcpy_safe_(A, sizeof(A), B, \
HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(B), \
"overflow while copying '" #B "' to '" #A "'", __FILE__, \
__LINE__))
#endif
@@ -288,24 +286,24 @@ static char *strncatbuff_ptr_(char *dest, const char *src, size_t n) {
* Append characters of "B" to "A", "A" having a maximum capacity of "S".
*/
#define strlcatbuff(A, B, S) \
strncat_safe_(A, S, B, HTS_SIZEOF_SRC_(B), (size_t) -1, \
"overflow while appending '" #B "' to '" #A "'", __FILE__, \
__LINE__)
strncat_safe_(A, S, B, HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(B), \
(size_t) -1, "overflow while appending '" #B "' to '" #A "'", \
__FILE__, __LINE__)
/**
* Append at most "N" characters of "B" to "A", "A" having a maximum capacity
* of "S".
*/
#define strlncatbuff(A, B, S, N) \
strncat_safe_(A, S, B, HTS_SIZEOF_SRC_(B), N, \
"overflow while appending '" #B "' to '" #A "'", __FILE__, \
strncat_safe_(A, S, B, HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(B), \
N, "overflow while appending '" #B "' to '" #A "'", __FILE__, \
__LINE__)
/**
* Copy characters of "B" to "A", "A" having a maximum capacity of "S".
*/
#define strlcpybuff(A, B, S) \
strcpy_safe_(A, S, B, HTS_SIZEOF_SRC_(B), \
strcpy_safe_(A, S, B, HTS_IS_NOT_CHAR_BUFFER(B) ? (size_t) -1 : sizeof(B), \
"overflow while copying '" #B "' to '" #A "'", __FILE__, \
__LINE__)
@@ -424,9 +422,7 @@ static HTS_INLINE HTS_UNUSED htsbuff htsbuff_ptr_(char *buf, size_t cap) {
*/
static HTS_INLINE HTS_UNUSED void htsbuff_catn(htsbuff *b, const char *s,
size_t n) {
/* the (size_t)-1 "no limit" sentinel would reach strnlen as a bound past
PTRDIFF_MAX */
const size_t add = n != (size_t) -1 ? strnlen(s, n) : strlen(s);
const size_t add = strnlen(s, n);
/* Overflow-safe: keep the (potentially huge) 'add' alone on one side. The
maintained invariant len < cap makes 'cap - len' >= 1 (no underflow), so
'add < cap - len' cannot wrap the way 'len + add < cap' could. */
@@ -460,49 +456,6 @@ static HTS_INLINE HTS_UNUSED const char *htsbuff_str(const htsbuff *b) {
return b->buf;
}
/**
* Callers that deliberately ignore truncation use this instead of
* slprintfbuff(), so it is not HTS_CHECK_RESULT.
*/
static HTS_INLINE HTS_UNUSED HTS_PRINTF_FUN(3, 0) hts_boolean
vslprintfbuff(char *dest, size_t size, const char *fmt, va_list args) {
int ret;
assertf(dest != NULL && size != 0);
ret = vsnprintf(dest, size, fmt, args);
/* pre-C99 runtimes (msvcrt _vsnprintf) return -1 and do not terminate */
dest[size - 1] = '\0';
return ret >= 0 && (size_t) ret < size ? HTS_TRUE : HTS_FALSE;
}
/**
* Formatted print into dest (capacity size, NUL included), truncating to fit
* and always NUL-terminating. Returns HTS_TRUE if the whole output fit; the
* result is the only truncation signal, so it must be acted on. Unlike
* strcpybuff() it never aborts, so it suits text built from remote input.
*/
static HTS_INLINE HTS_UNUSED HTS_CHECK_RESULT HTS_PRINTF_FUN(3, 4) hts_boolean
slprintfbuff(char *dest, size_t size, const char *fmt, ...) {
va_list args;
hts_boolean ret;
va_start(args, fmt);
ret = vslprintfbuff(dest, size, fmt, args);
va_end(args);
return ret;
}
/**
* slprintfbuff() over the in-scope array ARR (capacity = sizeof(ARR)).
* On GCC/Clang a pointer is a compile error; use slprintfbuff() for those.
*/
#if (defined(__GNUC__) && !defined(__cplusplus))
#define sprintfbuff(ARR, ...) \
slprintfbuff((ARR), sizeof(ARR) + htsbuff_must_be_array_(ARR), __VA_ARGS__)
#else
#define sprintfbuff(ARR, ...) slprintfbuff((ARR), sizeof(ARR), __VA_ARGS__)
#endif
/* Thin aliases over the libc allocator/memcpy (historical "t" suffix); no
added bounds checking. freet() also NULLs the freed pointer and tolerates
NULL. memcpybuff() despite the name is a raw memcpy: the caller owns the

File diff suppressed because it is too large Load Diff

View File

@@ -147,8 +147,7 @@ HTS_UNUSED static int LANG_LIST(const char *path, char *buffer, size_t size);
// 0- Init the URL catcher with standard port
// smallserver_init(&port,&return_host);
T_SOC smallserver_init_std(int *port_prox, char *adr_prox, int defaultPort,
const char *bindAddr) {
T_SOC smallserver_init_std(int *port_prox, char *adr_prox, int defaultPort) {
T_SOC soc;
if (defaultPort <= 0) {
@@ -161,12 +160,12 @@ T_SOC smallserver_init_std(int *port_prox, char *adr_prox, int defaultPort,
int i = 0;
do {
soc = smallserver_init(&try_to_listen_to[i], adr_prox, bindAddr);
soc = smallserver_init(&try_to_listen_to[i], adr_prox);
*port_prox = try_to_listen_to[i];
i++;
} while((soc == INVALID_SOCKET) && (try_to_listen_to[i] >= 0));
} else {
soc = smallserver_init(&defaultPort, adr_prox, bindAddr);
soc = smallserver_init(&defaultPort, adr_prox);
*port_prox = defaultPort;
}
return soc;
@@ -244,10 +243,9 @@ static int my_gethostname(char *h_loc, size_t size) {
}
// smallserver_init(&port,&return_host);
T_SOC smallserver_init(int *port, char *adr, const char *bindAddr) {
T_SOC smallserver_init(int *port, char *adr) {
T_SOC soc = INVALID_SOCKET;
char h_loc[256 + 2];
SOCaddr server;
commandRunning = commandEnd = commandReturn = commandReturnSet =
commandEndRequested = 0;
@@ -258,23 +256,25 @@ T_SOC smallserver_init(int *port, char *adr, const char *bindAddr) {
free(commandReturnCmdl);
commandReturnCmdl = NULL;
SOCaddr_initany(server);
if (bindAddr != NULL && *bindAddr != '\0') {
/* advertise the bound address, else the URL we print is unreachable */
if (strlen(bindAddr) >= sizeof(h_loc) || !gethost(bindAddr, &server)) {
return INVALID_SOCKET;
}
strcpybuff(h_loc, bindAddr);
} else if (my_gethostname(h_loc, 256) != 0) { // host name
return INVALID_SOCKET;
}
if (my_gethostname(h_loc, 256) == 0) { // host name
SOCaddr server;
if ((soc = (T_SOC) socket(SOCaddr_sinfamily(server), SOCK_STREAM, 0)) !=
INVALID_SOCKET) {
SOCaddr_initport(server, *port);
if (bind(soc, &SOCaddr_sockaddr(server), SOCaddr_size(server)) == 0) {
if (listen(soc, 10) >= 0) {
strcpy(adr, h_loc);
SOCaddr_initany(server);
if ((soc =
(T_SOC) socket(SOCaddr_sinfamily(server), SOCK_STREAM,
0)) != INVALID_SOCKET) {
SOCaddr_initport(server, *port);
if (bind(soc, &SOCaddr_sockaddr(server), SOCaddr_size(server)) == 0) {
if (listen(soc, 10) >= 0) {
strcpy(adr, h_loc);
} else {
#ifdef _WIN32
closesocket(soc);
#else
close(soc);
#endif
soc = INVALID_SOCKET;
}
} else {
#ifdef _WIN32
closesocket(soc);
@@ -283,13 +283,6 @@ T_SOC smallserver_init(int *port, char *adr, const char *bindAddr) {
#endif
soc = INVALID_SOCKET;
}
} else {
#ifdef _WIN32
closesocket(soc);
#else
close(soc);
#endif
soc = INVALID_SOCKET;
}
}
return soc;
@@ -318,104 +311,6 @@ typedef struct {
error_redirect = "/server/error.html"; \
} while(0)
/* Longest "sid" value worth unescaping: the expected one is an md5 hex digest,
so anything near this is already invalid and is rejected unread. */
#define SID_VALUE_MAX 64
/** Does the urlencoded request body present the expected session id?
True only if at least one "sid" field is present and every occurrence
matches, so it holds whichever one a later last-write-wins parse keeps.
Non-destructive: it runs before the body is tokenized in place. */
static hts_boolean body_sid_is_valid(const char *body, const char *expected) {
const char *s = body;
hts_boolean seen = HTS_FALSE;
while (s != NULL && *s != '\0') {
const char *const amp = strchr(s, '&');
const char *const eq = strchr(s, '=');
if (eq != NULL && (amp == NULL || eq < amp) && (size_t) (eq - s) == 3 &&
strncmp(s, "sid", 3) == 0) {
const size_t len = amp != NULL ? (size_t) (amp - eq - 1) : strlen(eq + 1);
hts_boolean match = HTS_FALSE;
if (len < SID_VALUE_MAX) {
char raw[SID_VALUE_MAX];
String value = STRING_EMPTY;
memcpy(raw, eq + 1, len);
raw[len] = '\0';
unescapehttp(raw, &value);
/* StringBuff is NULL until written, so an empty value lands here. */
if (StringBuff(value) != NULL &&
strcmp(StringBuff(value), expected) == 0) {
match = HTS_TRUE;
}
StringFree(value);
}
if (!match) {
return HTS_FALSE;
}
seen = HTS_TRUE;
}
s = amp != NULL ? amp + 1 : NULL;
}
return seen;
}
/** Append src to the NUL-terminated dst of capacity size (NUL included).
False, leaving dst untouched, if it would not fit: unlike strcatbuff() this
never aborts, because every piece appended here is client-supplied. */
static hts_boolean path_append(char *dst, size_t size, const char *src) {
const size_t used = strlen(dst);
const size_t len = strlen(src);
/* dst holds at most size-1 bytes, so "size - used" is >= 1 and the untrusted
len stays alone: "used + len < size" could wrap and pass. */
if (len >= size - used) {
return HTS_FALSE;
}
memcpy(dst + used, src, len + 1);
return HTS_TRUE;
}
/* Append c to dst as an HTML entity, or return HTS_FALSE if it needs none. */
static hts_boolean cat_html_escaped(String *dst, char c) {
switch (c) {
case '<':
StringCat(*dst, "&lt;");
break;
case '>':
StringCat(*dst, "&gt;");
break;
case '&':
StringCat(*dst, "&amp;");
break;
case '\'':
StringCat(*dst, "&#39;");
break;
default:
return HTS_FALSE;
}
return HTS_TRUE;
}
/* Append the value of a double-quoted command-line argument: escaped for HTML,
which the browser undoes when it posts the command line back, and for the
argv splitter, which does not. */
static void cat_cmdline_arg(String *output, const char *value) {
const char *a;
for (a = value; *a != '\0'; a++) {
if (*a == '\\' || *a == '\"') {
StringCat(*output, "\\");
}
if (!cat_html_escaped(output, *a)) {
StringMemcat(*output, a, 1);
}
}
}
int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
int timeout = 30;
int retour = 0;
@@ -427,9 +322,6 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
String tmpbuff = STRING_EMPTY;
String tmpbuff2 = STRING_EMPTY;
String fspath = STRING_EMPTY;
/* Project directory this server set up; the only root /website/ serves from,
and deliberately not cleared between requests. */
String website = STRING_EMPTY;
char catbuff[CATBUFF_SIZE];
/* Load strings */
@@ -503,7 +395,6 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
T_SOC soc_c;
LLint length = 0;
const char *error_redirect = NULL;
hts_boolean denied = HTS_FALSE;
line[0] = '\0';
buffer[0] = '\0';
@@ -615,22 +506,6 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
}
}
/* Authenticate the body before parsing it: every field it carries is
written straight into the global key store below, "command" included,
and that one reaches the engine. Checking afterwards cannot work — the
damage is already done, and the pre-seeded "sid" above would compare
equal to itself for a request that simply omits the field. */
if (meth && buffer[0]) {
intptr_t expected = 0;
if (!coucal_readptr(NewLangList, "_sid", &expected) ||
!body_sid_is_valid(buffer, (const char *) expected)) {
buffer[0] = '\0';
meth = 0;
denied = HTS_TRUE;
}
}
/* check variables */
if (meth && buffer[0]) {
char *s = buffer;
@@ -651,6 +526,20 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
}
}
/* Error check */
{
intptr_t adr = 0;
intptr_t adr2 = 0;
if (coucal_readptr(NewLangList, "sid", &adr)) {
if (coucal_readptr(NewLangList, "_sid", &adr2)) {
if (strcmp((char *) adr, (char *) adr2) != 0) {
meth = 0;
}
}
}
}
/* Check variables (internal) */
if (meth) {
int doLoad = 0;
@@ -841,11 +730,6 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
if (!structcheck(StringBuff(tmpbuff))) {
FILE *fp;
/* Both halves of fspath come from posted fields, so a ".."
in them would escape the mirror once served. */
if (strstr(StringBuff(fspath), "..") == NULL) {
StringCopy(website, StringBuff(fspath));
}
StringCat(tmpbuff, "winprofile.ini");
fp = fopen(StringBuff(tmpbuff), "wb");
if (fp != NULL) {
@@ -918,7 +802,7 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
/* Response */
if (meth) {
hts_boolean virtualpath = HTS_FALSE;
int virtualpath = 0;
char *pos;
char *url = strchr(line1, ' ');
@@ -929,11 +813,11 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
char *qpos;
/* get the URL */
fsfile[0] = '\0';
if (error_redirect == NULL) {
if ((qpos = strchr(url, '?'))) {
*qpos = '\0';
}
fsfile[0] = '\0';
if (strcmp(url, "/") == 0) {
file = "/server/index.html";
meth = 2;
@@ -946,7 +830,7 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
}
if (strncmp(file, "/website/", 9) == 0) {
virtualpath = HTS_TRUE;
virtualpath = 1;
}
/* override */
@@ -960,26 +844,18 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
}
}
/* the override above may have swapped a mirror path for a GUI page */
virtualpath = strncmp(file, "/website/", 9) == 0;
if (strlen(path) + strlen(file) + 32 < sizeof(fsfile)) {
if (strncmp(file, "/website/", 9) != 0) {
sprintf(fsfile, "%shtml%s", path, file);
} else {
intptr_t adr = 0;
if (!virtualpath) {
if (!path_append(fsfile, sizeof(fsfile), path) ||
!path_append(fsfile, sizeof(fsfile), "html") ||
!path_append(fsfile, sizeof(fsfile), file)) {
fsfile[0] = '\0';
}
} else if (StringNotEmpty(website)) {
/* Never the posted "projpath": a client root reads any file. */
if (!path_append(fsfile, sizeof(fsfile), StringBuff(website)) ||
!path_append(fsfile, sizeof(fsfile), "/") ||
!path_append(fsfile, sizeof(fsfile), file + 9)) {
fsfile[0] = '\0';
if (coucal_readptr(NewLangList, "projpath", &adr)) {
sprintf(fsfile, "%s%s", (char *) adr, file + 9);
}
}
}
/* path itself may hold ".." (webhttrack passes "<bin>/../share"), so
only the untrusted halves are checked: file here, website above. */
if (fsfile[0] && strstr(file, "..") == NULL
&& (fp = fopen(fsfile, "rb"))) {
char ok[] =
@@ -1027,16 +903,16 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
}
}
StringMemcat(headers, redir, strlen(redir));
/* client-supplied: a CR/LF here would split the response */
if (newfile[strcspn(newfile, "\r\n")] == '\0') {
StringCat(headers, "Location: ");
StringCat(headers, newfile);
StringCat(headers, "\r\n");
{
char tmp[256];
if (strlen(file) < sizeof(tmp) - 32) {
sprintf(tmp, "Location: %s\r\n", newfile);
StringMemcat(headers, tmp, strlen(tmp));
}
}
coucal_write(NewLangList, "redirect", (intptr_t) NULL);
} else if (!virtualpath && is_html(file)) {
/* GUI templates only: ${_sid} in a mirrored page would hand the
crawled site the session id that authenticates commands */
} else if (is_html(file)) {
int outputmode = 0;
StringMemcat(headers, ok, sizeof(ok) - 1);
@@ -1064,7 +940,6 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
int p;
int format = 0;
int listDefault = 0;
hts_boolean unquoted = HTS_FALSE;
name[0] = '\0';
strlncatbuff(name, str, sizeof(name_), n);
@@ -1074,12 +949,6 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
} else if ((p = strfield(name, "html:"))) {
name += p;
format = 1;
} else if ((p = strfield(name, "unquoted:"))) {
name += p;
unquoted = HTS_TRUE;
} else if ((p = strfield(name, "arg:"))) {
name += p;
format = 5;
} else if ((p = strfield(name, "list:"))) {
name += p;
format = 2;
@@ -1216,8 +1085,8 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
test:<if ==0>:<if ==1>:<if == 2>..
ztest:<if == 0 || !exist>:<if == 1>:<if == 2>..
*/
else if ((p = strfield(name, "test:")) ||
(p = strfield(name, "ztest:"))) {
else if ((p = strfield(name, "test:"))
|| (p = strfield(name, "ztest:"))) {
intptr_t adr = 0;
char *pos2;
int ztest = (name[0] == 'z');
@@ -1320,12 +1189,6 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
}
}
}
/* consumed here: it shares nothing with the list and
option formats below */
if (format == 5 && langstr != NULL && outputmode != -1) {
cat_cmdline_arg(&output, langstr);
langstr = NULL;
}
if (langstr && outputmode != -1) {
switch (format) {
case 0:
@@ -1343,18 +1206,18 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
StringMemcat(output, &c, 1);
}
a += 2;
} else if (unquoted && a[0] == '\"') {
/* the browser posts an entity back as a raw
quote, which would open a quoted run in the
argv splitter; a URI cannot hold one anyway */
StringCat(output, "%22");
} else if (outputmode &&
cat_html_escaped(&output, a[0])) {
/* appended as an entity */
} else if (outputmode && a[0] == '<') {
StringCat(output, "&lt;");
} else if (outputmode && a[0] == '>') {
StringCat(output, "&gt;");
} else if (outputmode && a[0] == '&') {
StringCat(output, "&amp;");
} else if (outputmode && a[0] == '\'') {
StringCat(output, "&#39;");
} else if (outputmode == 3 && a[0] == ' ') {
StringCat(output, "%20");
} else if (outputmode >= 2 &&
((unsigned char) a[0]) < 32) {
} else if (outputmode >= 2
&& ((unsigned char) a[0]) < 32) {
char tmp[32];
sprintf(tmp, "%%%02x", (unsigned char) a[0]);
@@ -1415,10 +1278,20 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
}
StringClear(tmpbuff);
break;
case '<':
StringCat(tmpbuff, "&lt;");
break;
case '>':
StringCat(tmpbuff, "&gt;");
break;
case '&':
StringCat(tmpbuff, "&amp;");
break;
case '\'':
StringCat(tmpbuff, "&#39;");
break;
default:
if (!cat_html_escaped(&tmpbuff, *fstr)) {
StringMemcat(tmpbuff, fstr, 1);
}
StringMemcat(tmpbuff, fstr, 1);
break;
}
fstr++;
@@ -1458,9 +1331,7 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
}
#endif
} else {
if (is_html(file)) {
StringMemcat(headers, ok, sizeof(ok) - 1);
} else if (is_text(file)) {
if (is_text(file)) {
StringMemcat(headers, ok_text, sizeof(ok_text) - 1);
} else if (is_js(file)) {
StringMemcat(headers, ok_js, sizeof(ok_js) - 1);
@@ -1497,11 +1368,6 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
StringCat(output, error);
}
}
} else if (denied) {
StringCat(headers, "HTTP/1.0 403 Forbidden\r\n"
"Server: httrack small server\r\n"
"Content-type: text/html\r\n");
StringCat(output, "Missing or invalid session id.\r\n");
} else {
#ifdef _DEBUG
char error_hdr[] =
@@ -1567,7 +1433,6 @@ int smallserver(T_SOC soc, char *url, char *method, char *data, char *path) {
StringFree(tmpbuff);
StringFree(tmpbuff2);
StringFree(fspath);
StringFree(website);
if (buffer)
free(buffer);

View File

@@ -43,11 +43,8 @@ Please visit our Website: http://www.httrack.com
// Fonctions
void socinput(T_SOC soc, char *s, int max);
/* Listen on bindAddr, or every interface if NULL/empty; adr (>= 258 bytes) gets
the address to advertise. INVALID_SOCKET on error. */
T_SOC smallserver_init_std(int *port_prox, char *adr_prox, int defaultPort,
const char *bindAddr);
T_SOC smallserver_init(int *port, char *adr, const char *bindAddr);
T_SOC smallserver_init_std(int *port_prox, char *adr_prox, int defaultPort);
T_SOC smallserver_init(int *port, char *adr);
int smallserver(T_SOC soc, char *url, char *method, char *data, char *path);
#define CATCH_RESPONSE \

File diff suppressed because it is too large Load Diff

View File

@@ -1,78 +0,0 @@
/* ------------------------------------------------------------ */
/*
HTTrack Website Copier, Offline Browser for Windows and Unix
Copyright (C) 2026 Xavier Roche and other contributors
SPDX-License-Identifier: GPL-3.0-or-later
This program is free software: you can redistribute it and/or modify
it under the terms of the GNU General Public License as published by
the Free Software Foundation, either version 3 of the License, or
(at your option) any later version.
This program is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
GNU General Public License for more details.
You should have received a copy of the GNU General Public License
along with this program. If not, see <http://www.gnu.org/licenses/>.
Ethical use: we kindly ask that you NOT use this software to harvest email
addresses or to collect any other private information about people. Doing so
would dishonor our work and waste the many hours we have spent on it.
Please visit our Website: http://www.httrack.com
*/
/* ------------------------------------------------------------ */
/* --single-file: post-mirror pass embedding each page's assets as data: URIs.
Internal, not installed. Unlike -%M (one MHT archive for the whole mirror)
this rewrites the saved pages in place and keeps page-to-page links
relative, so the mirror stays browsable and every page also stands alone. */
/* ------------------------------------------------------------ */
#ifndef HTS_SINGLEFILE_DEFH
#define HTS_SINGLEFILE_DEFH
#include "htsopt.h"
#include "htsstrings.h"
#ifdef __cplusplus
extern "C" {
#endif
/* --single-file-max-size default: a bigger asset stays an ordinary link. */
#define SINGLEFILE_DEFAULT_MAX_SIZE (10 * 1024 * 1024)
/* Never base64 more than this, whatever the cap: code64() sizes in int. */
#define SINGLEFILE_HARD_MAX_SIZE (256 * 1024 * 1024)
/* Total bytes one page may inline. Nested @import fans out multiplicatively,
so a few hundred bytes of hostile CSS can otherwise ask for gigabytes. */
#define SINGLEFILE_MAX_PAGE_SIZE (64 * 1024 * 1024)
/* Rewrite every HTML page the mirror produced. No-op unless opt->single_file;
call once the tree is final, after the update purge. */
void singlefile_process_mirror(httrackp *opt);
/* Rewrite one HTML document held in memory, appending the result to out.
root is the mirror directory that references may not escape; page_path is
the document's own path under it (both UTF-8, '/' or native separators).
Returns HTS_TRUE if at least one reference was replaced; out may still
differ from the input when that is HTS_FALSE, since a style or srcset value
is re-serialized in place. */
hts_boolean singlefile_rewrite_html(httrackp *opt, const char *root,
const char *page_path, const char *html,
size_t html_len, String *out);
/* singlefile_rewrite_html() over the file at page_path, rewritten in place.
Returns HTS_TRUE if the file changed. */
hts_boolean singlefile_rewrite_file(httrackp *opt, const char *root,
const char *page_path);
#ifdef __cplusplus
}
#endif
#endif

View File

@@ -1318,19 +1318,19 @@ HTSEXT_API hts_boolean hts_findnext(find_handle find) {
if (find) {
#ifdef _WIN32
if ((FindNextFileA(find->handle, &find->hdata)))
return HTS_TRUE;
return 1;
#else
char catbuff[CATBUFF_SIZE];
memset(&(find->filestat), 0, sizeof(find->filestat));
if ((find->dirp = readdir(find->hdir)))
if (!STAT(
concat(catbuff, sizeof(catbuff), find->path, find->dirp->d_name),
&find->filestat))
return HTS_TRUE;
if (find->dirp->d_name)
if (!STAT
(concat(catbuff, sizeof(catbuff), find->path, find->dirp->d_name), &find->filestat))
return 1;
#endif
}
return HTS_FALSE;
return 0;
}
HTSEXT_API int hts_findclose(find_handle find) {

View File

@@ -62,7 +62,6 @@ Please visit our Website: http://www.httrack.com
#include "htsmd5.c"
#include "md5.c"
#include "htscmdline.h"
#include "htsserver.h"
#include "htsurlport.h"
#include "htsweb.h"
@@ -89,7 +88,7 @@ Please visit our Website: http://www.httrack.com
static htsmutex refreshMutex = HTSMUTEX_INIT;
static int help_server(char *dest_path, int defaultPort, const char *bindAddr);
static int help_server(char *dest_path, int defaultPort);
extern int commandRunning;
extern int commandEnd;
extern int commandReturn;
@@ -154,8 +153,6 @@ int main(int argc, char *argv[]) {
int ret = 0;
int defaultPort = 0;
int parentPid = 0;
/* loopback by default: the handler trusts its input; --bind widens it */
const char *bindAddr = "127.0.0.1";
printf("Initializing the server..\n");
@@ -182,8 +179,7 @@ int main(int argc, char *argv[]) {
if (argc < 2 || (argc % 2) != 0) {
fprintf(stderr, "** Warning: use the webhttrack frontend if available\n");
fprintf(stderr,
"usage: %s [--port <port>] [--bind <address>] [--ppid parent-pid] "
"<path-to-html-root-dir> [key value [key value]..]\n",
"usage: %s [--port <port>] [--ppid parent-pid] <path-to-html-root-dir> [key value [key value]..]\n",
argv[0]);
fprintf(stderr, "example: %s /usr/share/httrack/\n", argv[0]);
return 1;
@@ -271,14 +267,6 @@ int main(int argc, char *argv[]) {
fprintf(stderr, "couldn't set the port number to %s\n", argv[i + 1]);
return -1;
}
} else if (strcmp(argv[i], "--bind") == 0 && i + 1 < argc) {
/* empty would fall back to every interface, silently undoing the default
*/
if (!strnotempty(argv[i + 1])) {
fprintf(stderr, "--bind needs an address\n");
return -1;
}
bindAddr = argv[i + 1];
} else if (strcmp(argv[i], "--ppid") == 0 && i + 1 < argc) {
if (sscanf(argv[i + 1], "%u", &parentPid) != 1) {
fprintf(stderr, "couldn't set the parent PID to %s\n", argv[i + 1]);
@@ -305,7 +293,7 @@ int main(int argc, char *argv[]) {
}
/* launch */
ret = help_server(argv[1], defaultPort, bindAddr);
ret = help_server(argv[1], defaultPort);
htsthread_wait_n(background_threads - 1);
hts_uninit();
@@ -320,8 +308,10 @@ int main(int argc, char *argv[]) {
static int webhttrack_runmain(httrackp * opt, int argc, char **argv);
static void back_launch_cmd(void *pP) {
char *cmd = (char *) pP;
char **argv;
char **argv = (char **) malloct(1024 * sizeof(char *));
int argc = 0;
int i = 0;
int g = 0;
//
httrackp *opt;
@@ -332,19 +322,28 @@ static void back_launch_cmd(void *pP) {
commandReturnCmdl = strdup(cmd);
/* split */
argv = hts_split_cmdline(cmd, &argc);
if (argv == NULL) {
if (commandReturnMsg)
free(commandReturnMsg);
commandReturnMsg = strdup("could not parse the command line");
commandReturn = -1;
commandRunning = 0;
commandEnd = 1;
free(cmd);
return;
argv[0] = strdup("webhttrack");
argv[1] = cmd;
argc++;
i = 0;
while(cmd[i]) {
if (cmd[i] == '\t' || cmd[i] == '\r' || cmd[i] == '\n') {
cmd[i] = ' ';
}
i++;
}
i = 0;
while(cmd[i]) {
if (cmd[i] == '\"')
g = !g;
if (cmd[i] == ' ') {
if (!g) {
cmd[i] = '\0';
argv[argc++] = cmd + i + 1;
}
}
i++;
}
/* drop the program name the posted command line carries */
argv[0] = strdupt("webhttrack");
/* init */
hts_init();
@@ -373,7 +372,6 @@ static void back_launch_cmd(void *pP) {
/* free */
free(cmd);
freet(argv[0]);
freet(argv);
return;
}
@@ -429,11 +427,11 @@ static int webhttrack_runmain(httrackp * opt, int argc, char **argv) {
return ret;
}
static int help_server(char *dest_path, int defaultPort, const char *bindAddr) {
static int help_server(char *dest_path, int defaultPort) {
int returncode = 0;
char adr_prox[HTS_URLMAXSIZE * 2];
int port_prox;
T_SOC soc = smallserver_init_std(&port_prox, adr_prox, defaultPort, bindAddr);
T_SOC soc = smallserver_init_std(&port_prox, adr_prox, defaultPort);
if (soc != INVALID_SOCKET) {
char url[HTS_URLMAXSIZE * 2];
@@ -673,10 +671,10 @@ int __cdecl htsshow_loop(t_hts_callbackarg * carg, httrackp * opt, lien_back * b
char *eps = strchr(back[i].url_adr, '/');
int count;
if (ep != NULL && eps != NULL && ep < eps &&
(count = (int) (ep - back[i].url_adr)) < 4) {
if (ep != NULL && ep < eps
&& (count = (int) (ep - back[i].url_adr)) < 4) {
proto[0] = '\0';
strncatbuff(proto, back[i].url_adr, count);
strncat(proto, back[i].url_adr, count);
}
}
snprintf(StatsBuffer[index].state, sizeof(StatsBuffer[index].state),

View File

@@ -42,7 +42,7 @@ Please visit our Website: http://www.httrack.com
typedef struct t_StatsBuffer {
char name[1024];
char file[1024];
char state[288]; // a short label plus back->info[256]
char state[256];
char url_sav[HTS_URLMAXSIZE * 2]; // pour cancel
char url_adr[HTS_URLMAXSIZE * 2];
char url_fil[HTS_URLMAXSIZE * 2];

View File

@@ -46,7 +46,6 @@ Please visit our Website: http://www.httrack.com
#include "httrack.h"
#include "htslib.h"
#include "htscharset.h" // after htslib.h: winsock2.h must precede windows.h
#include "htsbacktrace.h"
/* Static definitions */
static int fexist(const char *s);
@@ -69,10 +68,11 @@ static int linput(FILE * fp, char *s, int max);
#ifdef HAVE_UNISTD_H
#include <unistd.h>
#endif
#ifdef HAVE_SYS_IOCTL_H
#include <sys/ioctl.h>
#endif
#include <ctype.h>
#if (defined(__linux) && defined(HAVE_EXECINFO_H))
#include <execinfo.h>
#define USES_BACKTRACE
#endif
/* END specific definitions */
static void __cdecl htsshow_init(t_hts_callbackarg * carg);
@@ -185,43 +185,6 @@ static void vt_home(void) {
printf("%s%s", VT_RESET, VT_GOTOXY("1", "0"));
}
/* Last known terminal geometry; the defaults are the classic VT100 size. */
static int term_cols = 80;
static int term_rows = 24;
/* Returns HTS_TRUE if the terminal size changed since the last call. */
static hts_boolean vt_size_refresh(void) {
int cols = 0;
int rows = 0;
#ifdef _WIN32
CONSOLE_SCREEN_BUFFER_INFO info;
const HANDLE console = GetStdHandle(STD_OUTPUT_HANDLE);
if (console != INVALID_HANDLE_VALUE &&
GetConsoleScreenBufferInfo(console, &info)) {
cols = info.srWindow.Right - info.srWindow.Left + 1;
rows = info.srWindow.Bottom - info.srWindow.Top + 1;
}
#elif defined(TIOCGWINSZ)
struct winsize ws;
if (ioctl(fileno(stdout), TIOCGWINSZ, &ws) == 0) {
cols = ws.ws_col;
rows = ws.ws_row;
}
#endif
if (cols <= 0)
cols = 80;
if (rows <= 0)
rows = 24;
if (cols == term_cols && rows == term_rows)
return HTS_FALSE;
term_cols = cols;
term_rows = rows;
return HTS_TRUE;
}
//
/*
@@ -232,8 +195,7 @@ static hts_boolean vt_size_refresh(void) {
#define STYLE_STATTEXT VT_UNBOLD
#define STYLE_STATRESET VT_UNBOLD
#define NStatsBuffer 14
/* Rows the stats block and "Current job" take above the in-progress list. */
#define NStatsHeaderRows 7
#define MAX_LEN_INPROGRESS 40
static int use_show;
static httrackp *global_opt = NULL;
@@ -329,7 +291,6 @@ static int __cdecl htsshow_start(t_hts_callbackarg * carg, httrackp * opt) {
use_show = 0;
if (opt->verbosedisplay == HTS_VERBOSE_FULL) {
use_show = 1;
(void) vt_size_refresh();
vt_clear();
}
return 1;
@@ -435,10 +396,6 @@ static int __cdecl htsshow_loop(t_hts_callbackarg * carg, httrackp * opt, lien_b
prev_mytime = mytime;
/* A resize leaves stale wrapped text on screen: repaint everything. */
if (vt_size_refresh())
vt_clear();
st[0] = '\0';
qsec2str(st, stat_time);
vt_home();
@@ -478,13 +435,6 @@ static int __cdecl htsshow_loop(t_hts_callbackarg * carg, httrackp * opt, lien_b
//
t_StatsBuffer StatsBuffer[NStatsBuffer];
/* Keep the frame within the terminal, or it scrolls the stats away. */
const int nstats =
min(NStatsBuffer, max(0, term_rows - NStatsHeaderRows));
/* URL budget: half the width, i.e. the historical 40 at 80 columns. */
const int maxurl =
min((int) sizeof(StatsBuffer[0].name) - 1, max(16, term_cols / 2));
{
int i;
@@ -499,11 +449,10 @@ static int __cdecl htsshow_loop(t_hts_callbackarg * carg, httrackp * opt, lien_b
}
}
for(k = 0; k < 2; k++) { // 0: lien en cours 1: autres liens
for (j = 0; (j < 3) && (index < nstats); j++) { // passe de priorité
for(j = 0; (j < 3) && (index < NStatsBuffer); j++) { // passe de priorité
int _i;
for (_i = 0 + k; (_i < max(back_max * k, 1)) && (index < nstats);
_i++) { // no lien
for(_i = 0 + k; (_i < max(back_max * k, 1)) && (index < NStatsBuffer); _i++) { // no lien
int i = (back_index + _i) % back_max; // commencer par le "premier" (l'actuel)
if (back[i].status >= 0) { // signifie "lien actif"
@@ -578,14 +527,16 @@ static int __cdecl htsshow_loop(t_hts_callbackarg * carg, httrackp * opt, lien_b
}
}
if ((l = (int) strlen(s)) < maxurl)
if ((l = (int) strlen(s)) < MAX_LEN_INPROGRESS)
strcpybuff(StatsBuffer[index].name, s);
else {
// couper
StatsBuffer[index].name[0] = '\0';
strncatbuff(StatsBuffer[index].name, s, maxurl / 2 - 2);
strncatbuff(StatsBuffer[index].name, s,
MAX_LEN_INPROGRESS / 2 - 2);
strcatbuff(StatsBuffer[index].name, "...");
strcatbuff(StatsBuffer[index].name, s + l - maxurl / 2 + 2);
strcatbuff(StatsBuffer[index].name,
s + l - MAX_LEN_INPROGRESS / 2 + 2);
}
if (back[i].r.totalsize >= 0) { // taille prédéfinie
@@ -646,7 +597,7 @@ static int __cdecl htsshow_loop(t_hts_callbackarg * carg, httrackp * opt, lien_b
{
int i;
for (i = 0; i < nstats; i++) {
for(i = 0; i < NStatsBuffer; i++) {
if (strnotempty(StatsBuffer[i].state)) {
printf(VT_CLREOL " %s - \t%s%s \t%s / \t%s", StatsBuffer[i].state,
StatsBuffer[i].name, StatsBuffer[i].file, int2bytes(&strc,
@@ -877,6 +828,21 @@ static void sig_doback(int blind) { // mettre en backing
#undef FD_ERR
#define FD_ERR 2
static void print_backtrace(void) {
#ifdef USES_BACKTRACE
void *stack[256];
const int size = backtrace(stack, sizeof(stack)/sizeof(stack[0]));
if (size != 0) {
backtrace_symbols_fd(stack, size, FD_ERR);
}
#else
const char msg[] = "No stack trace available on this OS :(\n";
if (write(FD_ERR, msg, sizeof(msg) - 1) != sizeof(msg) - 1) {
/* sorry GCC */
}
#endif
}
static size_t print_num(char *buffer, int num) {
size_t i, j;
if (num < 0) {
@@ -910,7 +876,7 @@ static void sig_fatal(int code) {
size += print_num(&buffer[size], code);
buffer[size++] = '\n';
(void) (write(FD_ERR, buffer, size) == size);
hts_print_backtrace(FD_ERR);
print_backtrace();
(void) (write(FD_ERR, msgreport, sizeof(msgreport) - 1)
== sizeof(msgreport) - 1);
abort();
@@ -933,7 +899,6 @@ static void sig_leave(int code) {
}
static void signal_handlers(void) {
hts_backtrace_init();
#ifdef _WIN32
signal(SIGINT, sig_leave); // ^C
signal(SIGTERM, sig_finish); // kill <process>

View File

@@ -44,7 +44,7 @@ typedef struct t_StatsBuffer t_StatsBuffer;
struct t_StatsBuffer {
char name[1024];
char file[1024];
char state[288]; // a short label plus back->info[256]
char state[256];
char BIGSTK url_sav[HTS_URLMAXSIZE * 2]; // pour cancel
char BIGSTK url_adr[HTS_URLMAXSIZE * 2];
char BIGSTK url_fil[HTS_URLMAXSIZE * 2];

View File

@@ -100,7 +100,6 @@
<ItemGroup>
<ClCompile Include="httrack.c" />
<ClCompile Include="htsbacktrace.c" />
</ItemGroup>
<!-- Pulls in libhttrack.lib and builds it first. -->

View File

@@ -59,10 +59,9 @@
<ItemDefinitionGroup>
<ClCompile>
<!-- LIBHTTRACK_EXPORTS turns HTSEXT_API into __declspec(dllexport); ZLIB_DLL
imports zlib from its DLL, ZLIB_CONST makes its next_in const as the
autotools build does. HTS_USEBROTLI/HTS_USEZSTD enable the br and zstd
content codings. Windows 7 floor, matching WinHTTrack. -->
<PreprocessorDefinitions>WIN32;_WINDOWS;_MBCS;_USRDLL;LIBHTTRACK_EXPORTS;ZLIB_DLL;ZLIB_CONST;HTS_USEBROTLI=1;HTS_USEZSTD=1;WINVER=0x0601;_WIN32_WINNT=0x0601;_CRT_SECURE_NO_WARNINGS;_CRT_NONSTDC_NO_DEPRECATE;%(PreprocessorDefinitions)</PreprocessorDefinitions>
imports zlib from its DLL. HTS_USEBROTLI/HTS_USEZSTD enable the br and
zstd content codings. Windows 7 floor, matching WinHTTrack. -->
<PreprocessorDefinitions>WIN32;_WINDOWS;_MBCS;_USRDLL;LIBHTTRACK_EXPORTS;ZLIB_DLL;HTS_USEBROTLI=1;HTS_USEZSTD=1;WINVER=0x0601;_WIN32_WINNT=0x0601;_CRT_SECURE_NO_WARNINGS;_CRT_NONSTDC_NO_DEPRECATE;%(PreprocessorDefinitions)</PreprocessorDefinitions>
<AdditionalIncludeDirectories>$(MSBuildThisFileDirectory);$(MSBuildThisFileDirectory)coucal;%(AdditionalIncludeDirectories)</AdditionalIncludeDirectories>
<WarningLevel>Level3</WarningLevel>
<MultiProcessorCompilation>true</MultiProcessorCompilation>
@@ -122,7 +121,6 @@
<ClCompile Include="htshelp.c" />
<ClCompile Include="htsindex.c" />
<ClCompile Include="htslib.c" />
<ClCompile Include="htscmdline.c" />
<ClCompile Include="htsurlport.c" />
<ClCompile Include="htsmd5.c" />
<ClCompile Include="htsmodules.c" />
@@ -131,7 +129,6 @@
<ClCompile Include="htsproxy.c" />
<ClCompile Include="htsrobots.c" />
<ClCompile Include="htsselftest.c" />
<ClCompile Include="htssinglefile.c" />
<ClCompile Include="htssniff.c" />
<ClCompile Include="htsthread.c" />
<ClCompile Include="htstools.c" />

View File

@@ -688,7 +688,7 @@ static PT_Element proxytrack_process_DAV_Request(PT_Indexes indexes,
PT_Element elt = PT_ElementNew();
elt->statuscode = 405;
strcpybuff(elt->msg, "Method Not Allowed");
strcpy(elt->msg, "Method Not Allowed");
return elt;
}
@@ -771,7 +771,7 @@ static PT_Element proxytrack_process_DAV_Request(PT_Indexes indexes,
PT_ReadIndex(indexes, StringBuff(itemUrl) + 1, FETCH_HEADERS);
if (file != NULL && file->statuscode == HTTP_OK) {
size = file->size;
if (file->lastmodified[0] != '\0') {
if (file->lastmodified) {
timestamp = get_time_rfc822(file->lastmodified);
}
if (timestamp == (time_t) 0) {
@@ -785,7 +785,7 @@ static PT_Element proxytrack_process_DAV_Request(PT_Indexes indexes,
}
timestamp = timestampRep;
}
if (file->contenttype[0] != '\0') {
if (file->contenttype) {
mimeType = file->contenttype;
}
}
@@ -822,7 +822,7 @@ static PT_Element proxytrack_process_DAV_Request(PT_Indexes indexes,
elt->statuscode = 207; /* Multi-Status */
strcpy(elt->charset, "utf-8");
strcpy(elt->contenttype, "text/xml");
strcpybuff(elt->msg, "Multi-Status");
strcpy(elt->msg, "Multi-Status");
StringFree(response);
fprintf(stderr, "RESPONSE:\n%s\n", elt->adr);
@@ -879,7 +879,7 @@ static PT_Element proxytrack_process_HTTP_List(PT_Indexes indexes,
elt->statuscode = HTTP_OK;
strcpy(elt->charset, "iso-8859-1");
strcpy(elt->contenttype, "text/html");
strcpybuff(elt->msg, "OK");
strcpy(elt->msg, "OK");
StringFree(html);
return elt;
}

View File

@@ -407,7 +407,7 @@ PT_Element PT_Index_HTML_BuildRootInfo(PT_Indexes indexes) {
elt->statuscode = HTTP_OK;
strcpy(elt->charset, "iso-8859-1");
strcpy(elt->contenttype, "text/html");
strcpybuff(elt->msg, "OK");
strcpy(elt->msg, "OK");
StringFree(html);
return elt;
}
@@ -579,10 +579,9 @@ PT_Index PT_LoadCache(const char *filename) {
if (chain != NULL) {
const char *scheme = link_has_authority(chain->name) ? "" : "http://";
/* dropped rather than truncated: empty already reads as "unset" */
if (!sprintfbuff(index->slots.common.startUrl, "%s%s", scheme,
(const char *) chain->name))
index->slots.common.startUrl[0] = '\0';
snprintf(index->slots.common.startUrl,
sizeof(index->slots.common.startUrl), "%s%s", scheme,
(const char *) chain->name);
}
}
}
@@ -828,18 +827,6 @@ PT_Element PT_ElementNew(void) {
return r;
}
/* ProxyTrack's htsblk_failf(): a clipped, diagnostic-only failure reason. */
static void PT_Element_failf(PT_Element r, const char *fmt, ...)
HTS_PRINTF_FUN(2, 3);
static void PT_Element_failf(PT_Element r, const char *fmt, ...) {
va_list args;
va_start(args, fmt);
(void) vslprintfbuff(r->msg, sizeof(r->msg), fmt, args);
va_end(args);
}
PT_Element PT_ReadCache(PT_Index index, const char *url, int flags) {
if (index != NULL && SAFE_INDEX(index)) {
return _IndexFuncts[index->type].PT_ReadCache(index, url, flags);
@@ -878,17 +865,12 @@ static PT_Element PT_ReadCache__New(PT_Index index, const char *url, int flags)
sprintf(headers + headersSize, "%s: "LLintP"\r\n", field, (LLint)(value)); \
(headersSize) += (int) strlen(headers + headersSize); \
} while(0)
/* refvalue_size is mandatory: the cache line is bounded only by the line
buffer, not by the destination. Clip rather than reject, since an
engine-written field can be wider than ours. */
#define ZIP_READFIELD_STRING(line, value, refline, refvalue, refvalue_size) \
do { \
if (line[0] != '\0' && strfield2(line, refline)) { \
(refvalue)[0] = '\0'; \
strlncatbuff(refvalue, value, refvalue_size, (refvalue_size) - 1); \
line[0] = '\0'; \
} \
} while (0)
#define ZIP_READFIELD_STRING(line, value, refline, refvalue) do { \
if (line[0] != '\0' && strfield2(line, refline)) { \
strcpy(refvalue, value); \
line[0] = '\0'; \
} \
} while(0)
#define ZIP_READFIELD_INT(line, value, refline, refvalue) do { \
if (line[0] != '\0' && strfield2(line, refline)) { \
int intval = 0; \
@@ -990,12 +972,9 @@ int PT_LoadCache__New(PT_Index index_, const char *filename) {
const char *scheme =
link_has_authority(filenameIndex) ? "" : "http://";
/* dropped rather than truncated; try the next entry */
if (sprintfbuff(index->startUrl, "%s%s", scheme,
filenameIndex))
firstSeen = 1;
else
index->startUrl[0] = '\0';
firstSeen = 1;
snprintf(index->startUrl, sizeof(index->startUrl), "%s%s",
scheme, filenameIndex);
}
}
} else {
@@ -1091,23 +1070,16 @@ static PT_Element PT_ReadCache__New_u(PT_Index index_, const char *url,
value++;
ZIP_READFIELD_INT(line, value, "X-In-Cache", dataincache);
ZIP_READFIELD_INT(line, value, "X-Statuscode", r->statuscode);
ZIP_READFIELD_STRING(line, value, "X-StatusMessage", r->msg,
sizeof(r->msg));
ZIP_READFIELD_STRING(line, value, "X-StatusMessage", r->msg); // msg
ZIP_READFIELD_INT(line, value, "X-Size", r->size); // size
ZIP_READFIELD_STRING(line, value, "Content-Type", r->contenttype,
sizeof(r->contenttype));
ZIP_READFIELD_STRING(line, value, "X-Charset", r->charset,
sizeof(r->charset));
ZIP_READFIELD_STRING(line, value, "Last-Modified",
r->lastmodified, sizeof(r->lastmodified));
ZIP_READFIELD_STRING(line, value, "Etag", r->etag,
sizeof(r->etag));
ZIP_READFIELD_STRING(line, value, "Location", r->location,
sizeof(location_default));
ZIP_READFIELD_STRING(line, value, "Content-Type", r->contenttype); // contenttype
ZIP_READFIELD_STRING(line, value, "X-Charset", r->charset); // contenttype
ZIP_READFIELD_STRING(line, value, "Last-Modified", r->lastmodified); // last-modified
ZIP_READFIELD_STRING(line, value, "Etag", r->etag); // Etag
ZIP_READFIELD_STRING(line, value, "Location", r->location); // 'location' pour moved
ZIP_READFIELD_STRING(line, value, "Content-Disposition",
r->cdispo, sizeof(r->cdispo));
ZIP_READFIELD_STRING(line, value, "X-Save", previous_save_,
sizeof(previous_save_));
r->cdispo); // Content-disposition
ZIP_READFIELD_STRING(line, value, "X-Save", previous_save_); // Original save filename
if (line[0] != '\0') {
int len = r->headers ? ((int) strlen(r->headers)) : 0;
int nlen =
@@ -1163,9 +1135,8 @@ static PT_Element PT_ReadCache__New_u(PT_Index index_, const char *url,
snprintf(previous_save, sizeof(previous_save), "%s%s",
index->path, previous_save_ + index->fixedPath);
} else {
PT_Element_failf(
r, "Bogus fixePath prefix for %s (prefixLen=%d)",
previous_save_, (int) index->fixedPath);
snprintf(r->msg, sizeof(r->msg), "Bogus fixePath prefix for %s (prefixLen=%d)",
previous_save_, (int) index->fixedPath);
r->statuscode = STATUSCODE_INVALID;
}
} else {
@@ -1184,7 +1155,7 @@ static PT_Element PT_ReadCache__New_u(PT_Index index_, const char *url,
// Peut-on stocker le fichier directement sur disque?
if (ok) {
if (r->msg[0] == '\0') {
strcpybuff(r->msg, "Cache Read Error : Unexpected error");
strcpy(r->msg, "Cache Read Error : Unexpected error");
}
} else { // lire en mémoire
@@ -1202,27 +1173,24 @@ static PT_Element PT_ReadCache__New_u(PT_Index index_, const char *url,
int last_errno = errno;
r->statuscode = STATUSCODE_INVALID;
PT_Element_failf(r,
"Read error in cache disk data: %s",
strerror(last_errno));
sprintf(r->msg, "Read error in cache disk data: %s",
strerror(last_errno));
}
r->adr[r->size] = '\0';
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg,
"Read error (memory exhausted) from cache");
strcpy(r->msg,
"Read error (memory exhausted) from cache");
}
fclose(fp);
} else {
r->statuscode = STATUSCODE_INVALID;
PT_Element_failf(
r, "Read error (can't open '%s') from cache",
file_convert(catbuff, sizeof(catbuff),
previous_save));
snprintf(r->msg, sizeof(r->msg), "Read error (can't open '%s') from cache",
file_convert(catbuff, sizeof(catbuff), previous_save));
}
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cached file name is invalid");
strcpy(r->msg, "Cached file name is invalid");
}
}
} else {
@@ -1234,12 +1202,12 @@ static PT_Element PT_ReadCache__New_u(PT_Index index_, const char *url,
free(r->adr);
r->adr = NULL;
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Read Data");
strcpy(r->msg, "Cache Read Error : Read Data");
} else
*(r->adr + r->size) = '\0';
} else { // erreur
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Memory Error");
strcpy(r->msg, "Cache Memory Error");
}
}
}
@@ -1247,21 +1215,21 @@ static PT_Element PT_ReadCache__New_u(PT_Index index_, const char *url,
} // si save==null, ne rien charger (juste en tête)
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Read Header Data");
strcpy(r->msg, "Cache Read Error : Read Header Data");
}
unzCloseCurrentFile(index->zFile);
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Open File");
strcpy(r->msg, "Cache Read Error : Open File");
}
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Bad Offset");
strcpy(r->msg, "Cache Read Error : Bad Offset");
}
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "File Cache Entry Not Found");
strcpy(r->msg, "File Cache Entry Not Found");
}
if (r->location[0] != '\0') {
r->location = strdup(r->location);
@@ -1602,11 +1570,9 @@ static int PT_LoadCache__Old(PT_Index index_, const char *filename) {
const char *scheme =
link_has_authority(line) ? "" : "http://";
/* dropped rather than truncated; try the next entry */
if (sprintfbuff(index->startUrl, "%s%s", scheme, line))
firstSeen = 1;
else
index->startUrl[0] = '\0';
firstSeen = 1;
snprintf(index->startUrl, sizeof(index->startUrl), "%s%s",
scheme, line);
}
}
@@ -1827,9 +1793,8 @@ static PT_Element PT_ReadCache__Old_u(PT_Index index_, const char *url,
snprintf(previous_save, sizeof(previous_save), "%s%s",
index->path, previous_save_ + index->fixedPath);
} else {
PT_Element_failf(r,
"Bogus fixePath prefix for %s (prefixLen=%d)",
previous_save_, (int) index->fixedPath);
snprintf(r->msg, sizeof(r->msg), "Bogus fixePath prefix for %s (prefixLen=%d)",
previous_save_, (int) index->fixedPath);
r->statuscode = STATUSCODE_INVALID;
}
} else {
@@ -1855,18 +1820,17 @@ static PT_Element PT_ReadCache__Old_u(PT_Index index_, const char *url,
if (r->adr != NULL) {
if (r->size > 0 && fread(r->adr, 1, r->size, fp) != r->size) {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Read error in cache disk data");
strcpy(r->msg, "Read error in cache disk data");
}
r->adr[r->size] = '\0';
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg,
"Read error (memory exhausted) from cache");
strcpy(r->msg, "Read error (memory exhausted) from cache");
}
fclose(fp);
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Previous cache file not found (2)");
strcpy(r->msg, "Previous cache file not found (2)");
}
}
} else {
@@ -1878,30 +1842,30 @@ static PT_Element PT_ReadCache__Old_u(PT_Index index_, const char *url,
free(r->adr);
r->adr = NULL;
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Read Data");
strcpy(r->msg, "Cache Read Error : Read Data");
} else
r->adr[r->size] = '\0';
} else { // erreur
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Memory Error");
strcpy(r->msg, "Cache Memory Error");
}
}
}
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Bad Data");
strcpy(r->msg, "Cache Read Error : Bad Data");
}
} else { // erreur
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Read Header");
strcpy(r->msg, "Cache Read Error : Read Header");
}
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Seek Failed");
strcpy(r->msg, "Cache Read Error : Seek Failed");
}
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "File Cache Entry Not Found");
strcpy(r->msg, "File Cache Entry Not Found");
}
if (r->location[0] != '\0') {
r->location = strdup(r->location);
@@ -2164,15 +2128,12 @@ int PT_LoadCache__Arc(PT_Index index_, const char *filename) {
return 0;
}
/* Same contract as ZIP_READFIELD_STRING, for the ARC reader's header lines. */
#define HTTP_READFIELD_STRING(line, value, refline, refvalue, refvalue_size) \
do { \
if (line[0] != '\0' && strfield2(line, refline)) { \
(refvalue)[0] = '\0'; \
strlncatbuff(refvalue, value, refvalue_size, (refvalue_size) - 1); \
line[0] = '\0'; \
} \
} while (0)
#define HTTP_READFIELD_STRING(line, value, refline, refvalue) do { \
if (line[0] != '\0' && strfield2(line, refline)) { \
strcpy(refvalue, value); \
line[0] = '\0'; \
} \
} while(0)
#define HTTP_READFIELD_INT(line, value, refline, refvalue) do { \
if (line[0] != '\0' && strfield2(line, refline)) { \
int intval = 0; \
@@ -2230,7 +2191,7 @@ static PT_Element PT_ReadCache__Arc_u(PT_Index index_, const char *url,
}
if ((pos = getArcField(index->line, 2)) != NULL) {
r->msg[0] = '\0';
strncatbuff(r->msg, pos, sizeof(r->msg) - 1);
strncat(r->msg, pos, sizeof(pos) - 1);
}
while(linput(index->file, index->line, sizeof(index->line) - 1)
&& index->line[0] != '\0') {
@@ -2241,16 +2202,11 @@ static PT_Element PT_ReadCache__Arc_u(PT_Index index_, const char *url,
*value = '\0';
for(value++; *value == ' ' || *value == '\t'; value++) ;
HTTP_READFIELD_INT(line, value, "Content-Length", r->size); // size
HTTP_READFIELD_STRING(line, value, "Content-Type", r->contenttype,
sizeof(r->contenttype));
HTTP_READFIELD_STRING(line, value, "Last-Modified",
r->lastmodified, sizeof(r->lastmodified));
HTTP_READFIELD_STRING(line, value, "Etag", r->etag,
sizeof(r->etag));
HTTP_READFIELD_STRING(line, value, "Location", r->location,
sizeof(location_default));
HTTP_READFIELD_STRING(line, value, "Content-Disposition",
r->cdispo, sizeof(r->cdispo));
HTTP_READFIELD_STRING(line, value, "Content-Type", r->contenttype); // contenttype
HTTP_READFIELD_STRING(line, value, "Last-Modified", r->lastmodified); // last-modified
HTTP_READFIELD_STRING(line, value, "Etag", r->etag); // Etag
HTTP_READFIELD_STRING(line, value, "Location", r->location); // 'location' pour moved
HTTP_READFIELD_STRING(line, value, "Content-Disposition", r->cdispo); // Content-disposition
if (line[0] != '\0') {
int len = r->headers ? ((int) strlen(r->headers)) : 0;
int nlen =
@@ -2290,7 +2246,7 @@ static PT_Element PT_ReadCache__Arc_u(PT_Index index_, const char *url,
fetchSize = dataLength - metaSize;
} else if (fetchSize > dataLength - metaSize) {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Truncated Data");
strcpy(r->msg, "Cache Read Error : Truncated Data");
}
r->size = 0;
if (r->statuscode != STATUSCODE_INVALID) {
@@ -2303,13 +2259,12 @@ static PT_Element PT_ReadCache__Arc_u(PT_Index index_, const char *url,
int last_errno = errno;
r->statuscode = STATUSCODE_INVALID;
PT_Element_failf(r, "Read error in cache disk data: %s",
strerror(last_errno));
sprintf(r->msg, "Read error in cache disk data: %s",
strerror(last_errno));
}
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg,
"Read error (memory exhausted) from cache");
strcpy(r->msg, "Read error (memory exhausted) from cache");
}
}
}
@@ -2317,21 +2272,21 @@ static PT_Element PT_ReadCache__Arc_u(PT_Index index_, const char *url,
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Read Header Error");
strcpy(r->msg, "Cache Read Error : Read Header Error");
}
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Read Header Error");
strcpy(r->msg, "Cache Read Error : Read Header Error");
}
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "Cache Read Error : Seek Error");
strcpy(r->msg, "Cache Read Error : Seek Error");
}
} else {
r->statuscode = STATUSCODE_INVALID;
strcpybuff(r->msg, "File Cache Entry Not Found");
strcpy(r->msg, "File Cache Entry Not Found");
}
if (r->location[0] != '\0') {
r->location = strdup(r->location);

View File

@@ -97,7 +97,6 @@
<ItemGroup>
<ClCompile Include="htsserver.c" />
<ClCompile Include="htsweb.c" />
<ClCompile Include="htscmdline.c" />
<ClCompile Include="htsurlport.c" />
</ItemGroup>

View File

@@ -1,8 +0,0 @@
#!/bin/bash
#
set -euo pipefail
# webhttrack posts its command line as one string: the argv split must grow past
# 1024 arguments and keep a quote inside a value out of the option parser.
httrack -O /dev/null -#test=cmdline-split run | grep -q "cmdline-split self-test OK"

View File

@@ -1,25 +0,0 @@
#!/bin/bash
#
# Buffer capacity for a 64-bit file size ('httrack -#test=growsize'): a -%S list
# file past 4GB must size its buffer exactly or be refused, never wrap.
set -euo pipefail
out=$(httrack -#test=growsize)
echo "$out"
test "$out" == "growsize self-test OK"
tmp=$(mktemp -d "${TMPDIR:-/tmp}/httrack_growsize.XXXXXX")
trap 'rm -rf "$tmp"' EXIT HUP INT QUIT PIPE TERM
echo '<html><body>hi</body></html>' >"$tmp/index.html"
printf -- '-*/zzmarker*\n' >"$tmp/rules.txt"
# the rules file lands in the URL/filter string, echoed back by the banner
run=$(httrack -O "$tmp/out" --quiet -n "-%S" "$tmp/rules.txt" \
"file://$tmp/index.html" 2>&1) || true
printf '%s\n' "$run" | grep -q 'zzmarker' || {
echo "FAIL: -%S rules file was not loaded"
printf '%s\n' "$run"
exit 1
}

View File

@@ -47,18 +47,3 @@ case "$err" in
exit 1
;;
esac
# An array source with no NUL must abort on the source bound rather than run
# off the array: the array half of the source-capacity selection.
err=$(httrack -#test=strsafe overflow-src "x" 2>&1) || true
case "$err" in
*"strsafe: NOT aborted"*)
echo "unterminated source array was NOT caught" >&2
exit 1
;;
*"size < sizeof_source"*) ;;
*)
echo "expected htssafe source-bound abort, got: $err" >&2
exit 1
;;
esac

View File

@@ -1,21 +0,0 @@
#!/bin/bash
#
# UTF-8 <-> system codepage ('httrack -#test=syscharset'), the pair an MBCS GUI
# uses at the Win32 ANSI entry points. WIN32 only, skipped elsewhere.
set -euo pipefail
rc=0
out=$(httrack -O /dev/null -#test=syscharset) || rc=$?
echo "$out"
[ "$rc" = 77 ] && exit 77
[ "$rc" = 0 ] || exit "$rc"
case "$out" in
"syscharset: acp="*": OK") ;;
*)
echo "FAIL: unexpected report"
exit 1
;;
esac

View File

@@ -1,32 +0,0 @@
#!/bin/bash
# An ARC entry's HTTP reason phrase must survive a proxytrack --convert
# round-trip; it used to be clipped to sizeof(char*) - 1 bytes.
set -euo pipefail
dir=$(mktemp -d)
trap 'rm -rf "$dir"' EXIT
printf 'HTTP/1.1 404 Not Found Here At All\r\nContent-Type: text/html\r\nLast-Modified: Wed, 01 Jan 2025 00:00:00 GMT\r\nContent-Length: 5\r\n\r\n' >"$dir/hdr"
printf 'hello' >"$dir/body"
alen=$(($(wc -c <"$dir/hdr") + $(wc -c <"$dir/body")))
# ARC 1.0: filedesc record and version block, then per entry
# <nl> <URL-record> <nl> <headers> <body>; the record's last field is that length.
{
printf 'filedesc://t.arc 0.0.0.0 20250101000000 text/plain 200 - - 0 t.arc 9\n'
printf '2 0 test\n'
printf '\n\n'
printf 'http://example.com/page.html 0.0.0.0 20250101000000 text/html 404 - - 0 t.arc %d\n' "$alen"
cat "$dir/hdr" "$dir/body"
} >"$dir/in.arc"
proxytrack --convert "$dir/out.arc" "$dir/in.arc" >/dev/null 2>&1
# HTTP/1.0 is the writer's own prefix (the input says 1.1), so a verbatim copy
# of the input cannot satisfy this match.
grep -aqF 'HTTP/1.0 404 Not Found Here At All' "$dir/out.arc" || {
echo "reason phrase lost in the ARC round-trip:" >&2
grep -a '^HTTP/' "$dir/out.arc" >&2 || echo "(no status line at all)" >&2
exit 1
}

View File

@@ -60,20 +60,13 @@ test -n "${url:-}" || fail "htsserver did not start: $(cat "${srvlog}")"
# Post a "start" whose -O dir is 'café' in the form's ISO-8859-1 charset.
"${python}" - "${url}" "${work}" <<'PY'
import re, sys, urllib.parse, urllib.request
import sys, urllib.parse, urllib.request
url, work = sys.argv[1], sys.argv[2]
outdir = work + "/caf\xe9" # 'café' as the single byte the browser would send
# Port 1 refuses at once: only the decoded -O dir matters, not the fetch.
cmd = "httrack --quiet --robots=0 http://127.0.0.1:1/x.html -O " + outdir
# The body is refused without the session id the server renders into each form.
# Note the wizard page is under /server/; a bare /step4.html is the doc page.
page = urllib.request.urlopen(url + "server/step4.html", timeout=20).read()
m = re.search(rb'name="sid" value="([0-9a-f]+)"', page)
if not m:
raise SystemExit("no session id in server/step4.html")
fields = [("sid", m.group(1).decode()),
("path", work), ("projname", "proj"), ("command_do", "start"),
fields = [("path", work), ("projname", "proj"), ("command_do", "start"),
("winprofile", "x"), ("command", cmd)]
body = "&".join("%s=%s" % (k, urllib.parse.quote(v, safe="", encoding="latin-1"))
for k, v in fields)

View File

@@ -1,70 +0,0 @@
#!/bin/bash
# #97: a terminal resize must repaint the whole screen, and the frame must
# follow the new geometry instead of the hardcoded 80x24 one.
set -eu
: "${top_srcdir:=..}"
testdir=$(cd "$(dirname "$0")" && pwd)
# shellcheck source=tests/testlib.sh
. "${testdir}/testlib.sh"
server=$(nativepath "${testdir}/local-server.py")
root=$(nativepath "${testdir}/server-root")
driver=$(nativepath "${testdir}/pty-resize.py")
if is_windows; then
echo "no pty on Windows; skipping" >&2
exit 77
fi
python=$(find_python) || ! echo "python3 not found; skipping" >&2 || exit 77
tmpdir=$(mktemp -d "${TMPDIR:-/tmp}/httrack_resize.XXXXXX") || exit 1
serverpid=
cleanup() {
stop_server "$serverpid"
rm -rf "$tmpdir"
}
trap cleanup EXIT HUP INT QUIT PIPE TERM
serverlog="${tmpdir}/server.log"
: >"$serverlog"
"$python" "$server" --root "$root" >"$serverlog" 2>&1 &
serverpid=$!
port=
for _ in $(seq 1 50); do
line=$(head -n1 "$serverlog" 2>/dev/null)
if test "${line%% *}" == "PORT"; then
port="${line#PORT }"
break
fi
kill -0 "$serverpid" 2>/dev/null || {
echo "server exited early: $(cat "$serverlog")"
exit 1
}
sleep 0.1
done
test -n "$port" || {
echo "could not discover server port"
exit 1
}
which httrack >/dev/null || {
echo "could not find httrack"
exit 1
}
# Trickled pages under a long path: the display keeps animating while the pty is
# resized under it, and the URL is truncated at 80 columns but not at 200.
deep="127.0.0.1:${port}/trickle/deep/a-long-directory-segment/another-long-segment"
out="${tmpdir}/crawl"
mkdir "$out"
rc=0
# The in-progress rows show host:port, not the scheme.
"$python" "$driver" "${tmpdir}/pty.raw" "${deep}/p" -- \
httrack "http://${deep}/index.html" \
-O "$out" -%v --robots=0 --disable-security-limits \
--timeout=30 --retries=1 -c4 --max-time 60 || rc=$?
test "$rc" -eq 0 || {
echo "pty capture tail:"
tail -c 600 "${tmpdir}/pty.raw" | cat -v
exit 1
}

View File

@@ -1,185 +0,0 @@
#!/bin/bash
#
# htsserver's POST redirect: the Location header is built from a client-supplied
# value, and the listen address is loopback unless --bind widens it.
set -euo pipefail
testdir=$(cd "$(dirname "$0")" && pwd)
distdir=${top_srcdir:-$(cd "${testdir}/.." && pwd)}
distdir=$(cd "${distdir}" && pwd)
# run_with_timeout/kill_tree: timeout(1) is absent on macOS
# shellcheck source=tests/testlib.sh
. "${testdir}/testlib.sh"
fail() {
echo "FAIL: $*" >&2
exit 1
}
command -v htsserver >/dev/null || fail "no htsserver in PATH"
command -v python3 >/dev/null || {
echo "python3 not found; skipping" >&2
exit 77
}
log=$(mktemp)
# start() runs inside a command substitution, so its $! never reaches this
# shell. Take the pid from the server's own announcement instead: a missed kill
# leaves an orphan holding the CI job open long after the suite has passed.
srvpid() { sed -n 's/^PID=//p' "${log}" 2>/dev/null | head -1; }
cleanup() {
stop
rm -f "${log}"
}
trap cleanup EXIT HUP INT QUIT PIPE TERM
freeport() {
python3 -c 'import socket
s = socket.socket()
s.bind(("127.0.0.1", 0))
print(s.getsockname()[1])
s.close()'
}
# Start htsserver, echo the announced URL. Extra args are passed through.
start() {
local port url
port=$(freeport)
: >"${log}"
(
trap '' TERM TTOU
exec htsserver "${distdir}/" --port "${port}" "$@" >"${log}" 2>&1
) &
for _ in $(seq 1 40); do
url=$(sed -n 's/^URL=//p' "${log}" 2>/dev/null) && test -n "${url}" && break
sleep 0.25
done
test -n "${url:-}" || fail "htsserver did not come up: $(cat "${log}")"
echo "${url}"
}
stop() {
local pid
pid=$(srvpid)
test -z "${pid}" || kill -9 "${pid}" 2>/dev/null || true
}
# The session id gates the request body, so a POST has to carry one. The server
# renders it into every form, which is where a browser picks it up too.
scrape_sid() {
python3 -c 'import socket, sys
s = socket.create_connection(("127.0.0.1", int(sys.argv[1])), 10)
s.settimeout(20)
s.sendall(b"GET /server/index.html HTTP/1.0\r\nHost: 127.0.0.1\r\n\r\n")
out = b""
while True:
b = s.recv(65536)
if not b:
break
out += b
s.close()
sys.stdout.write(out.decode("latin-1"))' "$1" |
sed -n 's/.*name="sid" value="\([0-9a-f]*\)".*/\1/p' | head -1
}
# POST redirect=$1 to 127.0.0.1:$2 with session id $3, print the raw headers.
post_redirect() {
python3 -c 'import socket, sys, urllib.parse
body = ("sid=" + sys.argv[3] + "&redirect="
+ urllib.parse.quote(sys.argv[1], safe=""))
req = ("POST / HTTP/1.0\r\nHost: 127.0.0.1\r\n"
"Content-type: application/x-www-form-urlencoded\r\n"
"Content-length: %d\r\n\r\n%s" % (len(body), body))
s = socket.create_connection(("127.0.0.1", int(sys.argv[2])), 10)
s.settimeout(20)
s.sendall(req.encode())
out = b""
# stop at the end of the header block: the server need not close the connection,
# and an unbounded recv() hung the macOS runner for over an hour.
while b"\r\n\r\n" not in out:
b = s.recv(65536)
if not b:
break
out += b
s.close()
sys.stdout.write(out.split(b"\r\n\r\n")[0].decode("latin-1"))' "$1" "$2" "$3"
}
portof() { echo "${1##*:}" | tr -d /; }
# "free"/"inuse"/"unusable": can $1 still be bound on 127.0.0.2? A wildcard
# listener takes the port on every address, a loopback-only one does not, so
# this discriminates where the announced URL cannot.
probe_alias() {
python3 -c 'import socket, sys
s = socket.socket()
try:
s.bind(("127.0.0.2", int(sys.argv[1])))
except OSError as e:
print("unusable" if e.errno in (99, 49, 10049) else "inuse")
else:
print("free")
finally:
s.close()' "$1"
}
# Non-repeating and not a round number: a truncation, a reorder or a dropped
# interior byte all stay visible.
long=$(python3 -c 'print("".join(chr(33 + i % 90) for i in range(4097)))')
url=$(start)
port=$(portof "${url}")
sid=$(scrape_sid "${port}")
test "${#sid}" -eq 32 || fail "did not scrape a 32-hex sid (got '${sid}')"
resp=$(post_redirect "${long}" "${port}" "${sid}") ||
fail "no response to an oversized redirect value"
stop
loc=$(printf '%s' "${resp}" | sed -n 's/^Location: //p' | tr -d '\r')
test "${loc}" = "${long}" ||
fail "oversized redirect not echoed whole (got ${#loc} bytes, want ${#long})"
# CR/LF must suppress the header outright. Sanitising it instead would keep the
# grep for an injected header quiet while still emitting a mangled Location.
url=$(start)
port=$(portof "${url}")
sid=$(scrape_sid "${port}")
resp=$(post_redirect '/foo
X-Injected: pwned' "${port}" "${sid}") || fail "no response to a CRLF redirect value"
stop
printf '%s' "${resp}" | grep -qi 'X-Injected' &&
fail "CRLF in the redirect value reached the response"
test "$(printf '%s' "${resp}" | grep -ci '^Location:')" -eq 0 ||
fail "CRLF redirect still emitted a Location header"
test "$(printf '%s' "${resp}" | grep -c '^HTTP/1\.')" -eq 1 ||
fail "CRLF redirect did not yield exactly one status line"
# Loopback unless asked otherwise. Assert the socket, not the announcement.
url=$(start)
port=$(portof "${url}")
alias_state=$(probe_alias "${port}")
stop
if test "${alias_state}" = unusable; then
echo "no 127.0.0.2 alias; skipping the listen-address assertions" >&2
else
test "${alias_state}" = free ||
fail "default listen address is not loopback-only (127.0.0.2:${port} taken)"
case "${url}" in
http://127.0.0.1:*) ;;
*) fail "default announcement is not loopback: ${url}" ;;
esac
url=$(start --bind 0.0.0.0)
port=$(portof "${url}")
alias_state=$(probe_alias "${port}")
stop
test "${alias_state}" = inuse ||
fail "--bind 0.0.0.0 did not take the wildcard (127.0.0.2:${port} ${alias_state})"
fi
# An empty --bind must not quietly fall back to every interface. Bounded: if it
# were accepted the server would listen instead of exiting.
run_with_timeout 15 htsserver "${distdir}/" --bind "" >"${log}" 2>&1 || true
grep -q -- "--bind needs an address" "${log}" ||
fail "empty --bind not refused: $(cat "${log}")"
echo "PASS"

View File

@@ -1,142 +0,0 @@
#!/bin/bash
#
# htsserver's session id must gate the request body: every field in it is
# written into the global key store, and "command" from there reaches the engine.
set -euo pipefail
testdir=$(cd "$(dirname "$0")" && pwd)
distdir=${top_srcdir:-$(cd "${testdir}/.." && pwd)}
distdir=$(cd "${distdir}" && pwd)
fail() {
echo "FAIL: $*" >&2
exit 1
}
command -v htsserver >/dev/null || fail "no htsserver in PATH"
command -v python3 >/dev/null || {
echo "python3 not found; skipping" >&2
exit 77
}
srv=
log=$(mktemp)
cleanup() {
test -z "${srv}" || kill -9 "${srv}" 2>/dev/null || true
rm -f "${log}"
}
trap cleanup EXIT HUP INT QUIT PIPE TERM
freeport() {
python3 -c 'import socket
s = socket.socket()
s.bind(("127.0.0.1", 0))
print(s.getsockname()[1])
s.close()'
}
# Echo the announced URL. Runs in a command substitution, so it is a
# subshell and cannot export the pid: the caller reads it back with srvpid.
start() {
local port url
port=$(freeport)
: >"${log}"
(
trap '' TERM TTOU
exec htsserver "${distdir}/" --port "${port}" >"${log}" 2>&1
) &
for _ in $(seq 1 40); do
url=$(sed -n 's/^URL=//p' "${log}" 2>/dev/null) && test -n "${url}" && break
sleep 0.25
done
test -n "${url:-}" || fail "htsserver did not come up: $(cat "${log}")"
echo "${url}"
}
# The server reports its own pid; the aliveness assertions below hang off it.
srvpid() { sed -n 's/^PID=//p' "${log}" | head -1; }
alive() { kill -0 "$1" 2>/dev/null; }
portof() { echo "${1##*:}" | tr -d /; }
# Raw request to 127.0.0.1:$1; $2 is the body ("" for a GET of the $3 page,
# default index). Prints the reply.
request() {
python3 -c 'import socket, sys
port, body = int(sys.argv[1]), sys.argv[2]
page = sys.argv[3] if len(sys.argv) > 3 else "index"
if body:
req = ("POST / HTTP/1.0\r\nHost: 127.0.0.1\r\n"
"Content-type: application/x-www-form-urlencoded\r\n"
"Content-length: %d\r\n\r\n%s" % (len(body), body))
else:
req = ("GET /server/%s.html HTTP/1.0\r\nHost: 127.0.0.1\r\n\r\n" % page)
s = socket.create_connection(("127.0.0.1", port), 10)
s.settimeout(30)
s.sendall(req.encode())
out = b""
while True:
b = s.recv(65536)
if not b:
break
out += b
s.close()
sys.stdout.write(out.decode("latin-1"))' "$1" "$2" ${3+"$3"}
}
# The server hands the sid to any client in every form; scraping it is the
# legitimate flow, and it is what makes the accept case below meaningful.
scrape_sid() {
request "$1" "" | sed -n 's/.*name="sid" value="\([0-9a-f]*\)".*/\1/p' | head -1
}
url=$(start)
port=$(portof "${url}")
srv=$(srvpid)
test -n "${srv}" || fail "htsserver did not report its pid"
alive "${srv}" || fail "htsserver is not running"
sid=$(scrape_sid "${port}")
# Control: without a real sid every assertion below would pass vacuously.
test "${#sid}" -eq 32 || fail "did not scrape a 32-hex sid from the page (got '${sid}')"
# Accept: the legitimate flow still works. "redirect" is the cheapest field
# with a reply that is visible in the headers.
resp=$(request "${port}" "sid=${sid}&redirect=/accepted")
printf '%s' "${resp}" | grep -q '^Location: /accepted' ||
fail "a body carrying the correct sid was refused"
# Refuse: missing and wrong. A missing one is the regression under test — the
# expected value used to be pre-seeded into the compared key, so omitting the
# field compared equal to itself.
for bad in "redirect=/nosid" "sid=&redirect=/empty" \
"sid=00000000000000000000000000000000&redirect=/wrong"; do
resp=$(request "${port}" "${bad}")
printf '%s' "${resp}" | grep -q '^Location:' &&
fail "body accepted without a valid sid: ${bad}"
# A refusal has to be a well-formed reply, not a headerless fragment: the
# release build used to emit only Content-length, which reads as a protocol
# error to any client and hides the reason.
printf '%s' "${resp}" | grep -q '^HTTP/1\.0 403 ' ||
fail "refusal was not a 403: ${bad}"
done
# Suppressing the reply is not the same as refusing the write: the body is
# applied to one global key store that later requests render from, and the
# command dispatcher reads it from there. Probe the store itself — step3
# interpolates ${projname} into its title — rather than the refused reply.
title() { request "$1" "" step3; }
request "${port}" "projname=UNAUTHWRITE" >/dev/null 2>&1 || true
title "${port}" | grep -q 'UNAUTHWRITE' &&
fail "a body without a valid sid was written to the key store"
# Paired accept case: without it the assertion above passes even if the server
# simply ignores every body, which would prove nothing.
request "${port}" "sid=${sid}&projname=AUTHWRITE" >/dev/null 2>&1 || true
title "${port}" | grep -q 'AUTHWRITE' ||
fail "a body carrying the correct sid was not written to the key store"
echo "PASS"

View File

@@ -1,90 +0,0 @@
#!/bin/bash
#
# proxytrack's WebDAV PROPFIND fallback must behave identically whether or not
# the cache entry supplies Content-Type and Last-Modified.
set -euo pipefail
: "${top_srcdir:=..}"
# shellcheck source=tests/testlib.sh
. "$top_srcdir/tests/testlib.sh"
python=$(find_python) || {
echo "python3 missing, skipping"
exit 77
}
command -v curl >/dev/null 2>&1 || {
echo "curl missing, skipping"
exit 77
}
# First test to run proxytrack as a live server, and MSYS cannot reap a native
# listener: the orphan wedged the whole Windows suite for 48 minutes (#595).
if is_windows; then
echo "windows: cannot reap a backgrounded proxytrack, skipping"
exit 77
fi
dir=$(mktemp -d)
ptpid=
cleanup() {
stop_server "$ptpid"
rm -rf "$dir"
}
trap cleanup EXIT
# Neither header is present, so file->contenttype and ->lastmodified stay "".
printf 'HTTP/1.1 200 OK\r\nContent-Length: 5\r\n\r\n' >"$dir/hdr"
printf 'hello' >"$dir/body"
alen=$(($(wc -c <"$dir/hdr") + $(wc -c <"$dir/body")))
{
printf 'filedesc://t.arc 0.0.0.0 20250101000000 text/plain 200 - - 0 t.arc 9\n'
printf '2 0 test\n'
printf '\n\n'
printf 'http://example.com/page.html 0.0.0.0 20250101000000 text/html 200 - - 0 t.arc %d\n' "$alen"
cat "$dir/hdr" "$dir/body"
} >"$dir/in.arc"
freeport() {
"$python" -c 'import socket; s=socket.socket(); s.bind(("127.0.0.1",0)); print(s.getsockname()[1]); s.close()'
}
proxyport=$(freeport)
icpport=$(freeport)
proxytrack "127.0.0.1:$proxyport" "127.0.0.1:$icpport" "$dir/in.arc" >"$dir/pt.log" 2>&1 &
ptpid=$!
waited=0
until grep -qE "HTTP Proxy installed on|Unable to (initialize a temporary server|create the server)" "$dir/pt.log"; do
kill -0 "$ptpid" 2>/dev/null || {
echo "FAIL: proxytrack exited before listening"
cat "$dir/pt.log"
exit 1
}
test "$waited" -lt 50 || {
echo "FAIL: proxytrack never announced its listen port"
exit 1
}
sleep 0.1
waited=$((waited + 1))
done
grep -q "HTTP Proxy installed on" "$dir/pt.log" || {
echo "FAIL: proxytrack failed to bind"
cat "$dir/pt.log"
exit 1
}
# --max-time: an unbounded read here would wedge the runner rather than fail
curl -s --max-time 30 -X PROPFIND -H "Depth: 1" \
"http://127.0.0.1:$proxyport/webdav/example.com/" >"$dir/resp.xml"
grep -q '<getcontenttype>application/octet-stream</getcontenttype>' "$dir/resp.xml" || {
echo "FAIL: missing Content-Type did not fall back to application/octet-stream"
cat "$dir/resp.xml"
exit 1
}
grep -q '<getlastmodified>Wed, 01 Jan 2025 00:00:00 GMT</getlastmodified>' "$dir/resp.xml" || {
echo "FAIL: missing Last-Modified did not fall back to the index timestamp"
cat "$dir/resp.xml"
exit 1
}
echo "OK: WebDAV listing falls back correctly for an entry with no Content-Type/Last-Modified"

View File

@@ -1,97 +0,0 @@
#!/bin/bash
# A fatal signal prints the raw backtrace, then names the frames
# backtrace_symbols_fd() leaves as module+offset (-fvisibility=hidden hides them).
set -eu
testdir=$(cd "$(dirname "$0")" && pwd)
# shellcheck source=tests/testlib.sh
. "${testdir}/testlib.sh"
if is_windows; then
echo "no backtrace() on Windows; skipping" >&2
exit 77
fi
command -v httrack >/dev/null || {
echo "could not find httrack" >&2
exit 1
}
tmpdir=$(mktemp -d "${TMPDIR:-/tmp}/httrack_symbolize.XXXXXX") || exit 1
trap 'rm -rf "$tmpdir"' EXIT HUP INT QUIT PIPE TERM
out="${tmpdir}/trace"
raw="${tmpdir}/trace-optout"
overflow="this string is far too long for the buffer"
# A frame glibc could not name. Stop at the offset: it prints the trailing
# "[0xADDR]" with or without a leading space depending on its version.
rawframe='(+0x[0-9a-f][0-9a-f]*)'
# The strsafe selftest aborts on purpose, which lands in sig_fatal.
rc=0
run_with_timeout 60 httrack -#test=strsafe overflow "$overflow" >"$out" 2>&1 || rc=$?
test "$rc" -ne 124 || {
echo "the crash handler did not finish within the deadline" >&2
exit 1
}
grep -q '^Caught signal ' "$out" || {
echo "no 'Caught signal' line:" >&2
cat "$out" >&2
exit 1
}
if grep -q 'No stack trace available on this OS' "$out"; then
echo "no backtrace() on this platform; skipping" >&2
exit 77
fi
# Unconditional and first: symbolizing must never cost the raw trace.
grep -q "$rawframe" "$out" || {
echo "raw module+offset frames are gone:" >&2
cat "$out" >&2
exit 1
}
command -v addr2line >/dev/null || {
echo "addr2line not installed; skipping" >&2
exit 77
}
# Control: resolve the same frames ourselves, so a stripped build cannot turn
# the assertion below into a vacuous pass.
oracle="${tmpdir}/oracle"
sed -n 's/^\([^(]*\)(+\(0x[0-9a-f]*\)).*$/\1 \2/p' "$out" |
while read -r mod off; do
test -f "$mod" || continue
addr2line -Cfip -e "$mod" "$off" 2>/dev/null || true
done >"$oracle"
grep -qE 'st_strsafe|strcpy_safe_' "$oracle" || {
echo "addr2line names no hidden symbol in this build; skipping" >&2
exit 77
}
# The payload. Dropping the raw frames first is what makes it specific: both
# names are static, so .dynsym could never have carried them. Not anchored on
# the "0xOFF:" line, the inline chain puts the name on either half.
grep -v "$rawframe" "$out" | grep -qE 'st_strsafe|strcpy_safe_' || {
echo "the handler did not name the hidden frames:" >&2
cat "$out" >&2
exit 1
}
# Opt-out: raw trace kept, nothing spawned.
export HTTRACK_NO_SYMBOLIZE=1
rc=0
run_with_timeout 60 httrack -#test=strsafe overflow "$overflow" >"$raw" 2>&1 || rc=$?
unset HTTRACK_NO_SYMBOLIZE
test "$rc" -ne 124 || {
echo "the opt-out run did not finish within the deadline" >&2
exit 1
}
grep -q "$rawframe" "$raw" || {
echo "the opt-out lost the raw trace:" >&2
cat "$raw" >&2
exit 1
}
! grep -q '^0x[0-9a-f]*: ' "$raw" || {
echo "the opt-out still symbolized:" >&2
cat "$raw" >&2
exit 1
}

View File

@@ -1,142 +0,0 @@
#!/bin/bash
#
# The wizard's size fields must render as distinct caps: sizemax is -M, othermax
# then maxhtml share -m.
set -euo pipefail
testdir=$(cd "$(dirname "$0")" && pwd)
distdir=${top_srcdir:-$(cd "${testdir}/.." && pwd)}
distdir=$(cd "${distdir}" && pwd)
# shellcheck source=tests/testlib.sh
. "${testdir}/testlib.sh"
fail() {
echo "FAIL: $*" >&2
exit 1
}
command -v htsserver >/dev/null || fail "no htsserver in PATH"
python=$(find_python) || {
echo "python3 not found; skipping" >&2
exit 77
}
work=$(mktemp -d "${TMPDIR:-/tmp}/webhttrack_maxsize.XXXXXX") || fail "no tmpdir"
srvlog=$(mktemp)
srv=
cleanup() {
# htsserver keeps SIGTERM ignored across its exec, so only -9 reaps it.
test -z "${srv}" || kill -9 "${srv}" 2>/dev/null || true
wait "${srv}" 2>/dev/null || true # absorb bash's async "Killed" notice
srv=
rm -rf "${work}" "${srvlog}"
}
trap cleanup EXIT HUP INT QUIT PIPE TERM
# webhttrack server on a pre-picked port; an isolated HOME keeps a stray
# ~/.httrack.ini out of it.
sport=$("${python}" -c 'import socket
s = socket.socket()
s.bind(("127.0.0.1", 0))
print(s.getsockname()[1])
s.close()')
(
trap '' TERM TTOU
export HOME="${work}"
exec htsserver "${distdir}/" --port "${sport}" >"${srvlog}" 2>&1
) &
srv=$!
for _ in $(seq 1 40); do
url=$(sed -n 's/^URL=//p' "${srvlog}") && test -n "${url}" && break
kill -0 "${srv}" 2>/dev/null || break
sleep 0.25
done
test -n "${url:-}" || fail "htsserver did not start: $(cat "${srvlog}")"
# Audit what step4.html renders back; without command_do nothing is launched.
"${python}" - "${url}" <<'PY' || fail "wizard rendered the wrong options (see above)"
import re, sys, urllib.parse, urllib.request
url = sys.argv[1]
html, other, site = "111111", "222222", "333333"
def post(fields):
# The body is refused without the session id the server puts in each form.
form = urllib.request.urlopen(url + "server/index.html", timeout=20).read()
m = re.search(rb'name="sid" value="([0-9a-f]+)"', form)
if not m:
raise SystemExit("no session id in server/index.html")
fields = [("sid", m.group(1).decode())] + fields
body = "&".join("%s=%s" % (k, urllib.parse.quote(v)) for k, v in fields)
req = urllib.request.Request(url + "server/step4.html",
data=body.encode("latin-1"), method="POST")
return urllib.request.urlopen(req, timeout=20).read().decode("latin-1")
def textarea(page, name):
m = re.search(r'<textarea name="%s".*?>(.*?)</textarea>' % name, page, re.S)
if m is None:
sys.exit("no %s textarea in the rendered step4.html" % name)
return m.group(1)
def want(ok, msg, ctx):
if not ok:
sys.exit("%s\nrendered:%s" % (msg, ctx))
page = post([("maxhtml", html), ("othermax", other), ("sizemax", site),
("dos", "2")])
cmd = textarea(page, "command")
want("--max-size=" + site in cmd, "site cap not emitted as --max-size", cmd)
want("--max-files=" + site not in cmd, "site cap still an --max-files", cmd)
# Positive controls: both per-file caps still ride -m, one option each.
want("--max-files=" + other in cmd, "non-HTML cap not emitted as --max-files",
cmd)
want("--max-files=," + html in cmd, "HTML cap not emitted as --max-files=,",
cmd)
want(cmd.count("--max-files=") == 2, "unexpected --max-files count", cmd)
want(cmd.index("--max-files=" + other) < cmd.index("--max-files=," + html),
"the bare --max-files must precede the --max-files=, form", cmd)
# winprofile.ini feeds the same three caps to the Windows GUI, one key each.
ini = textarea(page, "winprofile").replace("\r\n", "\n") # the ini is CRLF
want("\nDos=2" in ini, "Dos not written from the dos field", ini)
want("\nMaxHtml=" + html + "\n" in ini, "MaxHtml not written from maxhtml", ini)
want("\nMaxOther=" + other + "\n" in ini, "MaxOther not written from othermax",
ini)
want("\nMaxAll=" + site + "\n" in ini, "MaxAll not written from sizemax", ini)
# A bare --max-files= (an unguarded empty field) parses as -m with no digits,
# silently clearing the html cap.
page = post([("maxhtml", ""), ("othermax", ""), ("sizemax", ""), ("dos", "2")])
cmd = textarea(page, "command")
want("--max-rate=" in cmd, "empty size fields rendered no command line", cmd)
want("--max-files" not in cmd and "--max-size" not in cmd,
"empty size fields still rendered a size option", cmd)
# A mistyped ${LANG_OK] key renders the OK button's label empty.
opts = urllib.request.urlopen(url + "server/option2b.html",
timeout=20).read().decode("latin-1")
want("LANG_OK" not in opts, "unexpanded LANG_OK in option2b.html", "")
want('<input type="submit" value="OK"' in opts, "no OK label in option2b.html",
"")
PY
cleanup
# A bare -m<n> resets the html limit, so only the bare-then-comma order keeps
# both caps live. basic.html is 487 bytes, well past the 10-byte html cap.
crawl() {
bash "${distdir}/tests/local-crawl.sh" "$@"
}
crawl --errors 1 --log-found 'File too big' \
httrack 'BASEURL/simple/basic.html' '--max-files=,10'
crawl --errors 1 --log-found 'File too big' \
httrack 'BASEURL/simple/basic.html' '--max-files=500000' '--max-files=,10'
crawl --errors 0 --found 'simple/basic.html' --log-not-found 'File too big' \
httrack 'BASEURL/simple/basic.html' '--max-files=,10' '--max-files=500000'
echo "PASS"

View File

@@ -1,135 +0,0 @@
#!/bin/bash
#
# A browser will not follow an http: page to a file: URL, so the GUI's "browse
# the mirror" link has to go through the server's own /website/ route.
set -euo pipefail
testdir=$(cd "$(dirname "$0")" && pwd)
distdir=${top_srcdir:-$(cd "${testdir}/.." && pwd)}
distdir=$(cd "${distdir}" && pwd)
# shellcheck source=tests/testlib.sh
. "${testdir}/testlib.sh"
fail() {
echo "FAIL: $*" >&2
exit 1
}
command -v htsserver >/dev/null || fail "no htsserver in PATH"
python=$(find_python) || {
echo "python3 not found; skipping" >&2
exit 77
}
# Control: on a path that does not resolve, grep "passes" without reading anything.
test -f "${distdir}/html/server/finished.html" || fail "no GUI pages under ${distdir}"
bad=$(grep -l 'file://' "${distdir}"/html/server/finished.html || true)
test -z "${bad}" || fail "file: link left in: ${bad}"
work=$(mktemp -d "${TMPDIR:-/tmp}/webhttrack_browse.XXXXXX") || fail "no tmpdir"
srvlog=$(mktemp)
srv=
srvpid=
cleanup() {
# htsserver keeps SIGTERM ignored across its exec, so only -9 reaps it.
test -z "${srvpid}" || kill -9 "${srvpid}" 2>/dev/null || true
test -z "${srv}" || kill -9 "${srv}" 2>/dev/null || true
wait "${srv}" 2>/dev/null || true # absorb bash's async "Killed" notice
rm -rf "${work}" "${srvlog}"
}
trap cleanup EXIT HUP INT QUIT PIPE TERM
# Stand in for a finished mirror: the crawl itself is not under test.
proj="${work}/websites/proj"
mkdir -p "${proj}"
printf 'MIRROR-PROBE-OK\n' >"${proj}/probe.txt"
printf '<html><body>MIRROR-INDEX-OK</body></html>\n' >"${proj}/index.html"
# An isolated HOME keeps a stray ~/.httrack.ini out of the server's settings.
sport=$("${python}" -c 'import socket
s = socket.socket()
s.bind(("127.0.0.1", 0))
print(s.getsockname()[1])
s.close()')
(
trap '' TERM TTOU
export HOME="${work}"
exec htsserver "${distdir}/" --port "${sport}" >"${srvlog}" 2>&1
) &
srv=$!
for _ in $(seq 1 40); do
url=$(sed -n 's/^URL=//p' "${srvlog}") && test -n "${url}" && break
kill -0 "${srv}" 2>/dev/null || break
sleep 0.25
done
test -n "${url:-}" || fail "htsserver did not start: $(cat "${srvlog}")"
srvpid=$(sed -n 's/^PID=//p' "${srvlog}") # absent on Windows
# htsserver resolves the posted paths itself, so hand it native ones.
"${python}" - "${url}" "$(nativepath "${work}/websites")" <<'PY' || fail "browse-link checks failed"
import re, sys, urllib.error, urllib.parse, urllib.request
url, base = sys.argv[1].rstrip("/"), sys.argv[2]
rc = 0
def check(ok, what):
global rc
print(("ok: " if ok else "FAIL: ") + what)
if not ok:
rc = 1
def get(path):
# The UI is served ISO-8859-1, so stay in bytes.
return urllib.request.urlopen(url + path, timeout=20).read()
# No crawl has run, so no commandRunning/commandEnd override swaps the page out.
finished = get("/server/finished.html").decode("latin-1")
# Control: an empty or unrendered page would pass "no file:" on its own.
check("HTTrack Website Copier" in finished, "finished.html rendered")
check("file://" not in finished, "finished.html has no file: link")
# Control: /website/ 404s until a project is set, so the fetches below are real.
try:
get("/website/probe.txt")
check(False, "/website/ served with no project set")
except urllib.error.HTTPError as e:
check(e.code == 404, "/website/ is bound to a project (got %d)" % e.code)
# Point the server at the mirror the way step4.html does; a body without the
# session id is refused outright. Only a profile save records the served root,
# so post one; command_do=save runs no crawl.
sid = re.search(r'name="sid" value="([0-9a-f]+)"', finished)
if sid is None:
print("FAIL: no sid in the rendered form")
sys.exit(1)
fields = [("sid", sid.group(1)), ("path", base), ("projname", "proj"),
("command", "httrack"), ("command_do", "save"), ("winprofile", "x")]
body = "&".join("%s=%s" % (k, urllib.parse.quote(v, safe="")) for k, v in fields)
urllib.request.urlopen(urllib.request.Request(
url + "/server/step4.html", data=body.encode("latin-1"), method="POST"),
timeout=20).read()
check(get("/website/probe.txt") == b"MIRROR-PROBE-OK\n", "mirror file over http")
check("MIRROR-INDEX-OK" in get("/website/index.html").decode("latin-1"),
"mirror index over http")
body = get("/server/finished.html").decode("latin-1")
# Match the anchor by its mirror-path label: the list below it links /website/
# too, so a bare "is the route mentioned" check passes on the unfixed page.
check(re.search(r'href="/website/index\.html"[^>]*>\s*' +
re.escape(base + "/proj"), body) is not None,
"the mirror-path link points at the served mirror")
check("file://" not in body, "finished.html has no file: link")
sys.exit(rc)
PY
# A leaked htsserver wedges the parallel harness behind a green log.
cleanup
! kill -0 "${srv}" 2>/dev/null || fail "htsserver ${srv} survived"
echo "PASS"

View File

@@ -1,148 +0,0 @@
#!/bin/bash
#
# The wizard renders its httrack command line as one string, which the engine
# splits back into argv: a double quote in a field must reach the split escaped,
# or the rest of the value is parsed as fresh options (-V runs a shell command).
set -euo pipefail
testdir=$(cd "$(dirname "$0")" && pwd)
distdir=${top_srcdir:-$(cd "${testdir}/.." && pwd)}
distdir=$(cd "${distdir}" && pwd)
fail() {
echo "FAIL: $*" >&2
exit 1
}
command -v htsserver >/dev/null || fail "no htsserver in PATH"
command -v python3 >/dev/null || {
echo "python3 not found; skipping" >&2
exit 77
}
srv=
log=$(mktemp)
cleanup() {
test -z "${srv}" || kill -9 "${srv}" 2>/dev/null || true
rm -f "${log}"
}
trap cleanup EXIT HUP INT QUIT PIPE TERM
freeport() {
python3 -c 'import socket
s = socket.socket()
s.bind(("127.0.0.1", 0))
print(s.getsockname()[1])
s.close()'
}
# Echo the announced URL. Runs in a command substitution, so it is a
# subshell and cannot export the pid: the caller reads it back with srvpid.
start() {
local port url
port=$(freeport)
: >"${log}"
(
trap '' TERM TTOU
exec htsserver "${distdir}/" --port "${port}" >"${log}" 2>&1
) &
for _ in $(seq 1 40); do
url=$(sed -n 's/^URL=//p' "${log}" 2>/dev/null) && test -n "${url}" && break
sleep 0.25
done
test -n "${url:-}" || fail "htsserver did not come up: $(cat "${log}")"
echo "${url}"
}
srvpid() { sed -n 's/^PID=//p' "${log}" | head -1; }
portof() { echo "${1##*:}" | tr -d /; }
# Raw request to 127.0.0.1:$1; $2 is the body ("" for a GET of the $3 page,
# default index). Prints the reply.
request() {
python3 -c 'import socket, sys
port, body = int(sys.argv[1]), sys.argv[2]
page = sys.argv[3] if len(sys.argv) > 3 else "index"
if body:
req = ("POST / HTTP/1.0\r\nHost: 127.0.0.1\r\n"
"Content-type: application/x-www-form-urlencoded\r\n"
"Content-length: %d\r\n\r\n%s" % (len(body), body))
else:
req = ("GET /server/%s.html HTTP/1.0\r\nHost: 127.0.0.1\r\n\r\n" % page)
s = socket.create_connection(("127.0.0.1", port), 10)
s.settimeout(30)
s.sendall(req.encode())
out = b""
while True:
b = s.recv(65536)
if not b:
break
out += b
s.close()
sys.stdout.write(out.decode("latin-1"))' "$1" "$2" ${3+"$3"}
}
scrape_sid() {
request "$1" "" | sed -n 's/.*name="sid" value="\([0-9a-f]*\)".*/\1/p' | head -1
}
# urlencode the key=value pairs given as arguments
formencode() {
python3 -c 'import sys, urllib.parse
print(urllib.parse.urlencode([tuple(a.split("=", 1)) for a in sys.argv[1:]]))' "$@"
}
url=$(start)
port=$(portof "${url}")
srv=$(srvpid)
test -n "${srv}" || fail "htsserver did not report its pid"
sid=$(scrape_sid "${port}")
test "${#sid}" -eq 32 || fail "did not scrape a 32-hex sid from the page (got '${sid}')"
# Fill the wizard fields the command line quotes, then read back the generated
# command line: the user-agent carries a break-out attempt, the footer a
# backslash (which must survive the escape round trip), the project name a plain
# value.
body=$(formencode "sid=${sid}" 'user=Moz" -V "touch /tmp/pwn' 'footer=a\b"c' \
"path=/tmp/p" "projname=plain proj" 'urls=http://x/a"b' 'url2=+*.png"')
request "${port}" "${body}" >/dev/null
cmdline=$(request "${port}" "" step4 |
sed -n '/<textarea name="command"/,/<\/textarea>/p')
# Control: without the fields in the page every assertion below is vacuous.
grep -q -- '--user-agent' <<<"${cmdline}" ||
fail "no --user-agent in the generated command line (probe blind)"
# A plain value is passed through untouched: the escaping must not mangle the
# ordinary case.
grep -qF -- '--path "/tmp/p/plain proj"' <<<"${cmdline}" ||
fail "a plain quoted value was not passed through: ${cmdline}"
# The quote is escaped, so the split keeps it inside the value...
grep -qF -- '--user-agent "Moz\" -V \"touch /tmp/pwn"' <<<"${cmdline}" ||
fail "the quote in the user-agent was not escaped: ${cmdline}"
# ...and the raw form, which would end the argument and hand -V to the option
# parser, is gone.
grep -qF -- '--user-agent "Moz" -V "touch' <<<"${cmdline}" &&
fail "the user-agent still closes its argument early: ${cmdline}"
# A backslash is escaped too, or the split would eat it along with the quote
# that follows.
grep -qF -- '--footer "a\\b\"c"' <<<"${cmdline}" ||
fail "the backslash in the footer was not escaped: ${cmdline}"
# The url and filter fields sit outside quotes, where a backslash cannot escape
# anything: one quote there flips the parity of every quote after it, so the
# escaping above would protect nothing. They must not emit a raw quote at all.
grep -qF -- 'http://x/a%22b' <<<"${cmdline}" ||
fail "the quote in the url field was not neutralised: ${cmdline}"
grep -qF -- '+*.png%22' <<<"${cmdline}" ||
fail "the quote in the filter field was not neutralised: ${cmdline}"
grep -qF -- 'http://x/a"b' <<<"${cmdline}" &&
fail "the url field still emits a raw quote: ${cmdline}"
echo "PASS"

View File

@@ -1,167 +0,0 @@
#!/bin/bash
#
# htsserver serves the crawled mirror under /website/ alongside its own GUI:
# mirrored pages must skip the ${...} expander, or ${_sid} leaks the session id.
set -euo pipefail
testdir=$(cd "$(dirname "$0")" && pwd)
distdir=${top_srcdir:-$(cd "${testdir}/.." && pwd)}
distdir=$(cd "${distdir}" && pwd)
fail() {
echo "FAIL: $*" >&2
exit 1
}
command -v htsserver >/dev/null || fail "no htsserver in PATH"
command -v python3 >/dev/null || {
echo "python3 not found; skipping" >&2
exit 77
}
log=$(mktemp)
work=$(mktemp -d)
csrv=
# start() runs in a command substitution, so its $! never reaches this shell. A
# missed kill leaves an orphan holding the CI job open long after a green suite.
srvpid() { sed -n 's/^PID=//p' "${log}" 2>/dev/null | head -1; }
cleanup() {
local pid
pid=$(srvpid)
test -z "${pid}" || kill -9 "${pid}" 2>/dev/null || true
test -z "${csrv}" || kill -9 "${csrv}" 2>/dev/null || true
wait "${csrv}" 2>/dev/null || true # absorb bash's async "Killed" notice
rm -rf "${log}" "${work}"
}
trap cleanup EXIT HUP INT QUIT PIPE TERM
freeport() {
python3 -c 'import socket
s = socket.socket()
s.bind(("127.0.0.1", 0))
print(s.getsockname()[1])
s.close()'
}
# Echo the announced URL.
start() {
local port url
port=$(freeport)
: >"${log}"
(
trap '' TERM TTOU
exec htsserver "${distdir}/" --port "${port}" >"${log}" 2>&1
) &
for _ in $(seq 1 40); do
url=$(sed -n 's/^URL=//p' "${log}" 2>/dev/null) && test -n "${url}" && break
sleep 0.25
done
test -n "${url:-}" || fail "htsserver did not come up: $(cat "${log}")"
echo "${url}"
}
portof() { echo "${1##*:}" | tr -d /; }
# GET $2 from 127.0.0.1:$1, headers into $3 and the body, byte for byte, into $4.
fetch() {
python3 -c 'import socket, sys
s = socket.create_connection(("127.0.0.1", int(sys.argv[1])), 10)
s.settimeout(20)
s.sendall(("GET %s HTTP/1.0\r\nHost: 127.0.0.1\r\n\r\n" % sys.argv[2]).encode())
out = b""
while True:
b = s.recv(65536)
if not b:
break
out += b
s.close()
head, _, body = out.partition(b"\r\n\r\n")
open(sys.argv[3], "wb").write(head)
open(sys.argv[4], "wb").write(body)' "$1" "$2" "$3" "$4"
}
# POST the remaining args as urlencoded key=value fields to 127.0.0.1:$1.
post() {
local port=$1
shift
python3 -c 'import socket, sys, urllib.parse
body = "&".join("%s=%s" % (k, urllib.parse.quote(v, safe=""))
for k, v in (a.split("=", 1) for a in sys.argv[2:])).encode()
s = socket.create_connection(("127.0.0.1", int(sys.argv[1])), 10)
s.settimeout(30)
s.sendall(b"POST /step4.html HTTP/1.0\r\nHost: 127.0.0.1\r\n"
b"Content-type: application/x-www-form-urlencoded\r\n"
b"Content-length: %d\r\n\r\n" % len(body) + body)
while s.recv(65536):
pass
s.close()' "${port}" "$@" >/dev/null
}
# LF-only, and a line ending in a backslash: the expander rewrites both, so a
# mangled reply fails the byte comparison even where no directive is present.
mirror="${work}/proj/hostile.html"
hdr="${work}/hdr"
body="${work}/body"
# The server merges $HOME/.httrack.ini into the same store on the first request;
# point it somewhere empty so a developer's own file cannot shadow the fields.
export HOME="${work}"
url=$(start)
port=$(portof "${url}")
# Positive control: the GUI's own templates must still expand. This is also
# where the real token comes from, so its absence below can be asserted.
fetch "${port}" /server/index.html "${hdr}" "${body}"
sid=$(sed -n 's/.*name="sid" value="\([0-9a-f]*\)".*/\1/p' "${body}" | head -1)
test "${#sid}" -eq 32 || fail "GUI page did not expand \${sid} (got '${sid}')"
# /website/ serves only the root the server itself recorded, so save a profile
# (no command_do=start, so nothing crawls) to create it, then plant the file.
post "${port}" "sid=${sid}" command=httrack command_do=save winprofile=x \
"path=${work}" projname=proj
test -f "${work}/proj/hts-cache/winprofile.ini" ||
fail "profile save did not create the project: $(cat "${log}")"
mkdir -p "$(dirname "${mirror}")"
# shellcheck disable=SC2016 # the directives are the payload, not shell expansions
printf 'Hostile mirrored page.\nsid=${_sid} copy=${sid}\ntrailing backslash: \\\n' \
>"${mirror}"
fetch "${port}" /website/hostile.html "${hdr}" "${body}"
grep -q '^HTTP/1\.0 200 ' "${hdr}" || fail "mirrored page not served: $(head -1 "${hdr}")"
grep -qF "${sid}" "${body}" &&
fail "the session id was expanded into mirrored content"
# shellcheck disable=SC2016 # the directives are the payload, not shell expansions
grep -qF '${_sid}' "${body}" || fail "\${_sid} did not survive verbatim"
# shellcheck disable=SC2016
grep -qF '${sid}' "${body}" || fail "\${sid} did not survive verbatim"
cmp -s "${mirror}" "${body}" || fail "mirrored file not served byte for byte"
# The mirror stays browsable: verbatim must not mean served as a download.
grep -qi '^Content-type: text/html' "${hdr}" ||
fail "mirrored page lost its text/html type: $(cat "${hdr}")"
# The other direction: while a crawl runs, every .html request is overridden to
# the GUI's own refresh page, so a /website/ URL stops naming mirrored content.
# /trickle/ dribbles for a minute, which holds the crawl open for the probe.
clog="${work}/content.log"
python3 "${testdir}/local-server.py" --root "${work}" >"${clog}" 2>&1 &
csrv=$!
for _ in $(seq 1 40); do
cport=$(sed -n 's/^PORT //p' "${clog}") && test -n "${cport}" && break
kill -0 "${csrv}" 2>/dev/null || break
sleep 0.25
done
test -n "${cport:-}" || fail "content server did not come up: $(cat "${clog}")"
post "${port}" "sid=${sid}" "path=${work}" projname=crawl winprofile=x \
command_do=start \
"command=httrack --quiet --robots=0 http://127.0.0.1:${cport}/trickle/ -O ${work}/crawl"
fetch "${port}" /website/hostile.html "${hdr}" "${body}"
grep -q '^HTTP/1\.0 200 ' "${hdr}" ||
fail "running crawl: /website/ was not overridden to the GUI page: $(head -1 "${hdr}")"
grep -qF "'crawl' - HTTrack Website Copier" "${body}" ||
fail "the overridden GUI page was not the expanded refresh page"
echo "PASS"

View File

@@ -1,168 +0,0 @@
#!/bin/bash
#
# /website/ is served from the project directory htsserver set up itself, never
# from a root the request body names, and composing that path must stay bounded.
set -euo pipefail
testdir=$(cd "$(dirname "$0")" && pwd)
distdir=${top_srcdir:-$(cd "${testdir}/.." && pwd)}
distdir=$(cd "${distdir}" && pwd)
fail() {
echo "FAIL: $*" >&2
exit 1
}
command -v htsserver >/dev/null || fail "no htsserver in PATH"
command -v python3 >/dev/null || {
echo "python3 not found; skipping" >&2
exit 77
}
srv=
log=$(mktemp)
base=$(mktemp -d)
cleanup() {
test -z "${srv}" || kill -9 "${srv}" 2>/dev/null || true
rm -f "${log}"
rm -rf "${base}"
}
trap cleanup EXIT HUP INT QUIT PIPE TERM
freeport() {
python3 -c 'import socket
s = socket.socket()
s.bind(("127.0.0.1", 0))
print(s.getsockname()[1])
s.close()'
}
# Echo the announced URL. Runs in a command substitution, so it is a
# subshell and cannot export the pid: the caller reads it back with srvpid.
start() {
local port url
port=$(freeport)
: >"${log}"
(
trap '' TERM TTOU
exec htsserver "${distdir}/" --port "${port}" >"${log}" 2>&1
) &
for _ in $(seq 1 40); do
url=$(sed -n 's/^URL=//p' "${log}" 2>/dev/null) && test -n "${url}" && break
sleep 0.25
done
test -n "${url:-}" || fail "htsserver did not come up: $(cat "${log}")"
echo "${url}"
}
# The server reports its own pid; the aliveness assertion below hangs off it.
srvpid() { sed -n 's/^PID=//p' "${log}" | head -1; }
alive() { kill -0 "$1" 2>/dev/null; }
portof() { echo "${1##*:}" | tr -d /; }
# Raw request to 127.0.0.1:$1: GET the path $2, or POST the body $3 to / when
# $2 is empty. Prints the reply.
request() {
python3 -c 'import socket, sys
port, path, body = int(sys.argv[1]), sys.argv[2], sys.argv[3]
if path:
req = "GET %s HTTP/1.0\r\nHost: 127.0.0.1\r\n\r\n" % path
else:
req = ("POST / HTTP/1.0\r\nHost: 127.0.0.1\r\n"
"Content-type: application/x-www-form-urlencoded\r\n"
"Content-length: %d\r\n\r\n%s" % (len(body), body))
s = socket.create_connection(("127.0.0.1", port), 10)
s.settimeout(30)
s.sendall(req.encode())
out = b""
while True:
b = s.recv(65536)
if not b:
break
out += b
s.close()
sys.stdout.write(out.decode("latin-1"))' "$1" "$2" "$3"
}
get() { request "$1" "$2" ""; }
post() { request "$1" "" "$2"; }
url=$(start)
port=$(portof "${url}")
srv=$(srvpid)
test -n "${srv}" || fail "htsserver did not report its pid"
# Every request body is gated by the session id (78_webhttrack-sid.test).
sid=$(get "${port}" /server/index.html |
sed -n 's/.*name="sid" value="\([0-9a-f]*\)".*/\1/p' | head -1)
test "${#sid}" -eq 32 || fail "did not scrape a 32-hex sid (got '${sid}')"
# A file the mirror must never expose, next to the project that may.
echo "SECRETMARKER" >"${base}/secret.txt"
mkdir -p "${base}/proj"
echo "LOGMARKER" >"${base}/proj/hts-log.txt"
# error_redirect is the branch that skipped the fsfile clearing, so fail the save
# with a component over NAME_MAX (mkdir refuses it whatever the uid). Must precede
# any successful save: commandEnd then swaps the error page for the finished one.
get "${port}" /server/style.css >/dev/null # leaves a path behind in fsfile
post "${port}" "sid=${sid}&command=httrack&command_do=save&winprofile=x&path=${base}/$(printf '%0300d' 0)&projname=p" |
grep -q '^Location: /server/error.html' ||
fail "the refused save did not redirect to the error page"
# No project yet, so no root to serve from.
post "${port}" "sid=${sid}&projpath=${base}/" >/dev/null
get "${port}" /website/secret.txt | grep -q SECRETMARKER &&
fail "a posted projpath served a file outside any project"
# Positive control: step4's "save settings" flow registers the project without
# crawling, and browsing its mirror keeps working. Without it the assertions
# around it would pass on a server that never serves /website/ at all.
body="sid=${sid}&command=httrack&command_do=save&winprofile=x"
body="${body}&path=${base}&projname=proj&projpath=${base}/proj/"
post "${port}" "${body}" >/dev/null
test -f "${base}/proj/hts-cache/winprofile.ini" ||
fail "the project was not registered: $(cat "${log}")"
get "${port}" /website/hts-log.txt | grep -q LOGMARKER ||
fail "the registered project's mirror is not served"
# Same request with the root repointed: the project is legitimate, projpath is
# not what decides where the bytes come from.
post "${port}" "sid=${sid}&projpath=${base}/" >/dev/null
get "${port}" /website/secret.txt | grep -q SECRETMARKER &&
fail "a posted projpath repointed the served root"
post "${port}" "sid=${sid}&projpath=/etc/" >/dev/null
get "${port}" /website/passwd | grep -q '^root:' &&
fail "a posted projpath read an arbitrary system file"
# A ".." anywhere in the recorded root would escape the mirror on every later
# request, so the save must be refused and the previous root kept.
post "${port}" "sid=${sid}&command=httrack&command_do=save&winprofile=x&path=${base}/proj&projname=.." >/dev/null
get "${port}" /website/secret.txt | grep -q SECRETMARKER &&
fail "a '..' in the saved project path escaped the mirror root"
get "${port}" /website/hts-log.txt | grep -q LOGMARKER ||
fail "rejecting the '..' root also lost the previous one"
# The root is now the server's own, but it is still built from two posted
# fields: 800-odd bytes of them used to be sprintf'd into a 1024-byte buffer.
seg=$(printf '%0200d' 0)
longpath="${base}/${seg}/${seg}/${seg}/${seg}"
fspath="${longpath}/proj"
# structcheck() refuses a root over HTS_URLMAXSIZE, so the URL carries the rest.
test "$((${#fspath} + 11))" -le 1024 ||
fail "the long project path (${#fspath}) would not pass structcheck"
longurl=$(printf '%0800d' 0)
test "$((${#fspath} + 1 + ${#longurl}))" -gt 1024 ||
fail "the composed path (${#fspath} + ${#longurl}) would not overflow"
post "${port}" "sid=${sid}&command=httrack&command_do=save&winprofile=x&path=${longpath}&projname=proj" >/dev/null
test -f "${fspath}/hts-cache/winprofile.ini" ||
fail "the long-path project was not registered: $(cat "${log}")"
get "${port}" "/website/${longurl}" >/dev/null 2>&1 || true
alive "${srv}" || fail "an over-long project path crashed the server: $(cat "${log}")"
get "${port}" /server/index.html | grep -q '200 OK' ||
fail "the server stopped answering after the over-long project path"
echo "PASS"

View File

@@ -1,111 +0,0 @@
#!/bin/bash
# A cached header longer than proxytrack's field must be clipped to that field,
# not overflow into the next one (contenttype[64] abuts the location pointer).
set -euo pipefail
testdir=$(cd "$(dirname "$0")" && pwd)
# shellcheck source=tests/testlib.sh
. "${testdir}/testlib.sh"
dir=$(mktemp -d)
trap 'rm -rf "$dir"' EXIT
pad() { printf "%${1}s" '' | tr ' ' "$2"; }
# Longest surviving run of char $2 in file $1, or 0.
runlen() {
grep -ao "$2\\+" "$1" | awk '{ print length($0) }' | sort -rn | head -n1 || true
}
# --- ARC reader (HTTP_READFIELD_STRING), no python needed -------------------
# Each header overshoots its own destination, so a per-field clip is the only
# way to get every expected length right at once.
{
printf 'HTTP/1.1 200 OK\r\n'
printf 'Content-Type: text/html%s\r\n' "$(pad 400 A)"
printf 'Etag: %s\r\n' "$(pad 400 E)"
printf 'Content-Disposition: %s\r\n' "$(pad 400 D)"
printf 'Last-Modified: Wed, 01 Jan 2025 00:00:00 GMT\r\n'
printf 'Content-Length: 5\r\n\r\n'
} >"$dir/hdr"
printf 'hello' >"$dir/body"
alen=$(($(wc -c <"$dir/hdr") + $(wc -c <"$dir/body")))
{
printf 'filedesc://t.arc 0.0.0.0 20250101000000 text/plain 200 - - 0 t.arc 9\n'
printf '2 0 test\n'
printf '\n\n'
printf 'http://example.com/p.html 0.0.0.0 20250101000000 text/html 200 - - 0 t.arc %d\n' "$alen"
cat "$dir/hdr" "$dir/body"
} >"$dir/in.arc"
proxytrack --convert "$dir/out.arc" "$dir/in.arc" >/dev/null 2>&1 || {
echo "FAIL: proxytrack crashed reading an ARC with over-long headers" >&2
exit 1
}
test -s "$dir/out.arc" || {
echo "FAIL: ARC entry dropped instead of clipped" >&2
exit 1
}
# Only contenttype survives into the ARC output; the over-long Etag and
# Content-Disposition above are still load-bearing, since bounding just one
# field leaves the others smashing the element (the run above would crash).
got=$(runlen "$dir/out.arc" A)
test "${got:-0}" = 54 || {
echo "FAIL: ARC content-type clipped to ${got:-0}, expected 54" >&2
exit 1
}
# --- ZIP reader (ZIP_READFIELD_STRING) --------------------------------------
python=$(find_python) || {
echo "python3 missing; ARC half only" >&2
exit 0
}
"$python" - "$dir/in.zip" <<'EOF'
import sys, zipfile
body = b"hello world"
meta = (
"X-In-Cache: 1\r\n"
"X-StatusCode: 200\r\n"
"X-StatusMessage: ok%s\r\n"
"X-Size: %d\r\n"
"Content-Type: text/html%s\r\n"
"X-Charset: %s\r\n"
"Etag: %s\r\n"
"Content-Disposition: %s\r\n"
# a parseable date is required: the ARC writer dereferences it unchecked
"Last-Modified: Wed, 01 Jan 2025 00:00:00 GMT\r\n"
"X-Save: out/page.html\r\n"
% ("M" * 400, len(body), "A" * 400, "C" * 400, "E" * 400, "D" * 400)
).encode()
zi = zipfile.ZipInfo("example.com/page.html")
zi.compress_type = zipfile.ZIP_STORED
zi.extra = meta
with zipfile.ZipFile(sys.argv[1], "w") as z:
z.writestr(zi, body)
EOF
proxytrack --convert "$dir/out2.arc" "$dir/in.zip" >/dev/null 2>&1 || {
echo "FAIL: proxytrack crashed reading a zip cache with over-long headers" >&2
exit 1
}
test -s "$dir/out2.arc" || {
echo "FAIL: zip entry dropped instead of clipped" >&2
exit 1
}
# three destinations, two capacities: msg[1024] keeps all 400 M's where a
# one-size-fits-all clip would cut it to 63 like the others
while read -r ch want; do
got=$(runlen "$dir/out2.arc" "$ch")
test "${got:-0}" = "$want" || {
echo "FAIL: zip field '$ch' clipped to ${got:-0}, expected $want" >&2
exit 1
}
done <<'EOF'
A 54
C 63
M 400
EOF

Some files were not shown because too many files have changed in this diff Show More