mirror of
https://github.com/xroche/httrack.git
synced 2026-07-23 17:19:17 +03:00
Compare commits
17 Commits
fix/footer
...
master
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6223739cba | ||
|
|
301f5c2f2f | ||
|
|
dd321171b4 | ||
|
|
2f158c05d0 | ||
|
|
6c74d94802 | ||
|
|
00b0f5728c | ||
|
|
93314bcef9 | ||
|
|
d3d3bce8af | ||
|
|
b3e51d753b | ||
|
|
5099efc1cf | ||
|
|
e1d4c35ee6 | ||
|
|
69c562bf0c | ||
|
|
d83ed3fdee | ||
|
|
ce278a4184 | ||
|
|
f6f46e88b2 | ||
|
|
1e0c009273 | ||
|
|
f894a64ff8 |
@@ -473,6 +473,15 @@ then:</small><br>
|
||||
lifts the built-in caps; use it only against infrastructure you are allowed to
|
||||
load.</small></p>
|
||||
|
||||
<h4>Save a WARC archive of the crawl</h4>
|
||||
<p><tt>httrack https://example.com/ --warc --path mydir</tt><br>
|
||||
<small>Writes a standard WARC/1.1 file (<tt>httrack-<timestamp>.warc.gz</tt>) in
|
||||
the project folder alongside the browsable mirror, not instead of it. Set the name
|
||||
with <tt>--warc-file NAME</tt> and split a large crawl with <tt>--warc-max-size N</tt>;
|
||||
add <tt>--warc-cdx</tt> for a sorted CDXJ index, or <tt>--wacz</tt> to bundle the
|
||||
archive, index and pages into one WACZ for replay tools such as
|
||||
replayweb.page.</small></p>
|
||||
|
||||
<h4>HTTrack as a fetch tool</h4>
|
||||
<p><tt>httrack --get https://host/file.bin --path tmp</tt><br>
|
||||
<small><tt>--get</tt> fetches one file with cache, index, depth, cookies and robots
|
||||
|
||||
@@ -108,16 +108,16 @@ offline browser : copy websites to a local directory</p>
|
||||
--footer</b> ] [ <b>-%l, --language</b> ] [ <b>-%a,
|
||||
--accept</b> ] [ <b>-%X, --headers</b> ] [ <b>-C,
|
||||
--cache[=N]</b> ] [ <b>-k, --store-all-in-cache</b> ] [
|
||||
<b>-%n, --do-not-recatch</b> ] [ <b>-%v, --display</b> ] [
|
||||
<b>-Q, --do-not-log</b> ] [ <b>-q, --quiet</b> ] [ <b>-z,
|
||||
--extra-log</b> ] [ <b>-Z, --debug-log</b> ] [ <b>-v,
|
||||
--verbose</b> ] [ <b>-f, --file-log</b> ] [ <b>-f2,
|
||||
--single-log</b> ] [ <b>-I, --index</b> ] [ <b>-%i,
|
||||
--build-top-index</b> ] [ <b>-%I, --search-index</b> ] [
|
||||
<b>-pN, --priority[=N]</b> ] [ <b>-S, --stay-on-same-dir</b>
|
||||
] [ <b>-D, --can-go-down</b> ] [ <b>-U, --can-go-up</b> ] [
|
||||
<b>-B, --can-go-up-and-down</b> ] [ <b>-a,
|
||||
--stay-on-same-address</b> ] [ <b>-d,
|
||||
<b>-%r, --warc</b> ] [ <b>-%n, --do-not-recatch</b> ] [
|
||||
<b>-%v, --display</b> ] [ <b>-Q, --do-not-log</b> ] [ <b>-q,
|
||||
--quiet</b> ] [ <b>-z, --extra-log</b> ] [ <b>-Z,
|
||||
--debug-log</b> ] [ <b>-v, --verbose</b> ] [ <b>-f,
|
||||
--file-log</b> ] [ <b>-f2, --single-log</b> ] [ <b>-I,
|
||||
--index</b> ] [ <b>-%i, --build-top-index</b> ] [ <b>-%I,
|
||||
--search-index</b> ] [ <b>-pN, --priority[=N]</b> ] [ <b>-S,
|
||||
--stay-on-same-dir</b> ] [ <b>-D, --can-go-down</b> ] [
|
||||
<b>-U, --can-go-up</b> ] [ <b>-B, --can-go-up-and-down</b> ]
|
||||
[ <b>-a, --stay-on-same-address</b> ] [ <b>-d,
|
||||
--stay-on-same-domain</b> ] [ <b>-l, --stay-on-same-tld</b>
|
||||
] [ <b>-e, --go-everywhere</b> ] [ <b>-%H,
|
||||
--debug-headers</b> ] [ <b>-%!,
|
||||
@@ -1129,6 +1129,20 @@ update before) (--cache[=N])</p></td></tr>
|
||||
<td width="4%">
|
||||
|
||||
|
||||
<p>-%r</p></td>
|
||||
<td width="5%"></td>
|
||||
<td width="82%">
|
||||
|
||||
|
||||
<p>write an ISO-28500 WARC/1.1 archive; --warc-file NAME
|
||||
sets the output name, --warc-max-size N rotates segments
|
||||
past N bytes, --warc-cdx also writes a sorted CDXJ index,
|
||||
--wacz packages it all as a WACZ file (--warc)</p></td></tr>
|
||||
<tr valign="top" align="left">
|
||||
<td width="9%"></td>
|
||||
<td width="4%">
|
||||
|
||||
|
||||
<p>-%n</p></td>
|
||||
<td width="5%"></td>
|
||||
<td width="82%">
|
||||
|
||||
@@ -103,6 +103,17 @@ ${do:end-if}
|
||||
> ${LANG_I61}
|
||||
<br><br>
|
||||
|
||||
<input type="checkbox" name="warc" ${checked:warc}
|
||||
title='${html:LANG_WARCTIP}' onMouseOver="info('${html:LANG_WARCTIP}'); return true" onMouseOut="info(' '); return true"
|
||||
> ${LANG_WARC}
|
||||
<br><br>
|
||||
|
||||
${LANG_WARCFILE}
|
||||
<input name="warcfile" value="${warcfile}" size="40"
|
||||
title='${html:LANG_WARCFILETIP}' onMouseOver="info('${html:LANG_WARCFILETIP}'); return true" onMouseOut="info(' '); return true"
|
||||
>
|
||||
<br><br>
|
||||
|
||||
<input type="checkbox" name="norecatch" ${checked:norecatch}
|
||||
title='${html:LANG_I5b}' onMouseOver="info('${html:LANG_I5b}'); return true" onMouseOut="info(' '); return true"
|
||||
> ${LANG_I34b}
|
||||
|
||||
@@ -141,6 +141,8 @@ ${do:copy:KeepSlashes:keepslashes}
|
||||
${do:copy:KeepQueryOrder:keepqueryorder}
|
||||
${do:copy:StripQuery:stripquery}
|
||||
${do:copy:StoreAllInCache:cache2}
|
||||
${do:copy:Warc:warc}
|
||||
${do:copy:WarcFile:warcfile}
|
||||
${do:copy:LogType:logtype}
|
||||
${do:copy:UseHTTPProxyForFTP:ftpprox}
|
||||
${do:copy:ProxyType:proxytype}
|
||||
|
||||
@@ -187,6 +187,8 @@ ${do:end-if}
|
||||
${test:toler:--tolerant}
|
||||
${test:http10:--http-10}
|
||||
${test:cache2:--store-all-in-cache}
|
||||
${test:warc:--warc}
|
||||
${test:warcfile:--warc-file "}${html:warcfile}${test:warcfile:"}
|
||||
${test:norecatch:--do-not-recatch}
|
||||
${test:logf:--single-log}
|
||||
${test:logtype:::--extra-log:--debug-log}
|
||||
@@ -237,6 +239,8 @@ KeepSlashes=${ztest:keepslashes:0:1}
|
||||
KeepQueryOrder=${ztest:keepqueryorder:0:1}
|
||||
StripQuery=${stripquery}
|
||||
StoreAllInCache=${ztest:cache2:0:1}
|
||||
Warc=${ztest:warc:0:1}
|
||||
WarcFile=${warcfile}
|
||||
LogType=${logtype}
|
||||
UseHTTPProxyForFTP=${ztest:ftpprox:0:1}
|
||||
ProxyType=${proxytype}
|
||||
|
||||
8
lang.def
8
lang.def
@@ -1034,3 +1034,11 @@ LANG_STRIPQUERY
|
||||
Strip query keys:
|
||||
LANG_STRIPQUERYTIP
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
LANG_WARC
|
||||
Write a WARC archive of the crawl
|
||||
LANG_WARCTIP
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
LANG_WARCFILE
|
||||
WARC archive name:
|
||||
LANG_WARCFILETIP
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Ïðåìàõâàíå íà êëþ÷îâå îò çàÿâêàòà:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Êëþ÷îâå îò çàÿâêàòà, ðàçäåëåíè ñúñ çàïåòàÿ, êîèòî äà ñå ïðåìàõíàò îò èìåòî íà çàïèñàíèÿ ôàéë (íàïðèìåð sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Çàïèñâàíå íà WARC àðõèâ íà îáõîæäàíåòî
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Çàïèñâàíå íà âñåêè èçòåãëåí îòãîâîð è â WARC/1.1 àðõèâ ïî ISO-28500, äî îãëåäàëîòî.
|
||||
WARC archive name:
|
||||
Èìå íà WARC àðõèâà:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Íåçàäúëæèòåëíî áàçîâî èìå çà WARC àðõèâà; îñòàâåòå ïðàçíî çà àâòîìàòè÷íî èìåíóâàíå â èçõîäíàòà äèðåêòîðèÿ.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Eliminar claves de query string:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Claves de query string, separadas por comas, que se eliminarán del nombre de los archivos guardados (p. ej. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Escribir un archivo WARC del rastreo
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Guardar también cada respuesta descargada en un archivo WARC/1.1 ISO-28500, junto a la réplica.
|
||||
WARC archive name:
|
||||
Nombre del archivo WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nombre base opcional para el archivo WARC; déjelo en blanco para nombrarlo automáticamente en el directorio de salida.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Odebrat klíèe dotazu:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Klíèe dotazu oddìlené èárkami, které se vynechají z pojmenování ukládaných souborù (napø. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Zapsat archiv WARC z procházení
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Uložit také každou staženou odpovìï do archivu WARC/1.1 podle ISO-28500 vedle zrcadla.
|
||||
WARC archive name:
|
||||
Název archivu WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Volitelný základní název archivu WARC; ponechte prázdné pro automatické pojmenování ve výstupním adresáøi.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
移除查詢鍵:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
以逗號分隔的查詢鍵,將其從儲存檔案的命名中移除 (例如 sid,utm_source)。
|
||||
Write a WARC archive of the crawl
|
||||
寫入此次抓取的 WARC 封存檔
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
同時將每個已擷取的回應儲存為 ISO-28500 WARC/1.1 封存檔,置於鏡像網站旁。
|
||||
WARC archive name:
|
||||
WARC 封存檔名稱:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC 封存檔的選用基本名稱;留空則於輸出目錄中自動命名。
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
剥离查询键:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
用逗号分隔的查询键,将其从保存文件的命名中删除 (例如 sid,utm_source)。
|
||||
Write a WARC archive of the crawl
|
||||
写入本次抓取的 WARC 归档
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
同时将每个已获取的响应保存为 ISO-28500 WARC/1.1 归档,置于镜像站点旁边。
|
||||
WARC archive name:
|
||||
WARC 归档名称:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC 归档的可选基本名称;留空则在输出目录中自动命名。
|
||||
|
||||
@@ -958,3 +958,11 @@ Strip query keys:
|
||||
Ukloniti kljuèeve upita:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Zarezom odvojeni kljuèevi upita koji se izostavljaju iz naziva spremljenih datoteka (npr. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Zapi¹i WARC arhivu obilaska
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Spremi i svaki preuzeti odgovor u ISO-28500 WARC/1.1 arhivu, uz zrcalo.
|
||||
WARC archive name:
|
||||
Naziv WARC arhive:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Neobavezni osnovni naziv WARC arhive; ostavite prazno za automatsko imenovanje u izlaznom direktoriju.
|
||||
|
||||
@@ -1004,3 +1004,11 @@ Strip query keys:
|
||||
Fjern forespørgselsnøgler:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kommaseparerede forespørgselsnøgler, der udelades i navngivningen af gemte filer (f.eks. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Skriv et WARC-arkiv af gennemsøgningen
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Gem også hvert hentet svar i et ISO-28500 WARC/1.1-arkiv ved siden af spejlet.
|
||||
WARC archive name:
|
||||
Navn på WARC-arkiv:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Valgfrit basisnavn til WARC-arkivet; lad feltet stå tomt for automatisk navngivning i outputmappen.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Query-Schlüssel entfernen:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kommagetrennte Query-Schlüssel, die bei der Benennung gespeicherter Dateien entfallen (z. B. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
WARC-Archiv des Crawls schreiben
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Jede heruntergeladene Antwort zusätzlich in einem ISO-28500-WARC/1.1-Archiv neben dem Spiegel speichern.
|
||||
WARC archive name:
|
||||
Name des WARC-Archivs:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Optionaler Basisname für das WARC-Archiv; leer lassen, um es automatisch im Ausgabeverzeichnis zu benennen.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Eemalda päringuvõtmed:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Komadega eraldatud päringuvõtmed, mis jäetakse salvestatud faili nimest välja (nt sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Kirjuta läbimise WARC-arhiiv
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Salvesta iga alla laaditud vastus ka ISO-28500 WARC/1.1 arhiivi peegli kõrvale.
|
||||
WARC archive name:
|
||||
WARC-arhiivi nimi:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC-arhiivi valikuline põhinimi; jäta tühjaks, et see väljundkataloogis automaatselt nimetada.
|
||||
|
||||
@@ -1004,3 +1004,11 @@ Strip query keys:
|
||||
Strip query keys:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Write a WARC archive of the crawl
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
WARC archive name:
|
||||
WARC archive name:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
|
||||
@@ -958,3 +958,11 @@ Strip query keys:
|
||||
Poista kyselyavaimet:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Pilkuin erotellut kyselyavaimet, jotka jätetään pois tallennettujen tiedostojen nimeämisestä (esim. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Kirjoita imuroinnin WARC-arkisto
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Tallenna myös jokainen noudettu vastaus ISO-28500 WARC/1.1 -arkistoon peilin viereen.
|
||||
WARC archive name:
|
||||
WARC-arkiston nimi:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Valinnainen WARC-arkiston perusnimi; jätä tyhjäksi, jotta se nimetään automaattisesti tulostehakemistoon.
|
||||
|
||||
@@ -1004,3 +1004,11 @@ Strip query keys:
|
||||
Supprimer les clés de query string :
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Clés de query string à retirer du nommage des fichiers enregistrés, séparées par des virgules (par ex. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Écrire une archive WARC du crawl
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Enregistrer aussi chaque réponse téléchargée dans une archive WARC/1.1 (ISO-28500), à côté du miroir.
|
||||
WARC archive name:
|
||||
Nom de l'archive WARC :
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nom de base optionnel pour l'archive WARC ; laissez vide pour le générer automatiquement dans le répertoire de sortie.
|
||||
|
||||
@@ -958,3 +958,11 @@ Strip query keys:
|
||||
Αφαίρεση κλειδιών ερωτήματος:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Κλειδιά ερωτήματος χωρισμένα με κόμμα, που θα αφαιρεθούν από την ονομασία των αποθηκευμένων αρχείων (π.χ. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Εγγραφή αρχείου WARC της ανίχνευσης
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Αποθήκευση κάθε ληφθείσας απόκρισης και σε αρχείο WARC/1.1 ISO-28500, δίπλα στο είδωλο.
|
||||
WARC archive name:
|
||||
Όνομα αρχείου WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Προαιρετικό βασικό όνομα για το αρχείο WARC. Αφήστε το κενό για αυτόματη ονομασία στον κατάλογο εξόδου.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Rimuovi chiavi della query string:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Chiavi della query string, separate da virgole, da rimuovere dai nomi dei file salvati (ad es. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Scrivi un archivio WARC della scansione
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Salva anche ogni risposta scaricata in un archivio WARC/1.1 ISO-28500, accanto al mirror.
|
||||
WARC archive name:
|
||||
Nome dell'archivio WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nome di base facoltativo per l'archivio WARC; lascia vuoto per assegnarlo automaticamente nella directory di output.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
削除するクエリキー:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
保存ファイル名の生成から除外するクエリキーをカンマ区切りで指定します (例: sid,utm_source)。
|
||||
Write a WARC archive of the crawl
|
||||
クロールの WARC アーカイブを書き出す
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
取得した各レスポンスを ISO-28500 WARC/1.1 アーカイブとしてミラーの隣にも保存します。
|
||||
WARC archive name:
|
||||
WARC アーカイブ名:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC アーカイブの任意のベース名。空欄にすると出力ディレクトリ内で自動的に名前が付けられます。
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
¾âáâàÐÝØ ÚÛãçÕÒØ ÞÔ ÑÐàÐúÕâÞ:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
ºÛãçÕÒØ ÞÔ ÑÐàÐúÕâÞ, ÞÔÔÕÛÕÝØ áÞ ×ÐߨàÚÐ, èâÞ áÕ ÞâáâàÐÝãÒÐÐâ ÞÔ ØÜÕâÞ ÝÐ ×ÐçãÒÐÝÐâÐ ÔÐâÞâÕÚÐ (ÝÐ ßàØÜÕà sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
·ÐßØèØ WARC ÐàåØÒÐ ÝÐ ßàÕÑÐàãÒÐúÕâÞ
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
·ÐçãÒÐø ÓÞ áÕÚÞø ßàÕ×ÕÜÕÝ ÞÔÓÞÒÞà Ø ÒÞ ISO-28500 WARC/1.1 ÐàåØÒÐ, ßÞÚàÐø ÞÓÛÕÔÐÛÞâÞ.
|
||||
WARC archive name:
|
||||
¸ÜÕ ÝÐ WARC ÐàåØÒÐâÐ:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
¸×ÑÞàÝÞ ÞáÝÞÒÝÞ ØÜÕ ×Ð WARC ÐàåØÒÐâÐ; ÞáâÐÒÕâÕ ßàÐ×ÝÞ ×Ð ÐÒâÞÜÐâáÚÞ ØÜÕÝãÒÐúÕ ÒÞ Ø×ÛÕ×ÝØÞâ ÔØàÕÚâÞàØãÜ.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Lekérdezési kulcsok eltávolítása:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Vesszõvel elválasztott lekérdezési kulcsok, amelyeket el kell hagyni a mentett fájl elnevezésébõl (pl. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
A bejárás WARC archívumának írása
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Minden letöltött válasz mentése ISO-28500 WARC/1.1 archívumba is, a tükör mellé.
|
||||
WARC archive name:
|
||||
WARC archívum neve:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
A WARC archívum opcionális alapneve; hagyja üresen az automatikus elnevezéshez a kimeneti könyvtárban.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Query-sleutels verwijderen:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Door komma's gescheiden query-sleutels die bij het benoemen van opgeslagen bestanden worden weggelaten (bijv. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Schrijf een WARC-archief van de crawl
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Sla ook elke opgehaalde respons op in een ISO-28500 WARC/1.1-archief, naast de mirror.
|
||||
WARC archive name:
|
||||
Naam van WARC-archief:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Optionele basisnaam voor het WARC-archief; laat leeg om het automatisch een naam te geven in de uitvoermap.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Fjern spørrenøkler:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kommaseparerte spørrenøkler som utelates i navngivingen av lagrede filer (f.eks. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Skriv et WARC-arkiv av gjennomgangen
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Lagre også hvert nedlastet svar i et ISO-28500 WARC/1.1-arkiv ved siden av speilet.
|
||||
WARC archive name:
|
||||
Navn på WARC-arkiv:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Valgfritt basisnavn for WARC-arkivet; la feltet stå tomt for automatisk navngivning i utdatamappen.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Usuñ klucze zapytania:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Rozdzielone przecinkami klucze zapytania pomijane przy nazywaniu zapisanych plików (np. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Zapisz archiwum WARC z indeksowania
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Zapisz te¿ ka¿d± pobran± odpowied¼ do archiwum WARC/1.1 ISO-28500, obok kopii lustrzanej.
|
||||
WARC archive name:
|
||||
Nazwa archiwum WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Opcjonalna nazwa bazowa archiwum WARC; pozostaw puste, aby nazwaæ je automatycznie w katalogu wyj¶ciowym.
|
||||
|
||||
@@ -1004,3 +1004,11 @@ Strip query keys:
|
||||
Remover chaves da query string:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Chaves da query string, separadas por vírgulas, a serem removidas da nomeação dos arquivos salvos (ex.: sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Gravar um arquivo WARC do rastreamento
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Salvar também cada resposta baixada em um arquivo WARC/1.1 ISO-28500, ao lado do espelho.
|
||||
WARC archive name:
|
||||
Nome do arquivo WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nome base opcional para o arquivo WARC; deixe em branco para nomeá-lo automaticamente no diretório de saída.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Remover chaves da query string:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Chaves da query string, separadas por vírgulas, a remover da nomeação dos ficheiros guardados (por ex. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Escrever um arquivo WARC do rastreio
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Guardar também cada resposta transferida num arquivo WARC/1.1 ISO-28500, ao lado do espelho.
|
||||
WARC archive name:
|
||||
Nome do arquivo WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nome base opcional para o arquivo WARC; deixe em branco para o nomear automaticamente no diretório de saída.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Elimina cheile din query string:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Chei din query string, separate prin virgula, de eliminat din denumirea fisierelor salvate (de ex. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Scrie o arhiva WARC a parcurgerii
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Salveaza si fiecare raspuns descarcat intr-o arhiva WARC/1.1 ISO-28500, langa oglinda.
|
||||
WARC archive name:
|
||||
Numele arhivei WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nume de baza optional pentru arhiva WARC; lasati gol pentru a-l denumi automat in directorul de iesire.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Óäàëÿòü êëþ÷è çàïðîñà:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Êëþ÷è çàïðîñà ÷åðåç çàïÿòóþ, óäàëÿåìûå èç èìåíè ñîõðàíÿåìîãî ôàéëà (íàïðèìåð, sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Çàïèñàòü WARC-àðõèâ îáõîäà
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Ñîõðàíÿòü êàæäûé çàãðóæåííûé îòâåò òàêæå â àðõèâ WARC/1.1 ISO-28500 ðÿäîì ñ çåðêàëîì.
|
||||
WARC archive name:
|
||||
Èìÿ WARC-àðõèâà:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Íåîáÿçàòåëüíîå áàçîâîå èìÿ WARC-àðõèâà; îñòàâüòå ïóñòûì äëÿ àâòîìàòè÷åñêîãî èìåíîâàíèÿ â âûõîäíîì êàòàëîãå.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Odstráni» kµúèe dotazu:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kµúèe dotazu oddelené èiarkami, ktoré sa vynechajú z pomenovania ukladaných súborov (napr. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Zapísa» archív WARC z prehµadávania
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Ulo¾i» aj ka¾dú stiahnutú odpoveï do archívu WARC/1.1 ISO-28500 vedµa zrkadla.
|
||||
WARC archive name:
|
||||
Názov archívu WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Voliteµný základný názov archívu WARC; ponechajte prázdne pre automatické pomenovanie vo výstupnom adresári.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Odstrani kljuce poizvedbe:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Z vejicami loceni kljuci poizvedbe, ki se izpustijo pri poimenovanju shranjenih datotek (npr. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Zapisi arhiv WARC iz pregledovanja
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Shrani tudi vsak preneseni odgovor v arhiv WARC/1.1 ISO-28500 poleg zrcala.
|
||||
WARC archive name:
|
||||
Ime arhiva WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Neobvezno osnovno ime arhiva WARC; pustite prazno za samodejno poimenovanje v izhodni mapi.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Ta bort frågenycklar:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kommaseparerade frågenycklar som utelämnas vid namngivningen av sparade filer (t.ex. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Skriv ett WARC-arkiv av genomsökningen
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Spara även varje hämtat svar i ett ISO-28500 WARC/1.1-arkiv, bredvid spegeln.
|
||||
WARC archive name:
|
||||
WARC-arkivets namn:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Valfritt basnamn för WARC-arkivet; lämna tomt för att namnge det automatiskt i utdatakatalogen.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Sorgu anahtarlarýný çýkar:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kaydedilen dosya adlandýrmasýndan çýkarýlacak, virgülle ayrýlmýþ sorgu anahtarlarý (örn. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Taramanýn WARC arþivini yaz
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Ýndirilen her yanýtý ayrýca aynanýn yanýna bir ISO-28500 WARC/1.1 arþivine kaydet.
|
||||
WARC archive name:
|
||||
WARC arþivi adý:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC arþivi için isteðe baðlý temel ad; çýktý dizininde otomatik adlandýrma için boþ býrakýn.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Âèëó÷àòè êëþ÷³ çàïèòó:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Êëþ÷³ çàïèòó ÷åðåç êîìó, ÿê³ âèëó÷àþòüñÿ ç ³ìåí³ çáåðåæåíîãî ôàéëó (íàïðèêëàä, sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Çàïèñàòè WARC-àðõ³â îáõîäó
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Çáåð³ãàòè êîæíó çàâàíòàæåíó â³äïîâ³äü òàêîæ ó àðõ³â WARC/1.1 ISO-28500 ïîðÿä ³ç äçåðêàëîì.
|
||||
WARC archive name:
|
||||
²ì'ÿ WARC-àðõ³âó:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Íåîáîâ'ÿçêîâà áàçîâà íàçâà WARC-àðõ³âó; çàëèøòå ïîðîæí³ì äëÿ àâòîìàòè÷íîãî íàéìåíóâàííÿ ó âèõ³äíîìó êàòàëîç³.
|
||||
|
||||
@@ -956,3 +956,11 @@ Strip query keys:
|
||||
Olib tashlanadigan so’rov kalitlari:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Saqlangan fayl nomidan olib tashlanadigan, vergul bilan ajratilgan so’rov kalitlari (masalan, sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Qidiruvning WARC arxivini yozish
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Har bir yuklab olingan javobni ISO-28500 WARC/1.1 arxiviga ham, ko'zgu yonida saqlash.
|
||||
WARC archive name:
|
||||
WARC arxivi nomi:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC arxivi uchun ixtiyoriy asosiy nom; chiqish katalogida avtomatik nomlash uchun bo'sh qoldiring.
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
.\"
|
||||
.\" This file is generated by man/makeman.sh; do not edit by hand.
|
||||
.\" SPDX-License-Identifier: GPL-3.0-or-later
|
||||
.TH httrack 1 "22 July 2026" "httrack website copier"
|
||||
.TH httrack 1 "23 July 2026" "httrack website copier"
|
||||
.SH NAME
|
||||
httrack \- offline browser : copy websites to a local directory
|
||||
.SH SYNOPSIS
|
||||
@@ -74,6 +74,7 @@ httrack \- offline browser : copy websites to a local directory
|
||||
[ \fB\-%X, \-\-headers\fR ]
|
||||
[ \fB\-C, \-\-cache[=N]\fR ]
|
||||
[ \fB\-k, \-\-store\-all\-in\-cache\fR ]
|
||||
[ \fB\-%r, \-\-warc\fR ]
|
||||
[ \fB\-%n, \-\-do\-not\-recatch\fR ]
|
||||
[ \fB\-%v, \-\-display\fR ]
|
||||
[ \fB\-Q, \-\-do\-not\-log\fR ]
|
||||
@@ -276,6 +277,8 @@ additional HTTP header line (\-%X "X\-Magic: 42" (\-\-headers <param>)
|
||||
create/use a cache for updates and retries (C0 no cache,C1 cache is prioritary,* C2 test update before) (\-\-cache[=N])
|
||||
.IP \-k
|
||||
store all files in cache (not useful if files on disk) (\-\-store\-all\-in\-cache)
|
||||
.IP \-%r
|
||||
write an ISO\-28500 WARC/1.1 archive; \-\-warc\-file NAME sets the output name, \-\-warc\-max\-size N rotates segments past N bytes, \-\-warc\-cdx also writes a sorted CDXJ index, \-\-wacz packages it all as a WACZ file (\-\-warc)
|
||||
.IP \-%n
|
||||
do not re\-download locally erased files (\-\-do\-not\-recatch)
|
||||
.IP \-%v
|
||||
|
||||
@@ -63,7 +63,7 @@ libhttrack_la_SOURCES = htscore.c htsparse.c htsback.c htscache.c \
|
||||
htshelp.c htslib.c htsurlport.c htscoremain.c \
|
||||
htsname.c htsrobots.c htstools.c htswizard.c \
|
||||
htsalias.c htsthread.c htsindex.c htsbauth.c \
|
||||
htsmd5.c htscodec.c htsproxy.c htszlib.c htswrap.c htsconcat.c \
|
||||
htsmd5.c htscodec.c htswarc.c htsproxy.c htszlib.c htswrap.c htsconcat.c \
|
||||
htsmodules.c htscharset.c punycode.c htsencoding.c htssniff.c \
|
||||
md5.c \
|
||||
minizip/ioapi.c minizip/mztools.c minizip/unzip.c minizip/zip.c \
|
||||
@@ -74,7 +74,7 @@ libhttrack_la_SOURCES = htscore.c htsparse.c htsback.c htscache.c \
|
||||
htshelp.h htsindex.h htslib.h htsurlport.h htsmd5.h \
|
||||
htsmodules.h htsname.h htsnet.h htssniff.h \
|
||||
htsopt.h htsrobots.h htsthread.h \
|
||||
htstools.h htswizard.h htswrap.h htscodec.h htsproxy.h htszlib.h \
|
||||
htstools.h htswizard.h htswrap.h htscodec.h htswarc.h htsproxy.h htszlib.h \
|
||||
htsstrings.h htsarrays.h httrack-library.h \
|
||||
htscharset.h punycode.h htsencoding.h \
|
||||
htsentities.h htsentities.sh htsbasiccharsets.sh htscodepages.h \
|
||||
|
||||
@@ -114,6 +114,15 @@ const char *hts_optalias[][4] = {
|
||||
"strip [host/pattern=]key1,key2,... from URLs"},
|
||||
{"cookies-file", "-%K", "param1",
|
||||
"load extra cookies from a Netscape cookies.txt"},
|
||||
{"warc", "-%r", "single", "write an ISO-28500 WARC/1.1 archive of the crawl"},
|
||||
{"warc-file", "-%rf", "param1", "write a WARC archive to the given base name"},
|
||||
{"warc-max-size", "-%rs", "param1",
|
||||
"rotate the WARC archive once a segment passes N bytes (0: single file)"},
|
||||
{"warc-cdx", "-%rc", "single",
|
||||
"write a sorted CDXJ index next to the WARC archive"},
|
||||
{"warc-cdxj", "-%rc", "single", ""},
|
||||
{"wacz", "-%rz", "single",
|
||||
"package the WARC archive, CDXJ index and pages as a WACZ file"},
|
||||
{"why", "-%Y", "param1",
|
||||
"explain which filter rule accepts or rejects a URL, then exit"},
|
||||
{"pause", "-%G", "param1",
|
||||
|
||||
@@ -37,6 +37,7 @@ Please visit our Website: http://www.httrack.com
|
||||
/* specific definitions */
|
||||
#include "htsnet.h"
|
||||
#include "htscore.h"
|
||||
#include "htswarc.h"
|
||||
#include "htsthread.h"
|
||||
#include <time.h>
|
||||
/* END specific definitions */
|
||||
@@ -749,9 +750,20 @@ int back_finalize(httrackp * opt, cache_back * cache, struct_back * sback,
|
||||
fexist_utf8(back[p].url_sav))
|
||||
filenote(&opt->state.strc, back[p].url_sav, NULL);
|
||||
}
|
||||
/* Keep the compressed spool so the WARC record stores the body
|
||||
verbatim (Content-Encoding preserved) instead of unlinking it.
|
||||
*/
|
||||
if (StringNotEmpty(opt->warc_file)) {
|
||||
warc_adopt_rawspool(&back[p].r, back[p].tmpfile);
|
||||
if (back[p].r.warc_rawpath != NULL)
|
||||
back[p].tmpfile =
|
||||
NULL; /* adopted: freed via warc_free_request */
|
||||
}
|
||||
/* ensure that no remaining temporary file exists */
|
||||
unlink(back[p].tmpfile);
|
||||
back[p].tmpfile = NULL;
|
||||
if (back[p].tmpfile != NULL) {
|
||||
unlink(back[p].tmpfile);
|
||||
back[p].tmpfile = NULL;
|
||||
}
|
||||
}
|
||||
// stats
|
||||
HTS_STAT.total_packed += back[p].compressed_size;
|
||||
@@ -979,6 +991,10 @@ int back_finalize(httrackp * opt, cache_back * cache, struct_back * sback,
|
||||
// status finished callback
|
||||
RUN_CALLBACK1(opt, xfrstatus, &back[p]);
|
||||
|
||||
// WARC archive of the transaction (request + response/revisit)
|
||||
if (StringNotEmpty(opt->warc_file))
|
||||
warc_write_backtransaction(opt, &back[p]);
|
||||
|
||||
return 0;
|
||||
} else { // testmode
|
||||
if (back[p].r.statuscode / 100 >= 3) { /* Store 3XX, 4XX, 5XX test response codes, but NOT 2XX */
|
||||
@@ -1055,6 +1071,11 @@ void back_copy_static(const lien_back * src, lien_back * dst) {
|
||||
dst->r.soc = INVALID_SOCKET;
|
||||
dst->r.adr = NULL;
|
||||
dst->r.headers = NULL;
|
||||
dst->r.warc_reqhdr = NULL;
|
||||
dst->r.warc_resphdr = NULL;
|
||||
dst->r.warc_rawpath =
|
||||
NULL; /* the spool stays owned by src (no double-unlink) */
|
||||
dst->r.warc_truncated = 0;
|
||||
dst->r.out = NULL;
|
||||
dst->r.location = dst->location_buffer;
|
||||
dst->r.fp = NULL;
|
||||
@@ -1118,6 +1139,10 @@ int back_unserialize(FILE * fp, lien_back ** dst) {
|
||||
(*dst)->chunk_adr = NULL;
|
||||
(*dst)->r.adr = NULL;
|
||||
(*dst)->r.out = NULL;
|
||||
(*dst)->r.warc_reqhdr = NULL;
|
||||
(*dst)->r.warc_resphdr = NULL;
|
||||
(*dst)->r.warc_rawpath = NULL;
|
||||
(*dst)->r.warc_truncated = 0;
|
||||
(*dst)->r.location = (*dst)->location_buffer;
|
||||
(*dst)->r.fp = NULL;
|
||||
(*dst)->r.soc = INVALID_SOCKET;
|
||||
@@ -1585,6 +1610,7 @@ int back_clear_entry(lien_back * back) {
|
||||
freet(back->r.headers);
|
||||
back->r.headers = NULL;
|
||||
}
|
||||
warc_free_request(&back->r);
|
||||
// Tout nettoyer
|
||||
memset(back, 0, sizeof(lien_back));
|
||||
back->r.soc = INVALID_SOCKET;
|
||||
@@ -2509,6 +2535,19 @@ void back_wait(struct_back * sback, httrackp * opt, cache_back * cache,
|
||||
|
||||
for (i = 0; i < (unsigned int) back_max; i++) {
|
||||
if (back[i].status > 0 && back[i].status < STATUS_FTP_TRANSFER) {
|
||||
/* A cap-truncated body is deliberate, not broken: archive what arrived
|
||||
with WARC-Truncated before the abort overwrites the slot's real 2xx
|
||||
status. HTTrack still treats the slot as incomplete afterwards. */
|
||||
if (StringNotEmpty(opt->warc_file) && back[i].r.statuscode > 0 &&
|
||||
back[i].r.warc_resphdr != NULL && back[i].r.size > 0 &&
|
||||
!(back[i].r.is_write && IS_DELAYED_EXT(back[i].url_sav))) {
|
||||
if (back[i].r.is_write && back[i].r.out != NULL)
|
||||
fflush(back[i].r.out);
|
||||
back[i].r.warc_truncated = (limit == HTS_MIRROR_LIMIT_SIZE)
|
||||
? WARC_TRUNC_LENGTH
|
||||
: WARC_TRUNC_TIME;
|
||||
warc_write_backtransaction(opt, &back[i]);
|
||||
}
|
||||
if (back[i].r.soc != INVALID_SOCKET) {
|
||||
deletehttp(&back[i].r);
|
||||
}
|
||||
@@ -3631,6 +3670,10 @@ void back_wait(struct_back * sback, httrackp * opt, cache_back * cache,
|
||||
deleteaddr(&back[i].r);
|
||||
back[i].r.headers = block;
|
||||
}
|
||||
// Stash the raw response headers for WARC (deletehttp frees
|
||||
// r.headers when the socket closes, before back_finalize)
|
||||
if (StringNotEmpty(opt->warc_file))
|
||||
warc_stash_response(&back[i].r, back[i].r.headers);
|
||||
|
||||
/*
|
||||
Status code and header-response hacks
|
||||
|
||||
@@ -39,6 +39,7 @@ Please visit our Website: http://www.httrack.com
|
||||
|
||||
/* File defs */
|
||||
#include "htscore.h"
|
||||
#include "htswarc.h"
|
||||
|
||||
/* specific definitions */
|
||||
#include "htsbase.h"
|
||||
@@ -698,7 +699,7 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
int primary_len = 8192;
|
||||
|
||||
if (StringNotEmpty(opt->filelist)) {
|
||||
primary_len += max(0, fsize(StringBuff(opt->filelist)) * 2);
|
||||
primary_len += max(0, fsize_utf8(StringBuff(opt->filelist)) * 2);
|
||||
}
|
||||
primary_len += (int) strlen(url1) * 2;
|
||||
|
||||
@@ -855,19 +856,19 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
char *filelist_buff = NULL;
|
||||
size_t filelist_sz = 0;
|
||||
const char *filelist_err = NULL; /* failure reason, NULL on success */
|
||||
const LLint fs = fsize(StringBuff(opt->filelist));
|
||||
const LLint fs = fsize_utf8(StringBuff(opt->filelist));
|
||||
|
||||
if (fs < 0) {
|
||||
/* fsize() hides the cause; redo stat() for a precise errno (#49) */
|
||||
struct stat st;
|
||||
filelist_err = stat(StringBuff(opt->filelist), &st) != 0
|
||||
STRUCT_STAT st;
|
||||
filelist_err = STAT(StringBuff(opt->filelist), &st) != 0
|
||||
? strerror(errno)
|
||||
: "not a regular file";
|
||||
} else if ((filelist_sz = llint_to_size_t(fs)) == (size_t) -1) {
|
||||
filelist_err = "file too large";
|
||||
filelist_sz = 0;
|
||||
} else {
|
||||
FILE *fp = fopen(StringBuff(opt->filelist), "rb");
|
||||
FILE *fp = FOPEN(StringBuff(opt->filelist), "rb");
|
||||
|
||||
if (fp == NULL) {
|
||||
filelist_err = strerror(errno);
|
||||
@@ -975,10 +976,9 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
}
|
||||
// statistiques
|
||||
if (opt->makestat) {
|
||||
makestat_fp =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), "hts-stats.txt"),
|
||||
"wb");
|
||||
makestat_fp = FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-stats.txt"),
|
||||
"wb");
|
||||
if (makestat_fp != NULL) {
|
||||
fprintf(makestat_fp, "HTTrack statistics report, every minutes" LF LF);
|
||||
fflush(makestat_fp);
|
||||
@@ -986,10 +986,9 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
}
|
||||
// tracking -- débuggage
|
||||
if (opt->maketrack) {
|
||||
maketrack_fp =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), "hts-track.txt"),
|
||||
"wb");
|
||||
maketrack_fp = FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-track.txt"),
|
||||
"wb");
|
||||
if (maketrack_fp != NULL) {
|
||||
fprintf(maketrack_fp, "HTTrack tracking report, every minutes" LF LF);
|
||||
fflush(maketrack_fp);
|
||||
@@ -1638,7 +1637,8 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
|
||||
/* Remove file if being processed */
|
||||
if (is_loaded_from_file) {
|
||||
(void) unlink(fconv(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), savename()));
|
||||
(void) UNLINK(
|
||||
fconv(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), savename()));
|
||||
is_loaded_from_file = 0;
|
||||
}
|
||||
|
||||
@@ -1951,8 +1951,8 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
} else */
|
||||
|
||||
/* External modules */
|
||||
if (opt->parsejava && (opt->parsejava & HTSPARSE_NO_CLASS) == 0
|
||||
&& fexist(savename())) {
|
||||
if (opt->parsejava && (opt->parsejava & HTSPARSE_NO_CLASS) == 0 &&
|
||||
fexist_utf8(savename())) {
|
||||
char BIGSTK buff_err_msg[1024];
|
||||
htsmoduleStruct BIGSTK str;
|
||||
|
||||
@@ -1993,7 +1993,7 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
} // text/html ou autre
|
||||
|
||||
/* Post-processing */
|
||||
if (fexist(savename())) {
|
||||
if (fexist_utf8(savename())) {
|
||||
usercommand(opt, 0, NULL, savename(), urladr(), urlfil());
|
||||
}
|
||||
|
||||
@@ -2084,18 +2084,16 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
//
|
||||
opt->state._hts_in_html_parsing = 3;
|
||||
//
|
||||
old_lst =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.lst"), "rb");
|
||||
old_lst = FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/old.lst"),
|
||||
"rb");
|
||||
if (old_lst) {
|
||||
const size_t sz = llint_to_size_t(
|
||||
fsize(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.lst")));
|
||||
new_lst =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.lst"), "rb");
|
||||
const size_t sz = llint_to_size_t(fsize_utf8(
|
||||
fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.lst")));
|
||||
new_lst = FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.lst"),
|
||||
"rb");
|
||||
if (new_lst != NULL && sz != (size_t) -1) {
|
||||
/* +1 for the NUL below: new.lst is read raw, and the strstr()
|
||||
that follows needs a terminated C string. */
|
||||
@@ -2115,9 +2113,9 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
strcpybuff(file, StringBuff(opt->path_html));
|
||||
strcatbuff(file, line + 1);
|
||||
file[strlen(file) - 1] = '\0';
|
||||
if (fexist(file)) { // toujours sur disque: virer
|
||||
if (fexist_utf8(file)) { // toujours sur disque: virer
|
||||
hts_log_print(opt, LOG_INFO, "Purging %s", file);
|
||||
remove(file);
|
||||
UNLINK(file);
|
||||
purge = 1;
|
||||
}
|
||||
}
|
||||
@@ -2138,7 +2136,8 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
|
||||
strcpybuff(file, StringBuff(opt->path_html));
|
||||
strcatbuff(file, line + 1);
|
||||
while((strnotempty(file)) && (rmdir(file) == 0)) { // ok, éliminé (existait)
|
||||
while ((strnotempty(file)) &&
|
||||
(RMDIR(file) == 0)) { // ok, éliminé (existait)
|
||||
purge = 1;
|
||||
if (opt->log) {
|
||||
hts_log_print(opt, LOG_INFO, "Purging directory %s/",
|
||||
@@ -2245,6 +2244,7 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
|
||||
// ending
|
||||
usercommand(opt, 0, NULL, NULL, NULL, NULL);
|
||||
warc_close_opt(opt);
|
||||
|
||||
// désallocation mémoire & buffers
|
||||
XH_uninit;
|
||||
@@ -2960,8 +2960,8 @@ static void postprocess_file(httrackp *opt, const char *save, const char *adr,
|
||||
if (adr != NULL && strcmp(adr, "primary") == 0) {
|
||||
adr = NULL;
|
||||
}
|
||||
if (save != NULL && opt != NULL && adr != NULL && adr[0]
|
||||
&& strnotempty(save) && fexist(save)) {
|
||||
if (save != NULL && opt != NULL && adr != NULL && adr[0] &&
|
||||
strnotempty(save) && fexist_utf8(save)) {
|
||||
const char *rsc_save = save;
|
||||
const char *rsc_fil = strrchr(fil, '/');
|
||||
int n;
|
||||
@@ -2983,13 +2983,11 @@ static void postprocess_file(httrackp *opt, const char *save, const char *adr,
|
||||
|
||||
if (!opt->state.mimehtml_created) {
|
||||
opt->state.mimefp =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_html), "index.mht"),
|
||||
"wb");
|
||||
(void) unlink(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_html),
|
||||
"index.eml"));
|
||||
FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_html), "index.mht"),
|
||||
"wb");
|
||||
(void) UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_html), "index.eml"));
|
||||
#ifndef _WIN32
|
||||
if (symlink("index.mht",
|
||||
fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_html),
|
||||
@@ -3628,6 +3626,12 @@ HTSEXT_API int copy_htsopt(const httrackp * from, httrackp * to) {
|
||||
if (StringNotEmpty(from->cookies_file))
|
||||
StringCopyS(to->cookies_file, from->cookies_file);
|
||||
|
||||
if (StringNotEmpty(from->warc_file))
|
||||
StringCopyS(to->warc_file, from->warc_file);
|
||||
to->warc_max_size = from->warc_max_size;
|
||||
to->warc_cdx = from->warc_cdx;
|
||||
to->warc_wacz = from->warc_wacz;
|
||||
|
||||
if (from->pause_max_ms > 0) {
|
||||
to->pause_min_ms = from->pause_min_ms;
|
||||
to->pause_max_ms = from->pause_max_ms;
|
||||
|
||||
@@ -40,6 +40,7 @@ Please visit our Website: http://www.httrack.com
|
||||
#include "htscore.h"
|
||||
#include "htsdefines.h"
|
||||
#include "htsalias.h"
|
||||
#include "htswarc.h"
|
||||
#include "htsbauth.h"
|
||||
#include "htswrap.h"
|
||||
#include "htsmodules.h"
|
||||
@@ -441,11 +442,10 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
|| (strnotempty(StringBuff(opt->path_html))))
|
||||
loops++; // do not loop once again and do not include rc file (O option exists)
|
||||
else {
|
||||
if ((!fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/doit.log"))) || (argv_url > 0)) {
|
||||
if ((!fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/doit.log"))) ||
|
||||
(argv_url > 0)) {
|
||||
if (!optinclude_file(
|
||||
fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), HTS_HTTRACKRC),
|
||||
@@ -472,16 +472,12 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
} // traiter -O
|
||||
|
||||
/* load doit.log and insert in current command line */
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/doit.log"))
|
||||
&& (argv_url <= 0)) {
|
||||
FILE *fp =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/doit.log"), "rb");
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/doit.log")) &&
|
||||
(argv_url <= 0)) {
|
||||
FILE *fp = FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/doit.log"),
|
||||
"rb");
|
||||
if (fp) {
|
||||
int insert_after = 1; /* insérer après nom au début */
|
||||
|
||||
@@ -530,23 +526,16 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
|
||||
/* Interrupted mirror detected */
|
||||
if (!opt->quiet) {
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-in_progress.lock"))) {
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-in_progress.lock"))) {
|
||||
/* Old cache */
|
||||
if ((fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/old.dat")))
|
||||
&&
|
||||
(fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/old.ndx")))) {
|
||||
if ((fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/old.dat"))) &&
|
||||
(fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/old.ndx")))) {
|
||||
if (opt->log != NULL) {
|
||||
fprintf(opt->log, "Warning!\n");
|
||||
fprintf(opt->log,
|
||||
@@ -575,111 +564,82 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
if (argv[i][1] == '-') { // --xxx
|
||||
if ((strfield2(argv[i] + 2, "clean")) || (strfield2(argv[i] + 2, "tide"))) { // nettoyer
|
||||
argv[i][1] = '\0';
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), "hts-log.txt")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-log.txt"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), "hts-err.txt")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-err.txt"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_html), "index.html")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_html),
|
||||
"index.html"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-log.txt")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-log.txt"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-err.txt")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-err.txt"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_html), "index.html")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_html), "index.html"));
|
||||
/* */
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.zip")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.zip"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.zip")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.zip"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.dat")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.dat"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.ndx")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.ndx"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.dat")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.dat"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.ndx")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.ndx"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.lst")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.lst"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.lst")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.lst"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.txt")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.txt"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.txt")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.txt"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/doit.log")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/doit.log"));
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-in_progress.lock")))
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-in_progress.lock"));
|
||||
rmdir(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), "hts-cache"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.zip")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.zip"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/old.zip")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/old.zip"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.dat")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.dat"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.ndx")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.ndx"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/old.dat")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/old.dat"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/old.ndx")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/old.ndx"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.lst")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.lst"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/old.lst")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/old.lst"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.txt")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.txt"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/old.txt")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/old.txt"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/doit.log")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/doit.log"));
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-in_progress.lock")))
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-in_progress.lock"));
|
||||
RMDIR(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache"));
|
||||
//
|
||||
} else if (strfield2(argv[i] + 2, "catchurl")) { // capture d'URL via proxy temporaire!
|
||||
argv_url = 1; // forcer a passer les parametres
|
||||
@@ -736,29 +696,27 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
#endif
|
||||
if (argv_url == 0) {
|
||||
// Présence d'un cache, que faire?..
|
||||
if ((fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.zip")))
|
||||
||
|
||||
(fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.dat"))
|
||||
&&
|
||||
fexist(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.ndx")))
|
||||
) { // il existe déja un cache précédent.. renommer
|
||||
if (fexist(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/doit.log"))) { // un cache est présent
|
||||
if ((fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.zip"))) ||
|
||||
(fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.dat")) &&
|
||||
fexist_utf8(
|
||||
fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.ndx")))) { // il existe déja un cache
|
||||
// précédent.. renommer
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/doit.log"))) { // un cache est présent
|
||||
if (x_argvblk != NULL) {
|
||||
int m;
|
||||
|
||||
// établir mode - mode cache: 1 (cache valide) 2 (cache à vérifier)
|
||||
if (fexist(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-in_progress.lock"))) { // cache prioritaire
|
||||
if (fexist_utf8(
|
||||
fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-in_progress.lock"))) { // cache prioritaire
|
||||
m = 1;
|
||||
} else {
|
||||
m = 2;
|
||||
@@ -793,7 +751,7 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
htsmain_free();
|
||||
return -1;
|
||||
}
|
||||
} else { // log existe pas
|
||||
} else { // log existe pas
|
||||
HTS_PANIC_PRINTF("A cache has been found, but no command line");
|
||||
printf
|
||||
("Please launch httrack with proper parameters to reuse the cache\n");
|
||||
@@ -801,7 +759,7 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
} else { // aucune URL définie et pas de cache
|
||||
} else { // aucune URL définie et pas de cache
|
||||
if (opt->quiet) {
|
||||
help(argv[0], !opt->quiet);
|
||||
htsmain_free();
|
||||
@@ -816,26 +774,21 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
}
|
||||
} else { // plus de 2 paramètres
|
||||
// un fichier log existe?
|
||||
if (fexist(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-in_progress.lock"))) { // fichier lock?
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-in_progress.lock"))) { // fichier lock?
|
||||
|
||||
opt->cache = HTS_CACHE_PRIORITY; // cache prioritaire
|
||||
if (opt->quiet == 0) {
|
||||
if ((fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.zip")))
|
||||
||
|
||||
(fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.dat"))
|
||||
&&
|
||||
fexist(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.ndx")))
|
||||
) {
|
||||
if ((fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.zip"))) ||
|
||||
(fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.dat")) &&
|
||||
fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.ndx")))) {
|
||||
HT_REQUEST_START;
|
||||
HT_PRINT("There is a lock-file in the directory ");
|
||||
HT_PRINT(StringBuff(opt->path_log));
|
||||
@@ -849,24 +802,19 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if (fexist(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_html), "index.html"))) {
|
||||
} else if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_html), "index.html"))) {
|
||||
opt->cache = HTS_CACHE_TEST_UPDATE;
|
||||
if (opt->quiet == 0) {
|
||||
if ((fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.zip")))
|
||||
||
|
||||
(fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.dat"))
|
||||
&&
|
||||
fexist(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.ndx")))
|
||||
) {
|
||||
if ((fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.zip"))) ||
|
||||
(fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.dat")) &&
|
||||
fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/new.ndx")))) {
|
||||
HT_REQUEST_START;
|
||||
HT_PRINT
|
||||
("There is an index.html and a hts-cache folder in the directory ");
|
||||
@@ -1499,13 +1447,13 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
LLint fz;
|
||||
|
||||
na++;
|
||||
fz = fsize(argv[na]);
|
||||
fz = fsize_utf8(argv[na]);
|
||||
if (fz < 0) {
|
||||
HTS_PANIC_PRINTF("File url list could not be opened");
|
||||
htsmain_free();
|
||||
return -1;
|
||||
} else {
|
||||
FILE *fp = fopen(argv[na], "rb");
|
||||
FILE *fp = FOPEN(argv[na], "rb");
|
||||
|
||||
if (fp != NULL) {
|
||||
int cl = (int) strlen(url);
|
||||
@@ -1789,6 +1737,54 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
StringCopy(opt->cookies_file, argv[na]);
|
||||
}
|
||||
break;
|
||||
case 'r': // warc / warc-file: write an ISO-28500 WARC archive
|
||||
if (*(com + 1) == 'f') { // --warc-file NAME: explicit basename
|
||||
com++;
|
||||
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
|
||||
HTS_PANIC_PRINTF(
|
||||
"Option warc-file needs a blank space and a WARC name");
|
||||
htsmain_free();
|
||||
return -1;
|
||||
}
|
||||
na++;
|
||||
if (strlen(argv[na]) >= 1024) {
|
||||
HTS_PANIC_PRINTF("WARC file name too long");
|
||||
htsmain_free();
|
||||
return -1;
|
||||
}
|
||||
StringCopy(opt->warc_file, argv[na]);
|
||||
} else if (*(com + 1) == 's') { // --warc-max-size N: rotation
|
||||
com++;
|
||||
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
|
||||
HTS_PANIC_PRINTF(
|
||||
"Option warc-max-size needs a blank space and a size");
|
||||
htsmain_free();
|
||||
return -1;
|
||||
}
|
||||
na++;
|
||||
{ // reject non-numeric/negative/overflow; keep default 0
|
||||
// (single file)
|
||||
char *end;
|
||||
LLint v;
|
||||
errno = 0;
|
||||
v = strtoll(argv[na], &end, 10);
|
||||
if (isdigit((unsigned char) argv[na][0]) && *end == '\0' &&
|
||||
errno != ERANGE)
|
||||
opt->warc_max_size = v;
|
||||
}
|
||||
} else if (*(com + 1) == 'c') { // --warc-cdx: sorted CDXJ index
|
||||
com++;
|
||||
opt->warc_cdx = 1;
|
||||
} else if (*(com + 1) == 'z') { // --wacz: WACZ package
|
||||
com++;
|
||||
opt->warc_wacz = 1;
|
||||
opt->warc_cdx = 1; // WACZ embeds the CDXJ index
|
||||
if (!StringNotEmpty(opt->warc_file))
|
||||
StringCopy(opt->warc_file, WARC_AUTONAME);
|
||||
} else { // --warc: auto-named archive under the output dir
|
||||
StringCopy(opt->warc_file, WARC_AUTONAME);
|
||||
}
|
||||
break;
|
||||
case 'Y': // why: explain the filter verdict for a URL, no crawl
|
||||
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
|
||||
HTS_PANIC_PRINTF("Option why needs a blank space and a URL");
|
||||
@@ -2001,7 +1997,7 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
/*cache */ &cache, /*hash */ NULL, /*ptr */
|
||||
0, /*numero_passe */ 0, /*mime_type */
|
||||
NULL) != -1) {
|
||||
if (fexist(afs.save)) {
|
||||
if (fexist_utf8(afs.save)) {
|
||||
fprintf(stdout, "Content-location: %s\r\n",
|
||||
afs.save);
|
||||
}
|
||||
@@ -2099,18 +2095,16 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
uLong repaired = 0;
|
||||
uLong repairedBytes = 0;
|
||||
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.zip"))) {
|
||||
if (fexist_utf8(fconcat(
|
||||
OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/new.zip"))) {
|
||||
name =
|
||||
fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/new.zip");
|
||||
} else
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.zip"))) {
|
||||
} else if (fexist_utf8(fconcat(OPT_GET_BUFF(opt),
|
||||
OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/old.zip"))) {
|
||||
name =
|
||||
fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/old.zip");
|
||||
@@ -2129,10 +2123,11 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/repair.tmp"), &repaired,
|
||||
&repairedBytes) == Z_OK) {
|
||||
unlink(name);
|
||||
rename(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/repair.zip"), name);
|
||||
UNLINK(name);
|
||||
RENAME(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-cache/repair.zip"),
|
||||
name);
|
||||
fprintf(stderr,
|
||||
"Cache: %d bytes successfully recovered in %d entries\n",
|
||||
(int) repairedBytes, (int) repaired);
|
||||
@@ -2350,10 +2345,9 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
hts_cache_reconcile(opt, CACHE_RECONCILE_INTERRUPTED);
|
||||
// Débuggage des en têtes
|
||||
if (_DEBUG_HEAD) {
|
||||
ioinfo =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), "hts-ioinfo.txt"),
|
||||
"wb");
|
||||
ioinfo = FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-ioinfo.txt"),
|
||||
"wb");
|
||||
}
|
||||
|
||||
{
|
||||
@@ -2428,9 +2422,9 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
/* readme for information purpose */
|
||||
{
|
||||
FILE *fp =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/readme.txt"), "wb");
|
||||
FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/readme.txt"),
|
||||
"wb");
|
||||
if (fp) {
|
||||
fprintf(fp, "What's in this folder?" LF);
|
||||
fprintf(fp, "" LF);
|
||||
@@ -2482,17 +2476,16 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
int i;
|
||||
|
||||
#ifdef _WIN32
|
||||
mkdir(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), "hts-cache"));
|
||||
hts_mkdir_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache"));
|
||||
#else
|
||||
mkdir(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), "hts-cache"),
|
||||
HTS_PROTECT_FOLDER);
|
||||
#endif
|
||||
fp =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log),
|
||||
"hts-cache/doit.log"), "wb");
|
||||
fp = FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-cache/doit.log"),
|
||||
"wb");
|
||||
if (fp) {
|
||||
for(i = 0 + 1; i < argc; i++) {
|
||||
if (((strchr(argv[i], ' ') != NULL)
|
||||
@@ -2537,7 +2530,7 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
}
|
||||
}
|
||||
// petit message dans le lock
|
||||
if ((fp = fopen(n_lock, "wb")) != NULL) {
|
||||
if ((fp = FOPEN(n_lock, "wb")) != NULL) {
|
||||
int i;
|
||||
|
||||
fprintf(fp, "Mirror in progress since %s .. please wait!" LF, t);
|
||||
@@ -2697,16 +2690,15 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
char *f = OPT_GET_BUFF(opt);
|
||||
|
||||
sprintf(f, "%s/%s", CACHE_REFNAME, entry->d_name);
|
||||
(void)
|
||||
unlink(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), f));
|
||||
(void) UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), f));
|
||||
}
|
||||
}
|
||||
if (dir != NULL) {
|
||||
(void) closedir(dir);
|
||||
}
|
||||
(void)
|
||||
rmdir(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), CACHE_REFNAME));
|
||||
(void) RMDIR(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), CACHE_REFNAME));
|
||||
}
|
||||
|
||||
/* Info for wrappers */
|
||||
@@ -2734,7 +2726,7 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
}
|
||||
}
|
||||
// supprimer lock
|
||||
remove(n_lock);
|
||||
UNLINK(n_lock);
|
||||
}
|
||||
|
||||
if (x_argvblk)
|
||||
|
||||
@@ -412,9 +412,9 @@ void help_catchurl(const char *dest_path) {
|
||||
do {
|
||||
snprintf(dest, sizeof(dest), "%s%s%d", dest_path, "hts-post", i);
|
||||
i++;
|
||||
} while(fexist(dest));
|
||||
} while (fexist_utf8(dest));
|
||||
{
|
||||
FILE *fp = fopen(dest, "wb");
|
||||
FILE *fp = FOPEN(dest, "wb");
|
||||
|
||||
if (fp) {
|
||||
fwrite(data, strlen(data), 1, fp);
|
||||
@@ -593,6 +593,10 @@ void help(const char *app, int more) {
|
||||
infomsg
|
||||
(" C create/use a cache for updates and retries (C0 no cache,C1 cache is prioritary,* C2 test update before)");
|
||||
infomsg(" k store all files in cache (not useful if files on disk)");
|
||||
infomsg(" %r write an ISO-28500 WARC/1.1 archive; --warc-file NAME sets the "
|
||||
"output name, --warc-max-size N rotates segments past N bytes, "
|
||||
"--warc-cdx also writes a sorted CDXJ index, --wacz packages it all "
|
||||
"as a WACZ file");
|
||||
infomsg(" %n do not re-download locally erased files");
|
||||
infomsg
|
||||
(" %v display on screen filenames downloaded (in realtime) - * %v1 short version - %v2 full animation");
|
||||
|
||||
@@ -349,9 +349,13 @@ void index_finish(const char *indexpath, int mode) {
|
||||
|
||||
// Write new file
|
||||
if (mode == 1) // TEXT
|
||||
fp = fopen(concat(catbuff, sizeof(catbuff), indexpath, "index.txt"), "wb");
|
||||
fp = FOPEN(
|
||||
concat(catbuff, sizeof(catbuff), indexpath, "index.txt"),
|
||||
"wb");
|
||||
else // HTML
|
||||
fp = fopen(concat(catbuff, sizeof(catbuff), indexpath, "sindex.html"), "wb");
|
||||
fp = FOPEN(
|
||||
concat(catbuff, sizeof(catbuff), indexpath, "sindex.html"),
|
||||
"wb");
|
||||
if (fp) {
|
||||
char current_word[KEYW_LEN + 32];
|
||||
char word[KEYW_LEN + 32];
|
||||
|
||||
123
src/htslib.c
123
src/htslib.c
@@ -36,6 +36,7 @@ Please visit our Website: http://www.httrack.com
|
||||
// Fichier librairie .c
|
||||
|
||||
#include "htscore.h"
|
||||
#include "htswarc.h"
|
||||
|
||||
/* specific definitions */
|
||||
#include "htsbase.h"
|
||||
@@ -1201,6 +1202,11 @@ int http_sendhead(httrackp * opt, t_cookie * cookie, int mode,
|
||||
} // Fin test pas postfile
|
||||
//
|
||||
|
||||
// Stash the raw request for the WARC request record (freed at
|
||||
// back_clear_entry)
|
||||
if (StringNotEmpty(opt->warc_file))
|
||||
warc_stash_request(retour, bstr.buffer);
|
||||
|
||||
// Callback
|
||||
{
|
||||
int test_head =
|
||||
@@ -2114,6 +2120,8 @@ htsblk http_test(httrackp * opt, const char *adr, const char *fil, char *loc) {
|
||||
#if HTS_DEBUG_CLOSESOCK
|
||||
DEBUG_W("http_test: deletehttp\n");
|
||||
#endif
|
||||
// this probe's htsblk is discarded by callers, so free any WARC stash here
|
||||
warc_free_request(&retour);
|
||||
deletehttp(&retour);
|
||||
retour.soc = INVALID_SOCKET;
|
||||
}
|
||||
@@ -6010,6 +6018,8 @@ HTSEXT_API httrackp *hts_create_opt(void) {
|
||||
StringCopy(opt->footer, HTS_DEFAULT_FOOTER);
|
||||
StringCopy(opt->strip_query, "");
|
||||
StringCopy(opt->cookies_file, "");
|
||||
StringCopy(opt->warc_file, "");
|
||||
opt->warc_max_size = 0; /* no rotation unless --warc-max-size sets it */
|
||||
StringCopy(opt->why_url, "");
|
||||
opt->pause_min_ms = 0;
|
||||
opt->pause_max_ms = 0;
|
||||
@@ -6161,6 +6171,7 @@ HTSEXT_API void hts_free_opt(httrackp * opt) {
|
||||
StringFree(opt->strip_query);
|
||||
StringFree(opt->cookies_file);
|
||||
StringFree(opt->why_url);
|
||||
StringFree(opt->warc_file);
|
||||
|
||||
StringFree(opt->path_html);
|
||||
StringFree(opt->path_html_utf8);
|
||||
@@ -6420,21 +6431,29 @@ HTSEXT_API int hts_resetvar(void) {
|
||||
#ifdef _WIN32
|
||||
|
||||
typedef struct dirent dirent;
|
||||
static LPWSTR hts_pathToUCS2(const char *path);
|
||||
|
||||
DIR *opendir(const char *name) {
|
||||
WIN32_FILE_ATTRIBUTE_DATA st;
|
||||
DIR *dir;
|
||||
size_t len;
|
||||
int i;
|
||||
LPWSTR wname;
|
||||
|
||||
if (name == NULL || *name == '\0') {
|
||||
errno = ENOENT;
|
||||
return NULL;
|
||||
}
|
||||
if (!GetFileAttributesEx(name, GetFileExInfoStandard, &st)
|
||||
|| (st.dwFileAttributes & FILE_ATTRIBUTE_DIRECTORY) == 0) {
|
||||
// Wide \\?\ path: no MAX_PATH cap, no CP_ACP mis-decode (#133,#630).
|
||||
wname = hts_pathToUCS2(name);
|
||||
if (wname == NULL ||
|
||||
!GetFileAttributesExW(wname, GetFileExInfoStandard, &st) ||
|
||||
(st.dwFileAttributes & FILE_ATTRIBUTE_DIRECTORY) == 0) {
|
||||
freet(wname);
|
||||
errno = ENOENT;
|
||||
return NULL;
|
||||
}
|
||||
freet(wname);
|
||||
dir = calloc(sizeof(DIR), 1);
|
||||
if (dir == NULL) {
|
||||
errno = ENOMEM;
|
||||
@@ -6454,19 +6473,29 @@ DIR *opendir(const char *name) {
|
||||
}
|
||||
|
||||
struct dirent *readdir(DIR * dir) {
|
||||
WIN32_FIND_DATAA find;
|
||||
WIN32_FIND_DATAW find;
|
||||
|
||||
if (dir->h == INVALID_HANDLE_VALUE) {
|
||||
dir->h = FindFirstFileA(dir->name, &find);
|
||||
// \\?\-prefix so a long/non-ASCII directory enumerates instead of ENOENT.
|
||||
LPWSTR wname = hts_pathToUCS2(dir->name);
|
||||
|
||||
dir->h =
|
||||
wname != NULL ? FindFirstFileW(wname, &find) : INVALID_HANDLE_VALUE;
|
||||
freet(wname);
|
||||
} else {
|
||||
if (!FindNextFile(dir->h, &find)) {
|
||||
if (!FindNextFileW(dir->h, &find)) {
|
||||
FindClose(dir->h);
|
||||
dir->h = INVALID_HANDLE_VALUE;
|
||||
}
|
||||
}
|
||||
if (dir->h != INVALID_HANDLE_VALUE) {
|
||||
char *u = hts_convertUCS2StringToUTF8(find.cFileName, -1);
|
||||
|
||||
dir->entry.d_name[0] = 0;
|
||||
strncat(dir->entry.d_name, find.cFileName, HTS_DIRENT_SIZE - 1);
|
||||
if (u != NULL) {
|
||||
strncat(dir->entry.d_name, u, HTS_DIRENT_SIZE - 1);
|
||||
freet(u);
|
||||
}
|
||||
return &dir->entry;
|
||||
}
|
||||
errno = ENOENT;
|
||||
@@ -6499,9 +6528,61 @@ static void copyWchar(LPWSTR dest, const char *src) {
|
||||
dest[i] = '\0';
|
||||
}
|
||||
|
||||
/* UTF-8 path -> UCS-2 for the _w* file APIs. At/above HTS_WIN_LONGPATH_MIN,
|
||||
\\?\-prefix it via GetFullPathNameW to clear MAX_PATH (#133); else unchanged.
|
||||
Any prefixing failure falls back to the plain converted path. */
|
||||
#define HTS_WIN_LONGPATH_MIN 240 /* stay clear of MAX_PATH (260) */
|
||||
|
||||
static LPWSTR hts_pathToUCS2(const char *path) {
|
||||
LPWSTR wpath = hts_convertUTF8StringToUCS2(path, (int) strlen(path), NULL);
|
||||
|
||||
if (wpath == NULL) {
|
||||
return NULL;
|
||||
}
|
||||
const size_t len = wcslen(wpath);
|
||||
// Already "\\?\" or "\\.\": don't re-prefix.
|
||||
const int verbatim = len >= 4 && wpath[0] == L'\\' && wpath[1] == L'\\' &&
|
||||
(wpath[2] == L'?' || wpath[2] == L'.') &&
|
||||
wpath[3] == L'\\';
|
||||
|
||||
if (len < HTS_WIN_LONGPATH_MIN || verbatim) {
|
||||
return wpath;
|
||||
}
|
||||
|
||||
const DWORD need = GetFullPathNameW(wpath, 0, NULL, NULL); /* incl NUL */
|
||||
LPWSTR full = need != 0 ? malloct((size_t) need * sizeof(WCHAR)) : NULL;
|
||||
|
||||
if (full == NULL) {
|
||||
return wpath; /* fall back to the plain path */
|
||||
}
|
||||
const DWORD written = GetFullPathNameW(wpath, need, full, NULL);
|
||||
|
||||
if (written == 0 || written >= need || full[0] == L'\0') {
|
||||
freet(full);
|
||||
return wpath;
|
||||
}
|
||||
|
||||
const int isUNC = full[0] == L'\\' && full[1] == L'\\';
|
||||
// UNC "\\srv\share" -> "\\?\UNC\srv\share": the prefix subsumes the "\\".
|
||||
const WCHAR *const pfx = isUNC ? L"\\\\?\\UNC\\" : L"\\\\?\\";
|
||||
const WCHAR *const body = isUNC ? full + 2 : full;
|
||||
const size_t pfxLen = wcslen(pfx), bodyLen = wcslen(body);
|
||||
LPWSTR out = malloct((pfxLen + bodyLen + 1) * sizeof(WCHAR));
|
||||
|
||||
if (out == NULL) {
|
||||
freet(full);
|
||||
return wpath;
|
||||
}
|
||||
memcpybuff(out, pfx, pfxLen * sizeof(WCHAR));
|
||||
memcpybuff(out + pfxLen, body, (bodyLen + 1) * sizeof(WCHAR));
|
||||
freet(full);
|
||||
freet(wpath);
|
||||
return out;
|
||||
}
|
||||
|
||||
FILE *hts_fopen_utf8(const char *path, const char *mode) {
|
||||
WCHAR wmode[32];
|
||||
LPWSTR wpath = hts_convertUTF8StringToUCS2(path, (int) strlen(path), NULL);
|
||||
LPWSTR wpath = hts_pathToUCS2(path);
|
||||
|
||||
assertf(strlen(mode) < sizeof(wmode) / sizeof(WCHAR));
|
||||
copyWchar(wmode, mode);
|
||||
@@ -6517,7 +6598,7 @@ FILE *hts_fopen_utf8(const char *path, const char *mode) {
|
||||
}
|
||||
|
||||
int hts_stat_utf8(const char *path, STRUCT_STAT * buf) {
|
||||
LPWSTR wpath = hts_convertUTF8StringToUCS2(path, (int) strlen(path), NULL);
|
||||
LPWSTR wpath = hts_pathToUCS2(path);
|
||||
|
||||
if (wpath != NULL) {
|
||||
const int result = _wstat64(wpath, buf);
|
||||
@@ -6531,7 +6612,7 @@ int hts_stat_utf8(const char *path, STRUCT_STAT * buf) {
|
||||
}
|
||||
|
||||
int hts_unlink_utf8(const char *path) {
|
||||
LPWSTR wpath = hts_convertUTF8StringToUCS2(path, (int) strlen(path), NULL);
|
||||
LPWSTR wpath = hts_pathToUCS2(path);
|
||||
|
||||
if (wpath != NULL) {
|
||||
const int result = _wunlink(wpath);
|
||||
@@ -6545,10 +6626,8 @@ int hts_unlink_utf8(const char *path) {
|
||||
}
|
||||
|
||||
int hts_rename_utf8(const char *oldpath, const char *newpath) {
|
||||
LPWSTR woldpath =
|
||||
hts_convertUTF8StringToUCS2(oldpath, (int) strlen(oldpath), NULL);
|
||||
LPWSTR wnewpath =
|
||||
hts_convertUTF8StringToUCS2(newpath, (int) strlen(newpath), NULL);
|
||||
LPWSTR woldpath = hts_pathToUCS2(oldpath);
|
||||
LPWSTR wnewpath = hts_pathToUCS2(newpath);
|
||||
if (woldpath != NULL && wnewpath != NULL) {
|
||||
const int result = _wrename(woldpath, wnewpath);
|
||||
|
||||
@@ -6565,8 +6644,22 @@ int hts_rename_utf8(const char *oldpath, const char *newpath) {
|
||||
}
|
||||
}
|
||||
|
||||
int hts_rmdir_utf8(const char *path) {
|
||||
LPWSTR wpath = hts_pathToUCS2(path);
|
||||
|
||||
if (wpath != NULL) {
|
||||
const int result = _wrmdir(wpath);
|
||||
|
||||
free(wpath);
|
||||
return result;
|
||||
} else {
|
||||
// Fallback on conversion error.
|
||||
return _rmdir(path);
|
||||
}
|
||||
}
|
||||
|
||||
int hts_mkdir_utf8(const char *path) {
|
||||
LPWSTR wpath = hts_convertUTF8StringToUCS2(path, (int) strlen(path), NULL);
|
||||
LPWSTR wpath = hts_pathToUCS2(path);
|
||||
|
||||
if (wpath != NULL) {
|
||||
const int result = _wmkdir(wpath);
|
||||
@@ -6581,7 +6674,7 @@ int hts_mkdir_utf8(const char *path) {
|
||||
|
||||
HTSEXT_API int hts_utime_utf8(const char *path, const STRUCT_UTIMBUF * times) {
|
||||
STRUCT_UTIMBUF mtimes = *times;
|
||||
LPWSTR wpath = hts_convertUTF8StringToUCS2(path, (int) strlen(path), NULL);
|
||||
LPWSTR wpath = hts_pathToUCS2(path);
|
||||
|
||||
if (wpath != NULL) {
|
||||
const int result = _wutime(wpath, &mtimes);
|
||||
|
||||
@@ -609,7 +609,9 @@ static HTS_UNUSED size_t llint_to_size_t(LLint o) {
|
||||
|
||||
/* dirent() compatibility */
|
||||
#ifdef _WIN32
|
||||
#define HTS_DIRENT_SIZE 256
|
||||
/* Holds a UTF-8 d_name: MAX_PATH (260) UTF-16 units expand to <=3 bytes each.
|
||||
Windows-only struct, ABI free to break. */
|
||||
#define HTS_DIRENT_SIZE 1024
|
||||
struct dirent {
|
||||
ino_t d_ino; /* ignored */
|
||||
off_t d_off; /* ignored */
|
||||
|
||||
@@ -47,6 +47,7 @@ Please visit our Website: http://www.httrack.com
|
||||
#include "htszlib.h"
|
||||
#endif
|
||||
#include <ctype.h>
|
||||
#include <limits.h>
|
||||
|
||||
#define ADD_STANDARD_PATH \
|
||||
{ /* ajout nom */\
|
||||
@@ -1475,15 +1476,37 @@ int url_savename(lien_adrfilsave *const afs,
|
||||
sizeof(afs->save) - (size_t) (lastDot - afs->save));
|
||||
}
|
||||
}
|
||||
// enforce 260-character path limit before inserting destination path
|
||||
// note: 12 characters at least for WIN32, and 12 for ".99.delayed"
|
||||
// (MSDN) "When using an API to create a directory, the specified path
|
||||
// cannot be so long that you cannot append an 8.3 file name
|
||||
// (that is, the directory name cannot exceed MAX_PATH minus 12)."
|
||||
#define HTS_MAX_PATH_LEN ( 260 - 12 - 12 )
|
||||
// Cap the save path: the final parent+name is copied into a fixed buffer that
|
||||
// aborts() on overflow (htssafe.h), so clamp every ceiling to fit it.
|
||||
#define HTS_SAVE_BUFSIZE (HTS_URLMAXSIZE * 2) /* sizeof(afs->save) */
|
||||
#define HTS_PATH_TAIL_RESERVE 64 /* collision suffix + ".delayed" + NUL */
|
||||
#ifdef _WIN32
|
||||
// MAX_PATH minus 8.3 headroom (MSDN) minus the ".delayed" marker; raising it
|
||||
// needs the engine to "\\?\"-prefix its paths, which is separate work.
|
||||
#define HTS_MAX_PATH_LEN (260 - 12 - 12)
|
||||
#define MAX_SEG_LEN 48
|
||||
#else
|
||||
// #133: use the platform's own PATH_MAX/NAME_MAX (Linux/Android 4096, macOS
|
||||
// 1024) rather than the far smaller Windows MAX_PATH.
|
||||
#ifdef PATH_MAX
|
||||
#define HTS_PATH_MAX_ PATH_MAX
|
||||
#else
|
||||
#define HTS_PATH_MAX_ 1024
|
||||
#endif
|
||||
#ifdef NAME_MAX
|
||||
#define HTS_NAME_MAX_ NAME_MAX
|
||||
#else
|
||||
#define HTS_NAME_MAX_ 255
|
||||
#endif
|
||||
#define HTS_MAX_PATH_LEN \
|
||||
((HTS_PATH_MAX_ - HTS_PATH_TAIL_RESERVE) < \
|
||||
(HTS_SAVE_BUFSIZE - HTS_PATH_TAIL_RESERVE) \
|
||||
? (HTS_PATH_MAX_ - HTS_PATH_TAIL_RESERVE) \
|
||||
: (HTS_SAVE_BUFSIZE - HTS_PATH_TAIL_RESERVE))
|
||||
#define MAX_SEG_LEN (HTS_NAME_MAX_ > 64 ? HTS_NAME_MAX_ - 16 : HTS_NAME_MAX_)
|
||||
#endif
|
||||
#define MIN_LAST_SEG_RESERVE 12
|
||||
#define MAX_LAST_SEG_RESERVE 24
|
||||
#define MAX_SEG_LEN 48
|
||||
if (hts_stringLengthUTF8(afs->save) +
|
||||
hts_stringLengthUTF8(StringBuff(opt->path_html_utf8)) >=
|
||||
HTS_MAX_PATH_LEN) {
|
||||
@@ -1493,7 +1516,7 @@ int url_savename(lien_adrfilsave *const afs,
|
||||
if (wsave != NULL) {
|
||||
const size_t parentLen =
|
||||
hts_stringLengthUTF8(StringBuff(opt->path_html_utf8));
|
||||
// parent path length is not insane (otherwise, ignore and pick 200 as
|
||||
// parent path length is not insane (otherwise, ignore and pick 200 as
|
||||
// suffix length)
|
||||
const size_t maxLen =
|
||||
parentLen <
|
||||
@@ -1584,9 +1607,37 @@ int url_savename(lien_adrfilsave *const afs,
|
||||
// Re-check again ending space or dot after cut (see bug #5)
|
||||
cleanEndingSpaceOrDot(afs->save);
|
||||
}
|
||||
// The cut above counts UTF-8 codepoints, but parent+name lands in a fixed
|
||||
// byte buffer that aborts() on overflow (htssafe.h). A multibyte name can
|
||||
// pass the codepoint cap yet overflow in bytes, so hard-cut on a codepoint
|
||||
// boundary to keep parent+name inside the buffer regardless (#133).
|
||||
{
|
||||
const size_t parentBytes = strlen(StringBuff(opt->path_html_utf8));
|
||||
const size_t cap = HTS_SAVE_BUFSIZE - HTS_PATH_TAIL_RESERVE;
|
||||
// Shrink only the name. A parent that alone fills the buffer is left to the
|
||||
// existing prepend abort, not collapsed to an empty name that would collide
|
||||
// across URLs and overrun the unbounded collision-suffix sprintf.
|
||||
if (parentBytes < cap) {
|
||||
size_t budget = cap - parentBytes;
|
||||
if (strlen(afs->save) > budget) {
|
||||
while (budget > 0 && ((unsigned char) afs->save[budget] & 0xC0) == 0x80)
|
||||
budget--; // back off a continuation byte, never split a char
|
||||
afs->save[budget] = '\0';
|
||||
cleanEndingSpaceOrDot(afs->save);
|
||||
}
|
||||
}
|
||||
}
|
||||
#undef MAX_UTF8_SEQ_CHARS
|
||||
#undef MIN_LAST_SEG_RESERVE
|
||||
#undef MAX_LAST_SEG_RESERVE
|
||||
#undef MAX_SEG_LEN
|
||||
#undef HTS_MAX_PATH_LEN
|
||||
#undef HTS_PATH_TAIL_RESERVE
|
||||
#undef HTS_SAVE_BUFSIZE
|
||||
#ifndef _WIN32
|
||||
#undef HTS_PATH_MAX_
|
||||
#undef HTS_NAME_MAX_
|
||||
#endif
|
||||
|
||||
// chemin primaire éventuel A METTRE AVANT
|
||||
if (strnotempty(StringBuff(opt->path_html_utf8))) {
|
||||
|
||||
17
src/htsopt.h
17
src/htsopt.h
@@ -256,6 +256,7 @@ struct htsoptstate {
|
||||
unsigned int debug_state;
|
||||
unsigned int tmpnameid; /**< counter for temporary file names */
|
||||
int is_ended; /**< mirror has finished */
|
||||
void *warc; /**< open WARC writer (warc_writer*), or NULL */
|
||||
};
|
||||
|
||||
/* Library handles */
|
||||
@@ -538,6 +539,14 @@ struct httrackp {
|
||||
int pause_max_ms; /**< inter-file pause upper bound, ms */
|
||||
String why_url; /**< URL to diagnose (--why): print the deciding filter rule
|
||||
and exit without crawling */
|
||||
String warc_file; /**< WARC output: WARC_AUTONAME for --warc, or the
|
||||
--warc-file basename (appended at the tail: ABI) */
|
||||
LLint warc_max_size; /**< --warc-max-size: rotate the archive past this many
|
||||
bytes (<=0: single file). Tail: ABI */
|
||||
hts_boolean warc_cdx; /**< --warc-cdx: write a sorted CDXJ index next to the
|
||||
archive. Tail: ABI */
|
||||
hts_boolean warc_wacz; /**< --wacz: package archive+index+pages as a WACZ zip
|
||||
(implies --warc + --warc-cdx). Tail: ABI */
|
||||
};
|
||||
|
||||
/* Running statistics for a mirror. */
|
||||
@@ -661,6 +670,14 @@ struct htsblk {
|
||||
/* Restart-whole signal: a resume this response rejected (unusable 206) must
|
||||
retry with no Range, else a surviving partial/temp-ref loops (#581). */
|
||||
hts_boolean refetch_wholefile;
|
||||
char *warc_reqhdr; /**< stashed raw request header block for WARC (or NULL) */
|
||||
char *
|
||||
warc_resphdr; /**< stashed raw response header block for WARC (or NULL) */
|
||||
int warc_truncated; /**< WARC-Truncated reason for a cap-truncated body
|
||||
(WARC_TRUNC_*, 0=none). Tail: ABI */
|
||||
char *warc_rawpath; /**< verbatim WARC: spooled compressed body path, or NULL
|
||||
(owns the file; unlinked on free). Tail: ABI */
|
||||
LLint warc_rawsize; /**< byte length of warc_rawpath. Tail: ABI */
|
||||
/*char digest[32+2]; // md5 digest generated by the engine ("" if none) */
|
||||
};
|
||||
|
||||
|
||||
@@ -3656,7 +3656,7 @@ int hts_mirror_check_moved(htsmoduleStruct * str,
|
||||
!ref_existed;
|
||||
if (fexist_utf8(heap(ptr)->sav)) {
|
||||
had_partial = 1;
|
||||
remove(heap(ptr)->sav);
|
||||
UNLINK(heap(ptr)->sav);
|
||||
}
|
||||
|
||||
// Re-get once, only if a partial existed and both Range triggers are
|
||||
@@ -3862,18 +3862,13 @@ void hts_mirror_process_user_interaction(htsmoduleStruct * str,
|
||||
int do_pause = 0;
|
||||
|
||||
// user pause lockfile : create hts-paused.lock --> HTTrack will be paused
|
||||
if (fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-stop.lock"))) {
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-stop.lock"))) {
|
||||
// remove lockfile
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-stop.lock"));
|
||||
if (!fexist
|
||||
(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-stop.lock"))) {
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-stop.lock"));
|
||||
if (!fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-stop.lock"))) {
|
||||
do_pause = 1;
|
||||
}
|
||||
}
|
||||
@@ -3902,11 +3897,9 @@ void hts_mirror_process_user_interaction(htsmoduleStruct * str,
|
||||
}
|
||||
}
|
||||
{
|
||||
FILE *fp =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-paused.lock"), "wb");
|
||||
FILE *fp = FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-paused.lock"),
|
||||
"wb");
|
||||
if (fp) {
|
||||
fspc(NULL, fp, "info"); // dater
|
||||
fprintf(fp,
|
||||
@@ -4222,18 +4215,17 @@ int hts_mirror_wait_for_next_file(htsmoduleStruct * str,
|
||||
int a = 0;
|
||||
|
||||
*stre->last_info_shell_ = tl;
|
||||
if (fexist(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt), StringBuff(opt->path_log), "hts-autopsy"))) { // débuggage: teste si le robot est vivant
|
||||
if (fexist_utf8(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-autopsy"))) { // débuggage: teste si le
|
||||
// robot est vivant
|
||||
// (oui je sais un robot vivant.. mais bon.. il a le droit de vivre lui aussi)
|
||||
// (libérons les robots esclaves de l'internet!)
|
||||
remove(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-autopsy"));
|
||||
fp =
|
||||
fopen(fconcat
|
||||
(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log),
|
||||
"hts-isalive"), "wb");
|
||||
UNLINK(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-autopsy"));
|
||||
fp = FOPEN(fconcat(OPT_GET_BUFF(opt), OPT_GET_BUFF_SIZE(opt),
|
||||
StringBuff(opt->path_log), "hts-isalive"),
|
||||
"wb");
|
||||
a = 1;
|
||||
}
|
||||
if ((*stre->info_shell_) || a) {
|
||||
|
||||
1406
src/htsselftest.c
1406
src/htsselftest.c
File diff suppressed because it is too large
Load Diff
@@ -1027,6 +1027,17 @@ HTSEXT_API int hts_buildtopindex(httrackp * opt, const char *path,
|
||||
freet(category);
|
||||
category = NULL;
|
||||
}
|
||||
#ifdef _WIN32
|
||||
/* category is ANSI-codepage, doc is utf-8: convert (#216) */
|
||||
else {
|
||||
char *cat_utf8 = hts_convertStringSystemToUTF8(
|
||||
category, strlen(category));
|
||||
if (cat_utf8 != NULL) {
|
||||
freet(category);
|
||||
category = cat_utf8;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
if (category == NULL) {
|
||||
@@ -1044,7 +1055,18 @@ HTSEXT_API int hts_buildtopindex(httrackp * opt, const char *path,
|
||||
oldchain->next = chain;
|
||||
}
|
||||
chain->next = NULL;
|
||||
#ifdef _WIN32
|
||||
/* name is ANSI-codepage, doc is utf-8: convert (#216) */
|
||||
{
|
||||
const char *const name = hts_findgetname(h);
|
||||
char *name_utf8 =
|
||||
hts_convertStringSystemToUTF8(name, strlen(name));
|
||||
strcpybuff(chain->name, name_utf8 != NULL ? name_utf8 : name);
|
||||
freet(name_utf8);
|
||||
}
|
||||
#else
|
||||
strcpybuff(chain->name, hts_findgetname(h));
|
||||
#endif
|
||||
chain->category = category;
|
||||
chain->level = level;
|
||||
}
|
||||
|
||||
1799
src/htswarc.c
Normal file
1799
src/htswarc.c
Normal file
File diff suppressed because it is too large
Load Diff
131
src/htswarc.h
Normal file
131
src/htswarc.h
Normal file
@@ -0,0 +1,131 @@
|
||||
/* ------------------------------------------------------------ */
|
||||
/*
|
||||
HTTrack Website Copier, Offline Browser for Windows and Unix
|
||||
Copyright (C) 2026 Xavier Roche and other contributors
|
||||
|
||||
SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License
|
||||
along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
Ethical use: we kindly ask that you NOT use this software to harvest email
|
||||
addresses or to collect any other private information about people. Doing so
|
||||
would dishonor our work and waste the many hours we have spent on it.
|
||||
|
||||
Please visit our Website: http://www.httrack.com
|
||||
*/
|
||||
|
||||
/* ------------------------------------------------------------ */
|
||||
/* HTTrack WARC/1.1 output writer (ISO 28500). Internal, not installed.
|
||||
All WARC record serialization, gzip-member framing, digests, UUID and
|
||||
revisit dedup live here; the engine only stashes the request, frees it,
|
||||
and calls one entry point per finished transaction. */
|
||||
/* ------------------------------------------------------------ */
|
||||
|
||||
#ifndef HTS_WARC_DEFH
|
||||
#define HTS_WARC_DEFH
|
||||
|
||||
#include "htsopt.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* opt->warc_file sentinel: --warc with no argument => auto-name the archive
|
||||
under the project's output directory at open time. */
|
||||
#define WARC_AUTONAME "\001auto"
|
||||
|
||||
/* htsblk.warc_truncated / WARC-Truncated reason tokens (ISO 28500 sec 5.13).
|
||||
A cap-truncated body is still archived, tagged with why it was cut short. */
|
||||
#define WARC_TRUNC_NONE 0
|
||||
#define WARC_TRUNC_LENGTH 1 /* hit a size cap (-M mirror / -m per-file) */
|
||||
#define WARC_TRUNC_TIME 2 /* hit the mirror time cap (-E/--max-time) */
|
||||
#define WARC_TRUNC_DISCONNECT 3 /* connection dropped mid-body */
|
||||
|
||||
/* WARC-Truncated token for a warc_truncated code, or NULL for none. */
|
||||
const char *warc_truncated_reason(int code);
|
||||
|
||||
typedef struct warc_writer warc_writer;
|
||||
|
||||
/* Stash the raw request header block (bstr.buffer) on r for the later WARC
|
||||
request record; frees any prior stash. No-op when reqhdr is NULL. */
|
||||
void warc_stash_request(htsblk *r, const char *reqhdr);
|
||||
|
||||
/* Stash the raw response header block: deletehttp frees r.headers when the
|
||||
socket closes, before back_finalize, so keep a WARC-owned copy. */
|
||||
void warc_stash_response(htsblk *r, const char *resphdr);
|
||||
|
||||
/* Free both stashed header blocks (idempotent, NULL-safe). */
|
||||
void warc_free_request(htsblk *r);
|
||||
|
||||
/* Adopt the de-chunked compressed spool at tmpfile_path onto
|
||||
r->warc_rawpath/warc_rawsize (strdupt; frees any prior) so the WARC record
|
||||
stores the body verbatim. No-op leaving warc_rawpath NULL when tmpfile_path
|
||||
is empty or the spool is missing/empty (the record then stores the decoded
|
||||
in-memory/on-disk body instead). */
|
||||
void warc_adopt_rawspool(htsblk *r, const char *tmpfile_path);
|
||||
|
||||
/* Emit the request + response (or revisit) records for one finished
|
||||
transaction. Lazily opens the writer into opt->state.warc; a no-op (logged
|
||||
once) if the archive cannot be created. */
|
||||
void warc_write_backtransaction(httrackp *opt, lien_back *back);
|
||||
|
||||
/* Close and free the writer held in opt->state.warc, if any. */
|
||||
void warc_close_opt(httrackp *opt);
|
||||
|
||||
/* --- Direct writer API (used by the hooks above and the self-test). --- */
|
||||
|
||||
/* Create the archive at path (auto-named when path is WARC_AUTONAME), writing
|
||||
the warcinfo record. .warc.gz => one gzip member per record; .warc => raw.
|
||||
Returns NULL on failure. */
|
||||
warc_writer *warc_open(httrackp *opt, const char *path);
|
||||
|
||||
/* Flush, close and free the writer (NULL-safe). */
|
||||
void warc_close(warc_writer *w);
|
||||
|
||||
/* SURT-canonicalize url into out[outsz] (the CDXJ sort key). Returns 0 on
|
||||
success, -1 on error or truncation. Exposed for the -#test=warc-surt test. */
|
||||
int warc_surt(const char *url, char *out, size_t outsz);
|
||||
|
||||
/* Write one transaction's request + response (or revisit) records.
|
||||
target_uri: absolute URL fetched.
|
||||
ip: numeric peer IP, or NULL/"" to omit.
|
||||
req_hdr: exact request header block sent, or NULL to skip the request.
|
||||
resp_hdr: raw received response header block (status line + headers).
|
||||
body/body_len: decoded in-memory body, or NULL when on disk.
|
||||
body_path: file re-read for the body when body==NULL (may be NULL).
|
||||
is_update_unchanged: nonzero for a 304 server-not-modified revisit.
|
||||
truncated: a WARC_TRUNC_* reason to tag a cap-truncated body, else 0.
|
||||
The body is stored verbatim: Content-Encoding is kept and Content-Length set
|
||||
to body_len, so body/body_len must be the as-received (coded) bytes.
|
||||
Returns 0 on success, -1 on error. */
|
||||
int warc_write_transaction(warc_writer *w, const char *target_uri,
|
||||
const char *ip, const char *req_hdr,
|
||||
const char *resp_hdr, const char *body,
|
||||
size_t body_len, const char *body_path,
|
||||
int statuscode, int is_update_unchanged,
|
||||
int truncated);
|
||||
|
||||
/* Write one non-HTTP capture as a single WARC 'resource' record: the block is
|
||||
the raw payload (no HTTP envelope), Content-Type is the payload's own MIME.
|
||||
Used for ftp:// transfers. truncated is a WARC_TRUNC_* reason or 0.
|
||||
Returns 0 on success, -1 on error. */
|
||||
int warc_write_resource(warc_writer *w, const char *target_uri, const char *ip,
|
||||
const char *content_type, const char *body,
|
||||
size_t body_len, const char *body_path, int truncated);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -758,6 +758,9 @@ HTSEXT_API int hts_rename_utf8(const char *oldpath, const char *newpath);
|
||||
|
||||
HTSEXT_API int hts_mkdir_utf8(const char *pathname);
|
||||
|
||||
#define RMDIR hts_rmdir_utf8
|
||||
HTSEXT_API int hts_rmdir_utf8(const char *pathname);
|
||||
|
||||
#define UTIME(A, B) hts_utime_utf8(A, B)
|
||||
|
||||
typedef struct _utimbuf STRUCT_UTIMBUF;
|
||||
@@ -772,6 +775,7 @@ typedef struct stat STRUCT_STAT;
|
||||
#define UNLINK unlink
|
||||
#define RENAME rename
|
||||
#define MKDIR(F) mkdir(F, HTS_ACCESS_FOLDER)
|
||||
#define RMDIR rmdir
|
||||
|
||||
typedef struct utimbuf STRUCT_UTIMBUF;
|
||||
|
||||
|
||||
@@ -135,6 +135,7 @@
|
||||
<ClCompile Include="htswizard.c" />
|
||||
<ClCompile Include="htswrap.c" />
|
||||
<ClCompile Include="htszlib.c" />
|
||||
<ClCompile Include="htswarc.c" />
|
||||
<ClCompile Include="md5.c" />
|
||||
<ClCompile Include="minizip\ioapi.c" />
|
||||
<ClCompile Include="minizip\iowin32.c" />
|
||||
|
||||
11
tests/01_engine-direnum.test
Executable file
11
tests/01_engine-direnum.test
Executable file
@@ -0,0 +1,11 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# Drives -#test=direnum: enumerate a long+non-ASCII directory through the
|
||||
# opendir/readdir wrappers, checking each child round-trips as UTF-8 (#133,#630).
|
||||
dir=$(mktemp -d)
|
||||
trap 'rm -rf "$dir"' EXIT
|
||||
|
||||
httrack -O /dev/null -#test=direnum "$dir" | grep -q "direnum:.*OK"
|
||||
@@ -8,9 +8,14 @@ set -euo pipefail
|
||||
dir=$(mktemp -d)
|
||||
trap 'rm -rf "$dir"' EXIT
|
||||
|
||||
out=$(httrack -#test=fsize "$dir")
|
||||
rc=0
|
||||
out=$(httrack -#test=fsize "$dir") || rc=$?
|
||||
echo "$out"
|
||||
|
||||
# 77 = platform can't host the 5GB probe (EFBIG); skip, don't fail.
|
||||
[ "$rc" = 77 ] && exit 77
|
||||
[ "$rc" = 0 ] || exit "$rc"
|
||||
|
||||
want="fsize: width=8,8 size=5368709120,5368709120 psize=5368709120 absent=-1"
|
||||
test "$out" == "$want" || {
|
||||
echo "FAIL: unexpected size report (want '$want')"
|
||||
|
||||
11
tests/01_engine-longpath-io.test
Normal file
11
tests/01_engine-longpath-io.test
Normal file
@@ -0,0 +1,11 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# Drives -#test=longpath: a >MAX_PATH round trip exercising hts_pathToUCS2's
|
||||
# \\?\ prefixing on Windows (#133); a positive control on POSIX.
|
||||
dir=$(mktemp -d)
|
||||
trap 'rm -rf "$dir"' EXIT
|
||||
|
||||
httrack -O /dev/null -#test=longpath "$dir" | grep -q "longpath:.*OK"
|
||||
12
tests/01_engine-mirror-io.test
Executable file
12
tests/01_engine-mirror-io.test
Executable file
@@ -0,0 +1,12 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# Drives -#test=mirrorio: rounds a file through a path that is both long
|
||||
# (>MAX_PATH) and non-ASCII, exercising the mirror I/O wrappers the engine's
|
||||
# raw file ops now route to on Windows (#133, #630). Positive control on POSIX.
|
||||
dir=$(mktemp -d)
|
||||
trap 'rm -rf "$dir"' EXIT
|
||||
|
||||
httrack -O /dev/null -#test=mirrorio "$dir" | grep -q "mirrorio:.*OK"
|
||||
@@ -149,10 +149,19 @@ full '/page.php?id=3&sid=42' 'text/html' '/dev/null/www.example.com/page475b.htm
|
||||
strip=sid 'prior=www.example.com|/page.php?id=4|/dev/null/www.example.com/PAGE-PRIOR.html'
|
||||
|
||||
# Hostile fils stay rooted under the mirror: ../ (raw or %2e-encoded) drops out,
|
||||
# control characters become spaces, oversized names cap at 210 chars (the cap
|
||||
# can chop the extension off entirely).
|
||||
# control characters become spaces.
|
||||
full '/../../etc/passwd' 'text/html' '/dev/null/www.example.com///etc/passwd.html'
|
||||
full '/%2e%2e/%2e%2e/etc/passwd' 'text/html' '/dev/null/www.example.com///etc/passwd.html'
|
||||
full '/x.php' 'application/pdf' '/dev/null/www.example.com///evil.exe' 'cdispo=../../evil.exe'
|
||||
name $'/evil\rname\t.php' 'text/html' 'evil name .html'
|
||||
name "/$(printf 'a%.0s' {1..300}).php" 'text/html' "$(printf 'a%.0s' {1..210})"
|
||||
|
||||
# #133: oversized names are capped only on Windows (MAX_PATH, chopping the
|
||||
# extension); elsewhere the platform PATH_MAX leaves a 300-char name whole.
|
||||
case "$(uname -s 2>/dev/null)" in
|
||||
MINGW* | MSYS* | CYGWIN*)
|
||||
name "/$(printf 'a%.0s' {1..300}).php" 'text/html' "$(printf 'a%.0s' {1..210})"
|
||||
;;
|
||||
*)
|
||||
name "/$(printf 'a%.0s' {1..300}).php" 'text/html' "$(printf 'a%.0s' {1..300}).html"
|
||||
;;
|
||||
esac
|
||||
|
||||
8
tests/01_engine-warc-surt.test
Executable file
8
tests/01_engine-warc-surt.test
Executable file
@@ -0,0 +1,8 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# SURT canonicalization of the CDXJ sort key (--warc-cdx). Pure string work,
|
||||
# so it runs under the MSan-instrumented 01_engine glob.
|
||||
httrack -O /dev/null -#test=warc-surt | grep -q "warc-surt: OK"
|
||||
19
tests/01_zlib-warc-cdx.test
Executable file
19
tests/01_zlib-warc-cdx.test
Executable file
@@ -0,0 +1,19 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# --warc-cdx CDXJ index over synthetic transactions: sorted, one line per
|
||||
# response/revisit/resource, each offset/length inflates to the right member.
|
||||
|
||||
httrack_bin=$(cd "$(dirname "$(command -v httrack)")" && pwd)/httrack
|
||||
|
||||
scratch=$(mktemp -d)
|
||||
trap 'rm -rf "$scratch"' EXIT
|
||||
|
||||
out=$("$httrack_bin" -O /dev/null -#test=warc-cdx "$scratch/")
|
||||
echo "$out"
|
||||
case "$out" in
|
||||
*": OK") ;;
|
||||
*) exit 1 ;;
|
||||
esac
|
||||
26
tests/01_zlib-warc-wacz.test
Executable file
26
tests/01_zlib-warc-wacz.test
Executable file
@@ -0,0 +1,26 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# --wacz packaging over synthetic transactions: the WACZ unzips in-process to
|
||||
# the fixed layout, every entry is ZIP STORE, each datapackage sha256 recomputes
|
||||
# from the stored bytes, and the digest chains datapackage.json. WACZ needs the
|
||||
# OpenSSL SHA-256 digests, so the self-test is absent on non-OpenSSL builds.
|
||||
|
||||
httrack_bin=$(cd "$(dirname "$(command -v httrack)")" && pwd)/httrack
|
||||
|
||||
if ! "$httrack_bin" -#test 2>&1 | grep -q '^ warc-wacz'; then
|
||||
echo "warc-wacz self-test unavailable (build without OpenSSL); skipping"
|
||||
exit 77
|
||||
fi
|
||||
|
||||
scratch=$(mktemp -d)
|
||||
trap 'rm -rf "$scratch"' EXIT
|
||||
|
||||
out=$("$httrack_bin" -O /dev/null -#test=warc-wacz "$scratch/")
|
||||
echo "$out"
|
||||
case "$out" in
|
||||
*": OK") ;;
|
||||
*) exit 1 ;;
|
||||
esac
|
||||
22
tests/01_zlib-warc.test
Executable file
22
tests/01_zlib-warc.test
Executable file
@@ -0,0 +1,22 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# WARC/1.1 writer self-test: framing, Content-Length, gzip members, round-trip,
|
||||
# revisit dedup, plus v1.1 WARC-Truncated, ftp resource records, and
|
||||
# --warc-max-size rotation, over synthetic transactions.
|
||||
|
||||
httrack_bin=$(cd "$(dirname "$(command -v httrack)")" && pwd)/httrack
|
||||
|
||||
scratch=$(mktemp -d)
|
||||
trap 'rm -rf "$scratch"' EXIT
|
||||
|
||||
for t in warc warc-trunc warc-ftp warc-rotate warc-verbatim; do
|
||||
out=$("$httrack_bin" -O /dev/null "-#test=$t" "$scratch/")
|
||||
echo "$out"
|
||||
case "$out" in
|
||||
*": OK") ;;
|
||||
*) exit 1 ;;
|
||||
esac
|
||||
done
|
||||
@@ -3,30 +3,68 @@
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# #623: url_savename enforces the 236-char path ceiling by cutting the tail of
|
||||
# the last segment, where the mandatory ".delayed" placeholder marker lives. A
|
||||
# cut marker fails IS_DELAYED_EXT, so back_delayed_rename never renames the file
|
||||
# to its final name and the download is lost. The marker must survive the cut.
|
||||
|
||||
# statuscode=302 status=-1 = a redirect answer still downloading: no type is
|
||||
# resolved, so the name gets a ".<id>.delayed" placeholder (see 01_engine-savename).
|
||||
CEIL=236
|
||||
# #623: url_savename shortens an over-ceiling path by cutting the tail of the
|
||||
# last segment, where the mandatory ".<id>.delayed" placeholder lives. A cut
|
||||
# marker fails IS_DELAYED_EXT, so back_delayed_rename never renames the file and
|
||||
# the download is lost. The marker must survive the cut.
|
||||
#
|
||||
# The ceiling is the Windows MAX_PATH on Windows but the (far larger) save
|
||||
# buffer elsewhere (#133). Each CLI arg is capped at HTS_CDLMAXSIZE (1024), so
|
||||
# on POSIX we lengthen the output dir as well to overrun the ceiling.
|
||||
|
||||
httrack_bin=$(cd "$(dirname "$(command -v httrack)")" && pwd)/httrack
|
||||
scratch=$(mktemp -d)
|
||||
trap 'rm -rf "$scratch"' EXIT
|
||||
cd "$scratch"
|
||||
|
||||
hexseg=$(printf 'a1b2c3d4e5f60718%.0s' {1..8}) # 128 hex chars
|
||||
|
||||
# engine ceiling is the Windows MAX_PATH on Windows, the save buffer elsewhere
|
||||
case "$(uname -s 2>/dev/null)" in
|
||||
MINGW* | MSYS* | CYGWIN*)
|
||||
ceil=236
|
||||
outdir=/dev/null
|
||||
deep="/d1$(printf 'a%.0s' {1..40})/d2$(printf 'b%.0s' {1..40})"
|
||||
deep="$deep/d3$(printf 'c%.0s' {1..40})/d4$(printf 'd%.0s' {1..40})"
|
||||
long="$deep/$(printf 'z%.0s' {1..90})"
|
||||
hexpath="$deep/$hexseg"
|
||||
;;
|
||||
*)
|
||||
# overrun the save-buffer ceiling with a long dir + long name
|
||||
ceil=$((1024 * 2 - 64)) # HTS_URLMAXSIZE*2 - HTS_PATH_TAIL_RESERVE
|
||||
outdir="$scratch/$(printf 'D%.0s' $(seq 1 $((1000 - ${#scratch}))))"
|
||||
long="/d1/$(printf 'z%.0s' $(seq 1 985))"
|
||||
hexpath="/d1/$(printf 'y%.0s' $(seq 1 850))/$hexseg"
|
||||
;;
|
||||
esac
|
||||
|
||||
run() {
|
||||
"$httrack_bin" -O /dev/null -#test=savename "$@" | sed -n 's/^savename: //p'
|
||||
"$httrack_bin" -O "$outdir" -#test=savename "$@" | sed -n 's/^savename: //p'
|
||||
}
|
||||
|
||||
# A deep path ending in a long segment that overruns the ceiling.
|
||||
deep="/d1$(printf 'a%.0s' {1..40})/d2$(printf 'b%.0s' {1..40})"
|
||||
deep="$deep/d3$(printf 'c%.0s' {1..40})/d4$(printf 'd%.0s' {1..40})"
|
||||
long="$deep/$(printf 'z%.0s' {1..90})"
|
||||
# untruncated "outdir/host/name" is longer than the ceiling, so truncation must
|
||||
# fire; out then sits at or under it. Together these keep the marker check below
|
||||
# non-vacuous.
|
||||
check_truncated() {
|
||||
local out=$1 fil=$2
|
||||
local untrunc=$((${#outdir} + 1 + 15 + ${#fil})) # host = www.example.com
|
||||
test "$untrunc" -gt "$ceil" ||
|
||||
{
|
||||
echo "FAIL: input too short to truncate ($untrunc <= $ceil)"
|
||||
exit 1
|
||||
}
|
||||
test "${#out}" -le "$ceil" ||
|
||||
{
|
||||
echo "FAIL: truncated name ${#out} > $ceil ceiling: '$out'"
|
||||
exit 1
|
||||
}
|
||||
}
|
||||
|
||||
out="$(run "$long" text/html statuscode=302 status=-1)"
|
||||
# statuscode=302 status=-1 = a redirect still downloading: no type is resolved,
|
||||
# so the name gets a ".<id>.delayed" placeholder (see 01_engine-savename).
|
||||
long_delayed="$long.ext"
|
||||
out="$(run "$long_delayed" text/html statuscode=302 status=-1)"
|
||||
check_truncated "$out" "$long_delayed"
|
||||
case "$out" in
|
||||
*.delayed) ;;
|
||||
*)
|
||||
@@ -34,16 +72,11 @@ case "$out" in
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
test "${#out}" -le "$CEIL" || {
|
||||
echo "FAIL: truncated name ${#out} > $CEIL ceiling: '$out'"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# #133-style hashed name: an all-hex last segment must keep the marker too (the
|
||||
# ".<id>." tag is not mistaken for part of the hash).
|
||||
hexseg=$(printf 'a1b2c3d4e5f60718%.0s' {1..8})
|
||||
hexpath="/d1$(printf 'x%.0s' {1..30})/d2$(printf 'y%.0s' {1..30})/$hexseg"
|
||||
out="$(run "$hexpath" text/html statuscode=302 status=-1)"
|
||||
check_truncated "$out" "$hexpath"
|
||||
case "$out" in
|
||||
*.delayed) ;;
|
||||
*)
|
||||
@@ -51,17 +84,10 @@ case "$out" in
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
test "${#out}" -le "$CEIL" || {
|
||||
echo "FAIL: hashed name ${#out} > $CEIL ceiling: '$out'"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# A non-delayed name of the same shape still truncates, with no marker to keep.
|
||||
out="$(run "$long.html" text/html)"
|
||||
test "${#out}" -le "$CEIL" || {
|
||||
echo "FAIL: non-delayed name ${#out} > $CEIL ceiling: '$out'"
|
||||
exit 1
|
||||
}
|
||||
out="$(run "$long_delayed.html" text/html)"
|
||||
check_truncated "$out" "$long_delayed.html"
|
||||
case "$out" in
|
||||
*.delayed)
|
||||
echo "FAIL: non-delayed name grew a .delayed marker: '$out'"
|
||||
|
||||
21
tests/73_local-warc.test
Executable file
21
tests/73_local-warc.test
Executable file
@@ -0,0 +1,21 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# A --warc crawl writes a standards-conformant WARC/1.1 archive, and an
|
||||
# --update re-crawl of an unchanged (all-304) site emits revisit records.
|
||||
# The stdlib validator (no warcio) is the real gate: it byte-compares the
|
||||
# fresh page.html response body against what the server served, asserts the
|
||||
# encoding headers were stripped and the payload digest matches, then checks
|
||||
# the update pass turned the unchanged assets into revisits (no full response).
|
||||
|
||||
set -eu
|
||||
|
||||
: "${top_srcdir:=..}"
|
||||
|
||||
# page.html body served by tests/local-server.py (route_mini304_page).
|
||||
export WARC_VALIDATE_BODY="page.html=3c68746d6c3e3c626f64793e74696e7920636163686561626c6520706167653c2f626f64793e3c2f68746d6c3e0a"
|
||||
export WARC_VALIDATE_NORESP="index.html page.html"
|
||||
|
||||
bash "$top_srcdir/tests/local-crawl.sh" --errors 0 --rerun --warc-validate \
|
||||
--log-found 'no files updated' \
|
||||
--found 'mini304/index.html' --found 'mini304/page.html' \
|
||||
httrack 'BASEURL/mini304/index.html' --warc-file warc-out
|
||||
22
tests/74_local-warc-verbatim.test
Executable file
22
tests/74_local-warc-verbatim.test
Executable file
@@ -0,0 +1,22 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# A --warc crawl stores compressed bodies verbatim (the default): the response
|
||||
# record keeps Content-Encoding: gzip and its stored bytes inflate back to the
|
||||
# served page. The validator's --verbatim mode is the differential gate:
|
||||
# inflate(stored) must equal the decoded body the server compressed.
|
||||
#
|
||||
# Two fixtures cover both adoption branches of back_finalize: page.html (text/html)
|
||||
# takes the in-memory branch; data.bin (application/octet-stream) is streamed to
|
||||
# disk, so it exercises the is_write direct-to-disk spool adoption.
|
||||
|
||||
set -eu
|
||||
|
||||
: "${top_srcdir:=..}"
|
||||
|
||||
# decoded bodies served (gzip-coded) by route_warcgz_page and route_warcgz_data.
|
||||
export WARC_VALIDATE_BODY="warcgz/page.html=3c68746d6c3e3c626f64793e766572626174696d20677a6970207061676520666f72205741524320737472617465677920413c2f626f64793e3c2f68746d6c3e0a warcgz/data.bin=766572626174696d20677a6970206f637465742d73747265616d20626f647920666f722074686520574152432069735f777269746520706174680a"
|
||||
export WARC_VALIDATE_VERBATIM=1
|
||||
|
||||
bash "$top_srcdir/tests/local-crawl.sh" --errors 0 --warc-validate \
|
||||
--found 'warcgz/index.html' --found 'warcgz/page.html' --found 'warcgz/data.bin' \
|
||||
httrack 'BASEURL/warcgz/index.html' --warc-file warc-out
|
||||
15
tests/74_local-warc-wacz.test
Executable file
15
tests/74_local-warc-wacz.test
Executable file
@@ -0,0 +1,15 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# A --wacz crawl packages the WARC archive, its CDXJ index and a generated
|
||||
# pages.jsonl into a single WACZ file at crawl end. The stdlib validator is the
|
||||
# real gate: STORE-mode entries, the fixed layout, recomputed sha256 digests and
|
||||
# the datapackage-digest chain (plus py-wacz/pywb when importable). The crawl
|
||||
# skips cleanly on a build without OpenSSL (no conformant SHA-256 -> no package).
|
||||
|
||||
set -eu
|
||||
|
||||
: "${top_srcdir:=..}"
|
||||
|
||||
bash "$top_srcdir/tests/local-crawl.sh" --errors 0 --wacz-validate \
|
||||
--found 'mini304/index.html' --found 'mini304/page.html' \
|
||||
httrack 'BASEURL/mini304/index.html' --warc-file warc-out --wacz
|
||||
64
tests/75_engine-longpath-posix.test
Executable file
64
tests/75_engine-longpath-posix.test
Executable file
@@ -0,0 +1,64 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# #133: the save-path ceiling used to be the Windows MAX_PATH (236 usable) on
|
||||
# every platform, needlessly hashing long names on Linux/macOS/Android. Off
|
||||
# Windows a name well under the platform PATH_MAX must now be kept whole.
|
||||
|
||||
httrack_bin=$(cd "$(dirname "$(command -v httrack)")" && pwd)/httrack
|
||||
scratch=$(mktemp -d)
|
||||
trap 'rm -rf "$scratch"' EXIT
|
||||
cd "$scratch"
|
||||
|
||||
# ~300 chars: over the old 236 ceiling, well under any POSIX PATH_MAX. The
|
||||
# distinctive 90-char last segment lets us tell "kept whole" from "hashed".
|
||||
seg="$(printf 'Z%.0s' {1..90})"
|
||||
deep="/a/$(printf 'b%.0s' {1..100})/c/$(printf 'd%.0s' {1..100})"
|
||||
out="$("$httrack_bin" -O /dev/null -#test=savename "$deep/$seg.html" text/html |
|
||||
sed -n 's/^savename: //p')"
|
||||
|
||||
case "$(uname -s 2>/dev/null)" in
|
||||
MINGW* | MSYS* | CYGWIN*)
|
||||
# Windows still caps at MAX_PATH: the distinctive tail is cut
|
||||
case "$out" in
|
||||
*"$seg"*)
|
||||
echo "FAIL: Windows should have shortened the long name: '$out'"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
;;
|
||||
*)
|
||||
# POSIX: the full name survives, no hashing
|
||||
case "$out" in
|
||||
*"$seg"*) ;;
|
||||
*)
|
||||
echo "FAIL: long POSIX name was truncated: '$out'"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
test "${#out}" -gt 236 ||
|
||||
{
|
||||
echo "FAIL: POSIX name not longer than the old ceiling: '$out'"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# byte-safety: a multibyte name whose codepoint count is under the ceiling
|
||||
# but whose bytes plus a long -O dir overflow the save buffer must be cut to
|
||||
# fit, not abort the crawl (SIGABRT before the byte-cut guard).
|
||||
bigdir="/x/$(printf 'D%.0s' $(seq 1 1015))" # ~1020 bytes
|
||||
cjk="/$(printf '\xe4\xb8\xad%.0s' $(seq 1 339))" # 1018 bytes, 339 codepoints
|
||||
if ! mb="$("$httrack_bin" -O "$bigdir" -#test=savename "$cjk" text/html 2>/dev/null)"; then
|
||||
echo "FAIL: aborted on a multibyte over-buffer name"
|
||||
exit 1
|
||||
fi
|
||||
mb="${mb#savename: }"
|
||||
if [ -z "$mb" ] || [ "${#mb}" -ge 2048 ]; then
|
||||
echo "FAIL: multibyte name not bounded (len=${#mb})"
|
||||
exit 1
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
|
||||
echo "longpath-posix OK"
|
||||
@@ -3,7 +3,7 @@
|
||||
# silently drop it from the dist tarball and break "make distcheck".
|
||||
EXTRA_DIST = $(TESTS) crawl-test.sh run-all-tests.sh check-network.sh \
|
||||
proxy-https-server.py socks5-server.py proxy-connect-server.py \
|
||||
proxytestlib.py tls-stall-server.py \
|
||||
proxytestlib.py tls-stall-server.py warc-validate.py wacz-validate.py \
|
||||
local-crawl.sh local-server.py testlib.sh server.crt server.key \
|
||||
server-root/simple/basic.html server-root/simple/link.html \
|
||||
server-root/stripquery/index.html server-root/stripquery/a.html \
|
||||
@@ -64,6 +64,9 @@ TESTS = \
|
||||
01_engine-expandhome.test \
|
||||
01_engine-fsize.test \
|
||||
01_engine-redirect.test \
|
||||
01_engine-longpath-io.test \
|
||||
01_engine-mirror-io.test \
|
||||
01_engine-direnum.test \
|
||||
01_engine-relative.test \
|
||||
01_engine-robots.test \
|
||||
01_engine-savename.test \
|
||||
@@ -78,8 +81,12 @@ TESTS = \
|
||||
01_engine-unescape-bounds.test \
|
||||
01_engine-useragent.test \
|
||||
01_engine-version-macros.test \
|
||||
01_engine-warc-surt.test \
|
||||
01_engine-xfread.test \
|
||||
01_zlib-acceptencoding.test \
|
||||
01_zlib-warc.test \
|
||||
01_zlib-warc-cdx.test \
|
||||
01_zlib-warc-wacz.test \
|
||||
01_zlib-contentcodings.test \
|
||||
01_zlib-cache.test \
|
||||
01_zlib-cache-corrupt.test \
|
||||
@@ -155,6 +162,10 @@ TESTS = \
|
||||
68_webhttrack-outdir-charset.test \
|
||||
69_local-intl-logdir.test \
|
||||
71_local-crange-repaircache.test \
|
||||
72_watchdog-crawl.test
|
||||
72_watchdog-crawl.test \
|
||||
73_local-warc.test \
|
||||
74_local-warc-wacz.test \
|
||||
74_local-warc-verbatim.test \
|
||||
75_engine-longpath-posix.test
|
||||
|
||||
CLEANFILES = check-network_sh.cache
|
||||
|
||||
@@ -47,6 +47,8 @@ key="${testdir}/server.key"
|
||||
|
||||
tls=
|
||||
verbose=
|
||||
warc_validate=
|
||||
wacz_validate=
|
||||
html_subdir=
|
||||
outdir_intl=
|
||||
rerun=
|
||||
@@ -114,6 +116,10 @@ while test "$pos" -lt "$nargs"; do
|
||||
--debug) verbose=1 ;;
|
||||
--rerun) rerun=1 ;; # run httrack a second time (update pass) before auditing
|
||||
--rerun-dead) rerun_dead=1 ;; # re-run with the server stopped (cache rollback)
|
||||
# validate the produced .warc.gz (see the validation block near the end)
|
||||
--warc-validate) warc_validate=1 ;;
|
||||
# validate the produced .wacz package (stdlib, plus py-wacz/pywb if present)
|
||||
--wacz-validate) wacz_validate=1 ;;
|
||||
--no-purge)
|
||||
nopurge=1
|
||||
audit+=("--no-purge")
|
||||
@@ -259,6 +265,13 @@ test "$crawlres" -eq 0 || ! result "httrack exited $crawlres" || {
|
||||
result "OK"
|
||||
grep -iE "^[0-9:]*[[:space:]]Error:" "${logroot}/hts-log.txt" >&2
|
||||
|
||||
# Snapshot the first-pass WARC before an update pass overwrites it: the fresh
|
||||
# crawl carries the full response bodies, the update pass only revisits.
|
||||
if test -n "$warc_validate"; then
|
||||
w1=$(find "$mirrorroot" -maxdepth 2 -name '*.warc.gz' 2>/dev/null | sort | tail -n1)
|
||||
test -z "$w1" || cp "$w1" "${tmpdir}/warc-pass1.gz"
|
||||
fi
|
||||
|
||||
# --- optional second pass: re-mirror into the same dir (cache/update path) ----
|
||||
if test -n "$rerun"; then
|
||||
info "re-running httrack (update pass)"
|
||||
@@ -350,6 +363,65 @@ done
|
||||
test -n "$hostroot" || die "could not find host root under $out"
|
||||
debug "host root: $hostroot"
|
||||
|
||||
# --- optional WARC validation (stdlib validator, no warcio) ------------------
|
||||
# WARC_VALIDATE_BODY="URLSUB=HEX" byte-checks a fresh-crawl response body;
|
||||
# WARC_VALIDATE_NORESP="URLSUB..." asserts those assets are revisits post-update.
|
||||
if test -n "$warc_validate"; then
|
||||
validator=$(nativepath "${testdir}/warc-validate.py")
|
||||
warc=$(find "$mirrorroot" -maxdepth 2 \( -name '*.warc.gz' -o -name '*.warc' \) 2>/dev/null | sort | tail -n1)
|
||||
test -n "$warc" || die "no WARC file produced under $mirrorroot"
|
||||
|
||||
# Fresh-crawl file (snapshot if an update pass overwrote it): full responses.
|
||||
fresh="${tmpdir}/warc-pass1.gz"
|
||||
test -f "$fresh" || fresh="$warc"
|
||||
declare -a bodyargs=()
|
||||
# WARC_VALIDATE_BODY holds one or more whitespace-separated SUB=HEX specs.
|
||||
for spec in ${WARC_VALIDATE_BODY:-}; do
|
||||
bodyargs+=(--expect-body-hex "$spec")
|
||||
done
|
||||
# compressed asset: assert the stored (verbatim) body inflates to the served
|
||||
# body and keeps Content-Encoding, instead of expecting a decoded body.
|
||||
test -n "${WARC_VALIDATE_VERBATIM:-}" && bodyargs+=(--verbatim)
|
||||
info "validating fresh WARC (response bodies)"
|
||||
"$python" "$validator" "$(nativepath "$fresh")" "${bodyargs[@]}" >&2 ||
|
||||
die "fresh WARC validation failed"
|
||||
result "OK"
|
||||
|
||||
# Final file: after an update pass the unchanged assets must be revisits.
|
||||
if test -n "$rerun"; then
|
||||
declare -a revargs=(--expect-revisit)
|
||||
for sub in ${WARC_VALIDATE_NORESP:-}; do
|
||||
revargs+=(--no-response-for "$sub")
|
||||
done
|
||||
info "validating update WARC (revisits)"
|
||||
"$python" "$validator" "$(nativepath "$warc")" "${revargs[@]}" >&2 ||
|
||||
die "update WARC validation failed"
|
||||
result "OK"
|
||||
fi
|
||||
|
||||
if command -v warcio >/dev/null 2>&1; then
|
||||
info "warcio check (optional)"
|
||||
if warcio check -v "$warc" >&2; then result "OK"; else die "warcio check failed"; fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- optional WACZ validation (--wacz) --------------------------------------
|
||||
if test -n "$wacz_validate"; then
|
||||
wacz=$(find "$mirrorroot" -maxdepth 2 -name '*.wacz' 2>/dev/null | sort | tail -n1)
|
||||
if test -z "$wacz"; then
|
||||
# No package: only acceptable when the build lacks OpenSSL (SHA-256).
|
||||
if grep -aqi "WACZ requires an OpenSSL" "${logroot}/hts-log.txt"; then
|
||||
info "no .wacz produced (build without OpenSSL); skipping"
|
||||
exit 77
|
||||
fi
|
||||
die "no .wacz file produced under $mirrorroot"
|
||||
fi
|
||||
validator=$(nativepath "${testdir}/wacz-validate.py")
|
||||
info "validating WACZ package"
|
||||
"$python" "$validator" "$(nativepath "$wacz")" >&2 || die "WACZ validation failed"
|
||||
result "OK"
|
||||
fi
|
||||
|
||||
# No crawl, even a cancelled one, may leave engine temporaries: .delayed (#107,
|
||||
# #483), or the .z/.u content-coding temps (#557).
|
||||
info "checking for leftover engine temporaries"
|
||||
|
||||
@@ -635,6 +635,34 @@ class Handler(SimpleHTTPRequestHandler):
|
||||
extra_headers=[("Content-Encoding", "gzip")],
|
||||
)
|
||||
|
||||
# A gzip-coded HTML page whose decoded body is known, for the verbatim-WARC
|
||||
# differential (stored compressed bytes must inflate to this).
|
||||
WARCGZ_BODY = b"<html><body>verbatim gzip page for WARC strategy A</body></html>\n"
|
||||
|
||||
# A NON-html gzip-coded asset: HTTrack streams it straight to disk, so the
|
||||
# verbatim spool adoption runs on the is_write (direct-to-disk) branch of
|
||||
# back_finalize, not the in-memory branch route_warcgz_page exercises.
|
||||
WARCGZ_BIN_BODY = b"verbatim gzip octet-stream body for the WARC is_write path\n"
|
||||
|
||||
def route_warcgz_index(self):
|
||||
self.send_html(
|
||||
'\t<a href="page.html">page</a>\n' '\t<a href="data.bin">data</a>\n'
|
||||
)
|
||||
|
||||
def route_warcgz_page(self):
|
||||
self.send_raw(
|
||||
gzip.compress(self.WARCGZ_BODY),
|
||||
"text/html",
|
||||
extra_headers=[("Content-Encoding", "gzip")],
|
||||
)
|
||||
|
||||
def route_warcgz_data(self):
|
||||
self.send_raw(
|
||||
gzip.compress(self.WARCGZ_BIN_BODY),
|
||||
"application/octet-stream",
|
||||
extra_headers=[("Content-Encoding", "gzip")],
|
||||
)
|
||||
|
||||
# --- content codings ---------------------------------------------------
|
||||
# Canned br/zstd bodies (no brotli/zstd module in the stdlib): both decode
|
||||
# to CODEC_BODY. Regenerate with the brotli/zstd CLIs over that string.
|
||||
@@ -1542,6 +1570,9 @@ class Handler(SimpleHTTPRequestHandler):
|
||||
"/gated/index.php": route_gated_index,
|
||||
"/gated/secret.php": route_gated_secret,
|
||||
"/robots.txt": route_robots,
|
||||
"/warcgz/index.html": route_warcgz_index,
|
||||
"/warcgz/page.html": route_warcgz_page,
|
||||
"/warcgz/data.bin": route_warcgz_data,
|
||||
"/codec/index.html": route_codec_index,
|
||||
"/codec/br.html": route_codec_br,
|
||||
"/codec/zstd.html": route_codec_zstd,
|
||||
|
||||
95
tests/wacz-validate.py
Executable file
95
tests/wacz-validate.py
Executable file
@@ -0,0 +1,95 @@
|
||||
#!/usr/bin/env python3
|
||||
# Validate a WACZ package with the stdlib only (no py-wacz needed): every entry
|
||||
# is ZIP STORE, the fixed layout is present, each datapackage resource hash and
|
||||
# size recomputes, the digest chains datapackage.json, and pages.jsonl carries
|
||||
# the json-pages-1.0 header. If py-wacz (`wacz validate`) is importable it runs
|
||||
# too; both gates must pass. Exit 0 = valid, nonzero = invalid.
|
||||
import sys
|
||||
import json
|
||||
import zipfile
|
||||
import hashlib
|
||||
|
||||
|
||||
def fail(msg):
|
||||
print("wacz-validate: FAIL: %s" % msg, file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
fail("usage: wacz-validate.py FILE.wacz")
|
||||
path = sys.argv[1]
|
||||
z = zipfile.ZipFile(path)
|
||||
names = z.namelist()
|
||||
|
||||
for info in z.infolist():
|
||||
if info.compress_type != zipfile.ZIP_STORED:
|
||||
fail("%s is not STORE mode (%d)" % (info.filename, info.compress_type))
|
||||
|
||||
need_arc = any(
|
||||
n.startswith("archive/") and n.endswith((".warc.gz", ".warc")) for n in names
|
||||
)
|
||||
for req, ok in (
|
||||
("archive/*.warc.gz", need_arc),
|
||||
("indexes/index.cdx", "indexes/index.cdx" in names),
|
||||
("pages/pages.jsonl", "pages/pages.jsonl" in names),
|
||||
("datapackage.json", "datapackage.json" in names),
|
||||
("datapackage-digest.json", "datapackage-digest.json" in names),
|
||||
):
|
||||
if not ok:
|
||||
fail("missing %s (entries: %s)" % (req, names))
|
||||
|
||||
lines = [ln for ln in z.read("pages/pages.jsonl").split(b"\n") if ln.strip()]
|
||||
if not lines or json.loads(lines[0]).get("format") != "json-pages-1.0":
|
||||
fail("pages.jsonl header is not json-pages-1.0: %r" % lines[:1])
|
||||
body = [json.loads(ln) for ln in lines[1:]]
|
||||
if not body:
|
||||
fail("pages.jsonl has no page rows after the header")
|
||||
for row in body:
|
||||
if "url" not in row or "ts" not in row:
|
||||
fail("pages.jsonl row missing url/ts: %r" % row)
|
||||
|
||||
dp = json.loads(z.read("datapackage.json"))
|
||||
if dp.get("profile") != "data-package":
|
||||
fail("profile != data-package: %r" % dp.get("profile"))
|
||||
if dp.get("wacz_version") != "1.1.1":
|
||||
fail("wacz_version != 1.1.1: %r" % dp.get("wacz_version"))
|
||||
resources = dp.get("resources", [])
|
||||
if not resources:
|
||||
fail("datapackage has no resources")
|
||||
for r in resources:
|
||||
data = z.read(r["path"])
|
||||
h = "sha256:" + hashlib.sha256(data).hexdigest()
|
||||
if h != r["hash"]:
|
||||
fail("%s hash %s != %s" % (r["path"], h, r["hash"]))
|
||||
if len(data) != r["bytes"]:
|
||||
fail("%s bytes %d != %d" % (r["path"], len(data), r["bytes"]))
|
||||
|
||||
dig = json.loads(z.read("datapackage-digest.json"))
|
||||
want = "sha256:" + hashlib.sha256(z.read("datapackage.json")).hexdigest()
|
||||
if dig.get("path") != "datapackage.json" or dig.get("hash") != want:
|
||||
fail("digest chain broken: %r" % dig)
|
||||
|
||||
print(
|
||||
"wacz-validate: OK (%d entries, %d resources, stdlib)"
|
||||
% (len(names), len(resources))
|
||||
)
|
||||
|
||||
# Optional stricter gate when py-wacz is present.
|
||||
try:
|
||||
import wacz # noqa: F401
|
||||
except Exception:
|
||||
return
|
||||
try:
|
||||
from wacz.main import main as wacz_main # type: ignore
|
||||
except Exception:
|
||||
return
|
||||
print("wacz-validate: running py-wacz validate")
|
||||
rc = wacz_main(["validate", "-f", path])
|
||||
if rc not in (0, None):
|
||||
fail("py-wacz validate returned %r" % rc)
|
||||
print("wacz-validate: py-wacz OK")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
186
tests/warc-validate.py
Executable file
186
tests/warc-validate.py
Executable file
@@ -0,0 +1,186 @@
|
||||
#!/usr/bin/env python3
|
||||
# Structural + semantic WARC/1.1 validator, Python stdlib only (no warcio):
|
||||
# walks the concatenated gzip members with zlib (gzip.decompress would fuse them
|
||||
# and lose per-record boundaries) and checks each record against the spec.
|
||||
#
|
||||
# Options:
|
||||
# --expect-revisit at least one revisit record must be present
|
||||
# --expect-body-hex SUB=HEX a response whose WARC-Target-URI contains SUB must
|
||||
# have an entity body byte-equal to bytes.fromhex(HEX),
|
||||
# no Content-Encoding/Transfer-Encoding header, and a
|
||||
# WARC-Payload-Digest matching sha1(body) when present
|
||||
# --no-response-for SUB the asset containing SUB must be a revisit: no
|
||||
# response may target it, and a revisit must
|
||||
# --verbatim compressed asset: --expect-body-hex instead keeps
|
||||
# Content-Encoding, checks the HTTP Content-Length is
|
||||
# the stored (compressed) length, asserts the stored
|
||||
# body inflates to HEX (the served plaintext), and
|
||||
# requires the payload digest when the file emits any
|
||||
import base64
|
||||
import hashlib
|
||||
import sys
|
||||
import zlib
|
||||
|
||||
|
||||
def records(data):
|
||||
"""Yield each record's decompressed bytes (one gzip member per record)."""
|
||||
if data[:2] != b"\x1f\x8b": # uncompressed .warc: split on the record magic
|
||||
parts = data.split(b"WARC/1.")
|
||||
for p in parts[1:]:
|
||||
yield b"WARC/1." + p
|
||||
return
|
||||
while data:
|
||||
d = zlib.decompressobj(zlib.MAX_WBITS | 16)
|
||||
block = d.decompress(data) + d.flush()
|
||||
yield block
|
||||
data = d.unused_data
|
||||
|
||||
|
||||
def field(header, name):
|
||||
for line in header.split(b"\r\n"):
|
||||
if line.lower().startswith(name.lower() + b":"):
|
||||
return line.split(b":", 1)[1].strip()
|
||||
return None
|
||||
|
||||
|
||||
def opt_values(argv, name):
|
||||
out = []
|
||||
for i, a in enumerate(argv):
|
||||
if a == name and i + 1 < len(argv):
|
||||
out.append(argv[i + 1])
|
||||
return out
|
||||
|
||||
|
||||
def check_body(rec, http_hdr, body, sub, want):
|
||||
if b"Content-Encoding" in http_hdr or b"Transfer-Encoding" in http_hdr:
|
||||
sys.exit("record for %s kept a content/transfer-encoding header" % sub)
|
||||
if body != want:
|
||||
sys.exit(
|
||||
"body mismatch for %s: got %d bytes, expected %d"
|
||||
% (sub, len(body), len(want))
|
||||
)
|
||||
pd = field(rec[: rec.find(b"\r\n\r\n")], b"WARC-Payload-Digest")
|
||||
if pd is not None and pd.startswith(b"sha1:"):
|
||||
want_b32 = base64.b32encode(hashlib.sha1(want).digest()).decode("ascii")
|
||||
if pd[5:].decode("ascii") != want_b32:
|
||||
sys.exit("WARC-Payload-Digest mismatch for %s" % sub)
|
||||
|
||||
|
||||
def check_body_verbatim(rec, http_hdr, body, sub, want, digests_emitted):
|
||||
"""Verbatim: the stored body is the coded octets, Content-Encoding is kept,
|
||||
the HTTP Content-Length equals the stored (compressed) length, inflating the
|
||||
body yields the served plaintext, and the payload digest is over the coded
|
||||
body. The differential: inflate(stored) == the body the server compressed."""
|
||||
if b"Content-Encoding" not in http_hdr:
|
||||
sys.exit("verbatim record for %s dropped Content-Encoding" % sub)
|
||||
if b"Transfer-Encoding" in http_hdr:
|
||||
sys.exit("verbatim record for %s kept Transfer-Encoding" % sub)
|
||||
hcl = field(http_hdr, b"Content-Length")
|
||||
if hcl is None or int(hcl) != len(body):
|
||||
sys.exit("verbatim record for %s: HTTP Content-Length != stored body" % sub)
|
||||
try:
|
||||
decoded = zlib.decompress(body, zlib.MAX_WBITS | 16)
|
||||
except Exception as exc:
|
||||
sys.exit("verbatim record for %s: body did not inflate: %s" % (sub, exc))
|
||||
if decoded != want:
|
||||
sys.exit(
|
||||
"verbatim decoded mismatch for %s: got %d bytes, expected %d"
|
||||
% (sub, len(decoded), len(want))
|
||||
)
|
||||
pd = field(rec[: rec.find(b"\r\n\r\n")], b"WARC-Payload-Digest")
|
||||
# Catch a regression that drops the digest on the verbatim path, but only
|
||||
# when this file emits digests at all (an OpenSSL build; none otherwise).
|
||||
if digests_emitted and pd is None:
|
||||
sys.exit("verbatim record for %s: missing WARC-Payload-Digest" % sub)
|
||||
if pd is not None and pd.startswith(b"sha1:"):
|
||||
want_b32 = base64.b32encode(hashlib.sha1(body).digest()).decode("ascii")
|
||||
if pd[5:].decode("ascii") != want_b32:
|
||||
sys.exit("WARC-Payload-Digest (compressed) mismatch for %s" % sub)
|
||||
|
||||
|
||||
def main():
|
||||
argv = sys.argv[1:]
|
||||
expect_revisit = "--expect-revisit" in argv
|
||||
verbatim = "--verbatim" in argv
|
||||
body_specs = [s.split("=", 1) for s in opt_values(argv, "--expect-body-hex")]
|
||||
no_resp = opt_values(argv, "--no-response-for")
|
||||
path = [a for a in argv if not a.startswith("--") and "=" not in a][0]
|
||||
data = open(path, "rb").read()
|
||||
|
||||
# digests are emitted only on an OpenSSL build; detect it once so --verbatim
|
||||
# can require the payload digest exactly when the file carries any.
|
||||
digests_emitted = any(
|
||||
b"WARC-Payload-Digest" in r[: r.find(b"\r\n\r\n")] for r in records(data)
|
||||
)
|
||||
|
||||
total = revisits = responses = infos = 0
|
||||
body_hits = {sub: False for sub, _ in body_specs}
|
||||
revisit_hits = {sub: False for sub in no_resp}
|
||||
for rec in records(data):
|
||||
total += 1
|
||||
if not rec.startswith(b"WARC/1."):
|
||||
sys.exit("record %d: bad magic %r" % (total, rec[:16]))
|
||||
sep = rec.find(b"\r\n\r\n")
|
||||
if sep < 0:
|
||||
sys.exit("record %d: no header terminator" % total)
|
||||
hdr_end = sep + 4
|
||||
header = rec[:sep]
|
||||
cl = field(header, b"Content-Length")
|
||||
if cl is None:
|
||||
sys.exit("record %d: no Content-Length" % total)
|
||||
block_len = int(cl)
|
||||
if hdr_end + block_len + 4 != len(rec):
|
||||
sys.exit(
|
||||
"record %d: Content-Length %d != block length %d"
|
||||
% (total, block_len, len(rec) - hdr_end - 4)
|
||||
)
|
||||
if rec[hdr_end + block_len :] != b"\r\n\r\n":
|
||||
sys.exit("record %d: missing \\r\\n\\r\\n trailer" % total)
|
||||
wtype = field(header, b"WARC-Type")
|
||||
uri = field(header, b"WARC-Target-URI") or b""
|
||||
if wtype == b"warcinfo":
|
||||
infos += 1
|
||||
elif wtype == b"response":
|
||||
responses += 1
|
||||
for sub in no_resp:
|
||||
if sub.encode() in uri:
|
||||
sys.exit("unexpected full response for %s (want revisit)" % sub)
|
||||
block = rec[hdr_end : hdr_end + block_len]
|
||||
bsep = block.find(b"\r\n\r\n")
|
||||
http_hdr, body = block[:bsep], block[bsep + 4 :]
|
||||
for sub, hexval in body_specs:
|
||||
if sub.encode() in uri:
|
||||
want = bytes.fromhex(hexval)
|
||||
if verbatim:
|
||||
check_body_verbatim(
|
||||
rec, http_hdr, body, sub, want, digests_emitted
|
||||
)
|
||||
else:
|
||||
check_body(rec, http_hdr, body, sub, want)
|
||||
body_hits[sub] = True
|
||||
elif wtype == b"revisit":
|
||||
revisits += 1
|
||||
for sub in no_resp:
|
||||
if sub.encode() in uri:
|
||||
revisit_hits[sub] = True
|
||||
|
||||
if total < 1:
|
||||
sys.exit("no records found")
|
||||
if infos != 1:
|
||||
sys.exit("expected exactly one warcinfo record, got %d" % infos)
|
||||
if expect_revisit and revisits < 1:
|
||||
sys.exit("expected at least one revisit record, found none")
|
||||
for sub, hit in body_hits.items():
|
||||
if not hit:
|
||||
sys.exit("no response record found for --expect-body-hex %s" % sub)
|
||||
for sub, hit in revisit_hits.items():
|
||||
if not hit:
|
||||
sys.exit("no revisit record found for unchanged asset %s" % sub)
|
||||
print(
|
||||
"warc-validate: %d records OK (%d response, %d revisit)"
|
||||
% (total, responses, revisits)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -46,9 +46,12 @@ cat >"$stubdir/x-www-browser" <<EOF
|
||||
echo "stub browser invoked with: \$1" >&2
|
||||
# Also fetch an option page and require a rendered title='' tooltip: proves the
|
||||
# option template expands and the \${html:} filter escapes into the attribute.
|
||||
# option9 additionally proves the WARC control renders with its expanded label.
|
||||
opturl="\${1%/}/server/option2.html"
|
||||
warcurl="\${1%/}/server/option9.html"
|
||||
if body="\$(curl -fsSL --max-time 20 "\$1")" && printf '%s' "\$body" | grep -qai httrack && printf '%s' "\$body" | grep -qaF step2.html &&
|
||||
opt="\$(curl -fsSL --max-time 20 "\$opturl")" && printf '%s' "\$opt" | grep -qaF "title='"; then
|
||||
opt="\$(curl -fsSL --max-time 20 "\$opturl")" && printf '%s' "\$opt" | grep -qaF "title='" &&
|
||||
warc="\$(curl -fsSL --max-time 20 "\$warcurl")" && printf '%s' "\$warc" | grep -qaF 'name="warcfile"' && printf '%s' "\$warc" | grep -qaF WARC; then
|
||||
echo PASS >"$marker"
|
||||
else
|
||||
echo "FAIL: unexpected response from \$1" >"$marker"
|
||||
|
||||
Reference in New Issue
Block a user