mirror of
https://github.com/xroche/httrack.git
synced 2026-07-23 01:01:41 +03:00
Compare commits
3 Commits
master
...
fix/footer
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4c93947f44 | ||
|
|
680ece6518 | ||
|
|
76c989abaf |
@@ -90,19 +90,18 @@ offline browser : copy websites to a local directory</p>
|
||||
] [ <b>-NN, --structure[=N]</b> ] [ <b>-%N,
|
||||
--delayed-type-check</b> ] [ <b>-%D,
|
||||
--cached-delayed-type-check</b> ] [ <b>-%M, --mime-html</b>
|
||||
] [ <b>-%r, --warc</b> ] [ <b>-LN, --long-names[=N]</b> ] [
|
||||
<b>-KN, --keep-links[=N]</b> ] [ <b>-x,
|
||||
--replace-external</b> ] [ <b>-%x, --disable-passwords</b> ]
|
||||
[ <b>-%q, --include-query-string</b> ] [ <b>-%g,
|
||||
--strip-query</b> ] [ <b>-o, --generate-errors</b> ] [
|
||||
<b>-X, --purge-old[=N]</b> ] [ <b>-%p, --preserve</b> ] [
|
||||
<b>-%T, --utf8-conversion</b> ] [ <b>-bN, --cookies[=N]</b>
|
||||
] [ <b>-%K, --cookies-file</b> ] [ <b>-%Y, --why</b> ] [
|
||||
<b>-u, --check-type[=N]</b> ] [ <b>-j, --parse-java[=N]</b>
|
||||
] [ <b>-sN, --robots[=N]</b> ] [ <b>-%h, --http-10</b> ] [
|
||||
<b>-%k, --keep-alive</b> ] [ <b>-%z,
|
||||
--disable-compression</b> ] [ <b>-%B, --tolerant</b> ] [
|
||||
<b>-%s, --updatehack</b> ] [ <b>-%u, --urlhack</b> ] [
|
||||
] [ <b>-LN, --long-names[=N]</b> ] [ <b>-KN,
|
||||
--keep-links[=N]</b> ] [ <b>-x, --replace-external</b> ] [
|
||||
<b>-%x, --disable-passwords</b> ] [ <b>-%q,
|
||||
--include-query-string</b> ] [ <b>-%g, --strip-query</b> ] [
|
||||
<b>-o, --generate-errors</b> ] [ <b>-X, --purge-old[=N]</b>
|
||||
] [ <b>-%p, --preserve</b> ] [ <b>-%T, --utf8-conversion</b>
|
||||
] [ <b>-bN, --cookies[=N]</b> ] [ <b>-%K, --cookies-file</b>
|
||||
] [ <b>-%Y, --why</b> ] [ <b>-u, --check-type[=N]</b> ] [
|
||||
<b>-j, --parse-java[=N]</b> ] [ <b>-sN, --robots[=N]</b> ] [
|
||||
<b>-%h, --http-10</b> ] [ <b>-%k, --keep-alive</b> ] [
|
||||
<b>-%z, --disable-compression</b> ] [ <b>-%B, --tolerant</b>
|
||||
] [ <b>-%s, --updatehack</b> ] [ <b>-%u, --urlhack</b> ] [
|
||||
<b>-%A, --assume</b> ] [ <b>-@iN, --protocol[=N]</b> ] [
|
||||
<b>-%w, --disable-module</b> ] [ <b>-F, --user-agent</b> ] [
|
||||
<b>-%R, --referer</b> ] [ <b>-%E, --from</b> ] [ <b>-%F,
|
||||
@@ -649,19 +648,6 @@ don’t wait) (--cached-delayed-type-check)</p></td></tr>
|
||||
<td width="4%">
|
||||
|
||||
|
||||
<p>-%r</p></td>
|
||||
<td width="5%"></td>
|
||||
<td width="82%">
|
||||
|
||||
|
||||
<p>write an ISO-28500 WARC/1.1 archive; --warc-file NAME
|
||||
sets the output name, --warc-max-size N rotates segments
|
||||
past N bytes (--warc)</p></td></tr>
|
||||
<tr valign="top" align="left">
|
||||
<td width="9%"></td>
|
||||
<td width="4%">
|
||||
|
||||
|
||||
<p>-%t</p></td>
|
||||
<td width="5%"></td>
|
||||
<td width="82%">
|
||||
|
||||
@@ -103,17 +103,6 @@ ${do:end-if}
|
||||
> ${LANG_I61}
|
||||
<br><br>
|
||||
|
||||
<input type="checkbox" name="warc" ${checked:warc}
|
||||
title='${html:LANG_WARCTIP}' onMouseOver="info('${html:LANG_WARCTIP}'); return true" onMouseOut="info(' '); return true"
|
||||
> ${LANG_WARC}
|
||||
<br><br>
|
||||
|
||||
${LANG_WARCFILE}
|
||||
<input name="warcfile" value="${warcfile}" size="40"
|
||||
title='${html:LANG_WARCFILETIP}' onMouseOver="info('${html:LANG_WARCFILETIP}'); return true" onMouseOut="info(' '); return true"
|
||||
>
|
||||
<br><br>
|
||||
|
||||
<input type="checkbox" name="norecatch" ${checked:norecatch}
|
||||
title='${html:LANG_I5b}' onMouseOver="info('${html:LANG_I5b}'); return true" onMouseOut="info(' '); return true"
|
||||
> ${LANG_I34b}
|
||||
|
||||
@@ -141,8 +141,6 @@ ${do:copy:KeepSlashes:keepslashes}
|
||||
${do:copy:KeepQueryOrder:keepqueryorder}
|
||||
${do:copy:StripQuery:stripquery}
|
||||
${do:copy:StoreAllInCache:cache2}
|
||||
${do:copy:Warc:warc}
|
||||
${do:copy:WarcFile:warcfile}
|
||||
${do:copy:LogType:logtype}
|
||||
${do:copy:UseHTTPProxyForFTP:ftpprox}
|
||||
${do:copy:ProxyType:proxytype}
|
||||
|
||||
@@ -187,8 +187,6 @@ ${do:end-if}
|
||||
${test:toler:--tolerant}
|
||||
${test:http10:--http-10}
|
||||
${test:cache2:--store-all-in-cache}
|
||||
${test:warc:--warc}
|
||||
${test:warcfile:--warc-file "}${html:warcfile}${test:warcfile:"}
|
||||
${test:norecatch:--do-not-recatch}
|
||||
${test:logf:--single-log}
|
||||
${test:logtype:::--extra-log:--debug-log}
|
||||
@@ -239,8 +237,6 @@ KeepSlashes=${ztest:keepslashes:0:1}
|
||||
KeepQueryOrder=${ztest:keepqueryorder:0:1}
|
||||
StripQuery=${stripquery}
|
||||
StoreAllInCache=${ztest:cache2:0:1}
|
||||
Warc=${ztest:warc:0:1}
|
||||
WarcFile=${warcfile}
|
||||
LogType=${logtype}
|
||||
UseHTTPProxyForFTP=${ztest:ftpprox:0:1}
|
||||
ProxyType=${proxytype}
|
||||
|
||||
8
lang.def
8
lang.def
@@ -1034,11 +1034,3 @@ LANG_STRIPQUERY
|
||||
Strip query keys:
|
||||
LANG_STRIPQUERYTIP
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
LANG_WARC
|
||||
Write a WARC archive of the crawl
|
||||
LANG_WARCTIP
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
LANG_WARCFILE
|
||||
WARC archive name:
|
||||
LANG_WARCFILETIP
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Ïðåìàõâàíå íà êëþ÷îâå îò çàÿâêàòà:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Êëþ÷îâå îò çàÿâêàòà, ðàçäåëåíè ñúñ çàïåòàÿ, êîèòî äà ñå ïðåìàõíàò îò èìåòî íà çàïèñàíèÿ ôàéë (íàïðèìåð sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Çàïèñâàíå íà WARC àðõèâ íà îáõîæäàíåòî
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Çàïèñâàíå íà âñåêè èçòåãëåí îòãîâîð è â WARC/1.1 àðõèâ ïî ISO-28500, äî îãëåäàëîòî.
|
||||
WARC archive name:
|
||||
Èìå íà WARC àðõèâà:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Íåçàäúëæèòåëíî áàçîâî èìå çà WARC àðõèâà; îñòàâåòå ïðàçíî çà àâòîìàòè÷íî èìåíóâàíå â èçõîäíàòà äèðåêòîðèÿ.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Eliminar claves de query string:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Claves de query string, separadas por comas, que se eliminarán del nombre de los archivos guardados (p. ej. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Escribir un archivo WARC del rastreo
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Guardar también cada respuesta descargada en un archivo WARC/1.1 ISO-28500, junto a la réplica.
|
||||
WARC archive name:
|
||||
Nombre del archivo WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nombre base opcional para el archivo WARC; déjelo en blanco para nombrarlo automáticamente en el directorio de salida.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Odebrat klíèe dotazu:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Klíèe dotazu oddìlené èárkami, které se vynechají z pojmenování ukládaných souborù (napø. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Zapsat archiv WARC z procházení
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Uložit také každou staženou odpovìï do archivu WARC/1.1 podle ISO-28500 vedle zrcadla.
|
||||
WARC archive name:
|
||||
Název archivu WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Volitelný základní název archivu WARC; ponechte prázdné pro automatické pojmenování ve výstupním adresáøi.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
移除查詢鍵:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
以逗號分隔的查詢鍵,將其從儲存檔案的命名中移除 (例如 sid,utm_source)。
|
||||
Write a WARC archive of the crawl
|
||||
寫入此次抓取的 WARC 封存檔
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
同時將每個已擷取的回應儲存為 ISO-28500 WARC/1.1 封存檔,置於鏡像網站旁。
|
||||
WARC archive name:
|
||||
WARC 封存檔名稱:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC 封存檔的選用基本名稱;留空則於輸出目錄中自動命名。
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
剥离查询键:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
用逗号分隔的查询键,将其从保存文件的命名中删除 (例如 sid,utm_source)。
|
||||
Write a WARC archive of the crawl
|
||||
写入本次抓取的 WARC 归档
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
同时将每个已获取的响应保存为 ISO-28500 WARC/1.1 归档,置于镜像站点旁边。
|
||||
WARC archive name:
|
||||
WARC 归档名称:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC 归档的可选基本名称;留空则在输出目录中自动命名。
|
||||
|
||||
@@ -958,11 +958,3 @@ Strip query keys:
|
||||
Ukloniti kljuèeve upita:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Zarezom odvojeni kljuèevi upita koji se izostavljaju iz naziva spremljenih datoteka (npr. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Zapi¹i WARC arhivu obilaska
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Spremi i svaki preuzeti odgovor u ISO-28500 WARC/1.1 arhivu, uz zrcalo.
|
||||
WARC archive name:
|
||||
Naziv WARC arhive:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Neobavezni osnovni naziv WARC arhive; ostavite prazno za automatsko imenovanje u izlaznom direktoriju.
|
||||
|
||||
@@ -1004,11 +1004,3 @@ Strip query keys:
|
||||
Fjern forespørgselsnøgler:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kommaseparerede forespørgselsnøgler, der udelades i navngivningen af gemte filer (f.eks. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Skriv et WARC-arkiv af gennemsøgningen
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Gem også hvert hentet svar i et ISO-28500 WARC/1.1-arkiv ved siden af spejlet.
|
||||
WARC archive name:
|
||||
Navn på WARC-arkiv:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Valgfrit basisnavn til WARC-arkivet; lad feltet stå tomt for automatisk navngivning i outputmappen.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Query-Schlüssel entfernen:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kommagetrennte Query-Schlüssel, die bei der Benennung gespeicherter Dateien entfallen (z. B. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
WARC-Archiv des Crawls schreiben
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Jede heruntergeladene Antwort zusätzlich in einem ISO-28500-WARC/1.1-Archiv neben dem Spiegel speichern.
|
||||
WARC archive name:
|
||||
Name des WARC-Archivs:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Optionaler Basisname für das WARC-Archiv; leer lassen, um es automatisch im Ausgabeverzeichnis zu benennen.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Eemalda päringuvõtmed:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Komadega eraldatud päringuvõtmed, mis jäetakse salvestatud faili nimest välja (nt sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Kirjuta läbimise WARC-arhiiv
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Salvesta iga alla laaditud vastus ka ISO-28500 WARC/1.1 arhiivi peegli kõrvale.
|
||||
WARC archive name:
|
||||
WARC-arhiivi nimi:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC-arhiivi valikuline põhinimi; jäta tühjaks, et see väljundkataloogis automaatselt nimetada.
|
||||
|
||||
@@ -1004,11 +1004,3 @@ Strip query keys:
|
||||
Strip query keys:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Write a WARC archive of the crawl
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
WARC archive name:
|
||||
WARC archive name:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
|
||||
@@ -958,11 +958,3 @@ Strip query keys:
|
||||
Poista kyselyavaimet:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Pilkuin erotellut kyselyavaimet, jotka jätetään pois tallennettujen tiedostojen nimeämisestä (esim. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Kirjoita imuroinnin WARC-arkisto
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Tallenna myös jokainen noudettu vastaus ISO-28500 WARC/1.1 -arkistoon peilin viereen.
|
||||
WARC archive name:
|
||||
WARC-arkiston nimi:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Valinnainen WARC-arkiston perusnimi; jätä tyhjäksi, jotta se nimetään automaattisesti tulostehakemistoon.
|
||||
|
||||
@@ -1004,11 +1004,3 @@ Strip query keys:
|
||||
Supprimer les clés de query string :
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Clés de query string à retirer du nommage des fichiers enregistrés, séparées par des virgules (par ex. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Écrire une archive WARC du crawl
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Enregistrer aussi chaque réponse téléchargée dans une archive WARC/1.1 (ISO-28500), à côté du miroir.
|
||||
WARC archive name:
|
||||
Nom de l'archive WARC :
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nom de base optionnel pour l'archive WARC ; laissez vide pour le générer automatiquement dans le répertoire de sortie.
|
||||
|
||||
@@ -958,11 +958,3 @@ Strip query keys:
|
||||
Αφαίρεση κλειδιών ερωτήματος:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Κλειδιά ερωτήματος χωρισμένα με κόμμα, που θα αφαιρεθούν από την ονομασία των αποθηκευμένων αρχείων (π.χ. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Εγγραφή αρχείου WARC της ανίχνευσης
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Αποθήκευση κάθε ληφθείσας απόκρισης και σε αρχείο WARC/1.1 ISO-28500, δίπλα στο είδωλο.
|
||||
WARC archive name:
|
||||
Όνομα αρχείου WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Προαιρετικό βασικό όνομα για το αρχείο WARC. Αφήστε το κενό για αυτόματη ονομασία στον κατάλογο εξόδου.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Rimuovi chiavi della query string:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Chiavi della query string, separate da virgole, da rimuovere dai nomi dei file salvati (ad es. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Scrivi un archivio WARC della scansione
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Salva anche ogni risposta scaricata in un archivio WARC/1.1 ISO-28500, accanto al mirror.
|
||||
WARC archive name:
|
||||
Nome dell'archivio WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nome di base facoltativo per l'archivio WARC; lascia vuoto per assegnarlo automaticamente nella directory di output.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
削除するクエリキー:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
保存ファイル名の生成から除外するクエリキーをカンマ区切りで指定します (例: sid,utm_source)。
|
||||
Write a WARC archive of the crawl
|
||||
クロールの WARC アーカイブを書き出す
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
取得した各レスポンスを ISO-28500 WARC/1.1 アーカイブとしてミラーの隣にも保存します。
|
||||
WARC archive name:
|
||||
WARC アーカイブ名:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC アーカイブの任意のベース名。空欄にすると出力ディレクトリ内で自動的に名前が付けられます。
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
¾âáâàÐÝØ ÚÛãçÕÒØ ÞÔ ÑÐàÐúÕâÞ:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
ºÛãçÕÒØ ÞÔ ÑÐàÐúÕâÞ, ÞÔÔÕÛÕÝØ áÞ ×ÐߨàÚÐ, èâÞ áÕ ÞâáâàÐÝãÒÐÐâ ÞÔ ØÜÕâÞ ÝÐ ×ÐçãÒÐÝÐâÐ ÔÐâÞâÕÚÐ (ÝÐ ßàØÜÕà sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
·ÐßØèØ WARC ÐàåØÒÐ ÝÐ ßàÕÑÐàãÒÐúÕâÞ
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
·ÐçãÒÐø ÓÞ áÕÚÞø ßàÕ×ÕÜÕÝ ÞÔÓÞÒÞà Ø ÒÞ ISO-28500 WARC/1.1 ÐàåØÒÐ, ßÞÚàÐø ÞÓÛÕÔÐÛÞâÞ.
|
||||
WARC archive name:
|
||||
¸ÜÕ ÝÐ WARC ÐàåØÒÐâÐ:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
¸×ÑÞàÝÞ ÞáÝÞÒÝÞ ØÜÕ ×Ð WARC ÐàåØÒÐâÐ; ÞáâÐÒÕâÕ ßàÐ×ÝÞ ×Ð ÐÒâÞÜÐâáÚÞ ØÜÕÝãÒÐúÕ ÒÞ Ø×ÛÕ×ÝØÞâ ÔØàÕÚâÞàØãÜ.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Lekérdezési kulcsok eltávolítása:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Vesszõvel elválasztott lekérdezési kulcsok, amelyeket el kell hagyni a mentett fájl elnevezésébõl (pl. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
A bejárás WARC archívumának írása
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Minden letöltött válasz mentése ISO-28500 WARC/1.1 archívumba is, a tükör mellé.
|
||||
WARC archive name:
|
||||
WARC archívum neve:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
A WARC archívum opcionális alapneve; hagyja üresen az automatikus elnevezéshez a kimeneti könyvtárban.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Query-sleutels verwijderen:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Door komma's gescheiden query-sleutels die bij het benoemen van opgeslagen bestanden worden weggelaten (bijv. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Schrijf een WARC-archief van de crawl
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Sla ook elke opgehaalde respons op in een ISO-28500 WARC/1.1-archief, naast de mirror.
|
||||
WARC archive name:
|
||||
Naam van WARC-archief:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Optionele basisnaam voor het WARC-archief; laat leeg om het automatisch een naam te geven in de uitvoermap.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Fjern spørrenøkler:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kommaseparerte spørrenøkler som utelates i navngivingen av lagrede filer (f.eks. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Skriv et WARC-arkiv av gjennomgangen
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Lagre også hvert nedlastet svar i et ISO-28500 WARC/1.1-arkiv ved siden av speilet.
|
||||
WARC archive name:
|
||||
Navn på WARC-arkiv:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Valgfritt basisnavn for WARC-arkivet; la feltet stå tomt for automatisk navngivning i utdatamappen.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Usuñ klucze zapytania:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Rozdzielone przecinkami klucze zapytania pomijane przy nazywaniu zapisanych plików (np. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Zapisz archiwum WARC z indeksowania
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Zapisz te¿ ka¿d± pobran± odpowied¼ do archiwum WARC/1.1 ISO-28500, obok kopii lustrzanej.
|
||||
WARC archive name:
|
||||
Nazwa archiwum WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Opcjonalna nazwa bazowa archiwum WARC; pozostaw puste, aby nazwaæ je automatycznie w katalogu wyj¶ciowym.
|
||||
|
||||
@@ -1004,11 +1004,3 @@ Strip query keys:
|
||||
Remover chaves da query string:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Chaves da query string, separadas por vírgulas, a serem removidas da nomeação dos arquivos salvos (ex.: sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Gravar um arquivo WARC do rastreamento
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Salvar também cada resposta baixada em um arquivo WARC/1.1 ISO-28500, ao lado do espelho.
|
||||
WARC archive name:
|
||||
Nome do arquivo WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nome base opcional para o arquivo WARC; deixe em branco para nomeá-lo automaticamente no diretório de saída.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Remover chaves da query string:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Chaves da query string, separadas por vírgulas, a remover da nomeação dos ficheiros guardados (por ex. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Escrever um arquivo WARC do rastreio
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Guardar também cada resposta transferida num arquivo WARC/1.1 ISO-28500, ao lado do espelho.
|
||||
WARC archive name:
|
||||
Nome do arquivo WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nome base opcional para o arquivo WARC; deixe em branco para o nomear automaticamente no diretório de saída.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Elimina cheile din query string:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Chei din query string, separate prin virgula, de eliminat din denumirea fisierelor salvate (de ex. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Scrie o arhiva WARC a parcurgerii
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Salveaza si fiecare raspuns descarcat intr-o arhiva WARC/1.1 ISO-28500, langa oglinda.
|
||||
WARC archive name:
|
||||
Numele arhivei WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Nume de baza optional pentru arhiva WARC; lasati gol pentru a-l denumi automat in directorul de iesire.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Óäàëÿòü êëþ÷è çàïðîñà:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Êëþ÷è çàïðîñà ÷åðåç çàïÿòóþ, óäàëÿåìûå èç èìåíè ñîõðàíÿåìîãî ôàéëà (íàïðèìåð, sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Çàïèñàòü WARC-àðõèâ îáõîäà
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Ñîõðàíÿòü êàæäûé çàãðóæåííûé îòâåò òàêæå â àðõèâ WARC/1.1 ISO-28500 ðÿäîì ñ çåðêàëîì.
|
||||
WARC archive name:
|
||||
Èìÿ WARC-àðõèâà:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Íåîáÿçàòåëüíîå áàçîâîå èìÿ WARC-àðõèâà; îñòàâüòå ïóñòûì äëÿ àâòîìàòè÷åñêîãî èìåíîâàíèÿ â âûõîäíîì êàòàëîãå.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Odstráni» kµúèe dotazu:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kµúèe dotazu oddelené èiarkami, ktoré sa vynechajú z pomenovania ukladaných súborov (napr. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Zapísa» archív WARC z prehµadávania
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Ulo¾i» aj ka¾dú stiahnutú odpoveï do archívu WARC/1.1 ISO-28500 vedµa zrkadla.
|
||||
WARC archive name:
|
||||
Názov archívu WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Voliteµný základný názov archívu WARC; ponechajte prázdne pre automatické pomenovanie vo výstupnom adresári.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Odstrani kljuce poizvedbe:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Z vejicami loceni kljuci poizvedbe, ki se izpustijo pri poimenovanju shranjenih datotek (npr. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Zapisi arhiv WARC iz pregledovanja
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Shrani tudi vsak preneseni odgovor v arhiv WARC/1.1 ISO-28500 poleg zrcala.
|
||||
WARC archive name:
|
||||
Ime arhiva WARC:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Neobvezno osnovno ime arhiva WARC; pustite prazno za samodejno poimenovanje v izhodni mapi.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Ta bort frågenycklar:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kommaseparerade frågenycklar som utelämnas vid namngivningen av sparade filer (t.ex. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Skriv ett WARC-arkiv av genomsökningen
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Spara även varje hämtat svar i ett ISO-28500 WARC/1.1-arkiv, bredvid spegeln.
|
||||
WARC archive name:
|
||||
WARC-arkivets namn:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Valfritt basnamn för WARC-arkivet; lämna tomt för att namnge det automatiskt i utdatakatalogen.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Sorgu anahtarlarýný çýkar:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Kaydedilen dosya adlandýrmasýndan çýkarýlacak, virgülle ayrýlmýþ sorgu anahtarlarý (örn. sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Taramanýn WARC arþivini yaz
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Ýndirilen her yanýtý ayrýca aynanýn yanýna bir ISO-28500 WARC/1.1 arþivine kaydet.
|
||||
WARC archive name:
|
||||
WARC arþivi adý:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC arþivi için isteðe baðlý temel ad; çýktý dizininde otomatik adlandýrma için boþ býrakýn.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Âèëó÷àòè êëþ÷³ çàïèòó:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Êëþ÷³ çàïèòó ÷åðåç êîìó, ÿê³ âèëó÷àþòüñÿ ç ³ìåí³ çáåðåæåíîãî ôàéëó (íàïðèêëàä, sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Çàïèñàòè WARC-àðõ³â îáõîäó
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Çáåð³ãàòè êîæíó çàâàíòàæåíó â³äïîâ³äü òàêîæ ó àðõ³â WARC/1.1 ISO-28500 ïîðÿä ³ç äçåðêàëîì.
|
||||
WARC archive name:
|
||||
²ì'ÿ WARC-àðõ³âó:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
Íåîáîâ'ÿçêîâà áàçîâà íàçâà WARC-àðõ³âó; çàëèøòå ïîðîæí³ì äëÿ àâòîìàòè÷íîãî íàéìåíóâàííÿ ó âèõ³äíîìó êàòàëîç³.
|
||||
|
||||
@@ -956,11 +956,3 @@ Strip query keys:
|
||||
Olib tashlanadigan so’rov kalitlari:
|
||||
Comma-separated query keys to drop from the saved-file naming (e.g. sid,utm_source).
|
||||
Saqlangan fayl nomidan olib tashlanadigan, vergul bilan ajratilgan so’rov kalitlari (masalan, sid,utm_source).
|
||||
Write a WARC archive of the crawl
|
||||
Qidiruvning WARC arxivini yozish
|
||||
Also save every fetched response into an ISO-28500 WARC/1.1 archive, next to the mirror.
|
||||
Har bir yuklab olingan javobni ISO-28500 WARC/1.1 arxiviga ham, ko'zgu yonida saqlash.
|
||||
WARC archive name:
|
||||
WARC arxivi nomi:
|
||||
Optional base name for the WARC archive; leave blank to auto-name it under the output directory.
|
||||
WARC arxivi uchun ixtiyoriy asosiy nom; chiqish katalogida avtomatik nomlash uchun bo'sh qoldiring.
|
||||
|
||||
@@ -40,7 +40,6 @@ httrack \- offline browser : copy websites to a local directory
|
||||
[ \fB\-%N, \-\-delayed\-type\-check\fR ]
|
||||
[ \fB\-%D, \-\-cached\-delayed\-type\-check\fR ]
|
||||
[ \fB\-%M, \-\-mime\-html\fR ]
|
||||
[ \fB\-%r, \-\-warc\fR ]
|
||||
[ \fB\-LN, \-\-long\-names[=N]\fR ]
|
||||
[ \fB\-KN, \-\-keep\-links[=N]\fR ]
|
||||
[ \fB\-x, \-\-replace\-external\fR ]
|
||||
@@ -198,8 +197,6 @@ delayed type check, don't make any link test but wait for files download to star
|
||||
cached delayed type check, don't wait for remote type during updates, to speedup them (%D0 wait, * %D1 don't wait) (\-\-cached\-delayed\-type\-check)
|
||||
.IP \-%M
|
||||
generate a RFC MIME\-encapsulated full\-archive (.mht) (\-\-mime\-html)
|
||||
.IP \-%r
|
||||
write an ISO\-28500 WARC/1.1 archive; \-\-warc\-file NAME sets the output name, \-\-warc\-max\-size N rotates segments past N bytes (\-\-warc)
|
||||
.IP \-%t
|
||||
keep the original file extension, don't rewrite it from the MIME type (%t0 rewrite)
|
||||
.IP \-LN
|
||||
|
||||
@@ -63,7 +63,7 @@ libhttrack_la_SOURCES = htscore.c htsparse.c htsback.c htscache.c \
|
||||
htshelp.c htslib.c htsurlport.c htscoremain.c \
|
||||
htsname.c htsrobots.c htstools.c htswizard.c \
|
||||
htsalias.c htsthread.c htsindex.c htsbauth.c \
|
||||
htsmd5.c htscodec.c htswarc.c htsproxy.c htszlib.c htswrap.c htsconcat.c \
|
||||
htsmd5.c htscodec.c htsproxy.c htszlib.c htswrap.c htsconcat.c \
|
||||
htsmodules.c htscharset.c punycode.c htsencoding.c htssniff.c \
|
||||
md5.c \
|
||||
minizip/ioapi.c minizip/mztools.c minizip/unzip.c minizip/zip.c \
|
||||
@@ -74,7 +74,7 @@ libhttrack_la_SOURCES = htscore.c htsparse.c htsback.c htscache.c \
|
||||
htshelp.h htsindex.h htslib.h htsurlport.h htsmd5.h \
|
||||
htsmodules.h htsname.h htsnet.h htssniff.h \
|
||||
htsopt.h htsrobots.h htsthread.h \
|
||||
htstools.h htswizard.h htswrap.h htscodec.h htswarc.h htsproxy.h htszlib.h \
|
||||
htstools.h htswizard.h htswrap.h htscodec.h htsproxy.h htszlib.h \
|
||||
htsstrings.h htsarrays.h httrack-library.h \
|
||||
htscharset.h punycode.h htsencoding.h \
|
||||
htsentities.h htsentities.sh htsbasiccharsets.sh htscodepages.h \
|
||||
|
||||
@@ -114,10 +114,6 @@ const char *hts_optalias[][4] = {
|
||||
"strip [host/pattern=]key1,key2,... from URLs"},
|
||||
{"cookies-file", "-%K", "param1",
|
||||
"load extra cookies from a Netscape cookies.txt"},
|
||||
{"warc", "-%r", "single", "write an ISO-28500 WARC/1.1 archive of the crawl"},
|
||||
{"warc-file", "-%rf", "param1", "write a WARC archive to the given base name"},
|
||||
{"warc-max-size", "-%rs", "param1",
|
||||
"rotate the WARC archive once a segment passes N bytes (0: single file)"},
|
||||
{"why", "-%Y", "param1",
|
||||
"explain which filter rule accepts or rejects a URL, then exit"},
|
||||
{"pause", "-%G", "param1",
|
||||
|
||||
@@ -37,7 +37,6 @@ Please visit our Website: http://www.httrack.com
|
||||
/* specific definitions */
|
||||
#include "htsnet.h"
|
||||
#include "htscore.h"
|
||||
#include "htswarc.h"
|
||||
#include "htsthread.h"
|
||||
#include <time.h>
|
||||
/* END specific definitions */
|
||||
@@ -980,10 +979,6 @@ int back_finalize(httrackp * opt, cache_back * cache, struct_back * sback,
|
||||
// status finished callback
|
||||
RUN_CALLBACK1(opt, xfrstatus, &back[p]);
|
||||
|
||||
// WARC archive of the transaction (request + response/revisit)
|
||||
if (StringNotEmpty(opt->warc_file))
|
||||
warc_write_backtransaction(opt, &back[p]);
|
||||
|
||||
return 0;
|
||||
} else { // testmode
|
||||
if (back[p].r.statuscode / 100 >= 3) { /* Store 3XX, 4XX, 5XX test response codes, but NOT 2XX */
|
||||
@@ -1060,9 +1055,6 @@ void back_copy_static(const lien_back * src, lien_back * dst) {
|
||||
dst->r.soc = INVALID_SOCKET;
|
||||
dst->r.adr = NULL;
|
||||
dst->r.headers = NULL;
|
||||
dst->r.warc_reqhdr = NULL;
|
||||
dst->r.warc_resphdr = NULL;
|
||||
dst->r.warc_truncated = 0;
|
||||
dst->r.out = NULL;
|
||||
dst->r.location = dst->location_buffer;
|
||||
dst->r.fp = NULL;
|
||||
@@ -1126,9 +1118,6 @@ int back_unserialize(FILE * fp, lien_back ** dst) {
|
||||
(*dst)->chunk_adr = NULL;
|
||||
(*dst)->r.adr = NULL;
|
||||
(*dst)->r.out = NULL;
|
||||
(*dst)->r.warc_reqhdr = NULL;
|
||||
(*dst)->r.warc_resphdr = NULL;
|
||||
(*dst)->r.warc_truncated = 0;
|
||||
(*dst)->r.location = (*dst)->location_buffer;
|
||||
(*dst)->r.fp = NULL;
|
||||
(*dst)->r.soc = INVALID_SOCKET;
|
||||
@@ -1596,7 +1585,6 @@ int back_clear_entry(lien_back * back) {
|
||||
freet(back->r.headers);
|
||||
back->r.headers = NULL;
|
||||
}
|
||||
warc_free_request(&back->r);
|
||||
// Tout nettoyer
|
||||
memset(back, 0, sizeof(lien_back));
|
||||
back->r.soc = INVALID_SOCKET;
|
||||
@@ -2521,19 +2509,6 @@ void back_wait(struct_back * sback, httrackp * opt, cache_back * cache,
|
||||
|
||||
for (i = 0; i < (unsigned int) back_max; i++) {
|
||||
if (back[i].status > 0 && back[i].status < STATUS_FTP_TRANSFER) {
|
||||
/* A cap-truncated body is deliberate, not broken: archive what arrived
|
||||
with WARC-Truncated before the abort overwrites the slot's real 2xx
|
||||
status. HTTrack still treats the slot as incomplete afterwards. */
|
||||
if (StringNotEmpty(opt->warc_file) && back[i].r.statuscode > 0 &&
|
||||
back[i].r.warc_resphdr != NULL && back[i].r.size > 0 &&
|
||||
!(back[i].r.is_write && IS_DELAYED_EXT(back[i].url_sav))) {
|
||||
if (back[i].r.is_write && back[i].r.out != NULL)
|
||||
fflush(back[i].r.out);
|
||||
back[i].r.warc_truncated = (limit == HTS_MIRROR_LIMIT_SIZE)
|
||||
? WARC_TRUNC_LENGTH
|
||||
: WARC_TRUNC_TIME;
|
||||
warc_write_backtransaction(opt, &back[i]);
|
||||
}
|
||||
if (back[i].r.soc != INVALID_SOCKET) {
|
||||
deletehttp(&back[i].r);
|
||||
}
|
||||
@@ -3656,10 +3631,6 @@ void back_wait(struct_back * sback, httrackp * opt, cache_back * cache,
|
||||
deleteaddr(&back[i].r);
|
||||
back[i].r.headers = block;
|
||||
}
|
||||
// Stash the raw response headers for WARC (deletehttp frees
|
||||
// r.headers when the socket closes, before back_finalize)
|
||||
if (StringNotEmpty(opt->warc_file))
|
||||
warc_stash_response(&back[i].r, back[i].r.headers);
|
||||
|
||||
/*
|
||||
Status code and header-response hacks
|
||||
|
||||
@@ -39,7 +39,6 @@ Please visit our Website: http://www.httrack.com
|
||||
|
||||
/* File defs */
|
||||
#include "htscore.h"
|
||||
#include "htswarc.h"
|
||||
|
||||
/* specific definitions */
|
||||
#include "htsbase.h"
|
||||
@@ -2246,7 +2245,6 @@ int httpmirror(char *url1, httrackp * opt) {
|
||||
|
||||
// ending
|
||||
usercommand(opt, 0, NULL, NULL, NULL, NULL);
|
||||
warc_close_opt(opt);
|
||||
|
||||
// désallocation mémoire & buffers
|
||||
XH_uninit;
|
||||
@@ -3630,10 +3628,6 @@ HTSEXT_API int copy_htsopt(const httrackp * from, httrackp * to) {
|
||||
if (StringNotEmpty(from->cookies_file))
|
||||
StringCopyS(to->cookies_file, from->cookies_file);
|
||||
|
||||
if (StringNotEmpty(from->warc_file))
|
||||
StringCopyS(to->warc_file, from->warc_file);
|
||||
to->warc_max_size = from->warc_max_size;
|
||||
|
||||
if (from->pause_max_ms > 0) {
|
||||
to->pause_min_ms = from->pause_min_ms;
|
||||
to->pause_max_ms = from->pause_max_ms;
|
||||
|
||||
@@ -40,7 +40,6 @@ Please visit our Website: http://www.httrack.com
|
||||
#include "htscore.h"
|
||||
#include "htsdefines.h"
|
||||
#include "htsalias.h"
|
||||
#include "htswarc.h"
|
||||
#include "htsbauth.h"
|
||||
#include "htswrap.h"
|
||||
#include "htsmodules.h"
|
||||
@@ -1790,45 +1789,6 @@ static int hts_main_internal(int argc, char **argv, httrackp * opt) {
|
||||
StringCopy(opt->cookies_file, argv[na]);
|
||||
}
|
||||
break;
|
||||
case 'r': // warc / warc-file: write an ISO-28500 WARC archive
|
||||
if (*(com + 1) == 'f') { // --warc-file NAME: explicit basename
|
||||
com++;
|
||||
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
|
||||
HTS_PANIC_PRINTF(
|
||||
"Option warc-file needs a blank space and a WARC name");
|
||||
htsmain_free();
|
||||
return -1;
|
||||
}
|
||||
na++;
|
||||
if (strlen(argv[na]) >= 1024) {
|
||||
HTS_PANIC_PRINTF("WARC file name too long");
|
||||
htsmain_free();
|
||||
return -1;
|
||||
}
|
||||
StringCopy(opt->warc_file, argv[na]);
|
||||
} else if (*(com + 1) == 's') { // --warc-max-size N: rotation
|
||||
com++;
|
||||
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
|
||||
HTS_PANIC_PRINTF(
|
||||
"Option warc-max-size needs a blank space and a size");
|
||||
htsmain_free();
|
||||
return -1;
|
||||
}
|
||||
na++;
|
||||
{ // reject non-numeric/negative/overflow; keep default 0
|
||||
// (single file)
|
||||
char *end;
|
||||
LLint v;
|
||||
errno = 0;
|
||||
v = strtoll(argv[na], &end, 10);
|
||||
if (isdigit((unsigned char) argv[na][0]) && *end == '\0' &&
|
||||
errno != ERANGE)
|
||||
opt->warc_max_size = v;
|
||||
}
|
||||
} else { // --warc: auto-named archive under the output dir
|
||||
StringCopy(opt->warc_file, WARC_AUTONAME);
|
||||
}
|
||||
break;
|
||||
case 'Y': // why: explain the filter verdict for a URL, no crawl
|
||||
if ((na + 1 >= argc) || (argv[na + 1][0] == '-')) {
|
||||
HTS_PANIC_PRINTF("Option why needs a blank space and a URL");
|
||||
|
||||
@@ -524,8 +524,6 @@ void help(const char *app, int more) {
|
||||
infomsg
|
||||
(" %D cached delayed type check, don't wait for remote type during updates, to speedup them (%D0 wait, * %D1 don't wait)");
|
||||
infomsg(" %M generate a RFC MIME-encapsulated full-archive (.mht)");
|
||||
infomsg(" %r write an ISO-28500 WARC/1.1 archive; --warc-file NAME sets the "
|
||||
"output name, --warc-max-size N rotates segments past N bytes");
|
||||
infomsg(" %t keep the original file extension, don't rewrite it from the "
|
||||
"MIME type (%t0 rewrite)");
|
||||
infomsg
|
||||
|
||||
11
src/htslib.c
11
src/htslib.c
@@ -36,7 +36,6 @@ Please visit our Website: http://www.httrack.com
|
||||
// Fichier librairie .c
|
||||
|
||||
#include "htscore.h"
|
||||
#include "htswarc.h"
|
||||
|
||||
/* specific definitions */
|
||||
#include "htsbase.h"
|
||||
@@ -1202,11 +1201,6 @@ int http_sendhead(httrackp * opt, t_cookie * cookie, int mode,
|
||||
} // Fin test pas postfile
|
||||
//
|
||||
|
||||
// Stash the raw request for the WARC request record (freed at
|
||||
// back_clear_entry)
|
||||
if (StringNotEmpty(opt->warc_file))
|
||||
warc_stash_request(retour, bstr.buffer);
|
||||
|
||||
// Callback
|
||||
{
|
||||
int test_head =
|
||||
@@ -2120,8 +2114,6 @@ htsblk http_test(httrackp * opt, const char *adr, const char *fil, char *loc) {
|
||||
#if HTS_DEBUG_CLOSESOCK
|
||||
DEBUG_W("http_test: deletehttp\n");
|
||||
#endif
|
||||
// this probe's htsblk is discarded by callers, so free any WARC stash here
|
||||
warc_free_request(&retour);
|
||||
deletehttp(&retour);
|
||||
retour.soc = INVALID_SOCKET;
|
||||
}
|
||||
@@ -6018,8 +6010,6 @@ HTSEXT_API httrackp *hts_create_opt(void) {
|
||||
StringCopy(opt->footer, HTS_DEFAULT_FOOTER);
|
||||
StringCopy(opt->strip_query, "");
|
||||
StringCopy(opt->cookies_file, "");
|
||||
StringCopy(opt->warc_file, "");
|
||||
opt->warc_max_size = 0; /* no rotation unless --warc-max-size sets it */
|
||||
StringCopy(opt->why_url, "");
|
||||
opt->pause_min_ms = 0;
|
||||
opt->pause_max_ms = 0;
|
||||
@@ -6171,7 +6161,6 @@ HTSEXT_API void hts_free_opt(httrackp * opt) {
|
||||
StringFree(opt->strip_query);
|
||||
StringFree(opt->cookies_file);
|
||||
StringFree(opt->why_url);
|
||||
StringFree(opt->warc_file);
|
||||
|
||||
StringFree(opt->path_html);
|
||||
StringFree(opt->path_html_utf8);
|
||||
|
||||
10
src/htsopt.h
10
src/htsopt.h
@@ -256,7 +256,6 @@ struct htsoptstate {
|
||||
unsigned int debug_state;
|
||||
unsigned int tmpnameid; /**< counter for temporary file names */
|
||||
int is_ended; /**< mirror has finished */
|
||||
void *warc; /**< open WARC writer (warc_writer*), or NULL */
|
||||
};
|
||||
|
||||
/* Library handles */
|
||||
@@ -539,10 +538,6 @@ struct httrackp {
|
||||
int pause_max_ms; /**< inter-file pause upper bound, ms */
|
||||
String why_url; /**< URL to diagnose (--why): print the deciding filter rule
|
||||
and exit without crawling */
|
||||
String warc_file; /**< WARC output: WARC_AUTONAME for --warc, or the
|
||||
--warc-file basename (appended at the tail: ABI) */
|
||||
LLint warc_max_size; /**< --warc-max-size: rotate the archive past this many
|
||||
bytes (<=0: single file). Tail: ABI */
|
||||
};
|
||||
|
||||
/* Running statistics for a mirror. */
|
||||
@@ -666,11 +661,6 @@ struct htsblk {
|
||||
/* Restart-whole signal: a resume this response rejected (unusable 206) must
|
||||
retry with no Range, else a surviving partial/temp-ref loops (#581). */
|
||||
hts_boolean refetch_wholefile;
|
||||
char *warc_reqhdr; /**< stashed raw request header block for WARC (or NULL) */
|
||||
char *
|
||||
warc_resphdr; /**< stashed raw response header block for WARC (or NULL) */
|
||||
int warc_truncated; /**< WARC-Truncated reason for a cap-truncated body
|
||||
(WARC_TRUNC_*, 0=none). Tail: ABI */
|
||||
/*char digest[32+2]; // md5 digest generated by the engine ("" if none) */
|
||||
};
|
||||
|
||||
|
||||
@@ -57,7 +57,6 @@ Please visit our Website: http://www.httrack.com
|
||||
#include "htssniff.h"
|
||||
#include "htscodec.h"
|
||||
#include "htsproxy.h"
|
||||
#include "htswarc.h"
|
||||
#if HTS_USEZLIB
|
||||
#include "htszlib.h"
|
||||
#endif
|
||||
@@ -1302,13 +1301,6 @@ static int st_copyopt(httrackp *opt, int argc, char **argv) {
|
||||
if (strcmp(StringBuff(to->cookies_file), "/tmp/jar.txt") != 0)
|
||||
err = 1;
|
||||
|
||||
/* warc_file: same String deep-copy path as cookies_file */
|
||||
StringCopy(from->warc_file, "run.warc.gz");
|
||||
StringCopy(to->warc_file, "");
|
||||
copy_htsopt(from, to);
|
||||
if (strcmp(StringBuff(to->warc_file), "run.warc.gz") != 0)
|
||||
err = 1;
|
||||
|
||||
/* #185 pause pair: copied when enabled (max>0), the 0 sentinel skips */
|
||||
from->pause_min_ms = 5000;
|
||||
from->pause_max_ms = 10000;
|
||||
@@ -3264,498 +3256,6 @@ static int st_ftpuser(httrackp *opt, int argc, char **argv) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Bounded substring search (records carry NUL bytes; strstr won't do). */
|
||||
static const char *warc_memstr(const char *hay, const char *needle,
|
||||
size_t haylen, size_t nlen) {
|
||||
if (nlen == 0 || haylen < nlen)
|
||||
return NULL;
|
||||
{
|
||||
size_t i;
|
||||
for (i = 0; i + nlen <= haylen; i++) {
|
||||
if (memcmp(hay + i, needle, nlen) == 0)
|
||||
return hay + i;
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Slurp a whole file into a malloc'd buffer; sets *len. NULL on error. */
|
||||
static unsigned char *warc_slurp(const char *path, size_t *len) {
|
||||
FILE *f = FOPEN(path, "rb");
|
||||
unsigned char *buf;
|
||||
long sz;
|
||||
if (f == NULL)
|
||||
return NULL;
|
||||
if (fseek(f, 0, SEEK_END) != 0 || (sz = ftell(f)) < 0) {
|
||||
fclose(f);
|
||||
return NULL;
|
||||
}
|
||||
rewind(f);
|
||||
buf = malloct((size_t) sz + 1);
|
||||
if (buf == NULL) {
|
||||
fclose(f);
|
||||
return NULL;
|
||||
}
|
||||
*len = fread(buf, 1, (size_t) sz, f);
|
||||
fclose(f);
|
||||
return buf;
|
||||
}
|
||||
|
||||
/* Inflate one gzip member at *in (limit end); returns the decompressed record
|
||||
in a malloc'd buffer (*out_len), advancing *in past the member. NULL at end
|
||||
or on error (*out_len distinguishes: 0 and NULL = clean end). */
|
||||
static unsigned char *warc_next_member(const unsigned char **in,
|
||||
const unsigned char *end,
|
||||
size_t *out_len) {
|
||||
z_stream zs;
|
||||
unsigned char *out = NULL;
|
||||
size_t len = 0;
|
||||
int zerr;
|
||||
*out_len = 0;
|
||||
if (*in >= end)
|
||||
return NULL;
|
||||
memset(&zs, 0, sizeof(zs));
|
||||
if (inflateInit2(&zs, 15 + 32) != Z_OK)
|
||||
return NULL;
|
||||
zs.next_in = (const Bytef *) *in;
|
||||
zs.avail_in = (uInt) (end - *in);
|
||||
do {
|
||||
unsigned char tmp[8192];
|
||||
size_t got;
|
||||
zs.next_out = tmp;
|
||||
zs.avail_out = sizeof(tmp);
|
||||
zerr = inflate(&zs, Z_NO_FLUSH);
|
||||
if (zerr != Z_OK && zerr != Z_STREAM_END) {
|
||||
freet(out);
|
||||
inflateEnd(&zs);
|
||||
return NULL;
|
||||
}
|
||||
got = sizeof(tmp) - zs.avail_out;
|
||||
if (got > 0) {
|
||||
unsigned char *n = realloct(out, len + got + 1);
|
||||
if (n == NULL) {
|
||||
freet(out);
|
||||
inflateEnd(&zs);
|
||||
return NULL;
|
||||
}
|
||||
out = n;
|
||||
memcpy(out + len, tmp, got);
|
||||
len += got;
|
||||
}
|
||||
} while (zerr != Z_STREAM_END);
|
||||
*in = (const unsigned char *) zs.next_in; /* start of the next member */
|
||||
inflateEnd(&zs);
|
||||
if (out != NULL)
|
||||
out[len] = '\0';
|
||||
*out_len = len;
|
||||
return out;
|
||||
}
|
||||
|
||||
/* Feed a synthetic transaction and validate the resulting .warc.gz against the
|
||||
WARC/1.1 spec: each record a self-standing gzip member starting WARC/1.,
|
||||
Content-Length == block length, the \r\n\r\n trailer intact, the response
|
||||
body round-trips, and the encoding headers are stripped (strategy B). */
|
||||
static int st_warc(httrackp *opt, int argc, char **argv) {
|
||||
char path[HTS_URLMAXSIZE];
|
||||
warc_writer *w;
|
||||
unsigned char *data;
|
||||
size_t data_len = 0;
|
||||
const unsigned char *p, *end;
|
||||
int err = 0, nrec = 0, nresp = 0, nreq = 0, nrevisit = 0, ninfo = 0;
|
||||
int seen_a_body = 0, body_occurrences = 0, a2_bodyless = 0, nm_cl_ok = 0;
|
||||
static const char a_body[] = "Hello, WARC!\n";
|
||||
|
||||
if (argc < 1) {
|
||||
fprintf(stderr, "warc: needs a writable directory\n");
|
||||
return 1;
|
||||
}
|
||||
fconcat(path, sizeof(path), argv[0], "warc-selftest.warc.gz");
|
||||
|
||||
w = warc_open(opt, path);
|
||||
assertf(w != NULL);
|
||||
|
||||
/* 200 HTML: bogus Content-Length + gzip/chunked encodings must be stripped.
|
||||
*/
|
||||
warc_write_transaction(
|
||||
w, "http://test.local/a.html", "127.0.0.1",
|
||||
"GET /a.html HTTP/1.1\r\nHost: test.local\r\n\r\n",
|
||||
"HTTP/1.1 200 OK\r\nContent-Type: text/html\r\nContent-Encoding: "
|
||||
"gzip\r\nTransfer-Encoding: chunked\r\nContent-Length: 999\r\n\r\n",
|
||||
a_body, sizeof(a_body) - 1, NULL, 200, 0, 0);
|
||||
|
||||
/* 302 redirect: header-only, no body. */
|
||||
warc_write_transaction(
|
||||
w, "http://test.local/r", "127.0.0.1",
|
||||
"GET /r HTTP/1.1\r\nHost: test.local\r\n\r\n",
|
||||
"HTTP/1.1 302 Found\r\nLocation: http://test.local/a.html\r\n\r\n", NULL,
|
||||
0, NULL, 302, 0, 0);
|
||||
|
||||
/* 200 binary, chunked coding on the wire (already de-chunked here). */
|
||||
warc_write_transaction(
|
||||
w, "http://test.local/b.bin", "127.0.0.1",
|
||||
"GET /b.bin HTTP/1.1\r\nHost: test.local\r\n\r\n",
|
||||
"HTTP/1.1 200 OK\r\nContent-Type: application/octet-stream\r\n"
|
||||
"Transfer-Encoding: chunked\r\n\r\n",
|
||||
"\x00\x01\x02\x03\x04", 5, NULL, 200, 0, 0);
|
||||
|
||||
/* 200 with a body shorter than the declared Content-Length (rewritten). */
|
||||
warc_write_transaction(
|
||||
w, "http://test.local/trunc", "127.0.0.1",
|
||||
"GET /trunc HTTP/1.1\r\nHost: test.local\r\n\r\n",
|
||||
"HTTP/1.1 200 OK\r\nContent-Type: text/plain\r\nContent-Length: "
|
||||
"100\r\n\r\n",
|
||||
"short", 5, NULL, 200, 0, 0);
|
||||
|
||||
/* Same payload as a.html at a new URL: identical-payload-digest revisit
|
||||
(OpenSSL builds only; a plain build writes a second full response). */
|
||||
warc_write_transaction(w, "http://test.local/a2.html", "127.0.0.1",
|
||||
"GET /a2.html HTTP/1.1\r\nHost: test.local\r\n\r\n",
|
||||
"HTTP/1.1 200 OK\r\nContent-Type: text/html\r\n\r\n",
|
||||
a_body, sizeof(a_body) - 1, NULL, 200, 0, 0);
|
||||
|
||||
/* 304 revisit with an EMPTY response-header block: the block is just the
|
||||
2-byte separator, so declared Content-Length must be exactly 2 (F3). */
|
||||
warc_write_transaction(w, "http://test.local/nm", "127.0.0.1",
|
||||
"GET /nm HTTP/1.1\r\nHost: test.local\r\n\r\n", "",
|
||||
NULL, 0, NULL, 304, 1, 0);
|
||||
|
||||
warc_close(w);
|
||||
|
||||
data = warc_slurp(path, &data_len);
|
||||
assertf(data != NULL);
|
||||
p = data;
|
||||
end = data + data_len;
|
||||
|
||||
while (p < end) {
|
||||
size_t rlen = 0;
|
||||
unsigned char *rec = warc_next_member(&p, end, &rlen);
|
||||
const char *sep, *cl;
|
||||
long long block_len = 0; /* 0 on a parse failure; err is already set */
|
||||
size_t hdr_len;
|
||||
if (rec == NULL) {
|
||||
if (rlen == 0)
|
||||
break; /* clean end */
|
||||
err = 1;
|
||||
break;
|
||||
}
|
||||
nrec++;
|
||||
/* magic */
|
||||
if (rlen < 8 || memcmp(rec, "WARC/1.", 7) != 0)
|
||||
err = 1;
|
||||
/* record header ends at the first blank line */
|
||||
sep = warc_memstr((char *) rec, "\r\n\r\n", rlen, 4);
|
||||
if (sep == NULL) {
|
||||
err = 1;
|
||||
freet(rec);
|
||||
continue;
|
||||
}
|
||||
hdr_len = (size_t) ((const unsigned char *) sep - rec) + 4;
|
||||
/* Content-Length must equal the actual block length */
|
||||
cl = warc_memstr((char *) rec, "Content-Length:", hdr_len, 15);
|
||||
if (cl == NULL || sscanf(cl + 15, "%lld", &block_len) != 1)
|
||||
err = 1;
|
||||
else {
|
||||
if (hdr_len + (size_t) block_len + 4 != rlen)
|
||||
err = 1; /* header + block + trailing CRLFCRLF */
|
||||
else if (memcmp(rec + hdr_len + block_len, "\r\n\r\n", 4) != 0)
|
||||
err = 1; /* trailer intact */
|
||||
}
|
||||
if (warc_memstr((char *) rec, "WARC-Type: warcinfo", hdr_len, 19) != NULL)
|
||||
ninfo++;
|
||||
if (warc_memstr((char *) rec, "WARC-Type: request", hdr_len, 18) != NULL)
|
||||
nreq++;
|
||||
if (warc_memstr((char *) rec, "WARC-Type: response", hdr_len, 19) != NULL)
|
||||
nresp++;
|
||||
if (warc_memstr((char *) rec, "WARC-Type: revisit", hdr_len, 18) != NULL)
|
||||
nrevisit++;
|
||||
/* F1: the full body must appear exactly once across the whole file (a
|
||||
revisit must not re-embed it). */
|
||||
if (warc_memstr((char *) rec, a_body, rlen, sizeof(a_body) - 1) != NULL)
|
||||
body_occurrences++;
|
||||
/* F1: the a2.html identical-payload-digest revisit carries no body. */
|
||||
if (warc_memstr((char *) rec, "WARC-Target-URI: http://test.local/a2.html",
|
||||
hdr_len, 42) != NULL &&
|
||||
warc_memstr((char *) rec, "WARC-Type: revisit", hdr_len, 18) != NULL)
|
||||
a2_bodyless =
|
||||
(warc_memstr((char *) rec, a_body, rlen, sizeof(a_body) - 1) == NULL);
|
||||
/* F3: the empty-header 304 revisit block is exactly the 2-byte separator
|
||||
(the request record shares this target URI, so match the revisit only).
|
||||
*/
|
||||
if (warc_memstr((char *) rec, "WARC-Target-URI: http://test.local/nm",
|
||||
hdr_len, 37) != NULL &&
|
||||
warc_memstr((char *) rec, "WARC-Type: revisit", hdr_len, 18) != NULL)
|
||||
nm_cl_ok = (block_len == 2);
|
||||
/* a.html response body must round-trip and carry no encoding headers */
|
||||
if (warc_memstr((char *) rec, "WARC-Target-URI: http://test.local/a.html",
|
||||
hdr_len, 41) != NULL &&
|
||||
warc_memstr((char *) rec, "msgtype=response", hdr_len, 16) != NULL) {
|
||||
const char *bsep = warc_memstr((char *) rec + hdr_len, "\r\n\r\n",
|
||||
(size_t) block_len, 4);
|
||||
if (bsep == NULL)
|
||||
err = 1;
|
||||
else {
|
||||
size_t bodyoff = (size_t) (bsep - (char *) rec) + 4;
|
||||
size_t got = rlen - 4 - bodyoff; /* minus record trailer */
|
||||
if (got != sizeof(a_body) - 1 ||
|
||||
memcmp(rec + bodyoff, a_body, got) != 0)
|
||||
err = 1;
|
||||
seen_a_body = 1;
|
||||
}
|
||||
if (warc_memstr((char *) rec, "Content-Encoding", hdr_len + block_len,
|
||||
16) != NULL ||
|
||||
warc_memstr((char *) rec, "Transfer-Encoding", hdr_len + block_len,
|
||||
17) != NULL)
|
||||
err = 1;
|
||||
}
|
||||
freet(rec);
|
||||
}
|
||||
freet(data);
|
||||
|
||||
/* warcinfo + 6 transactions (response/revisit + request each) = 13 records.
|
||||
*/
|
||||
if (ninfo != 1 || nreq != 6 || nrec != 13 || !seen_a_body || !nm_cl_ok)
|
||||
err = 1;
|
||||
#if HTS_USEOPENSSL
|
||||
/* a.html + b.bin + trunc + 302 are full responses; a2.html deduped to a
|
||||
revisit (bodyless), nm is the 304 revisit; the body appears exactly once.
|
||||
*/
|
||||
if (nrevisit != 2 || nresp != 4 || !a2_bodyless || body_occurrences != 1)
|
||||
err = 1;
|
||||
#else
|
||||
/* No digests: a2.html is a second full response, so the body appears twice
|
||||
and only the 304 nm is a revisit. */
|
||||
if (nrevisit != 1 || nresp != 5 || body_occurrences != 2)
|
||||
err = 1;
|
||||
(void) a2_bodyless; /* only meaningful with digests */
|
||||
#endif
|
||||
|
||||
printf("warc: %d records (%d response, %d request, %d revisit): %s\n", nrec,
|
||||
nresp, nreq, nrevisit, err ? "FAIL" : "OK");
|
||||
return err;
|
||||
}
|
||||
|
||||
/* Parse a record's header/block split; sets *hdr_len and *block_len, returns 0
|
||||
when Content-Length matches the actual block bytes, -1 otherwise. */
|
||||
static int warc_rec_split(const unsigned char *rec, size_t rlen,
|
||||
size_t *hdr_len, long long *block_len) {
|
||||
const char *sep = warc_memstr((const char *) rec, "\r\n\r\n", rlen, 4);
|
||||
const char *cl;
|
||||
*block_len = 0;
|
||||
if (sep == NULL)
|
||||
return -1;
|
||||
*hdr_len = (size_t) ((const unsigned char *) sep - rec) + 4;
|
||||
cl = warc_memstr((const char *) rec, "Content-Length:", *hdr_len, 15);
|
||||
if (cl == NULL || sscanf(cl + 15, "%lld", block_len) != 1 ||
|
||||
*hdr_len + (size_t) *block_len + 4 != rlen)
|
||||
return -1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* A cap-truncated body is still archived, tagged WARC-Truncated (v1.1). Assert
|
||||
the sole response carries "WARC-Truncated: length" and every record parses.
|
||||
*/
|
||||
static int st_warc_trunc(httrackp *opt, int argc, char **argv) {
|
||||
char path[HTS_URLMAXSIZE];
|
||||
warc_writer *w;
|
||||
unsigned char *data;
|
||||
size_t data_len = 0;
|
||||
const unsigned char *p, *end;
|
||||
int err = 0, found_trunc = 0, nresp = 0;
|
||||
static const char body[] = "partial body bytes\n";
|
||||
|
||||
if (argc < 1) {
|
||||
fprintf(stderr, "warc-trunc: needs a writable directory\n");
|
||||
return 1;
|
||||
}
|
||||
fconcat(path, sizeof(path), argv[0], "warc-trunc.warc.gz");
|
||||
w = warc_open(opt, path);
|
||||
assertf(w != NULL);
|
||||
warc_write_transaction(
|
||||
w, "http://test.local/big.bin", "127.0.0.1",
|
||||
"GET /big.bin HTTP/1.1\r\nHost: test.local\r\n\r\n",
|
||||
"HTTP/1.1 200 OK\r\nContent-Type: application/octet-stream\r\n\r\n", body,
|
||||
sizeof(body) - 1, NULL, 200, 0, WARC_TRUNC_LENGTH);
|
||||
warc_close(w);
|
||||
|
||||
data = warc_slurp(path, &data_len);
|
||||
assertf(data != NULL);
|
||||
p = data;
|
||||
end = data + data_len;
|
||||
while (p < end) {
|
||||
size_t rlen = 0, hdr_len = 0;
|
||||
long long block_len = 0;
|
||||
unsigned char *rec = warc_next_member(&p, end, &rlen);
|
||||
if (rec == NULL) {
|
||||
if (rlen != 0)
|
||||
err = 1;
|
||||
break;
|
||||
}
|
||||
if (warc_rec_split(rec, rlen, &hdr_len, &block_len) != 0)
|
||||
err = 1;
|
||||
else if (warc_memstr((char *) rec, "WARC-Type: response", hdr_len, 19) !=
|
||||
NULL) {
|
||||
nresp++;
|
||||
if (warc_memstr((char *) rec, "WARC-Truncated: length", hdr_len, 22) !=
|
||||
NULL)
|
||||
found_trunc = 1;
|
||||
}
|
||||
freet(rec);
|
||||
}
|
||||
freet(data);
|
||||
if (!found_trunc || nresp != 1)
|
||||
err = 1;
|
||||
printf("warc-trunc: %s\n", err ? "FAIL" : "OK");
|
||||
return err;
|
||||
}
|
||||
|
||||
/* An ftp:// capture is ONE resource record: WARC-Type: resource, the payload's
|
||||
own Content-Type, block == payload, and no request/response pair. */
|
||||
static int st_warc_ftp(httrackp *opt, int argc, char **argv) {
|
||||
char path[HTS_URLMAXSIZE];
|
||||
warc_writer *w;
|
||||
unsigned char *data;
|
||||
size_t data_len = 0;
|
||||
const unsigned char *p, *end;
|
||||
int err = 0, nresource = 0, nresp = 0, nreq = 0;
|
||||
static const char body[] = "\x00\x01"
|
||||
"FTP payload"
|
||||
"\x02\x03";
|
||||
|
||||
if (argc < 1) {
|
||||
fprintf(stderr, "warc-ftp: needs a writable directory\n");
|
||||
return 1;
|
||||
}
|
||||
fconcat(path, sizeof(path), argv[0], "warc-ftp.warc.gz");
|
||||
w = warc_open(opt, path);
|
||||
assertf(w != NULL);
|
||||
warc_write_resource(w, "ftp://ftp.local/file.bin", "127.0.0.1",
|
||||
"application/octet-stream", body, sizeof(body) - 1, NULL,
|
||||
0);
|
||||
warc_close(w);
|
||||
|
||||
data = warc_slurp(path, &data_len);
|
||||
assertf(data != NULL);
|
||||
p = data;
|
||||
end = data + data_len;
|
||||
while (p < end) {
|
||||
size_t rlen = 0, hdr_len = 0;
|
||||
long long block_len = 0;
|
||||
unsigned char *rec = warc_next_member(&p, end, &rlen);
|
||||
if (rec == NULL) {
|
||||
if (rlen != 0)
|
||||
err = 1;
|
||||
break;
|
||||
}
|
||||
if (warc_rec_split(rec, rlen, &hdr_len, &block_len) != 0)
|
||||
err = 1;
|
||||
if (warc_memstr((char *) rec, "WARC-Type: resource", hdr_len, 19) != NULL) {
|
||||
nresource++;
|
||||
if ((size_t) block_len != sizeof(body) - 1 ||
|
||||
memcmp(rec + hdr_len, body, sizeof(body) - 1) != 0)
|
||||
err = 1; /* block is the raw payload, no HTTP envelope */
|
||||
if (warc_memstr((char *) rec, "WARC-Target-URI: ftp://ftp.local/file.bin",
|
||||
hdr_len, 41) == NULL ||
|
||||
warc_memstr((char *) rec, "Content-Type: application/octet-stream",
|
||||
hdr_len, 38) == NULL)
|
||||
err = 1;
|
||||
}
|
||||
if (warc_memstr((char *) rec, "WARC-Type: response", hdr_len, 19) != NULL)
|
||||
nresp++;
|
||||
if (warc_memstr((char *) rec, "WARC-Type: request", hdr_len, 18) != NULL)
|
||||
nreq++;
|
||||
freet(rec);
|
||||
}
|
||||
freet(data);
|
||||
if (nresource != 1 || nresp != 0 || nreq != 0)
|
||||
err = 1;
|
||||
printf("warc-ftp: resource=%d response=%d request=%d: %s\n", nresource, nresp,
|
||||
nreq, err ? "FAIL" : "OK");
|
||||
return err;
|
||||
}
|
||||
|
||||
/* --warc-max-size rotates into <base>-00000.warc.gz, -00001, ...; each segment
|
||||
is independently valid and begins with its own warcinfo. */
|
||||
static int st_warc_rotate(httrackp *opt, int argc, char **argv) {
|
||||
char path[HTS_URLMAXSIZE];
|
||||
char seg[HTS_URLMAXSIZE];
|
||||
warc_writer *w;
|
||||
LLint saved_max;
|
||||
unsigned char body[600];
|
||||
unsigned int rng = 0x12345678u;
|
||||
int err = 0, nseg = 0, i;
|
||||
size_t j;
|
||||
|
||||
if (argc < 1) {
|
||||
fprintf(stderr, "warc-rotate: needs a writable directory\n");
|
||||
return 1;
|
||||
}
|
||||
for (j = 0; j < sizeof(body);
|
||||
j++) { /* incompressible: gzip can't shrink it */
|
||||
rng ^= rng << 13;
|
||||
rng ^= rng >> 17;
|
||||
rng ^= rng << 5;
|
||||
body[j] = (unsigned char) (rng >> 24);
|
||||
}
|
||||
fconcat(path, sizeof(path), argv[0], "warc-rot.warc.gz");
|
||||
saved_max = opt->warc_max_size;
|
||||
opt->warc_max_size =
|
||||
1000; /* a couple records per segment => several segments */
|
||||
w = warc_open(opt, path);
|
||||
assertf(w != NULL);
|
||||
for (i = 0; i < 8; i++) {
|
||||
char uri[64];
|
||||
snprintf(uri, sizeof(uri), "http://test.local/f%d.bin", i);
|
||||
warc_write_transaction(
|
||||
w, uri, "127.0.0.1", "GET / HTTP/1.1\r\nHost: test.local\r\n\r\n",
|
||||
"HTTP/1.1 200 OK\r\nContent-Type: application/octet-stream\r\n\r\n",
|
||||
(const char *) body, sizeof(body), NULL, 200, 0, 0);
|
||||
}
|
||||
warc_close(w);
|
||||
opt->warc_max_size = saved_max;
|
||||
|
||||
for (i = 0;; i++) {
|
||||
char fname[64];
|
||||
unsigned char *data;
|
||||
size_t data_len = 0;
|
||||
const unsigned char *p, *pend;
|
||||
int first = 1;
|
||||
snprintf(fname, sizeof(fname), "warc-rot-%05d.warc.gz", i);
|
||||
fconcat(seg, sizeof(seg), argv[0], fname);
|
||||
data = warc_slurp(seg, &data_len);
|
||||
if (data == NULL)
|
||||
break; /* past the last segment */
|
||||
nseg++;
|
||||
p = data;
|
||||
pend = data + data_len;
|
||||
while (p < pend) {
|
||||
size_t rlen = 0, hdr_len = 0;
|
||||
long long block_len = 0;
|
||||
unsigned char *rec = warc_next_member(&p, pend, &rlen);
|
||||
if (rec == NULL) {
|
||||
if (rlen != 0)
|
||||
err = 1;
|
||||
break;
|
||||
}
|
||||
if (warc_rec_split(rec, rlen, &hdr_len, &block_len) != 0)
|
||||
err = 1;
|
||||
if (first) { /* each segment leads with its own warcinfo */
|
||||
if (warc_memstr((char *) rec, "WARC-Type: warcinfo", hdr_len, 19) ==
|
||||
NULL)
|
||||
err = 1;
|
||||
first = 0;
|
||||
}
|
||||
freet(rec);
|
||||
}
|
||||
freet(data);
|
||||
if (first) /* empty segment */
|
||||
err = 1;
|
||||
}
|
||||
if (nseg < 2)
|
||||
err = 1;
|
||||
printf("warc-rotate: %d segments: %s\n", nseg, err ? "FAIL" : "OK");
|
||||
return err;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------ */
|
||||
/* Registry: name -> handler, with a usage hint and a one-line description. */
|
||||
/* ------------------------------------------------------------ */
|
||||
@@ -3875,14 +3375,6 @@ static const struct selftest_entry {
|
||||
{"ftp-line", "", "get_ftp_line bounds a hostile FTP reply line",
|
||||
st_ftpline},
|
||||
{"ftp-userpass", "", "ftp_split_userpass bounds URL userinfo", st_ftpuser},
|
||||
{"warc", "<dir>", "WARC/1.1 writer: framing, digests, revisit dedup",
|
||||
st_warc},
|
||||
{"warc-trunc", "<dir>", "WARC-Truncated on a cap-truncated body",
|
||||
st_warc_trunc},
|
||||
{"warc-ftp", "<dir>", "ftp resource record (no HTTP envelope)",
|
||||
st_warc_ftp},
|
||||
{"warc-rotate", "<dir>", "--warc-max-size segment rotation",
|
||||
st_warc_rotate},
|
||||
};
|
||||
|
||||
static void list_selftests(void) {
|
||||
|
||||
995
src/htswarc.c
995
src/htswarc.c
@@ -1,995 +0,0 @@
|
||||
/* ------------------------------------------------------------ */
|
||||
/*
|
||||
HTTrack Website Copier, Offline Browser for Windows and Unix
|
||||
Copyright (C) 2026 Xavier Roche and other contributors
|
||||
|
||||
SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License
|
||||
along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
Ethical use: we kindly ask that you NOT use this software to harvest email
|
||||
addresses or to collect any other private information about people. Doing so
|
||||
would dishonor our work and waste the many hours we have spent on it.
|
||||
|
||||
Please visit our Website: http://www.httrack.com
|
||||
*/
|
||||
|
||||
/* ------------------------------------------------------------ */
|
||||
/* HTTrack WARC/1.1 output writer (ISO 28500). See warc.h.
|
||||
Strategy B (see design): the response record stores the decoded body and a
|
||||
normalized header block (Content-Encoding/Transfer-Encoding stripped,
|
||||
Content-Length rewritten to the decoded length). Valid, replayable WARC. */
|
||||
/* ------------------------------------------------------------ */
|
||||
|
||||
#define HTS_INTERNAL_BYTECODE
|
||||
|
||||
#include "htswarc.h"
|
||||
|
||||
#include "htscore.h"
|
||||
#include "htslib.h"
|
||||
#include "htstools.h"
|
||||
#include "htssafe.h"
|
||||
#include "htszlib.h"
|
||||
#include "coucal/coucal.h"
|
||||
|
||||
#include <stdarg.h>
|
||||
#include <stdint.h>
|
||||
#include <time.h>
|
||||
|
||||
#if HTS_USEOPENSSL
|
||||
#include <openssl/evp.h>
|
||||
#include <openssl/rand.h>
|
||||
#endif
|
||||
|
||||
/* opt->state.warc value meaning "open failed once, do not retry". */
|
||||
#define WARC_DISABLED ((void *) ~(uintptr_t) 0)
|
||||
|
||||
struct warc_writer {
|
||||
FILE *f;
|
||||
int gz; /* 1: one gzip member per record; 0: raw */
|
||||
uint64_t offset; /* running byte offset (member starts, for a future index) */
|
||||
uint64_t counter; /* monotonic record counter */
|
||||
uint64_t rng; /* PRNG state for the UUID fallback */
|
||||
char info_id[64]; /* warcinfo WARC-Record-ID, referenced by every record */
|
||||
coucal seen; /* base32 payload digest -> "uri\001date" (revisit dedup) */
|
||||
/* --warc-max-size rotation: NAME-00000.warc.gz, -00001, ... (wget-style). */
|
||||
uint64_t max_size; /* rotate once a segment reaches this; 0: single file */
|
||||
char *seg_base; /* segment path without the .warc[.gz] suffix, or NULL */
|
||||
const char *seg_ext; /* ".warc.gz" or ".warc" */
|
||||
unsigned seg; /* current segment number */
|
||||
char *info_fields; /* warcinfo body, re-emitted at each new segment */
|
||||
};
|
||||
|
||||
const char *warc_truncated_reason(int code) {
|
||||
switch (code) {
|
||||
case WARC_TRUNC_LENGTH:
|
||||
return "length";
|
||||
case WARC_TRUNC_TIME:
|
||||
return "time";
|
||||
case WARC_TRUNC_DISCONNECT:
|
||||
return "disconnect";
|
||||
default:
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- growable byte buffer (overflow-safe, project allocators) ---- */
|
||||
|
||||
typedef struct {
|
||||
char *data;
|
||||
size_t len;
|
||||
size_t cap;
|
||||
} wbuf;
|
||||
|
||||
static void wbuf_free(wbuf *b) {
|
||||
freet(b->data);
|
||||
b->len = b->cap = 0;
|
||||
}
|
||||
|
||||
/* Append n bytes; returns 0 on success, -1 on OOM/overflow. */
|
||||
static int wbuf_add(wbuf *b, const void *p, size_t n) {
|
||||
if (n > (size_t) -1 - b->len)
|
||||
return -1;
|
||||
if (b->len + n > b->cap) {
|
||||
size_t ncap = b->cap ? b->cap : 256;
|
||||
char *nd;
|
||||
while (ncap < b->len + n) {
|
||||
if (ncap > (size_t) -1 / 2)
|
||||
return -1;
|
||||
ncap *= 2;
|
||||
}
|
||||
nd = realloct(b->data, ncap);
|
||||
if (nd == NULL)
|
||||
return -1;
|
||||
b->data = nd;
|
||||
b->cap = ncap;
|
||||
}
|
||||
memcpy(b->data + b->len, p, n);
|
||||
b->len += n;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int wbuf_puts(wbuf *b, const char *s) {
|
||||
return wbuf_add(b, s, strlen(s));
|
||||
}
|
||||
|
||||
static int wbuf_printf(wbuf *b, const char *fmt, ...) HTS_PRINTF_FUN(2, 3);
|
||||
|
||||
static int wbuf_printf(wbuf *b, const char *fmt, ...) {
|
||||
char tmp[1024];
|
||||
int n;
|
||||
va_list ap;
|
||||
va_start(ap, fmt);
|
||||
n = vsnprintf(tmp, sizeof(tmp), fmt, ap);
|
||||
va_end(ap);
|
||||
if (n < 0 || (size_t) n >= sizeof(tmp))
|
||||
return -1;
|
||||
return wbuf_add(b, tmp, (size_t) n);
|
||||
}
|
||||
|
||||
/* ---- gzip-per-record member writer (mirrors ae_write_packed) ---- */
|
||||
|
||||
typedef struct {
|
||||
warc_writer *w;
|
||||
z_stream strm;
|
||||
int active; /* deflate stream initialized */
|
||||
} member;
|
||||
|
||||
static int member_begin(member *m, warc_writer *w) {
|
||||
m->w = w;
|
||||
m->active = 0;
|
||||
if (w->gz) {
|
||||
memset(&m->strm, 0, sizeof(m->strm));
|
||||
/* windowBits=31 => full RFC1952 gzip member */
|
||||
if (deflateInit2(&m->strm, Z_DEFAULT_COMPRESSION, Z_DEFLATED, 31, 8,
|
||||
Z_DEFAULT_STRATEGY) != Z_OK)
|
||||
return -1;
|
||||
m->active = 1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int member_write(member *m, const void *p, size_t n) {
|
||||
if (!m->w->gz)
|
||||
return (n == 0 || fwrite(p, 1, n, m->w->f) == n) ? 0 : -1;
|
||||
m->strm.next_in = (const Bytef *) p;
|
||||
while (n > 0) {
|
||||
unsigned char out[8192];
|
||||
size_t got;
|
||||
uInt chunk = (n > (uInt) -1) ? (uInt) -1 : (uInt) n;
|
||||
m->strm.avail_in = chunk;
|
||||
do {
|
||||
m->strm.next_out = out;
|
||||
m->strm.avail_out = sizeof(out);
|
||||
if (deflate(&m->strm, Z_NO_FLUSH) != Z_OK)
|
||||
return -1;
|
||||
got = sizeof(out) - m->strm.avail_out;
|
||||
if (got > 0 && fwrite(out, 1, got, m->w->f) != got)
|
||||
return -1;
|
||||
} while (m->strm.avail_out == 0);
|
||||
n -= chunk;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int member_end(member *m) {
|
||||
int rc = 0;
|
||||
if (m->active) {
|
||||
unsigned char out[8192];
|
||||
int zerr;
|
||||
m->strm.avail_in = 0;
|
||||
do {
|
||||
m->strm.next_out = out;
|
||||
m->strm.avail_out = sizeof(out);
|
||||
zerr = deflate(&m->strm, Z_FINISH);
|
||||
{
|
||||
size_t got = sizeof(out) - m->strm.avail_out;
|
||||
if (got > 0 && fwrite(out, 1, got, m->w->f) != got)
|
||||
rc = -1;
|
||||
}
|
||||
} while (zerr == Z_OK);
|
||||
if (zerr != Z_STREAM_END)
|
||||
rc = -1;
|
||||
deflateEnd(&m->strm);
|
||||
m->active = 0;
|
||||
}
|
||||
return rc;
|
||||
}
|
||||
|
||||
/* ---- SHA-1 + Base32 (digests are OpenSSL-only; omitted otherwise) ---- */
|
||||
|
||||
#if HTS_USEOPENSSL
|
||||
/* Streaming SHA-1 over the block (all regions) and the payload (body only). */
|
||||
typedef struct {
|
||||
EVP_MD_CTX *block;
|
||||
EVP_MD_CTX *payload;
|
||||
} digester;
|
||||
|
||||
static void base32_20(const unsigned char in[20], char out[33]) {
|
||||
static const char a[] = "ABCDEFGHIJKLMNOPQRSTUVWXYZ234567";
|
||||
int i, o = 0;
|
||||
uint64_t buf = 0;
|
||||
int bits = 0;
|
||||
for (i = 0; i < 20; i++) {
|
||||
buf = (buf << 8) | in[i];
|
||||
bits += 8;
|
||||
while (bits >= 5) {
|
||||
bits -= 5;
|
||||
out[o++] = a[(buf >> bits) & 0x1F];
|
||||
}
|
||||
}
|
||||
out[o] = '\0'; /* 20 bytes => exactly 32 base32 chars, no padding */
|
||||
}
|
||||
#endif
|
||||
|
||||
/* Stream a block to a sink. When http_section, the HTTP header bytes (may be
|
||||
empty) plus their terminating CRLF are emitted first; the separator is bound
|
||||
to http_section, not to http_hdr being non-NULL, so an empty header still
|
||||
emits (and is counted in) the 2-byte separator (F3). The on-disk body is
|
||||
written as EXACTLY body_len octets — capped if the file grew, zero-padded if
|
||||
it shrank — so the declared Content-Length always equals the bytes written
|
||||
across every pass (F2). region 0=header, 1=body. Returns 0 on success. */
|
||||
typedef int (*warc_sink)(void *ctx, int region, const void *p, size_t n);
|
||||
|
||||
static int stream_body_pad(warc_sink sink, void *ctx, size_t remaining) {
|
||||
static const char zeros[4096] = {0};
|
||||
while (remaining > 0) {
|
||||
size_t chunk = (remaining < sizeof(zeros)) ? remaining : sizeof(zeros);
|
||||
if (sink(ctx, 1, zeros, chunk) != 0)
|
||||
return -1;
|
||||
remaining -= chunk;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int stream_block(int http_section, const char *http_hdr,
|
||||
size_t http_hdr_len, int has_body, const char *body,
|
||||
size_t body_len, const char *body_path, warc_sink sink,
|
||||
void *ctx) {
|
||||
if (http_section) {
|
||||
if (http_hdr != NULL && http_hdr_len > 0 &&
|
||||
sink(ctx, 0, http_hdr, http_hdr_len) != 0)
|
||||
return -1;
|
||||
if (sink(ctx, 0, "\r\n", 2) != 0)
|
||||
return -1;
|
||||
}
|
||||
if (has_body) {
|
||||
if (body != NULL) {
|
||||
if (body_len > 0 && sink(ctx, 1, body, body_len) != 0)
|
||||
return -1;
|
||||
} else if (body_path != NULL) {
|
||||
char catbuff[CATBUFF_SIZE];
|
||||
size_t remaining = body_len;
|
||||
FILE *fp = FOPEN(fconv(catbuff, sizeof(catbuff), body_path), "rb");
|
||||
if (fp == NULL)
|
||||
return -1;
|
||||
while (remaining > 0) {
|
||||
char b[32768];
|
||||
size_t want = (remaining < sizeof(b)) ? remaining : sizeof(b);
|
||||
size_t nl = fread(b, 1, want, fp);
|
||||
if (nl == 0)
|
||||
break; /* short file: pad below so written == declared */
|
||||
if (sink(ctx, 1, b, nl) != 0) {
|
||||
fclose(fp);
|
||||
return -1;
|
||||
}
|
||||
remaining -= nl;
|
||||
}
|
||||
fclose(fp);
|
||||
if (stream_body_pad(sink, ctx, remaining) != 0)
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
#if HTS_USEOPENSSL
|
||||
static int digest_sink(void *ctx, int region, const void *p, size_t n) {
|
||||
digester *d = (digester *) ctx;
|
||||
if (d->block != NULL && EVP_DigestUpdate(d->block, p, n) != 1)
|
||||
return -1;
|
||||
if (region == 1 && d->payload != NULL &&
|
||||
EVP_DigestUpdate(d->payload, p, n) != 1)
|
||||
return -1;
|
||||
return 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
static int write_sink(void *ctx, int region, const void *p, size_t n) {
|
||||
(void) region;
|
||||
return member_write((member *) ctx, p, n);
|
||||
}
|
||||
|
||||
/* Base32 SHA-1 of a transaction payload (body only), for revisit dedup.
|
||||
Returns 1 and fills out[33] on success, 0 without OpenSSL or on error. */
|
||||
static int payload_digest_b32(const char *body, size_t body_len,
|
||||
const char *body_path, char out[33]) {
|
||||
#if HTS_USEOPENSSL
|
||||
digester d;
|
||||
unsigned char md[EVP_MAX_MD_SIZE];
|
||||
unsigned int mdlen = 0;
|
||||
int ok;
|
||||
d.block = NULL;
|
||||
d.payload = EVP_MD_CTX_new();
|
||||
if (d.payload == NULL)
|
||||
return 0;
|
||||
if (EVP_DigestInit_ex(d.payload, EVP_sha1(), NULL) != 1) {
|
||||
EVP_MD_CTX_free(d.payload);
|
||||
return 0;
|
||||
}
|
||||
ok = stream_block(0, NULL, 0, 1, body, body_len, body_path, digest_sink,
|
||||
&d) == 0 &&
|
||||
EVP_DigestFinal_ex(d.payload, md, &mdlen) == 1 && mdlen == 20;
|
||||
EVP_MD_CTX_free(d.payload);
|
||||
if (!ok)
|
||||
return 0;
|
||||
base32_20(md, out);
|
||||
return 1;
|
||||
#else
|
||||
(void) body;
|
||||
(void) body_len;
|
||||
(void) body_path;
|
||||
(void) out;
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
/* ---- misc record helpers ---- */
|
||||
|
||||
static void warc_fill_random(warc_writer *w, unsigned char *b, size_t n) {
|
||||
size_t i;
|
||||
#if HTS_USEOPENSSL
|
||||
if (n <= (size_t) 0x7fffffff && RAND_bytes(b, (int) n) == 1)
|
||||
return;
|
||||
#endif
|
||||
for (i = 0; i < n; i++) {
|
||||
w->rng ^= w->rng << 13;
|
||||
w->rng ^= w->rng >> 7;
|
||||
w->rng ^= w->rng << 17;
|
||||
b[i] = (unsigned char) (w->rng >> 24);
|
||||
}
|
||||
}
|
||||
|
||||
/* urn:uuid: v4-shaped record id (uniqueness within the run is what matters). */
|
||||
static void warc_make_id(warc_writer *w, char out[64]) {
|
||||
unsigned char b[16];
|
||||
w->counter++;
|
||||
warc_fill_random(w, b, sizeof(b));
|
||||
b[6] = (unsigned char) ((b[6] & 0x0F) | 0x40);
|
||||
b[8] = (unsigned char) ((b[8] & 0x3F) | 0x80);
|
||||
snprintf(out, 64,
|
||||
"<urn:uuid:%02x%02x%02x%02x-%02x%02x-%02x%02x-%02x%02x-"
|
||||
"%02x%02x%02x%02x%02x%02x>",
|
||||
b[0], b[1], b[2], b[3], b[4], b[5], b[6], b[7], b[8], b[9], b[10],
|
||||
b[11], b[12], b[13], b[14], b[15]);
|
||||
}
|
||||
|
||||
static void warc_now_iso8601(char out[32]) {
|
||||
time_t t = time(NULL);
|
||||
struct tm tmv;
|
||||
#if defined(_WIN32)
|
||||
struct tm *g = gmtime(&t);
|
||||
if (g != NULL)
|
||||
tmv = *g;
|
||||
else
|
||||
memset(&tmv, 0, sizeof(tmv));
|
||||
#else
|
||||
if (gmtime_r(&t, &tmv) == NULL)
|
||||
memset(&tmv, 0, sizeof(tmv));
|
||||
#endif
|
||||
strftime(out, 32, "%Y-%m-%dT%H:%M:%SZ", &tmv);
|
||||
}
|
||||
|
||||
/* Case-insensitive "does line start with name:" test. */
|
||||
static int header_is(const char *line, size_t line_len, const char *name) {
|
||||
size_t nl = strlen(name);
|
||||
if (line_len < nl + 1)
|
||||
return 0;
|
||||
if (strncasecmp(line, name, nl) != 0)
|
||||
return 0;
|
||||
return line[nl] == ':';
|
||||
}
|
||||
|
||||
/* Build the normalized HTTP header block from raw resp_hdr into out (no
|
||||
trailing CRLF terminator). Always drops Transfer-Encoding (hop-by-hop).
|
||||
When set_cl>=0, also drops Content-Encoding and the original Content-Length
|
||||
and appends "Content-Length: <set_cl>". Returns 0 on success. */
|
||||
static int normalize_http_headers(const char *resp_hdr, long long set_cl,
|
||||
wbuf *out) {
|
||||
const char *p = resp_hdr;
|
||||
int first = 1;
|
||||
if (resp_hdr == NULL)
|
||||
return -1;
|
||||
while (*p != '\0') {
|
||||
const char *eol = strchr(p, '\n');
|
||||
size_t len = (eol != NULL) ? (size_t) (eol - p) : strlen(p);
|
||||
size_t raw = len;
|
||||
if (len > 0 && p[len - 1] == '\r')
|
||||
len--; /* strip CR; re-added as CRLF below */
|
||||
if (len == 0)
|
||||
break; /* blank line: end of headers */
|
||||
if (first) {
|
||||
first = 0; /* status line: keep verbatim */
|
||||
} else if (header_is(p, len, "Transfer-Encoding")) {
|
||||
goto next;
|
||||
} else if (set_cl >= 0 && (header_is(p, len, "Content-Encoding") ||
|
||||
header_is(p, len, "Content-Length"))) {
|
||||
goto next;
|
||||
}
|
||||
if (wbuf_add(out, p, len) != 0 || wbuf_add(out, "\r\n", 2) != 0)
|
||||
return -1;
|
||||
next:
|
||||
if (eol == NULL)
|
||||
break;
|
||||
p = eol + 1;
|
||||
(void) raw;
|
||||
}
|
||||
if (set_cl >= 0 && wbuf_printf(out, "Content-Length: %lld\r\n", set_cl) != 0)
|
||||
return -1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Close the current segment and open the next; writes its warcinfo. */
|
||||
static int warc_rotate(warc_writer *w);
|
||||
|
||||
/* Emit one full WARC record. When http_section, the block carries an HTTP
|
||||
header block (http_hdr, possibly empty) + a CRLF separator; body follows when
|
||||
has_body. block_len is derived here (single source of truth: separator and
|
||||
payload are counted exactly as stream_block emits them), so a declared
|
||||
Content-Length can never desync from the written bytes. The payload digest
|
||||
(body-only) is passed in when already known. truncated is a WARC-Truncated
|
||||
reason token or NULL. On success the record id is copied to out_id (may be
|
||||
NULL). */
|
||||
static int warc_emit(warc_writer *w, const char *type, const char *content_type,
|
||||
const char *target_uri, const char *ip,
|
||||
const char *concurrent_to, const char *refers_uri,
|
||||
const char *refers_date, const char *profile,
|
||||
const char *payload_digest, const char *truncated,
|
||||
int http_section, const char *http_hdr,
|
||||
size_t http_hdr_len, int has_body, const char *body,
|
||||
size_t body_len, const char *body_path, char out_id[64]) {
|
||||
wbuf hdr;
|
||||
member m;
|
||||
char id[64], date[32];
|
||||
size_t sep = http_section ? 2 : 0;
|
||||
size_t payload = has_body ? body_len : 0;
|
||||
size_t block_len;
|
||||
int rc = -1;
|
||||
#if HTS_USEOPENSSL
|
||||
digester d;
|
||||
unsigned char md[EVP_MAX_MD_SIZE];
|
||||
unsigned int mdlen = 0;
|
||||
char block_b32[33];
|
||||
int have_block_digest = 0;
|
||||
#endif
|
||||
|
||||
/* Rotate to the next segment before this record when the current one is full;
|
||||
never split a record, and never rotate a warcinfo (it opens a segment). */
|
||||
if (w->max_size > 0 && w->seg_base != NULL && w->offset >= w->max_size &&
|
||||
strcmp(type, "warcinfo") != 0) {
|
||||
if (warc_rotate(w) != 0)
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* F4: overflow-safe block length; http_hdr_len+sep is provably small. */
|
||||
if (payload > (size_t) -1 - http_hdr_len - sep)
|
||||
return -1;
|
||||
block_len = http_hdr_len + sep + payload;
|
||||
|
||||
memset(&hdr, 0, sizeof(hdr));
|
||||
warc_make_id(w, id);
|
||||
warc_now_iso8601(date);
|
||||
|
||||
#if HTS_USEOPENSSL
|
||||
/* Block digest over the whole block, in one streaming pass. */
|
||||
d.block = EVP_MD_CTX_new();
|
||||
d.payload = NULL;
|
||||
if (d.block != NULL && EVP_DigestInit_ex(d.block, EVP_sha1(), NULL) == 1 &&
|
||||
stream_block(http_section, http_hdr, http_hdr_len, has_body, body,
|
||||
body_len, body_path, digest_sink, &d) == 0 &&
|
||||
EVP_DigestFinal_ex(d.block, md, &mdlen) == 1 && mdlen == 20) {
|
||||
base32_20(md, block_b32);
|
||||
have_block_digest = 1;
|
||||
}
|
||||
if (d.block != NULL)
|
||||
EVP_MD_CTX_free(d.block);
|
||||
#endif
|
||||
|
||||
if (wbuf_puts(&hdr, "WARC/1.1\r\n") != 0 ||
|
||||
wbuf_printf(&hdr, "WARC-Type: %s\r\n", type) != 0 ||
|
||||
wbuf_printf(&hdr, "WARC-Record-ID: %s\r\n", id) != 0 ||
|
||||
wbuf_printf(&hdr, "WARC-Date: %s\r\n", date) != 0)
|
||||
goto done;
|
||||
if (content_type != NULL &&
|
||||
wbuf_printf(&hdr, "Content-Type: %s\r\n", content_type) != 0)
|
||||
goto done;
|
||||
if (wbuf_printf(&hdr, "Content-Length: %llu\r\n",
|
||||
(unsigned long long) block_len) != 0)
|
||||
goto done;
|
||||
if (w->info_id[0] != '\0' && strcmp(type, "warcinfo") != 0 &&
|
||||
wbuf_printf(&hdr, "WARC-Warcinfo-ID: %s\r\n", w->info_id) != 0)
|
||||
goto done;
|
||||
if (target_uri != NULL && target_uri[0] != '\0' &&
|
||||
wbuf_printf(&hdr, "WARC-Target-URI: %s\r\n", target_uri) != 0)
|
||||
goto done;
|
||||
if (ip != NULL && ip[0] != '\0' &&
|
||||
wbuf_printf(&hdr, "WARC-IP-Address: %s\r\n", ip) != 0)
|
||||
goto done;
|
||||
if (concurrent_to != NULL && concurrent_to[0] != '\0' &&
|
||||
wbuf_printf(&hdr, "WARC-Concurrent-To: %s\r\n", concurrent_to) != 0)
|
||||
goto done;
|
||||
if (profile != NULL &&
|
||||
wbuf_printf(&hdr, "WARC-Profile: %s\r\n", profile) != 0)
|
||||
goto done;
|
||||
if (refers_uri != NULL && refers_uri[0] != '\0' &&
|
||||
wbuf_printf(&hdr, "WARC-Refers-To-Target-URI: %s\r\n", refers_uri) != 0)
|
||||
goto done;
|
||||
if (refers_date != NULL && refers_date[0] != '\0' &&
|
||||
wbuf_printf(&hdr, "WARC-Refers-To-Date: %s\r\n", refers_date) != 0)
|
||||
goto done;
|
||||
#if HTS_USEOPENSSL
|
||||
if (have_block_digest &&
|
||||
wbuf_printf(&hdr, "WARC-Block-Digest: sha1:%s\r\n", block_b32) != 0)
|
||||
goto done;
|
||||
#endif
|
||||
if (payload_digest != NULL && payload_digest[0] != '\0' &&
|
||||
wbuf_printf(&hdr, "WARC-Payload-Digest: sha1:%s\r\n", payload_digest) !=
|
||||
0)
|
||||
goto done;
|
||||
if (truncated != NULL &&
|
||||
wbuf_printf(&hdr, "WARC-Truncated: %s\r\n", truncated) != 0)
|
||||
goto done;
|
||||
if (wbuf_puts(&hdr, "\r\n") != 0)
|
||||
goto done;
|
||||
|
||||
if (member_begin(&m, w) != 0)
|
||||
goto done;
|
||||
if (member_write(&m, hdr.data, hdr.len) != 0 ||
|
||||
stream_block(http_section, http_hdr, http_hdr_len, has_body, body,
|
||||
body_len, body_path, write_sink, &m) != 0 ||
|
||||
member_write(&m, "\r\n\r\n", 4) != 0) {
|
||||
member_end(&m);
|
||||
goto done;
|
||||
}
|
||||
if (member_end(&m) != 0)
|
||||
goto done;
|
||||
|
||||
{
|
||||
long pos = ftell(w->f);
|
||||
if (pos >= 0)
|
||||
w->offset = (uint64_t) pos;
|
||||
}
|
||||
if (out_id != NULL)
|
||||
strlcpybuff(out_id, id, 64);
|
||||
rc = 0;
|
||||
done:
|
||||
wbuf_free(&hdr);
|
||||
return rc;
|
||||
}
|
||||
|
||||
/* ---- segment rotation (--warc-max-size) ---- */
|
||||
|
||||
/* Emit the warcinfo that heads a segment; sets w->info_id for its records. */
|
||||
static int warc_write_warcinfo_record(warc_writer *w) {
|
||||
w->info_id[0] = '\0'; /* warcinfo itself carries no WARC-Warcinfo-ID */
|
||||
return warc_emit(w, "warcinfo", "application/warc-fields", NULL, NULL, NULL,
|
||||
NULL, NULL, NULL, NULL, NULL, 0, NULL, 0, 1, w->info_fields,
|
||||
w->info_fields != NULL ? strlen(w->info_fields) : 0, NULL,
|
||||
w->info_id);
|
||||
}
|
||||
|
||||
static int warc_rotate(warc_writer *w) {
|
||||
char namebuf[HTS_URLMAXSIZE * 2];
|
||||
char catbuff[CATBUFF_SIZE];
|
||||
if (w->f != NULL) {
|
||||
fclose(w->f);
|
||||
w->f = NULL;
|
||||
}
|
||||
w->seg++;
|
||||
snprintf(namebuf, sizeof(namebuf), "%s-%05u%s", w->seg_base, w->seg,
|
||||
w->seg_ext);
|
||||
w->f = FOPEN(fconv(catbuff, sizeof(catbuff), namebuf), "wb");
|
||||
if (w->f == NULL)
|
||||
return -1;
|
||||
w->offset = 0;
|
||||
return warc_write_warcinfo_record(w);
|
||||
}
|
||||
|
||||
/* ---- request stash (engine hooks) ---- */
|
||||
|
||||
void warc_stash_request(htsblk *r, const char *reqhdr) {
|
||||
if (r == NULL)
|
||||
return;
|
||||
freet(r->warc_reqhdr);
|
||||
if (reqhdr != NULL)
|
||||
r->warc_reqhdr = strdupt(reqhdr);
|
||||
}
|
||||
|
||||
void warc_stash_response(htsblk *r, const char *resphdr) {
|
||||
if (r == NULL)
|
||||
return;
|
||||
freet(r->warc_resphdr);
|
||||
if (resphdr != NULL)
|
||||
r->warc_resphdr = strdupt(resphdr);
|
||||
}
|
||||
|
||||
void warc_free_request(htsblk *r) {
|
||||
if (r != NULL) {
|
||||
freet(r->warc_reqhdr);
|
||||
freet(r->warc_resphdr);
|
||||
}
|
||||
}
|
||||
|
||||
/* ---- open / close ---- */
|
||||
|
||||
warc_writer *warc_open(httrackp *opt, const char *path) {
|
||||
warc_writer *w;
|
||||
char namebuf[HTS_URLMAXSIZE * 2];
|
||||
char catbuff[CATBUFF_SIZE];
|
||||
wbuf info;
|
||||
const char *robots;
|
||||
size_t plen;
|
||||
|
||||
if (path == NULL)
|
||||
return NULL;
|
||||
|
||||
/* --warc with no name: <output>/httrack-<timestamp>.warc.gz */
|
||||
if (strcmp(path, WARC_AUTONAME) == 0) {
|
||||
char ts[32];
|
||||
time_t t = time(NULL);
|
||||
struct tm tmv;
|
||||
#if defined(_WIN32)
|
||||
struct tm *g = gmtime(&t);
|
||||
if (g != NULL)
|
||||
tmv = *g;
|
||||
else
|
||||
memset(&tmv, 0, sizeof(tmv));
|
||||
#else
|
||||
if (gmtime_r(&t, &tmv) == NULL)
|
||||
memset(&tmv, 0, sizeof(tmv));
|
||||
#endif
|
||||
strftime(ts, sizeof(ts), "%Y%m%d%H%M%S", &tmv);
|
||||
snprintf(catbuff, sizeof(catbuff), "httrack-%s.warc.gz", ts);
|
||||
path =
|
||||
fconcat(namebuf, sizeof(namebuf), StringBuff(opt->path_html), catbuff);
|
||||
} else {
|
||||
/* --warc-file NAME: append .warc.gz unless already a .warc/.warc.gz name;
|
||||
place bare basenames under the output directory (like the auto name). */
|
||||
size_t l = strlen(path);
|
||||
int has_warc = (l >= 5 && strcasecmp(path + l - 5, ".warc") == 0);
|
||||
int has_gz = (l >= 3 && strcasecmp(path + l - 3, ".gz") == 0);
|
||||
char named[HTS_URLMAXSIZE];
|
||||
if (has_warc || has_gz)
|
||||
strlcpybuff(named, path, sizeof(named));
|
||||
else
|
||||
snprintf(named, sizeof(named), "%s.warc.gz", path);
|
||||
if (strchr(named, '/') == NULL && strchr(named, '\\') == NULL) {
|
||||
path =
|
||||
fconcat(namebuf, sizeof(namebuf), StringBuff(opt->path_html), named);
|
||||
} else {
|
||||
strlcpybuff(namebuf, named, sizeof(namebuf));
|
||||
path = namebuf;
|
||||
}
|
||||
}
|
||||
|
||||
w = calloct(1, sizeof(*w));
|
||||
if (w == NULL)
|
||||
return NULL;
|
||||
plen = strlen(path);
|
||||
w->gz = (plen >= 3 && strcasecmp(path + plen - 3, ".gz") == 0);
|
||||
w->rng = (uint64_t) time(NULL) ^ ((uint64_t) (uintptr_t) w << 16) ^
|
||||
0x9e3779b97f4a7c15ULL;
|
||||
w->seen = coucal_new(0);
|
||||
if (w->seen != NULL)
|
||||
coucal_value_is_malloc(w->seen, 1);
|
||||
w->max_size = (opt->warc_max_size > 0) ? (uint64_t) opt->warc_max_size : 0;
|
||||
|
||||
/* Build the warcinfo body once; each segment re-emits it. */
|
||||
robots = (opt->robots == HTS_ROBOTS_NEVER) ? "ignore" : "obey";
|
||||
memset(&info, 0, sizeof(info));
|
||||
if (wbuf_printf(&info,
|
||||
"software: HTTrack/%s (+https://www.httrack.com/)\r\n"
|
||||
"format: WARC file version 1.1\r\n"
|
||||
"conformsTo: http://iipc.github.io/warc-specifications/"
|
||||
"specifications/warc-format/warc-1.1/\r\n"
|
||||
"robots: %s\r\n",
|
||||
HTTRACK_VERSION, robots) != 0 ||
|
||||
(StringNotEmpty(opt->path_html) &&
|
||||
wbuf_printf(&info, "isPartOf: %s\r\n", StringBuff(opt->path_html)) !=
|
||||
0) ||
|
||||
wbuf_add(&info, "", 1) != 0) { /* NUL-terminate for info_fields */
|
||||
wbuf_free(&info);
|
||||
warc_close(w);
|
||||
return NULL;
|
||||
}
|
||||
w->info_fields = strdupt(info.data);
|
||||
wbuf_free(&info);
|
||||
if (w->info_fields == NULL) {
|
||||
warc_close(w);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Rotation on: the first segment is <base>-00000<ext> (wget-style); split the
|
||||
resolved path into base + suffix so later segments reuse the base. */
|
||||
if (w->max_size > 0) {
|
||||
size_t l = strlen(path);
|
||||
size_t baselen;
|
||||
if (l >= 8 && strcasecmp(path + l - 8, ".warc.gz") == 0) {
|
||||
baselen = l - 8;
|
||||
w->seg_ext = ".warc.gz";
|
||||
} else if (l >= 5 && strcasecmp(path + l - 5, ".warc") == 0) {
|
||||
baselen = l - 5;
|
||||
w->seg_ext = ".warc";
|
||||
} else {
|
||||
baselen = l;
|
||||
w->seg_ext = w->gz ? ".warc.gz" : ".warc";
|
||||
}
|
||||
w->seg_base = malloct(baselen + 1);
|
||||
if (w->seg_base == NULL) {
|
||||
warc_close(w);
|
||||
return NULL;
|
||||
}
|
||||
memcpy(w->seg_base, path, baselen);
|
||||
w->seg_base[baselen] = '\0';
|
||||
w->seg = 0;
|
||||
snprintf(namebuf, sizeof(namebuf), "%s-%05u%s", w->seg_base, w->seg,
|
||||
w->seg_ext);
|
||||
path = namebuf;
|
||||
}
|
||||
|
||||
w->f = FOPEN(fconv(catbuff, sizeof(catbuff), path), "wb");
|
||||
if (w->f == NULL) {
|
||||
warc_close(w);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
if (warc_write_warcinfo_record(w) != 0) {
|
||||
warc_close(w);
|
||||
return NULL;
|
||||
}
|
||||
return w;
|
||||
}
|
||||
|
||||
void warc_close(warc_writer *w) {
|
||||
if (w == NULL)
|
||||
return;
|
||||
if (w->f != NULL)
|
||||
fclose(w->f);
|
||||
if (w->seen != NULL)
|
||||
coucal_delete(&w->seen);
|
||||
freet(w->seg_base);
|
||||
freet(w->info_fields);
|
||||
freet(w);
|
||||
}
|
||||
|
||||
void warc_close_opt(httrackp *opt) {
|
||||
if (opt->state.warc != NULL && opt->state.warc != WARC_DISABLED) {
|
||||
warc_close((warc_writer *) opt->state.warc);
|
||||
}
|
||||
opt->state.warc = NULL;
|
||||
}
|
||||
|
||||
/* ---- one transaction ---- */
|
||||
|
||||
int warc_write_transaction(warc_writer *w, const char *target_uri,
|
||||
const char *ip, const char *req_hdr,
|
||||
const char *resp_hdr, const char *body,
|
||||
size_t body_len, const char *body_path,
|
||||
int statuscode, int is_update_unchanged,
|
||||
int truncated) {
|
||||
wbuf http;
|
||||
char resp_id[64];
|
||||
char pdig[33];
|
||||
int have_pdig;
|
||||
int is_revisit = 0;
|
||||
const char *profile = NULL;
|
||||
const char *refers_uri = NULL;
|
||||
const char *refers_date = NULL;
|
||||
char refers_buf[HTS_URLMAXSIZE * 2 + 64];
|
||||
int has_payload;
|
||||
int emit_body;
|
||||
int rc = -1;
|
||||
|
||||
if (resp_hdr == NULL)
|
||||
return -1;
|
||||
|
||||
/* A payload exists (for digesting) unless this is a bodyless 304. */
|
||||
has_payload = (body_len > 0 && (body != NULL || body_path != NULL) &&
|
||||
!is_update_unchanged);
|
||||
|
||||
/* Payload digest drives identical-payload-digest dedup (OpenSSL only). */
|
||||
have_pdig =
|
||||
has_payload ? payload_digest_b32(body, body_len, body_path, pdig) : 0;
|
||||
|
||||
if (is_update_unchanged) {
|
||||
is_revisit = 1;
|
||||
profile = "http://netpreserve.org/warc/1.1/revisit/server-not-modified";
|
||||
} else if (have_pdig && w->seen != NULL) {
|
||||
void *prev = NULL;
|
||||
if (coucal_read_pvoid(w->seen, pdig, &prev) && prev != NULL) {
|
||||
char *slot = (char *) prev;
|
||||
char *sep = strchr(slot, '\001');
|
||||
is_revisit = 1;
|
||||
profile =
|
||||
"http://netpreserve.org/warc/1.1/revisit/identical-payload-digest";
|
||||
if (sep != NULL) {
|
||||
size_t n = (size_t) (sep - slot);
|
||||
if (n < sizeof(refers_buf)) {
|
||||
memcpy(refers_buf, slot, n);
|
||||
refers_buf[n] = '\0';
|
||||
refers_uri = refers_buf;
|
||||
refers_date = sep + 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* Both revisit kinds (server-304 and identical-payload-digest) are bodyless;
|
||||
only a full response carries the payload (F1). */
|
||||
emit_body = has_payload && !is_revisit;
|
||||
|
||||
/* Normalize headers: full response rewrites Content-Length to the decoded
|
||||
body length and strips Content-Encoding; a revisit keeps them (no body). */
|
||||
memset(&http, 0, sizeof(http));
|
||||
if (normalize_http_headers(resp_hdr, emit_body ? (long long) body_len : -1,
|
||||
&http) != 0) {
|
||||
wbuf_free(&http);
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* Response first: its id links the request via WARC-Concurrent-To. A revisit
|
||||
is a deliberate dedup, not a truncation, so tag WARC-Truncated only on a
|
||||
full (body-carrying) response. */
|
||||
resp_id[0] = '\0';
|
||||
if (warc_emit(w, is_revisit ? "revisit" : "response",
|
||||
"application/http;msgtype=response", target_uri, ip, NULL,
|
||||
refers_uri, refers_date, profile, have_pdig ? pdig : NULL,
|
||||
emit_body ? warc_truncated_reason(truncated) : NULL, 1,
|
||||
http.data, http.len, emit_body, body, body_len, body_path,
|
||||
resp_id) != 0) {
|
||||
wbuf_free(&http);
|
||||
return -1;
|
||||
}
|
||||
wbuf_free(&http);
|
||||
|
||||
if (req_hdr != NULL && req_hdr[0] != '\0') {
|
||||
size_t rlen = strlen(req_hdr);
|
||||
if (warc_emit(w, "request", "application/http;msgtype=request", target_uri,
|
||||
NULL, resp_id, NULL, NULL, NULL, NULL, NULL, 0, NULL, 0, 1,
|
||||
req_hdr, rlen, NULL, NULL) != 0)
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* Record this payload for later identical-payload-digest revisits. */
|
||||
if (!is_revisit && have_pdig && w->seen != NULL && target_uri != NULL) {
|
||||
char date[32];
|
||||
char *slot;
|
||||
size_t need;
|
||||
warc_now_iso8601(date);
|
||||
need = strlen(target_uri) + 1 + strlen(date) + 1;
|
||||
slot = malloct(need);
|
||||
if (slot != NULL) {
|
||||
snprintf(slot, need, "%s\001%s", target_uri, date);
|
||||
if (coucal_write_pvoid(w->seen, pdig, slot) == 0) {
|
||||
/* replaced an existing entry: coucal freed the old value */
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
(void) statuscode;
|
||||
rc = 0;
|
||||
return rc;
|
||||
}
|
||||
|
||||
/* ---- one non-HTTP capture ---- */
|
||||
|
||||
int warc_write_resource(warc_writer *w, const char *target_uri, const char *ip,
|
||||
const char *content_type, const char *body,
|
||||
size_t body_len, const char *body_path, int truncated) {
|
||||
char pdig[33];
|
||||
int has_body = (body_len > 0 && (body != NULL || body_path != NULL));
|
||||
int have_pdig =
|
||||
has_body ? payload_digest_b32(body, body_len, body_path, pdig) : 0;
|
||||
/* resource: the block is the raw payload, its own MIME is the record's
|
||||
Content-Type, and there is no HTTP request/response envelope. */
|
||||
return warc_emit(w, "resource",
|
||||
(content_type != NULL && content_type[0] != '\0')
|
||||
? content_type
|
||||
: "application/octet-stream",
|
||||
target_uri, ip, NULL, NULL, NULL, NULL,
|
||||
have_pdig ? pdig : NULL, warc_truncated_reason(truncated), 0,
|
||||
NULL, 0, has_body, body, body_len, body_path, NULL);
|
||||
}
|
||||
|
||||
/* ---- engine emit hook ---- */
|
||||
|
||||
void warc_write_backtransaction(httrackp *opt, lien_back *back) {
|
||||
warc_writer *w;
|
||||
char uri[HTS_URLMAXSIZE * 4 + 16];
|
||||
char ip[128];
|
||||
const char *body;
|
||||
size_t body_len;
|
||||
const char *body_path;
|
||||
const char *resp_hdr;
|
||||
char synth[512];
|
||||
int is_unchanged;
|
||||
int is_ftp;
|
||||
|
||||
if (opt->state.warc == WARC_DISABLED)
|
||||
return;
|
||||
if (opt->state.warc == NULL) {
|
||||
w = warc_open(opt, StringBuff(opt->warc_file));
|
||||
if (w == NULL) {
|
||||
opt->state.warc = WARC_DISABLED;
|
||||
hts_log_print(opt, LOG_ERROR, "could not create WARC archive %s",
|
||||
StringBuff(opt->warc_file));
|
||||
return;
|
||||
}
|
||||
opt->state.warc = w;
|
||||
}
|
||||
w = (warc_writer *) opt->state.warc;
|
||||
|
||||
if (back->r.statuscode <= 0)
|
||||
return;
|
||||
|
||||
is_ftp = strfield(back->url_adr, "ftp://") != 0;
|
||||
snprintf(uri, sizeof(uri), "%s%s%s",
|
||||
link_has_authority(back->url_adr) ? "" : "http://", back->url_adr,
|
||||
back->url_fil);
|
||||
|
||||
ip[0] = '\0';
|
||||
SOCaddr_inetntoa(ip, sizeof(ip), back->r.address);
|
||||
|
||||
if (!back->r.is_write) {
|
||||
body = back->r.adr;
|
||||
body_len = (back->r.size > 0) ? (size_t) back->r.size : 0;
|
||||
body_path = NULL;
|
||||
} else {
|
||||
LLint fs;
|
||||
body = NULL;
|
||||
body_path = back->url_sav;
|
||||
fs = fsize_utf8(body_path);
|
||||
/* F4: an on-disk body past size_t (LLP32/ILP32, >4GB) would wrap the length
|
||||
used as Content-Length; drop the body rather than desync the record. */
|
||||
if (fs > 0 && (uint64_t) fs <= (uint64_t) (size_t) -1)
|
||||
body_len = (size_t) fs;
|
||||
else
|
||||
body_len = 0;
|
||||
}
|
||||
|
||||
/* FTP has no HTTP envelope: one resource record carrying the payload. */
|
||||
if (is_ftp) {
|
||||
warc_write_resource(w, uri, ip, back->r.contenttype, body, body_len,
|
||||
body_path, back->r.warc_truncated);
|
||||
return;
|
||||
}
|
||||
|
||||
is_unchanged = (back->r.notmodified && opt->is_update) ? 1 : 0;
|
||||
|
||||
/* Prefer the stashed raw headers; synthesize a minimal status line for the
|
||||
header-less (HTTP/0.9-style) responses that never carried a header block.
|
||||
*/
|
||||
resp_hdr = back->r.warc_resphdr;
|
||||
if (resp_hdr == NULL) {
|
||||
snprintf(synth, sizeof(synth), "HTTP/1.1 %d %s\r\nContent-Type: %s\r\n\r\n",
|
||||
back->r.statuscode, back->r.msg[0] ? back->r.msg : "OK",
|
||||
back->r.contenttype[0] ? back->r.contenttype
|
||||
: "application/octet-stream");
|
||||
resp_hdr = synth;
|
||||
}
|
||||
|
||||
warc_write_transaction(w, uri, ip, back->r.warc_reqhdr, resp_hdr, body,
|
||||
body_len, body_path, back->r.statuscode, is_unchanged,
|
||||
back->r.warc_truncated);
|
||||
}
|
||||
118
src/htswarc.h
118
src/htswarc.h
@@ -1,118 +0,0 @@
|
||||
/* ------------------------------------------------------------ */
|
||||
/*
|
||||
HTTrack Website Copier, Offline Browser for Windows and Unix
|
||||
Copyright (C) 2026 Xavier Roche and other contributors
|
||||
|
||||
SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License
|
||||
along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
Ethical use: we kindly ask that you NOT use this software to harvest email
|
||||
addresses or to collect any other private information about people. Doing so
|
||||
would dishonor our work and waste the many hours we have spent on it.
|
||||
|
||||
Please visit our Website: http://www.httrack.com
|
||||
*/
|
||||
|
||||
/* ------------------------------------------------------------ */
|
||||
/* HTTrack WARC/1.1 output writer (ISO 28500). Internal, not installed.
|
||||
All WARC record serialization, gzip-member framing, digests, UUID and
|
||||
revisit dedup live here; the engine only stashes the request, frees it,
|
||||
and calls one entry point per finished transaction. */
|
||||
/* ------------------------------------------------------------ */
|
||||
|
||||
#ifndef HTS_WARC_DEFH
|
||||
#define HTS_WARC_DEFH
|
||||
|
||||
#include "htsopt.h"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
/* opt->warc_file sentinel: --warc with no argument => auto-name the archive
|
||||
under the project's output directory at open time. */
|
||||
#define WARC_AUTONAME "\001auto"
|
||||
|
||||
/* htsblk.warc_truncated / WARC-Truncated reason tokens (ISO 28500 sec 5.13).
|
||||
A cap-truncated body is still archived, tagged with why it was cut short. */
|
||||
#define WARC_TRUNC_NONE 0
|
||||
#define WARC_TRUNC_LENGTH 1 /* hit a size cap (-M mirror / -m per-file) */
|
||||
#define WARC_TRUNC_TIME 2 /* hit the mirror time cap (-E/--max-time) */
|
||||
#define WARC_TRUNC_DISCONNECT 3 /* connection dropped mid-body */
|
||||
|
||||
/* WARC-Truncated token for a warc_truncated code, or NULL for none. */
|
||||
const char *warc_truncated_reason(int code);
|
||||
|
||||
typedef struct warc_writer warc_writer;
|
||||
|
||||
/* Stash the raw request header block (bstr.buffer) on r for the later WARC
|
||||
request record; frees any prior stash. No-op when reqhdr is NULL. */
|
||||
void warc_stash_request(htsblk *r, const char *reqhdr);
|
||||
|
||||
/* Stash the raw response header block: deletehttp frees r.headers when the
|
||||
socket closes, before back_finalize, so keep a WARC-owned copy. */
|
||||
void warc_stash_response(htsblk *r, const char *resphdr);
|
||||
|
||||
/* Free both stashed header blocks (idempotent, NULL-safe). */
|
||||
void warc_free_request(htsblk *r);
|
||||
|
||||
/* Emit the request + response (or revisit) records for one finished
|
||||
transaction. Lazily opens the writer into opt->state.warc; a no-op (logged
|
||||
once) if the archive cannot be created. */
|
||||
void warc_write_backtransaction(httrackp *opt, lien_back *back);
|
||||
|
||||
/* Close and free the writer held in opt->state.warc, if any. */
|
||||
void warc_close_opt(httrackp *opt);
|
||||
|
||||
/* --- Direct writer API (used by the hooks above and the self-test). --- */
|
||||
|
||||
/* Create the archive at path (auto-named when path is WARC_AUTONAME), writing
|
||||
the warcinfo record. .warc.gz => one gzip member per record; .warc => raw.
|
||||
Returns NULL on failure. */
|
||||
warc_writer *warc_open(httrackp *opt, const char *path);
|
||||
|
||||
/* Flush, close and free the writer (NULL-safe). */
|
||||
void warc_close(warc_writer *w);
|
||||
|
||||
/* Write one transaction's request + response (or revisit) records.
|
||||
target_uri: absolute URL fetched.
|
||||
ip: numeric peer IP, or NULL/"" to omit.
|
||||
req_hdr: exact request header block sent, or NULL to skip the request.
|
||||
resp_hdr: raw received response header block (status line + headers).
|
||||
body/body_len: decoded in-memory body, or NULL when on disk.
|
||||
body_path: file re-read for the body when body==NULL (may be NULL).
|
||||
is_update_unchanged: nonzero for a 304 server-not-modified revisit.
|
||||
truncated: a WARC_TRUNC_* reason to tag a cap-truncated body, else 0.
|
||||
Returns 0 on success, -1 on error. */
|
||||
int warc_write_transaction(warc_writer *w, const char *target_uri,
|
||||
const char *ip, const char *req_hdr,
|
||||
const char *resp_hdr, const char *body,
|
||||
size_t body_len, const char *body_path,
|
||||
int statuscode, int is_update_unchanged,
|
||||
int truncated);
|
||||
|
||||
/* Write one non-HTTP capture as a single WARC 'resource' record: the block is
|
||||
the raw payload (no HTTP envelope), Content-Type is the payload's own MIME.
|
||||
Used for ftp:// transfers. truncated is a WARC_TRUNC_* reason or 0.
|
||||
Returns 0 on success, -1 on error. */
|
||||
int warc_write_resource(warc_writer *w, const char *target_uri, const char *ip,
|
||||
const char *content_type, const char *body,
|
||||
size_t body_len, const char *body_path, int truncated);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -135,7 +135,6 @@
|
||||
<ClCompile Include="htswizard.c" />
|
||||
<ClCompile Include="htswrap.c" />
|
||||
<ClCompile Include="htszlib.c" />
|
||||
<ClCompile Include="htswarc.c" />
|
||||
<ClCompile Include="md5.c" />
|
||||
<ClCompile Include="minizip\ioapi.c" />
|
||||
<ClCompile Include="minizip\iowin32.c" />
|
||||
|
||||
@@ -1,22 +0,0 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# WARC/1.1 writer self-test: framing, Content-Length, gzip members, round-trip,
|
||||
# revisit dedup, plus v1.1 WARC-Truncated, ftp resource records, and
|
||||
# --warc-max-size rotation, over synthetic transactions.
|
||||
|
||||
httrack_bin=$(cd "$(dirname "$(command -v httrack)")" && pwd)/httrack
|
||||
|
||||
scratch=$(mktemp -d)
|
||||
trap 'rm -rf "$scratch"' EXIT
|
||||
|
||||
for t in warc warc-trunc warc-ftp warc-rotate; do
|
||||
out=$("$httrack_bin" -O /dev/null "-#test=$t" "$scratch/")
|
||||
echo "$out"
|
||||
case "$out" in
|
||||
*": OK") ;;
|
||||
*) exit 1 ;;
|
||||
esac
|
||||
done
|
||||
@@ -1,21 +0,0 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# A --warc crawl writes a standards-conformant WARC/1.1 archive, and an
|
||||
# --update re-crawl of an unchanged (all-304) site emits revisit records.
|
||||
# The stdlib validator (no warcio) is the real gate: it byte-compares the
|
||||
# fresh page.html response body against what the server served, asserts the
|
||||
# encoding headers were stripped and the payload digest matches, then checks
|
||||
# the update pass turned the unchanged assets into revisits (no full response).
|
||||
|
||||
set -eu
|
||||
|
||||
: "${top_srcdir:=..}"
|
||||
|
||||
# page.html body served by tests/local-server.py (route_mini304_page).
|
||||
export WARC_VALIDATE_BODY="page.html=3c68746d6c3e3c626f64793e74696e7920636163686561626c6520706167653c2f626f64793e3c2f68746d6c3e0a"
|
||||
export WARC_VALIDATE_NORESP="index.html page.html"
|
||||
|
||||
bash "$top_srcdir/tests/local-crawl.sh" --errors 0 --rerun --warc-validate \
|
||||
--log-found 'no files updated' \
|
||||
--found 'mini304/index.html' --found 'mini304/page.html' \
|
||||
httrack 'BASEURL/mini304/index.html' --warc-file warc-out
|
||||
@@ -3,7 +3,7 @@
|
||||
# silently drop it from the dist tarball and break "make distcheck".
|
||||
EXTRA_DIST = $(TESTS) crawl-test.sh run-all-tests.sh check-network.sh \
|
||||
proxy-https-server.py socks5-server.py proxy-connect-server.py \
|
||||
proxytestlib.py tls-stall-server.py warc-validate.py \
|
||||
proxytestlib.py tls-stall-server.py \
|
||||
local-crawl.sh local-server.py testlib.sh server.crt server.key \
|
||||
server-root/simple/basic.html server-root/simple/link.html \
|
||||
server-root/stripquery/index.html server-root/stripquery/a.html \
|
||||
@@ -80,7 +80,6 @@ TESTS = \
|
||||
01_engine-version-macros.test \
|
||||
01_engine-xfread.test \
|
||||
01_zlib-acceptencoding.test \
|
||||
01_zlib-warc.test \
|
||||
01_zlib-contentcodings.test \
|
||||
01_zlib-cache.test \
|
||||
01_zlib-cache-corrupt.test \
|
||||
@@ -156,7 +155,6 @@ TESTS = \
|
||||
68_webhttrack-outdir-charset.test \
|
||||
69_local-intl-logdir.test \
|
||||
71_local-crange-repaircache.test \
|
||||
72_watchdog-crawl.test \
|
||||
73_local-warc.test
|
||||
72_watchdog-crawl.test
|
||||
|
||||
CLEANFILES = check-network_sh.cache
|
||||
|
||||
@@ -47,7 +47,6 @@ key="${testdir}/server.key"
|
||||
|
||||
tls=
|
||||
verbose=
|
||||
warc_validate=
|
||||
html_subdir=
|
||||
outdir_intl=
|
||||
rerun=
|
||||
@@ -115,8 +114,6 @@ while test "$pos" -lt "$nargs"; do
|
||||
--debug) verbose=1 ;;
|
||||
--rerun) rerun=1 ;; # run httrack a second time (update pass) before auditing
|
||||
--rerun-dead) rerun_dead=1 ;; # re-run with the server stopped (cache rollback)
|
||||
# validate the produced .warc.gz (see the validation block near the end)
|
||||
--warc-validate) warc_validate=1 ;;
|
||||
--no-purge)
|
||||
nopurge=1
|
||||
audit+=("--no-purge")
|
||||
@@ -262,13 +259,6 @@ test "$crawlres" -eq 0 || ! result "httrack exited $crawlres" || {
|
||||
result "OK"
|
||||
grep -iE "^[0-9:]*[[:space:]]Error:" "${logroot}/hts-log.txt" >&2
|
||||
|
||||
# Snapshot the first-pass WARC before an update pass overwrites it: the fresh
|
||||
# crawl carries the full response bodies, the update pass only revisits.
|
||||
if test -n "$warc_validate"; then
|
||||
w1=$(find "$mirrorroot" -maxdepth 2 -name '*.warc.gz' 2>/dev/null | sort | tail -n1)
|
||||
test -z "$w1" || cp "$w1" "${tmpdir}/warc-pass1.gz"
|
||||
fi
|
||||
|
||||
# --- optional second pass: re-mirror into the same dir (cache/update path) ----
|
||||
if test -n "$rerun"; then
|
||||
info "re-running httrack (update pass)"
|
||||
@@ -360,44 +350,6 @@ done
|
||||
test -n "$hostroot" || die "could not find host root under $out"
|
||||
debug "host root: $hostroot"
|
||||
|
||||
# --- optional WARC validation (stdlib validator, no warcio) ------------------
|
||||
# WARC_VALIDATE_BODY="URLSUB=HEX" byte-checks a fresh-crawl response body;
|
||||
# WARC_VALIDATE_NORESP="URLSUB..." asserts those assets are revisits post-update.
|
||||
if test -n "$warc_validate"; then
|
||||
validator=$(nativepath "${testdir}/warc-validate.py")
|
||||
warc=$(find "$mirrorroot" -maxdepth 2 \( -name '*.warc.gz' -o -name '*.warc' \) 2>/dev/null | sort | tail -n1)
|
||||
test -n "$warc" || die "no WARC file produced under $mirrorroot"
|
||||
|
||||
# Fresh-crawl file (snapshot if an update pass overwrote it): full responses.
|
||||
fresh="${tmpdir}/warc-pass1.gz"
|
||||
test -f "$fresh" || fresh="$warc"
|
||||
declare -a bodyargs=()
|
||||
if test -n "${WARC_VALIDATE_BODY:-}"; then
|
||||
bodyargs=(--expect-body-hex "$WARC_VALIDATE_BODY")
|
||||
fi
|
||||
info "validating fresh WARC (response bodies)"
|
||||
"$python" "$validator" "$(nativepath "$fresh")" "${bodyargs[@]}" >&2 ||
|
||||
die "fresh WARC validation failed"
|
||||
result "OK"
|
||||
|
||||
# Final file: after an update pass the unchanged assets must be revisits.
|
||||
if test -n "$rerun"; then
|
||||
declare -a revargs=(--expect-revisit)
|
||||
for sub in ${WARC_VALIDATE_NORESP:-}; do
|
||||
revargs+=(--no-response-for "$sub")
|
||||
done
|
||||
info "validating update WARC (revisits)"
|
||||
"$python" "$validator" "$(nativepath "$warc")" "${revargs[@]}" >&2 ||
|
||||
die "update WARC validation failed"
|
||||
result "OK"
|
||||
fi
|
||||
|
||||
if command -v warcio >/dev/null 2>&1; then
|
||||
info "warcio check (optional)"
|
||||
if warcio check -v "$warc" >&2; then result "OK"; else die "warcio check failed"; fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# No crawl, even a cancelled one, may leave engine temporaries: .delayed (#107,
|
||||
# #483), or the .z/.u content-coding temps (#557).
|
||||
info "checking for leftover engine temporaries"
|
||||
|
||||
@@ -1,136 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
# Structural + semantic WARC/1.1 validator, Python stdlib only (no warcio):
|
||||
# walks the concatenated gzip members with zlib (gzip.decompress would fuse them
|
||||
# and lose per-record boundaries) and checks each record against the spec.
|
||||
#
|
||||
# Options:
|
||||
# --expect-revisit at least one revisit record must be present
|
||||
# --expect-body-hex SUB=HEX a response whose WARC-Target-URI contains SUB must
|
||||
# have an entity body byte-equal to bytes.fromhex(HEX),
|
||||
# no Content-Encoding/Transfer-Encoding header, and a
|
||||
# WARC-Payload-Digest matching sha1(body) when present
|
||||
# --no-response-for SUB the asset containing SUB must be a revisit: no
|
||||
# response may target it, and a revisit must
|
||||
import base64
|
||||
import hashlib
|
||||
import sys
|
||||
import zlib
|
||||
|
||||
|
||||
def records(data):
|
||||
"""Yield each record's decompressed bytes (one gzip member per record)."""
|
||||
if data[:2] != b"\x1f\x8b": # uncompressed .warc: split on the record magic
|
||||
parts = data.split(b"WARC/1.")
|
||||
for p in parts[1:]:
|
||||
yield b"WARC/1." + p
|
||||
return
|
||||
while data:
|
||||
d = zlib.decompressobj(zlib.MAX_WBITS | 16)
|
||||
block = d.decompress(data) + d.flush()
|
||||
yield block
|
||||
data = d.unused_data
|
||||
|
||||
|
||||
def field(header, name):
|
||||
for line in header.split(b"\r\n"):
|
||||
if line.lower().startswith(name.lower() + b":"):
|
||||
return line.split(b":", 1)[1].strip()
|
||||
return None
|
||||
|
||||
|
||||
def opt_values(argv, name):
|
||||
out = []
|
||||
for i, a in enumerate(argv):
|
||||
if a == name and i + 1 < len(argv):
|
||||
out.append(argv[i + 1])
|
||||
return out
|
||||
|
||||
|
||||
def check_body(rec, http_hdr, body, sub, want):
|
||||
if b"Content-Encoding" in http_hdr or b"Transfer-Encoding" in http_hdr:
|
||||
sys.exit("record for %s kept a content/transfer-encoding header" % sub)
|
||||
if body != want:
|
||||
sys.exit(
|
||||
"body mismatch for %s: got %d bytes, expected %d"
|
||||
% (sub, len(body), len(want))
|
||||
)
|
||||
pd = field(rec[: rec.find(b"\r\n\r\n")], b"WARC-Payload-Digest")
|
||||
if pd is not None and pd.startswith(b"sha1:"):
|
||||
want_b32 = base64.b32encode(hashlib.sha1(want).digest()).decode("ascii")
|
||||
if pd[5:].decode("ascii") != want_b32:
|
||||
sys.exit("WARC-Payload-Digest mismatch for %s" % sub)
|
||||
|
||||
|
||||
def main():
|
||||
argv = sys.argv[1:]
|
||||
expect_revisit = "--expect-revisit" in argv
|
||||
body_specs = [s.split("=", 1) for s in opt_values(argv, "--expect-body-hex")]
|
||||
no_resp = opt_values(argv, "--no-response-for")
|
||||
path = [a for a in argv if not a.startswith("--") and "=" not in a][0]
|
||||
data = open(path, "rb").read()
|
||||
|
||||
total = revisits = responses = infos = 0
|
||||
body_hits = {sub: False for sub, _ in body_specs}
|
||||
revisit_hits = {sub: False for sub in no_resp}
|
||||
for rec in records(data):
|
||||
total += 1
|
||||
if not rec.startswith(b"WARC/1."):
|
||||
sys.exit("record %d: bad magic %r" % (total, rec[:16]))
|
||||
sep = rec.find(b"\r\n\r\n")
|
||||
if sep < 0:
|
||||
sys.exit("record %d: no header terminator" % total)
|
||||
hdr_end = sep + 4
|
||||
header = rec[:sep]
|
||||
cl = field(header, b"Content-Length")
|
||||
if cl is None:
|
||||
sys.exit("record %d: no Content-Length" % total)
|
||||
block_len = int(cl)
|
||||
if hdr_end + block_len + 4 != len(rec):
|
||||
sys.exit(
|
||||
"record %d: Content-Length %d != block length %d"
|
||||
% (total, block_len, len(rec) - hdr_end - 4)
|
||||
)
|
||||
if rec[hdr_end + block_len :] != b"\r\n\r\n":
|
||||
sys.exit("record %d: missing \\r\\n\\r\\n trailer" % total)
|
||||
wtype = field(header, b"WARC-Type")
|
||||
uri = field(header, b"WARC-Target-URI") or b""
|
||||
if wtype == b"warcinfo":
|
||||
infos += 1
|
||||
elif wtype == b"response":
|
||||
responses += 1
|
||||
for sub in no_resp:
|
||||
if sub.encode() in uri:
|
||||
sys.exit("unexpected full response for %s (want revisit)" % sub)
|
||||
block = rec[hdr_end : hdr_end + block_len]
|
||||
bsep = block.find(b"\r\n\r\n")
|
||||
http_hdr, body = block[:bsep], block[bsep + 4 :]
|
||||
for sub, hexval in body_specs:
|
||||
if sub.encode() in uri:
|
||||
check_body(rec, http_hdr, body, sub, bytes.fromhex(hexval))
|
||||
body_hits[sub] = True
|
||||
elif wtype == b"revisit":
|
||||
revisits += 1
|
||||
for sub in no_resp:
|
||||
if sub.encode() in uri:
|
||||
revisit_hits[sub] = True
|
||||
|
||||
if total < 1:
|
||||
sys.exit("no records found")
|
||||
if infos != 1:
|
||||
sys.exit("expected exactly one warcinfo record, got %d" % infos)
|
||||
if expect_revisit and revisits < 1:
|
||||
sys.exit("expected at least one revisit record, found none")
|
||||
for sub, hit in body_hits.items():
|
||||
if not hit:
|
||||
sys.exit("no response record found for --expect-body-hex %s" % sub)
|
||||
for sub, hit in revisit_hits.items():
|
||||
if not hit:
|
||||
sys.exit("no revisit record found for unchanged asset %s" % sub)
|
||||
print(
|
||||
"warc-validate: %d records OK (%d response, %d revisit)"
|
||||
% (total, responses, revisits)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -46,12 +46,9 @@ cat >"$stubdir/x-www-browser" <<EOF
|
||||
echo "stub browser invoked with: \$1" >&2
|
||||
# Also fetch an option page and require a rendered title='' tooltip: proves the
|
||||
# option template expands and the \${html:} filter escapes into the attribute.
|
||||
# option9 additionally proves the WARC control renders with its expanded label.
|
||||
opturl="\${1%/}/server/option2.html"
|
||||
warcurl="\${1%/}/server/option9.html"
|
||||
if body="\$(curl -fsSL --max-time 20 "\$1")" && printf '%s' "\$body" | grep -qai httrack && printf '%s' "\$body" | grep -qaF step2.html &&
|
||||
opt="\$(curl -fsSL --max-time 20 "\$opturl")" && printf '%s' "\$opt" | grep -qaF "title='" &&
|
||||
warc="\$(curl -fsSL --max-time 20 "\$warcurl")" && printf '%s' "\$warc" | grep -qaF 'name="warcfile"' && printf '%s' "\$warc" | grep -qaF WARC; then
|
||||
opt="\$(curl -fsSL --max-time 20 "\$opturl")" && printf '%s' "\$opt" | grep -qaF "title='"; then
|
||||
echo PASS >"$marker"
|
||||
else
|
||||
echo "FAIL: unexpected response from \$1" >"$marker"
|
||||
|
||||
Reference in New Issue
Block a user