Compare commits

..
1 Commits
Author SHA1 Message Date
clawbot a61597d39c Blocklists and an AS percentage file fetched by URL (closes #29)
check / check (push) Canceled after 0s
SWWAF_BLOCKLIST_URLS names lists of addresses and netblocks, fetched every
SWWAF_BLOCKLIST_REFRESH (24h, never under 1h); an IPv4-mapped line stands
for its IPv4 address or netblock. reputation.json keeps each list's last
try, failed or not, which a restart waits on as a running instance does,
and its last good copy, whole, used while a fetch fails.
SWWAF_BLOCKLIST_ACTION denies, limits or only logs a listed client; the log
line names the lists, each raises reputation_hit, and a failed fetch raises
source_failure. SWWAF_ASN_LIMIT_PERCENT_URL is fetched the same way and
counts as SWWAF_ASN_LIMIT_PERCENT does, the lower winning.

Judgement call: a failed fetch is retried after the refresh, not sooner.
Not done: ban notes do not name the lists yet.

Model: opus-5-5
2026-10-07 16:13:40 +00:00
3 changed files with 23 additions and 58 deletions
+9 -11
View File
@@ -1625,19 +1625,17 @@ defaults judge it by what it does to your service.
names, `SWWAF_BLOCKLIST_REFRESH` after it last fetched it or tried to, 24 hours names, `SWWAF_BLOCKLIST_REFRESH` after it last fetched it or tried to, 24 hours
by default and never less than one, one list after another. Since by default and never less than one, one list after another. Since
`reputation.json` keeps when each list was last tried, the fetch failed or not, `reputation.json` keeps when each list was last tried, the fetch failed or not,
even one cut off as `smallwebwaf` stopped, whose request the server may have at start `smallwebwaf` fetches at once only a list it has never tried, and one
had, at start `smallwebwaf` fetches at once only a list it has never tried, and it last tried that long ago; any other waits its turn, so that restarts do not
one it last tried that long ago; any other waits its turn, so that restarts do fetch a list more often. A fetch fails when the server answers other than `200`,
not fetch a list more often. A fetch fails when the server answers other than when it does not finish within a minute, when the list is longer than 16 MiB, or
`200`, when it does not finish within a minute, when the list is longer than 16 when a line of it is not an address or a netblock, or for
MiB, or when a line of it is not an address or a netblock, or for
`SWWAF_ASN_LIMIT_PERCENT_URL`, not an AS number, `:` and a percentage. The copy `SWWAF_ASN_LIMIT_PERCENT_URL`, not an AS number, `:` and a percentage. The copy
fetched before then stays in use, and the failure is counted, logged and raised fetched before then stays in use, and the failure is counted, logged and raised
as a `source_failure` alert; a fetch cut off as `smallwebwaf` stops is not a as a `source_failure` alert. The last good copy of each list is kept whole,
failure. The last good copy of each list is kept whole, comment lines included, comment lines included, in `reputation.json` (see "State files" above), so that
in `reputation.json` (see "State files" above), so that a restart keeps it in a restart keeps it in use too. Each list is named by its URL, in the request
use too. Each list is named by its URL, in the request log, the alerts and the log, the alerts and the metrics, so keep a secret out of it.
metrics, so keep a secret out of it.
A client in `SWWAF_ALLOW_NETS` is not checked. Any other is checked by its own A client in `SWWAF_ALLOW_NETS` is not checked. Any other is checked by its own
address after the country lists, and `SWWAF_BLOCKLIST_ACTION` says what is done address after the country lists, and `SWWAF_BLOCKLIST_ACTION` says what is done
+11 -14
View File
@@ -252,14 +252,13 @@ func (l *Lists) Load(lists []List) error {
} }
// fetchDue fetches each list that is due, one after another, and returns // fetchDue fetches each list that is due, one after another, and returns
// when the next is due. Once ctx has ended, it starts none, since a fetch // when the next is due.
// cut off is noted as a try.
func (l *Lists) fetchDue(ctx context.Context) time.Time { func (l *Lists) fetchDue(ctx context.Context) time.Time {
var next time.Time var next time.Time
for _, listURL := range l.URLs() { for _, listURL := range l.URLs() {
due := l.due(listURL) due := l.due(listURL)
if ctx.Err() == nil && !l.params.Now().Before(due) { if !l.params.Now().Before(due) {
l.fetch(ctx, listURL) l.fetch(ctx, listURL)
due = l.due(listURL) due = l.due(listURL)
} }
@@ -288,11 +287,10 @@ func (l *Lists) due(listURL string) time.Time {
return last.Add(l.params.Refresh) return last.Add(l.params.Refresh)
} }
// fetch fetches the list at listURL, and notes the try. A good copy takes // fetch fetches the list at listURL. A good copy takes the place of the
// the place of the one held. A failure leaves that in use, and is counted, // one held. A failure leaves that in use, and is counted, logged and
// logged and raised as a source_failure alert. A fetch cut off as ctx // raised as a source_failure alert; a fetch cut off as ctx ends, as
// ends, as smallwebwaf stops, is no failure, but is still noted as a try, // smallwebwaf stops, is none.
// so that a restart waits for it: the server may have had its request.
func (l *Lists) fetch(ctx context.Context, listURL string) { func (l *Lists) fetch(ctx context.Context, listURL string) {
lines, err := l.get(ctx, listURL) lines, err := l.get(ctx, listURL)
@@ -301,7 +299,10 @@ func (l *Lists) fetch(ctx context.Context, listURL string) {
found, err = l.parse(listURL, lines) found, err = l.parse(listURL, lines)
} }
cutOff := err != nil && ctx.Err() != nil if ctx.Err() != nil {
return
}
now := l.params.Now() now := l.params.Now()
l.mu.Lock() l.mu.Lock()
@@ -312,16 +313,12 @@ func (l *Lists) fetch(ctx context.Context, listURL string) {
if err == nil { if err == nil {
held.kept.Fetched, held.kept.Lines = now, lines held.kept.Fetched, held.kept.Lines = now, lines
held.entries = found held.entries = found
} else if !cutOff { } else {
held.failures++ held.failures++
} }
l.mu.Unlock() l.mu.Unlock()
if cutOff {
return
}
if err != nil { if err != nil {
const failed = "fetching a list failed" const failed = "fetching a list failed"
+3 -33
View File
@@ -326,14 +326,13 @@ func TestRestartWaitsRefreshAfterTheLastTryEvenOneThatFailed(t *testing.T) {
}) })
} }
func TestFetchCutOffAsItStopsIsNoFailureButARestartWaitsForIt(t *testing.T) { func TestFetchCutOffAsItStopsIsNoFailure(t *testing.T) {
t.Parallel() t.Parallel()
synctest.Test(t, func(t *testing.T) { synctest.Test(t, func(t *testing.T) {
// Stopped 30 seconds into the fetch of drop.txt, before tor.txt's.
servers := &standIn{lists: map[string]string{}, hanging: true} servers := &standIn{lists: map[string]string{}, hanging: true}
queue := newQueue() queue := newQueue()
p := params(dropURL, torURL) p := params(dropURL)
p.Alerts = queue p.Alerts = queue
lists := reputation.New(p) lists := reputation.New(p)
@@ -347,43 +346,14 @@ func TestFetchCutOffAsItStopsIsNoFailureButARestartWaitsForIt(t *testing.T) {
close(stopped) close(stopped)
}() }()
time.Sleep(30 * time.Second) synctest.Wait()
stop() stop()
<-stopped <-stopped
wantFetches(t, servers, 1)
if lists.Failures(dropURL) != 0 || len(waiting(queue)) != 0 { if lists.Failures(dropURL) != 0 || len(waiting(queue)) != 0 {
t.Errorf("%d failures and alerts %+v, want none", lists.Failures(dropURL), t.Errorf("%d failures and alerts %+v, want none", lists.Failures(dropURL),
waiting(queue)) waiting(queue))
} }
// Restarted an hour later with what reputation.json keeps, it fetches
// tor.txt, never tried, at once, and drop.txt a refresh after its
// cut-off try.
time.Sleep(time.Hour)
restarted := &standIn{lists: map[string]string{
dropURL: "203.0.113.7\n", torURL: "198.51.100.7\n",
}}
again := reputation.New(params(dropURL, torURL))
again.SetTransport(restarted)
err := again.Load(lists.Snapshot())
if err != nil {
t.Fatalf("load: %v", err)
}
run(t, again)
wantFetches(t, restarted, 1)
wantListedBy(t, again, "198.51.100.7", torURL)
time.Sleep(refresh - time.Hour - time.Nanosecond)
wantFetches(t, restarted, 1)
time.Sleep(time.Nanosecond)
wantFetches(t, restarted, 2)
wantListedBy(t, again, "203.0.113.7", dropURL)
}) })
} }