0.7.11: names covered, analyst files
This commit is contained in:
1 parent
447e39e80f
commit
ffe6ae1a2c
11 files changed
+1852
-154
No files matched your search
+31
-11
@@ -55,8 +55,11 @@ env:
|
||||
PYTHON_IMAGE: python:3.12-slim-bookworm
|
||||
GITLEAKS_IMAGE: zricethezav/gitleaks:v8.30.1
|
||||
# Memory of the build container: the biggest sources (UT1, newly
|
||||
# registered domains) hold millions of names. Raise it if the build is
|
||||
# killed (exit code 137).
|
||||
# registered domains) hold millions of names. Counting the names only one
|
||||
# source brings takes a table that grows by steps: 176 MiB at most up to
|
||||
# 5.6 million distinct names, 352 MiB up to 10 million; beyond, the build
|
||||
# gives that count up by itself (704 MiB would be needed) and says so.
|
||||
# Raise it if the build is killed (exit code 137).
|
||||
BUILD_MEMORY: 3g
|
||||
LISTS_URL: https://gitea.tips-of-mine.com/warda-dns/warda-lists/raw/branch/dist
|
||||
|
||||
@@ -75,22 +78,32 @@ jobs:
|
||||
|
||||
# The tests of the builder: names read as Warda reads them, the three
|
||||
# plain formats, a UT1-style archive, exceptions, allow.txt,
|
||||
# protect.txt, the guard, the downloads (local server). No network.
|
||||
# protect.txt, the files of warda-analyst, the figures of the
|
||||
# manifest, the guard, the downloads (local server). No network.
|
||||
- name: Unit tests
|
||||
shell: bash
|
||||
run: |
|
||||
tar -c scripts tests licenses taxonomy.toml sources.toml extra allow.txt protect.txt \
|
||||
files=(scripts tests licenses taxonomy.toml sources.toml extra allow.txt protect.txt)
|
||||
if [ -d analyst ]; then files+=(analyst); fi
|
||||
tar -c "${files[@]}" \
|
||||
| docker run --rm -i --network none -w /work "$PYTHON_IMAGE" sh -ec '
|
||||
tar -x
|
||||
python3 -m unittest discover -s scripts -v'
|
||||
|
||||
# taxonomy.toml and sources.toml: every category and site type used
|
||||
# exists, every source has a licence, a known format and a valid
|
||||
# address; extra/, allow.txt and protect.txt hold valid domains.
|
||||
- name: Validate the TOML and the owner files
|
||||
# address; extra/, allow.txt and protect.txt hold valid domains; the
|
||||
# files warda-analyst commits in analyst/ are well formed, never add
|
||||
# and remove the same name, stay within their budgets, and nothing
|
||||
# else is in that directory. The directory goes in the archive when
|
||||
# it exists: the daily run takes this workflow from the default
|
||||
# branch and the files from main, which may not have it yet.
|
||||
- name: Validate the TOML, the owner files and the analyst files
|
||||
shell: bash
|
||||
run: |
|
||||
tar -c scripts licenses taxonomy.toml sources.toml extra allow.txt protect.txt \
|
||||
files=(scripts licenses taxonomy.toml sources.toml extra allow.txt protect.txt)
|
||||
if [ -d analyst ]; then files+=(analyst); fi
|
||||
tar -c "${files[@]}" \
|
||||
| docker run --rm -i --network none -w /work "$PYTHON_IMAGE" sh -ec '
|
||||
tar -x
|
||||
python3 scripts/build.py --check'
|
||||
@@ -99,7 +112,9 @@ jobs:
|
||||
- name: Offline build from the fixtures
|
||||
shell: bash
|
||||
run: |
|
||||
tar -c scripts tests licenses taxonomy.toml extra allow.txt protect.txt \
|
||||
files=(scripts tests licenses taxonomy.toml extra allow.txt protect.txt)
|
||||
if [ -d analyst ]; then files+=(analyst); fi
|
||||
tar -c "${files[@]}" \
|
||||
| docker run --rm -i --network none -w /work "$PYTHON_IMAGE" sh -ec '
|
||||
tar -x
|
||||
python3 scripts/build.py --sources tests/fixtures/sources.toml \
|
||||
@@ -174,9 +189,12 @@ jobs:
|
||||
# enabled source (User-Agent warda-lists/1.0) and builds dist/. A file
|
||||
# over 120 MiB (Warda refuses a list over 128 MiB) or a category over
|
||||
# its budget of domains (max_domains, taxonomy.toml) fails the build,
|
||||
# even with force. The container is limited to BUILD_MEMORY, swap
|
||||
# even with force; when only the names of warda-analyst (analyst/)
|
||||
# cause it, or trip the guard, the lists are built without them and
|
||||
# the log says IGNORED. The container is limited to BUILD_MEMORY, swap
|
||||
# included. The guard compares with the manifest published before
|
||||
# (absent the first time): a list that loses more than half of its
|
||||
# (absent the first time), on the lists without the names of
|
||||
# warda-analyst: a list that loses more than half of its
|
||||
# domains, or a base list too small, stops here and nothing is
|
||||
# published; the boxes keep the lists of the day before. The log of
|
||||
# the build goes to standard error, the archive of dist/ to standard
|
||||
@@ -191,7 +209,9 @@ jobs:
|
||||
if [ "$FORCE" = "true" ]; then args+=(--force); fi
|
||||
if [ "$ALLOW_PARTIAL" = "true" ]; then args+=(--allow-partial); fi
|
||||
rm -rf dist
|
||||
tar -c scripts licenses taxonomy.toml sources.toml extra allow.txt protect.txt \
|
||||
files=(scripts licenses taxonomy.toml sources.toml extra allow.txt protect.txt)
|
||||
if [ -d analyst ]; then files+=(analyst); fi
|
||||
tar -c "${files[@]}" \
|
||||
| docker run --rm -i --memory "$BUILD_MEMORY" --memory-swap "$BUILD_MEMORY" \
|
||||
-w /work "$PYTHON_IMAGE" sh -ec '
|
||||
tar -x >&2
|
||||
|
||||
+1
-1
@@ -131,7 +131,7 @@ Changing one of these rules is a decision of the owner.
|
||||
| --- | --- | --- | --- | --- | --- |
|
||||
| AdAway | `https://adaway.org/hosts.txt` | Hosts list | CC BY 3.0. Header of <https://raw.githubusercontent.com/AdAway/adaway.github.io/master/hosts.txt> | `ads.txt` holds GPL-3.0 data: the build refuses CC BY 3.0 with it (`--check` fails). 18 of its 6,540 names are new for `base.txt` | AdAway moving to CC BY 4.0, or the owner accepting this mix in `combine_licences` after legal advice |
|
||||
| Liste FR + EasyList | `https://easylist-downloads.adblockplus.org/liste_fr%2Beasylist.txt` | (second list) Advertising Lists | Liste FR: CC BY-SA 3.0; EasyList: GPL-3.0 or later / CC BY-SA 3.0 or later. Header of <https://raw.githubusercontent.com/easylist/listefr/master/liste_fr.txt>; <https://easylist.to/pages/licence.html> | Same refusal (CC BY-SA 3.0 with GPL), and the fault of the parser told in section 5 (measured: `youtube.com` and `yahoo.com` would be listed from Liste FR alone) | The owner accepting the path CC BY-SA 3.0, then 4.0, then GPL-3.0 in `combine_licences`; and the parser fixed |
|
||||
| SNAFU (RooneyMcNibNug) | `https://raw.githubusercontent.com/RooneyMcNibNug/pihole-stuff/master/SNAFU.txt` | Suspicious Lists | WTFPL (header of the file) | WTFPL allows everything but is not in the accepted list; 75,000 names of one person (18,000 new for `base.txt`), false positives not known | The owner adding `WTFPL` to `ALLOWED_LICENCES` with its text in `licenses/` |
|
||||
| SNAFU (RooneyMcNibNug) | `https://raw.githubusercontent.com/RooneyMcNibNug/pihole-stuff/master/SNAFU.txt` | Suspicious Lists | WTFPL (header of the file) | WTFPL allows everything but is not in the accepted list; 75,000 names of one person (18,000 new for `base.txt`), false positives not known | The owner adding `WTFPL` to `licences` of `taxonomy.toml` (where the accepted licences are listed) with its text in `licenses/` |
|
||||
| Dandelion Sprout's Anti-Malware hosts | `https://raw.githubusercontent.com/DandelionSprout/adfilt/master/Alternate%20versions%20Anti-Malware%20List/AntiMalwareHosts.txt` | Malicious Lists | "Dandelicence" 1.4: use "with or without commercial purpose" allowed, with its own conditions (the list must stay reachable in 100 countries; no sale of the unchanged list as a limited paid product; requests of credit or removal from the projects it borrows from must be followed). <https://github.com/DandelionSprout/adfilt/blob/master/LICENSE.md> | Not an SPDX licence, not in the accepted list. It would add 9,200 names to `security.txt` | The owner reading the Dandelicence and adding it to the accepted list, or written permission of the author |
|
||||
|
||||
### 5. The parser reads the list wrongly
|
||||
|
||||
@@ -15,8 +15,9 @@ The documentation of the Warda project is in the repository
|
||||
1. `sources.toml` names the lists to download (the only file to edit to
|
||||
add one).
|
||||
2. The CI downloads them every day, reads them the way Warda reads them,
|
||||
puts each domain in its categories, adds the domains of `extra/`,
|
||||
removes those of `allow.txt` and `protect.txt`, and removes the
|
||||
puts each domain in its categories, adds the domains of `extra/` and of
|
||||
`analyst/add/`, removes those of `allow.txt`, of `protect.txt` and, in
|
||||
the protection lists only, of `analyst/remove.txt`, and removes the
|
||||
subdomains already covered by a parent.
|
||||
3. The result, `dist/`, is published on the branch `dist`. The boxes
|
||||
download it from there.
|
||||
@@ -147,7 +148,10 @@ The licence is the exact SPDX id of the list, one of: `GPL-3.0-only`,
|
||||
`ISC`, `0BSD`, `BSD-2-Clause`, `BSD-3-Clause`, `Apache-2.0`. Anything else
|
||||
is refused (unknown or proprietary licences, `NC`, `ND`); `GPL-3.0` alone
|
||||
is ambiguous and refused. Each accepted licence has its text in
|
||||
`licenses/`.
|
||||
`licenses/`. The accepted ids are written in one place, `licences` of
|
||||
`taxonomy.toml`: the build reads them there and publishes them in
|
||||
`manifest.json`; an `NC` or `ND` licence is refused even there, and an id
|
||||
without its text in `licenses/` fails `--check`.
|
||||
|
||||
A source that cannot be downloaded fails the whole build: nothing is
|
||||
published and the boxes keep the lists of the day before. A source gone
|
||||
@@ -164,6 +168,65 @@ for good gets `enabled = false`.
|
||||
One domain per line, `#` starts a comment. A line that is not a valid
|
||||
domain fails the build (a typo is never ignored silently).
|
||||
|
||||
## The files of warda-analyst
|
||||
|
||||
The service warda-analyst writes two kinds of files, by a commit on
|
||||
`develop`, each time its team approves or withdraws a name. They are never
|
||||
edited by hand, and are not the files of the owner (`analyst/README.md`).
|
||||
|
||||
| File | Role |
|
||||
| --- | --- |
|
||||
| `analyst/add/<category>.txt` | names the analysis of Warda confirmed, added to that category as `extra/<category>.txt` does (a category that public sources may fill: never `csam`, never `base`) |
|
||||
| `analyst/remove.txt` | names confirmed as harmless and wrongly blocked: removed with their subdomains from the protection lists only, the categories of `remove_from` (`[analyst]` in `taxonomy.toml`: `ads`, `tracking`, `phishing`, `security`) |
|
||||
|
||||
A removal never weakens a content category: the name leaves the
|
||||
categories of `remove_from`, the bundles built from them (`base`) and
|
||||
what `includes` brings into them (`security` includes `phishing`), and
|
||||
stays in every other category (`adult`, `gambling`…) and in the site
|
||||
types. The team clears a name because it is no threat and no tracker,
|
||||
which says nothing of what the site is about. Only `allow.txt` removes a
|
||||
name from every list.
|
||||
|
||||
Lines starting with `#` are comments; every other line is exactly one
|
||||
name in its normalised form (lower case, no trailing dot, no space, no
|
||||
empty line, nothing after the name), and the names are sorted (in the
|
||||
order of their bytes) and unique.
|
||||
`--check` refuses anything else; a name added and removed at once,
|
||||
whichever covers the other (the same name in an `add` file and in
|
||||
`remove.txt`, an added name under a removed one, a removed name under an
|
||||
added one); more names than the budgets `[analyst]` of `taxonomy.toml`,
|
||||
`max_add` (50,000, all the `add` files together) and `max_remove`
|
||||
(20,000); and anything in `analyst/` that is not `README.md`,
|
||||
`remove.txt` or `add/<category>.txt` (a file the build would not read
|
||||
must not look published). A missing file or directory means no name. A
|
||||
refused commit is not promoted to `main`: the daily build goes on with
|
||||
the files of `main`.
|
||||
|
||||
Files that passed the check never stop the build; what it does not apply
|
||||
is said in the log and in `manifest.json` (`analyst`):
|
||||
|
||||
- A line of `remove.txt` is applied only if it takes at most `max_effect`
|
||||
names (100, `[analyst]`) out of the lists it applies to, itself and its
|
||||
subdomains, those lists together. The check cannot know it: `co.uk` or
|
||||
`github.io` are valid names. A line over the limit is left out, with a
|
||||
warning, and listed in `analyst.skipped`.
|
||||
- A removed name that one of those categories still blocks, because a
|
||||
source or `extra/` lists a parent of it, is listed in `analyst.covered`
|
||||
(1,000 entries at most), with a warning.
|
||||
- The budgets and the guard are checked on the lists built without
|
||||
these files: added names cannot hide a collapse of the sources. If a
|
||||
list goes over its budget or over the size limit, or trips the guard,
|
||||
only once these files are applied, the lists are built without any of
|
||||
them: `analyst.status` says `ignored: <why>` and the log `IGNORED`.
|
||||
What fails without them fails as before.
|
||||
|
||||
`allow.txt` and `protect.txt` still win over an added name. The added
|
||||
names are attributed to the source `warda-analyst`
|
||||
(https://warda-dns.com) in the headers of the lists, in `LICENSES.md` and
|
||||
in `manifest.json` (`sources` of each file): own data of this repository,
|
||||
shown as `maintainer` as the `extra/` files are, which adds no licence to
|
||||
a file. No source of `sources.toml` may take that name.
|
||||
|
||||
## The two taxonomies
|
||||
|
||||
Both are in `taxonomy.toml`, with French and English labels.
|
||||
@@ -211,15 +274,50 @@ Published on the branch `dist`, at
|
||||
| `<category>.txt` | one per blocking category, e.g. `.../raw/branch/dist/security.txt` |
|
||||
| `base.txt` | `ads` + `tracking` |
|
||||
| `types/<type>.txt` | one per site type |
|
||||
| `manifest.json` | per file: count, SHA-256, size, sources, licence of the file and exact licences of its data, address; per source: status and counts |
|
||||
| `manifest.json` | per file: entries (`count`), names (`names`), SHA-256, size, sources, licence of the file and exact licences of its data, address; per source: status and figures; the taxonomy, the names of the owner and the counts of warda-analyst (below) |
|
||||
| `LICENSES.md` | the attribution of every source and the licence of every file |
|
||||
| `licenses/<SPDX id>.txt` | the text of every licence used (GPL: full text; CC: notice and link to the legal code) |
|
||||
| `README.md` | the list of the files, their counts and addresses |
|
||||
| `README.md` | the files with their entries, names and addresses; the sources with their figures |
|
||||
|
||||
Each list: one domain per line, sorted, unique; a domain blocks its
|
||||
subdomains too (a subdomain whose parent is listed is removed). The header
|
||||
lines start with `#`: title, generation time (UTC), count, licence (and
|
||||
the exact licences of the data), sources.
|
||||
lines start with `#`: title, file, generation time (`# Generated:`, UTC,
|
||||
such as `2026-10-04T03:21:07Z`), entries (`# Domains:`), names
|
||||
(`# Names:`), licence (and the exact licences of the data), sources.
|
||||
|
||||
Two figures for each file:
|
||||
|
||||
- **Domains** (`count`): the entries of the file, what a box loads.
|
||||
- **Names** (`names`): the distinct names the file stands for: the names
|
||||
of its sources, of `extra/` and of `analyst/add/`, once `allow.txt`,
|
||||
`analyst/remove.txt` and `protect.txt` are applied, before the names a
|
||||
parent already covers are dropped. Never smaller than the entries. It
|
||||
is the figure to compare with a tool that counts one per name.
|
||||
|
||||
And for each source read, beside `domains` (the names read from it),
|
||||
`unique`: how many of its names no other source of the build brings (0
|
||||
for a disabled or failed source). `domains - unique` is what it shares.
|
||||
A name that the UT1 archive holds in two folders counts twice in `domains`
|
||||
and once in `unique`.
|
||||
|
||||
`manifest.json` (`schema` 1) also holds what warda-analyst reads before it
|
||||
proposes a list or a name:
|
||||
|
||||
| Key | Content |
|
||||
| --- | --- |
|
||||
| `taxonomy.groups` | the groups of sources, in the order of `taxonomy.toml` |
|
||||
| `taxonomy.categories` | per category: `file`, `max_domains`, `public_sources` |
|
||||
| `taxonomy.bundles` | per bundle, its categories (`base`: `ads`, `tracking`) |
|
||||
| `taxonomy.formats` | the formats a source may have |
|
||||
| `taxonomy.licences` | the SPDX ids a source may carry |
|
||||
| `owner` | the names of `allow.txt` and of `protect.txt` |
|
||||
| `analyst.status` | `applied`, or `ignored: <why>` when the lists were built without the files of `analyst/` (`add` is then empty, `remove` 0) |
|
||||
| `analyst.add` | per category, the names of its `analyst/add/` file (only the files that hold some) |
|
||||
| `analyst.remove` | the lines of `analyst/remove.txt` applied |
|
||||
| `analyst.skipped` | the lines not applied because they would take more than `max_effect` names out: `{"name", "names"}`, sorted by name |
|
||||
| `analyst.covered` | the removed names a category of `remove_from` still blocks through a parent: `{"name", "parent", "file"}`, sorted, 1,000 at most |
|
||||
| `analyst.max_add`, `analyst.max_remove`, `analyst.max_effect` | the limits of `[analyst]` in `taxonomy.toml` |
|
||||
| `analyst.remove_from` | the categories `analyst/remove.txt` applies to, in the order of `taxonomy.toml` |
|
||||
|
||||
Two limits fail the build, with the file named, even with `force`:
|
||||
|
||||
@@ -233,6 +331,9 @@ Two limits fail the build, with the file named, even with `force`:
|
||||
big source, or raise `max_domains` of that category knowingly. Site
|
||||
types have no budget.
|
||||
|
||||
A limit reached only because of the names of warda-analyst does not fail
|
||||
the build: the lists are built without them (above).
|
||||
|
||||
The choices made for the budgets (comments in `sources.toml`): The Block
|
||||
List Project malware list is disabled (2.6 million names, largely stale,
|
||||
covered by HaGeZi TIF medium); the UT1 folders `adult` (5 million loose
|
||||
@@ -276,14 +377,16 @@ The CI compares the new build with the `manifest.json` published before:
|
||||
if a category loses more than half of its domains (lists of at least 100
|
||||
domains), or if `base.txt` has fewer than 50,000 domains, nothing is
|
||||
published. The build by hand (*Run workflow*) has two options: `force`
|
||||
(publish anyway) and `allow_partial` (go on when a source fails).
|
||||
(publish anyway) and `allow_partial` (go on when a source fails). The
|
||||
guard counts the lists without the names of warda-analyst; removals of
|
||||
`analyst/remove.txt` that would trip it are not published (above).
|
||||
|
||||
## Build by hand
|
||||
|
||||
Python 3.11 or later, standard library only.
|
||||
|
||||
```sh
|
||||
python3 scripts/build.py --check # validate the TOML and the owner files
|
||||
python3 scripts/build.py --check # validate the TOML, the owner and analyst files
|
||||
python3 scripts/build.py # download and build dist/
|
||||
python3 scripts/build.py --offline DIR # read each source from DIR/<name>
|
||||
python3 -m unittest discover -s scripts -v # the tests (no network)
|
||||
@@ -291,12 +394,19 @@ python3 -m unittest discover -s scripts -v # the tests (no network)
|
||||
|
||||
Other options: `--previous URL-or-file` (the guard), `--force`,
|
||||
`--allow-partial`, `--min-base N`, `--max-file-mb N`, `--out DIR`,
|
||||
`--base-url URL`.
|
||||
`--base-url URL`, `--no-source-stats` (does not count `unique`, then absent
|
||||
from `manifest.json`).
|
||||
|
||||
The build keeps every list in memory (millions of names for UT1 and the
|
||||
newly registered domains): give it a few GB. The CI limits its container
|
||||
with `docker run --memory 3g --memory-swap 3g` (`BUILD_MEMORY`); a build
|
||||
killed with exit code 137 needs more.
|
||||
killed with exit code 137 needs more. Counting `unique` is the only figure
|
||||
that costs memory, and a few seconds. Its table grows by steps, with the
|
||||
number of distinct names of the sources (printed in the log): 176 MiB at
|
||||
most up to 5.6 million names, 352 MiB from there to 10 million (measured
|
||||
by the tests: `WARDA_LISTS_STATS_NAMES=6000000`). Over 10 million it would
|
||||
take 704 MiB: the build then gives `unique` up by itself, with a warning,
|
||||
as `--no-source-stats` does on request.
|
||||
|
||||
## CI
|
||||
|
||||
@@ -304,8 +414,9 @@ killed with exit code 137 needs more.
|
||||
pull request to `main`, every day at 03:17 UTC and by hand:
|
||||
|
||||
- **test**: the unit tests, the validation of `taxonomy.toml`,
|
||||
`sources.toml` and the files of the owner, and an offline build from
|
||||
`tests/fixtures/`. Python runs in a container without network.
|
||||
`sources.toml`, the files of the owner and those of warda-analyst, and
|
||||
an offline build from `tests/fixtures/`. Python runs in a container
|
||||
without network.
|
||||
- **build** (pushes of `main`, the daily run and the run by hand, whatever
|
||||
the branch they start from: Gitea runs the schedule on the default
|
||||
branch): it always checks out `main`, builds with the memory limit, runs
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
# analyst/ — the files written by warda-analyst
|
||||
|
||||
**Never edit these files by hand.** The service warda-analyst writes them,
|
||||
by a commit on the branch `develop`, each time its team approves or
|
||||
withdraws a name. It rewrites a whole file at each publication: a change
|
||||
made by hand is lost at the next one. A name to add or to remove by hand
|
||||
goes in the files of the owner (`extra/`, `allow.txt`).
|
||||
|
||||
| File | Role |
|
||||
| --- | --- |
|
||||
| `add/<category>.txt` | names the analysis of Warda confirmed, added to that category as `extra/<category>.txt` does (never `csam`, never the bundle `base`) |
|
||||
| `remove.txt` | names confirmed as harmless and wrongly blocked: removed with their subdomains from the protection lists only (below) |
|
||||
|
||||
A missing file, or no file at all, means no name. Nothing else may be in
|
||||
this directory but this `README.md`.
|
||||
|
||||
## Where the removals apply
|
||||
|
||||
A name of `remove.txt` is removed, with its subdomains, from the
|
||||
categories named by `remove_from` of `[analyst]` in `taxonomy.toml` (`ads`,
|
||||
`tracking`, `phishing`, `security`) and from nothing else. It leaves the
|
||||
bundles built from them (`base`) and does not come back through an
|
||||
`includes` (`security` includes `phishing`). It is never removed from
|
||||
another category (`adult`, `gambling`…) nor from a site type: the team
|
||||
clears a name because it is no threat and no tracker, which says nothing
|
||||
of what the site is about. Only `allow.txt`, a file of the owner, removes
|
||||
a name from every list.
|
||||
|
||||
## How they are checked
|
||||
|
||||
`python3 scripts/build.py --check` (the CI, on every push) refuses:
|
||||
|
||||
- anything else in this directory: a file with another name
|
||||
(`removed.txt`, `add/ads.TXT`), a sub-directory, a link, `remove.txt`
|
||||
that is not a file;
|
||||
- a line that is neither a comment (it starts with `#`) nor exactly one
|
||||
name in its normalised form: lower case, no trailing dot, no space, no
|
||||
empty line, no `*.`, no comment after the name, no byte order mark, no
|
||||
carriage return;
|
||||
- names that are not sorted (in the order of their bytes, as `sort.Strings`
|
||||
of Go and `sorted` of Python give it), or a name written twice;
|
||||
- a file `add/<x>.txt` whose `<x>` is not a category that public sources
|
||||
may fill;
|
||||
- a name added and removed at once, whichever covers the other: the same
|
||||
name in an `add/` file and in `remove.txt`, an added name under a
|
||||
removed one, a removed name under an added one;
|
||||
- more names than the budgets `max_add` (all the `add/` files together) and
|
||||
`max_remove` of `[analyst]` in `taxonomy.toml`.
|
||||
|
||||
A refused commit fails the CI of `develop` and is not promoted to `main`.
|
||||
The lists are still built every day from `main`, with the files of this
|
||||
directory as they are there: those of the last commit that passed.
|
||||
|
||||
## In the build
|
||||
|
||||
Files that passed the check never stop the build, and what the build does
|
||||
not apply is said in its log and in `manifest.json` (`analyst`):
|
||||
|
||||
- **A line of `remove.txt` that would take too much out is not applied.**
|
||||
The build counts the names under each line in the lists the removals
|
||||
apply to; over `max_effect` (100, `[analyst]` in `taxonomy.toml`) the
|
||||
line is left out
|
||||
and listed in `analyst.skipped` with its count. `co.uk` or `github.io`
|
||||
are valid names: only the build knows that thousands of sites are under
|
||||
them.
|
||||
- **A removed name that stays blocked is reported.** When a source or
|
||||
`extra/` lists a parent of it in one of those lists, the boxes still
|
||||
block it: the build warns
|
||||
and lists it in `analyst.covered` (the name, the parent, the file; 1,000
|
||||
entries at most).
|
||||
- **The guards look at the sources only.** The budget of each file, the
|
||||
minimum size of `base.txt` and the guard against the previous manifest
|
||||
are checked on the lists built without these files: added names cannot
|
||||
hide a collapse of the sources.
|
||||
- **What goes wrong because of these files is not published.** If a list
|
||||
goes over its budget or over the size limit, if `base.txt` goes under
|
||||
its minimum, or if a list loses more than half of its entries, only
|
||||
once these files are applied, the lists are built without any of them:
|
||||
`analyst.status` is then `ignored: <why>` instead of `applied`, and the
|
||||
log says `IGNORED`.
|
||||
- `allow.txt` and `protect.txt` still win over an added name.
|
||||
|
||||
In the headers of the lists, `LICENSES.md` and `manifest.json`, the added
|
||||
names are attributed to the source `warda-analyst`
|
||||
(https://warda-dns.com), as own data of this repository; `manifest.json`
|
||||
gives their count per category (`analyst.add`).
|
||||
+628
-121
File diff suppressed because it is too large.
Load diff
+910
-7
@@ -3,6 +3,10 @@
|
||||
|
||||
No network: the sources are the fixtures of tests/fixtures/, a UT1-style
|
||||
archive made here, and a local HTTP server for the downloads.
|
||||
|
||||
The memory of the statistics is measured on a small synthetic set by
|
||||
default; WARDA_LISTS_STATS_NAMES=6000000 measures it at the size of the
|
||||
real build (about 20 seconds, 1.2 GB for the test itself).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -11,11 +15,14 @@ import hashlib
|
||||
import http.server
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import random
|
||||
import shutil
|
||||
import sys
|
||||
import tarfile
|
||||
import tempfile
|
||||
import threading
|
||||
import tracemalloc
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
@@ -113,8 +120,11 @@ class LicenceTest(unittest.TestCase):
|
||||
c(bad)
|
||||
|
||||
def test_every_licence_has_its_text(self):
|
||||
for lic in build.ALLOWED_LICENCES:
|
||||
self.assertTrue((ROOT / "licenses" / f"{lic}.txt").is_file(), lic)
|
||||
tax = build.load_taxonomy(ROOT / "taxonomy.toml")
|
||||
self.assertEqual(len(tax.licences), 16)
|
||||
build.check_licence_texts(tax.licences, ROOT / "licenses")
|
||||
with self.assertRaisesRegex(build.BuildError, r"no text in .* for the licence\(s\) WTFPL"):
|
||||
build.check_licence_texts([*tax.licences, "WTFPL"], ROOT / "licenses")
|
||||
|
||||
|
||||
class RepositoryFilesTest(unittest.TestCase):
|
||||
@@ -140,6 +150,10 @@ class RepositoryFilesTest(unittest.TestCase):
|
||||
self.assertFalse(tax.categories["csam"]["public_sources"])
|
||||
for c in tax.categories.values():
|
||||
self.assertTrue(c["warda"])
|
||||
self.assertEqual(tax.licences[0], "GPL-3.0-only")
|
||||
self.assertIn("Apache-2.0", tax.licences)
|
||||
self.assertEqual((tax.analyst_max_add, tax.analyst_max_remove, tax.analyst_max_effect), (50000, 20000, 100))
|
||||
self.assertEqual(tax.analyst_remove_from, ["ads", "tracking", "phishing", "security"])
|
||||
|
||||
def test_sources_and_owner_files(self):
|
||||
tax = build.load_taxonomy(ROOT / "taxonomy.toml")
|
||||
@@ -162,6 +176,11 @@ class RepositoryFilesTest(unittest.TestCase):
|
||||
owner = build.load_owner_files(ROOT, tax)
|
||||
self.assertIn("warda-dns.com", owner.protect)
|
||||
build.load_sources(FIXTURES / "sources.toml", tax)
|
||||
# analyst/ ships its README only until warda-analyst writes there;
|
||||
# whatever it holds must pass the check.
|
||||
self.assertTrue((ROOT / "analyst" / "README.md").is_file())
|
||||
build.load_analyst_files(ROOT, tax)
|
||||
build.load_analyst_files(FIXTURES, tax)
|
||||
|
||||
def test_check_command(self):
|
||||
self.assertEqual(build.main(["--check"]), 0)
|
||||
@@ -215,6 +234,80 @@ homepage = "https://example.org/"
|
||||
+ 'categories = ["ads"]\n'))
|
||||
self.assertIn("needs a [source.map]", self.errors(self.SOURCE.replace('"domains"', '"ut1"')))
|
||||
self.assertIn("only for format", self.errors(self.SOURCE + 'categories = ["ads"]\n[source.map]\na = "ads"\n'))
|
||||
self.assertIn("kept for the files of analyst/", self.errors(self.SOURCE.replace('"x"', '"warda-analyst"')
|
||||
+ 'categories = ["ads"]\n'))
|
||||
|
||||
def taxonomy_errors(self, old: str, new: str) -> str:
|
||||
text = (ROOT / "taxonomy.toml").read_text(encoding="utf-8")
|
||||
self.assertEqual(text.count(old), 1, old)
|
||||
p = self.tmp / "taxonomy.toml"
|
||||
p.write_text(text.replace(old, new), encoding="utf-8")
|
||||
with self.assertRaises(build.BuildError) as cm:
|
||||
build.load_taxonomy(p)
|
||||
return str(cm.exception)
|
||||
|
||||
def test_licences_of_the_taxonomy(self):
|
||||
# The licences are written in taxonomy.toml only: a source under one
|
||||
# that is not there is refused, and the list cannot accept NC or ND.
|
||||
self.assertIn("licences is missing", self.taxonomy_errors("licences = [", "unused = ["))
|
||||
self.assertIn("'CC-BY-NC-4.0' forbids commercial use or modification",
|
||||
self.taxonomy_errors(' "MIT",\n', ' "CC-BY-NC-4.0",\n'))
|
||||
self.assertIn("'ISC' is written twice", self.taxonomy_errors(' "MIT",\n', ' "ISC",\n'))
|
||||
self.assertIn("7 is not an SPDX id", self.taxonomy_errors(' "MIT",\n', ' 7,\n'))
|
||||
text = (ROOT / "taxonomy.toml").read_text(encoding="utf-8")
|
||||
p = self.tmp / "taxonomy.toml"
|
||||
p.write_text(text.replace(' "MIT",\n', ""), encoding="utf-8")
|
||||
tax = build.load_taxonomy(p)
|
||||
self.assertNotIn("MIT", tax.licences)
|
||||
src = self.tmp / "sources.toml"
|
||||
src.write_text(self.SOURCE + 'categories = ["ads"]\n', encoding="utf-8")
|
||||
with self.assertRaisesRegex(build.BuildError, r"licence 'MIT' is not accepted; accepted SPDX ids "
|
||||
r"\(licences in taxonomy.toml\): 0BSD, Apache-2.0"):
|
||||
build.load_sources(src, tax)
|
||||
|
||||
def test_analyst_budgets_of_the_taxonomy(self):
|
||||
self.assertIn("[analyst]: max_add must be an integer, 0 or more",
|
||||
self.taxonomy_errors("max_add = 50000", "max_add = -1"))
|
||||
self.assertIn("[analyst]: max_remove must be an integer, 0 or more",
|
||||
self.taxonomy_errors("max_remove = 20000", 'max_remove = "many"'))
|
||||
self.assertIn("[analyst]: max_effect must be an integer, 0 or more",
|
||||
self.taxonomy_errors("max_effect = 100", "max_effect = true"))
|
||||
self.assertIn("[analyst]: unknown key 'max'", self.taxonomy_errors("max_add = 50000", "max = 50000"))
|
||||
# Without the keys: the same values.
|
||||
text = (ROOT / "taxonomy.toml").read_text(encoding="utf-8")
|
||||
keys = "max_add = 50000\nmax_remove = 20000\nmax_effect = 100\n"
|
||||
self.assertEqual(text.count(keys), 1)
|
||||
p = self.tmp / "taxonomy.toml"
|
||||
p.write_text(text.replace(keys, ""), encoding="utf-8")
|
||||
tax = build.load_taxonomy(p)
|
||||
self.assertEqual((tax.analyst_max_add, tax.analyst_max_remove, tax.analyst_max_effect), (50000, 20000, 100))
|
||||
p.write_text(text.replace(keys, "max_effect = 0\n"), encoding="utf-8")
|
||||
tax = build.load_taxonomy(p)
|
||||
self.assertEqual((tax.analyst_max_add, tax.analyst_max_remove, tax.analyst_max_effect), (50000, 20000, 0))
|
||||
|
||||
def test_remove_from_of_the_taxonomy(self):
|
||||
# Where the removals of the analyst apply is written by the owner:
|
||||
# there is no default, and only categories that sources fill.
|
||||
given = 'remove_from = ["ads", "tracking", "phishing", "security"]'
|
||||
missing = "[analyst]: remove_from is missing (the categories analyst/remove.txt applies to)"
|
||||
self.assertIn(missing, self.taxonomy_errors(given + "\n", ""))
|
||||
self.assertIn(missing, self.taxonomy_errors(given, "remove_from = []"))
|
||||
self.assertIn(missing, self.taxonomy_errors(given, 'remove_from = "ads"'))
|
||||
self.assertIn(missing, self.taxonomy_errors(
|
||||
"[analyst]\nmax_add = 50000\nmax_remove = 20000\nmax_effect = 100\n" + given + "\n", ""))
|
||||
self.assertIn("remove_from: 'nope' is not a category", self.taxonomy_errors(given, 'remove_from = ["nope"]'))
|
||||
self.assertIn("remove_from: 'base' is not a category (a bundle: name its categories)",
|
||||
self.taxonomy_errors(given, 'remove_from = ["ads", "base"]'))
|
||||
self.assertIn("remove_from: 7 is not a category", self.taxonomy_errors(given, 'remove_from = ["ads", 7]'))
|
||||
self.assertIn("remove_from: 'csam' has no public source, nothing is removed from it",
|
||||
self.taxonomy_errors(given, 'remove_from = ["csam"]'))
|
||||
self.assertIn("remove_from: 'ads' is written twice",
|
||||
self.taxonomy_errors(given, 'remove_from = ["ads", "ads"]'))
|
||||
# Kept in the order of the categories, however it is written.
|
||||
p = self.tmp / "taxonomy.toml"
|
||||
p.write_text((ROOT / "taxonomy.toml").read_text(encoding="utf-8").replace(
|
||||
given, 'remove_from = ["security", "ads"]'), encoding="utf-8")
|
||||
self.assertEqual(build.load_taxonomy(p).analyst_remove_from, ["ads", "security"])
|
||||
|
||||
def test_owner_file_errors(self):
|
||||
(self.tmp / "extra").mkdir()
|
||||
@@ -234,9 +327,10 @@ def make_ut1(path: Path) -> None:
|
||||
files = {
|
||||
"blacklists/adult/domains": "porn.example.com\nwww.porn.example.com\n10.0.0.1\n",
|
||||
"blacklists/adult/urls": "porn.example.com/path\n",
|
||||
"blacklists/gambling/domains": "casino2.example.com\napproved.example.fr\nlegal-casino.example.com\n",
|
||||
"blacklists/gambling/domains": ("casino.example.com\ncasino2.example.com\napproved.example.fr\n"
|
||||
"legal-casino.example.com\n"),
|
||||
"blacklists/arjel/domains": "approved.example.fr\n",
|
||||
"blacklists/shopping/domains": "shop.example.com\n",
|
||||
"blacklists/shopping/domains": "shop.example.com\nwww.shop.example.com\n",
|
||||
"blacklists/agressif/domains": "hate.example.com\n",
|
||||
"blacklists/unused/domains": "unused.example.com\n",
|
||||
}
|
||||
@@ -272,8 +366,8 @@ missing_folder = ["drugs"]
|
||||
'''
|
||||
|
||||
|
||||
class BuildTest(unittest.TestCase):
|
||||
"""A whole offline build."""
|
||||
class OfflineBuildCase(unittest.TestCase):
|
||||
"""A copy of the repository with the fixture sources, built offline."""
|
||||
|
||||
def setUp(self):
|
||||
self.tmp = Path(tempfile.mkdtemp())
|
||||
@@ -300,6 +394,33 @@ class BuildTest(unittest.TestCase):
|
||||
return build.main(["--root", str(self.root), "--offline", str(self.offline), "--out", str(self.out),
|
||||
"--min-base", "1", *extra])
|
||||
|
||||
def built(self, *extra: str) -> tuple[dict, str]:
|
||||
"""Build, which must succeed: the manifest and the log."""
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO) as err:
|
||||
self.assertEqual(self.run_build(*extra), 0, err.getvalue())
|
||||
return json.loads((self.out / "manifest.json").read_text()), err.getvalue()
|
||||
|
||||
def check_fails(self) -> str:
|
||||
"""--check, which must fail with a message, never a traceback: its log."""
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO) as err:
|
||||
self.assertEqual(build.main(["--root", str(self.root), "--check"]), 1)
|
||||
return err.getvalue()
|
||||
|
||||
def lists(self) -> dict[str, list[str]]:
|
||||
"""Every list of dist/, without the line of its build time."""
|
||||
return {str(f.relative_to(self.out)): [line for line in f.read_text().splitlines()
|
||||
if not line.startswith("# Generated: ")]
|
||||
for f in sorted(self.out.rglob("*.txt")) if f.parent.name != "licenses"}
|
||||
|
||||
def set_taxonomy(self, old: str, new: str) -> None:
|
||||
text = (self.root / "taxonomy.toml").read_text(encoding="utf-8")
|
||||
self.assertEqual(text.count(old), 1, old)
|
||||
(self.root / "taxonomy.toml").write_text(text.replace(old, new), encoding="utf-8")
|
||||
|
||||
|
||||
class BuildTest(OfflineBuildCase):
|
||||
"""A whole offline build."""
|
||||
|
||||
def test_build(self):
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO) as err:
|
||||
self.assertEqual(self.run_build(), 0)
|
||||
@@ -369,7 +490,7 @@ class BuildTest(unittest.TestCase):
|
||||
self.assertEqual(manifest["sources"]["fx-ut1"]["group"], "Other Lists")
|
||||
self.assertEqual(manifest["files"]["newdomains.txt"]["max_domains"], 3500000)
|
||||
self.assertIsNone(manifest["files"]["types/blog.txt"]["max_domains"])
|
||||
self.assertEqual(manifest["sources"]["fx-domains"]["domains"], 11)
|
||||
self.assertEqual(manifest["sources"]["fx-domains"]["domains"], 12)
|
||||
self.assertEqual(manifest["sources"]["fx-ut1"]["unmapped_folders"], ["unused"])
|
||||
self.assertEqual(manifest["protected_removed"]["ads.txt"], ["google.com", "www.google.com"])
|
||||
|
||||
@@ -463,6 +584,788 @@ class BuildTest(unittest.TestCase):
|
||||
self.assertTrue((self.out / "precious.txt").exists())
|
||||
|
||||
|
||||
NO_ANALYST = {"status": "applied", "add": {}, "remove": 0, "skipped": [], "covered": [],
|
||||
"max_add": 50000, "max_remove": 20000, "max_effect": 100,
|
||||
"remove_from": ["ads", "tracking", "phishing", "security"]}
|
||||
|
||||
|
||||
class StatisticsTest(OfflineBuildCase):
|
||||
"""The figures of the manifest, of the headers and of dist/README.md."""
|
||||
|
||||
def test_names_and_count(self):
|
||||
manifest, log = self.built()
|
||||
files = manifest["files"]
|
||||
# fx-domains gives 12 names to ads: 3 go with allow.txt, 2 with
|
||||
# protect.txt, and sub.ads.example.com is covered by its parent.
|
||||
self.assertEqual((files["ads.txt"]["count"], files["ads.txt"]["names"]), (6, 7))
|
||||
self.assertEqual((files["base.txt"]["count"], files["base.txt"]["names"]), (6, 7))
|
||||
# login.phish.example.com is covered; security includes phishing.
|
||||
self.assertEqual((files["phishing.txt"]["count"], files["phishing.txt"]["names"]), (2, 3))
|
||||
self.assertEqual((files["security.txt"]["count"], files["security.txt"]["names"]), (2, 3))
|
||||
self.assertEqual((files["adult.txt"]["count"], files["adult.txt"]["names"]), (1, 2))
|
||||
self.assertEqual((files["gambling.txt"]["count"], files["gambling.txt"]["names"]), (5, 5))
|
||||
self.assertEqual((files["csam.txt"]["count"], files["csam.txt"]["names"]), (0, 0))
|
||||
self.assertEqual((files["types/blog.txt"]["count"], files["types/blog.txt"]["names"]), (1, 1))
|
||||
# A site type too: www.shop.example.com is a name, not an entry.
|
||||
self.assertEqual((files["types/e-commerce.txt"]["count"], files["types/e-commerce.txt"]["names"]), (1, 2))
|
||||
for rel, info in files.items():
|
||||
self.assertGreaterEqual(info["names"], info["count"], rel)
|
||||
self.assertEqual(manifest["schema"], 1)
|
||||
self.assertIn("ads.txt: 6 domains for 7 names", log)
|
||||
|
||||
header = (self.out / "ads.txt").read_text().splitlines()
|
||||
self.assertRegex(header[2], r"^# Generated: \d{4}-\d\d-\d\dT\d\d:\d\d:\d\dZ$")
|
||||
self.assertEqual(header[2], f"# Generated: {manifest['generated']}")
|
||||
self.assertEqual(header[3:5], ["# Domains: 6", "# Names: 7"])
|
||||
for rel in files:
|
||||
head = (self.out / rel).read_text().splitlines()[:5]
|
||||
self.assertEqual(head[3:], [f"# Domains: {files[rel]['count']}", f"# Names: {files[rel]['names']}"], rel)
|
||||
|
||||
readme = (self.out / "README.md").read_text()
|
||||
self.assertIn("| File | Content | Warda category | Domains | Names | Address |", readme)
|
||||
self.assertRegex(readme, r"\| ads.txt \| Advertising \(Publicité\) \| base \| 6 \| 7 \| \S+/ads.txt \|")
|
||||
self.assertRegex(readme, r"\| adult.txt \| [^|]+ \| adult \| 1 \| 2 \|")
|
||||
|
||||
def test_names_of_a_bundle(self):
|
||||
# example.net covers tracker.example.net in ads, where it is dropped,
|
||||
# while tracking keeps it as an entry: base stands for it once, and
|
||||
# for sub.ads.example.com, which both categories dropped.
|
||||
(self.root / "extra" / "ads.txt").write_text("example.net\n")
|
||||
files = self.built()[0]["files"]
|
||||
self.assertEqual((files["ads.txt"]["count"], files["ads.txt"]["names"]), (6, 8))
|
||||
self.assertEqual((files["tracking.txt"]["count"], files["tracking.txt"]["names"]), (6, 7))
|
||||
self.assertEqual((files["base.txt"]["count"], files["base.txt"]["names"]), (6, 8))
|
||||
self.assertNotIn("tracker.example.net", lines_of(self.out / "base.txt"))
|
||||
self.assertIn("tracker.example.net", lines_of(self.out / "tracking.txt"))
|
||||
|
||||
def test_unique_of_a_source(self):
|
||||
manifest, log = self.built()
|
||||
# phish2.example.com is in fx-domains and fx-hosts, casino.example.com
|
||||
# in fx-adblock and fx-ut1. In fx-ut1, approved.example.fr is read in
|
||||
# two folders: counted twice in domains, once in unique.
|
||||
figures = {name: (info.get("domains"), info.get("unique")) for name, info in manifest["sources"].items()}
|
||||
self.assertEqual(figures, {"fx-domains": (12, 11), "fx-hosts": (3, 2), "fx-adblock": (2, 1),
|
||||
"fx-disabled": (None, 0), "fx-ut1": (10, 8)})
|
||||
self.assertIn("24 distinct names in the sources", log)
|
||||
readme = (self.out / "README.md").read_text()
|
||||
self.assertIn("| Source | Group | Status | Domains | Unique |", readme)
|
||||
self.assertIn("| fx-hosts | Phishing List | ok | 3 | 2 |", readme)
|
||||
self.assertIn("| fx-disabled | Drugs List | disabled | - | 0 |", readme)
|
||||
|
||||
def test_unique_of_a_failed_source(self):
|
||||
(self.offline / "fx-hosts.txt").unlink()
|
||||
sources = self.built("--allow-partial")[0]["sources"]
|
||||
self.assertEqual(sources["fx-hosts"]["unique"], 0)
|
||||
self.assertEqual(sources["fx-domains"]["unique"], 12) # nobody else brings phish2.example.com now
|
||||
|
||||
def test_no_source_stats(self):
|
||||
manifest, log = self.built("--no-source-stats")
|
||||
for name, info in manifest["sources"].items():
|
||||
self.assertNotIn("unique", info, name)
|
||||
self.assertNotIn("distinct names", log)
|
||||
self.assertEqual(manifest["sources"]["fx-domains"]["domains"], 12)
|
||||
self.assertEqual(manifest["files"]["ads.txt"]["names"], 7)
|
||||
self.assertIn("# Names: 7", (self.out / "ads.txt").read_text().splitlines())
|
||||
self.assertIn("| fx-hosts | Phishing List | ok | 3 | - |", (self.out / "README.md").read_text())
|
||||
|
||||
def test_unique_is_given_up_over_the_limit(self):
|
||||
# Over the limit the table of the count would double: the build goes
|
||||
# on without 'unique', as with --no-source-stats, and says so.
|
||||
with mock.patch.object(build, "UNIQUE_MAX_NAMES", 23):
|
||||
manifest, log = self.built()
|
||||
self.assertIn("WARNING more than 23 distinct names in the sources", log)
|
||||
for name, info in manifest["sources"].items():
|
||||
self.assertNotIn("unique", info, name)
|
||||
self.assertEqual(manifest["files"]["ads.txt"]["names"], 7)
|
||||
with mock.patch.object(build, "UNIQUE_MAX_NAMES", 24):
|
||||
manifest, log = self.built()
|
||||
self.assertIn("24 distinct names in the sources", log)
|
||||
self.assertEqual(manifest["sources"]["fx-hosts"]["unique"], 2)
|
||||
|
||||
def test_taxonomy_owner_and_analyst_of_the_manifest(self):
|
||||
tax = build.load_taxonomy(self.root / "taxonomy.toml")
|
||||
manifest, _ = self.built()
|
||||
got = manifest["taxonomy"]
|
||||
self.assertEqual(sorted(got), ["bundles", "categories", "formats", "groups", "licences"])
|
||||
self.assertEqual(got["groups"], list(tax.source_groups))
|
||||
self.assertEqual(got["groups"][0], "Hosts list")
|
||||
self.assertEqual(list(got["categories"]), list(tax.categories))
|
||||
self.assertEqual(got["categories"]["newdomains"],
|
||||
{"file": "newdomains.txt", "max_domains": 3500000, "public_sources": True})
|
||||
self.assertEqual(got["categories"]["csam"],
|
||||
{"file": "csam.txt", "max_domains": 1500000, "public_sources": False})
|
||||
for cid, c in got["categories"].items():
|
||||
self.assertEqual(c["max_domains"], manifest["files"][c["file"]]["max_domains"], cid)
|
||||
self.assertEqual(got["bundles"], {"base": ["ads", "tracking"]})
|
||||
self.assertEqual(got["formats"], ["domains", "hosts", "adblock", "ut1"])
|
||||
self.assertEqual(got["licences"], tax.licences)
|
||||
self.assertEqual(manifest["owner"], {"allow": ["allowed.example.com", "sub.porn.example.com",
|
||||
"wild.example.org"],
|
||||
"protect": ["google.com", "www.google.com"]})
|
||||
# No analyst/ directory at all: no name.
|
||||
self.assertFalse((self.root / "analyst").exists())
|
||||
self.assertEqual(manifest["analyst"], NO_ANALYST)
|
||||
|
||||
def test_unique_names(self):
|
||||
ads, tracking, blog = (build.Target("category", "ads"), build.Target("category", "tracking"),
|
||||
build.Target("type", "blog"))
|
||||
results = {
|
||||
"one": build.Ingested(domains={ads: {"a.example.com", "b.example.com"},
|
||||
tracking: {"a.example.com", "c.example.com"}}),
|
||||
"two": build.Ingested(domains={blog: {"c.example.com", "d.example.com"}}),
|
||||
"three": build.Ingested(domains={ads: {"c.example.com"}}),
|
||||
"four": build.Ingested(),
|
||||
}
|
||||
self.assertEqual(build.unique_names(results, 4), ({"one": 2, "two": 1, "three": 0, "four": 0}, 4))
|
||||
# One name too many: given up before the set that brings it is read.
|
||||
self.assertEqual(build.unique_names(results, 3), (None, 3))
|
||||
self.assertGreater(11184810, build.UNIQUE_MAX_NAMES) # where the table of 2**24 slots doubles
|
||||
|
||||
def test_memory_of_unique_names(self):
|
||||
# The nightly build runs in 3 GB: counting the unique names may add
|
||||
# 500 MB at 6 million distinct names, that is 87 bytes a name. The
|
||||
# default size sits just over a growth step of the dictionary, where
|
||||
# a name costs the most. Only the count is traced: the names exist
|
||||
# before, as the sources do in a build.
|
||||
n = int(os.environ.get("WARDA_LISTS_STATS_NAMES", "180000"))
|
||||
target = build.Target("category", "ads")
|
||||
names = [f"h{i:07d}.example.com" for i in range(n)]
|
||||
results = {
|
||||
"first": build.Ingested(domains={target: set(names[:2 * n // 3])}),
|
||||
"second": build.Ingested(domains={target: set(names[n // 3:])}),
|
||||
}
|
||||
tracemalloc.start()
|
||||
try:
|
||||
unique, distinct = build.unique_names(results, n)
|
||||
peak = tracemalloc.get_traced_memory()[1]
|
||||
finally:
|
||||
tracemalloc.stop()
|
||||
self.assertEqual((unique, distinct), ({"first": n // 3, "second": n - 2 * n // 3}, n))
|
||||
print(f"unique_names: {n} names, peak {peak / 1048576:.1f} MiB, {peak / n:.1f} bytes a name",
|
||||
file=sys.stderr)
|
||||
self.assertLess(peak / n, 500 * 1024 * 1024 / 6000000)
|
||||
|
||||
|
||||
class SettleTest(unittest.TestCase):
|
||||
"""A file with and without the files of warda-analyst, against a plain
|
||||
rewriting of the rules."""
|
||||
|
||||
UNIVERSE = [f"{sub}{name}" for name in ("a.com", "b.org", "c.net", "www.d.io")
|
||||
for sub in ("", "x.", "y.", "z.x.", "w.z.x.")]
|
||||
|
||||
@staticmethod
|
||||
def under(names: set[str], roots: set[str]) -> set[str]:
|
||||
return {d for d in names if d in roots or any(p in roots for p in build.parents(d))}
|
||||
|
||||
@staticmethod
|
||||
def entries(names: set[str]) -> set[str]:
|
||||
return {d for d in names if not any(p in names for p in build.parents(d))}
|
||||
|
||||
def test_against_the_rules(self):
|
||||
rng = random.Random(20261004)
|
||||
changed = 0
|
||||
for _ in range(2000):
|
||||
def some(most: int) -> set[str]:
|
||||
return set(rng.sample(self.UNIVERSE, rng.randint(0, most)))
|
||||
names, allow, protect, remove = some(14), some(2), some(2), some(3)
|
||||
closure = build.protected_closure(protect)
|
||||
gone = self.under(set(self.UNIVERSE), remove)
|
||||
confirmed = some(4) - gone # the check refuses a name both added and removed
|
||||
|
||||
kept = names - self.under(names, allow)
|
||||
wanted = confirmed - self.under(confirmed, allow)
|
||||
before = kept - set(closure)
|
||||
after = (before - gone) | (wanted - set(closure))
|
||||
|
||||
s = set(names)
|
||||
done = build.settle(s, allow, closure, confirmed, gone)
|
||||
self.assertEqual(s, self.entries(after))
|
||||
self.assertEqual(done.names, len(after))
|
||||
self.assertEqual(done.covered, after - self.entries(after))
|
||||
self.assertEqual(done.plain_covered, before - self.entries(before))
|
||||
self.assertEqual(done.protected, sorted((kept | wanted) & set(closure)))
|
||||
if done.plain is None:
|
||||
self.assertEqual(self.entries(before), s)
|
||||
self.assertEqual(len(before), done.names)
|
||||
continue
|
||||
changed += 1
|
||||
self.assertEqual((s - done.plain.out) | done.plain.back, self.entries(before))
|
||||
self.assertLessEqual(done.plain.out, s)
|
||||
self.assertFalse(done.plain.back & s)
|
||||
self.assertEqual((done.plain.count, done.plain.names), (len(self.entries(before)), len(before)))
|
||||
self.assertEqual(done.plain.protected, sorted(kept & set(closure)))
|
||||
self.assertGreater(changed, 1000)
|
||||
|
||||
def test_removals(self):
|
||||
lists = [{"a.x.com", "b.x.com", "y.org"}, {"a.x.com", "c.a.x.com", "deep.y.org"}, {"z.net"}]
|
||||
# x.com takes 3 distinct names out (a.x.com is in two lists), a.x.com
|
||||
# 2 of them, y.org 2, z.net 1, and nothing is under q.io.
|
||||
remove = {"x.com", "a.x.com", "y.org", "z.net", "q.io"}
|
||||
gone, applied, skipped = build.removals(remove, 3, lists)
|
||||
self.assertEqual((gone, applied, skipped),
|
||||
({"a.x.com", "b.x.com", "c.a.x.com", "y.org", "deep.y.org", "z.net"}, remove, []))
|
||||
gone, applied, skipped = build.removals(remove, 2, lists)
|
||||
self.assertEqual(skipped, [("x.com", 3)])
|
||||
self.assertEqual(applied, remove - {"x.com"})
|
||||
self.assertEqual(gone, {"a.x.com", "c.a.x.com", "y.org", "deep.y.org", "z.net"}) # b.x.com stays
|
||||
gone, applied, skipped = build.removals(remove, 0, lists)
|
||||
self.assertEqual((gone, applied), (set(), {"q.io"}))
|
||||
self.assertEqual(skipped, [("a.x.com", 2), ("x.com", 3), ("y.org", 2), ("z.net", 1)])
|
||||
self.assertEqual(build.removals(set(), 0, lists), (set(), set(), []))
|
||||
|
||||
|
||||
class ComputeTest(unittest.TestCase):
|
||||
"""Whole builds in memory against a plain rewriting of the rules: which
|
||||
lists the removals of warda-analyst reach, the includes, the bundle,
|
||||
max_effect, and the way back to the lists without the analyst."""
|
||||
|
||||
CATEGORIES = ("ads", "tracking", "phishing", "security", "adult", "gambling")
|
||||
|
||||
def reference(self, tax, src, blog, allow, protect, add, remove):
|
||||
"""What every file must hold: {rel: (names, protected)}, and the
|
||||
lines of remove applied."""
|
||||
under, closure = SettleTest.under, set(build.protected_closure(protect))
|
||||
|
||||
def members(cid):
|
||||
return (cid, *tax.categories[cid].get("includes", []))
|
||||
|
||||
reached = set().union(*(src[m] for cid in tax.analyst_remove_from for m in members(cid)))
|
||||
applied = {r for r in remove if len(under(reached, {r})) <= tax.analyst_max_effect}
|
||||
out = {}
|
||||
for cid in self.CATEGORIES:
|
||||
names = set().union(*(src[m] for m in members(cid)))
|
||||
names -= under(names, allow)
|
||||
added = set().union(*(add.get(m, set()) for m in members(cid)))
|
||||
added -= under(added, allow)
|
||||
# protect.txt reports what the sources and the analyst list,
|
||||
# whether or not a removal would have taken it out as well.
|
||||
protected = sorted((names | added) & closure)
|
||||
if cid in tax.analyst_remove_from:
|
||||
names -= under(names, applied)
|
||||
out[f"{cid}.txt"] = ((names | added) - closure, protected)
|
||||
out["base.txt"] = (out["ads.txt"][0] | out["tracking.txt"][0], [])
|
||||
out["types/blog.txt"] = (blog - under(blog, allow), [])
|
||||
return out, applied
|
||||
|
||||
def check(self, outputs, want):
|
||||
got = {o.rel: o for o in outputs}
|
||||
for rel, (names, protected) in want.items():
|
||||
o = got[rel]
|
||||
self.assertEqual(o.domains, SettleTest.entries(names), rel)
|
||||
self.assertEqual((o.count, o.names), (len(o.domains), len(names)), rel)
|
||||
self.assertEqual(o.protected, protected, rel)
|
||||
|
||||
def test_against_the_rules(self):
|
||||
rng = random.Random(41)
|
||||
universe = SettleTest.UNIVERSE
|
||||
ignored = 0
|
||||
for round_ in range(600):
|
||||
def some(most: int) -> set[str]:
|
||||
return set(rng.sample(universe, rng.randint(0, most)))
|
||||
tax = build.load_taxonomy(ROOT / "taxonomy.toml") if round_ % 100 == 0 else tax
|
||||
tax.analyst_max_effect = rng.choice((0, 1, 2, 100))
|
||||
tax.analyst_remove_from = [c for c in ("ads", "tracking", "phishing", "security") if rng.random() < 0.8]
|
||||
src = {cid: some(8) for cid in self.CATEGORIES}
|
||||
blog, allow, protect, remove = some(6), some(2), some(2), some(3)
|
||||
# As the check wants them: no name both added and removed.
|
||||
add = {cid: some(3) - SettleTest.under(set(universe), remove) for cid in ("ads", "phishing", "gambling")}
|
||||
add = {cid: {d for d in names if not any(d in build.parents(r) for r in remove)}
|
||||
for cid, names in add.items()}
|
||||
|
||||
results = {f"s-{cid}": build.Ingested(domains={build.Target("category", cid): set(names)})
|
||||
for cid, names in src.items()}
|
||||
results["s-blog"] = build.Ingested(domains={build.Target("type", "blog"): set(blog)})
|
||||
owner = build.OwnerFiles({}, {}, allow, protect)
|
||||
analyst = build.AnalystFiles({cid: names for cid, names in add.items() if names}, remove)
|
||||
outputs, applied, skipped = build.compute(tax, [], owner, analyst, results)
|
||||
|
||||
want, want_applied = self.reference(tax, src, blog, allow, protect, add, remove)
|
||||
self.assertEqual(applied, want_applied)
|
||||
self.assertEqual({name for name, _ in skipped}, remove - want_applied)
|
||||
self.check(outputs, want)
|
||||
for o in outputs:
|
||||
if o.kind == "type":
|
||||
self.assertIsNone(o.plain, o.rel) # the analyst never reaches a site type
|
||||
|
||||
plain, _ = self.reference(tax, src, blog, allow, protect, {}, set())
|
||||
ignored += any(o.plain for o in outputs)
|
||||
build.without_analyst(outputs)
|
||||
self.check(outputs, plain)
|
||||
self.assertGreater(ignored, 400)
|
||||
|
||||
|
||||
class AnalystTest(OfflineBuildCase):
|
||||
"""The files written by warda-analyst: analyst/add/ and analyst/remove.txt."""
|
||||
|
||||
def setUp(self):
|
||||
super().setUp()
|
||||
self.analyst = self.root / "analyst"
|
||||
shutil.copytree(FIXTURES / "analyst", self.analyst)
|
||||
|
||||
def lists_without_analyst(self) -> dict[str, list[str]]:
|
||||
"""The lists of a build of the same repository without analyst/."""
|
||||
aside = self.tmp / "analyst-aside"
|
||||
self.analyst.rename(aside)
|
||||
manifest, _ = self.built()
|
||||
self.assertEqual(manifest["analyst"], NO_ANALYST)
|
||||
aside.rename(self.analyst)
|
||||
return self.lists()
|
||||
|
||||
def test_added_and_removed(self):
|
||||
(self.analyst / "add" / "drugs.txt").write_text("# only the analyst fills it\nweed.example.com\n")
|
||||
(self.analyst / "README.md").write_text("not a list\n")
|
||||
manifest, log = self.built()
|
||||
files = manifest["files"]
|
||||
|
||||
# Added as extra/ adds: the name, its subdomain covered by it, and a
|
||||
# protected name that protect.txt still removes.
|
||||
self.assertEqual(lines_of(self.out / "phishing.txt"),
|
||||
["confirmed.example.org", "phish.example.com", "phish2.example.com"])
|
||||
self.assertEqual(lines_of(self.out / "security.txt"), lines_of(self.out / "phishing.txt"))
|
||||
self.assertEqual((files["phishing.txt"]["count"], files["phishing.txt"]["names"]), (3, 5))
|
||||
self.assertEqual(manifest["protected_removed"]["phishing.txt"], ["www.google.com"])
|
||||
self.assertEqual(manifest["protected_removed"]["security.txt"], ["www.google.com"])
|
||||
self.assertIn("PROTECTED: removed from phishing.txt: www.google.com", log)
|
||||
|
||||
# Attributed to the source warda-analyst, own data of the repository
|
||||
# as extra/ is: no licence added to the file.
|
||||
self.assertEqual(files["phishing.txt"]["sources"], ["fx-hosts", "warda-analyst"])
|
||||
self.assertEqual(files["security.txt"]["sources"], ["fx-hosts", "warda-analyst"])
|
||||
self.assertEqual(files["phishing.txt"]["licences"], ["Unlicense"])
|
||||
self.assertEqual(files["phishing.txt"]["licence"], "Unlicense")
|
||||
header = (self.out / "phishing.txt").read_text().splitlines()
|
||||
self.assertIn("# Licence: Unlicense (data under Unlicense)", header)
|
||||
self.assertIn("# Source: fx-hosts (Unlicense) https://example.org/", header)
|
||||
self.assertIn("# Source: warda-analyst (maintainer) https://warda-dns.com", header)
|
||||
self.assertEqual(lines_of(self.out / "drugs.txt"), ["weed.example.com"])
|
||||
self.assertEqual(files["drugs.txt"]["sources"], ["warda-analyst"])
|
||||
self.assertEqual(files["drugs.txt"]["licences"], [])
|
||||
self.assertEqual(files["drugs.txt"]["licence"], "maintainer additions only")
|
||||
self.assertIn("# Licence: maintainer additions only", (self.out / "drugs.txt").read_text().splitlines())
|
||||
self.assertNotIn("warda-analyst", manifest["sources"])
|
||||
licenses = (self.out / "LICENSES.md").read_text()
|
||||
self.assertIn("| phishing.txt | Unlicense | Unlicense | fx-hosts, warda-analyst |", licenses)
|
||||
self.assertIn("| drugs.txt | maintainer additions only | - | warda-analyst |", licenses)
|
||||
self.assertIn("`warda-analyst` (names confirmed by the analysis of Warda, https://warda-dns.com)", licenses)
|
||||
|
||||
# Removed from the protection lists: tracker.example.net, a subdomain
|
||||
# of example.net, leaves ads, tracking and their bundle. A content
|
||||
# category is not concerned: casino.example.com stays in gambling.
|
||||
gambling = lines_of(self.out / "gambling.txt")
|
||||
self.assertEqual(gambling, ["approved.example.fr", "bet.example.com", "casino.example.com",
|
||||
"casino2.example.com", "legal-casino.example.com"])
|
||||
self.assertEqual(files["gambling.txt"]["names"], 5)
|
||||
for rel in ("ads.txt", "tracking.txt", "base.txt"):
|
||||
self.assertNotIn("tracker.example.net", lines_of(self.out / rel), rel)
|
||||
self.assertEqual((files[rel]["count"], files[rel]["names"]), (5, 6), rel)
|
||||
|
||||
self.assertEqual(manifest["analyst"], NO_ANALYST | {"add": {"drugs": 1, "phishing": 3}, "remove": 2})
|
||||
self.assertNotIn("IGNORED", log)
|
||||
# The sources are counted without the files of the analyst.
|
||||
self.assertEqual(manifest["sources"]["fx-hosts"]["unique"], 2)
|
||||
|
||||
def test_owner_wins_over_an_added_name(self):
|
||||
# allow.txt wins over an added name; an exception of a source does
|
||||
# not win over an addition.
|
||||
(self.analyst / "add" / "gambling.txt").write_text("allowed.example.com\napproved.example.fr\n")
|
||||
(self.root / "extra" / "gambling.txt").unlink()
|
||||
self.built()
|
||||
gambling = lines_of(self.out / "gambling.txt")
|
||||
self.assertNotIn("allowed.example.com", gambling)
|
||||
self.assertIn("approved.example.fr", gambling)
|
||||
|
||||
def test_removed_from_the_protection_lists_only(self):
|
||||
# The team cleared tracker.example.net: no tracker there. The same
|
||||
# name, and a subdomain of it, are also listed as adult content, as
|
||||
# gambling and as a blog: that is another question, they stay.
|
||||
(self.root / "extra" / "adult.txt").write_text("deep.tracker.example.net\ntracker.example.net\n")
|
||||
(self.root / "extra" / "gambling.txt").write_text("approved.example.fr\nsub.tracker.example.net\n")
|
||||
(self.root / "extra" / "security.txt").write_text("sub.tracker.example.net\n")
|
||||
(self.root / "extra" / "types" / "blog.txt").write_text("blog.example.com\ntracker.example.net\n")
|
||||
(self.analyst / "remove.txt").write_text("tracker.example.net\n")
|
||||
without = self.lists_without_analyst()
|
||||
for rel in ("ads.txt", "tracking.txt", "base.txt"):
|
||||
self.assertIn("tracker.example.net", lines_of(self.out / rel), rel)
|
||||
self.assertIn("sub.tracker.example.net", lines_of(self.out / "security.txt"))
|
||||
|
||||
manifest, _ = self.built()
|
||||
files = manifest["files"]
|
||||
for rel in ("ads.txt", "tracking.txt", "base.txt"):
|
||||
self.assertNotIn("tracker.example.net", lines_of(self.out / rel), rel)
|
||||
self.assertEqual((files[rel]["count"], files[rel]["names"]), (5, 6), rel)
|
||||
self.assertNotIn("sub.tracker.example.net", lines_of(self.out / "security.txt")) # its subdomains too
|
||||
self.assertEqual(lines_of(self.out / "adult.txt"), ["porn.example.com", "tracker.example.net"])
|
||||
self.assertEqual(files["adult.txt"]["names"], 4) # www.porn and deep.tracker are names only
|
||||
self.assertIn("sub.tracker.example.net", lines_of(self.out / "gambling.txt"))
|
||||
self.assertEqual(lines_of(self.out / "types" / "blog.txt"), ["blog.example.com", "tracker.example.net"])
|
||||
# Nothing else than the protection lists and their bundle changed.
|
||||
got = self.lists()
|
||||
changed = sorted(rel for rel in got if got[rel] != without[rel])
|
||||
self.assertEqual(changed, ["ads.txt", "base.txt", "phishing.txt", "security.txt", "tracking.txt"])
|
||||
self.assertEqual(manifest["analyst"]["remove"], 1)
|
||||
|
||||
def test_removed_name_does_not_come_back_through_an_include(self):
|
||||
# security includes phishing: a name removed from security does not
|
||||
# come back with the names of phishing, even when phishing itself
|
||||
# is not a list the removals apply to.
|
||||
(self.analyst / "remove.txt").write_text("phish.example.com\n")
|
||||
manifest, _ = self.built()
|
||||
self.assertNotIn("phish.example.com", lines_of(self.out / "phishing.txt"))
|
||||
self.assertNotIn("phish.example.com", lines_of(self.out / "security.txt"))
|
||||
self.assertEqual(manifest["files"]["security.txt"]["names"], 3) # login.phish.example.com went with it
|
||||
self.set_taxonomy('remove_from = ["ads", "tracking", "phishing", "security"]', 'remove_from = ["security"]')
|
||||
manifest, _ = self.built()
|
||||
self.assertEqual(manifest["analyst"]["remove_from"], ["security"])
|
||||
self.assertIn("phish.example.com", lines_of(self.out / "phishing.txt"))
|
||||
self.assertNotIn("phish.example.com", lines_of(self.out / "security.txt"))
|
||||
self.assertEqual((manifest["files"]["phishing.txt"]["names"], manifest["files"]["security.txt"]["names"]),
|
||||
(5, 3))
|
||||
|
||||
def test_removed_name_still_covered(self):
|
||||
# A removal that cannot work is said: a protection list holds a
|
||||
# parent of the name, which stays blocked. In the log and in the
|
||||
# manifest. A parent in a content category is not a failure.
|
||||
(self.analyst / "remove.txt").write_text(
|
||||
"deep.phish.example.com\nsub.ads.example.com\nwww.casino2.example.com\n")
|
||||
manifest, log = self.built()
|
||||
self.assertEqual(manifest["analyst"]["covered"], [
|
||||
{"name": "deep.phish.example.com", "parent": "phish.example.com", "file": "phishing.txt"},
|
||||
{"name": "deep.phish.example.com", "parent": "phish.example.com", "file": "security.txt"},
|
||||
{"name": "sub.ads.example.com", "parent": "ads.example.com", "file": "ads.txt"},
|
||||
{"name": "sub.ads.example.com", "parent": "ads.example.com", "file": "tracking.txt"}])
|
||||
self.assertEqual(manifest["analyst"]["remove"], 3)
|
||||
self.assertIn("WARNING analyst/remove.txt: sub.ads.example.com is still covered by ads.example.com "
|
||||
"in ads.txt", log)
|
||||
self.assertNotIn("casino2.example.com", log)
|
||||
self.assertEqual(manifest["files"]["ads.txt"]["names"], 6) # the name itself is gone
|
||||
# The list of the manifest is bounded; the log says how many are left out.
|
||||
with mock.patch.object(build, "ANALYST_MAX_COVERED", 1):
|
||||
manifest, log = self.built()
|
||||
self.assertEqual([(c["name"], c["file"]) for c in manifest["analyst"]["covered"]],
|
||||
[("deep.phish.example.com", "phishing.txt")])
|
||||
self.assertIn("WARNING analyst/remove.txt: 3 more removed names are still covered by a parent", log)
|
||||
|
||||
def test_suffix_is_not_applied(self):
|
||||
# co.uk is a valid name, and the check cannot know what is under it.
|
||||
# The build counts: 101 names of the lists the removals apply to,
|
||||
# over max_effect (100). The names of the other lists do not count.
|
||||
sites = [f"site{i:03d}.co.uk" for i in range(101)]
|
||||
(self.root / "extra" / "security.txt").write_text("\n".join(sites[:60]) + "\n")
|
||||
(self.root / "extra" / "ads.txt").write_text("\n".join(sites[40:]) + "\n")
|
||||
(self.root / "extra" / "jobsearch.txt").write_text(
|
||||
"jobs.example.com\n" + "".join(f"job{i:03d}.co.uk\n" for i in range(300)))
|
||||
(self.analyst / "remove.txt").write_text("co.uk\nexample.net\n")
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO):
|
||||
self.assertEqual(build.main(["--root", str(self.root), "--check"]), 0)
|
||||
manifest, log = self.built()
|
||||
self.assertEqual(manifest["analyst"], NO_ANALYST | {
|
||||
"add": {"phishing": 3}, "remove": 1, "skipped": [{"name": "co.uk", "names": 101}]})
|
||||
self.assertIn("WARNING analyst/remove.txt: co.uk is NOT applied: it would take 101 names out of "
|
||||
"ads, tracking, phishing, security, more than 100 (max_effect of [analyst] in taxonomy.toml)",
|
||||
log)
|
||||
self.assertEqual([d for d in lines_of(self.out / "ads.txt") if d.endswith(".co.uk")], sites[40:])
|
||||
self.assertNotIn("tracker.example.net", lines_of(self.out / "ads.txt")) # the other line is applied
|
||||
# 100 names: applied, in the protection lists and nowhere else.
|
||||
(self.root / "extra" / "ads.txt").write_text("\n".join(sites[40:100]) + "\n")
|
||||
manifest, log = self.built()
|
||||
self.assertEqual((manifest["analyst"]["remove"], manifest["analyst"]["skipped"]), (2, []))
|
||||
for rel in ("ads.txt", "base.txt", "security.txt"):
|
||||
self.assertEqual([d for d in lines_of(self.out / rel) if d.endswith(".co.uk")], [], rel)
|
||||
self.assertEqual(manifest["files"]["jobsearch.txt"]["count"], 301)
|
||||
self.assertNotIn("NOT applied", log)
|
||||
# The owner sets the limit.
|
||||
self.set_taxonomy("max_effect = 100", "max_effect = 99")
|
||||
manifest, _ = self.built()
|
||||
self.assertEqual(manifest["analyst"]["skipped"], [{"name": "co.uk", "names": 100}])
|
||||
self.assertEqual(manifest["analyst"]["max_effect"], 99)
|
||||
|
||||
def test_empty_files_change_nothing(self):
|
||||
# A build without analyst/, and one with a directory that holds no
|
||||
# name: the same lists, byte for byte (but the time of the build).
|
||||
shutil.rmtree(self.analyst)
|
||||
self.assertEqual(self.built()[0]["analyst"], NO_ANALYST)
|
||||
without = self.lists()
|
||||
self.assertEqual(len(without), 49)
|
||||
(self.analyst / "add").mkdir(parents=True)
|
||||
(self.analyst / "README.md").write_text("read me\n")
|
||||
manifest, _ = self.built()
|
||||
self.assertEqual(self.lists(), without)
|
||||
(self.analyst / "add" / "phishing.txt").write_text("# nothing confirmed yet\n")
|
||||
(self.analyst / "remove.txt").write_text("# header\n# no name\n")
|
||||
manifest, _ = self.built()
|
||||
self.assertEqual(self.lists(), without)
|
||||
self.assertEqual(manifest["analyst"], NO_ANALYST)
|
||||
(self.analyst / "remove.txt").write_text("")
|
||||
self.assertEqual(self.built()[0]["analyst"], NO_ANALYST)
|
||||
|
||||
def test_empty_or_missing(self):
|
||||
tax = build.load_taxonomy(self.root / "taxonomy.toml")
|
||||
# Comments only, a file without its last end of line, no file, no
|
||||
# directory: all fine.
|
||||
(self.analyst / "add" / "phishing.txt").write_text("# nothing confirmed yet\n")
|
||||
(self.analyst / "remove.txt").write_text("# header\nexample.net")
|
||||
got = build.load_analyst_files(self.root, tax)
|
||||
self.assertEqual((got.add, got.remove), ({}, {"example.net"}))
|
||||
shutil.rmtree(self.analyst / "add")
|
||||
(self.analyst / "remove.txt").unlink()
|
||||
got = build.load_analyst_files(self.root, tax)
|
||||
self.assertEqual((got.add, got.remove, got.added()), ({}, set(), 0))
|
||||
shutil.rmtree(self.analyst)
|
||||
got = build.load_analyst_files(self.root, tax)
|
||||
self.assertEqual((got.add, got.remove), ({}, set()))
|
||||
|
||||
def test_check_counts_the_names(self):
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO) as err:
|
||||
self.assertEqual(build.main(["--root", str(self.root), "--check"]), 0)
|
||||
self.assertIn("3 added and 2 removed by warda-analyst", err.getvalue())
|
||||
|
||||
def test_format_errors(self):
|
||||
(self.analyst / "add" / "ads.txt").write_text(
|
||||
"# header\n"
|
||||
"b.example.com\n"
|
||||
"a.example.com\n" # line 3: not sorted
|
||||
"a.example.com\n" # line 4: twice
|
||||
"Upper.example.com\n" # line 5: not normalised
|
||||
"dot.example.com.\n" # line 6
|
||||
"*.wild.example.com\n" # line 7
|
||||
"c.example.com # why\n" # line 8: a comment after the name
|
||||
"\n" # line 9: an empty line
|
||||
"two.example.com names.example.com\n"
|
||||
" d.example.com\n" # line 11
|
||||
"bücher.example\n" # line 12
|
||||
"com\n") # line 13
|
||||
(self.analyst / "remove.txt").write_bytes(b"crlf.example.com\r\nzz.example.com\n")
|
||||
log = self.check_fails()
|
||||
where = str(self.analyst / "add" / "ads.txt")
|
||||
self.assertIn(f"{where}:3: a.example.com comes after b.example.com: the names must be sorted", log)
|
||||
self.assertIn(f"{where}:4: a.example.com is written twice", log)
|
||||
for n, line in ((5, "Upper.example.com"), (6, "dot.example.com."), (7, "*.wild.example.com"),
|
||||
(8, "c.example.com # why"), (9, ""), (10, "two.example.com names.example.com"),
|
||||
(11, " d.example.com"), (12, "bücher.example"), (13, "com")):
|
||||
self.assertIn(f"{where}:{n}: {line!r} is not one name in its normalised form", log)
|
||||
self.assertNotIn(f"{where}:1:", log)
|
||||
self.assertNotIn(f"{where}:2:", log)
|
||||
self.assertIn(f"{self.analyst / 'remove.txt'}:1: 'crlf.example.com\\r' is not one name", log)
|
||||
self.assertNotIn("remove.txt:2:", log)
|
||||
# The offline build stops on the same errors.
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO):
|
||||
self.assertEqual(self.run_build(), 1)
|
||||
self.assertFalse(self.out.exists())
|
||||
|
||||
def test_byte_order_mark(self):
|
||||
# Nothing is repaired: the mark is not part of a comment nor of a name.
|
||||
(self.analyst / "remove.txt").write_bytes(b"\xef\xbb\xbf# header\nexample.net\n")
|
||||
log = self.check_fails()
|
||||
self.assertIn(f"{self.analyst / 'remove.txt'}:1: '\\ufeff# header' is not one name in its normalised form",
|
||||
log)
|
||||
(self.analyst / "remove.txt").write_bytes(b"\xef\xbb\xbfexample.net\n")
|
||||
self.assertIn("remove.txt:1: '\\ufeffexample.net' is not one name", self.check_fails())
|
||||
|
||||
def test_many_errors_are_cut(self):
|
||||
(self.analyst / "remove.txt").write_text("".join(f"Bad{i:02d}.example.com\n" for i in range(30)))
|
||||
log = self.check_fails()
|
||||
self.assertEqual(log.count("is not one name in its normalised form"), 20)
|
||||
self.assertIn(f"{self.analyst / 'remove.txt'}: 10 more errors in this file", log)
|
||||
|
||||
def test_not_utf8(self):
|
||||
(self.analyst / "remove.txt").write_bytes(b"caf\xe9.example.com\n")
|
||||
self.assertIn("remove.txt: not UTF-8 text", self.check_fails())
|
||||
|
||||
def test_category_errors(self):
|
||||
for name in ("base", "csam", "nope"):
|
||||
(self.analyst / "add" / f"{name}.txt").write_text("a.example.com\n")
|
||||
log = self.check_fails()
|
||||
self.assertIn("base.txt: 'base' is not a category of taxonomy.toml (a bundle: name one of its categories)",
|
||||
log)
|
||||
self.assertIn("csam.txt: 'csam' cannot be filled in this public repository", log)
|
||||
self.assertIn("nope.txt: 'nope' is not a category of taxonomy.toml", log)
|
||||
|
||||
def test_nothing_else_in_the_directory(self):
|
||||
# A decision of the team in a file the build would not read must
|
||||
# not pass: every unexpected entry is an error, with its path.
|
||||
only = "not expected here: only README.md, remove.txt (a file) and add/<category>.txt"
|
||||
in_add = "not expected here: only files named <category>.txt"
|
||||
(self.analyst / "removed.txt").write_text("example.net\n")
|
||||
(self.analyst / "Remove.txt").write_text("example.net\n")
|
||||
(self.analyst / "notes").mkdir()
|
||||
(self.analyst / "add" / "ads.TXT").write_text("a.example.com\n")
|
||||
(self.analyst / "add" / "ads").write_text("a.example.com\n")
|
||||
(self.analyst / "add" / "types").mkdir()
|
||||
(self.analyst / "add" / "tracking.txt").mkdir()
|
||||
(self.analyst / "add" / "link.txt").symlink_to(self.analyst / "add" / "phishing.txt")
|
||||
log = self.check_fails()
|
||||
for name in ("removed.txt", "Remove.txt", "notes"):
|
||||
self.assertIn(f"{self.analyst / name}: {only}", log)
|
||||
for name in ("ads.TXT", "ads", "types", "tracking.txt", "link.txt"):
|
||||
self.assertIn(f"{self.analyst / 'add' / name}: {in_add}", log)
|
||||
self.assertNotIn("Traceback", log)
|
||||
|
||||
def test_wrong_kind_of_entry(self):
|
||||
only = "not expected here: only README.md, remove.txt (a file) and add/<category>.txt"
|
||||
# remove.txt as a directory, add as a file, README.md as a directory.
|
||||
(self.analyst / "remove.txt").unlink()
|
||||
(self.analyst / "remove.txt").mkdir()
|
||||
shutil.rmtree(self.analyst / "add")
|
||||
(self.analyst / "add").write_text("a.example.com\n")
|
||||
(self.analyst / "README.md").mkdir()
|
||||
log = self.check_fails()
|
||||
for name in ("remove.txt", "add", "README.md"):
|
||||
self.assertIn(f"{self.analyst / name}: {only}", log)
|
||||
# analyst itself as a file.
|
||||
shutil.rmtree(self.analyst)
|
||||
self.analyst.write_text("example.net\n")
|
||||
self.assertIn(f"error: {self.analyst}: must be a directory", self.check_fails())
|
||||
|
||||
def test_added_and_removed_at_once(self):
|
||||
# In both directions: the same name, an added name under a removed
|
||||
# one (it would be removed), a removed name under an added one (it
|
||||
# would stay blocked).
|
||||
(self.analyst / "add" / "ads.txt").write_text(
|
||||
"fine.example.com\nother.example.org\nsame.example.com\nsub.gone.example.com\n")
|
||||
(self.analyst / "remove.txt").write_text(
|
||||
"deep.sub.fine.example.com\ngone.example.com\nsame.example.com\nunrelated.example.org\n")
|
||||
log = self.check_fails()
|
||||
where, removed = self.analyst / "add" / "ads.txt", self.analyst / "remove.txt"
|
||||
self.assertIn(f"{where}: same.example.com is also removed by {removed}", log)
|
||||
self.assertIn(f"{where}: sub.gone.example.com is under gone.example.com, which {removed} removes", log)
|
||||
self.assertIn(f"{where}: fine.example.com still blocks deep.sub.fine.example.com, which {removed} removes",
|
||||
log)
|
||||
self.assertEqual(log.count(str(where)), 3)
|
||||
self.assertNotIn("other.example.org", log)
|
||||
self.assertNotIn("unrelated.example.org", log)
|
||||
|
||||
def test_budgets_of_the_analyst(self):
|
||||
(self.analyst / "add" / "ads.txt").write_text("one.example.com\n")
|
||||
self.set_taxonomy("max_add = 50000", "max_add = 3") # 3 in phishing, 1 in ads
|
||||
log = self.check_fails()
|
||||
self.assertIn(f"{self.analyst / 'add'}: 4 names, more than the budget of 3 (max_add of [analyst]", log)
|
||||
self.assertNotIn("max_remove", log)
|
||||
self.set_taxonomy("max_add = 3", "max_add = 4")
|
||||
self.set_taxonomy("max_remove = 20000", "max_remove = 1")
|
||||
log = self.check_fails()
|
||||
self.assertIn(f"{self.analyst / 'remove.txt'}: 2 names, more than the budget of 1 (max_remove of [analyst]",
|
||||
log)
|
||||
self.assertNotIn("max_add", log)
|
||||
# 0 refuses every name.
|
||||
self.set_taxonomy("max_remove = 1", "max_remove = 0")
|
||||
self.assertIn("2 names, more than the budget of 0", self.check_fails())
|
||||
self.set_taxonomy("max_remove = 0", "max_remove = 2")
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO):
|
||||
self.assertEqual(build.main(["--root", str(self.root), "--check"]), 0)
|
||||
|
||||
def assert_ignored(self, manifest: dict, log: str, why: str, without: dict[str, list[str]]) -> None:
|
||||
"""The build went on without the files of the analyst, and says so."""
|
||||
self.assertEqual(manifest["analyst"], NO_ANALYST | {"status": f"ignored: {why}"})
|
||||
self.assertIn(f"WARNING the files of analyst/ are IGNORED, the lists are built without them: {why}", log)
|
||||
self.assertEqual(self.lists(), without)
|
||||
for rel, info in manifest["files"].items():
|
||||
self.assertNotIn("warda-analyst", info["sources"], rel)
|
||||
self.assertEqual(info["count"], len(lines_of(self.out / rel)), rel)
|
||||
self.assertNotIn("phishing.txt", manifest["protected_removed"])
|
||||
self.assertNotIn("warda-analyst", (self.out / "LICENSES.md").read_text().split("## Licence of each file")[1])
|
||||
|
||||
def test_budget_reached_with_the_analyst(self):
|
||||
# phishing.txt: 2 entries from fx-hosts, a third from the analyst,
|
||||
# one too many. The files passed the check: the build must go on,
|
||||
# without them, and say it.
|
||||
without = self.lists_without_analyst()
|
||||
self.set_taxonomy('id = "phishing"\n', 'id = "phishing"\nmax_domains = 2\n')
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO):
|
||||
self.assertEqual(build.main(["--root", str(self.root), "--check"]), 0)
|
||||
manifest, log = self.built()
|
||||
self.assert_ignored(manifest, log, "phishing.txt: 3 domains, more than its budget of 2", without)
|
||||
self.assertEqual(manifest["files"]["phishing.txt"]["count"], 2)
|
||||
self.assertIn("tracker.example.net", lines_of(self.out / "ads.txt")) # the removals are ignored too
|
||||
self.assertIn("phishing.txt: 2 domains for 3 names", log)
|
||||
|
||||
def test_budget_reached_without_the_analyst(self):
|
||||
# gambling.txt is over its budget whatever the analyst does: this
|
||||
# fails as it always did, on the figure of the sources.
|
||||
self.set_taxonomy("default_max_domains = 1500000", "default_max_domains = 3")
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO) as err:
|
||||
self.assertEqual(self.run_build("--force"), 1)
|
||||
self.assertIn("gambling.txt: 5 domains, more than its budget of 3 (max_domains in taxonomy.toml",
|
||||
err.getvalue())
|
||||
self.assertFalse(self.out.exists())
|
||||
|
||||
def test_guard_tripped_by_the_removals(self):
|
||||
# 106 entries, 120 before: fine. The analyst removes 60 of them, one
|
||||
# line each: more than half of the list would be lost.
|
||||
sites = [f"n{i:03d}.example.org" for i in range(100)]
|
||||
(self.root / "extra" / "tracking.txt").write_text("\n".join(sites) + "\n")
|
||||
previous = self.tmp / "previous.json"
|
||||
previous.write_text(json.dumps({"files": {"tracking.txt": {"kind": "category", "count": 120}}}))
|
||||
without = self.lists_without_analyst()
|
||||
(self.analyst / "remove.txt").write_text("\n".join(sites[:60]) + "\n")
|
||||
manifest, log = self.built("--previous", str(previous))
|
||||
self.assert_ignored(manifest, log, "tracking.txt: 46 domains, was 120 (lost more than 50%)", without)
|
||||
self.assertEqual(manifest["files"]["tracking.txt"]["count"], 106)
|
||||
# 46 names removed: 60 are left, half of 120. Applied.
|
||||
(self.analyst / "remove.txt").write_text("\n".join(sites[:46]) + "\n")
|
||||
manifest, log = self.built("--previous", str(previous))
|
||||
self.assertEqual((manifest["analyst"]["status"], manifest["analyst"]["remove"]), ("applied", 46))
|
||||
self.assertEqual(manifest["files"]["tracking.txt"]["count"], 60)
|
||||
|
||||
def test_base_too_small_with_the_removals(self):
|
||||
# ads.example.com goes with its subdomain, a name that its parent
|
||||
# covered: the bundle comes back to its 6 entries and 7 names.
|
||||
without = self.lists_without_analyst()
|
||||
(self.analyst / "remove.txt").write_text("ads.example.com\n")
|
||||
manifest, log = self.built("--min-base", "6")
|
||||
self.assertEqual(manifest["files"]["base.txt"]["names"], 7)
|
||||
self.assertEqual(manifest["files"]["base.txt"]["count"], 6)
|
||||
self.assert_ignored(manifest, log, "base.txt: 5 domains, fewer than the minimum 6", without)
|
||||
|
||||
def test_guard_is_not_hidden_by_the_analyst(self):
|
||||
# The sources of phishing collapse from 120 entries to 2. The 60
|
||||
# names of the analyst would bring the list over half of 120: the
|
||||
# guard looks at the sources only, and stops the build.
|
||||
previous = self.tmp / "previous.json"
|
||||
previous.write_text(json.dumps({"files": {"phishing.txt": {"kind": "category", "count": 120}}}))
|
||||
(self.analyst / "add" / "phishing.txt").write_text("".join(f"n{i:02d}.example.org\n" for i in range(60)))
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO) as err:
|
||||
self.assertEqual(self.run_build("--previous", str(previous)), 1)
|
||||
self.assertIn("GUARD: phishing.txt: 2 domains, was 120 (lost more than 50%)", err.getvalue())
|
||||
self.assertNotIn("IGNORED", err.getvalue())
|
||||
self.assertFalse(self.out.exists())
|
||||
# The owner forces the publication: with the files of the analyst.
|
||||
manifest, log = self.built("--previous", str(previous), "--force")
|
||||
self.assertEqual(manifest["files"]["phishing.txt"]["count"], 62)
|
||||
self.assertEqual(manifest["analyst"]["status"], "applied")
|
||||
# And base.txt too small is not hidden by added names either.
|
||||
(self.analyst / "add" / "ads.txt").write_text("".join(f"n{i:02d}.example.org\n" for i in range(60)))
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO) as err:
|
||||
self.assertEqual(self.run_build("--min-base", "50"), 1)
|
||||
self.assertIn("GUARD: base.txt: 6 domains, fewer than the minimum 50", err.getvalue())
|
||||
|
||||
def test_forced_guard_does_not_cover_a_budget(self):
|
||||
# The owner forces the publication of a list that collapsed: that
|
||||
# does not let the names of the analyst bring it over its budget.
|
||||
without = self.lists_without_analyst()
|
||||
previous = self.tmp / "previous.json"
|
||||
previous.write_text(json.dumps({"files": {"phishing.txt": {"kind": "category", "count": 120}}}))
|
||||
self.set_taxonomy('id = "phishing"\n', 'id = "phishing"\nmax_domains = 2\n')
|
||||
manifest, log = self.built("--previous", str(previous), "--force")
|
||||
self.assert_ignored(manifest, log, "phishing.txt: 3 domains, more than its budget of 2", without)
|
||||
self.assertIn("GUARD ignored (--force)", log)
|
||||
|
||||
def test_list_too_big_with_the_analyst(self):
|
||||
without = self.lists_without_analyst()
|
||||
biggest = max(f.stat().st_size for f in self.out.rglob("*.txt") if f.parent.name != "licenses")
|
||||
limit = str((biggest + 200.5) / 1048576)
|
||||
(self.analyst / "add" / "drugs.txt").write_text(
|
||||
"".join(f"n{i:02d}.a-rather-long-name-of-a-site.example.org\n" for i in range(60)))
|
||||
manifest, log = self.built("--max-file-mb", limit)
|
||||
self.assertRegex(manifest["analyst"]["status"],
|
||||
r"^ignored: drugs\.txt: \d+ bytes \(0\.0 MiB\), more than the limit of 0 MiB$")
|
||||
self.assert_ignored(manifest, log, manifest["analyst"]["status"].removeprefix("ignored: "), without)
|
||||
self.assertEqual(lines_of(self.out / "drugs.txt"), [])
|
||||
# Too big without the analyst as well: this fails as before.
|
||||
with mock.patch("sys.stderr", new_callable=io.StringIO) as err:
|
||||
self.assertEqual(self.run_build("--max-file-mb", "0.0005"), 1)
|
||||
self.assertIn("MiB (Warda refuses a list over 128 MiB): split the category", err.getvalue())
|
||||
|
||||
|
||||
class Handler(http.server.BaseHTTPRequestHandler):
|
||||
agents: list[str] = []
|
||||
flaky = 0
|
||||
|
||||
+2
-1
@@ -52,7 +52,8 @@
|
||||
# GPL-2.0-or-later, CC-BY-SA-4.0, CC-BY-SA-3.0, CC-BY-4.0,
|
||||
# CC-BY-3.0, CC0-1.0, Unlicense, MIT, ISC, 0BSD,
|
||||
# BSD-2-Clause, BSD-3-Clause, Apache-2.0 (their texts are in
|
||||
# licenses/). Anything else is refused: unknown or
|
||||
# licenses/; the list itself is "licences" in
|
||||
# taxonomy.toml). Anything else is refused: unknown or
|
||||
# proprietary licences, NC (Warda has a paid offer), ND.
|
||||
# Check the LICENSE file of the project: "GPL-3.0" alone is
|
||||
# ambiguous, write GPL-3.0-only unless it says "or later".
|
||||
|
||||
@@ -21,6 +21,10 @@
|
||||
#
|
||||
# Only ids made of a-z, 0-9 and "-" are allowed. Changing an id changes the
|
||||
# address of its file: the Warda boxes that download it must follow.
|
||||
#
|
||||
# This file also holds the licences a source may carry (licences, below),
|
||||
# the budget of the files written by warda-analyst ([analyst]) and the
|
||||
# groups that file the sources ([[source_group]], at the end).
|
||||
|
||||
# Budget of a category or bundle file: the lists live in the memory of the
|
||||
# Warda boxes, a Raspberry Pi among them. A file over its budget fails the
|
||||
@@ -29,6 +33,34 @@
|
||||
# (never loaded to block).
|
||||
default_max_domains = 1500000
|
||||
|
||||
# The licences accepted for a source (SPDX ids): the only place where they
|
||||
# are written. scripts/build.py reads them here and publishes them in
|
||||
# manifest.json (taxonomy.licences), where warda-analyst reads them. Each
|
||||
# one has its text in licenses/<id>.txt, copied to dist/licenses/. Anything
|
||||
# else is refused: unknown or proprietary licences, and those that forbid
|
||||
# what the build does (NC: Warda has a paid offer; ND: the lists are
|
||||
# modified), which the build refuses here too. A new licence must also be
|
||||
# known to combine_licences of scripts/build.py when it cannot be mixed
|
||||
# with the GPL.
|
||||
licences = [
|
||||
"GPL-3.0-only",
|
||||
"GPL-3.0-or-later",
|
||||
"GPL-2.0-only",
|
||||
"GPL-2.0-or-later",
|
||||
"CC-BY-SA-4.0",
|
||||
"CC-BY-SA-3.0",
|
||||
"CC-BY-4.0",
|
||||
"CC-BY-3.0",
|
||||
"CC0-1.0",
|
||||
"Unlicense",
|
||||
"MIT",
|
||||
"ISC",
|
||||
"0BSD",
|
||||
"BSD-2-Clause",
|
||||
"BSD-3-Clause",
|
||||
"Apache-2.0",
|
||||
]
|
||||
|
||||
# --- Why a category is blocked -------------------------------------------
|
||||
|
||||
[[reason]]
|
||||
@@ -231,6 +263,36 @@ description_fr = "Publicité et pistage : la liste que chaque boîtier Warda blo
|
||||
warda = "base"
|
||||
includes = ["ads", "tracking"]
|
||||
|
||||
# --- Files written by warda-analyst -----------------------------------------
|
||||
# analyst/add/<category>.txt and analyst/remove.txt are written by the
|
||||
# service warda-analyst after a decision of the team, never by hand. Their
|
||||
# budgets, in names: over one of them, --check fails on develop and the
|
||||
# commit is not promoted to main.
|
||||
# max_add the names of all the analyst/add/ files together
|
||||
# max_remove the names of analyst/remove.txt
|
||||
# And a limit applied by the build, which alone knows what the sources hold:
|
||||
# max_effect the most names of the lists one line of analyst/remove.txt
|
||||
# may take out, itself and its subdomains. A line over it is
|
||||
# not applied (the build says it, manifest.json lists it): it
|
||||
# is a suffix that many sites share (co.uk, github.io), not a
|
||||
# name wrongly blocked.
|
||||
# And where the removals apply:
|
||||
# remove_from the categories analyst/remove.txt takes its names out of
|
||||
# (with their subdomains): the protection lists only. The
|
||||
# team clears a name because it is harmless (no malware, no
|
||||
# tracker); that says nothing of what the site is about, so
|
||||
# the name stays in the content categories (adult, gambling…)
|
||||
# and in the site types. The bundles built from these
|
||||
# categories (base) follow, and so does what "includes"
|
||||
# brings into them (security includes phishing). allow.txt
|
||||
# keeps its own meaning: every list.
|
||||
|
||||
[analyst]
|
||||
max_add = 50000
|
||||
max_remove = 20000
|
||||
max_effect = 100
|
||||
remove_from = ["ads", "tracking", "phishing", "security"]
|
||||
|
||||
# --- Site types (classification only, never blocked by themselves) --------
|
||||
|
||||
[[group]]
|
||||
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
# Written by warda-analyst (fixture of the tests). Never edit by hand.
|
||||
confirmed.example.org
|
||||
login.confirmed.example.org
|
||||
www.google.com
|
||||
Vendored
+3
@@ -0,0 +1,3 @@
|
||||
# Written by warda-analyst (fixture of the tests). Never edit by hand.
|
||||
casino.example.com
|
||||
example.net
|
||||
+1
@@ -17,3 +17,4 @@ google.com
|
||||
maps.google.com
|
||||
allowed.example.com
|
||||
deep.allowed.example.com
|
||||
phish2.example.com
|
||||
Reference in new issue
Block a user