2019-11-11 11:19:46 +01:00
#!/usr/bin/env bash
2019-12-07 18:45:48 +01:00
function log( ) {
echo -e " \033[33m $@ \033[0m "
}
2019-12-03 08:48:12 +01:00
if [ ! -f temp/all_resolved.csv ]
then
echo "Run ./resolve_subdomains.sh first!"
exit 1
fi
# Gather all the rules for filtering
2019-12-07 18:45:48 +01:00
log "Compiling rules…"
2019-12-03 08:48:12 +01:00
cat rules_adblock/*.txt | grep -v '^!' | grep -v '^\[Adblock' | sort -u > temp/all_rules_adblock.txt
./adblock_to_domain_list.py --input temp/all_rules_adblock.txt --output rules/from_adblock.cache.list
cat rules_hosts/*.txt | grep -v '^#' | grep -v '^$' | cut -d ' ' -f2 > rules/from_hosts.cache.list
2019-12-05 19:15:24 +01:00
cat rules/*.list | grep -v '^#' | grep -v '^$' | sort -u > temp/all_rules_multi.list
cat rules/first-party.list | grep -v '^#' | grep -v '^$' | sort -u > temp/all_rules_first.list
2019-12-07 13:51:23 +01:00
cat rules_ip/*.txt | grep -v '^#' | grep -v '^$' | sort -u > temp/all_ip_rules_multi.txt
cat rules_ip/first-party.txt | grep -v '^#' | grep -v '^$' | sort -u > temp/all_ip_rules_first.txt
2019-12-02 19:03:08 +01:00
2019-12-07 18:45:48 +01:00
log "Filtering first-party tracking domains…"
2019-12-07 13:51:23 +01:00
./filter_subdomains.py --rules temp/all_rules_first.list --rules-ip temp/all_ip_rules_first.txt --input temp/all_resolved_sorted.csv --output temp/firstparty-trackers.list
2019-12-03 08:48:12 +01:00
sort -u temp/firstparty-trackers.list > dist/firstparty-trackers.txt
2019-12-03 15:35:21 +01:00
2019-12-07 18:45:48 +01:00
log "Filtering first-party curated tracking domains…"
2019-12-07 13:51:23 +01:00
./filter_subdomains.py --rules temp/all_rules_first.list --rules-ip temp/all_ip_rules_first.txt --input temp/all_resolved_sorted.csv --no-explicit --output temp/firstparty-only-trackers.list
2019-12-03 08:48:12 +01:00
sort -u temp/firstparty-only-trackers.list > dist/firstparty-only-trackers.txt
2019-11-11 11:19:46 +01:00
2019-12-07 18:45:48 +01:00
log "Filtering multi-party tracking domains…"
2019-12-07 13:51:23 +01:00
./filter_subdomains.py --rules temp/all_rules_multi.list --rules-ip temp/all_ip_rules_multi.txt --input temp/all_resolved_sorted.csv --output temp/multiparty-trackers.list
2019-12-05 19:15:24 +01:00
sort -u temp/multiparty-trackers.list > dist/multiparty-trackers.txt
2019-12-07 18:45:48 +01:00
log "Filtering multi-party curated tracking domains…"
2019-12-07 13:51:23 +01:00
./filter_subdomains.py --rules temp/all_rules_multi.list --rules-ip temp/all_ip_rules_multi.txt --input temp/all_resolved_sorted.csv --no-explicit --output temp/multiparty-only-trackers.list
2019-12-05 19:15:24 +01:00
sort -u temp/multiparty-only-trackers.list > dist/multiparty-only-trackers.txt
2019-11-11 11:19:46 +01:00
# Format the blocklist so it can be used as a hostlist
2019-11-15 08:57:31 +01:00
function generate_hosts {
basename = " $1 "
description = " $2 "
2019-12-05 20:51:53 +01:00
description2 = " $3 "
2019-11-15 08:57:31 +01:00
(
echo "# First-party trackers host list"
echo " # $description "
2019-12-05 19:15:24 +01:00
echo " # $description2 "
2019-11-15 08:57:31 +01:00
echo "#"
echo "# About first-party trackers: https://git.frogeye.fr/geoffrey/eulaurarien#whats-a-first-party-tracker"
echo "# Source code: https://git.frogeye.fr/geoffrey/eulaurarien"
echo "#"
2019-12-03 21:45:29 +01:00
echo "# In case of false positives/negatives, or any other question,"
echo "# contact me the way you like: https://geoffrey.frogeye.fr"
echo "#"
2019-11-15 08:57:31 +01:00
echo "# Latest version:"
2019-12-05 19:15:24 +01:00
echo "# - First-party trackers : https://hostfiles.frogeye.fr/firstparty-trackers-hosts.txt"
echo "# - … excluding redirected: https://hostfiles.frogeye.fr/firstparty-only-trackers-hosts.txt"
echo "# - First and third party : https://hostfiles.frogeye.fr/multiparty-trackers-hosts.txt"
echo "# - … excluding redirected: https://hostfiles.frogeye.fr/multiparty-only-trackers-hosts.txt"
2019-11-15 08:57:31 +01:00
echo "#"
echo " # Generation date: $( date -Isec) "
2019-12-03 15:35:21 +01:00
echo " # Generation software: eulaurarien $( git describe --tags) "
2019-11-15 08:57:31 +01:00
echo " # Number of source websites: $( wc -l temp/all_websites.list | cut -d' ' -f1) "
echo " # Number of source subdomains: $( wc -l temp/all_subdomains.list | cut -d' ' -f1) "
2019-12-05 19:38:26 +01:00
echo "#"
2019-12-05 19:15:24 +01:00
echo " # Number of known first-party trackers: $( wc -l temp/all_rules_first.list | cut -d' ' -f1) "
echo " # Number of first-party subdomains: $( wc -l dist/firstparty-trackers.txt | cut -d' ' -f1) "
echo " # … excluding redirected: $( wc -l dist/firstparty-only-trackers.txt | cut -d' ' -f1) "
2019-12-05 19:38:26 +01:00
echo "#"
echo " # Number of known multi-party trackers: $( wc -l temp/all_rules_multi.list | cut -d' ' -f1) "
2019-12-05 19:15:24 +01:00
echo " # Number of multi-party subdomains: $( wc -l dist/multiparty-trackers.txt | cut -d' ' -f1) "
echo " # … excluding redirected: $( wc -l dist/multiparty-only-trackers.txt | cut -d' ' -f1) "
2019-11-15 08:57:31 +01:00
echo
cat " dist/ $basename .txt " | while read host;
do
echo " 0.0.0.0 $host "
done
) > " dist/ $basename -hosts.txt "
}
2019-12-05 19:15:24 +01:00
generate_hosts "firstparty-trackers" "Generated from a curated list of first-party trackers" ""
generate_hosts "firstparty-only-trackers" "Generated from a curated list of first-party trackers" "Only contain the first chain of redirection."
generate_hosts "multiparty-trackers" "Generated from known third-party trackers." "Also contains trackers used as third-party."
generate_hosts "multiparty-only-trackers" "Generated from known third-party trackers." "Do not contain trackers used in third-party. Use in combination with third-party lists."