diff --git a/.editorconfig b/.editorconfig new file mode 100644 index 0000000..c6c8b36 --- /dev/null +++ b/.editorconfig @@ -0,0 +1,9 @@ +root = true + +[*] +indent_style = space +indent_size = 2 +end_of_line = lf +charset = utf-8 +trim_trailing_whitespace = true +insert_final_newline = true diff --git a/auto_update_github_action.yml b/auto_update_github_action.yml index e36c4ce..5d8b92b 100644 --- a/auto_update_github_action.yml +++ b/auto_update_github_action.yml @@ -8,6 +8,9 @@ on: - main workflow_dispatch: +env: + NODE_ENV: production + jobs: cgps: runs-on: ubuntu-latest diff --git a/cf_gateway_rule_create.js b/cf_gateway_rule_create.js index d29ba2f..d069a00 100644 --- a/cf_gateway_rule_create.js +++ b/cf_gateway_rule_create.js @@ -1,21 +1,11 @@ -import { createZeroTrustRule, getZeroTrustLists } from './lib/api.js'; +import { createZeroTrustRule, getZeroTrustLists } from "./lib/api.js"; -;(async() => { - const { result: lists } = await getZeroTrustLists(); - const filtered_lists = lists.filter(list => list.name.startsWith('CGPS List')); +const { result: lists } = await getZeroTrustLists(); +const wirefilterExpression = lists.reduce((previous, current) => { + if (!current.name.startsWith("CGPS List")) return previous; - let wirefilter_expression = ''; + return `${previous} any(dns.domains[*] in \$${current.id}) or `; +}, ""); - // Build the wirefilter expression - for (const list of filtered_lists) { - wirefilter_expression += `any(dns.domains[*] in \$${list.id}) or `; - } - // Remove the trailing ' or ' - if (wirefilter_expression.endsWith(' or ')) { - wirefilter_expression = wirefilter_expression.slice(0, -4); - } - wirefilter_expression = wirefilter_expression.trim().replace('\n', ''); - if (!process.env.CI) console.log(`Firewall expression contains ${wirefilter_expression.length} characters, and checks against ${filtered_lists.length} filter lists.`) - - await createZeroTrustRule(wirefilter_expression); -})(); +// Remove the trailing ' or ' +await createZeroTrustRule(wirefilterExpression.slice(0, -4)); diff --git a/cf_gateway_rule_delete.js b/cf_gateway_rule_delete.js index 5c04aea..e8aedf5 100644 --- a/cf_gateway_rule_delete.js +++ b/cf_gateway_rule_delete.js @@ -1,12 +1,16 @@ -import { deleteZeroTrustRule, getZeroTrustRules } from './lib/api.js'; +import { deleteZeroTrustRule, getZeroTrustRules } from "./lib/api.js"; -;(async() => { - const { result: rules } = await getZeroTrustRules(); - const [filtered_rule] = rules.filter(rule => rule.name === "CGPS Filter Lists"); +const { result: rules } = await getZeroTrustRules(); +const cgpsRule = rules.find(({ name }) => name === "CGPS Filter Lists"); - if (!filtered_rule) return console.warn("No rule with matching name found - this is not an issue if you haven't run the create script yet. Exiting."); +(async () => { + if (!cgpsRule) { + console.warn( + "No rule with matching name found - this is not an issue if you haven't run the create script yet. Exiting." + ); + return; + } - console.log(`Deleting rule`, process.env.CI ? "(redacted, running in CI)" : `"${filtered_rule.name}" with ID ${filtered_rule.id}`); - - await deleteZeroTrustRule(filtered_rule.id); + console.log(`Deleting rule ${cgpsRule.name}`); + await deleteZeroTrustRule(cgpsRule.id); })(); diff --git a/cf_list_create.js b/cf_list_create.js index 1ea61f6..3671642 100644 --- a/cf_list_create.js +++ b/cf_list_create.js @@ -1,115 +1,137 @@ -import fs from 'fs'; -import { DRY_RUN, FAST_MODE, LIST_ITEM_LIMIT } from './lib/constants.js'; -import { createZeroTrustListsAtOnce, createZeroTrustListsOneByOne } from './lib/api.js'; -import { truncateArray } from './lib/utils.js'; +import { resolve } from "path"; -if (!process.env.CI) console.log(`List item limit set to ${LIST_ITEM_LIMIT}`); +import { + createZeroTrustListsAtOnce, + createZeroTrustListsOneByOne, +} from "./lib/api.js"; +import { + DRY_RUN, + FAST_MODE, + LIST_ITEM_LIMIT, + LIST_ITEM_SIZE, +} from "./lib/constants.js"; +import { normalizeDomain } from "./lib/helpers.js"; +import { + extractDomain, + isComment, + isValidDomain, + readFile, +} from "./lib/utils.js"; -let whitelist = []; // Define an empty array for the whitelist +const allowlistFilename = "whitelist.csv"; +const blocklistFilename = "input.csv"; +const allowlist = new Map(); +const blocklist = new Map(); +const domains = []; +let processedDomainCount = 0; +let unnecessaryDomainCount = 0; +let duplicateDomainCount = 0; +let allowedDomainCount = 0; -// Read whitelist.csv and parse -fs.readFile('whitelist.csv', 'utf8', async (err, data) => { - if (err) { - console.warn('Error reading whitelist.csv:', err); - console.warn('Assuming whitelist is empty.') - } else { - // Convert into array and cleanup whitelist - const domainValidationPattern = /^(?!-)[A-Za-z0-9-]+([\-\.]{1}[a-z0-9]+)*\.[A-Za-z]{2,6}$/; - whitelist = data.split('\n').filter(domain => { - // Remove entire lines starting with "127.0.0.1" or "::1", empty lines or comments - return domain && !domain.startsWith('#') && !domain.startsWith('//') && !domain.startsWith('/*') && !domain.startsWith('*/') && !(domain === '\r'); - }).map(domain => { - // Remove "\r", "0.0.0.0 ", "127.0.0.1 ", "::1 " and similar from domain items - return domain - .replace('\r', '') - .replace('0.0.0.0 ', '') - .replace('127.0.0.1 ', '') - .replace('::1 ', '') - .replace(':: ', '') - .replace('||', '') - .replace('@@||', '') - .replace('^$important', '') - .replace('*.', '') - .replace('^', ''); - }).filter(domain => { - return domainValidationPattern.test(domain); - }); - console.log(`Found ${whitelist.length} valid domains in whitelist.`); - } +// Read allowlist +console.log(`Processing ${allowlistFilename}`); +await readFile(resolve(allowlistFilename), (line) => { + const _line = line.trim(); + + if (!_line) return; + + if (isComment(_line)) return; + + const domain = normalizeDomain(_line, true); + + if (!isValidDomain(domain)) return; + + allowlist.set(domain, 1); }); - -// Read input.csv and parse domains -fs.readFile('input.csv', 'utf8', async (err, data) => { - if (err) { - console.error('Error reading input.csv:', err); +// Read blocklist +console.log(`Processing ${blocklistFilename}`); +await readFile(resolve(blocklistFilename), (line, rl) => { + if (domains.length === LIST_ITEM_LIMIT) { return; } - // Convert into array and cleanup input - const domainValidationPattern = /^(?!-)[A-Za-z0-9-]+([\-\.]{1}[a-z0-9]+)*\.[A-Za-z]{2,6}$/; - let domains = data.split('\n').filter(domain => { - // Remove entire lines starting with "127.0.0.1" or "::1", empty lines or comments - return domain && !domain.startsWith('#') && !domain.startsWith('//') && !domain.startsWith('/*') && !domain.startsWith('*/') && !(domain === '\r'); - }).map(domain => { - // Remove "\r", "0.0.0.0 ", "127.0.0.1 ", "::1 " and similar from domain items - return domain - .replace('\r', '') - .replace('0.0.0.0 ', '') - .replace('127.0.0.1 ', '') - .replace('::1 ', '') - .replace(':: ', '') - .replace('^', '') - .replace('||', '') - .replace('@@||', '') - .replace('^$important', '') - .replace('*.', '') - .replace('^', ''); - }).filter(domain => { - return domainValidationPattern.test(domain); - }); + const _line = line.trim(); - // Check for duplicates in domains array - let duplicateDomainCount = 0; - let uniqueDomains = []; - let seen = new Set(); // Use a set to store seen values - for (let domain of domains) { - if (!seen.has(domain)) { // If the domain is not in the set - seen.add(domain); // Add it to the set - uniqueDomains.push(domain); // Push the domain to the uniqueDomains array - } else { // If the domain is in the set - duplicateDomainCount++; // Increment the duplicateDomainCount - } - } - if (duplicateDomainCount > 0) console.warn(`Found ${duplicateDomainCount} duplicate domains in input.csv - removing`); + if (!_line) return; - // Replace domains array with uniqueDomains array - domains = uniqueDomains; + // Check if the current line is a comment in any format + if (isComment(_line)) return; + + // Remove prefixes and suffixes in hosts, wildcard or adblock format + const domain = normalizeDomain(_line); + + // Check if it is a valid domain which is not a URL or does not contain + // characters like * in the middle of the domain + if (!isValidDomain(domain)) return; + + processedDomainCount++; + + // Get all the levels of the domain and check from the highest + // because we are blocking all subdomains + // Example: fourth.third.example.com => ["example.com", "third.example.com", "fourth.third.example.com"] + const anyDomainExists = extractDomain(domain) + .reverse() + .some((item) => { + if (blocklist.has(item)) { + if (item === domain) { + // The exact domain is already blocked + console.log(`Found ${item} in blocklist already - Skipping`); + duplicateDomainCount++; + } else { + // The higher-level domain is already blocked + // so it's not necessary to block this domain + console.log( + `Found ${item} in blocklist already - Skipping ${domain}` + ); + unnecessaryDomainCount++; + } + + return true; + } - // Remove domains from the domains array that are present in the whitelist array - let whitelistedDomainCount = 0; - domains = domains.filter(domain => { - if (whitelist.includes(domain)) { - whitelistedDomainCount++; return false; - } - return true; - }); - if (whitelistedDomainCount > 0) console.warn(`Found ${whitelistedDomainCount} domains in input.csv that are present in the whitelist - removing them`); + }); - // Trim array to 300,000 domains if it's longer than that - if (domains.length > LIST_ITEM_LIMIT) { - console.warn(`${domains.length} domains found in input.csv - input has to be trimmed to ${LIST_ITEM_LIMIT} domains`); - domains = truncateArray(domains, LIST_ITEM_LIMIT); + if (anyDomainExists) return; + + if (allowlist.has(domain)) { + console.log(`Found ${domain} in allowlist - Skipping`); + allowedDomainCount++; + return; } - const listsToCreate = Math.ceil(domains.length / 1000); + blocklist.set(domain, 1); + domains.push(domain); - if (!process.env.CI) console.log(`Found ${domains.length} valid domains in input.csv after cleanup - ${listsToCreate} list(s) will be created`); + if (domains.length === LIST_ITEM_LIMIT) { + console.log( + "Maximum number of blocked domains reached - Stopping processing blocklist..." + ); + rl.close(); + } +}); - // If we are dry-running, stop here because we don't want to create lists - // TODO: we should probably continue, just without making any real requests to Cloudflare - if (DRY_RUN) return console.log('Dry run complete - no lists were created. If this was not intended, please remove the DRY_RUN environment variable and try again.'); +console.log("\n\n"); +console.log(`Number of processed domains: ${processedDomainCount}`); +console.log(`Number of duplicate domains: ${duplicateDomainCount}`); +console.log(`Number of unnecessary domains: ${unnecessaryDomainCount}`); +console.log(`Number of blocked domains: ${domains.length}`); +console.log(`Number of allowed domains: ${allowedDomainCount}`); +console.log( + `Number of lists which will be created: ${Math.ceil( + domains.length / LIST_ITEM_SIZE + )}` +); +console.log("\n\n"); + +(async () => { + if (DRY_RUN) { + console.log( + "Dry run complete - no lists were created. If this was not intended, please remove the DRY_RUN environment variable and try again." + ); + return; + } if (FAST_MODE) { await createZeroTrustListsAtOnce(domains); @@ -117,4 +139,4 @@ fs.readFile('input.csv', 'utf8', async (err, data) => { } await createZeroTrustListsOneByOne(domains); -}); +})(); diff --git a/cf_list_delete.js b/cf_list_delete.js index ced83e9..e9d3cb6 100644 --- a/cf_list_delete.js +++ b/cf_list_delete.js @@ -1,18 +1,37 @@ -import { deleteZeroTrustListsAtOnce, deleteZeroTrustListsOneByOne, getZeroTrustLists } from "./lib/api.js"; +import { + deleteZeroTrustListsAtOnce, + deleteZeroTrustListsOneByOne, + getZeroTrustLists, +} from "./lib/api.js"; import { FAST_MODE } from "./lib/constants.js"; -;(async() => { - const { result: lists } = await getZeroTrustLists(); - if (!lists) return console.warn("No file lists found - this is not an issue if it's your first time running this script. Exiting."); - const cgps_lists = lists.filter(list => list.name.startsWith('CGPS List')); - if (!cgps_lists.length) return console.warn("No lists with matching name found - this is not an issue if you haven't created any filter lists before. Exiting."); +(async () => { + const { result: lists } = await getZeroTrustLists(); - if (!process.env.CI) console.log(`Got ${lists.length} lists, ${cgps_lists.length} of which are CGPS lists that will be deleted.`); + if (!lists) { + console.warn( + "No file lists found - this is not an issue if it's your first time running this script. Exiting." + ); + return; + } - if (FAST_MODE) { - await deleteZeroTrustListsAtOnce(cgps_lists); - return; - } + const cgpsLists = lists.filter(({ name }) => name.startsWith("CGPS List")); - await deleteZeroTrustListsOneByOne(cgps_lists); + if (!cgpsLists.length) { + console.warn( + "No lists with matching name found - this is not an issue if you haven't created any filter lists before. Exiting." + ); + return; + } + + console.log( + `Got ${lists.length} lists, ${cgpsLists.length} of which are CGPS lists that will be deleted.` + ); + + if (FAST_MODE) { + await deleteZeroTrustListsAtOnce(cgpsLists); + return; + } + + await deleteZeroTrustListsOneByOne(cgpsLists); })(); diff --git a/lib/helpers.js b/lib/helpers.js index 5fb9e89..93e6ed0 100644 --- a/lib/helpers.js +++ b/lib/helpers.js @@ -44,3 +44,22 @@ const request = async (url, options) => { */ export const requestGateway = (path, options) => request(`${API_HOST}/accounts/${ACCOUNT_ID}/gateway${path}`, options); + +/** + * Normalizes a domain. + * @param {string} value The value to be normalized. + * @param {boolean} isAllowlisting Whether the value is to be whitelisted. + * @returns {string} + */ +export const normalizeDomain = (value, isAllowlisting) => { + const normalized = value + .replace(/(0\.0\.0\.0|127\.0\.0\.1|::1|::)\s+/, "") + .replace("||", "") + .replace("^$important", "") + .replace("*.", "") + .replace("^", ""); + + if (isAllowlisting) return normalized.replace("@@||", ""); + + return normalized; +}; diff --git a/lib/utils.js b/lib/utils.js index 1d13e0c..b5b3b6b 100644 --- a/lib/utils.js +++ b/lib/utils.js @@ -1,3 +1,8 @@ +import { once } from "events"; +import { createReadStream } from "fs"; +import { basename } from "path"; +import { createInterface } from "readline"; + /** * Sleeps for a specified amount of time. * @param {number} [ms=350] The amount of time in ms. @@ -6,9 +11,65 @@ export const sleep = (ms = 350) => new Promise((resolve) => setTimeout(resolve, ms)); /** - * Truncates an array to the specified size. - * @param {any[]} arr The array to be truncated. - * @param {number} size The size to which the array will be truncated. - * @returns {any[]} + * Checks if the value is a valid domain. + * @param {string} value The value to be checked. */ -export const truncateArray = (arr, size) => arr.slice(0, size); +export const isValidDomain = (value) => + /^(?!-)[A-Za-z0-9-]+([\-\.]{1}[a-z0-9]+)*\.[A-Za-z]{2,6}$/.test(value); + +/** + * Extracts all subdomains from a domain including itself. + * @param {string} domain The domain to be extracted. + * @returns {string[]} + */ +export const extractDomain = (domain) => + domain.split(".").reduce((previous, current, index, array) => { + const nextIndex = index + 1; + + if (nextIndex > array.length - 1) return previous; + + const domain = [current, ...array.slice(nextIndex)].join("."); + + previous.push(domain); + + return previous; + }, []); + +/** + * Checks if the value is a comment. + * @param {string} value The value to be checked. + */ +export const isComment = (value) => + value.startsWith("#") || + value.startsWith("//") || + value.startsWith("!") || + value.startsWith("/*") || + value.startsWith("*/"); + +/** + * @callback onLine + * @param {string} line The current line. + * @param {ReturnType} rl The readline interface. + */ + +/** + * Asynchronously reads a file line by line. + * @param {string} filePath The path to the file. + * @param {onLine} onLine The callback executed on each line read. + */ +export const readFile = async (filePath, onLine) => { + try { + const rl = createInterface({ + input: createReadStream(filePath), + crlfDelay: Infinity, + }); + + rl.on("line", (line) => onLine(line, rl)); + + await once(rl, "close"); + } catch (err) { + console.error( + `Error occurred while reading ${basename(filePath)} - ${err.toString()}` + ); + } +};