Merge pull request #23 from hlqviet/refactor/domain-process

Domain process code refactor
This commit is contained in:
mrrfv
2023-09-17 17:23:28 +02:00
committed by GitHub
8 changed files with 266 additions and 139 deletions
+9
View File
@@ -0,0 +1,9 @@
root = true
[*]
indent_style = space
indent_size = 2
end_of_line = lf
charset = utf-8
trim_trailing_whitespace = true
insert_final_newline = true
+3
View File
@@ -8,6 +8,9 @@ on:
- main - main
workflow_dispatch: workflow_dispatch:
env:
NODE_ENV: production
jobs: jobs:
cgps: cgps:
runs-on: ubuntu-latest runs-on: ubuntu-latest
+8 -18
View File
@@ -1,21 +1,11 @@
import { createZeroTrustRule, getZeroTrustLists } from './lib/api.js'; import { createZeroTrustRule, getZeroTrustLists } from "./lib/api.js";
;(async() => { const { result: lists } = await getZeroTrustLists();
const { result: lists } = await getZeroTrustLists(); const wirefilterExpression = lists.reduce((previous, current) => {
const filtered_lists = lists.filter(list => list.name.startsWith('CGPS List')); if (!current.name.startsWith("CGPS List")) return previous;
let wirefilter_expression = ''; return `${previous} any(dns.domains[*] in \$${current.id}) or `;
}, "");
// Build the wirefilter expression // Remove the trailing ' or '
for (const list of filtered_lists) { await createZeroTrustRule(wirefilterExpression.slice(0, -4));
wirefilter_expression += `any(dns.domains[*] in \$${list.id}) or `;
}
// Remove the trailing ' or '
if (wirefilter_expression.endsWith(' or ')) {
wirefilter_expression = wirefilter_expression.slice(0, -4);
}
wirefilter_expression = wirefilter_expression.trim().replace('\n', '');
if (!process.env.CI) console.log(`Firewall expression contains ${wirefilter_expression.length} characters, and checks against ${filtered_lists.length} filter lists.`)
await createZeroTrustRule(wirefilter_expression);
})();
+12 -8
View File
@@ -1,12 +1,16 @@
import { deleteZeroTrustRule, getZeroTrustRules } from './lib/api.js'; import { deleteZeroTrustRule, getZeroTrustRules } from "./lib/api.js";
;(async() => { const { result: rules } = await getZeroTrustRules();
const { result: rules } = await getZeroTrustRules(); const cgpsRule = rules.find(({ name }) => name === "CGPS Filter Lists");
const [filtered_rule] = rules.filter(rule => rule.name === "CGPS Filter Lists");
if (!filtered_rule) return console.warn("No rule with matching name found - this is not an issue if you haven't run the create script yet. Exiting."); (async () => {
if (!cgpsRule) {
console.warn(
"No rule with matching name found - this is not an issue if you haven't run the create script yet. Exiting."
);
return;
}
console.log(`Deleting rule`, process.env.CI ? "(redacted, running in CI)" : `"${filtered_rule.name}" with ID ${filtered_rule.id}`); console.log(`Deleting rule ${cgpsRule.name}`);
await deleteZeroTrustRule(cgpsRule.id);
await deleteZeroTrustRule(filtered_rule.id);
})(); })();
+118 -96
View File
@@ -1,115 +1,137 @@
import fs from 'fs'; import { resolve } from "path";
import { DRY_RUN, FAST_MODE, LIST_ITEM_LIMIT } from './lib/constants.js';
import { createZeroTrustListsAtOnce, createZeroTrustListsOneByOne } from './lib/api.js';
import { truncateArray } from './lib/utils.js';
if (!process.env.CI) console.log(`List item limit set to ${LIST_ITEM_LIMIT}`); import {
createZeroTrustListsAtOnce,
createZeroTrustListsOneByOne,
} from "./lib/api.js";
import {
DRY_RUN,
FAST_MODE,
LIST_ITEM_LIMIT,
LIST_ITEM_SIZE,
} from "./lib/constants.js";
import { normalizeDomain } from "./lib/helpers.js";
import {
extractDomain,
isComment,
isValidDomain,
readFile,
} from "./lib/utils.js";
let whitelist = []; // Define an empty array for the whitelist const allowlistFilename = "whitelist.csv";
const blocklistFilename = "input.csv";
const allowlist = new Map();
const blocklist = new Map();
const domains = [];
let processedDomainCount = 0;
let unnecessaryDomainCount = 0;
let duplicateDomainCount = 0;
let allowedDomainCount = 0;
// Read whitelist.csv and parse // Read allowlist
fs.readFile('whitelist.csv', 'utf8', async (err, data) => { console.log(`Processing ${allowlistFilename}`);
if (err) { await readFile(resolve(allowlistFilename), (line) => {
console.warn('Error reading whitelist.csv:', err); const _line = line.trim();
console.warn('Assuming whitelist is empty.')
} else { if (!_line) return;
// Convert into array and cleanup whitelist
const domainValidationPattern = /^(?!-)[A-Za-z0-9-]+([\-\.]{1}[a-z0-9]+)*\.[A-Za-z]{2,6}$/; if (isComment(_line)) return;
whitelist = data.split('\n').filter(domain => {
// Remove entire lines starting with "127.0.0.1" or "::1", empty lines or comments const domain = normalizeDomain(_line, true);
return domain && !domain.startsWith('#') && !domain.startsWith('//') && !domain.startsWith('/*') && !domain.startsWith('*/') && !(domain === '\r');
}).map(domain => { if (!isValidDomain(domain)) return;
// Remove "\r", "0.0.0.0 ", "127.0.0.1 ", "::1 " and similar from domain items
return domain allowlist.set(domain, 1);
.replace('\r', '')
.replace('0.0.0.0 ', '')
.replace('127.0.0.1 ', '')
.replace('::1 ', '')
.replace(':: ', '')
.replace('||', '')
.replace('@@||', '')
.replace('^$important', '')
.replace('*.', '')
.replace('^', '');
}).filter(domain => {
return domainValidationPattern.test(domain);
});
console.log(`Found ${whitelist.length} valid domains in whitelist.`);
}
}); });
// Read blocklist
// Read input.csv and parse domains console.log(`Processing ${blocklistFilename}`);
fs.readFile('input.csv', 'utf8', async (err, data) => { await readFile(resolve(blocklistFilename), (line, rl) => {
if (err) { if (domains.length === LIST_ITEM_LIMIT) {
console.error('Error reading input.csv:', err);
return; return;
} }
// Convert into array and cleanup input const _line = line.trim();
const domainValidationPattern = /^(?!-)[A-Za-z0-9-]+([\-\.]{1}[a-z0-9]+)*\.[A-Za-z]{2,6}$/;
let domains = data.split('\n').filter(domain => {
// Remove entire lines starting with "127.0.0.1" or "::1", empty lines or comments
return domain && !domain.startsWith('#') && !domain.startsWith('//') && !domain.startsWith('/*') && !domain.startsWith('*/') && !(domain === '\r');
}).map(domain => {
// Remove "\r", "0.0.0.0 ", "127.0.0.1 ", "::1 " and similar from domain items
return domain
.replace('\r', '')
.replace('0.0.0.0 ', '')
.replace('127.0.0.1 ', '')
.replace('::1 ', '')
.replace(':: ', '')
.replace('^', '')
.replace('||', '')
.replace('@@||', '')
.replace('^$important', '')
.replace('*.', '')
.replace('^', '');
}).filter(domain => {
return domainValidationPattern.test(domain);
});
// Check for duplicates in domains array if (!_line) return;
let duplicateDomainCount = 0;
let uniqueDomains = [];
let seen = new Set(); // Use a set to store seen values
for (let domain of domains) {
if (!seen.has(domain)) { // If the domain is not in the set
seen.add(domain); // Add it to the set
uniqueDomains.push(domain); // Push the domain to the uniqueDomains array
} else { // If the domain is in the set
duplicateDomainCount++; // Increment the duplicateDomainCount
}
}
if (duplicateDomainCount > 0) console.warn(`Found ${duplicateDomainCount} duplicate domains in input.csv - removing`);
// Replace domains array with uniqueDomains array // Check if the current line is a comment in any format
domains = uniqueDomains; if (isComment(_line)) return;
// Remove prefixes and suffixes in hosts, wildcard or adblock format
const domain = normalizeDomain(_line);
// Check if it is a valid domain which is not a URL or does not contain
// characters like * in the middle of the domain
if (!isValidDomain(domain)) return;
processedDomainCount++;
// Get all the levels of the domain and check from the highest
// because we are blocking all subdomains
// Example: fourth.third.example.com => ["example.com", "third.example.com", "fourth.third.example.com"]
const anyDomainExists = extractDomain(domain)
.reverse()
.some((item) => {
if (blocklist.has(item)) {
if (item === domain) {
// The exact domain is already blocked
console.log(`Found ${item} in blocklist already - Skipping`);
duplicateDomainCount++;
} else {
// The higher-level domain is already blocked
// so it's not necessary to block this domain
console.log(
`Found ${item} in blocklist already - Skipping ${domain}`
);
unnecessaryDomainCount++;
}
return true;
}
// Remove domains from the domains array that are present in the whitelist array
let whitelistedDomainCount = 0;
domains = domains.filter(domain => {
if (whitelist.includes(domain)) {
whitelistedDomainCount++;
return false; return false;
} });
return true;
});
if (whitelistedDomainCount > 0) console.warn(`Found ${whitelistedDomainCount} domains in input.csv that are present in the whitelist - removing them`);
// Trim array to 300,000 domains if it's longer than that if (anyDomainExists) return;
if (domains.length > LIST_ITEM_LIMIT) {
console.warn(`${domains.length} domains found in input.csv - input has to be trimmed to ${LIST_ITEM_LIMIT} domains`); if (allowlist.has(domain)) {
domains = truncateArray(domains, LIST_ITEM_LIMIT); console.log(`Found ${domain} in allowlist - Skipping`);
allowedDomainCount++;
return;
} }
const listsToCreate = Math.ceil(domains.length / 1000); blocklist.set(domain, 1);
domains.push(domain);
if (!process.env.CI) console.log(`Found ${domains.length} valid domains in input.csv after cleanup - ${listsToCreate} list(s) will be created`); if (domains.length === LIST_ITEM_LIMIT) {
console.log(
"Maximum number of blocked domains reached - Stopping processing blocklist..."
);
rl.close();
}
});
// If we are dry-running, stop here because we don't want to create lists console.log("\n\n");
// TODO: we should probably continue, just without making any real requests to Cloudflare console.log(`Number of processed domains: ${processedDomainCount}`);
if (DRY_RUN) return console.log('Dry run complete - no lists were created. If this was not intended, please remove the DRY_RUN environment variable and try again.'); console.log(`Number of duplicate domains: ${duplicateDomainCount}`);
console.log(`Number of unnecessary domains: ${unnecessaryDomainCount}`);
console.log(`Number of blocked domains: ${domains.length}`);
console.log(`Number of allowed domains: ${allowedDomainCount}`);
console.log(
`Number of lists which will be created: ${Math.ceil(
domains.length / LIST_ITEM_SIZE
)}`
);
console.log("\n\n");
(async () => {
if (DRY_RUN) {
console.log(
"Dry run complete - no lists were created. If this was not intended, please remove the DRY_RUN environment variable and try again."
);
return;
}
if (FAST_MODE) { if (FAST_MODE) {
await createZeroTrustListsAtOnce(domains); await createZeroTrustListsAtOnce(domains);
@@ -117,4 +139,4 @@ fs.readFile('input.csv', 'utf8', async (err, data) => {
} }
await createZeroTrustListsOneByOne(domains); await createZeroTrustListsOneByOne(domains);
}); })();
+31 -12
View File
@@ -1,18 +1,37 @@
import { deleteZeroTrustListsAtOnce, deleteZeroTrustListsOneByOne, getZeroTrustLists } from "./lib/api.js"; import {
deleteZeroTrustListsAtOnce,
deleteZeroTrustListsOneByOne,
getZeroTrustLists,
} from "./lib/api.js";
import { FAST_MODE } from "./lib/constants.js"; import { FAST_MODE } from "./lib/constants.js";
;(async() => { (async () => {
const { result: lists } = await getZeroTrustLists(); const { result: lists } = await getZeroTrustLists();
if (!lists) return console.warn("No file lists found - this is not an issue if it's your first time running this script. Exiting.");
const cgps_lists = lists.filter(list => list.name.startsWith('CGPS List'));
if (!cgps_lists.length) return console.warn("No lists with matching name found - this is not an issue if you haven't created any filter lists before. Exiting.");
if (!process.env.CI) console.log(`Got ${lists.length} lists, ${cgps_lists.length} of which are CGPS lists that will be deleted.`); if (!lists) {
console.warn(
"No file lists found - this is not an issue if it's your first time running this script. Exiting."
);
return;
}
if (FAST_MODE) { const cgpsLists = lists.filter(({ name }) => name.startsWith("CGPS List"));
await deleteZeroTrustListsAtOnce(cgps_lists);
return;
}
await deleteZeroTrustListsOneByOne(cgps_lists); if (!cgpsLists.length) {
console.warn(
"No lists with matching name found - this is not an issue if you haven't created any filter lists before. Exiting."
);
return;
}
console.log(
`Got ${lists.length} lists, ${cgpsLists.length} of which are CGPS lists that will be deleted.`
);
if (FAST_MODE) {
await deleteZeroTrustListsAtOnce(cgpsLists);
return;
}
await deleteZeroTrustListsOneByOne(cgpsLists);
})(); })();
+19
View File
@@ -44,3 +44,22 @@ const request = async (url, options) => {
*/ */
export const requestGateway = (path, options) => export const requestGateway = (path, options) =>
request(`${API_HOST}/accounts/${ACCOUNT_ID}/gateway${path}`, options); request(`${API_HOST}/accounts/${ACCOUNT_ID}/gateway${path}`, options);
/**
* Normalizes a domain.
* @param {string} value The value to be normalized.
* @param {boolean} isAllowlisting Whether the value is to be whitelisted.
* @returns {string}
*/
export const normalizeDomain = (value, isAllowlisting) => {
const normalized = value
.replace(/(0\.0\.0\.0|127\.0\.0\.1|::1|::)\s+/, "")
.replace("||", "")
.replace("^$important", "")
.replace("*.", "")
.replace("^", "");
if (isAllowlisting) return normalized.replace("@@||", "");
return normalized;
};
+66 -5
View File
@@ -1,3 +1,8 @@
import { once } from "events";
import { createReadStream } from "fs";
import { basename } from "path";
import { createInterface } from "readline";
/** /**
* Sleeps for a specified amount of time. * Sleeps for a specified amount of time.
* @param {number} [ms=350] The amount of time in ms. * @param {number} [ms=350] The amount of time in ms.
@@ -6,9 +11,65 @@ export const sleep = (ms = 350) =>
new Promise((resolve) => setTimeout(resolve, ms)); new Promise((resolve) => setTimeout(resolve, ms));
/** /**
* Truncates an array to the specified size. * Checks if the value is a valid domain.
* @param {any[]} arr The array to be truncated. * @param {string} value The value to be checked.
* @param {number} size The size to which the array will be truncated.
* @returns {any[]}
*/ */
export const truncateArray = (arr, size) => arr.slice(0, size); export const isValidDomain = (value) =>
/^(?!-)[A-Za-z0-9-]+([\-\.]{1}[a-z0-9]+)*\.[A-Za-z]{2,6}$/.test(value);
/**
* Extracts all subdomains from a domain including itself.
* @param {string} domain The domain to be extracted.
* @returns {string[]}
*/
export const extractDomain = (domain) =>
domain.split(".").reduce((previous, current, index, array) => {
const nextIndex = index + 1;
if (nextIndex > array.length - 1) return previous;
const domain = [current, ...array.slice(nextIndex)].join(".");
previous.push(domain);
return previous;
}, []);
/**
* Checks if the value is a comment.
* @param {string} value The value to be checked.
*/
export const isComment = (value) =>
value.startsWith("#") ||
value.startsWith("//") ||
value.startsWith("!") ||
value.startsWith("/*") ||
value.startsWith("*/");
/**
* @callback onLine
* @param {string} line The current line.
* @param {ReturnType<createInterface>} rl The readline interface.
*/
/**
* Asynchronously reads a file line by line.
* @param {string} filePath The path to the file.
* @param {onLine} onLine The callback executed on each line read.
*/
export const readFile = async (filePath, onLine) => {
try {
const rl = createInterface({
input: createReadStream(filePath),
crlfDelay: Infinity,
});
rl.on("line", (line) => onLine(line, rl));
await once(rl, "close");
} catch (err) {
console.error(
`Error occurred while reading ${basename(filePath)} - ${err.toString()}`
);
}
};