refactored domain processing code to improve performance and memory usage

This commit is contained in:
Viet Huynh
2023-11-19 17:46:39 +07:00
parent 6eb55bf6a1
commit 70a2dfa89f
2 changed files with 34 additions and 44 deletions
+25 -35
View File
@@ -73,40 +73,31 @@ await readFile(resolve(`./${blocklistFilename}`), (line, rl) => {
processedDomainCount++;
// Get all the levels of the domain and check from the highest
// because we are blocking all subdomains
// Example: fourth.third.example.com => ["example.com", "third.example.com", "fourth.third.example.com"]
const anyDomainExists = extractDomain(domain)
.reverse()
.some((item) => {
if (blocklist.has(item)) {
if (item === domain) {
// The exact domain is already blocked
console.log(`Found ${item} in blocklist already - Skipping`);
duplicateDomainCount++;
} else {
// The higher-level domain is already blocked
// so it's not necessary to block this domain
console.log(
`Found ${item} in blocklist already - Skipping ${domain}`
);
unnecessaryDomainCount++;
}
return true;
}
return false;
});
if (anyDomainExists) return;
if (allowlist.has(domain)) {
console.log(`Found ${domain} in allowlist - Skipping`);
allowedDomainCount++;
return;
}
if (blocklist.has(domain)) {
console.log(`Found ${domain} in blocklist already - Skipping`);
duplicateDomainCount++;
return;
}
// Get all the levels of the domain and check from the highest
// because we are blocking all subdomains
// Example: fourth.third.example.com => ["example.com", "third.example.com", "fourth.third.example.com"]
for (const item of extractDomain(domain).slice(1)) {
if (!blocklist.has(item)) continue;
// The higher-level domain is already blocked
// so it's not necessary to block this domain
console.log(`Found ${item} in blocklist already - Skipping ${domain}`);
unnecessaryDomainCount++;
return;
}
blocklist.set(domain, 1);
domains.push(domain);
@@ -124,8 +115,8 @@ console.log("\n\n");
console.log(`Number of processed domains: ${processedDomainCount}`);
console.log(`Number of duplicate domains: ${duplicateDomainCount}`);
console.log(`Number of unnecessary domains: ${unnecessaryDomainCount}`);
console.log(`Number of blocked domains: ${domains.length}`);
console.log(`Number of allowed domains: ${allowedDomainCount}`);
console.log(`Number of blocked domains: ${domains.length}`);
console.log(`Number of lists to be created: ${numberOfLists}`);
console.log("\n\n");
@@ -143,12 +134,11 @@ console.log("\n\n");
if (FAST_MODE) {
await createZeroTrustListsAtOnce(domains);
// TODO: make this less repetitive
await notifyWebhook(`CF List Create script finished running (${domains.length} domains, ${numberOfLists} lists)`);
return;
} else {
await createZeroTrustListsOneByOne(domains);
}
await createZeroTrustListsOneByOne(domains);
await notifyWebhook(`CF List Create script finished running (${domains.length} domains, ${numberOfLists} lists)`);
await notifyWebhook(
`CF List Create script finished running (${domains.length} domains, ${numberOfLists} lists)`
);
})();
+9 -9
View File
@@ -19,18 +19,18 @@ export const isValidDomain = (value) =>
* @param {string} domain The domain to be extracted.
* @returns {string[]}
*/
export const extractDomain = (domain) =>
domain.split(".").reduce((previous, current, index, array) => {
const nextIndex = index + 1;
export const extractDomain = (domain) => {
const parts = domain.split(".");
const extractedDomains = [];
if (nextIndex > array.length - 1) return previous;
for (let i = 0; i < parts.length; i++) {
const subdomains = parts.slice(i).join(".");
const domain = [current, ...array.slice(nextIndex)].join(".");
extractedDomains.unshift(subdomains);
}
previous.push(domain);
return previous;
}, []);
return extractedDomains;
};
/**
* Checks if the value is a comment.