diff --git a/README.md b/README.md index 01ee4d5..8723c52 100644 --- a/README.md +++ b/README.md @@ -15,7 +15,10 @@ From the project root: ```bash python3 -m venv .venv -source .venv/bin/activate # Windows: .venv\Scripts\activate +# Unix: +source .venv/bin/activate +# Windows: +.venv\Scripts\activate pip install -e . ``` diff --git a/data/mins_depts_test_check.csv b/data/mins_depts_test_check.csv new file mode 100644 index 0000000..e4825ab --- /dev/null +++ b/data/mins_depts_test_check.csv @@ -0,0 +1,50 @@ +Type,Institution Name,URL,Scrape_URL +Ministry,Ministry of Defence,defence.lk, +Ministry,"Ministry of Finance, Planning and Economic Development",treasury.gov.lk, +Ministry,Ministry of Digital Economy,midec.gov.lk, +Ministry,"Ministry of Foreign Affairs, Foreign Employment and Tourism",mfa.gov.lk,https://www.mfa.gov.lk/en +Ministry,"Ministry of Education, Higher Education and Vocational Education",moe.gov.lk,https://moe.gov.lk/en/ +Ministry,"Ministry of Agriculture, Livestock, Land and Irrigation",agrimin.gov.lk, +Ministry,"Ministry of Transport, Highways and Urban Development",transport.gov.lk,https://transport.gov.lk/web/index.php?lang=en +Ministry,"Ministry of Public Administration, Provincial Councils and Local Government",pubad.gov.lk,https://pubad.gov.lk/web/index.php?lang=en +Ministry,Ministry of Health and Mass Media,health.gov.lk,https://www.health.gov.lk/home/ +Ministry,Ministry of Justice and National Integration,moj.gov.lk, +Ministry,Ministry of Public Security and Parliamentary Affairs,pubsec.gov.lk, +Ministry,Ministry of Industry and Entrepreneurship Development,industry.gov.lk,https://www.industry.gov.lk/web/ +Ministry,"Ministry of Trade, Commerce, Food Security and Cooperative Development",ment.gov.lk, +Ministry,Ministry of Environment,env.gov.lk,https://env.gov.lk/web/index.php?lang=en +Ministry,Ministry of Energy,energy.gov.lk, +Ministry,"Ministry of Housing, Construction and Water Supply",housingmin.gov.lk, +Ministry,Ministry of Labor,labourmin.gov.lk, +Ministry,Ministry of Plantation and Community Infrastructure,plantationindustries.gov.lk, +Ministry,Ministry of Ports and Civil Aviation,portaviation.gov.lk, +Ministry,"Ministry of Fisheries, Aquatic and Ocean Resources",fisheries.gov.lk, +Ministry,Ministry of Women and Child Affairs,childwomen.gov.lk, +Ministry,"Ministry of Rural Development, Social Security and Community Empowerment",socialservices.gov.lk,https://socialservices.gov.lk/web/index.php?lang=en +Ministry,"Ministry of Buddhasasana, Religious and Cultural Affairs",mbra.gov.lk, +Ministry,Ministry of Youth Affairs and Sports,sportsmin.gov.lk, +Ministry,Ministry of Science and Technology,most.gov.lk,https://most.gov.lk/web/index.php?lang=en +Department,Department of Immigration and Emigration,immigration.gov.lk,https://immigration.gov.lk/index_e.php +Department,Department of Motor Traffic,dmt.gov.lk, +Department,Department of Registration of Persons,drp.gov.lk,https://drp.gov.lk/en/home.php +Department,Department of Registrar General,rgd.gov.lk,https://rgd.gov.lk/web/index.php?lang=en +Department,Department of Inland Revenue,ird.gov.lk, +Department,Sri Lanka Customs,customs.gov.lk, +Department,Department of Pensions,pensions.gov.lk, +Department,Department of Census and Statistics,statistics.gov.lk, +Department,Department of Government Information,dgi.gov.lk, +Department,Department of Government Printing,documents.gov.lk, +Department,Department of Excise,excise.gov.lk, +Department,Department of National Budget,nationalbudget.gov.lk, +Department,Department of Meteorology,meteo.gov.lk, +Department,Department of Wildlife Conservation,dwc.gov.lk, +Department,Department of Forest Conservation,forestdept.gov.lk, +Department,Department of National Zoological Gardens,colombozoo.gov.lk, +Department,Department of Examinations,doenets.lk, +Department,Department of Archaeology,archaeology.gov.lk, +Department,Department of Samurdhi,samurdhi.gov.lk,https://samurdhi.gov.lk/web/index.php?lang=en +Department,LSF,opensource.lk, +Expired,Expired SSL,expired.badssl.com, +Self-Signed,Self-Signed SSL,self-signed.badssl.com, +Wrong Host,Wrong Host SSL,wrong.host.badssl.com, +Never SSL,Never SSL,neverssl.com, \ No newline at end of file diff --git a/data/mins_depts_test_check_scored.csv b/data/mins_depts_test_check_scored.csv new file mode 100644 index 0000000..d4e3902 --- /dev/null +++ b/data/mins_depts_test_check_scored.csv @@ -0,0 +1,50 @@ +Type,Institution Name,URL,Scrape_URL,footer_vendor,footer_vendor_error +Ministry,Ministry of Defence,defence.lk,,unknown, +Ministry,"Ministry of Finance, Planning and Economic Development",treasury.gov.lk,,found,elysiancrest.com +Ministry,Ministry of Digital Economy,midec.gov.lk,,unreachable,Failed to fetch page +Ministry,"Ministry of Foreign Affairs, Foreign Employment and Tourism",mfa.gov.lk,https://www.mfa.gov.lk/en,found,Frontwalker +Ministry,"Ministry of Education, Higher Education and Vocational Education",moe.gov.lk,https://moe.gov.lk/en/,unknown, +Ministry,"Ministry of Agriculture, Livestock, Land and Irrigation",agrimin.gov.lk,,found,onzdev.com +Ministry,"Ministry of Transport, Highways and Urban Development",transport.gov.lk,https://transport.gov.lk/web/index.php?lang=en,found,Procons Infotech +Ministry,"Ministry of Public Administration, Provincial Councils and Local Government",pubad.gov.lk,https://pubad.gov.lk/web/index.php?lang=en,found,gnu.org +Ministry,Ministry of Health and Mass Media,health.gov.lk,https://www.health.gov.lk/home/,found,Weblankan & Arrogance Technologies (Pvt)Ltd +Ministry,Ministry of Justice and National Integration,moj.gov.lk,,found,Procons Infotech +Ministry,Ministry of Public Security and Parliamentary Affairs,pubsec.gov.lk,,unknown, +Ministry,Ministry of Industry and Entrepreneurship Development,industry.gov.lk,https://www.industry.gov.lk/web/,found,theekshana.lk +Ministry,"Ministry of Trade, Commerce, Food Security and Cooperative Development",ment.gov.lk,,unreachable,Failed to fetch page +Ministry,Ministry of Environment,env.gov.lk,https://env.gov.lk/web/index.php?lang=en,found,Procons Infotech. +Ministry,Ministry of Energy,energy.gov.lk,,unknown, +Ministry,"Ministry of Housing, Construction and Water Supply",housingmin.gov.lk,,unreachable,Failed to fetch page +Ministry,Ministry of Labor,labourmin.gov.lk,,unknown, +Ministry,Ministry of Plantation and Community Infrastructure,plantationindustries.gov.lk,,found,University of Kelaniya +Ministry,Ministry of Ports and Civil Aviation,portaviation.gov.lk,,unreachable,Failed to fetch page +Ministry,"Ministry of Fisheries, Aquatic and Ocean Resources",fisheries.gov.lk,,self_built,IT UNIT Last Update +Ministry,Ministry of Women and Child Affairs,childwomen.gov.lk,,unreachable,Failed to fetch page +Ministry,"Ministry of Rural Development, Social Security and Community Empowerment",socialservices.gov.lk,https://socialservices.gov.lk/web/index.php?lang=en,found,Procons Infotech +Ministry,"Ministry of Buddhasasana, Religious and Cultural Affairs",mbra.gov.lk,,unreachable,Failed to fetch page +Ministry,Ministry of Youth Affairs and Sports,sportsmin.gov.lk,,unreachable,Failed to fetch page +Ministry,Ministry of Science and Technology,most.gov.lk,https://most.gov.lk/web/index.php?lang=en,found,Procons Infotech +Department,Department of Immigration and Emigration,immigration.gov.lk,https://immigration.gov.lk/index_e.php,found,SLT +Department,Department of Motor Traffic,dmt.gov.lk,,found,Procons Infotech +Department,Department of Registration of Persons,drp.gov.lk,https://drp.gov.lk/en/home.php,unknown, +Department,Department of Registrar General,rgd.gov.lk,https://rgd.gov.lk/web/index.php?lang=en,unknown, +Department,Department of Inland Revenue,ird.gov.lk,,unreachable,Failed to fetch page +Department,Sri Lanka Customs,customs.gov.lk,,unknown, +Department,Department of Pensions,pensions.gov.lk,,found,Procons Infotech +Department,Department of Census and Statistics,statistics.gov.lk,,unreachable,Failed to fetch page +Department,Department of Government Information,dgi.gov.lk,,unknown, +Department,Department of Government Printing,documents.gov.lk,,unknown, +Department,Department of Excise,excise.gov.lk,,found,LankaCom +Department,Department of National Budget,nationalbudget.gov.lk,,unreachable,Failed to fetch page +Department,Department of Meteorology,meteo.gov.lk,,unknown, +Department,Department of Wildlife Conservation,dwc.gov.lk,,unreachable,Failed to fetch page +Department,Department of Forest Conservation,forestdept.gov.lk,,found,ICTA & Procons Infotech +Department,Department of National Zoological Gardens,colombozoo.gov.lk,,unreachable,Failed to fetch page +Department,Department of Examinations,doenets.lk,,unknown, +Department,Department of Archaeology,archaeology.gov.lk,,found,University of Kelaniya +Department,Department of Samurdhi,samurdhi.gov.lk,https://samurdhi.gov.lk/web/index.php?lang=en,found,Information and Communication Technology Agency +Department,LSF,opensource.lk,,unknown, +Expired,Expired SSL,expired.badssl.com,,unknown, +Self-Signed,Self-Signed SSL,self-signed.badssl.com,,unknown, +Wrong Host,Wrong Host SSL,wrong.host.badssl.com,,unknown, +Never SSL,Never SSL,neverssl.com,,unknown, diff --git a/data/mins_depts_test_scored.csv b/data/mins_depts_test_scored.csv index e959f7f..0e35a6f 100644 --- a/data/mins_depts_test_scored.csv +++ b/data/mins_depts_test_scored.csv @@ -1,50 +1,50 @@ -Type,Institution Name,URL,ssl_status,ssl_error -Ministry,Ministry of Defence,defence.lk,valid, -Ministry,"Ministry of Finance, Planning and Economic Development",treasury.gov.lk,valid, -Ministry,Ministry of Digital Economy,midec.gov.lk,unreachable,"[Errno 8] nodename nor servname provided, or not known" -Ministry,"Ministry of Foreign Affairs, Foreign Employment and Tourism",mfa.gov.lk,valid, -Ministry,"Ministry of Education, Higher Education and Vocational Education",moe.gov.lk,valid, -Ministry,"Ministry of Agriculture, Livestock, Land and Irrigation",agrimin.gov.lk,valid, -Ministry,"Ministry of Transport, Highways and Urban Development",transport.gov.lk,valid, -Ministry,"Ministry of Public Administration, Provincial Councils and Local Government",pubad.gov.lk,valid, -Ministry,Ministry of Health and Mass Media,health.gov.lk,valid, -Ministry,Ministry of Justice and National Integration,moj.gov.lk,valid, -Ministry,Ministry of Public Security and Parliamentary Affairs,pubsec.gov.lk,valid, -Ministry,Ministry of Industry and Entrepreneurship Development,industry.gov.lk,valid, -Ministry,"Ministry of Trade, Commerce, Food Security and Cooperative Development",ment.gov.lk,unreachable,"[Errno 8] nodename nor servname provided, or not known" -Ministry,Ministry of Environment,env.gov.lk,valid, -Ministry,Ministry of Energy,energy.gov.lk,valid, -Ministry,"Ministry of Housing, Construction and Water Supply",housingmin.gov.lk,unreachable,"[Errno 8] nodename nor servname provided, or not known" -Ministry,Ministry of Labor,labourmin.gov.lk,valid, -Ministry,Ministry of Plantation and Community Infrastructure,plantationindustries.gov.lk,invalid,"Hostname mismatch, certificate is not valid for 'plantationindustries.gov.lk'." -Ministry,Ministry of Ports and Civil Aviation,portaviation.gov.lk,unreachable,"[Errno 8] nodename nor servname provided, or not known" -Ministry,"Ministry of Fisheries, Aquatic and Ocean Resources",fisheries.gov.lk,valid, -Ministry,Ministry of Women and Child Affairs,childwomen.gov.lk,unreachable,"[Errno 8] nodename nor servname provided, or not known" -Ministry,"Ministry of Rural Development, Social Security and Community Empowerment",socialservices.gov.lk,valid, -Ministry,"Ministry of Buddhasasana, Religious and Cultural Affairs",mbra.gov.lk,unreachable,"[Errno 8] nodename nor servname provided, or not known" -Ministry,Ministry of Youth Affairs and Sports,sportsmin.gov.lk,unreachable,"[Errno 8] nodename nor servname provided, or not known" -Ministry,Ministry of Science and Technology,most.gov.lk,valid, -Department,Department of Immigration and Emigration,immigration.gov.lk,valid, -Department,Department of Motor Traffic,dmt.gov.lk,valid, -Department,Department of Registration of Persons,drp.gov.lk,valid, -Department,Department of Registrar General,rgd.gov.lk,valid, -Department,Department of Inland Revenue,ird.gov.lk,valid, -Department,Sri Lanka Customs,customs.gov.lk,valid, -Department,Department of Pensions,pensions.gov.lk,valid, -Department,Department of Census and Statistics,statistics.gov.lk,unreachable,[Errno 54] Connection reset by peer -Department,Department of Government Information,dgi.gov.lk,valid, -Department,Department of Government Printing,documents.gov.lk,valid, -Department,Department of Excise,excise.gov.lk,valid, -Department,Department of National Budget,nationalbudget.gov.lk,unreachable,"[Errno 8] nodename nor servname provided, or not known" -Department,Department of Meteorology,meteo.gov.lk,valid, -Department,Department of Wildlife Conservation,dwc.gov.lk,unreachable,"[Errno 8] nodename nor servname provided, or not known" -Department,Department of Forest Conservation,forestdept.gov.lk,expired,unable to get local issuer certificate -Department,Department of National Zoological Gardens,colombozoo.gov.lk,unreachable,"[Errno 8] nodename nor servname provided, or not known" -Department,Department of Examinations,doenets.lk,valid, -Department,Department of Archaeology,archaeology.gov.lk,valid, -Department,Department of Samurdhi,samurdhi.gov.lk,valid, -Department,LSF,opensource.lk,invalid,"Hostname mismatch, certificate is not valid for 'opensource.lk'." -Expired,Expired SSL,expired.badssl.com,unreachable,[Errno 54] Connection reset by peer -Self-Signed,Self-Signed SSL,self-signed.badssl.com,invalid,self-signed certificate -Wrong Host,Wrong Host SSL,wrong.host.badssl.com,invalid,"Hostname mismatch, certificate is not valid for 'wrong.host.badssl.com'." -Never SSL,Never SSL,neverssl.com,valid, +Type,Institution Name,URL,ssl_status,ssl_error,footer_vendor,footer_vendor_error +Ministry,Ministry of Defence,defence.lk,valid,,unknown, +Ministry,"Ministry of Finance, Planning and Economic Development",treasury.gov.lk,valid,,found,elysiancrest.com +Ministry,Ministry of Digital Economy,midec.gov.lk,unreachable,[Errno 11001] getaddrinfo failed,unreachable,Failed to fetch page +Ministry,"Ministry of Foreign Affairs, Foreign Employment and Tourism",mfa.gov.lk,valid,,unknown, +Ministry,"Ministry of Education, Higher Education and Vocational Education",moe.gov.lk,valid,,unknown, +Ministry,"Ministry of Agriculture, Livestock, Land and Irrigation",agrimin.gov.lk,valid,,found,Ones and Zeros +Ministry,"Ministry of Transport, Highways and Urban Development",transport.gov.lk,valid,,unknown, +Ministry,"Ministry of Public Administration, Provincial Councils and Local Government",pubad.gov.lk,valid,,unknown, +Ministry,Ministry of Health and Mass Media,health.gov.lk,valid,,unknown, +Ministry,Ministry of Justice and National Integration,moj.gov.lk,valid,,found,Procons Infotech +Ministry,Ministry of Public Security and Parliamentary Affairs,pubsec.gov.lk,valid,,unknown, +Ministry,Ministry of Industry and Entrepreneurship Development,industry.gov.lk,valid,,unknown, +Ministry,"Ministry of Trade, Commerce, Food Security and Cooperative Development",ment.gov.lk,unreachable,[Errno 11001] getaddrinfo failed,unreachable,Failed to fetch page +Ministry,Ministry of Environment,env.gov.lk,valid,,unknown, +Ministry,Ministry of Energy,energy.gov.lk,valid,,unknown, +Ministry,"Ministry of Housing, Construction and Water Supply",housingmin.gov.lk,unreachable,[Errno 11001] getaddrinfo failed,unreachable,Failed to fetch page +Ministry,Ministry of Labor,labourmin.gov.lk,valid,,unknown, +Ministry,Ministry of Plantation and Community Infrastructure,plantationindustries.gov.lk,invalid,"Hostname mismatch, certificate is not valid for 'plantationindustries.gov.lk'.",found,University of Kelaniya +Ministry,Ministry of Ports and Civil Aviation,portaviation.gov.lk,unreachable,[Errno 11001] getaddrinfo failed,unreachable,Failed to fetch page +Ministry,"Ministry of Fisheries, Aquatic and Ocean Resources",fisheries.gov.lk,valid,,self_built,IT UNIT Last Update +Ministry,Ministry of Women and Child Affairs,childwomen.gov.lk,unreachable,[Errno 11001] getaddrinfo failed,unreachable,Failed to fetch page +Ministry,"Ministry of Rural Development, Social Security and Community Empowerment",socialservices.gov.lk,valid,,unknown, +Ministry,"Ministry of Buddhasasana, Religious and Cultural Affairs",mbra.gov.lk,unreachable,[Errno 11001] getaddrinfo failed,unreachable,Failed to fetch page +Ministry,Ministry of Youth Affairs and Sports,sportsmin.gov.lk,unreachable,[Errno 11001] getaddrinfo failed,unreachable,Failed to fetch page +Ministry,Ministry of Science and Technology,most.gov.lk,valid,,unknown, +Department,Department of Immigration and Emigration,immigration.gov.lk,valid,,unknown, +Department,Department of Motor Traffic,dmt.gov.lk,valid,,found,Procons Infotech & gic.gov.lk +Department,Department of Registration of Persons,drp.gov.lk,valid,,unknown, +Department,Department of Registrar General,rgd.gov.lk,valid,,unknown, +Department,Department of Inland Revenue,ird.gov.lk,valid,,unreachable,Failed to fetch page +Department,Sri Lanka Customs,customs.gov.lk,valid,,unknown, +Department,Department of Pensions,pensions.gov.lk,valid,,found,Procons Infotech +Department,Department of Census and Statistics,statistics.gov.lk,unreachable,[WinError 10054] An existing connection was forcibly closed by the remote host,unreachable,Failed to fetch page +Department,Department of Government Information,dgi.gov.lk,valid,,unknown, +Department,Department of Government Printing,documents.gov.lk,valid,,unknown, +Department,Department of Excise,excise.gov.lk,valid,,found,LankaCom +Department,Department of National Budget,nationalbudget.gov.lk,unreachable,[Errno 11001] getaddrinfo failed,unreachable,Failed to fetch page +Department,Department of Meteorology,meteo.gov.lk,valid,,unknown, +Department,Department of Wildlife Conservation,dwc.gov.lk,unreachable,[Errno 11001] getaddrinfo failed,unreachable,Failed to fetch page +Department,Department of Forest Conservation,forestdept.gov.lk,expired,certificate has expired,found,icta.lk & Procons Infotech +Department,Department of National Zoological Gardens,colombozoo.gov.lk,unreachable,[Errno 11001] getaddrinfo failed,unreachable,Failed to fetch page +Department,Department of Examinations,doenets.lk,valid,,unknown, +Department,Department of Archaeology,archaeology.gov.lk,valid,,found,University of Kelaniya +Department,Department of Samurdhi,samurdhi.gov.lk,valid,,unknown, +Department,LSF,opensource.lk,valid,,unknown, +Expired,Expired SSL,expired.badssl.com,expired,certificate has expired,unknown, +Self-Signed,Self-Signed SSL,self-signed.badssl.com,invalid,self-signed certificate,unknown, +Wrong Host,Wrong Host SSL,wrong.host.badssl.com,invalid,"Hostname mismatch, certificate is not valid for 'wrong.host.badssl.com'.",unknown, +Never SSL,Never SSL,neverssl.com,valid,,unreachable,Failed to fetch page diff --git a/src/websitescorecard/checks/__init__.py b/src/websitescorecard/checks/__init__.py index fd94d37..37dd126 100644 --- a/src/websitescorecard/checks/__init__.py +++ b/src/websitescorecard/checks/__init__.py @@ -6,11 +6,13 @@ from websitescorecard.checks.base import Check from websitescorecard.checks.ssl import SSLCheck +from websitescorecard.checks.footer_vendor import FooterVendorCheck CheckFactory = Callable[[], Check] CHECK_REGISTRY: dict[str, CheckFactory] = { "ssl": SSLCheck, + "footer_vendor": FooterVendorCheck, } diff --git a/src/websitescorecard/checks/footer_vendor.py b/src/websitescorecard/checks/footer_vendor.py new file mode 100644 index 0000000..d13edf1 --- /dev/null +++ b/src/websitescorecard/checks/footer_vendor.py @@ -0,0 +1,332 @@ +from __future__ import annotations + +import re +from urllib.parse import urlparse + +from websitescorecard.checks.base import CheckResult +from websitescorecard.url_utils import parse_url + +_MISSING_DEPS: list[str] = [] + +try: + import httpx +except ImportError: + _MISSING_DEPS.append("httpx") + +try: + from bs4 import BeautifulSoup, Tag +except ImportError: + _MISSING_DEPS.append("beautifulsoup4") + +_CREDIT_TRIGGERS = re.compile( + r"(?:" + r"designed\s*(?:&|&|and)?\s*developed\s+by" + r"|developed\s+(?:in\s+association\s+with|by)" + r"|concept[,\s]+design\s*(?:&|&|and)?\s*development\s+by" + r"|designed\s+by" + r"|developed\s+by" + r"|built\s+by" + r"|created\s+by" + r"|maintained\s+by" + r"|managed\s+by" + r"|powered\s+by" + r"|hosted\s+by" + r"|website\s+by" + r"|solution\s+by:?" + r")", + re.IGNORECASE, +) + +_TEXT_PATTERNS: list[re.Pattern[str]] = [ + re.compile( + r"(?:designed\s*(?:&|and)?\s*developed|developed|designed|built" + r"|created|maintained|managed|powered|hosted|website)" + r"\s+(?:in\s+association\s+with\s+|by\s+)" + r"([A-Za-z0-9][A-Za-z0-9 &()\-\.]{1,60})", + re.IGNORECASE, + ), + re.compile( + r"concept[,\s]+design\s*(?:&|and)?\s*development\s+by\s+" + r"([A-Za-z0-9][A-Za-z0-9 &()\-\.]{1,60})", + re.IGNORECASE, + ), + re.compile( + r"solution\s+by:?\s*([A-Za-z0-9][A-Za-z0-9 &()\-\.]{1,60})", + re.IGNORECASE, + ), +] + +_SELF_BUILT = re.compile( + r"\b(?:ict\s+(?:directorate|unit|division|branch|centre|center)" + r"|it\s+(?:division|unit|department)" + r"|mis\s+(?:unit|division)" + r"|information\s+(?:technology|systems)\s+(?:unit|division|department))\b", + re.IGNORECASE, +) + +_FOOTER_ATTR = re.compile(r"footer|bottom|copyright|credit", re.IGNORECASE) + +_DOMAIN_NAME_MAP: dict[str, str] = { + "gic.gov.lk": "ICTA (GIC)", + "icta.lk": "ICTA", + "slts.lk": "SLT", +} + +_NOISE: frozenset[str] = frozenset({ + "government of sri lanka", + "all rights reserved", + "the government", + "sri lanka", + "republic of", + "ministry", + "department", + "best viewed", + "this website", + "official website", + "last updated", + "site map", + "privacy policy", + "terms", + "contact", +}) + +_CMS_NOISE: frozenset[str] = frozenset({ + "joomla", + "wordpress", + "drupal", + "wix", + "squarespace", + "gnu general public license", + "bootstrap", + "php", +}) + +_NAV_WORDS: frozenset[str] = frozenset({ + "top", "about", "home", "back", "next", "prev", + "previous", "more", "read", "click", "here", "visit", +}) + +_BLOCKED_DOMAINS: frozenset[str] = frozenset({ + +}) + +_FOOTER_FRACTION = 0.20 +_MAX_CHARS = 6_000 + +_HEADERS = { + "User-Agent": "Mozilla/5.0 (compatible; WebsiteScorecard/1.0)", + "Accept": "text/html,application/xhtml+xml,*/*;q=0.8", + "Accept-Language": "en-US,en;q=0.9", +} + +SCRAPE_URL_COLUMN = "Scrape_URL" + + +class FooterVendorCheck: + name = "footer_vendor" + column = "footer_vendor" + error_column = "footer_vendor_error" + + def __init__(self, timeout: float = 15.0) -> None: + if _MISSING_DEPS: + raise RuntimeError( + f"FooterVendorCheck requires: {', '.join(_MISSING_DEPS)}. " + f"Install with: pip install {' '.join(_MISSING_DEPS)}" + ) + self.timeout = timeout + + def run(self, url: str, scrape_url: str | None = None) -> CheckResult: + try: + parsed = parse_url(url) + except ValueError as exc: + return CheckResult(status="unreachable", error=str(exc)) + + site_domain = parsed.hostname + + fetch_targets: list[str] = [] + if scrape_url and scrape_url.strip(): + fetch_targets.append(scrape_url.strip()) + normalized_url = url.strip() + if normalized_url not in fetch_targets: + fetch_targets.append(normalized_url) + + any_reachable = False + for target in fetch_targets: + html = self._fetch(target) + if html is None: + continue + any_reachable = True + + soup = BeautifulSoup(html, "html.parser") + for tag in soup(["script", "style", "noscript"]): + tag.decompose() + + footer_el = _find_footer_element(soup) + vendors = _extract_vendors(footer_el, soup, site_domain=site_domain) + + if vendors: + combined = " & ".join(vendors) + if _SELF_BUILT.search(combined): + return CheckResult(status="self_built", error=combined) + return CheckResult(status="found", error=combined) + + if not any_reachable: + return CheckResult(status="unreachable", error="Failed to fetch page") + + return CheckResult(status="unknown", error=None) + + def _fetch(self, url: str) -> str | None: + try: + with httpx.Client( + timeout=self.timeout, + headers=_HEADERS, + follow_redirects=True, + verify=False, # noqa: S501 + ) as client: + for candidate in _candidate_urls(url): + try: + r = client.get(candidate) + if r.status_code < 400: + return r.text + except (httpx.RequestError, httpx.HTTPStatusError): + continue + except httpx.RequestError: + pass + return None + + +def _candidate_urls(url: str) -> list[str]: + raw = url.strip() + if "://" not in raw: + return [f"https://{raw}", f"http://{raw}"] + if raw.startswith("https://"): + return [raw, raw.replace("https://", "http://", 1)] + return [raw, raw.replace("http://", "https://", 1)] + + +def _find_footer_element(soup: "BeautifulSoup") -> "Tag | None": + el = soup.find("footer") + if el: + return el + el = soup.find(lambda tag: ( + (tag.get("id") and _FOOTER_ATTR.search(tag.get("id"))) or + (tag.get("class") and _FOOTER_ATTR.search( + " ".join(tag.get("class") if isinstance(tag.get("class"), list) else [tag.get("class")]) + )) + )) + return el or None + + +def _extract_vendors( + footer_el: "Tag | None", + soup: "BeautifulSoup", + site_domain: str, +) -> list[str]: + vendors: list[str] = [] + + if footer_el: + vendors = _vendors_from_links(footer_el, site_domain) + if not vendors: + text = footer_el.get_text(separator=" ", strip=True)[:_MAX_CHARS] + vendors = _vendors_from_text(text) + if not vendors: + vendors = _vendors_from_copyright_links(footer_el, site_domain) + + if not vendors: + body = soup.find("body") + if body: + full = body.get_text(separator=" ", strip=True) + start = max(0, int(len(full) * (1 - _FOOTER_FRACTION))) + vendors = _vendors_from_text(full[start:][:_MAX_CHARS]) + + return vendors + + +def _resolve_domain_name(domain: str, site_domain: str) -> str | None: + domain = domain.lstrip("www.") + if domain in site_domain or site_domain in domain: + return None + if domain in _BLOCKED_DOMAINS: + return None + if domain in _DOMAIN_NAME_MAP: + return _DOMAIN_NAME_MAP[domain] + return domain + + +def _link_name(a_tag: "Tag", site_domain: str) -> str | None: + text = a_tag.get_text(strip=True) + if text and not _is_noise(text): + return text + href = a_tag.get("href", "") + if href: + try: + domain = urlparse(href).netloc + return _resolve_domain_name(domain, site_domain) + except Exception: + pass + return None + + +def _vendors_from_links(container: "Tag", site_domain: str) -> list[str]: + vendors: list[str] = [] + for text_node in container.find_all(string=True): + if _CREDIT_TRIGGERS.search(text_node): + parent = text_node.parent + if parent: + if parent.name == "a": + name = _link_name(parent, site_domain) + if name and name not in vendors: + vendors.append(name) + else: + for a in parent.find_all("a"): + name = _link_name(a, site_domain) + if name and name not in vendors: + vendors.append(name) + return vendors + + +def _vendors_from_copyright_links(container: "Tag", site_domain: str) -> list[str]: + vendors: list[str] = [] + html_str = str(container) + copyright_re = re.compile(r"©.*?(?=©|$)", re.IGNORECASE | re.DOTALL) + for block_m in copyright_re.finditer(html_str): + block = block_m.group(0)[:600] + mini = BeautifulSoup(block, "html.parser") + for a in mini.find_all("a"): + href = a.get("href", "") + if not href: + continue + try: + domain = urlparse(href).netloc + except Exception: + continue + name = _resolve_domain_name(domain, site_domain) + if name and not _is_noise(name) and name not in vendors: + vendors.append(name) + return vendors + + +def _vendors_from_text(text: str) -> list[str]: + vendors: list[str] = [] + for pattern in _TEXT_PATTERNS: + for m in pattern.finditer(text): + candidate = m.group(1).strip().rstrip(".,;:|") + candidate = re.split(r"\s{2,}|\|", candidate)[0].strip() + candidate = re.sub(r"[\s.]+[\d.]+$", "", candidate).strip() + candidate = candidate.rstrip(".,;:|").strip() + if candidate and not _is_noise(candidate) and candidate not in vendors: + vendors.append(candidate) + return vendors + + +def _is_noise(s: str) -> bool: + lower = s.lower().strip() + if len(lower) < 3: + return True + if lower in _NAV_WORDS: + return True + if any(cms in lower for cms in _CMS_NOISE): + return True + if lower in _BLOCKED_DOMAINS: + return True + return any(noise in lower for noise in _NOISE) \ No newline at end of file diff --git a/src/websitescorecard/runner.py b/src/websitescorecard/runner.py index 9b8d55b..f5d13cb 100644 --- a/src/websitescorecard/runner.py +++ b/src/websitescorecard/runner.py @@ -9,6 +9,7 @@ from rich.progress import BarColumn, Progress, TaskProgressColumn, TextColumn, TimeElapsedColumn from websitescorecard.checks.base import Check +from websitescorecard.checks.footer_vendor import FooterVendorCheck, SCRAPE_URL_COLUMN from websitescorecard.csv_io import read_csv, write_csv @@ -31,11 +32,16 @@ def _output_columns(original: list[str], checks: list[Check], include_errors: bo return columns -def _run_checks_for_row(url: str, checks: list[Check]) -> dict[str, str]: +def _run_checks_for_row(url: str, checks: list[Check], row: dict[str, str]) -> dict[str, str]: results: dict[str, str] = {} for check in checks: try: - result = check.run(url) + if isinstance(check, FooterVendorCheck): + scrape_url = row.get(SCRAPE_URL_COLUMN, "") or None + result = check.run(url, scrape_url=scrape_url) + else: + result = check.run(url) + results[check.column] = result.status if check.error_column: results[check.error_column] = result.error or "" @@ -48,7 +54,7 @@ def _run_checks_for_row(url: str, checks: list[Check]) -> dict[str, str]: def _scan_row(index: int, row: dict[str, str], url_column: str, checks: list[Check]) -> tuple[int, dict[str, str]]: url = row.get(url_column, "") - check_results = _run_checks_for_row(url, checks) + check_results = _run_checks_for_row(url, checks, row) enriched = {**row, **check_results} return index, enriched @@ -85,4 +91,4 @@ def run_scan(config: ScanConfig) -> None: enriched_rows[index] = enriched progress.advance(task) - write_csv(config.output_path, output_columns, enriched_rows) # type: ignore[arg-type] + write_csv(config.output_path, output_columns, enriched_rows) # type: ignore[arg-type] \ No newline at end of file diff --git a/tests/test_footer_vendor.py b/tests/test_footer_vendor.py new file mode 100644 index 0000000..125c9e7 --- /dev/null +++ b/tests/test_footer_vendor.py @@ -0,0 +1,332 @@ +"""Tests for FooterVendorCheck.""" + +from __future__ import annotations + +from types import SimpleNamespace + +import pytest + +bs4 = pytest.importorskip("bs4") +httpx = pytest.importorskip("httpx") + +from bs4 import BeautifulSoup + +import websitescorecard.checks.footer_vendor as fv +from websitescorecard.checks.footer_vendor import FooterVendorCheck, SCRAPE_URL_COLUMN + +class _FakeResponse: + def __init__(self, status_code: int, text: str) -> None: + self.status_code = status_code + self.text = text + + +class _FakeClient: + """responses: url -> (status_code, html) or an Exception to raise.""" + + def __init__(self, responses: dict[str, object], calls: list[str], **kwargs) -> None: + self._responses = responses + self._calls = calls + + def __enter__(self) -> "_FakeClient": + return self + + def __exit__(self, *exc_info) -> bool: + return False + + def get(self, url: str) -> _FakeResponse: + self._calls.append(url) + outcome = self._responses.get(url) + if outcome is None: + raise httpx.RequestError(f"no fake route configured for {url}") + if isinstance(outcome, BaseException): + raise outcome + status_code, text = outcome + return _FakeResponse(status_code, text) + + +def _install_fake_httpx(monkeypatch: pytest.MonkeyPatch, responses: dict[str, object]) -> list[str]: + calls: list[str] = [] + monkeypatch.setattr( + fv, + "httpx", + SimpleNamespace( + Client=lambda **kw: _FakeClient(responses, calls, **kw), + RequestError=httpx.RequestError, + HTTPStatusError=httpx.HTTPStatusError, + ), + ) + return calls + +@pytest.mark.parametrize( + "url, expected", + [ + ("example.com", ["https://example.com", "http://example.com"]), + ("https://example.com", ["https://example.com", "http://example.com"]), + ("http://example.com", ["http://example.com", "https://example.com"]), + ], +) +def test_candidate_urls(url: str, expected: list[str]): + assert fv._candidate_urls(url) == expected + + +@pytest.mark.parametrize( + "text, expected", + [ + ("ab", True), # too short + ("Back", True), # nav word + ("Built with WordPress", True), # CMS noise + ("All Rights Reserved", True), # noise phrase + ("Acme Solutions", False), + ("vendorhost.net", False), + ], +) +def test_is_noise(text: str, expected: bool): + assert fv._is_noise(text) is expected + +def test_resolve_domain_name_returns_none_for_self_domain(): + assert fv._resolve_domain_name("sub.example.gov.lk", "example.gov.lk") is None + + +def test_resolve_domain_name_maps_known_domain_to_friendly_name(): + assert fv._resolve_domain_name("icta.lk", "example.gov.lk") == "ICTA" + + +def test_resolve_domain_name_returns_raw_domain_when_unmapped(): + assert fv._resolve_domain_name("vendorhost.net", "example.gov.lk") == "vendorhost.net" + + +def test_resolve_domain_name_skips_blocked_domains(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setattr(fv, "_BLOCKED_DOMAINS", frozenset({"blocked.example"})) + assert fv._resolve_domain_name("blocked.example", "example.gov.lk") is None + +def test_vendors_from_text_extracts_name_after_trigger_phrase(): + assert fv._vendors_from_text("Developed by Globex Technologies Pvt Ltd.") == [ + "Globex Technologies Pvt Ltd" + ] + +def test_vendors_from_text_dedupes_same_vendor_across_patterns(): + text = "Built by Acme Corp | Solution by: Acme Corp" + assert fv._vendors_from_text(text) == ["Acme Corp"] + +def test_vendors_from_text_filters_out_noise_candidates(): + assert fv._vendors_from_text("Designed by Privacy Policy") == [] + +def test_vendors_from_text_returns_empty_when_no_trigger_present(): + assert fv._vendors_from_text("Welcome to our website. Contact us.") == [] + +def test_vendors_from_links_uses_link_text_when_present(): + soup = BeautifulSoup( + '

Website by Creative Corp

', + "html.parser", + ) + vendors = fv._vendors_from_links(soup.find("div"), site_domain="example.gov.lk") + assert vendors == ["Creative Corp"] + +def test_vendors_from_links_falls_back_to_resolved_href_domain_when_text_empty(): + soup = BeautifulSoup( + '

Hosted by

', "html.parser" + ) + vendors = fv._vendors_from_links(soup.find("div"), site_domain="example.gov.lk") + assert vendors == ["ICTA"] + + +def test_vendors_from_links_ignores_self_links_and_missing_triggers(): + soup = BeautifulSoup( + '

Hosted by

' + '

Quick links: About

', + "html.parser", + ) + vendors = fv._vendors_from_links(soup.find("div"), site_domain="example.gov.lk") + assert vendors == [] + +def test_vendors_from_copyright_links_extracts_domain_near_copyright_symbol(): + soup = BeautifulSoup( + '
\u00a9 2024 Ministry of Foo. Site by ' + 'CreativeCorp
', + "html.parser", + ) + vendors = fv._vendors_from_copyright_links(soup.find("div"), site_domain="example.gov.lk") + assert vendors == ["creativecorp.io"] + + +def test_vendors_from_copyright_links_returns_empty_with_no_copyright_symbol(): + soup = BeautifulSoup( + '
All rights reserved. CreativeCorp
', + "html.parser", + ) + vendors = fv._vendors_from_copyright_links(soup.find("div"), site_domain="example.gov.lk") + assert vendors == [] + +@pytest.mark.parametrize( + "html, found", + [ + ("", True), + ('', True), + ('
F
', True), + ("
No footer here
", False), + ], +) +def test_find_footer_element(html: str, found: bool): + soup = BeautifulSoup(f"{html}", "html.parser") + el = fv._find_footer_element(soup) + assert (el is not None) is found + +def test_extract_vendors_uses_footer_links_first(): + soup = BeautifulSoup( + '', + "html.parser", + ) + footer_el = fv._find_footer_element(soup) + vendors = fv._extract_vendors(footer_el, soup, site_domain="example.gov.lk") + assert vendors == ["Acme Vendor"] + + +def test_extract_vendors_falls_back_to_footer_text_when_no_links(): + soup = BeautifulSoup( + "", + "html.parser", + ) + footer_el = fv._find_footer_element(soup) + vendors = fv._extract_vendors(footer_el, soup, site_domain="example.gov.lk") + assert vendors == ["ICT Unit"] + + +def test_extract_vendors_falls_back_to_body_tail_when_footer_has_nothing(): + html = ( + "" + "

Lots of unrelated filler content sits here to pad out the page body.

" + "" + "

Powered by TailVendor

" + "" + ) + soup = BeautifulSoup(html, "html.parser") + footer_el = fv._find_footer_element(soup) + vendors = fv._extract_vendors(footer_el, soup, site_domain="example.gov.lk") + assert vendors == ["TailVendor"] + + +def test_extract_vendors_returns_empty_for_blank_page(): + soup = BeautifulSoup("

Nothing interesting here.

", "html.parser") + footer_el = fv._find_footer_element(soup) + vendors = fv._extract_vendors(footer_el, soup, site_domain="example.gov.lk") + assert vendors == [] + +def test_run_returns_found_when_vendor_link_in_footer(monkeypatch: pytest.MonkeyPatch): + html = ( + "' + ) + calls = _install_fake_httpx(monkeypatch, {"https://example.gov.lk": (200, html)}) + + result = FooterVendorCheck().run("https://example.gov.lk") + + assert result.status == "found" + assert result.error == "Acme Vendor" + assert calls == ["https://example.gov.lk"] + + +def test_run_returns_self_built_when_vendor_matches_internal_unit(monkeypatch: pytest.MonkeyPatch): + html = "" + _install_fake_httpx(monkeypatch, {"https://example.gov.lk": (200, html)}) + + result = FooterVendorCheck().run("https://example.gov.lk") + + assert result.status == "self_built" + assert result.error == "ICT Unit" + + +def test_run_returns_unknown_when_page_loads_but_no_credit_found(monkeypatch: pytest.MonkeyPatch): + html = "" + _install_fake_httpx(monkeypatch, {"https://example.gov.lk": (200, html)}) + + result = FooterVendorCheck().run("https://example.gov.lk") + + assert result.status == "unknown" + assert result.error is None + + +def test_run_returns_unreachable_when_every_fetch_fails(monkeypatch: pytest.MonkeyPatch): + _install_fake_httpx(monkeypatch, {}) + + result = FooterVendorCheck().run("https://example.gov.lk") + + assert result.status == "unreachable" + assert result.error == "Failed to fetch page" + + +def test_run_returns_unreachable_when_url_cannot_be_parsed(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setattr( + fv, "parse_url", lambda url: (_ for _ in ()).throw(ValueError(f"could not parse {url!r}")) + ) + + result = FooterVendorCheck().run("not-a-url") + + assert result.status == "unreachable" + assert "could not parse" in result.error + + +def test_run_tries_http_fallback_when_https_fails(monkeypatch: pytest.MonkeyPatch): + html = "" + calls = _install_fake_httpx( + monkeypatch, + { + "https://example.gov.lk": httpx.RequestError("connection refused"), + "http://example.gov.lk": (200, html), + }, + ) + + result = FooterVendorCheck().run("https://example.gov.lk") + + assert result.status == "found" + assert result.error == "Fallback Co" + assert calls == ["https://example.gov.lk", "http://example.gov.lk"] + + +def test_run_prefers_scrape_url_over_main_url(monkeypatch: pytest.MonkeyPatch): + html = "" + calls = _install_fake_httpx( + monkeypatch, {"https://example.gov.lk/about": (200, html)} + ) + + result = FooterVendorCheck().run( + "https://example.gov.lk", scrape_url="https://example.gov.lk/about" + ) + + assert result.status == "found" + assert result.error == "Scrape Vendor" + assert calls == ["https://example.gov.lk/about"] + + +def test_run_falls_back_to_main_url_when_scrape_url_unreachable(monkeypatch: pytest.MonkeyPatch): + html = "" + _install_fake_httpx(monkeypatch, {"https://example.gov.lk": (200, html)}) + + result = FooterVendorCheck().run( + "https://example.gov.lk", scrape_url="https://bad.example.gov.lk/about" + ) + + assert result.status == "found" + assert result.error == "Main Vendor" + + +@pytest.mark.parametrize("scrape_url", [" ", "https://example.gov.lk"]) +def test_run_ignores_blank_or_duplicate_scrape_url(monkeypatch: pytest.MonkeyPatch, scrape_url: str): + html = "" + calls = _install_fake_httpx(monkeypatch, {"https://example.gov.lk": (200, html)}) + + result = FooterVendorCheck().run("https://example.gov.lk", scrape_url=scrape_url) + + assert result.status == "found" + assert calls == ["https://example.gov.lk"] + + +def test_init_raises_when_required_dependencies_missing(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setattr(fv, "_MISSING_DEPS", ["httpx", "beautifulsoup4"]) + + with pytest.raises(RuntimeError, match="httpx"): + FooterVendorCheck() + + +def test_scrape_url_column_constant_value(): + assert SCRAPE_URL_COLUMN == "Scrape_URL" \ No newline at end of file diff --git a/tests/test_runner.py b/tests/test_runner.py index f2f521f..59386a7 100644 --- a/tests/test_runner.py +++ b/tests/test_runner.py @@ -6,6 +6,7 @@ from pathlib import Path from websitescorecard.checks.base import CheckResult +from websitescorecard.checks.footer_vendor import FooterVendorCheck, SCRAPE_URL_COLUMN from websitescorecard.runner import ScanConfig, _run_checks_for_row, run_scan @@ -29,8 +30,20 @@ def run(self, url: str) -> CheckResult: raise RuntimeError("check exploded") +@dataclass +class _NoErrorColumnCheck: + """A check that doesn't report a separate error column.""" + + name = "noerr" + column = "noerr_status" + error_column = None + + def run(self, url: str) -> CheckResult: + return CheckResult(status="pass", error=None) + + def test_run_checks_for_row_records_error_when_check_raises(): - results = _run_checks_for_row("example.com", [_RaisingCheck()]) + results = _run_checks_for_row("example.com", [_RaisingCheck()], row={}) assert results == { "boom_status": "error", @@ -39,14 +52,15 @@ def test_run_checks_for_row_records_error_when_check_raises(): def test_run_checks_for_row_continues_after_one_check_raises(): - results = _run_checks_for_row("example.com", [_OkCheck(), _RaisingCheck()]) + results = _run_checks_for_row( + "example.com", [_OkCheck(), _RaisingCheck()], row={} + ) assert results["ok_status"] == "pass" assert results["ok_error"] == "" assert results["boom_status"] == "error" assert results["boom_error"] == "Unexpected error: check exploded" - - + def test_run_scan_completes_when_row_check_raises(tmp_path: Path): input_csv = tmp_path / "input.csv" output_csv = tmp_path / "output.csv" @@ -66,3 +80,4 @@ def test_run_scan_completes_when_row_check_raises(tmp_path: Path): assert lines[0] == "name,website,boom_status,boom_error" assert "error" in lines[1] assert "error" in lines[2] +