diff --git a/README.md b/README.md index 11aedef..59474ca 100644 --- a/README.md +++ b/README.md @@ -1,3 +1,10 @@ +## Fork development: SearXNG support + +This fork adds `--search searxng` with JSON results, bounded pagination, profile +URL deduplication and explicit failure reporting. See the +[SearXNG setup and usage guide](docs/searxng.md). Existing Google/Bing defaults +remain available; their current scraping behavior has not been verified. +

CrossLinked

diff --git a/crosslinked/__init__.py b/crosslinked/__init__.py index d75086d..716eba3 100644 --- a/crosslinked/__init__.py +++ b/crosslinked/__init__.py @@ -3,18 +3,21 @@ # License: GPLv3 import re import argparse +import math +import os from sys import exit from csv import reader from crosslinked import utils from crosslinked.logger import * from crosslinked.search import CrossLinked +from crosslinked.searxng import SearXNG, search_endpoint def banner(): VERSION = 'v0.3.0' - print(''' + print(r''' _____ _ _ _ / __ \ | | ({}) | | | | | / \/_ __ ___ ___ ___ | | _ _ __ | | _____ __| | @@ -36,16 +39,32 @@ def cli(): s = args.add_argument_group("Search arguments") s.add_argument('--search', dest='engine', default='google,bing', type=lambda x: utils.delimiter2list(x), help='Search Engine (Default=\'google,bing\')') + s.add_argument('--searxng-url', default=os.environ.get('SEARXNG_URL'), help='SearXNG instance URL (or SEARXNG_URL environment variable)') + s.add_argument('--max-pages', type=int, default=5, help='Maximum SearXNG pages (Default=5)') o = args.add_argument_group("Output arguments") - o.add_argument('-f', dest='nformat', type=str, required=True, help='Format names, ex: \'domain\{f}{last}\', \'{first}.{last}@domain.com\'') + o.add_argument('-f', dest='nformat', type=str, required=True, help="Format names, ex: 'domain\\{f}{last}', '{first}.{last}@domain.com'") o.add_argument('-o', dest='outfile', type=str, default='names', help='Change name of output file (omit_extension)') p = args.add_argument_group("Proxy arguments") pr = p.add_mutually_exclusive_group(required=False) pr.add_argument('--proxy', dest='proxy', action='append', default=[], help='Proxy requests (IP:Port)') pr.add_argument('--proxy-file', dest='proxy', default=False, type=lambda x: utils.file_exists(x), help='Load proxies from file for rotation') - return args.parse_args() + parsed = args.parse_args() + if not parsed.company_name: + args.error('company_name is required') + if any(engine not in ('google', 'bing', 'searxng') for engine in parsed.engine) or not parsed.engine: + args.error('--search supports google, bing, searxng') + if not math.isfinite(parsed.timeout) or parsed.timeout <= 0 or not math.isfinite(parsed.jitter) or parsed.jitter < 0 or parsed.max_pages < 1: + args.error('timeout and max-pages must be positive; jitter must be nonnegative') + if 'searxng' in parsed.engine and not parsed.company_name.endswith('.csv'): + if not parsed.searxng_url: + args.error('--search searxng requires --searxng-url or SEARXNG_URL') + try: + search_endpoint(parsed.searxng_url) + except ValueError as exc: + args.error(str(exc)) + return parsed def start_scrape(args): @@ -53,6 +72,13 @@ def start_scrape(args): Log.info("Searching {} for valid employee names at \"{}\"".format(', '.join(args.engine), args.company_name)) for search_engine in args.engine: + if search_engine == 'searxng': + c = SearXNG(args.company_name, args.searxng_url, args.timeout, + args.jitter, args.max_pages, args.proxy) + tmp += c.search() + if c.status in ('failed', 'partial'): + args.search_failed = True + continue c = CrossLinked(search_engine, args.company_name, args.timeout, 3, args.proxy, args.jitter) if search_engine in c.url.keys(): tmp += c.search() @@ -118,7 +144,13 @@ def main(): csv = setup_file_logger(args.outfile+".csv", log_name="cLinked_csv", file_mode='a') # names.csv appended data = start_parse(args) if args.company_name.endswith('.csv') else start_scrape(args) - format_names(args, data, txt) if len(data) > 0 else Log.warn('No results found') + if data: + format_names(args, data, txt) + elif not getattr(args, 'search_failed', False): + Log.warn('No results found') + if getattr(args, 'search_failed', False): + Log.warn('Search incomplete; any saved results may be partial') + exit(1) except KeyboardInterrupt: Log.warn("Key event detected, closing...") exit(0) diff --git a/crosslinked/search.py b/crosslinked/search.py index 3ebca2b..baf5fc0 100644 --- a/crosslinked/search.py +++ b/crosslinked/search.py @@ -1,4 +1,6 @@ import logging +import csv as csv_module +import io import requests import threading from time import sleep @@ -127,7 +129,9 @@ def log_results(self, d): self.results.append(d) # Search results are logged to names.csv but names.txt is not generated until end to prevent duplicates logging.debug('name: {:25} RawTxt: {}'.format(d['name'], d['text'])) - csv.info('"{}","{}","{}","{}","{}","{}",'.format(self.runtime, self.search_engine, d['name'], d['title'], d['url'], d['text'])) + row = io.StringIO() + csv_module.writer(row).writerow([self.runtime, self.search_engine, d['name'], d['title'], d['url'], d['text']]) + csv.info(row.getvalue().rstrip('\r\n')) def get_statuscode(resp): diff --git a/crosslinked/searxng.py b/crosslinked/searxng.py new file mode 100644 index 0000000..b009c66 --- /dev/null +++ b/crosslinked/searxng.py @@ -0,0 +1,111 @@ +"""SearXNG JSON provider. No LinkedIn requests or browser required.""" +import re +from time import monotonic, sleep +from urllib.parse import urlsplit, urlunsplit + +import requests +from bs4 import BeautifulSoup +from unidecode import unidecode + +from crosslinked.search import CrossLinked, get_proxy +from crosslinked.logger import Log + + +def search_endpoint(value): + parts = urlsplit(value) + if (parts.scheme not in ('http', 'https') or not parts.hostname + or parts.username or parts.password or parts.query or parts.fragment): + raise ValueError('SearXNG URL must be an HTTP(S) URL without credentials, query or fragment') + path = parts.path.rstrip('/') + if not path.endswith('/search'): + path += '/search' + return urlunsplit((parts.scheme, parts.netloc, path, '', '')) + + +class SearXNG(CrossLinked): + def __init__(self, target, endpoint, timeout=15, jitter=1, max_pages=5, proxies=None): + super().__init__('searxng', target, timeout, proxies=proxies or [], jitter=jitter) + self.endpoint = search_endpoint(endpoint) + self.max_pages = max_pages + self.status = 'pending' + self.warnings = [] + + def warn(self, message): + self.warnings.append(message) + Log.warn(message) + + def parse_result(self, item): + if not isinstance(item, dict): + return None + url, title = item.get('url'), item.get('title') + if not isinstance(url, str) or not isinstance(title, str): + return None + try: + parts = urlsplit(url) + host = (parts.hostname or '').lower() + except ValueError: + return None + if (parts.scheme not in ('http', 'https') or parts.username or parts.password + or not (host == 'linkedin.com' or host.endswith('.linkedin.com')) + or not re.fullmatch(r'/in/[^/]+/?', parts.path)): + return None + text = BeautifulSoup(title, 'html.parser').get_text(' ', strip=True) + text = re.sub(r'\s+', ' ', text).strip() + fields = re.split(r'\s+[-–—|]\s+', text) + name = fields[0].strip() + if len(name.split()) < 2 or 'linkedin' in name.lower() or not any(c.isalpha() for c in name): + return None + return {'name': unidecode(name).lower(), + 'title': fields[1] if len(fields) > 1 and fields[1].lower() != 'linkedin' else 'N/A', + 'url': 'https://www.linkedin.com' + parts.path.rstrip('/'), + 'text': text} + + def search(self): + deadline = monotonic() + self.timeout + seen_urls, seen_pages = set(), set() + self.status = 'complete' + with requests.Session() as session: + for page in range(1, self.max_pages + 1): + remaining = deadline - monotonic() + if remaining <= 0: + self.status = 'limited' + break + try: + response = session.get(self.endpoint, params={ + 'q': 'site:linkedin.com/in "{}"'.format(self.target.replace('"', ' ')), + 'format': 'json', 'pageno': page, 'categories': 'general'}, + timeout=min(remaining, 15), proxies=get_proxy(self.proxies)) + if response.status_code == 403: + raise ValueError('HTTP 403: enable JSON in SearXNG search.formats and check instance access') + response.raise_for_status() + payload = response.json() + if not isinstance(payload, dict) or not isinstance(payload.get('results'), list): + raise ValueError('Invalid SearXNG JSON: expected a results list') + except (requests.RequestException, ValueError) as exc: + self.status = 'failed' + self.warn('SearXNG request failed: {}'.format(exc)) + break + if payload.get('unresponsive_engines'): + self.status = 'partial' + self.warn('SearXNG upstream failures: {}'.format(payload['unresponsive_engines'])) + items = payload['results'] + if not items: + break + fingerprint = repr(items) + if fingerprint in seen_pages: + self.status = 'limited' if self.status == 'complete' else self.status + self.warn('SearXNG repeated a page; stopping pagination') + break + seen_pages.add(fingerprint) + for item in items: + record = self.parse_result(item) + if record and record['url'] not in seen_urls: + seen_urls.add(record['url']) + self.log_results(record) + Log.info('SearXNG page {}: {} unique profiles'.format(page, len(self.results))) + if page == self.max_pages: + self.status = 'limited' if self.status == 'complete' else self.status + else: + sleep(min(self.jitter, max(0, deadline - monotonic()))) + Log.info('SearXNG status: {}'.format(self.status)) + return self.results diff --git a/docs/searxng.md b/docs/searxng.md new file mode 100644 index 0000000..088eeb2 --- /dev/null +++ b/docs/searxng.md @@ -0,0 +1,227 @@ +# SearXNG provider + +Use a SearXNG instance you operate or are permitted to query. No public instance +is selected automatically. SearXNG depends on upstream search providers; this +integration does not contact LinkedIn directly. + +Merge this into the instance's `settings.yml`, preserving other settings, then restart: + +```yaml +search: + formats: + - html + - json +``` + +Configure upstream engines in SearXNG itself. Official documentation: +[installation](https://docs.searxng.org/admin/installation.html), +[JSON search API](https://docs.searxng.org/dev/search_api.html). + +## Kali WSL: setup on a new Windows device + +These steps use Docker Desktop's Linux engine, with CrossLinked running inside +Kali WSL. Docker Desktop hosts SearXNG; activating the Python environment does +not start Docker. + +1. Install Docker Desktop and Kali under WSL 2. +2. Open Docker Desktop and enable its WSL 2 engine. Under **Settings → Resources + → WSL Integration**, enable Kali, then apply the changes. +3. Enable **Start Docker Desktop when you sign in** in Docker Desktop settings + if you want the service available after Windows login. +4. Open Kali and check `sudo docker info`. If it cannot connect, start Docker + Desktop and check Kali integration before continuing. + +See [Docker Desktop for Windows](https://docs.docker.com/desktop/setup/install/windows-install/) +and [Docker WSL integration](https://docs.docker.com/desktop/features/wsl/). + +### Create SearXNG once + +Run these commands in Kali. The configuration command is for a **new device**; +do not overwrite an existing `settings.yml` containing your custom settings. + +```bash +mkdir -p ~/searxng/config +cd ~/searxng + +cat > config/settings.yml <&1 \ + | grep -iE 'duckduckgo|403|429|captcha|denied' +``` + +## Run from this repository (PowerShell alternative) + +```powershell +python -m venv venv +./venv/Scripts/python -m pip install requests beautifulsoup4 lxml Unidecode +./venv/Scripts/python crosslinked.py 'Example Company' --search searxng --searxng-url http://localhost:8080 -t 60 --max-pages 5 -f '{first}.{last}@example.com' +``` + +Alternatively set `$env:SEARXNG_URL = 'http://localhost:8080'`. URLs can include a +deployment subpath or end in `/search`; credentials, queries and fragments are +rejected. HTTPS verifies certificates. Existing proxy flags are supported. + +Output remains `names.txt` and `names.csv`. CSV records the search-result title +and normalized profile URL. Names and job titles are inferred from titles; +employment and generated email addresses are not verified. The first record for +a profile wins. Regional LinkedIn URLs and tracking parameters are deduplicated. + +## Status + +- `complete`: reached an empty page; does not establish exhaustive coverage. +- `limited`: page/time budget reached or a repeated page detected. +- `partial`: SearXNG reported upstream engine failures. +- `failed`: HTTP/network error, invalid JSON or an unexpected response schema. + +Partial/failed searches retain collected records and exit with code 1. HTTP 403 +includes JSON configuration guidance. Requests use the remaining search budget, +capped at 15 seconds per request. These are socket inactivity timeouts, not a +strict wall-clock deadline. Blocked requests are not retried automatically. + +## Verification + +```powershell +./venv/Scripts/python -m unittest discover -s tests -v +``` + +Tests simulate responses. Live coverage must be benchmarked with a configured +instance. This addition does not repair legacy Google/Bing scraping or implement +browser automation, YaCy, or the planned Pi extension. diff --git a/tests/test_cli_integration.py b/tests/test_cli_integration.py new file mode 100644 index 0000000..4de6e56 --- /dev/null +++ b/tests/test_cli_integration.py @@ -0,0 +1,64 @@ +"""Exercise actual HTTP, CLI exit codes, and output files against a local fixture.""" +import csv +import json +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +import subprocess +import sys +import tempfile +from threading import Thread +import unittest +from urllib.parse import parse_qs, urlsplit + + +class CliIntegrationTests(unittest.TestCase): + def test_http_to_files_and_failure_exit(self): + pages = [] + + class Handler(BaseHTTPRequestHandler): + def do_GET(self): + params = parse_qs(urlsplit(self.path).query) + pages.append(params) + if 'Failure' in params['q'][0]: + self.send_response(403) + self.end_headers() + return + payload = {'results': []} + if params['pageno'] == ['1']: + payload['results'] = [{'url': 'https://www.linkedin.com/in/jane-example', + 'title': 'Jane Example - Engineer | LinkedIn'}] + body = json.dumps(payload).encode() + self.send_response(200) + self.send_header('Content-Type', 'application/json') + self.send_header('Content-Length', str(len(body))) + self.end_headers() + self.wfile.write(body) + + def log_message(self, *args): + pass + + server = ThreadingHTTPServer(('127.0.0.1', 0), Handler) + thread = Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + with tempfile.TemporaryDirectory() as directory: + command = [sys.executable, str(Path(__file__).resolve().parents[1] / 'crosslinked.py'), + '--search', 'searxng', '--searxng-url', + 'http://127.0.0.1:{}'.format(server.server_port), + '-j', '0', '-f', '{first}.{last}@example.com'] + run = subprocess.run(command + ['Example'], cwd=directory, capture_output=True, text=True, timeout=15) + self.assertEqual(run.returncode, 0, run.stdout + run.stderr) + self.assertEqual(Path(directory, 'names.txt').read_text().strip(), 'jane.example@example.com') + with Path(directory, 'names.csv').open(newline='') as handle: + rows = list(csv.DictReader(handle)) + self.assertEqual(rows[0]['Search'], 'searxng') + self.assertEqual(rows[0]['Name'], 'jane example') + self.assertEqual([p['pageno'] for p in pages], [['1'], ['2']]) + run = subprocess.run(command + ['Failure'], cwd=directory, capture_output=True, text=True, timeout=15) + self.assertEqual(run.returncode, 1, run.stdout + run.stderr) + self.assertIn('HTTP 403', run.stdout) + self.assertNotIn('No results found', run.stdout) + finally: + server.shutdown() + server.server_close() + thread.join() diff --git a/tests/test_searxng.py b/tests/test_searxng.py new file mode 100644 index 0000000..ece22cf --- /dev/null +++ b/tests/test_searxng.py @@ -0,0 +1,120 @@ +import csv +import io +import unittest +from unittest.mock import Mock, patch +import requests +from crosslinked import cli +from crosslinked.searxng import SearXNG, search_endpoint + + +def result(title='Jane Doe - Engineer | LinkedIn'): + return {'url': 'https://www.linkedin.com/in/jane-doe', 'title': title} + + +def response(payload=None, status=200): + value = Mock(status_code=status) + value.json.return_value = payload + if status >= 400: + value.raise_for_status.side_effect = requests.HTTPError(str(status)) + return value + + +class ProviderTests(unittest.TestCase): + def provider(self, **kwargs): + return SearXNG('Example & Co', 'http://localhost:8080', jitter=0, **kwargs) + + def run_pages(self, provider, pages): + with patch('crosslinked.searxng.requests.Session') as session: + get = session.return_value.__enter__.return_value.get + get.side_effect = pages + return provider.search(), get.call_args_list + + def test_endpoint(self): + for value in ('http://localhost:8080', 'http://localhost:8080/search/'): + self.assertEqual(search_endpoint(value), 'http://localhost:8080/search') + self.assertEqual(search_endpoint('https://example.org/prefix'), 'https://example.org/prefix/search') + for value in ('file:///tmp', 'https://user:pass@example.org', 'https://example.org?q=x'): + with self.assertRaises(ValueError): + search_endpoint(value) + + def test_pagination_independent_of_extracted_count(self): + provider = self.provider() + records, calls = self.run_pages(provider, [ + response({'results': [{'url': 'https://example.org', 'title': 'Unrelated'}]}), + response({'results': [result()]}), response({'results': []})]) + self.assertEqual(len(records), 1) + self.assertEqual([c.kwargs['params']['pageno'] for c in calls], [1, 2, 3]) + self.assertEqual(calls[0].kwargs['params']['q'], 'site:linkedin.com/in "Example & Co"') + self.assertNotIn('verify', calls[0].kwargs) + self.assertEqual(provider.status, 'complete') + + def test_dedup_and_repeated_page(self): + provider = self.provider() + duplicate = result() + duplicate['url'] = 'https://au.linkedin.com/in/jane-doe/?trk=search' + payload = {'results': [result(), duplicate]} + records, calls = self.run_pages(provider, [response(payload), response(payload)]) + self.assertEqual(len(records), 1) + self.assertEqual(len(calls), 2) + self.assertEqual(provider.status, 'limited') + + def test_reject_nonprofiles(self): + provider = self.provider() + for url in ('https://notlinkedin.com/in/jane', 'https://linkedin.com.evil.org/in/jane', + 'https://linkedin.com/company/example', 'https://linkedin.com/in/', + 'https://linkedin.com/in/jane/details', 'javascript:alert(1)'): + self.assertIsNone(provider.parse_result({'url': url, 'title': 'Jane Doe'})) + for item in (None, {}, {'url': [], 'title': 2}): + self.assertIsNone(provider.parse_result(item)) + + def test_hyphenated_name(self): + record = self.provider().parse_result(result('Anne-Marie Dupont – Senior Engineer | LinkedIn')) + self.assertEqual(record['name'], 'anne-marie dupont') + self.assertEqual(record['title'], 'Senior Engineer') + + def test_failures_not_empty_success(self): + invalid = response() + invalid.json.side_effect = ValueError('Not JSON') + for page in (response(status=403), response(status=429), invalid, + response({'unexpected': []}), requests.Timeout('timeout')): + provider = self.provider() + records, _ = self.run_pages(provider, [page]) + self.assertEqual(records, []) + self.assertEqual(provider.status, 'failed') + + def test_partial_results_survive_failure(self): + provider = self.provider() + records, _ = self.run_pages(provider, [response({'results': [result()]}), response(status=503)]) + self.assertEqual(len(records), 1) + self.assertEqual(provider.status, 'failed') + + def test_upstream_failure(self): + provider = self.provider() + self.run_pages(provider, [response({'results': [], 'unresponsive_engines': [['bing', 'timeout']]})]) + self.assertEqual(provider.status, 'partial') + + def test_page_cap(self): + provider = self.provider(max_pages=1) + _, calls = self.run_pages(provider, [response({'results': [result()]})]) + self.assertEqual(len(calls), 1) + self.assertEqual(provider.status, 'limited') + + def test_csv_quotes(self): + provider = self.provider() + record = provider.parse_result(result('Jane Doe - Engineer, "Platform" | LinkedIn')) + with patch('crosslinked.search.csv.info') as log: + provider.log_results(record) + row = next(csv.reader(io.StringIO(log.call_args.args[0]))) + self.assertEqual(len(row), 6) + self.assertEqual(row[3], 'Engineer, "Platform"') + + def test_cli_validation(self): + for flags in (['--search', 'unknown'], ['--search', 'searxng'], ['-t', 'nan'], ['-j', '-1']): + with patch.dict('os.environ', {}, clear=True), patch('sys.argv', ['crosslinked', 'Example', '-f', '{first}.{last}'] + flags): + with self.assertRaises(SystemExit) as exc: + cli() + self.assertEqual(exc.exception.code, 2) + + +if __name__ == '__main__': + unittest.main()