diff --git a/README.md b/README.md
index 11aedef..59474ca 100644
--- a/README.md
+++ b/README.md
@@ -1,3 +1,10 @@
+## Fork development: SearXNG support
+
+This fork adds `--search searxng` with JSON results, bounded pagination, profile
+URL deduplication and explicit failure reporting. See the
+[SearXNG setup and usage guide](docs/searxng.md). Existing Google/Bing defaults
+remain available; their current scraping behavior has not been verified.
+
CrossLinked
diff --git a/crosslinked/__init__.py b/crosslinked/__init__.py
index d75086d..716eba3 100644
--- a/crosslinked/__init__.py
+++ b/crosslinked/__init__.py
@@ -3,18 +3,21 @@
# License: GPLv3
import re
import argparse
+import math
+import os
from sys import exit
from csv import reader
from crosslinked import utils
from crosslinked.logger import *
from crosslinked.search import CrossLinked
+from crosslinked.searxng import SearXNG, search_endpoint
def banner():
VERSION = 'v0.3.0'
- print('''
+ print(r'''
_____ _ _ _
/ __ \ | | ({}) | | | |
| / \/_ __ ___ ___ ___ | | _ _ __ | | _____ __| |
@@ -36,16 +39,32 @@ def cli():
s = args.add_argument_group("Search arguments")
s.add_argument('--search', dest='engine', default='google,bing', type=lambda x: utils.delimiter2list(x), help='Search Engine (Default=\'google,bing\')')
+ s.add_argument('--searxng-url', default=os.environ.get('SEARXNG_URL'), help='SearXNG instance URL (or SEARXNG_URL environment variable)')
+ s.add_argument('--max-pages', type=int, default=5, help='Maximum SearXNG pages (Default=5)')
o = args.add_argument_group("Output arguments")
- o.add_argument('-f', dest='nformat', type=str, required=True, help='Format names, ex: \'domain\{f}{last}\', \'{first}.{last}@domain.com\'')
+ o.add_argument('-f', dest='nformat', type=str, required=True, help="Format names, ex: 'domain\\{f}{last}', '{first}.{last}@domain.com'")
o.add_argument('-o', dest='outfile', type=str, default='names', help='Change name of output file (omit_extension)')
p = args.add_argument_group("Proxy arguments")
pr = p.add_mutually_exclusive_group(required=False)
pr.add_argument('--proxy', dest='proxy', action='append', default=[], help='Proxy requests (IP:Port)')
pr.add_argument('--proxy-file', dest='proxy', default=False, type=lambda x: utils.file_exists(x), help='Load proxies from file for rotation')
- return args.parse_args()
+ parsed = args.parse_args()
+ if not parsed.company_name:
+ args.error('company_name is required')
+ if any(engine not in ('google', 'bing', 'searxng') for engine in parsed.engine) or not parsed.engine:
+ args.error('--search supports google, bing, searxng')
+ if not math.isfinite(parsed.timeout) or parsed.timeout <= 0 or not math.isfinite(parsed.jitter) or parsed.jitter < 0 or parsed.max_pages < 1:
+ args.error('timeout and max-pages must be positive; jitter must be nonnegative')
+ if 'searxng' in parsed.engine and not parsed.company_name.endswith('.csv'):
+ if not parsed.searxng_url:
+ args.error('--search searxng requires --searxng-url or SEARXNG_URL')
+ try:
+ search_endpoint(parsed.searxng_url)
+ except ValueError as exc:
+ args.error(str(exc))
+ return parsed
def start_scrape(args):
@@ -53,6 +72,13 @@ def start_scrape(args):
Log.info("Searching {} for valid employee names at \"{}\"".format(', '.join(args.engine), args.company_name))
for search_engine in args.engine:
+ if search_engine == 'searxng':
+ c = SearXNG(args.company_name, args.searxng_url, args.timeout,
+ args.jitter, args.max_pages, args.proxy)
+ tmp += c.search()
+ if c.status in ('failed', 'partial'):
+ args.search_failed = True
+ continue
c = CrossLinked(search_engine, args.company_name, args.timeout, 3, args.proxy, args.jitter)
if search_engine in c.url.keys():
tmp += c.search()
@@ -118,7 +144,13 @@ def main():
csv = setup_file_logger(args.outfile+".csv", log_name="cLinked_csv", file_mode='a') # names.csv appended
data = start_parse(args) if args.company_name.endswith('.csv') else start_scrape(args)
- format_names(args, data, txt) if len(data) > 0 else Log.warn('No results found')
+ if data:
+ format_names(args, data, txt)
+ elif not getattr(args, 'search_failed', False):
+ Log.warn('No results found')
+ if getattr(args, 'search_failed', False):
+ Log.warn('Search incomplete; any saved results may be partial')
+ exit(1)
except KeyboardInterrupt:
Log.warn("Key event detected, closing...")
exit(0)
diff --git a/crosslinked/search.py b/crosslinked/search.py
index 3ebca2b..baf5fc0 100644
--- a/crosslinked/search.py
+++ b/crosslinked/search.py
@@ -1,4 +1,6 @@
import logging
+import csv as csv_module
+import io
import requests
import threading
from time import sleep
@@ -127,7 +129,9 @@ def log_results(self, d):
self.results.append(d)
# Search results are logged to names.csv but names.txt is not generated until end to prevent duplicates
logging.debug('name: {:25} RawTxt: {}'.format(d['name'], d['text']))
- csv.info('"{}","{}","{}","{}","{}","{}",'.format(self.runtime, self.search_engine, d['name'], d['title'], d['url'], d['text']))
+ row = io.StringIO()
+ csv_module.writer(row).writerow([self.runtime, self.search_engine, d['name'], d['title'], d['url'], d['text']])
+ csv.info(row.getvalue().rstrip('\r\n'))
def get_statuscode(resp):
diff --git a/crosslinked/searxng.py b/crosslinked/searxng.py
new file mode 100644
index 0000000..b009c66
--- /dev/null
+++ b/crosslinked/searxng.py
@@ -0,0 +1,111 @@
+"""SearXNG JSON provider. No LinkedIn requests or browser required."""
+import re
+from time import monotonic, sleep
+from urllib.parse import urlsplit, urlunsplit
+
+import requests
+from bs4 import BeautifulSoup
+from unidecode import unidecode
+
+from crosslinked.search import CrossLinked, get_proxy
+from crosslinked.logger import Log
+
+
+def search_endpoint(value):
+ parts = urlsplit(value)
+ if (parts.scheme not in ('http', 'https') or not parts.hostname
+ or parts.username or parts.password or parts.query or parts.fragment):
+ raise ValueError('SearXNG URL must be an HTTP(S) URL without credentials, query or fragment')
+ path = parts.path.rstrip('/')
+ if not path.endswith('/search'):
+ path += '/search'
+ return urlunsplit((parts.scheme, parts.netloc, path, '', ''))
+
+
+class SearXNG(CrossLinked):
+ def __init__(self, target, endpoint, timeout=15, jitter=1, max_pages=5, proxies=None):
+ super().__init__('searxng', target, timeout, proxies=proxies or [], jitter=jitter)
+ self.endpoint = search_endpoint(endpoint)
+ self.max_pages = max_pages
+ self.status = 'pending'
+ self.warnings = []
+
+ def warn(self, message):
+ self.warnings.append(message)
+ Log.warn(message)
+
+ def parse_result(self, item):
+ if not isinstance(item, dict):
+ return None
+ url, title = item.get('url'), item.get('title')
+ if not isinstance(url, str) or not isinstance(title, str):
+ return None
+ try:
+ parts = urlsplit(url)
+ host = (parts.hostname or '').lower()
+ except ValueError:
+ return None
+ if (parts.scheme not in ('http', 'https') or parts.username or parts.password
+ or not (host == 'linkedin.com' or host.endswith('.linkedin.com'))
+ or not re.fullmatch(r'/in/[^/]+/?', parts.path)):
+ return None
+ text = BeautifulSoup(title, 'html.parser').get_text(' ', strip=True)
+ text = re.sub(r'\s+', ' ', text).strip()
+ fields = re.split(r'\s+[-–—|]\s+', text)
+ name = fields[0].strip()
+ if len(name.split()) < 2 or 'linkedin' in name.lower() or not any(c.isalpha() for c in name):
+ return None
+ return {'name': unidecode(name).lower(),
+ 'title': fields[1] if len(fields) > 1 and fields[1].lower() != 'linkedin' else 'N/A',
+ 'url': 'https://www.linkedin.com' + parts.path.rstrip('/'),
+ 'text': text}
+
+ def search(self):
+ deadline = monotonic() + self.timeout
+ seen_urls, seen_pages = set(), set()
+ self.status = 'complete'
+ with requests.Session() as session:
+ for page in range(1, self.max_pages + 1):
+ remaining = deadline - monotonic()
+ if remaining <= 0:
+ self.status = 'limited'
+ break
+ try:
+ response = session.get(self.endpoint, params={
+ 'q': 'site:linkedin.com/in "{}"'.format(self.target.replace('"', ' ')),
+ 'format': 'json', 'pageno': page, 'categories': 'general'},
+ timeout=min(remaining, 15), proxies=get_proxy(self.proxies))
+ if response.status_code == 403:
+ raise ValueError('HTTP 403: enable JSON in SearXNG search.formats and check instance access')
+ response.raise_for_status()
+ payload = response.json()
+ if not isinstance(payload, dict) or not isinstance(payload.get('results'), list):
+ raise ValueError('Invalid SearXNG JSON: expected a results list')
+ except (requests.RequestException, ValueError) as exc:
+ self.status = 'failed'
+ self.warn('SearXNG request failed: {}'.format(exc))
+ break
+ if payload.get('unresponsive_engines'):
+ self.status = 'partial'
+ self.warn('SearXNG upstream failures: {}'.format(payload['unresponsive_engines']))
+ items = payload['results']
+ if not items:
+ break
+ fingerprint = repr(items)
+ if fingerprint in seen_pages:
+ self.status = 'limited' if self.status == 'complete' else self.status
+ self.warn('SearXNG repeated a page; stopping pagination')
+ break
+ seen_pages.add(fingerprint)
+ for item in items:
+ record = self.parse_result(item)
+ if record and record['url'] not in seen_urls:
+ seen_urls.add(record['url'])
+ self.log_results(record)
+ Log.info('SearXNG page {}: {} unique profiles'.format(page, len(self.results)))
+ if page == self.max_pages:
+ self.status = 'limited' if self.status == 'complete' else self.status
+ else:
+ sleep(min(self.jitter, max(0, deadline - monotonic())))
+ Log.info('SearXNG status: {}'.format(self.status))
+ return self.results
diff --git a/docs/searxng.md b/docs/searxng.md
new file mode 100644
index 0000000..088eeb2
--- /dev/null
+++ b/docs/searxng.md
@@ -0,0 +1,227 @@
+# SearXNG provider
+
+Use a SearXNG instance you operate or are permitted to query. No public instance
+is selected automatically. SearXNG depends on upstream search providers; this
+integration does not contact LinkedIn directly.
+
+Merge this into the instance's `settings.yml`, preserving other settings, then restart:
+
+```yaml
+search:
+ formats:
+ - html
+ - json
+```
+
+Configure upstream engines in SearXNG itself. Official documentation:
+[installation](https://docs.searxng.org/admin/installation.html),
+[JSON search API](https://docs.searxng.org/dev/search_api.html).
+
+## Kali WSL: setup on a new Windows device
+
+These steps use Docker Desktop's Linux engine, with CrossLinked running inside
+Kali WSL. Docker Desktop hosts SearXNG; activating the Python environment does
+not start Docker.
+
+1. Install Docker Desktop and Kali under WSL 2.
+2. Open Docker Desktop and enable its WSL 2 engine. Under **Settings → Resources
+ → WSL Integration**, enable Kali, then apply the changes.
+3. Enable **Start Docker Desktop when you sign in** in Docker Desktop settings
+ if you want the service available after Windows login.
+4. Open Kali and check `sudo docker info`. If it cannot connect, start Docker
+ Desktop and check Kali integration before continuing.
+
+See [Docker Desktop for Windows](https://docs.docker.com/desktop/setup/install/windows-install/)
+and [Docker WSL integration](https://docs.docker.com/desktop/features/wsl/).
+
+### Create SearXNG once
+
+Run these commands in Kali. The configuration command is for a **new device**;
+do not overwrite an existing `settings.yml` containing your custom settings.
+
+```bash
+mkdir -p ~/searxng/config
+cd ~/searxng
+
+cat > config/settings.yml <&1 \
+ | grep -iE 'duckduckgo|403|429|captcha|denied'
+```
+
+## Run from this repository (PowerShell alternative)
+
+```powershell
+python -m venv venv
+./venv/Scripts/python -m pip install requests beautifulsoup4 lxml Unidecode
+./venv/Scripts/python crosslinked.py 'Example Company' --search searxng --searxng-url http://localhost:8080 -t 60 --max-pages 5 -f '{first}.{last}@example.com'
+```
+
+Alternatively set `$env:SEARXNG_URL = 'http://localhost:8080'`. URLs can include a
+deployment subpath or end in `/search`; credentials, queries and fragments are
+rejected. HTTPS verifies certificates. Existing proxy flags are supported.
+
+Output remains `names.txt` and `names.csv`. CSV records the search-result title
+and normalized profile URL. Names and job titles are inferred from titles;
+employment and generated email addresses are not verified. The first record for
+a profile wins. Regional LinkedIn URLs and tracking parameters are deduplicated.
+
+## Status
+
+- `complete`: reached an empty page; does not establish exhaustive coverage.
+- `limited`: page/time budget reached or a repeated page detected.
+- `partial`: SearXNG reported upstream engine failures.
+- `failed`: HTTP/network error, invalid JSON or an unexpected response schema.
+
+Partial/failed searches retain collected records and exit with code 1. HTTP 403
+includes JSON configuration guidance. Requests use the remaining search budget,
+capped at 15 seconds per request. These are socket inactivity timeouts, not a
+strict wall-clock deadline. Blocked requests are not retried automatically.
+
+## Verification
+
+```powershell
+./venv/Scripts/python -m unittest discover -s tests -v
+```
+
+Tests simulate responses. Live coverage must be benchmarked with a configured
+instance. This addition does not repair legacy Google/Bing scraping or implement
+browser automation, YaCy, or the planned Pi extension.
diff --git a/tests/test_cli_integration.py b/tests/test_cli_integration.py
new file mode 100644
index 0000000..4de6e56
--- /dev/null
+++ b/tests/test_cli_integration.py
@@ -0,0 +1,64 @@
+"""Exercise actual HTTP, CLI exit codes, and output files against a local fixture."""
+import csv
+import json
+from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
+from pathlib import Path
+import subprocess
+import sys
+import tempfile
+from threading import Thread
+import unittest
+from urllib.parse import parse_qs, urlsplit
+
+
+class CliIntegrationTests(unittest.TestCase):
+ def test_http_to_files_and_failure_exit(self):
+ pages = []
+
+ class Handler(BaseHTTPRequestHandler):
+ def do_GET(self):
+ params = parse_qs(urlsplit(self.path).query)
+ pages.append(params)
+ if 'Failure' in params['q'][0]:
+ self.send_response(403)
+ self.end_headers()
+ return
+ payload = {'results': []}
+ if params['pageno'] == ['1']:
+ payload['results'] = [{'url': 'https://www.linkedin.com/in/jane-example',
+ 'title': 'Jane Example - Engineer | LinkedIn'}]
+ body = json.dumps(payload).encode()
+ self.send_response(200)
+ self.send_header('Content-Type', 'application/json')
+ self.send_header('Content-Length', str(len(body)))
+ self.end_headers()
+ self.wfile.write(body)
+
+ def log_message(self, *args):
+ pass
+
+ server = ThreadingHTTPServer(('127.0.0.1', 0), Handler)
+ thread = Thread(target=server.serve_forever, daemon=True)
+ thread.start()
+ try:
+ with tempfile.TemporaryDirectory() as directory:
+ command = [sys.executable, str(Path(__file__).resolve().parents[1] / 'crosslinked.py'),
+ '--search', 'searxng', '--searxng-url',
+ 'http://127.0.0.1:{}'.format(server.server_port),
+ '-j', '0', '-f', '{first}.{last}@example.com']
+ run = subprocess.run(command + ['Example'], cwd=directory, capture_output=True, text=True, timeout=15)
+ self.assertEqual(run.returncode, 0, run.stdout + run.stderr)
+ self.assertEqual(Path(directory, 'names.txt').read_text().strip(), 'jane.example@example.com')
+ with Path(directory, 'names.csv').open(newline='') as handle:
+ rows = list(csv.DictReader(handle))
+ self.assertEqual(rows[0]['Search'], 'searxng')
+ self.assertEqual(rows[0]['Name'], 'jane example')
+ self.assertEqual([p['pageno'] for p in pages], [['1'], ['2']])
+ run = subprocess.run(command + ['Failure'], cwd=directory, capture_output=True, text=True, timeout=15)
+ self.assertEqual(run.returncode, 1, run.stdout + run.stderr)
+ self.assertIn('HTTP 403', run.stdout)
+ self.assertNotIn('No results found', run.stdout)
+ finally:
+ server.shutdown()
+ server.server_close()
+ thread.join()
diff --git a/tests/test_searxng.py b/tests/test_searxng.py
new file mode 100644
index 0000000..ece22cf
--- /dev/null
+++ b/tests/test_searxng.py
@@ -0,0 +1,120 @@
+import csv
+import io
+import unittest
+from unittest.mock import Mock, patch
+import requests
+from crosslinked import cli
+from crosslinked.searxng import SearXNG, search_endpoint
+
+
+def result(title='Jane Doe - Engineer | LinkedIn'):
+ return {'url': 'https://www.linkedin.com/in/jane-doe', 'title': title}
+
+
+def response(payload=None, status=200):
+ value = Mock(status_code=status)
+ value.json.return_value = payload
+ if status >= 400:
+ value.raise_for_status.side_effect = requests.HTTPError(str(status))
+ return value
+
+
+class ProviderTests(unittest.TestCase):
+ def provider(self, **kwargs):
+ return SearXNG('Example & Co', 'http://localhost:8080', jitter=0, **kwargs)
+
+ def run_pages(self, provider, pages):
+ with patch('crosslinked.searxng.requests.Session') as session:
+ get = session.return_value.__enter__.return_value.get
+ get.side_effect = pages
+ return provider.search(), get.call_args_list
+
+ def test_endpoint(self):
+ for value in ('http://localhost:8080', 'http://localhost:8080/search/'):
+ self.assertEqual(search_endpoint(value), 'http://localhost:8080/search')
+ self.assertEqual(search_endpoint('https://example.org/prefix'), 'https://example.org/prefix/search')
+ for value in ('file:///tmp', 'https://user:pass@example.org', 'https://example.org?q=x'):
+ with self.assertRaises(ValueError):
+ search_endpoint(value)
+
+ def test_pagination_independent_of_extracted_count(self):
+ provider = self.provider()
+ records, calls = self.run_pages(provider, [
+ response({'results': [{'url': 'https://example.org', 'title': 'Unrelated'}]}),
+ response({'results': [result()]}), response({'results': []})])
+ self.assertEqual(len(records), 1)
+ self.assertEqual([c.kwargs['params']['pageno'] for c in calls], [1, 2, 3])
+ self.assertEqual(calls[0].kwargs['params']['q'], 'site:linkedin.com/in "Example & Co"')
+ self.assertNotIn('verify', calls[0].kwargs)
+ self.assertEqual(provider.status, 'complete')
+
+ def test_dedup_and_repeated_page(self):
+ provider = self.provider()
+ duplicate = result()
+ duplicate['url'] = 'https://au.linkedin.com/in/jane-doe/?trk=search'
+ payload = {'results': [result(), duplicate]}
+ records, calls = self.run_pages(provider, [response(payload), response(payload)])
+ self.assertEqual(len(records), 1)
+ self.assertEqual(len(calls), 2)
+ self.assertEqual(provider.status, 'limited')
+
+ def test_reject_nonprofiles(self):
+ provider = self.provider()
+ for url in ('https://notlinkedin.com/in/jane', 'https://linkedin.com.evil.org/in/jane',
+ 'https://linkedin.com/company/example', 'https://linkedin.com/in/',
+ 'https://linkedin.com/in/jane/details', 'javascript:alert(1)'):
+ self.assertIsNone(provider.parse_result({'url': url, 'title': 'Jane Doe'}))
+ for item in (None, {}, {'url': [], 'title': 2}):
+ self.assertIsNone(provider.parse_result(item))
+
+ def test_hyphenated_name(self):
+ record = self.provider().parse_result(result('Anne-Marie Dupont – Senior Engineer | LinkedIn'))
+ self.assertEqual(record['name'], 'anne-marie dupont')
+ self.assertEqual(record['title'], 'Senior Engineer')
+
+ def test_failures_not_empty_success(self):
+ invalid = response()
+ invalid.json.side_effect = ValueError('Not JSON')
+ for page in (response(status=403), response(status=429), invalid,
+ response({'unexpected': []}), requests.Timeout('timeout')):
+ provider = self.provider()
+ records, _ = self.run_pages(provider, [page])
+ self.assertEqual(records, [])
+ self.assertEqual(provider.status, 'failed')
+
+ def test_partial_results_survive_failure(self):
+ provider = self.provider()
+ records, _ = self.run_pages(provider, [response({'results': [result()]}), response(status=503)])
+ self.assertEqual(len(records), 1)
+ self.assertEqual(provider.status, 'failed')
+
+ def test_upstream_failure(self):
+ provider = self.provider()
+ self.run_pages(provider, [response({'results': [], 'unresponsive_engines': [['bing', 'timeout']]})])
+ self.assertEqual(provider.status, 'partial')
+
+ def test_page_cap(self):
+ provider = self.provider(max_pages=1)
+ _, calls = self.run_pages(provider, [response({'results': [result()]})])
+ self.assertEqual(len(calls), 1)
+ self.assertEqual(provider.status, 'limited')
+
+ def test_csv_quotes(self):
+ provider = self.provider()
+ record = provider.parse_result(result('Jane Doe - Engineer, "Platform" | LinkedIn'))
+ with patch('crosslinked.search.csv.info') as log:
+ provider.log_results(record)
+ row = next(csv.reader(io.StringIO(log.call_args.args[0])))
+ self.assertEqual(len(row), 6)
+ self.assertEqual(row[3], 'Engineer, "Platform"')
+
+ def test_cli_validation(self):
+ for flags in (['--search', 'unknown'], ['--search', 'searxng'], ['-t', 'nan'], ['-j', '-1']):
+ with patch.dict('os.environ', {}, clear=True), patch('sys.argv', ['crosslinked', 'Example', '-f', '{first}.{last}'] + flags):
+ with self.assertRaises(SystemExit) as exc:
+ cli()
+ self.assertEqual(exc.exception.code, 2)
+
+
+if __name__ == '__main__':
+ unittest.main()