Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,10 @@
## Fork development: SearXNG support

This fork adds `--search searxng` with JSON results, bounded pagination, profile
URL deduplication and explicit failure reporting. See the
[SearXNG setup and usage guide](docs/searxng.md). Existing Google/Bing defaults
remain available; their current scraping behavior has not been verified.

<div align="center">
<h1>CrossLinked</h1>
</div>
Expand Down
40 changes: 36 additions & 4 deletions crosslinked/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,18 +3,21 @@
# License: GPLv3
import re
import argparse
import math
import os
from sys import exit
from csv import reader
from crosslinked import utils
from crosslinked.logger import *
from crosslinked.search import CrossLinked
from crosslinked.searxng import SearXNG, search_endpoint


def banner():

VERSION = 'v0.3.0'

print('''
print(r'''
_____ _ _ _
/ __ \ | | ({}) | | | |
| / \/_ __ ___ ___ ___ | | _ _ __ | | _____ __| |
Expand All @@ -36,23 +39,46 @@ def cli():

s = args.add_argument_group("Search arguments")
s.add_argument('--search', dest='engine', default='google,bing', type=lambda x: utils.delimiter2list(x), help='Search Engine (Default=\'google,bing\')')
s.add_argument('--searxng-url', default=os.environ.get('SEARXNG_URL'), help='SearXNG instance URL (or SEARXNG_URL environment variable)')
s.add_argument('--max-pages', type=int, default=5, help='Maximum SearXNG pages (Default=5)')

o = args.add_argument_group("Output arguments")
o.add_argument('-f', dest='nformat', type=str, required=True, help='Format names, ex: \'domain\{f}{last}\', \'{first}.{last}@domain.com\'')
o.add_argument('-f', dest='nformat', type=str, required=True, help="Format names, ex: 'domain\\{f}{last}', '{first}.{last}@domain.com'")
o.add_argument('-o', dest='outfile', type=str, default='names', help='Change name of output file (omit_extension)')

p = args.add_argument_group("Proxy arguments")
pr = p.add_mutually_exclusive_group(required=False)
pr.add_argument('--proxy', dest='proxy', action='append', default=[], help='Proxy requests (IP:Port)')
pr.add_argument('--proxy-file', dest='proxy', default=False, type=lambda x: utils.file_exists(x), help='Load proxies from file for rotation')
return args.parse_args()
parsed = args.parse_args()
if not parsed.company_name:
args.error('company_name is required')
if any(engine not in ('google', 'bing', 'searxng') for engine in parsed.engine) or not parsed.engine:
args.error('--search supports google, bing, searxng')
if not math.isfinite(parsed.timeout) or parsed.timeout <= 0 or not math.isfinite(parsed.jitter) or parsed.jitter < 0 or parsed.max_pages < 1:
args.error('timeout and max-pages must be positive; jitter must be nonnegative')
if 'searxng' in parsed.engine and not parsed.company_name.endswith('.csv'):
if not parsed.searxng_url:
args.error('--search searxng requires --searxng-url or SEARXNG_URL')
try:
search_endpoint(parsed.searxng_url)
except ValueError as exc:
args.error(str(exc))
return parsed


def start_scrape(args):
tmp = []
Log.info("Searching {} for valid employee names at \"{}\"".format(', '.join(args.engine), args.company_name))

for search_engine in args.engine:
if search_engine == 'searxng':
c = SearXNG(args.company_name, args.searxng_url, args.timeout,
args.jitter, args.max_pages, args.proxy)
tmp += c.search()
if c.status in ('failed', 'partial'):
args.search_failed = True
continue
c = CrossLinked(search_engine, args.company_name, args.timeout, 3, args.proxy, args.jitter)
if search_engine in c.url.keys():
tmp += c.search()
Expand Down Expand Up @@ -118,7 +144,13 @@ def main():
csv = setup_file_logger(args.outfile+".csv", log_name="cLinked_csv", file_mode='a') # names.csv appended

data = start_parse(args) if args.company_name.endswith('.csv') else start_scrape(args)
format_names(args, data, txt) if len(data) > 0 else Log.warn('No results found')
if data:
format_names(args, data, txt)
elif not getattr(args, 'search_failed', False):
Log.warn('No results found')
if getattr(args, 'search_failed', False):
Log.warn('Search incomplete; any saved results may be partial')
exit(1)
except KeyboardInterrupt:
Log.warn("Key event detected, closing...")
exit(0)
Expand Down
6 changes: 5 additions & 1 deletion crosslinked/search.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,6 @@
import logging
import csv as csv_module
import io
import requests
import threading
from time import sleep
Expand Down Expand Up @@ -127,7 +129,9 @@ def log_results(self, d):
self.results.append(d)
# Search results are logged to names.csv but names.txt is not generated until end to prevent duplicates
logging.debug('name: {:25} RawTxt: {}'.format(d['name'], d['text']))
csv.info('"{}","{}","{}","{}","{}","{}",'.format(self.runtime, self.search_engine, d['name'], d['title'], d['url'], d['text']))
row = io.StringIO()
csv_module.writer(row).writerow([self.runtime, self.search_engine, d['name'], d['title'], d['url'], d['text']])
csv.info(row.getvalue().rstrip('\r\n'))


def get_statuscode(resp):
Expand Down
111 changes: 111 additions & 0 deletions crosslinked/searxng.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,111 @@
"""SearXNG JSON provider. No LinkedIn requests or browser required."""
import re
from time import monotonic, sleep
from urllib.parse import urlsplit, urlunsplit

import requests
from bs4 import BeautifulSoup
from unidecode import unidecode

from crosslinked.search import CrossLinked, get_proxy
from crosslinked.logger import Log


def search_endpoint(value):
parts = urlsplit(value)
if (parts.scheme not in ('http', 'https') or not parts.hostname
or parts.username or parts.password or parts.query or parts.fragment):
raise ValueError('SearXNG URL must be an HTTP(S) URL without credentials, query or fragment')
path = parts.path.rstrip('/')
if not path.endswith('/search'):
path += '/search'
return urlunsplit((parts.scheme, parts.netloc, path, '', ''))


class SearXNG(CrossLinked):
def __init__(self, target, endpoint, timeout=15, jitter=1, max_pages=5, proxies=None):
super().__init__('searxng', target, timeout, proxies=proxies or [], jitter=jitter)
self.endpoint = search_endpoint(endpoint)
self.max_pages = max_pages
self.status = 'pending'
self.warnings = []

def warn(self, message):
self.warnings.append(message)
Log.warn(message)

def parse_result(self, item):
if not isinstance(item, dict):
return None
url, title = item.get('url'), item.get('title')
if not isinstance(url, str) or not isinstance(title, str):
return None
try:
parts = urlsplit(url)
host = (parts.hostname or '').lower()
except ValueError:
return None
if (parts.scheme not in ('http', 'https') or parts.username or parts.password
or not (host == 'linkedin.com' or host.endswith('.linkedin.com'))
or not re.fullmatch(r'/in/[^/]+/?', parts.path)):
return None
text = BeautifulSoup(title, 'html.parser').get_text(' ', strip=True)
text = re.sub(r'\s+', ' ', text).strip()
fields = re.split(r'\s+[-–—|]\s+', text)
name = fields[0].strip()
if len(name.split()) < 2 or 'linkedin' in name.lower() or not any(c.isalpha() for c in name):
return None
return {'name': unidecode(name).lower(),
'title': fields[1] if len(fields) > 1 and fields[1].lower() != 'linkedin' else 'N/A',
'url': 'https://www.linkedin.com' + parts.path.rstrip('/'),
'text': text}

def search(self):
deadline = monotonic() + self.timeout
seen_urls, seen_pages = set(), set()
self.status = 'complete'
with requests.Session() as session:
for page in range(1, self.max_pages + 1):
remaining = deadline - monotonic()
if remaining <= 0:
self.status = 'limited'
break
try:
response = session.get(self.endpoint, params={
'q': 'site:linkedin.com/in "{}"'.format(self.target.replace('"', ' ')),
'format': 'json', 'pageno': page, 'categories': 'general'},
timeout=min(remaining, 15), proxies=get_proxy(self.proxies))
if response.status_code == 403:
raise ValueError('HTTP 403: enable JSON in SearXNG search.formats and check instance access')
response.raise_for_status()
payload = response.json()
if not isinstance(payload, dict) or not isinstance(payload.get('results'), list):
raise ValueError('Invalid SearXNG JSON: expected a results list')
except (requests.RequestException, ValueError) as exc:
self.status = 'failed'
self.warn('SearXNG request failed: {}'.format(exc))
break
if payload.get('unresponsive_engines'):
self.status = 'partial'
self.warn('SearXNG upstream failures: {}'.format(payload['unresponsive_engines']))
items = payload['results']
if not items:
break
fingerprint = repr(items)
if fingerprint in seen_pages:
self.status = 'limited' if self.status == 'complete' else self.status
self.warn('SearXNG repeated a page; stopping pagination')
break
seen_pages.add(fingerprint)
for item in items:
record = self.parse_result(item)
if record and record['url'] not in seen_urls:
seen_urls.add(record['url'])
self.log_results(record)
Log.info('SearXNG page {}: {} unique profiles'.format(page, len(self.results)))
if page == self.max_pages:
self.status = 'limited' if self.status == 'complete' else self.status
else:
sleep(min(self.jitter, max(0, deadline - monotonic())))
Log.info('SearXNG status: {}'.format(self.status))
return self.results
Loading