Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 14 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -217,6 +217,7 @@ OUTPUT:

CONFIGURATIONS:
-config string path to the httpx configuration file (default $HOME/.config/httpx/config.yaml)
-ptm, -page-type-model string path to a local dit model for page classification (requires -kb or -fpt; skips model download)
-r, -resolvers string[] list of custom resolver (file or comma separated)
-allow string[] allowed list of IP/CIDR's to process (file or comma separated)
-deny string[] denied list of IP/CIDR's to process (file or comma separated)
Expand Down Expand Up @@ -286,6 +287,19 @@ For details about running httpx, see https://docs.projectdiscovery.io/tools/http
### Using `httpx` as a library
`httpx` can be used as a library by creating an instance of the `Option` struct and populating it with the same options that would be specified via CLI. Once validated, the struct should be passed to a runner instance (to be closed at the end of the program) and the `RunEnumeration` method should be called. A minimal example of how to do it is in the [examples](examples/) folder.

### Using a local page classification model

For environments that cannot reach Hugging Face, provision the [dit model.json](https://huggingface.co/datasets/happyhackingspace/dit/resolve/main/model.json) on an accessible machine and copy it to the scanning environment. Select it with `-page-type-model` (`-ptm`):

```bash
httpx -l hosts.txt -kb -json -ptm /opt/httpx/model.json
httpx -l hosts.txt -fpt error,parked -ptm /opt/httpx/model.json
```

The path selects a local model and bypasses automatic model discovery and downloading. A missing or unreadable model, or invalid model JSON, stops initialization instead of falling back to a download. Classification remains opt-in: use `-kb` or `-fpt` (the deprecated `-fep` is also supported). Setting `-ptm` alone does not load the model or enable classification. Library users can set `Options.PageTypeModel` alongside `KnowledgeBase` or `OutputFilterPageType`. Without a custom path, the existing model discovery and download behavior is unchanged.

Explicit models must also pass a page and form classification check before scanning starts or existing output indexes are changed. This checks that the model can run, rather than certifying every part of its contents. A later classification error leaves that response without page-type information; diagnostics are available with `-debug`.

## Common Recipes

Below are practical one-liners for common use cases leveraging httpx's composable primitives. These recipes are validated in `runner/wellknown_recipes_test.go`.
Expand Down
50 changes: 50 additions & 0 deletions runner/classifier.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
package runner

import (
"fmt"

"github.com/happyhackingspace/dit"
)

// loadPageTypeModel checks that an explicit model can classify a page and form
// before runner initialization changes output files or creates resources. dit
// does not validate model structure and may panic during loading or inference.
func loadPageTypeModel(path string) (model *dit.Classifier, err error) {
defer func() {
if failure := recover(); failure != nil {
model = nil
err = fmt.Errorf("invalid page classification model: %v", failure)
}
}()
model, err = dit.Load(path)
if err != nil {
return nil, err
}
// Two fields exercise the optional field classifier's sequence transitions.
const probe = `<html><head><title>Example</title></head><body><form action="/login"><input type="text" name="username"><input type="password" name="password"><button type="submit">Login</button></form></body></html>`
result, err := extractPageType(model, probe)
if err != nil {
return nil, fmt.Errorf("invalid page classification model: %w", err)
}
if result == nil || result.Type == "" || len(result.Forms) != 1 || result.Forms[0].Type == "" {
return nil, fmt.Errorf("invalid page classification model: missing page or form prediction")
}
for _, label := range result.Forms[0].Fields {
if label == "" {
return nil, fmt.Errorf("invalid page classification model: missing field prediction")
}
}
return model, nil
}

// extractPageType contains dependency panics at the same boundary as inference
// errors. A startup probe cannot exercise every input-dependent model feature.
func extractPageType(model *dit.Classifier, html string) (result *dit.PageResult, err error) {
defer func() {
if failure := recover(); failure != nil {
result = nil
err = fmt.Errorf("page classification failed: %v", failure)
}
}()
return model.ExtractPageType(html)
}
228 changes: 228 additions & 0 deletions runner/classifier_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,228 @@
package runner

import (
"encoding/json"
"os"
"path/filepath"
"strings"
"testing"

"github.com/PuerkitoBio/goquery"
"github.com/happyhackingspace/dit"
"github.com/happyhackingspace/dit/classifier"
"github.com/happyhackingspace/dit/crf"
"github.com/stretchr/testify/require"
)

// localPageModel creates a small, valid model without downloading training data
// or the production model. One class makes the expected prediction deterministic.
func localPageModel(t *testing.T, path string) {
t.Helper()
doc, err := goquery.NewDocumentFromReader(strings.NewReader("<html><body><form><input name='username'></form>Example page</body></html>"))
require.NoError(t, err)
config := classifier.DefaultPageTypeTrainConfig()
config.MaxIter = 1
formConfig := classifier.DefaultFormTypeTrainConfig()
formConfig.MaxIter = 1
model := &classifier.FormFieldClassifier{
FormModel: classifier.TrainFormType([]*goquery.Selection{doc.Find("form")}, []string{"login"}, formConfig),
PageModel: classifier.TrainPageType(
[]*goquery.Document{doc}, [][]classifier.ClassifyResult{nil},
[]string{"https://example.com"}, []string{"error"}, config,
),
}
require.NoError(t, model.SaveModel(path))
}

func TestLocalPageTypeModel(t *testing.T) {
modelPath := filepath.Join(t.TempDir(), "custom model.json")
localPageModel(t, modelPath)

// A broken auto-discovered model must not override the explicit model path.
t.Chdir(t.TempDir())
require.NoError(t, os.WriteFile("model.json", []byte("invalid JSON"), 0600))

for _, tc := range []struct {
name string
options Options
}{
{name: "knowledge base", options: Options{KnowledgeBase: true}},
{name: "page type filter", options: Options{OutputFilterPageType: []string{"error"}}},
{name: "deprecated error page filter", options: Options{OutputFilterErrorPage: true}},
} {
t.Run(tc.name, func(t *testing.T) {
options := tc.options
options.PageTypeModel = modelPath
r, err := New(&options)
require.NoError(t, err)
t.Cleanup(r.Close)
kb := r.classifyPage("", "<html><body><form><input name='username'></form></body></html>", 0)
require.Equal(t, "error", kb["PageType"])
require.NotEmpty(t, kb["Forms"])
})
}
}

func TestLocalPageTypeModelErrors(t *testing.T) {
// A valid auto-discovered model must not hide an invalid explicit path.
t.Chdir(t.TempDir())
localPageModel(t, "model.json")
require.NoError(t, os.WriteFile("invalid.json", []byte("invalid JSON"), 0600))

for _, path := range []string{"missing.json", "invalid.json"} {
t.Run(path, func(t *testing.T) {
r, err := New(&Options{KnowledgeBase: true, PageTypeModel: path})
require.ErrorContains(t, err, "could not initialize page classifier")
require.Nil(t, r)
})
}
}

func TestPageTypeModelDoesNotEnableClassification(t *testing.T) {
r, err := New(&Options{PageTypeModel: filepath.Join(t.TempDir(), "missing.json")})
require.NoError(t, err)
t.Cleanup(r.Close)
require.Nil(t, r.ditClassifier)
require.NotContains(t, r.classifyPage("", "<html></html>", 0), "PageType")
}

func TestLocalPageTypeModelErrorsPreserveResponseIndexes(t *testing.T) {
for _, name := range []string{"missing", "invalid JSON", "{}", "null", `{"form_model":{},"page_model":{}}`} {
t.Run(name, func(t *testing.T) {
dir := t.TempDir()
modelPath := filepath.Join(dir, "model.json")
if name != "missing" {
require.NoError(t, os.WriteFile(modelPath, []byte(name), 0600))
}
indexes := []string{
filepath.Join(dir, "response", "index.txt"),
filepath.Join(dir, "screenshot", "index_screenshot.txt"),
}
for _, path := range indexes {
require.NoError(t, os.MkdirAll(filepath.Dir(path), 0700))
require.NoError(t, os.WriteFile(path, []byte("previous scan results"), 0600))
}

r, err := New(&Options{KnowledgeBase: true, PageTypeModel: modelPath, StoreResponseDir: dir})
require.ErrorContains(t, err, "could not initialize page classifier")
require.Nil(t, r)
for _, path := range indexes {
data, err := os.ReadFile(path)
require.NoError(t, err)
require.Equal(t, "previous scan results", string(data))
}
})
}
}

func TestPageTypeModelDefaultDiscovery(t *testing.T) {
t.Chdir(t.TempDir())
localPageModel(t, "model.json")
r, err := New(&Options{KnowledgeBase: true})
require.NoError(t, err)
t.Cleanup(r.Close)
require.Equal(t, "error", r.classifyPage("", "<html></html>", 0)["PageType"])
}

// TestLocalPageTypeModelMalformedStructures exercises failures beyond JSON parsing.
func TestLocalPageTypeModelMalformedStructures(t *testing.T) {
validPath := filepath.Join(t.TempDir(), "valid.json")
localPageModel(t, validPath)
valid, err := os.ReadFile(validPath)
require.NoError(t, err)
for _, tc := range []struct {
name string
mutate func(*classifier.UnifiedModel)
}{
{"missing form", func(m *classifier.UnifiedModel) { m.FormModel = nil }},
{"missing page", func(m *classifier.UnifiedModel) { m.PageModel = nil }},
{"empty form", func(m *classifier.UnifiedModel) { m.FormModel = &classifier.FormTypeModel{} }},
{"empty page", func(m *classifier.UnifiedModel) { m.PageModel = &classifier.PageTypeModel{} }},
{"missing page classes", func(m *classifier.UnifiedModel) { m.PageModel.Classes = nil }},
{"missing form coefficients", func(m *classifier.UnifiedModel) { m.FormModel.Coef = nil }},
{"missing page intercepts", func(m *classifier.UnifiedModel) { m.PageModel.Intercept = nil }},
{"missing vectorizer", func(m *classifier.UnifiedModel) {
p := &m.PageModel.Pipelines[0]
p.DictVec, p.CountVec, p.TfidfVec = nil, nil, nil
}},
{"empty field model", func(m *classifier.UnifiedModel) { m.FieldModel = &crf.Model{} }},
} {
t.Run(tc.name, func(t *testing.T) {
var model classifier.UnifiedModel
require.NoError(t, json.Unmarshal(valid, &model))
tc.mutate(&model)
data, err := json.Marshal(model)
require.NoError(t, err)
path := filepath.Join(t.TempDir(), "model.json")
require.NoError(t, os.WriteFile(path, data, 0600))
require.NotPanics(t, func() {
r, err := New(&Options{KnowledgeBase: true, PageTypeModel: path})
if r != nil {
t.Cleanup(r.Close)
}
require.ErrorContains(t, err, "could not initialize page classifier")
require.Nil(t, r)
})
})
}
}

// TestClassifyPageMalformedModelContainsPanic covers failures reached only at inference.
func TestClassifyPageMalformedModelContainsPanic(t *testing.T) {
path := filepath.Join(t.TempDir(), "model.json")
require.NoError(t, os.WriteFile(path, []byte(`{"form_model":{},"page_model":{}}`), 0600))
model, err := dit.Load(path)
require.NoError(t, err)
r := &Runner{ditClassifier: model}
require.NotPanics(t, func() {
require.Equal(t, map[string]any{"pHash": uint64(42)}, r.classifyPage("", "<html></html>", 42))
})
}

// TestClassifyPageLatentModelFailure covers a corrupt feature not used by the startup probe.
func TestClassifyPageLatentModelFailure(t *testing.T) {
path := filepath.Join(t.TempDir(), "model.json")
localPageModel(t, path)
data, err := os.ReadFile(path)
require.NoError(t, err)
var model classifier.UnifiedModel
require.NoError(t, json.Unmarshal(data, &model))
for _, pipeline := range model.PageModel.Pipelines {
if pipeline.Name == "page title" {
pipeline.TfidfVec.CountVec.Vocabulary["corruptfeature"] = -1
}
}
data, err = json.Marshal(model)
require.NoError(t, err)
require.NoError(t, os.WriteFile(path, data, 0600))
loaded, err := loadPageTypeModel(path)
require.NoError(t, err)
const html = "<html><head><title>corruptfeature</title></head></html>"
require.Panics(t, func() { _, _ = loaded.ExtractPageType(html) })
r := &Runner{ditClassifier: loaded}
require.NotPanics(t, func() {
require.Equal(t, map[string]any{"pHash": uint64(42)}, r.classifyPage("", html, 42))
})
// A failure does not mutate the shared classifier or disable later results.
require.Equal(t, "error", r.classifyPage("", "<html></html>", 0)["PageType"])
}

// TestLocalPageTypeModelWithFields accepts a valid optional field classifier.
func TestLocalPageTypeModelWithFields(t *testing.T) {
path := filepath.Join(t.TempDir(), "model.json")
localPageModel(t, path)
model, err := classifier.LoadClassifier(path)
require.NoError(t, err)
fieldModel := crf.NewModel()
fieldModel.Labels.Add("other")
fieldModel.NumLabels = 1
fieldModel.Weights = []float64{0}
model.FieldModel = &classifier.FieldTypeModel{CRF: fieldModel}
require.NoError(t, model.SaveModel(path))
loaded, err := loadPageTypeModel(path)
require.NoError(t, err)
result, err := extractPageType(loaded, `<html><form><input name="one"><input name="two"></form></html>`)
require.NoError(t, err)
require.Len(t, result.Forms, 1)
require.Equal(t, map[string]string{"one": "other", "two": "other"}, result.Forms[0].Fields)
}
4 changes: 4 additions & 0 deletions runner/options.go
Original file line number Diff line number Diff line change
Expand Up @@ -207,6 +207,9 @@ type Options struct {
// KnowledgeBase enables knowledge base classification using dit. It is
// implied by OutputFilterPageType/OutputFilterErrorPage, which need it.
KnowledgeBase bool
// PageTypeModel overrides the model used by knowledge base classification.
// It does not enable classification on its own.
PageTypeModel string
FilterOutDuplicates bool
OutputFilterContentLength string
InputRawRequest string
Expand Down Expand Up @@ -529,6 +532,7 @@ func ParseOptions() *Options {

flagSet.CreateGroup("configs", "Configurations",
flagSet.StringVar(&cfgFile, "config", "", "path to the httpx configuration file (default $HOME/.config/httpx/config.yaml)"),
flagSet.StringVarP(&options.PageTypeModel, "page-type-model", "ptm", "", "path to a local dit model for page classification (requires -kb or -fpt; skips model download)"),
flagSet.StringSliceVarP(&options.Resolvers, "resolvers", "r", nil, "list of custom resolver (file or comma separated)", goflags.NormalizedStringSliceOptions),
flagSet.Var(&options.Allow, "allow", "allowed list of IP/CIDR's to process (file or comma separated)"),
flagSet.Var(&options.Deny, "deny", "denied list of IP/CIDR's to process (file or comma separated)"),
Expand Down
12 changes: 10 additions & 2 deletions runner/runner.go
Original file line number Diff line number Diff line change
Expand Up @@ -146,6 +146,13 @@ func New(options *Options) (*Runner, error) {
interruptCh: make(chan struct{}),
}
var err error
// Load an explicit model before creating resources or clearing output indexes.
if options.classificationEnabled() && options.PageTypeModel != "" {
runner.ditClassifier, err = loadPageTypeModel(options.PageTypeModel)
if err != nil {
return nil, errors.Wrap(err, "could not initialize page classifier")
}
}
if options.Wappalyzer != nil {
runner.wappalyzer = options.Wappalyzer
} else if techDetectRequired(options) {
Expand Down Expand Up @@ -431,7 +438,7 @@ func New(options *Options) (*Runner, error) {
}

runner.simHashes = gcache.New[uint64, []string](1000).ARC().Build()
if options.classificationEnabled() {
if options.classificationEnabled() && runner.ditClassifier == nil {
ditClassifier, err := dit.New()
if err != nil {
return nil, errors.Wrap(err, "could not initialize page classifier")
Expand Down Expand Up @@ -666,8 +673,9 @@ func (r *Runner) classifyPage(headlessBody, body string, pHash uint64) map[strin
if headlessBody != "" {
html = headlessBody
}
result, err := r.ditClassifier.ExtractPageType(html)
result, err := extractPageType(r.ditClassifier, html)
if err != nil {
gologger.Debug().Msgf("Could not classify page: %s", err)
return kb
}
kb["PageType"] = fmt.Sprint(result.Type)
Expand Down