diff --git a/README.md b/README.md
index 298c8733..9135ca44 100644
--- a/README.md
+++ b/README.md
@@ -217,6 +217,7 @@ OUTPUT:
CONFIGURATIONS:
-config string path to the httpx configuration file (default $HOME/.config/httpx/config.yaml)
+ -ptm, -page-type-model string path to a local dit model for page classification (requires -kb or -fpt; skips model download)
-r, -resolvers string[] list of custom resolver (file or comma separated)
-allow string[] allowed list of IP/CIDR's to process (file or comma separated)
-deny string[] denied list of IP/CIDR's to process (file or comma separated)
@@ -286,6 +287,19 @@ For details about running httpx, see https://docs.projectdiscovery.io/tools/http
### Using `httpx` as a library
`httpx` can be used as a library by creating an instance of the `Option` struct and populating it with the same options that would be specified via CLI. Once validated, the struct should be passed to a runner instance (to be closed at the end of the program) and the `RunEnumeration` method should be called. A minimal example of how to do it is in the [examples](examples/) folder.
+### Using a local page classification model
+
+For environments that cannot reach Hugging Face, provision the [dit model.json](https://huggingface.co/datasets/happyhackingspace/dit/resolve/main/model.json) on an accessible machine and copy it to the scanning environment. Select it with `-page-type-model` (`-ptm`):
+
+```bash
+httpx -l hosts.txt -kb -json -ptm /opt/httpx/model.json
+httpx -l hosts.txt -fpt error,parked -ptm /opt/httpx/model.json
+```
+
+The path selects a local model and bypasses automatic model discovery and downloading. A missing or unreadable model, or invalid model JSON, stops initialization instead of falling back to a download. Classification remains opt-in: use `-kb` or `-fpt` (the deprecated `-fep` is also supported). Setting `-ptm` alone does not load the model or enable classification. Library users can set `Options.PageTypeModel` alongside `KnowledgeBase` or `OutputFilterPageType`. Without a custom path, the existing model discovery and download behavior is unchanged.
+
+Explicit models must also pass a page and form classification check before scanning starts or existing output indexes are changed. This checks that the model can run, rather than certifying every part of its contents. A later classification error leaves that response without page-type information; diagnostics are available with `-debug`.
+
## Common Recipes
Below are practical one-liners for common use cases leveraging httpx's composable primitives. These recipes are validated in `runner/wellknown_recipes_test.go`.
diff --git a/runner/classifier.go b/runner/classifier.go
new file mode 100644
index 00000000..80cbe8d0
--- /dev/null
+++ b/runner/classifier.go
@@ -0,0 +1,50 @@
+package runner
+
+import (
+ "fmt"
+
+ "github.com/happyhackingspace/dit"
+)
+
+// loadPageTypeModel checks that an explicit model can classify a page and form
+// before runner initialization changes output files or creates resources. dit
+// does not validate model structure and may panic during loading or inference.
+func loadPageTypeModel(path string) (model *dit.Classifier, err error) {
+ defer func() {
+ if failure := recover(); failure != nil {
+ model = nil
+ err = fmt.Errorf("invalid page classification model: %v", failure)
+ }
+ }()
+ model, err = dit.Load(path)
+ if err != nil {
+ return nil, err
+ }
+ // Two fields exercise the optional field classifier's sequence transitions.
+ const probe = `
Example`
+ result, err := extractPageType(model, probe)
+ if err != nil {
+ return nil, fmt.Errorf("invalid page classification model: %w", err)
+ }
+ if result == nil || result.Type == "" || len(result.Forms) != 1 || result.Forms[0].Type == "" {
+ return nil, fmt.Errorf("invalid page classification model: missing page or form prediction")
+ }
+ for _, label := range result.Forms[0].Fields {
+ if label == "" {
+ return nil, fmt.Errorf("invalid page classification model: missing field prediction")
+ }
+ }
+ return model, nil
+}
+
+// extractPageType contains dependency panics at the same boundary as inference
+// errors. A startup probe cannot exercise every input-dependent model feature.
+func extractPageType(model *dit.Classifier, html string) (result *dit.PageResult, err error) {
+ defer func() {
+ if failure := recover(); failure != nil {
+ result = nil
+ err = fmt.Errorf("page classification failed: %v", failure)
+ }
+ }()
+ return model.ExtractPageType(html)
+}
diff --git a/runner/classifier_test.go b/runner/classifier_test.go
new file mode 100644
index 00000000..90445a93
--- /dev/null
+++ b/runner/classifier_test.go
@@ -0,0 +1,228 @@
+package runner
+
+import (
+ "encoding/json"
+ "os"
+ "path/filepath"
+ "strings"
+ "testing"
+
+ "github.com/PuerkitoBio/goquery"
+ "github.com/happyhackingspace/dit"
+ "github.com/happyhackingspace/dit/classifier"
+ "github.com/happyhackingspace/dit/crf"
+ "github.com/stretchr/testify/require"
+)
+
+// localPageModel creates a small, valid model without downloading training data
+// or the production model. One class makes the expected prediction deterministic.
+func localPageModel(t *testing.T, path string) {
+ t.Helper()
+ doc, err := goquery.NewDocumentFromReader(strings.NewReader("Example page"))
+ require.NoError(t, err)
+ config := classifier.DefaultPageTypeTrainConfig()
+ config.MaxIter = 1
+ formConfig := classifier.DefaultFormTypeTrainConfig()
+ formConfig.MaxIter = 1
+ model := &classifier.FormFieldClassifier{
+ FormModel: classifier.TrainFormType([]*goquery.Selection{doc.Find("form")}, []string{"login"}, formConfig),
+ PageModel: classifier.TrainPageType(
+ []*goquery.Document{doc}, [][]classifier.ClassifyResult{nil},
+ []string{"https://example.com"}, []string{"error"}, config,
+ ),
+ }
+ require.NoError(t, model.SaveModel(path))
+}
+
+func TestLocalPageTypeModel(t *testing.T) {
+ modelPath := filepath.Join(t.TempDir(), "custom model.json")
+ localPageModel(t, modelPath)
+
+ // A broken auto-discovered model must not override the explicit model path.
+ t.Chdir(t.TempDir())
+ require.NoError(t, os.WriteFile("model.json", []byte("invalid JSON"), 0600))
+
+ for _, tc := range []struct {
+ name string
+ options Options
+ }{
+ {name: "knowledge base", options: Options{KnowledgeBase: true}},
+ {name: "page type filter", options: Options{OutputFilterPageType: []string{"error"}}},
+ {name: "deprecated error page filter", options: Options{OutputFilterErrorPage: true}},
+ } {
+ t.Run(tc.name, func(t *testing.T) {
+ options := tc.options
+ options.PageTypeModel = modelPath
+ r, err := New(&options)
+ require.NoError(t, err)
+ t.Cleanup(r.Close)
+ kb := r.classifyPage("", "", 0)
+ require.Equal(t, "error", kb["PageType"])
+ require.NotEmpty(t, kb["Forms"])
+ })
+ }
+}
+
+func TestLocalPageTypeModelErrors(t *testing.T) {
+ // A valid auto-discovered model must not hide an invalid explicit path.
+ t.Chdir(t.TempDir())
+ localPageModel(t, "model.json")
+ require.NoError(t, os.WriteFile("invalid.json", []byte("invalid JSON"), 0600))
+
+ for _, path := range []string{"missing.json", "invalid.json"} {
+ t.Run(path, func(t *testing.T) {
+ r, err := New(&Options{KnowledgeBase: true, PageTypeModel: path})
+ require.ErrorContains(t, err, "could not initialize page classifier")
+ require.Nil(t, r)
+ })
+ }
+}
+
+func TestPageTypeModelDoesNotEnableClassification(t *testing.T) {
+ r, err := New(&Options{PageTypeModel: filepath.Join(t.TempDir(), "missing.json")})
+ require.NoError(t, err)
+ t.Cleanup(r.Close)
+ require.Nil(t, r.ditClassifier)
+ require.NotContains(t, r.classifyPage("", "", 0), "PageType")
+}
+
+func TestLocalPageTypeModelErrorsPreserveResponseIndexes(t *testing.T) {
+ for _, name := range []string{"missing", "invalid JSON", "{}", "null", `{"form_model":{},"page_model":{}}`} {
+ t.Run(name, func(t *testing.T) {
+ dir := t.TempDir()
+ modelPath := filepath.Join(dir, "model.json")
+ if name != "missing" {
+ require.NoError(t, os.WriteFile(modelPath, []byte(name), 0600))
+ }
+ indexes := []string{
+ filepath.Join(dir, "response", "index.txt"),
+ filepath.Join(dir, "screenshot", "index_screenshot.txt"),
+ }
+ for _, path := range indexes {
+ require.NoError(t, os.MkdirAll(filepath.Dir(path), 0700))
+ require.NoError(t, os.WriteFile(path, []byte("previous scan results"), 0600))
+ }
+
+ r, err := New(&Options{KnowledgeBase: true, PageTypeModel: modelPath, StoreResponseDir: dir})
+ require.ErrorContains(t, err, "could not initialize page classifier")
+ require.Nil(t, r)
+ for _, path := range indexes {
+ data, err := os.ReadFile(path)
+ require.NoError(t, err)
+ require.Equal(t, "previous scan results", string(data))
+ }
+ })
+ }
+}
+
+func TestPageTypeModelDefaultDiscovery(t *testing.T) {
+ t.Chdir(t.TempDir())
+ localPageModel(t, "model.json")
+ r, err := New(&Options{KnowledgeBase: true})
+ require.NoError(t, err)
+ t.Cleanup(r.Close)
+ require.Equal(t, "error", r.classifyPage("", "", 0)["PageType"])
+}
+
+// TestLocalPageTypeModelMalformedStructures exercises failures beyond JSON parsing.
+func TestLocalPageTypeModelMalformedStructures(t *testing.T) {
+ validPath := filepath.Join(t.TempDir(), "valid.json")
+ localPageModel(t, validPath)
+ valid, err := os.ReadFile(validPath)
+ require.NoError(t, err)
+ for _, tc := range []struct {
+ name string
+ mutate func(*classifier.UnifiedModel)
+ }{
+ {"missing form", func(m *classifier.UnifiedModel) { m.FormModel = nil }},
+ {"missing page", func(m *classifier.UnifiedModel) { m.PageModel = nil }},
+ {"empty form", func(m *classifier.UnifiedModel) { m.FormModel = &classifier.FormTypeModel{} }},
+ {"empty page", func(m *classifier.UnifiedModel) { m.PageModel = &classifier.PageTypeModel{} }},
+ {"missing page classes", func(m *classifier.UnifiedModel) { m.PageModel.Classes = nil }},
+ {"missing form coefficients", func(m *classifier.UnifiedModel) { m.FormModel.Coef = nil }},
+ {"missing page intercepts", func(m *classifier.UnifiedModel) { m.PageModel.Intercept = nil }},
+ {"missing vectorizer", func(m *classifier.UnifiedModel) {
+ p := &m.PageModel.Pipelines[0]
+ p.DictVec, p.CountVec, p.TfidfVec = nil, nil, nil
+ }},
+ {"empty field model", func(m *classifier.UnifiedModel) { m.FieldModel = &crf.Model{} }},
+ } {
+ t.Run(tc.name, func(t *testing.T) {
+ var model classifier.UnifiedModel
+ require.NoError(t, json.Unmarshal(valid, &model))
+ tc.mutate(&model)
+ data, err := json.Marshal(model)
+ require.NoError(t, err)
+ path := filepath.Join(t.TempDir(), "model.json")
+ require.NoError(t, os.WriteFile(path, data, 0600))
+ require.NotPanics(t, func() {
+ r, err := New(&Options{KnowledgeBase: true, PageTypeModel: path})
+ if r != nil {
+ t.Cleanup(r.Close)
+ }
+ require.ErrorContains(t, err, "could not initialize page classifier")
+ require.Nil(t, r)
+ })
+ })
+ }
+}
+
+// TestClassifyPageMalformedModelContainsPanic covers failures reached only at inference.
+func TestClassifyPageMalformedModelContainsPanic(t *testing.T) {
+ path := filepath.Join(t.TempDir(), "model.json")
+ require.NoError(t, os.WriteFile(path, []byte(`{"form_model":{},"page_model":{}}`), 0600))
+ model, err := dit.Load(path)
+ require.NoError(t, err)
+ r := &Runner{ditClassifier: model}
+ require.NotPanics(t, func() {
+ require.Equal(t, map[string]any{"pHash": uint64(42)}, r.classifyPage("", "", 42))
+ })
+}
+
+// TestClassifyPageLatentModelFailure covers a corrupt feature not used by the startup probe.
+func TestClassifyPageLatentModelFailure(t *testing.T) {
+ path := filepath.Join(t.TempDir(), "model.json")
+ localPageModel(t, path)
+ data, err := os.ReadFile(path)
+ require.NoError(t, err)
+ var model classifier.UnifiedModel
+ require.NoError(t, json.Unmarshal(data, &model))
+ for _, pipeline := range model.PageModel.Pipelines {
+ if pipeline.Name == "page title" {
+ pipeline.TfidfVec.CountVec.Vocabulary["corruptfeature"] = -1
+ }
+ }
+ data, err = json.Marshal(model)
+ require.NoError(t, err)
+ require.NoError(t, os.WriteFile(path, data, 0600))
+ loaded, err := loadPageTypeModel(path)
+ require.NoError(t, err)
+ const html = "corruptfeature"
+ require.Panics(t, func() { _, _ = loaded.ExtractPageType(html) })
+ r := &Runner{ditClassifier: loaded}
+ require.NotPanics(t, func() {
+ require.Equal(t, map[string]any{"pHash": uint64(42)}, r.classifyPage("", html, 42))
+ })
+ // A failure does not mutate the shared classifier or disable later results.
+ require.Equal(t, "error", r.classifyPage("", "", 0)["PageType"])
+}
+
+// TestLocalPageTypeModelWithFields accepts a valid optional field classifier.
+func TestLocalPageTypeModelWithFields(t *testing.T) {
+ path := filepath.Join(t.TempDir(), "model.json")
+ localPageModel(t, path)
+ model, err := classifier.LoadClassifier(path)
+ require.NoError(t, err)
+ fieldModel := crf.NewModel()
+ fieldModel.Labels.Add("other")
+ fieldModel.NumLabels = 1
+ fieldModel.Weights = []float64{0}
+ model.FieldModel = &classifier.FieldTypeModel{CRF: fieldModel}
+ require.NoError(t, model.SaveModel(path))
+ loaded, err := loadPageTypeModel(path)
+ require.NoError(t, err)
+ result, err := extractPageType(loaded, ``)
+ require.NoError(t, err)
+ require.Len(t, result.Forms, 1)
+ require.Equal(t, map[string]string{"one": "other", "two": "other"}, result.Forms[0].Fields)
+}
diff --git a/runner/options.go b/runner/options.go
index f636b9d1..4fa46a0b 100644
--- a/runner/options.go
+++ b/runner/options.go
@@ -207,6 +207,9 @@ type Options struct {
// KnowledgeBase enables knowledge base classification using dit. It is
// implied by OutputFilterPageType/OutputFilterErrorPage, which need it.
KnowledgeBase bool
+ // PageTypeModel overrides the model used by knowledge base classification.
+ // It does not enable classification on its own.
+ PageTypeModel string
FilterOutDuplicates bool
OutputFilterContentLength string
InputRawRequest string
@@ -529,6 +532,7 @@ func ParseOptions() *Options {
flagSet.CreateGroup("configs", "Configurations",
flagSet.StringVar(&cfgFile, "config", "", "path to the httpx configuration file (default $HOME/.config/httpx/config.yaml)"),
+ flagSet.StringVarP(&options.PageTypeModel, "page-type-model", "ptm", "", "path to a local dit model for page classification (requires -kb or -fpt; skips model download)"),
flagSet.StringSliceVarP(&options.Resolvers, "resolvers", "r", nil, "list of custom resolver (file or comma separated)", goflags.NormalizedStringSliceOptions),
flagSet.Var(&options.Allow, "allow", "allowed list of IP/CIDR's to process (file or comma separated)"),
flagSet.Var(&options.Deny, "deny", "denied list of IP/CIDR's to process (file or comma separated)"),
diff --git a/runner/runner.go b/runner/runner.go
index ea0cac3f..93fa2b8b 100644
--- a/runner/runner.go
+++ b/runner/runner.go
@@ -146,6 +146,13 @@ func New(options *Options) (*Runner, error) {
interruptCh: make(chan struct{}),
}
var err error
+ // Load an explicit model before creating resources or clearing output indexes.
+ if options.classificationEnabled() && options.PageTypeModel != "" {
+ runner.ditClassifier, err = loadPageTypeModel(options.PageTypeModel)
+ if err != nil {
+ return nil, errors.Wrap(err, "could not initialize page classifier")
+ }
+ }
if options.Wappalyzer != nil {
runner.wappalyzer = options.Wappalyzer
} else if techDetectRequired(options) {
@@ -431,7 +438,7 @@ func New(options *Options) (*Runner, error) {
}
runner.simHashes = gcache.New[uint64, []string](1000).ARC().Build()
- if options.classificationEnabled() {
+ if options.classificationEnabled() && runner.ditClassifier == nil {
ditClassifier, err := dit.New()
if err != nil {
return nil, errors.Wrap(err, "could not initialize page classifier")
@@ -666,8 +673,9 @@ func (r *Runner) classifyPage(headlessBody, body string, pHash uint64) map[strin
if headlessBody != "" {
html = headlessBody
}
- result, err := r.ditClassifier.ExtractPageType(html)
+ result, err := extractPageType(r.ditClassifier, html)
if err != nil {
+ gologger.Debug().Msgf("Could not classify page: %s", err)
return kb
}
kb["PageType"] = fmt.Sprint(result.Type)