From 11d33422bc4c45fb1695bc94f9496e08dc1c96e8 Mon Sep 17 00:00:00 2001 From: root-Manas Date: Sun, 27 Sep 2026 10:48:14 +0530 Subject: [PATCH 1/3] feat(runner): support local page classification models --- README.md | 12 +++++ runner/classifier_test.go | 93 +++++++++++++++++++++++++++++++++++++++ runner/options.go | 4 ++ runner/runner.go | 7 ++- 4 files changed, 115 insertions(+), 1 deletion(-) create mode 100644 runner/classifier_test.go diff --git a/README.md b/README.md index 298c87333..20dbf75f8 100644 --- a/README.md +++ b/README.md @@ -217,6 +217,7 @@ OUTPUT: CONFIGURATIONS: -config string path to the httpx configuration file (default $HOME/.config/httpx/config.yaml) + -ptm, -page-type-model string path to a local dit model for page classification (requires -kb or -fpt; skips model download) -r, -resolvers string[] list of custom resolver (file or comma separated) -allow string[] allowed list of IP/CIDR's to process (file or comma separated) -deny string[] denied list of IP/CIDR's to process (file or comma separated) @@ -286,6 +287,17 @@ For details about running httpx, see https://docs.projectdiscovery.io/tools/http ### Using `httpx` as a library `httpx` can be used as a library by creating an instance of the `Option` struct and populating it with the same options that would be specified via CLI. Once validated, the struct should be passed to a runner instance (to be closed at the end of the program) and the `RunEnumeration` method should be called. A minimal example of how to do it is in the [examples](examples/) folder. +### Using a local page classification model + +For environments that cannot reach Hugging Face, provision the [dit model.json](https://huggingface.co/datasets/happyhackingspace/dit/resolve/main/model.json) on an accessible machine and copy it to the scanning environment. Select it with `-page-type-model` (`-ptm`): + +```bash +httpx -l hosts.txt -kb -json -ptm /opt/httpx/model.json +httpx -l hosts.txt -fpt error,parked -ptm /opt/httpx/model.json +``` + +The path selects a local model and bypasses automatic model discovery and downloading. A missing or unreadable model, or invalid model JSON, stops initialization instead of falling back to a download. Classification remains opt-in: use `-kb` or `-fpt` (the deprecated `-fep` is also supported). Setting `-ptm` alone does not load the model or enable classification. Library users can set `Options.PageTypeModel` alongside `KnowledgeBase` or `OutputFilterPageType`. Without a custom path, the existing model discovery and download behavior is unchanged. + ## Common Recipes Below are practical one-liners for common use cases leveraging httpx's composable primitives. These recipes are validated in `runner/wellknown_recipes_test.go`. diff --git a/runner/classifier_test.go b/runner/classifier_test.go new file mode 100644 index 000000000..950fce81f --- /dev/null +++ b/runner/classifier_test.go @@ -0,0 +1,93 @@ +package runner + +import ( + "os" + "path/filepath" + "strings" + "testing" + + "github.com/PuerkitoBio/goquery" + "github.com/happyhackingspace/dit/classifier" + "github.com/stretchr/testify/require" +) + +// localPageModel creates a small, valid model without downloading training data +// or the production model. One class makes the expected prediction deterministic. +func localPageModel(t *testing.T, path string) { + t.Helper() + doc, err := goquery.NewDocumentFromReader(strings.NewReader("
Example page")) + require.NoError(t, err) + config := classifier.DefaultPageTypeTrainConfig() + config.MaxIter = 1 + formConfig := classifier.DefaultFormTypeTrainConfig() + formConfig.MaxIter = 1 + model := &classifier.FormFieldClassifier{ + FormModel: classifier.TrainFormType([]*goquery.Selection{doc.Find("form")}, []string{"login"}, formConfig), + PageModel: classifier.TrainPageType( + []*goquery.Document{doc}, [][]classifier.ClassifyResult{nil}, + []string{"https://example.com"}, []string{"error"}, config, + ), + } + require.NoError(t, model.SaveModel(path)) +} + +func TestLocalPageTypeModel(t *testing.T) { + modelPath := filepath.Join(t.TempDir(), "custom model.json") + localPageModel(t, modelPath) + + // A broken auto-discovered model must not override the explicit model path. + t.Chdir(t.TempDir()) + require.NoError(t, os.WriteFile("model.json", []byte("invalid JSON"), 0600)) + + for _, tc := range []struct { + name string + options Options + }{ + {name: "knowledge base", options: Options{KnowledgeBase: true}}, + {name: "page type filter", options: Options{OutputFilterPageType: []string{"error"}}}, + {name: "deprecated error page filter", options: Options{OutputFilterErrorPage: true}}, + } { + t.Run(tc.name, func(t *testing.T) { + options := tc.options + options.PageTypeModel = modelPath + r, err := New(&options) + require.NoError(t, err) + t.Cleanup(r.Close) + kb := r.classifyPage("", "
", 0) + require.Equal(t, "error", kb["PageType"]) + require.NotEmpty(t, kb["Forms"]) + }) + } +} + +func TestLocalPageTypeModelErrors(t *testing.T) { + // A valid auto-discovered model must not hide an invalid explicit path. + t.Chdir(t.TempDir()) + localPageModel(t, "model.json") + require.NoError(t, os.WriteFile("invalid.json", []byte("invalid JSON"), 0600)) + + for _, path := range []string{"missing.json", "invalid.json"} { + t.Run(path, func(t *testing.T) { + r, err := New(&Options{KnowledgeBase: true, PageTypeModel: path}) + require.ErrorContains(t, err, "could not initialize page classifier") + require.Nil(t, r) + }) + } +} + +func TestPageTypeModelDoesNotEnableClassification(t *testing.T) { + r, err := New(&Options{PageTypeModel: filepath.Join(t.TempDir(), "missing.json")}) + require.NoError(t, err) + t.Cleanup(r.Close) + require.Nil(t, r.ditClassifier) + require.NotContains(t, r.classifyPage("", "", 0), "PageType") +} + +func TestPageTypeModelDefaultDiscovery(t *testing.T) { + t.Chdir(t.TempDir()) + localPageModel(t, "model.json") + r, err := New(&Options{KnowledgeBase: true}) + require.NoError(t, err) + t.Cleanup(r.Close) + require.Equal(t, "error", r.classifyPage("", "", 0)["PageType"]) +} diff --git a/runner/options.go b/runner/options.go index f636b9d16..4fa46a0b1 100644 --- a/runner/options.go +++ b/runner/options.go @@ -207,6 +207,9 @@ type Options struct { // KnowledgeBase enables knowledge base classification using dit. It is // implied by OutputFilterPageType/OutputFilterErrorPage, which need it. KnowledgeBase bool + // PageTypeModel overrides the model used by knowledge base classification. + // It does not enable classification on its own. + PageTypeModel string FilterOutDuplicates bool OutputFilterContentLength string InputRawRequest string @@ -529,6 +532,7 @@ func ParseOptions() *Options { flagSet.CreateGroup("configs", "Configurations", flagSet.StringVar(&cfgFile, "config", "", "path to the httpx configuration file (default $HOME/.config/httpx/config.yaml)"), + flagSet.StringVarP(&options.PageTypeModel, "page-type-model", "ptm", "", "path to a local dit model for page classification (requires -kb or -fpt; skips model download)"), flagSet.StringSliceVarP(&options.Resolvers, "resolvers", "r", nil, "list of custom resolver (file or comma separated)", goflags.NormalizedStringSliceOptions), flagSet.Var(&options.Allow, "allow", "allowed list of IP/CIDR's to process (file or comma separated)"), flagSet.Var(&options.Deny, "deny", "denied list of IP/CIDR's to process (file or comma separated)"), diff --git a/runner/runner.go b/runner/runner.go index ea0cac3f3..f0399064e 100644 --- a/runner/runner.go +++ b/runner/runner.go @@ -432,7 +432,12 @@ func New(options *Options) (*Runner, error) { runner.simHashes = gcache.New[uint64, []string](1000).ARC().Build() if options.classificationEnabled() { - ditClassifier, err := dit.New() + ditClassifier, err := func() (*dit.Classifier, error) { + if options.PageTypeModel != "" { + return dit.Load(options.PageTypeModel) + } + return dit.New() + }() if err != nil { return nil, errors.Wrap(err, "could not initialize page classifier") } From 79b4a47f6ad99c64d901d63aaa3395365ca43eb9 Mon Sep 17 00:00:00 2001 From: root-Manas Date: Sun, 27 Sep 2026 11:01:45 +0530 Subject: [PATCH 2/3] fix(runner): load explicit models before output setup --- runner/classifier_test.go | 29 +++++++++++++++++++++++++++++ runner/runner.go | 16 +++++++++------- 2 files changed, 38 insertions(+), 7 deletions(-) diff --git a/runner/classifier_test.go b/runner/classifier_test.go index 950fce81f..6ef0b0b9b 100644 --- a/runner/classifier_test.go +++ b/runner/classifier_test.go @@ -83,6 +83,35 @@ func TestPageTypeModelDoesNotEnableClassification(t *testing.T) { require.NotContains(t, r.classifyPage("", "", 0), "PageType") } +func TestLocalPageTypeModelErrorsPreserveResponseIndexes(t *testing.T) { + for _, name := range []string{"missing", "invalid JSON"} { + t.Run(name, func(t *testing.T) { + dir := t.TempDir() + modelPath := filepath.Join(dir, "model.json") + if name == "invalid JSON" { + require.NoError(t, os.WriteFile(modelPath, []byte("invalid JSON"), 0600)) + } + indexes := []string{ + filepath.Join(dir, "response", "index.txt"), + filepath.Join(dir, "screenshot", "index_screenshot.txt"), + } + for _, path := range indexes { + require.NoError(t, os.MkdirAll(filepath.Dir(path), 0700)) + require.NoError(t, os.WriteFile(path, []byte("previous scan results"), 0600)) + } + + r, err := New(&Options{KnowledgeBase: true, PageTypeModel: modelPath, StoreResponseDir: dir}) + require.ErrorContains(t, err, "could not initialize page classifier") + require.Nil(t, r) + for _, path := range indexes { + data, err := os.ReadFile(path) + require.NoError(t, err) + require.Equal(t, "previous scan results", string(data)) + } + }) + } +} + func TestPageTypeModelDefaultDiscovery(t *testing.T) { t.Chdir(t.TempDir()) localPageModel(t, "model.json") diff --git a/runner/runner.go b/runner/runner.go index f0399064e..9290c87f2 100644 --- a/runner/runner.go +++ b/runner/runner.go @@ -146,6 +146,13 @@ func New(options *Options) (*Runner, error) { interruptCh: make(chan struct{}), } var err error + // Load an explicit model before creating resources or clearing output indexes. + if options.classificationEnabled() && options.PageTypeModel != "" { + runner.ditClassifier, err = dit.Load(options.PageTypeModel) + if err != nil { + return nil, errors.Wrap(err, "could not initialize page classifier") + } + } if options.Wappalyzer != nil { runner.wappalyzer = options.Wappalyzer } else if techDetectRequired(options) { @@ -431,13 +438,8 @@ func New(options *Options) (*Runner, error) { } runner.simHashes = gcache.New[uint64, []string](1000).ARC().Build() - if options.classificationEnabled() { - ditClassifier, err := func() (*dit.Classifier, error) { - if options.PageTypeModel != "" { - return dit.Load(options.PageTypeModel) - } - return dit.New() - }() + if options.classificationEnabled() && runner.ditClassifier == nil { + ditClassifier, err := dit.New() if err != nil { return nil, errors.Wrap(err, "could not initialize page classifier") } From 536d1cdfb6d981ad943bfebb765b94bb71891152 Mon Sep 17 00:00:00 2001 From: root-Manas Date: Sun, 27 Sep 2026 11:29:34 +0530 Subject: [PATCH 3/3] fix(runner): check model readiness and contain inference panics --- README.md | 2 + runner/classifier.go | 50 +++++++++++++++++ runner/classifier_test.go | 112 +++++++++++++++++++++++++++++++++++++- runner/runner.go | 5 +- 4 files changed, 164 insertions(+), 5 deletions(-) create mode 100644 runner/classifier.go diff --git a/README.md b/README.md index 20dbf75f8..9135ca44c 100644 --- a/README.md +++ b/README.md @@ -298,6 +298,8 @@ httpx -l hosts.txt -fpt error,parked -ptm /opt/httpx/model.json The path selects a local model and bypasses automatic model discovery and downloading. A missing or unreadable model, or invalid model JSON, stops initialization instead of falling back to a download. Classification remains opt-in: use `-kb` or `-fpt` (the deprecated `-fep` is also supported). Setting `-ptm` alone does not load the model or enable classification. Library users can set `Options.PageTypeModel` alongside `KnowledgeBase` or `OutputFilterPageType`. Without a custom path, the existing model discovery and download behavior is unchanged. +Explicit models must also pass a page and form classification check before scanning starts or existing output indexes are changed. This checks that the model can run, rather than certifying every part of its contents. A later classification error leaves that response without page-type information; diagnostics are available with `-debug`. + ## Common Recipes Below are practical one-liners for common use cases leveraging httpx's composable primitives. These recipes are validated in `runner/wellknown_recipes_test.go`. diff --git a/runner/classifier.go b/runner/classifier.go new file mode 100644 index 000000000..80cbe8d04 --- /dev/null +++ b/runner/classifier.go @@ -0,0 +1,50 @@ +package runner + +import ( + "fmt" + + "github.com/happyhackingspace/dit" +) + +// loadPageTypeModel checks that an explicit model can classify a page and form +// before runner initialization changes output files or creates resources. dit +// does not validate model structure and may panic during loading or inference. +func loadPageTypeModel(path string) (model *dit.Classifier, err error) { + defer func() { + if failure := recover(); failure != nil { + model = nil + err = fmt.Errorf("invalid page classification model: %v", failure) + } + }() + model, err = dit.Load(path) + if err != nil { + return nil, err + } + // Two fields exercise the optional field classifier's sequence transitions. + const probe = `Example
` + result, err := extractPageType(model, probe) + if err != nil { + return nil, fmt.Errorf("invalid page classification model: %w", err) + } + if result == nil || result.Type == "" || len(result.Forms) != 1 || result.Forms[0].Type == "" { + return nil, fmt.Errorf("invalid page classification model: missing page or form prediction") + } + for _, label := range result.Forms[0].Fields { + if label == "" { + return nil, fmt.Errorf("invalid page classification model: missing field prediction") + } + } + return model, nil +} + +// extractPageType contains dependency panics at the same boundary as inference +// errors. A startup probe cannot exercise every input-dependent model feature. +func extractPageType(model *dit.Classifier, html string) (result *dit.PageResult, err error) { + defer func() { + if failure := recover(); failure != nil { + result = nil + err = fmt.Errorf("page classification failed: %v", failure) + } + }() + return model.ExtractPageType(html) +} diff --git a/runner/classifier_test.go b/runner/classifier_test.go index 6ef0b0b9b..90445a93e 100644 --- a/runner/classifier_test.go +++ b/runner/classifier_test.go @@ -1,13 +1,16 @@ package runner import ( + "encoding/json" "os" "path/filepath" "strings" "testing" "github.com/PuerkitoBio/goquery" + "github.com/happyhackingspace/dit" "github.com/happyhackingspace/dit/classifier" + "github.com/happyhackingspace/dit/crf" "github.com/stretchr/testify/require" ) @@ -84,12 +87,12 @@ func TestPageTypeModelDoesNotEnableClassification(t *testing.T) { } func TestLocalPageTypeModelErrorsPreserveResponseIndexes(t *testing.T) { - for _, name := range []string{"missing", "invalid JSON"} { + for _, name := range []string{"missing", "invalid JSON", "{}", "null", `{"form_model":{},"page_model":{}}`} { t.Run(name, func(t *testing.T) { dir := t.TempDir() modelPath := filepath.Join(dir, "model.json") - if name == "invalid JSON" { - require.NoError(t, os.WriteFile(modelPath, []byte("invalid JSON"), 0600)) + if name != "missing" { + require.NoError(t, os.WriteFile(modelPath, []byte(name), 0600)) } indexes := []string{ filepath.Join(dir, "response", "index.txt"), @@ -120,3 +123,106 @@ func TestPageTypeModelDefaultDiscovery(t *testing.T) { t.Cleanup(r.Close) require.Equal(t, "error", r.classifyPage("", "", 0)["PageType"]) } + +// TestLocalPageTypeModelMalformedStructures exercises failures beyond JSON parsing. +func TestLocalPageTypeModelMalformedStructures(t *testing.T) { + validPath := filepath.Join(t.TempDir(), "valid.json") + localPageModel(t, validPath) + valid, err := os.ReadFile(validPath) + require.NoError(t, err) + for _, tc := range []struct { + name string + mutate func(*classifier.UnifiedModel) + }{ + {"missing form", func(m *classifier.UnifiedModel) { m.FormModel = nil }}, + {"missing page", func(m *classifier.UnifiedModel) { m.PageModel = nil }}, + {"empty form", func(m *classifier.UnifiedModel) { m.FormModel = &classifier.FormTypeModel{} }}, + {"empty page", func(m *classifier.UnifiedModel) { m.PageModel = &classifier.PageTypeModel{} }}, + {"missing page classes", func(m *classifier.UnifiedModel) { m.PageModel.Classes = nil }}, + {"missing form coefficients", func(m *classifier.UnifiedModel) { m.FormModel.Coef = nil }}, + {"missing page intercepts", func(m *classifier.UnifiedModel) { m.PageModel.Intercept = nil }}, + {"missing vectorizer", func(m *classifier.UnifiedModel) { + p := &m.PageModel.Pipelines[0] + p.DictVec, p.CountVec, p.TfidfVec = nil, nil, nil + }}, + {"empty field model", func(m *classifier.UnifiedModel) { m.FieldModel = &crf.Model{} }}, + } { + t.Run(tc.name, func(t *testing.T) { + var model classifier.UnifiedModel + require.NoError(t, json.Unmarshal(valid, &model)) + tc.mutate(&model) + data, err := json.Marshal(model) + require.NoError(t, err) + path := filepath.Join(t.TempDir(), "model.json") + require.NoError(t, os.WriteFile(path, data, 0600)) + require.NotPanics(t, func() { + r, err := New(&Options{KnowledgeBase: true, PageTypeModel: path}) + if r != nil { + t.Cleanup(r.Close) + } + require.ErrorContains(t, err, "could not initialize page classifier") + require.Nil(t, r) + }) + }) + } +} + +// TestClassifyPageMalformedModelContainsPanic covers failures reached only at inference. +func TestClassifyPageMalformedModelContainsPanic(t *testing.T) { + path := filepath.Join(t.TempDir(), "model.json") + require.NoError(t, os.WriteFile(path, []byte(`{"form_model":{},"page_model":{}}`), 0600)) + model, err := dit.Load(path) + require.NoError(t, err) + r := &Runner{ditClassifier: model} + require.NotPanics(t, func() { + require.Equal(t, map[string]any{"pHash": uint64(42)}, r.classifyPage("", "", 42)) + }) +} + +// TestClassifyPageLatentModelFailure covers a corrupt feature not used by the startup probe. +func TestClassifyPageLatentModelFailure(t *testing.T) { + path := filepath.Join(t.TempDir(), "model.json") + localPageModel(t, path) + data, err := os.ReadFile(path) + require.NoError(t, err) + var model classifier.UnifiedModel + require.NoError(t, json.Unmarshal(data, &model)) + for _, pipeline := range model.PageModel.Pipelines { + if pipeline.Name == "page title" { + pipeline.TfidfVec.CountVec.Vocabulary["corruptfeature"] = -1 + } + } + data, err = json.Marshal(model) + require.NoError(t, err) + require.NoError(t, os.WriteFile(path, data, 0600)) + loaded, err := loadPageTypeModel(path) + require.NoError(t, err) + const html = "corruptfeature" + require.Panics(t, func() { _, _ = loaded.ExtractPageType(html) }) + r := &Runner{ditClassifier: loaded} + require.NotPanics(t, func() { + require.Equal(t, map[string]any{"pHash": uint64(42)}, r.classifyPage("", html, 42)) + }) + // A failure does not mutate the shared classifier or disable later results. + require.Equal(t, "error", r.classifyPage("", "", 0)["PageType"]) +} + +// TestLocalPageTypeModelWithFields accepts a valid optional field classifier. +func TestLocalPageTypeModelWithFields(t *testing.T) { + path := filepath.Join(t.TempDir(), "model.json") + localPageModel(t, path) + model, err := classifier.LoadClassifier(path) + require.NoError(t, err) + fieldModel := crf.NewModel() + fieldModel.Labels.Add("other") + fieldModel.NumLabels = 1 + fieldModel.Weights = []float64{0} + model.FieldModel = &classifier.FieldTypeModel{CRF: fieldModel} + require.NoError(t, model.SaveModel(path)) + loaded, err := loadPageTypeModel(path) + require.NoError(t, err) + result, err := extractPageType(loaded, `
`) + require.NoError(t, err) + require.Len(t, result.Forms, 1) + require.Equal(t, map[string]string{"one": "other", "two": "other"}, result.Forms[0].Fields) +} diff --git a/runner/runner.go b/runner/runner.go index 9290c87f2..93fa2b8b8 100644 --- a/runner/runner.go +++ b/runner/runner.go @@ -148,7 +148,7 @@ func New(options *Options) (*Runner, error) { var err error // Load an explicit model before creating resources or clearing output indexes. if options.classificationEnabled() && options.PageTypeModel != "" { - runner.ditClassifier, err = dit.Load(options.PageTypeModel) + runner.ditClassifier, err = loadPageTypeModel(options.PageTypeModel) if err != nil { return nil, errors.Wrap(err, "could not initialize page classifier") } @@ -673,8 +673,9 @@ func (r *Runner) classifyPage(headlessBody, body string, pHash uint64) map[strin if headlessBody != "" { html = headlessBody } - result, err := r.ditClassifier.ExtractPageType(html) + result, err := extractPageType(r.ditClassifier, html) if err != nil { + gologger.Debug().Msgf("Could not classify page: %s", err) return kb } kb["PageType"] = fmt.Sprint(result.Type)