diff --git a/internal/extract/generic/generic.go b/internal/extract/generic/generic.go new file mode 100644 index 0000000..05707f2 --- /dev/null +++ b/internal/extract/generic/generic.go @@ -0,0 +1,228 @@ +package generic + +import ( + "strings" + + "tangled.org/dunkirk.sh/pear/internal/extract/schema" + "tangled.org/dunkirk.sh/pear/internal/models" + + "golang.org/x/net/html" +) + +func Extract(body string) (*models.Recipe, bool) { + doc, err := html.Parse(strings.NewReader(body)) + if err != nil { + return nil, false + } + + recipe := &models.Recipe{} + recipe.ExtractionMethod = "generic" + + found := false + + if name := findByClass(doc, "recipe-title", "recipe-name"); name != "" { + recipe.Name = name + found = true + } + if desc := findByMetaContent(doc, "description"); desc != "" { + recipe.Description = desc + } + if img := findByItempropImage(doc); img != "" { + recipe.ImageURL = img + } else if img := findByMetaContent(doc, "og:image"); img != "" { + recipe.ImageURL = img + } + if yield := findByClass(doc, "serves"); yield != "" { + recipe.Yield = strings.TrimPrefix(yield, "Serves ") + } + + ingredients := collectIngredients(doc) + if len(ingredients) > 0 { + found = true + for _, ing := range ingredients { + recipe.Ingredients = append(recipe.Ingredients, schema.ParseIngredient(ing)) + } + } + + instructions := collectInstructions(doc) + if len(instructions) > 0 { + found = true + for _, instr := range instructions { + recipe.Instructions = append(recipe.Instructions, models.Instruction{Text: instr}) + } + } + + if !found { + return nil, false + } + + if recipe.Name == "" { + if title := findByMetaContent(doc, "og:title"); title != "" { + recipe.Name = title + } + } + + if recipe.Name == "" { + return nil, false + } + + return recipe, true +} + +func findByClass(n *html.Node, classes ...string) string { + var result string + var f func(*html.Node) + f = func(n *html.Node) { + if n.Type == html.ElementNode { + for _, class := range classes { + if hasClass(n, class) { + result = textContent(n) + return + } + } + } + for c := n.FirstChild; c != nil; c = c.NextSibling { + f(c) + if result != "" { + return + } + } + } + f(n) + return result +} + +func hasClass(n *html.Node, class string) bool { + for _, attr := range n.Attr { + if attr.Key == "class" { + for _, c := range strings.Fields(attr.Val) { + if c == class { + return true + } + } + } + } + return false +} + +func findByMetaContent(n *html.Node, name string) string { + var f func(*html.Node) string + f = func(n *html.Node) string { + if n.Type == html.ElementNode && n.Data == "meta" { + metaName, metaProp, metaContent := "", "", "" + for _, attr := range n.Attr { + switch attr.Key { + case "name": + metaName = attr.Val + case "property": + metaProp = attr.Val + case "content": + metaContent = attr.Val + } + } + if (metaName == name || metaProp == name) && metaContent != "" { + return metaContent + } + } + for c := n.FirstChild; c != nil; c = c.NextSibling { + if v := f(c); v != "" { + return v + } + } + return "" + } + return f(n) +} + +func findByItempropImage(n *html.Node) string { + var f func(*html.Node) string + f = func(n *html.Node) string { + if n.Type == html.ElementNode { + hasImageProp := false + for _, attr := range n.Attr { + if attr.Key == "itemprop" && attr.Val == "image" { + hasImageProp = true + break + } + } + if hasImageProp && n.Data == "img" { + for _, attr := range n.Attr { + if attr.Key == "src" { + return attr.Val + } + } + } + } + for c := n.FirstChild; c != nil; c = c.NextSibling { + if v := f(c); v != "" { + return v + } + } + return "" + } + return f(n) +} + +func collectIngredients(n *html.Node) []string { + container := findNodeByClass(n, "ingredients", "recipe-ingredients") + if container == nil { + return nil + } + var items []string + collectParagraphsAndListItems(container, &items) + return items +} + +func collectInstructions(n *html.Node) []string { + container := findNodeByClass(n, "directions", "instructions", "recipe-instructions", "recipe-directions") + if container == nil { + return nil + } + var items []string + collectParagraphsAndListItems(container, &items) + return items +} + +func findNodeByClass(n *html.Node, classes ...string) *html.Node { + var f func(*html.Node) *html.Node + f = func(n *html.Node) *html.Node { + if n.Type == html.ElementNode { + for _, class := range classes { + if hasClass(n, class) { + return n + } + } + } + for c := n.FirstChild; c != nil; c = c.NextSibling { + if found := f(c); found != nil { + return found + } + } + return nil + } + return f(n) +} + +func collectParagraphsAndListItems(n *html.Node, items *[]string) { + for c := n.FirstChild; c != nil; c = c.NextSibling { + if c.Type == html.ElementNode && (c.Data == "p" || c.Data == "li") { + text := strings.TrimSpace(textContent(c)) + if text != "" { + *items = append(*items, text) + } + } else { + collectParagraphsAndListItems(c, items) + } + } +} + +func textContent(n *html.Node) string { + if n.Type == html.TextNode { + return n.Data + } + var sb strings.Builder + for c := n.FirstChild; c != nil; c = c.NextSibling { + sb.WriteString(textContent(c)) + } + return strings.TrimSpace(sb.String()) +} diff --git a/internal/extract/generic/generic_test.go b/internal/extract/generic/generic_test.go new file mode 100644 index 0000000..4be61d6 --- /dev/null +++ b/internal/extract/generic/generic_test.go @@ -0,0 +1,58 @@ +package generic + +import ( + "testing" +) + +const testHTML = ` + +Smoothie +
+

Creamy Ginger Green Smoothie

+

Serves 1

+
+

Ingredients:

+

2 handfuls organic spinach

+

1 cup filtered water

+

1/2 avocado

+

1 medium banana

+
+
+

Directions:

+
+

Simply add all ingredients in a high-speed blender and blend until thick and creamy.

+

You may add ice if you'd like to chill further.

+

Enjoy!

+
+
+
+ +` + +func TestNutritionStripped(t *testing.T) { + recipe, ok := Extract(testHTML) + if !ok { + t.Fatal("Extract returned false") + } + t.Logf("Name: %q", recipe.Name) + t.Logf("Yield: %q", recipe.Yield) + t.Logf("ImageURL: %q", recipe.ImageURL) + t.Logf("Ingredients (%d):", len(recipe.Ingredients)) + for i, ing := range recipe.Ingredients { + t.Logf(" [%d] RawText=%q Name=%q", i, ing.RawText, ing.Name) + } + t.Logf("Instructions (%d):", len(recipe.Instructions)) + for i, instr := range recipe.Instructions { + t.Logf(" [%d] Text=%q", i, instr.Text) + } + + if len(recipe.Ingredients) == 0 { + t.Error("expected ingredients, got none") + } + if len(recipe.Instructions) == 0 { + t.Error("expected instructions, got none") + } + if recipe.Name != "Creamy Ginger Green Smoothie" { + t.Errorf("unexpected name: %q", recipe.Name) + } +} diff --git a/internal/extract/pipeline.go b/internal/extract/pipeline.go index 7e33a53..66e183b 100644 --- a/internal/extract/pipeline.go +++ b/internal/extract/pipeline.go @@ -11,6 +11,7 @@ import ( "strings" "time" + "tangled.org/dunkirk.sh/pear/internal/extract/generic" "tangled.org/dunkirk.sh/pear/internal/extract/hrecipe" "tangled.org/dunkirk.sh/pear/internal/extract/marmiton" "tangled.org/dunkirk.sh/pear/internal/extract/schema" @@ -64,39 +65,60 @@ func (p *Pipeline) Extract(targetURL string) *Result { lang := detectLanguage(body) - if recipe, ok := marmiton.Extract(body); ok { - recipe.SourceURL = targetURL - recipe.SourceDomain = domainOf(targetURL) - recipe.Language = lang - return &Result{Recipe: recipe} + type candidate struct { + recipe *models.Recipe + method string } + var fallbacks []candidate - if recipe, ok := wprm.Extract(body); ok { - recipe.SourceURL = targetURL - recipe.SourceDomain = domainOf(targetURL) - recipe.Language = lang - return &Result{Recipe: recipe} + tryExtract := func(r *models.Recipe, ok bool, method string) *Result { + if !ok || r == nil { + return nil + } + r.SourceURL = targetURL + r.SourceDomain = domainOf(targetURL) + r.Language = lang + r.Normalize() + if len(r.Instructions) > 0 { + return &Result{Recipe: r} + } + fallbacks = append(fallbacks, candidate{r, method}) + return nil } - if recipe, ok := schema.Extract(body); ok { - recipe.SourceURL = targetURL - recipe.SourceDomain = domainOf(targetURL) - recipe.Language = lang - return &Result{Recipe: recipe} + if r, ok := marmiton.Extract(body); true { + if result := tryExtract(r, ok, "marmiton"); result != nil { + return result + } } - - if recipe, ok := schema.ExtractMicrodata(body); ok { - recipe.SourceURL = targetURL - recipe.SourceDomain = domainOf(targetURL) - recipe.Language = lang - return &Result{Recipe: recipe} + if r, ok := wprm.Extract(body); true { + if result := tryExtract(r, ok, "wprm"); result != nil { + return result + } + } + if r, ok := schema.Extract(body); true { + if result := tryExtract(r, ok, "schema.org"); result != nil { + return result + } + } + if r, ok := schema.ExtractMicrodata(body); true { + if result := tryExtract(r, ok, "microdata"); result != nil { + return result + } + } + if r, ok := hrecipe.Extract(body); true { + if result := tryExtract(r, ok, "h-recipe"); result != nil { + return result + } + } + if r, ok := generic.Extract(body); true { + if result := tryExtract(r, ok, "generic"); result != nil { + return result + } } - if recipe, ok := hrecipe.Extract(body); ok { - recipe.SourceURL = targetURL - recipe.SourceDomain = domainOf(targetURL) - recipe.Language = lang - return &Result{Recipe: recipe} + if len(fallbacks) > 0 { + return &Result{Recipe: fallbacks[0].recipe} } return &Result{Error: fmt.Errorf("no recipe found on page - tried JSON-LD, microdata, and h-recipe extraction")} diff --git a/internal/extract/pipeline_test.go b/internal/extract/pipeline_test.go new file mode 100644 index 0000000..2f190b8 --- /dev/null +++ b/internal/extract/pipeline_test.go @@ -0,0 +1,60 @@ +package extract + +import ( + "testing" + + "tangled.org/dunkirk.sh/pear/internal/extract/generic" + "tangled.org/dunkirk.sh/pear/internal/extract/schema" +) + +func TestPipelineNutritionStripped(t *testing.T) { + const html = ` + + + + +

Creamy Ginger Green Smoothie

+
+

2 handfuls organic spinach

+

1 cup filtered water

+
+
+
+

Add all ingredients to a blender and blend until creamy.

+

Enjoy!

+
+
+` + + recipe, ok := schema.Extract(html) + if !ok { + t.Fatal("JSON-LD extractor should have found the recipe") + } + if len(recipe.Instructions) != 0 { + t.Errorf("expected no instructions from JSON-LD, got %d", len(recipe.Instructions)) + } + t.Logf("JSON-LD: Name=%q Instructions=%d Ingredients=%d", recipe.Name, len(recipe.Instructions), len(recipe.Ingredients)) + + recipe2, ok2 := generic.Extract(html) + if !ok2 { + t.Fatal("generic extractor should have found the recipe") + } + if len(recipe2.Instructions) == 0 { + t.Error("generic extractor should have found instructions") + } + if len(recipe2.Ingredients) == 0 { + t.Error("generic extractor should have found ingredients") + } + t.Logf("Generic: Name=%q Instructions=%d Ingredients=%d", recipe2.Name, len(recipe2.Instructions), len(recipe2.Ingredients)) +} diff --git a/internal/models/recipe.go b/internal/models/recipe.go index e6fea71..8d86763 100644 --- a/internal/models/recipe.go +++ b/internal/models/recipe.go @@ -1,6 +1,9 @@ package models -import "time" +import ( + "html" + "time" +) type Recipe struct { Name string @@ -36,4 +39,17 @@ type CachedRecipe struct { Recipe []byte ExtractionMethod string FetchedAt time.Time +} + +func (r *Recipe) Normalize() { + r.Name = html.UnescapeString(r.Name) + r.Description = html.UnescapeString(r.Description) + for i := range r.Ingredients { + r.Ingredients[i].RawText = html.UnescapeString(r.Ingredients[i].RawText) + r.Ingredients[i].Name = html.UnescapeString(r.Ingredients[i].Name) + r.Ingredients[i].Group = html.UnescapeString(r.Ingredients[i].Group) + } + for i := range r.Instructions { + r.Instructions[i].Text = html.UnescapeString(r.Instructions[i].Text) + } } \ No newline at end of file