public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Ingredients $recipe->resetIngredients(); $nodes = null; if (!$nodes || !$nodes->length) { $nodes = $xpath->query('//*[@id="recipe-ingredients"]//div[@class="view-content"]/*'); } if (!$nodes || !$nodes->length) { $nodes = $xpath->query('//*[@id="recipe-ingredients"]//div[@class="ingredient-lists separator-serated tab-content"]/*'); } foreach ($nodes as $node) { if ($node->nodeName == 'h3') { $line = $node->nodeValue; $line = RecipeParser_Text::formatSectionName($line); $recipe->addIngredientsSection($line); } else { if ($node->nodeName == 'ul') { foreach ($node->childNodes as $subnode) { $line = $subnode->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } } } } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Notes -- Collect the non-standard cook times and baking temps, // and also any tips/notes that appear at the end of the recipe instructions. $notes = array(); $nodes = $xpath->query('//*[@class="recipeTips"]//li'); foreach ($nodes as $node) { $value = RecipeParser_Text::FormatAsOneLine($node->nodeValue); $value = preg_replace("/^(Tip|Note)\\s*(.*)\$/", "\$2", $value); $notes[] = $value; } $nodes = $xpath->query('//*[@class="recipeInfo"]//*[@class="type"]'); foreach ($nodes as $node) { $value = RecipeParser_Text::formatAsOneLine($node->nodeValue); if (strpos($value, "Makes:") !== false) { continue; } $notes[] = $value; } $recipe->notes = implode("\n\n", $notes); // Adjust Photo URL for larger dimensions $recipe->photo_url = preg_replace("/\\/l_([^\\/]+)/", "/550_\$1", $recipe->photo_url); return $recipe; }
public static function parse($html, $url) { $recipe = new RecipeParser_Recipe(); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Title $nodes = $xpath->query('//*[@id="page-title"]'); if ($nodes->length) { $line = RecipeParser_Text::formatTitle($nodes->item(0)->nodeValue); $recipe->title = $line; } // Times $nodes = $xpath->query('//*[@class="field-recipe-time"]'); foreach ($nodes as $node) { $line = RecipeParser_Text::formatAsOneLine($node->nodeValue); if (strpos($line, "Hands-On Time") !== false) { $line = str_replace("Hands-On Time ", "", $line); $recipe->time["prep"] = RecipeParser_Times::toMinutes($line); } else { if (strpos($line, "Total Time") !== false) { $line = str_replace("Total Time ", "", $line); $recipe->time["total"] = RecipeParser_Times::toMinutes($line); } } } // Yield $nodes = $xpath->query('//*[@class="field-yield"]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $line = RecipeParser_Text::formatYield($line); $recipe->yield = $line; } // Ingredients $nodes = $xpath->query('//*[@class="field-ingredients"]'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } // Instructions $nodes = $xpath->query('//*[@class="field-instructions"]//li'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendInstruction($line); } // Photo $nodes = $xpath->query('//*[@property="og:image"]'); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute('content'); $recipe->photo_url = RecipeParser_Text::relativeToAbsolute($photo_url, $url); } return $recipe; }
public static function parse($html, $url) { $recipe = new RecipeParser_Recipe(); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Title $node_list = $doc->getElementsByTagName('title'); if ($node_list->length) { $value = $node_list->item(0)->nodeValue; $value = trim(str_replace("Cooks.com - Recipe - ", "", $value)); $value = trim(str_replace(" - Recipe - Cooks.com", "", $value)); $recipe->title = $value; } // This node contains all ingredients, section titles, and instructions $node_list = $xpath->query('//table[@class="hrecipe"]//td/div'); foreach ($node_list as $node) { // Can determine each piece of content by the "style" attributes. $style = $node->getAttribute("style"); // Ingredients found in a div, black text if (stripos($style, "color: BLACK;") !== false) { $ing_nodes = $xpath->query('./span[@class = "ingredient"]', $node); foreach ($ing_nodes as $ing_node) { $recipe->appendIngredient($ing_node->nodeValue); } // Instructions node } else { if ($node->getAttribute('class') == "instructions") { foreach ($node->childNodes as $child) { $line = $child->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendInstruction($line); } // Section title } else { if ($node->getAttribute("class") == "section") { $title = RecipeParser_Text::formatSectionName($node->nodeValue); $recipe->addIngredientsSection($title); if (count($recipe->instructions) > 0) { $recipe->addInstructionsSection($title); } } } } } return $recipe; }
public static function parse($html, $url) { // Get all of the standard microdata stuff we can find. $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // ---- OVERRIDES // Title $nodes = $xpath->query('//h3//strong'); if ($nodes->length) { $line = RecipeParser_Text::formatAsOneLine($nodes->item(0)->nodeValue); $recipe->title = $line; } // Yield $nodes = $xpath->query('//span[@itemprop="articleBody"]//p'); foreach ($nodes as $node) { $line = trim($node->nodeValue); if (strpos($line, "Yield") === 0 || strpos($line, "Serve") === 0) { $line = RecipeParser_Text::formatYield($line); $recipe->yield = $line; break; } } // Ingredients $nodes = $xpath->query('//span[@itemprop="articleBody"]//ul/li'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } // Instructions $nodes = $xpath->query('//span[@itemprop="articleBody"]//ol/li'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendInstruction($line); } // Image $nodes = $xpath->query('//meta[@property="og:image"]'); foreach ($nodes as $node) { $line = $node->getAttribute("content"); $recipe->photo_url = $line; break; } return $recipe; }
public static function parse($html, $url) { // Get all of the standard bits we can find. $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Titles include "recipe" if (preg_match("/ Recipe( - CHOW.com)?\$/", $recipe->title)) { $recipe->title = trim(preg_replace("/(.*) Recipe( - CHOW.com)?\$/", "\$1", $recipe->title)); } // Strip leading numbers from instructions for ($i = 0; $i < count($recipe->instructions); $i++) { for ($j = 0; $j < count($recipe->instructions[$i]['list']); $j++) { $recipe->instructions[$i]['list'][$j] = preg_replace("/^\\d+(\\w.*)\$/", "\$1", $recipe->instructions[$i]['list'][$j]); } } // Ingredients (If none parsed) if (!count($recipe->ingredients[0]['list'])) { $nodes = $xpath->query('//*[@id="ingredients_list"]//li'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } } // Instructions (If none parsed) if (!count($recipe->instructions[0]['list'])) { $nodes = $xpath->query('//*[@itemprop="recipeInstructions"]'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendInstruction($line); } } // Cleanup description if ($recipe->description) { $recipe->description = preg_replace("/^(Read our review of|This (dish|recipe) was featured as part|See more recipes) .*\$/m", "", $recipe->description); $recipe->description = preg_replace("/[\r\n]{3,}/", "\n\n", $recipe->description); $recipe->description = trim($recipe->description); } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataDataVocabulary::parse($html, $url); libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Yield, Ingredients, Instructions $found_instructions = false; $found_ingredients = false; $nodes = $xpath->query('//*[@class="field field-name-body field-type-text-with-summary field-label-hidden"]//*[@class="field-item even"]'); if ($nodes->length) { foreach ($nodes->item(0)->childNodes as $node) { $str = trim($node->nodeValue); // Yield if (!$recipe->yield && preg_match("/(makes|yields|serves|servings)/i", $str) && preg_match("/\\d/", $str)) { $recipe->yield = RecipeParser_Text::formatYield($str); continue; } // Ingredients and Instructions if ($str == "INGREDIENTS") { $found_ingredients = true; continue; } if ($str == "INSTRUCTIONS") { $found_instructions = true; continue; } if (!$found_ingredients) { continue; } else { if (!$found_instructions) { $str = RecipeParser_Text::formatAsOneLine($str); $recipe->appendIngredient($str); } else { $str = RecipeParser_Text::formatAsOneLine($str); $str = RecipeParser_Text::stripLeadingNumbers($str); $recipe->appendInstruction($str); } } } } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Overrides for data that isn't captured by their implementation of Schema.org. // Instructions $recipe->resetInstructions(); $nodes = $xpath->query('//*[@itemprop="recipeInstructions"]'); foreach ($nodes as $node) { $line = RecipeParser_Text::formatAsOneLine($node->nodeValue); $recipe->appendInstruction($line); } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataRdfDataVocabulary::parse($html, $url); libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Ingredients $recipe->resetIngredients(); $nodes = $xpath->query('//*[@class="ingredient"]'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } return $recipe; }
public static function parse($html, $url) { // Get all of the standard hrecipe stuff we can find. $recipe = RecipeParser_Parser_Microformat::parse($html, $url); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Yield $nodes = $xpath->query('//*[@name="resizeTo"]'); if ($nodes->length) { $line = trim($nodes->item(0)->getAttribute("value")) . " servings"; $recipe->yield = RecipeParser_Text::formatYield($line); } // Ingredients $recipe->resetIngredients(); $nodes = $xpath->query('//*[contains(concat(" ", normalize-space(@class), " "), " ingredient ")]'); foreach ($nodes as $node) { $parts = array(); foreach ($node->childNodes as $n) { $parts[] = $n->nodeValue; } $line = implode(' ', $parts); $line = str_replace(" ; ", "; ", $line); $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } // Instructions $recipe->resetInstructions(); $nodes = $xpath->query('//div[@class="display-field"]/p'); foreach ($nodes as $node) { $line = trim($node->nodeValue); if ($line == strtoupper($line)) { $line = RecipeParser_Text::formatSectionName($line); $recipe->addInstructionsSection($line); } else { $recipe->appendInstruction($line); } } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Ingredients $recipe->resetIngredients(); $sections = $xpath->query('//*[@id="ingredients"]//*[@class="group"]'); if ($sections->length) { // Sections foreach ($sections as $section_node) { $section_nodes = $xpath->query('.//h3', $section_node); if ($section_nodes->length) { $line = $section_nodes->item(0)->nodeValue; $line = RecipeParser_Text::formatSectionName($line); if (!empty($line)) { $recipe->addIngredientsSection($line); } } $ing_nodes = $xpath->query('.//li', $section_node); if ($ing_nodes->length) { foreach ($ing_nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } } } } // Notes $nodes = $xpath->query('.//*[@class = "body-c note-text"]'); if ($nodes->length) { $value = $nodes->item(0)->nodeValue; $value = trim(str_replace("Cook's Note", '', $value)); $recipe->notes = $value; } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataDataVocabulary::parse($html, $url); libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // // Some of the ingredient lines in on The Daily Meal do not adhere to // the usual microdata formatting. Here we fall back to looking for a // regular list within a higher-level ingredients div. // if (!empty($recipe->ingredients)) { $nodes = $xpath->query("//div[@class='content']/div[@class='ingredient']/ul/li"); foreach ($nodes as $node) { $value = RecipeParser_Text::formatAsOneLine($node->nodeValue); if (empty($value)) { continue; } if (RecipeParser_Text::matchSectionName($value)) { $value = RecipeParser_Text::formatSectionName($value); $recipe->addIngredientsSection($value); } else { $recipe->appendIngredient($value); } } } // // The Daily Meal provides servings details via Edamam's plugin. // if (!$recipe->yield) { $nodes = $xpath->query("//table[@class='edamam-data']/tr[2]/td[2]"); if ($nodes->length) { $recipe->yield = RecipeParser_Text::formatYield($nodes->item(0)->nodeValue); } } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Notes $nodes = $xpath->query('//div[@class="rd_editornote margin_bottom"]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $line = preg_replace("/Editor's Note:\\s+/", "", $line); $recipe->notes = $line; } // Override image $nodes = $xpath->query('//meta[@itemprop="image"]'); if ($nodes->length) { $line = $nodes->item(0)->getAttribute("content"); $recipe->photo_url = $line; } return $recipe; }
public static function parse($html, $url) { // Get all of the standard microdata stuff we can find. $recipe = RecipeParser_Parser_MicrodataDataVocabulary::parse($html, $url); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Ingredients $recipe->resetIngredients(); $nodes = $xpath->query('//div[@id="ingredients-box"]//ul/li'); foreach ($nodes as $node) { if ($node->getAttribute("itemprop")) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } else { $line = $node->nodeValue; $line = RecipeParser_Text::formatSEctionName($line); $recipe->addIngredientsSection($line); } } // Instructions $recipe->resetInstructions(); $nodes = $xpath->query('//*[@id="method-box"]//p'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); if ($line) { $recipe->appendInstruction($line); } } return $recipe; }
public static function parse($html, $url) { $recipe = new RecipeParser_Recipe(); libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Find the top-level node for Recipe microdata $microdata = null; $nodes = $xpath->query('//*[@itemtype="http://data-vocabulary.org/Recipe"]'); if ($nodes->length) { $microdata = $nodes->item(0); } // Parse elements if ($microdata) { // Title $nodes = $xpath->query('.//*[@itemprop="name"]', $microdata); if ($nodes->length) { $value = $nodes->item(0)->nodeValue; $value = RecipeParser_Text::formatTitle($value); $recipe->title = $value; } // Summary $nodes = $xpath->query('.//*[@itemprop="summary"]', $microdata); if ($nodes->length) { $value = trim($nodes->item(0)->nodeValue); $recipe->description = $value; } // Times $searches = array('prepTime' => 'prep', 'cookTime' => 'cook', 'totalTime' => 'total'); foreach ($searches as $itemprop => $time_key) { $nodes = $xpath->query('.//*[@itemprop="' . $itemprop . '"]', $microdata); if ($nodes->length) { if ($value = $nodes->item(0)->getAttribute('datetime')) { $value = RecipeParser_Text::iso8601ToMinutes($value); } else { if ($value = $nodes->item(0)->getAttribute('content')) { $value = RecipeParser_Text::iso8601ToMinutes($value); } else { $value = trim($nodes->item(0)->nodeValue); $value = RecipeParser_Times::toMinutes($value); } } if ($value) { $recipe->time[$time_key] = $value; } } } // Yield $line = ""; $nodes = $xpath->query('.//*[@itemprop="yield"]', $microdata); if ($nodes->length) { $line = trim($nodes->item(0)->nodeValue); } else { $nodes = $xpath->query('.//*[@itemprop="servingSize"]', $microdata); if ($nodes->length) { $line = trim($nodes->item(0)->nodeValue); } } if ($line) { $line = preg_replace('/\\s+/', ' ', $line); $recipe->yield = RecipeParser_Text::formatYield($line); } // Ingredients $nodes = null; // (data-vocabulary) if (!$nodes || !$nodes->length) { $nodes = $xpath->query('.//*[@itemprop="ingredient"]', $microdata); } if (!$nodes || !$nodes->length) { // non-standard $nodes = $xpath->query('.//*[@id="ingredients"]//li', $microdata); } if (!$nodes || !$nodes->length) { // non-standard $nodes = $xpath->query('.//*[@class="ingredients"]//li', $microdata); } foreach ($nodes as $node) { $value = $node->nodeValue; $value = RecipeParser_Text::formatAsOneLine($value); if (empty($value)) { continue; } if (RecipeParser_Text::matchSectionName($value)) { $value = RecipeParser_Text::formatSectionName($value); $recipe->addIngredientsSection($value); } else { $recipe->appendIngredient($value); } } // Instructions $found = false; // Look for markup that uses <li> tags for each instruction. if (!$found) { $nodes = $xpath->query('.//*[@itemprop="instructions"]//li', $microdata); if ($nodes->length) { RecipeParser_Text::parseInstructionsFromNodes($nodes, $recipe); $found = true; } } // Some sites will use an "instruction" class for each line. if (!$found) { $nodes = $xpath->query('.//*[@itemprop="instruction"]//*[contains(concat(" ", normalize-space(@class), " "), " instruction ")]', $microdata); if ($nodes->length) { RecipeParser_Text::parseInstructionsFromNodes($nodes, $recipe); $found = true; } } // Either multiple instrutions nodes, or one node with a blob of text. if (!$found) { $nodes = $xpath->query('.//*[@itemprop="instructions"]', $microdata); if ($nodes->length > 1) { // Multiple nodes RecipeParser_Text::parseInstructionsFromNodes($nodes, $recipe); $found = true; } else { if ($nodes->length == 1) { // Blob $str = $nodes->item(0)->nodeValue; RecipeParser_Text::parseInstructionsFromBlob($str, $recipe); $found = true; } } } // Photo $photo_url = ""; if (!$photo_url) { // try to find open graph url $nodes = $xpath->query('//meta[@property="og:image"]'); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute('content'); } } if (!$photo_url) { $nodes = $xpath->query('.//*[@itemprop="photo"]', $microdata); if ($nodes->length) { if ($nodes->item(0)->hasAttribute('src')) { $photo_url = $nodes->item(0)->getAttribute('src'); } else { if ($nodes->item(0)->hasAttribute('content')) { $photo_url = $nodes->item(0)->getAttribute('content'); } } } } if (!$photo_url) { // for <img> as sub-node of class="photo" $nodes = $xpath->query('.//*[@itemprop="photo"]//img', $microdata); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute('src'); } } if ($photo_url) { $recipe->photo_url = RecipeParser_Text::relativeToAbsolute($photo_url, $url); } // Credits $nodes = $xpath->query('.//*[@itemprop="author"]', $microdata); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $recipe->credits = RecipeParser_Text::formatCredits($line); } } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Yield $nodes = $xpath->query('//*[@class="prep_box"]'); foreach ($nodes as $node) { $line = $node->nodeValue; if (preg_match("/Number of Servings: (\\d+)/", $line, $m)) { $recipe->yield = RecipeParser_Text::formatYield($m[1]); } } // Instructions $recipe->resetInstructions(); $str = ""; $nodes = $xpath->query('//*[@itemprop="recipeInstructions"]'); if ($nodes->length) { $children = $nodes->item(0)->childNodes; // This is a piece of HTML that has <br> tags for breaks in each instruction. // Rather than just getting nodeValue, I want to preserve the <br> tags. So I'm // looking for them as nodes and appending them to the string. Any other nodes // (either #text or other, e.g. <a href="">) get passed along into the string as // nodeValue. foreach ($children as $child) { if ($child->nodeName == "br") { $str .= "<br>"; } else { $line = trim($child->nodeValue); if (!empty($line)) { $str .= $line; } } } $lines = explode("<br>", $str); foreach ($lines as $line) { if (empty($line)) { continue; } else { if (RecipeParser_Text::matchSectionName($line)) { $line = RecipeParser_Text::formatSectionName($line); $recipe->addInstructionsSection($line); } else { if (!empty($line)) { $line = RecipeParser_Text::formatAsOneLine($line); $line = RecipeParser_Text::stripLeadingNumbers($line); if (stripos($line, "Recipe submitted by SparkPeople") === 0) { continue; } if (stripos($line, "Number of Servings:") === 0) { continue; } $recipe->appendInstruction($line); } } } } } return $recipe; }
public static function parse($html, $url) { // Get all of the standard microdata stuff we can find. $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Ingredients $recipe->resetIngredients(); $nodes = $xpath->query('//div[@class="col6 ingredients"]/*'); foreach ($nodes as $node) { // Extract ingredients from <ul> <li>. if ($node->nodeName == 'ul') { $ing_nodes = $node->childNodes; foreach ($ing_nodes as $ing_node) { // Find <li> with itemprop="ingredients" for each ingredient. if ($ing_node->nodeName == 'li' && $ing_node->getAttribute("itemprop") == "ingredients") { $line = trim($ing_node->nodeValue); // Section titles might be all uppercase ingredients if ($line == strtoupper($line)) { $line = RecipeParser_Text::formatSectionName($line); $recipe->addIngredientsSection($line); continue; } // Ingredient lines if (stripos($line, "copyright") !== false) { continue; } else { if (stripos($line, "recipe follows") !== false) { continue; } else { $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } } // Section titles } else { if ($ing_node->nodeName == 'li' && $ing_node->getAttribute("class") == "subtitle") { $line = trim($ing_node->nodeValue); $line = RecipeParser_Text::formatSectionName($line); $recipe->addIngredientsSection($line); } } } continue; } } // Instructions $recipe->resetInstructions(); $nodes = $xpath->query('//*[@itemprop="recipeInstructions"]/*'); foreach ($nodes as $node) { if ($node->nodeName == "span") { $line = RecipeParser_Text::formatSectionName($node->nodeValue); $recipe->addInstructionsSection($line); } else { if ($node->nodeName == "p") { $line = RecipeParser_Text::formatAsOneLine($node->nodeValue); if (!preg_match("/^Photograph/i", $line)) { $recipe->appendInstruction($line); } } } } // See if we've captured a chef's photo, and delete it (if so). if ($recipe->photo_url) { $nodes = $xpath->query('//a[@itemprop="url"]/img[@itemprop="image"]'); if ($nodes->length > 0) { $url = $nodes->item(0)->getAttribute("src"); if ($recipe->photo_url == $url) { $recipe->photo_url = ""; } } } return $recipe; }
public static function parse($html, $url) { $recipe = new RecipeParser_Recipe(); libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Title $nodes = $xpath->query('//*[@property="v:name"]'); if ($nodes->length) { $recipe->title = trim($nodes->item(0)->nodeValue); } // Summary $nodes = $xpath->query('//*[@property="v:summary"]'); if ($nodes->length) { $value = trim($nodes->item(0)->nodeValue); $recipe->description = $value; } // Times $searches = array('v:prepTime' => 'prep', 'v:cookTime' => 'cook', 'v:totalTime' => 'total'); foreach ($searches as $itemprop => $time_key) { $nodes = $xpath->query('//*[@property="' . $itemprop . '"]'); if ($nodes->length) { if ($value = $nodes->item(0)->getAttribute('content')) { $value = RecipeParser_Text::iso8601ToMinutes($value); } else { $value = trim($nodes->item(0)->nodeValue); $value = RecipeParser_Times::toMinutes($value); } if ($value) { $recipe->time[$time_key] = $value; } } } // Yield $nodes = $xpath->query('//*[@property="v:yield"]'); if ($nodes->length) { $line = trim($nodes->item(0)->nodeValue); $line = preg_replace('/\\s+/', ' ', $line); $recipe->yield = RecipeParser_Text::formatYield($line); } // Ingredients $nodes = null; // (data-vocabulary) $nodes = $xpath->query('//*[@rel="v:ingredient"]'); foreach ($nodes as $node) { $value = $node->nodeValue; $value = RecipeParser_Text::formatAsOneLine($value); if (empty($value)) { continue; } if (RecipeParser_Text::matchSectionName($value)) { $value = RecipeParser_Text::formatSectionName($value); $recipe->addIngredientsSection($value); } else { $recipe->appendIngredient($value); } } // Instructions $found = false; // Some sites will use an "instruction" class for each line. if (!$found) { $nodes = $xpath->query('//*[@property="v:instructions"]//*[@property="v:instruction"]'); if ($nodes->length) { RecipeParser_Text::parseInstructionsFromNodes($nodes, $recipe); $found = true; } } // Look for markup that uses <li>, <p> or other tags for each instruction. $search_sub_nodes = array("p", "li"); while (!$found && ($tag = array_pop($search_sub_nodes))) { $nodes = $xpath->query('//*[@property="v:instructions"]//' . $tag); if ($nodes->length) { RecipeParser_Text::parseInstructionsFromNodes($nodes, $recipe); $found = true; } } // Either multiple instrutions nodes, or one node with a blob of text. if (!$found) { $nodes = $xpath->query('//*[@property="v:instructions"]'); if ($nodes->length > 1) { // Multiple nodes RecipeParser_Text::parseInstructionsFromNodes($nodes, $recipe); $found = true; } else { if ($nodes->length == 1) { // Blob $str = $nodes->item(0)->nodeValue; RecipeParser_Text::parseInstructionsFromBlob($str, $recipe); $found = true; } } } // Photo $photo_url = ""; $nodes = $xpath->query('//*[@rel="v:photo"]'); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute('src'); } if (!$photo_url) { // for <img> as sub-node of rel="v:photo" $nodes = $xpath->query('//*[@rel="v:photo"]//img'); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute('src'); } } if ($photo_url) { $recipe->photo_url = RecipeParser_Text::formatPhotoUrl($photo_url, $url); } // Credits $nodes = $xpath->query('//*[@property="v:author"]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $recipe->credits = RecipeParser_Text::formatCredits($line); } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // OVERRIDES for epicurious // Prep Times $nodes = $xpath->query('//*[@class="summary_data"]'); if ($nodes->length) { foreach ($nodes as $node) { if (preg_match('/ACTIVE/', $node->nodeValue)) { $ing_nodes = $node->childNodes; foreach ($ing_nodes as $ing_node) { if ($ing_node->nodeName == "span") { $recipe->prep_time = RecipeParser_Text::formatAsOneLine($ing_node->nodeValue); } } } else { if (preg_match('/TOTAL/', $node->nodeValue)) { $ing_nodes = $node->childNodes; foreach ($ing_nodes as $ing_node) { if ($ing_node->nodeName == "span") { $recipe->total_time = RecipeParser_Text::formatAsOneLine($ing_node->nodeValue); } } } } } } // Total Time $nodes = $xpath->query('//*[@itemprop="totalTime"]'); if ($nodes->length) { $value = $nodes->item(0)->getAttribute("content"); $recipe->time['total'] = RecipeParser_Text::iso8601ToMinutes($value); } // Ingredients $recipe->resetIngredients(); $nodes = $xpath->query('//div[@id = "ingredients"]/*'); foreach ($nodes as $node) { // <strong> contains ingredient section names if ($node->nodeName == 'strong') { $line = RecipeParser_Text::formatSectionName($node->nodeValue); $recipe->addIngredientsSection($line); continue; } // Extract ingredients from inside of <ul class="ingredientsList"> if ($node->nodeName == 'ul') { // Child nodes should all be <li> $ing_nodes = $node->childNodes; foreach ($ing_nodes as $ing_node) { if ($ing_node->nodeName == 'li') { $line = trim($ing_node->nodeValue); $recipe->appendIngredient($line); } } } } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_Microformat::parse($html, $url); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Description $description = ""; $nodes = $xpath->query('//div[@id="recipe"]/p/i'); foreach ($nodes as $node) { $line = trim($node->nodeValue); if (strpos($line, "Adapted from") === false) { $description .= $line . "\n\n"; } } $description = trim($description); $recipe->description = $description; // Ingredients $recipe->resetIngredients(); $lines = array(); // Add ingredients to blob $nodes = $xpath->query('//div[@id="recipe"]/blockquote/p'); foreach ($nodes as $node) { foreach ($node->childNodes as $child) { $line = trim($child->nodeValue); switch ($child->nodeName) { case "strong": case "b": if (strpos($line, ":") === false) { $line .= ":"; } $lines[] = $line; break; case "#text": case "div": case "p": $lines[] = $line; break; } } } foreach ($lines as $line) { if (RecipeParser_Text::matchSectionName($line)) { $recipe->addIngredientsSection(RecipeParser_Text::formatSectionName($line)); } else { $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } } // Instructions $recipe->resetInstructions(); $lines = array(); $nodes = $xpath->query('//div[@id="recipe"]/*'); $passed_ingredients = false; foreach ($nodes as $node) { if ($node->nodeName == "blockquote") { $passed_ingredients = true; continue; } if ($node->nodeName == "p") { if ($passed_ingredients) { $line = trim($node->nodeValue); // Finished with ingredients once we hit "Adapted" notes or any <p> // with a class attribute. if (stripos($line, "Adapted from") !== false) { break; } else { if ($node->getAttribute("class")) { break; } } // Servings? if (stripos($line, "Serves ") === 0) { $recipe->yield = RecipeParser_Text::formatYield($line); continue; } $recipe->appendInstruction(RecipeParser_Text::formatAsOneLine($node->nodeValue)); } } } return $recipe; }
public static function parse($html, $url) { if (strpos($url, "www.nytimes.com/recipes/") !== false) { // // "RECIPES" SECTION // $recipe = new RecipeParser_Recipe(); libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Title $nodes = $xpath->query('//h1[@class="recipe-title recipeName"]'); if ($nodes->length) { $value = $nodes->item(0)->nodeValue; $value = RecipeParser_Text::formatTitle($value); $recipe->title = $value; } // Yield $nodes = $xpath->query('//*[@itemprop="recipeYield"]'); if ($nodes->length) { $value = $nodes->item(0)->nodeValue; $value = RecipeParser_Text::formatYield($value); $recipe->yield = $value; } // Ingredients $nodes = $xpath->query('//div[@class="ingredientsGroup"]/*'); foreach ($nodes as $node) { if ($node->nodeName == "h3") { $value = trim($node->nodeValue); if (!preg_match('/^Ingredients:?$/i', $value)) { $value = RecipeParser_Text::formatSectionName($value); $recipe->addIngredientsSection($value); } } else { foreach ($node->childNodes as $child) { $value = trim($child->nodeValue); $recipe->appendIngredient($value); } } } // Instructions $nodes = $xpath->query('//*[@itemprop="recipeInstructions"]/dd'); foreach ($nodes as $node) { $value = $node->nodeValue; $value = RecipeParser_Text::formatAsOneLine($value); $recipe->appendInstruction($value); } // Notes if (!$recipe->notes) { $nodes = $xpath->query('//div[@class="yieldNotesGroup"]//*[@class="note"]'); if ($nodes->length) { $value = trim($nodes->item(0)->nodeValue); $value = preg_replace("/^Notes?:?\\s*/i", '', $value); $recipe->notes = trim($value); } } } else { // // DINING SECTION RECIPES // $recipe = new RecipeParser_Recipe(); libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Title $nodes = $xpath->query('//div[@id = "article"]//h1'); if ($nodes->length) { $value = trim($nodes->item(0)->nodeValue); $recipe->title = $value; } // Time and Yield $nodes = $xpath->query('//div[@id = "article"]//p'); foreach ($nodes as $node) { $text = trim($node->nodeValue); if (preg_match('/^Yield:? (.+)/', $text, $m)) { $recipe->yield = RecipeParser_Text::formatYield($m[1]); } else { if (preg_match('/^Time:? (.+)/', $text, $m)) { $str = trim($m[1]); $str = preg_replace('/About (.+)/', '$1', $str); $str = preg_replace('/(.+) plus.*/', '$1', $str); $recipe->time['total'] = RecipeParser_Times::toMinutes($str); } } } // Ingredients $nodes = $xpath->query('//div[@class="recipeIngredientsList"]/p'); foreach ($nodes as $node) { $line = trim($node->nodeValue); // Section names if ($line && $line == strtoupper($line)) { $line = RecipeParser_Text::formatSectionName($line); $recipe->addIngredientsSection($line); continue; } $recipe->appendIngredient($line); } // Instructions and notes $nodes = $xpath->query('//div[@class="articleBody"]//p'); if (!$nodes->length) { $nodes = $xpath->query('//div[@id="articleBody"]//p'); } $notes = ''; $in_notes_section = false; foreach ($nodes as $node) { $line = trim($node->nodeValue); // Skip some of the useless lines if (preg_match('/^(Adapted from|Time|Yield)/i', $line)) { continue; } // Instructions start with line numbers if (!$in_notes_section && preg_match('/^\\d+\\./', $line)) { $line = RecipeParser_Text::stripLeadingNumbers($line); $recipe->appendInstruction($line); continue; } // Look for lines that start the notes section. $note = ''; if (preg_match('/^Notes?:?(.*)/i', $line, $m)) { $in_notes_section = true; $note = trim($m[1]); } else { if ($in_notes_section) { $note = $line; } } if ($note) { $notes .= $note . "\n\n"; } } if ($notes) { $notes = str_replace(" ", " ", $notes); // Some unnecessary spaces $notes = trim($notes); $recipe->notes = $notes; } // Photo $nodes = $xpath->query('//div[@class="image"]//img'); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute('src'); $photo_url = str_replace('-articleInline.jpg', '-popup.jpg', $photo_url); $recipe->photo_url = RecipeParser_Text::formatPhotoUrl($photo_url, $url); } } return $recipe; }
public static function parse($html, $url) { $recipe = new RecipeParser_Recipe(); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Title $nodes = $xpath->query('//h1[@itemprop="name"]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $line = RecipeParser_Text::formatTitle($line); $recipe->title = $line; } // Description $nodes = $xpath->query('//*[@itemprop="description"]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->description = $line; } // Author $nodes = $xpath->query('//span[@itemprop="author"]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $line = RecipeParser_Text::formatCredits($line); $recipe->credits = $line; } // Prep Times $nodes = $xpath->query('//*[@itemprop="prepTime"]'); if ($nodes->length) { $value = $nodes->item(0)->getAttribute("content"); $recipe->time['prep'] = RecipeParser_Text::iso8601ToMinutes($value); } // Total Time $nodes = $xpath->query('//*[@itemprop="totalTime"]'); if ($nodes->length) { $value = $nodes->item(0)->getAttribute("content"); $recipe->time['total'] = RecipeParser_Text::iso8601ToMinutes($value); } // Yield $nodes = $xpath->query('//*[@itemprop="recipeyield"]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $recipe->yield = RecipeParser_Text::formatYield($line); } // Ingredients $nodes = $xpath->query('//*[@itemprop="ingredients"]'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } // Instructions $nodes = $xpath->query('//*[@itemprop="recipeinstructions"]/li'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendInstruction($line); } // Photo $nodes = $xpath->query('//meta[@property="og:image"]'); if ($nodes->length) { $line = $nodes->item(0)->getAttribute("content"); $recipe->photo_url = $line; } return $recipe; }
public static function parse($html, $url) { // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); // OVERRIDES FOR ABOUT.COM // Title $nodes = $xpath->query('//*[@itemprop="headline name"]'); if ($nodes->length) { $value = trim($nodes->item(0)->nodeValue); $recipe->title = RecipeParser_Text::formatTitle($value); } // Credits $nodes = $xpath->query('//*[@itemprop="author"]//*[@itemprop="name"]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $recipe->credits = RecipeParser_Text::formatCredits($line . ", About.com"); } // Ingredients $recipe->resetIngredients(); $nodes = $xpath->query('//*[@itemprop="ingredients"]'); foreach ($nodes as $node) { $value = $node->nodeValue; $value = RecipeParser_Text::formatAsOneLine($value); if (RecipeParser_Text::matchSectionName($value) || $node->childNodes->item(0)->nodeName == "strong" || $node->childNodes->item(0)->nodeName == "b") { $value = RecipeParser_Text::formatSectionName($value); $recipe->addIngredientsSection($value); } else { $recipe->appendIngredient($value); } } // Instructions $recipe->resetInstructions(); $nodes = $xpath->query('//div[@itemprop="recipeInstructions"]'); foreach ($nodes as $node) { $text = trim($node->nodeValue); $lines = preg_split("/[\n\r]+/", $text); for ($i = count($lines) - 1; $i >= 0; $i--) { $lines[$i] = trim($lines[$i]); // Remove ends of lines that have the word "recipes" squashed up against // another word, which seems to happen with long lists of related // recipe links. // Remove lines that have the phrase "Xxxxx Recipes and More". // Remove lines that have the phrase "Xxxxx Recipes | Xxxxx". // Remove mentions of newsletters. $lines[$i] = preg_replace("/(.*)recipes\\w/i", "\$1", $lines[$i]); $lines[$i] = preg_replace("/(.*)More .* Recipes.*/", "\$1", $lines[$i]); $lines[$i] = preg_replace("/(.*)Recipes and More.*/", "\$1", $lines[$i]); $lines[$i] = preg_replace("/(.*)Recipes \\| .*/", "\$1", $lines[$i]); $lines[$i] = preg_replace("/(.*)Recipe Newsletter.*/", "\$1", $lines[$i]); // Look for a line in the instructions that looks like a yield. if (strpos($lines[$i], "Makes ") === 0) { $recipe->yield = substr($lines[$i], 6); $lines[$i] = ''; continue; } } foreach ($lines as $line) { $line = trim($line); if (empty($line)) { continue; } if (strtolower($line) == "preparation") { continue; } // Match section names that read something like "---For the cake: Raise the oven temperature..." if (preg_match("/^(?:-{2,})?For the (.+)\\: (.*)\$/i", $line, $m)) { $section = $m[1]; $section = RecipeParser_Text::formatSectionName($section); $recipe->addInstructionsSection($section); // Reset the value of $line, without the section name. $line = ucfirst($m[2]); } $recipe->appendInstruction($line); } } return $recipe; }
public function test_format_one_line() { $str = "\tThis classic recipe comes\n \n\tfrom my mom -- \n the walnuts add a nice\ntexture to the bread.\n"; $test = "This classic recipe comes from my mom -- the walnuts add a nice texture to the bread."; $this->assertEquals($test, RecipeParser_Text::formatAsOneLine($str)); }
public static function parse($html, $url) { $recipe = new RecipeParser_Recipe(); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Title $nodes = $xpath->query('//*[@class="rTitle fn"]'); if ($nodes->length) { $line = RecipeParser_Text::formatTitle($nodes->item(0)->nodeValue); $recipe->title = $line; } // Yield $nodes = $xpath->query('//*[contains(concat(" ", normalize-space(@class), " "), " yield ")]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $recipe->yield = RecipeParser_Text::formatYield($line); } // Times $nodes = $xpath->query('//*[contains(concat(" ", normalize-space(@class), " "), " prepTime ")]/span'); if ($nodes->length) { $line = $nodes->item(1)->getAttribute("title"); $recipe->time['prep'] = RecipeParser_Text::iso8601ToMinutes($line); } $nodes = $xpath->query('//*[contains(concat(" ", normalize-space(@class), " "), " rspec-cook-time ")]/span'); if ($nodes->length) { $line = $nodes->item(1)->getAttribute("title"); $recipe->time['cook'] = RecipeParser_Text::iso8601ToMinutes($line); } $nodes = $xpath->query('//*[contains(concat(" ", normalize-space(@class), " "), " totaltime ")]/span'); if ($nodes->length) { $line = $nodes->item(1)->getAttribute("title"); $recipe->time['total'] = RecipeParser_Text::iso8601ToMinutes($line); } // Ingredients $nodes = $xpath->query('//*[@class="ingredient"]'); foreach ($nodes as $node) { $line = RecipeParser_Text::formatAsOneLine($node->nodeValue); $recipe->appendIngredient($line); } // Instructions $nodes = $xpath->query('//*[@class="instructions"]'); if ($nodes->length) { $blob = ""; foreach ($nodes->item(0)->childNodes as $node) { $blob .= RecipeParser_Text::formatAsOneLine($node->nodeValue) . " "; if ($node->nodeName == "p") { $blob .= "\n\n"; } } // Minor cleanup $blob = str_replace(" , ", ", ", $blob); $blob = str_replace(" . ", ". ", $blob); $blob = str_replace(" ", " ", $blob); foreach (explode("\n\n", $blob) as $line) { $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendInstruction($line); } } // Photo $nodes = $xpath->query('//a[@class="img-enlarge"]'); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute("href"); $photo_url = RecipeParser_Text::relativeToAbsolute($photo_url, $url); $recipe->photo_url = $photo_url; } return $recipe; }
public static function parse($html, $url) { // Get all of the standard microdata stuff we can find. $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); // Turn off libxml errors to prevent mismatched tag warnings. libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // --- Allrecipes allows for custom recipes that use a different // --- template than their standard content. This template is not currently // --- using schema.org/Recipe. So we'll look for fields that need to be // --- overridden. // Title if (!$recipe->title) { $node_list = $xpath->query('//h1[@id = "itemTitle"]'); if ($node_list->length) { $value = $node_list->item(0)->nodeValue; $value = trim($value); $recipe->title = $value; } } // Yield if (!$recipe->yield) { $node_list = $xpath->query('//div[@class = "servings-form"]//span[@class = "yield yieldform"]'); if ($node_list->length) { $value = $node_list->item(0)->nodeValue; $recipe->yield = $value; } } // Times $searches = array('liPrep' => 'prep', 'liCook' => 'cook', 'liTotal' => 'total'); foreach ($searches as $id_name => $time_key) { $nodes = $xpath->query('.//*[@id="' . $id_name . '"]'); if ($nodes->length) { $value = RecipeParser_Text::formatAsOneLine($nodes->item(0)->nodeValue); $value = trim(preg_replace("/(COOK|PREP|READY IN)/", "", $value)); $value = RecipeParser_Times::toMinutes($value); if ($value) { $recipe->time[$time_key] = $value; } } } // Ingredients if (!count($recipe->ingredients[0]["list"])) { $node_list = $xpath->query('//li[contains(concat(" ", normalize-space(@class), " "), " ingredient ")]'); foreach ($node_list as $node) { $line = trim(strip_tags($node->nodeValue)); if (preg_match("/^(.+):\$/", $line, $m)) { $recipe->addIngredientsSection(ucfirst(strtolower($m[1]))); } else { if ($line) { $recipe->appendIngredient($line); } } } } // Instructions if (!count($recipe->instructions[0]["list"])) { $nodes = $xpath->query('//div[@class="directions"]//ol/li'); foreach ($nodes as $node) { $line = RecipeParser_Text::formatAsOneLine($node->nodeValue); if (preg_match("/^(.+):\$/", $line, $m)) { $recipe->addInstructionsSection(ucfirst(strtolower($m[1]))); } else { if ($line) { $recipe->appendInstruction($line); } } } } // Photo URL // Get larger images if ($recipe->photo_url) { $recipe->photo_url = str_replace('/userphoto/small/', '/userphoto/big/', $recipe->photo_url); $recipe->photo_url = str_replace('/userphotos/140x140/', '/userphotos/250x250/', $recipe->photo_url); } return $recipe; }
public static function parse($html, $url) { $recipe = new RecipeParser_Recipe(); libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); $microdata = null; $nodes = $xpath->query('//*[contains(@itemtype, "//schema.org/Recipe") or contains(@itemtype, "//schema.org/recipe")]'); if ($nodes->length) { $microdata = $nodes->item(0); } // Parse elements if ($microdata) { // Title $nodes = $xpath->query('.//*[@itemprop="name"]', $microdata); if ($nodes->length) { $value = trim($nodes->item(0)->nodeValue); $recipe->title = RecipeParser_Text::formatTitle($value); } // Summary $nodes = $xpath->query('.//*[@itemprop="description"]', $microdata); if ($nodes->length) { $value = $nodes->item(0)->nodeValue; $value = RecipeParser_Text::formatAsParagraphs($value); $recipe->description = $value; } // Times $searches = array('prepTime' => 'prep', 'cookTime' => 'cook', 'totalTime' => 'total'); foreach ($searches as $itemprop => $time_key) { $nodes = $xpath->query('.//*[@itemprop="' . $itemprop . '"]', $microdata); if ($nodes->length) { if ($value = $nodes->item(0)->getAttribute('content')) { $value = RecipeParser_Text::iso8601ToMinutes($value); } else { if ($value = $nodes->item(0)->getAttribute('datetime')) { $value = RecipeParser_Text::iso8601ToMinutes($value); } else { $value = trim($nodes->item(0)->nodeValue); $value = RecipeParser_Times::toMinutes($value); } } if ($value) { $recipe->time[$time_key] = $value; } } } // Yield $nodes = $xpath->query('.//*[@itemprop="recipeYield"]', $microdata); if (!$nodes->length) { $nodes = $xpath->query('.//*[@itemprop="recipeyield"]', $microdata); } if ($nodes->length) { if ($nodes->item(0)->hasAttribute('content')) { $line = $nodes->item(0)->getAttribute('content'); } else { $line = $nodes->item(0)->nodeValue; } $recipe->yield = RecipeParser_Text::formatYield($line); } // Ingredients $nodes = $xpath->query('//*[@itemprop="ingredients"]'); foreach ($nodes as $node) { $value = $node->nodeValue; $value = RecipeParser_Text::formatAsOneLine($value); if (empty($value)) { continue; } if (strlen($value) > 150) { // probably a mistake, like a run-on of existing ingredients? continue; } if (RecipeParser_Text::matchSectionName($value)) { $value = RecipeParser_Text::formatSectionName($value); $recipe->addIngredientsSection($value); } else { $recipe->appendIngredient($value); } } // Instructions $found = false; // Look for markup that uses <li> tags for each instruction. if (!$found) { $nodes = $xpath->query('//*[@itemprop="recipeInstructions"]//li'); if ($nodes->length) { RecipeParser_Text::parseInstructionsFromNodes($nodes, $recipe); $found = true; } } // Look for instructions as direct descendents of "recipeInstructions". if (!$found) { $nodes = $xpath->query('//*[@itemprop="recipeInstructions"]/*'); if ($nodes->length) { RecipeParser_Text::parseInstructionsFromNodes($nodes, $recipe); $found = true; } } // Some sites will use an "instruction" class for each line. if (!$found) { $nodes = $xpath->query('.//*[@itemprop="recipeInstructions"]//*[contains(concat(" ", normalize-space(@class), " "), " instruction ")]'); if ($nodes->length) { RecipeParser_Text::parseInstructionsFromNodes($nodes, $recipe); $found = true; } } // Either multiple recipeInstructions nodes, or one node with a blob of text. if (!$found) { $nodes = $xpath->query('.//*[@itemprop="recipeInstructions"]'); if ($nodes->length > 1) { // Multiple nodes RecipeParser_Text::parseInstructionsFromNodes($nodes, $recipe); $found = true; } else { if ($nodes->length == 1) { // Blob $str = $nodes->item(0)->nodeValue; RecipeParser_Text::parseInstructionsFromBlob($str, $recipe); $found = true; } } } // Photo $photo_url = ""; if (!$photo_url) { // try to find open graph url $nodes = $xpath->query('//meta[@property="og:image"]'); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute('content'); } } if (!$photo_url) { $nodes = $xpath->query('.//*[@itemprop="image"]', $microdata); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute('src'); } } if (!$photo_url) { // for <img> as sub-node of class="photo" $nodes = $xpath->query('.//*[@itemprop="image"]//img', $microdata); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute('src'); } } if ($photo_url) { $recipe->photo_url = RecipeParser_Text::formatPhotoUrl($photo_url, $url); } // Credits $line = ""; $nodes = $xpath->query('.//*[@itemprop="author"]', $microdata); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; } $nodes = $xpath->query('.//*[@itemprop="publisher"]', $microdata); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; } $recipe->credits = RecipeParser_Text::formatCredits($line); } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Yield $nodes = $xpath->query('//li[@class="credit"]'); foreach ($nodes as $node) { $line = $node->nodeValue; if (stripos($line, "servings") !== false) { $line = preg_replace("/servings\\:?.*(\\d+)/i", "\$1", $line); $line = RecipeParser_Text::formatYield($line); $recipe->yield = $line; } } // Description $nodes = $xpath->query('//*[@itemprop="page-dek"]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->description = $line; } // Notes $line = ""; $nodes = $xpath->query('//*[@class="note-text"]'); foreach ($nodes as $node) { $line .= trim($node->nodeValue) . "\n\n"; } $line = rtrim($line); $recipe->notes = $line; // Ingredients $recipe->resetIngredients(); $sections = $xpath->query('//*[@class="components-group"]'); if ($sections->length) { // Sections foreach ($sections as $section_node) { $section_nodes = $xpath->query('.//*[@class="components-group-header"]', $section_node); if ($section_nodes->length) { $line = $section_nodes->item(0)->nodeValue; $line = RecipeParser_Text::formatSectionName($line); if (!empty($line)) { $recipe->addIngredientsSection($line); } } $ing_nodes = $xpath->query('.//*[@class="components-item"]', $section_node); if ($ing_nodes->length) { foreach ($ing_nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } } } } // Instructions $recipe->resetInstructions(); $nodes = $xpath->query('//*[@class="directions-item"]'); foreach ($nodes as $node) { $line = RecipeParser_Text::formatAsOneLine($node->nodeValue); $recipe->appendInstruction($line); } // Photo URL $nodes = $xpath->query('//img[@itemprop="image"]'); if ($nodes->length) { $photo_url = $nodes->item(0)->getAttribute("data-original"); $recipe->photo_url = RecipeParser_Text::relativeToAbsolute($photo_url, $url); } return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_MicrodataSchema::parse($html, $url); libxml_use_internal_errors(true); $doc = new DOMDocument(); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); // Times $nodes = $xpath->query('//*[@class="recipePartAttributes recipePartPrimaryAttributes"]//li'); if ($nodes->length) { foreach ($nodes as $node) { if (trim($node->childNodes->item(1)->nodeValue) == "Prep Time") { $line = trim($node->childNodes->item(3)->nodeValue); $recipe->time['prep'] = RecipeParser_Times::toMinutes($line); continue; } if (trim($node->childNodes->item(1)->nodeValue) == "Total Time") { $line = trim($node->childNodes->item(3)->nodeValue); $recipe->time['total'] = RecipeParser_Times::toMinutes($line); continue; } } } // Yield $nodes = $xpath->query('//*[@class="recipePartAttributes recipePartSecondaryAttributes"]//li'); if ($nodes->length) { foreach ($nodes as $node) { if (trim($node->childNodes->item(1)->nodeValue) == "Servings") { $line = trim($node->childNodes->item(3)->nodeValue); $recipe->yield = RecipeParser_Text::formatYield($line); } } } // Ingredients $recipe->resetIngredients(); $groups = $xpath->query('//*[@class="recipePartIngredientGroup"]'); foreach ($groups as $group) { $nodes = $xpath->query('.//h2', $group); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $line = RecipeParser_Text::formatSectionName($line); $recipe->addIngredientsSection($line); } $nodes = $xpath->query('.//*[@itemprop="ingredients"]', $group); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } } // Notes / footnotes $notes = array(); $nodes = $xpath->query('//div[@class="recipePartTipsInfo"]'); foreach ($nodes as $node) { $line = trim($node->nodeValue); $notes[] = $line; } $recipe->notes = implode("\n\n", $notes); $recipe->notes = RecipeParser_Text::formatAsParagraphs($recipe->notes); // Fix description $recipe->description = trim(preg_replace("/Servings \\# \\d+/", "", $recipe->description)); return $recipe; }
public static function parse($html, $url) { $recipe = RecipeParser_Parser_Microformat::parse($html, $url); libxml_use_internal_errors(true); $html = mb_convert_encoding($html, 'HTML-ENTITIES', "UTF-8"); $doc = new DOMDocument(); $doc->loadHTML('<?xml encoding="UTF-8">' . $html); $xpath = new DOMXPath($doc); if (!$recipe->title) { $nodes = $xpath->query('//div[@itemprop="name"]'); if ($nodes->length) { $line = $nodes->item(0)->nodeValue; $line = RecipeParser_Text::formatTitle($line); $recipe->title = $line; } } if (!$recipe->yield) { $nodes = $xpath->query('//div[@class="box"]/div'); foreach ($nodes as $node) { $line = trim($node->nodeValue); if (stripos($line, "makes") === 0) { $line = RecipeParser_Text::formatYield($line); $recipe->yield = $line; break; } } } if (!count($recipe->ingredients[0]["list"])) { $nodes = $xpath->query('//ul[@class="ingredients"]'); if ($nodes->length) { $nodes = $nodes->item(0)->childNodes; $str = ""; foreach ($nodes as $node) { if (in_array($node->nodeName, array("li"))) { $line = $node->nodeValue; $str .= $line . "<br>"; } } $lines = explode("<br>", $str); foreach ($lines as $line) { $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendIngredient($line); } } } if (!count($recipe->instructions[0]["list"])) { $nodes = $xpath->query('//div[@class="instructions"]/ol/li'); foreach ($nodes as $node) { $line = $node->nodeValue; $line = RecipeParser_Text::formatAsOneLine($line); $recipe->appendInstruction($line); } } if (!$recipe->photo_url) { $nodes = $xpath->query('//meta[@property="og:image"]'); foreach ($nodes as $node) { $line = $node->getAttribute("content"); if (strpos($line, "wp-content") !== false) { $recipe->photo_url = $line; break; } } } return $recipe; }