Please can somebody show me a simple example of parsing some HTML using libxml.
#import <libxml2/libxml/HTMLparser.h>
NSString *html = @"<ul>"
"<li><input type=\"image\" name=\"input1\" value=\"string1value\" /></li>"
"<li><input type=\"image\" name=\"input2\" value=\"string2value\" /></li>"
"</ul>"
"<span class=\"spantext\"><b>Hello World 1</b></span>"
"<span class=\"spantext\"><b>Hello World 2</b></span>";
1) Say I want to parse the value of the input whose name = input2.
Should output "string2value".
2) Say I want to parse the inner contents of each span tag whose class = spantext.
Should output: "Hello World 1" and "Hello World 2".
I used Ben Reeves' HTML Parser to achieve what I wanted:
NSError *error = nil;
NSString *html =
@"<ul>"
"<li><input type='image' name='input1' value='string1value' /></li>"
"<li><input type='image' name='input2' value='string2value' /></li>"
"</ul>"
"<span class='spantext'><b>Hello World 1</b></span>"
"<span class='spantext'><b>Hello World 2</b></span>";
HTMLParser *parser = [[HTMLParser alloc] initWithString:html error:&error];
if (error) {
NSLog(@"Error: %@", error);
return;
}
HTMLNode *bodyNode = [parser body];
NSArray *inputNodes = [bodyNode findChildTags:@"input"];
for (HTMLNode *inputNode in inputNodes) {
if ([[inputNode getAttributeNamed:@"name"] isEqualToString:@"input2"]) {
NSLog(@"%@", [inputNode getAttributeNamed:@"value"]); //Answer to first question
}
}
NSArray *spanNodes = [bodyNode findChildTags:@"span"];
for (HTMLNode *spanNode in spanNodes) {
if ([[spanNode getAttributeNamed:@"class"] isEqualToString:@"spantext"]) {
NSLog(@"%@", [spanNode allContents]); //Answer to second question
}
}
[parser release];
As Vladimir said, for the second point it's important to replace rawContents with Contents. rawContents will print the complete raw text node, i.e.:
<span class='spantext'><b>Hello World 1</b></span>