almost there...

This commit is contained in:
tilman
2017-05-14 15:10:03 +02:00
parent 8a590c06b9
commit 597b43bb6f
7 changed files with 672 additions and 13 deletions
@@ -0,0 +1,193 @@
package de.kompf.javaxml;
import java.io.*;
import java.net.URL;
import java.util.*;
import javax.xml.XMLConstants;
import javax.xml.namespace.NamespaceContext;
import javax.xml.parsers.*;
import javax.xml.xpath.*;
import org.w3c.dom.*;
import org.xml.sax.SAXException;
/**
* Sample code how to use to XPath API to extract information from XML data.
*
* @author Kompf
*
*/
public class XPathReader {
/**
* Evaluate an XML input stream using the given xpath.
*
* @param in The XML input stream.
* @param xpathExpr The xpath expression - must not contain any namespace
* prefixes.
* @param result A collection to append the results to.
*/
void eval(InputStream in, String xpathExpr, Collection<String> result)
throws ParserConfigurationException, SAXException, IOException,
XPathExpressionException {
DocumentBuilderFactory docFactory = DocumentBuilderFactory.newInstance();
docFactory.setNamespaceAware(false); // important!
DocumentBuilder docBuilder = docFactory.newDocumentBuilder();
Document doc = docBuilder.parse(in);
XPath xpath = XPathFactory.newInstance().newXPath();
XPathExpression expr = xpath.compile(xpathExpr);
NodeList nodeList = (NodeList) expr.evaluate(doc, XPathConstants.NODESET);
for (int i = 0; i < nodeList.getLength(); ++i) {
Node node = nodeList.item(i);
result.add(node.getNodeValue());
}
}
/**
* Evaluate an XML input stream using the given xpath. This implementation is
* namespace aware.
*
* @param in The XML input stream.
* @param xpathExpr The xpath expression - may contain namespace prefixes.
* @param nsCtx The namespace context to resolve the namespace prefixes from
* the xpath expression.
* @param result A collection to append the results to.
*/
void evalNamespaceAware(InputStream in, String xpathExpr,
NamespaceContext nsCtx, Collection<String> result)
throws ParserConfigurationException, SAXException, IOException,
XPathExpressionException {
DocumentBuilderFactory docFactory = DocumentBuilderFactory.newInstance();
docFactory.setNamespaceAware(true); // important!
DocumentBuilder docBuilder = docFactory.newDocumentBuilder();
Document doc = docBuilder.parse(in);
XPath xpath = XPathFactory.newInstance().newXPath();
xpath.setNamespaceContext(nsCtx);
XPathExpression expr = xpath.compile(xpathExpr);
NodeList nodeList = (NodeList) expr.evaluate(doc, XPathConstants.NODESET);
for (int i = 0; i < nodeList.getLength(); ++i) {
Node node = nodeList.item(i);
result.add(node.getNodeValue());
}
}
/**
* Print the result.
*
* @param result The collection of results.
* @param out The print stream to use.
*/
void printResult(Collection<String> result, PrintStream out) {
// print result
for (String name : result) {
out.println(name);
}
}
/**
* Read the titles of entries from the Heise news feed.
*/
void readHeiseFeed() throws Exception {
URL heiseFeed = new URL("http://www.heise.de/newsticker/heise-atom.xml");
InputStream in = heiseFeed.openStream();
Collection<String> result = new LinkedList<String>();
eval(in, "//entry/title/text()", result);
printResult(result, System.out);
}
/**
* Read the titles of entries from the Heise news feed. This is basically the
* same like {@link #readHeiseFeed()} but is aware of the namespace of the
* atom feed.
*/
void readHeiseFeedNamespaceAware() throws Exception {
URL heiseFeed = new URL("http://www.heise.de/newsticker/heise-atom.xml");
InputStream in = heiseFeed.openStream();
Collection<String> result = new LinkedList<String>();
evalNamespaceAware(in, "//ns:entry/ns:title/text()",
new SimpleNamespaceContext("ns", "http://www.w3.org/2005/Atom"), result);
printResult(result, System.out);
}
/**
* Read the titles of entries from the Twitter public time line. This is
* basically the same like {@link #readHeiseFeed()} but uses another URL.
*/
void readTwitterPublicTimeLine() throws Exception {
URL twitterFeed = new URL(
"http://api.twitter.com/1/statuses/public_timeline.atom");
InputStream in = twitterFeed.openStream();
Collection<String> result = new LinkedList<String>();
eval(in, "//entry/title/text()", result);
printResult(result, System.out);
}
/**
* Read all all names of 'way' elements from an OSM XML file. The names are
* kept sorted and unique using a TreeSet.
*/
void readOsmWayNames() throws Exception {
// OSM path names
URL osmUrl = new File("md.osm.xml").toURI().toURL();
InputStream in = osmUrl.openStream();
Collection<String> result = new TreeSet<String>();
eval(in, "/osm/way/tag[@k='name']/@v", result);
printResult(result, System.out);
}
/**
* MAIN.
*
* @param args ignored.
* @throws Exception If an error occurs.
*/
public static void main(String[] args) throws Exception {
XPathReader xPathReader = new XPathReader();
// Heise News
xPathReader.readHeiseFeed();
// Heise News namespace aware
//xPathReader.readHeiseFeedNamespaceAware();
// Twitter public time line
//xPathReader.readTwitterPublicTimeLine();
// OSM path names
//xPathReader.readOsmWayNames();
}
static class SimpleNamespaceContext implements NamespaceContext {
private String prefix;
private String uri;
public SimpleNamespaceContext(String prefix, String uri) {
this.prefix = prefix;
this.uri = uri;
}
public String getNamespaceURI(String prefix) {
if (this.prefix.equals(prefix)) {
return uri;
}
return XMLConstants.NULL_NS_URI;
}
public String getPrefix(String namespaceURI) {
if (uri.equals(namespaceURI)) {
return prefix;
}
return null;
}
@SuppressWarnings("unchecked")
public Iterator getPrefixes(String namespaceURI) {
List<String> prefixList = new ArrayList<String>();
if (uri.equals(namespaceURI)) {
prefixList.add(prefix);
}
return prefixList.iterator();
}
}
}
@@ -1,10 +1,150 @@
package de.tilman.AHTreeReader;
import java.io.ByteArrayInputStream;
import java.nio.charset.StandardCharsets;
import javax.xml.parsers.DocumentBuilder;
import javax.xml.parsers.DocumentBuilderFactory;
import javax.xml.xpath.XPath;
import javax.xml.xpath.XPathConstants;
import javax.xml.xpath.XPathExpression;
import javax.xml.xpath.XPathFactory;
import org.apache.http.HttpEntity;
import org.apache.http.HttpHost;
import org.apache.http.client.methods.CloseableHttpResponse;
import org.apache.http.client.methods.HttpPost;
import org.apache.http.entity.mime.MultipartEntityBuilder;
import org.apache.http.impl.client.CloseableHttpClient;
import org.apache.http.impl.client.HttpClients;
import org.apache.http.util.EntityUtils;
import org.w3c.dom.Document;
import org.w3c.dom.NodeList;
import de.tilman.AHTreeReader.Tree.TreeNode;
public class AHTreeReader {
String cookie = "PHPSESSID=5nccupenl4b7tps9048bge76h3";
Tree<String, String> tree = new Tree<String, String>();
CloseableHttpClient httpClient = HttpClients.createDefault();
HttpHost target = new HttpHost("media-assistant.animalhealth.bayer.com", 443, "https");
HttpPost httpPost;
HttpEntity httpEntity;
DocumentBuilderFactory factory;
DocumentBuilder builder;
XPath xpath = XPathFactory.newInstance().newXPath();
public AHTreeReader() {
System.out.println("Aloha, funky time");
// proxy configuration
/*
HttpHost proxy = new HttpHost("10.185.190.100", 8080, "http");
RequestConfig config = RequestConfig.custom().setProxy(proxy).build();
httpPost.setConfig(config);
*/
try {
factory = DocumentBuilderFactory.newInstance();
factory.setNamespaceAware(false);
builder = factory.newDocumentBuilder();
getMainCategories();
getSubcategories(tree.root);
} catch (Exception e) {
e.printStackTrace();
}
}
private void getSubcategories(TreeNode<String, String> node) throws Exception {
for (TreeNode<String, String> n : node.children) {
System.out.println("\n----------------");
System.out.println("Looking for " + n.key + "\n");
httpPost = new HttpPost("https://media-assistant.animalhealth.bayer.com/assets/php/ajax_getCategoryFolderSubnavi.php");
httpPost.setHeader("Cookie", cookie);
// parent node is transferred as form data o_0
httpEntity = MultipartEntityBuilder.create().addTextBody("parent", n.key).build();
httpPost.setEntity(httpEntity);
CloseableHttpResponse response = httpClient.execute(target, httpPost);
if (response.getStatusLine().getStatusCode() != 200) {
System.err.println("Error requesting children of " + n.key + ", " + n.value);
break;
}
String content = EntityUtils.toString(response.getEntity());
response.close();
// dirty optimization: cut out <ul> node
if (content.indexOf("<ul>") > -1) {
content = content.substring(content.indexOf("<ul>"), content.indexOf("</ul>") + 5);
// Document doc = builder.parse(new ByteArrayInputStream(content.getBytes(StandardCharsets.UTF_8)));
// XPathExpression expr = xpath.compile("/ul/li/span");
// NodeList nl = (NodeList) expr.evaluate(doc, XPathConstants.NODESET);
//
// for (int i=0; i < nl.getLength(); i++) {
// String categoryId = nl.item(i).getAttributes().getNamedItem("catId").getTextContent();
// String categoryName = nl.item(i).getTextContent();
//
// n.children.add(new TreeNode<String, String>(categoryId, categoryName));
// System.out.println(categoryId + ", " + categoryName);
// }
// unfortunately, the contents of the tags are not properly escaped, so we can't use the XML parser here...
// while (content.indexOf("<li") > -1) {
String elem = content.substring(content.indexOf("<li"), content.indexOf("</li>") + 5);
content = content.substring(content.indexOf("</li>") + 5);
String categoryId = elem.substring(elem.indexOf("catId=\"") + 7, elem.indexOf("\"><a "));
String categoryName = elem.substring(elem.indexOf("<a ") + 29, elem.indexOf("</a>"));
n.children.add(new TreeNode<String, String>(categoryId, categoryName));
// }
}
else {
System.out.println("No subcategories for " + n.key + ", " + n.value);
}
}
}
private void getMainCategories() throws Exception {
httpPost = new HttpPost("https://media-assistant.animalhealth.bayer.com/assets/php/ajax_loadCategoryTree.php");
httpPost.setHeader("Cookie", cookie);
CloseableHttpResponse response = httpClient.execute(target, httpPost);
System.out.println(response.getStatusLine());
String content = EntityUtils.toString(response.getEntity());
response.close();
// simply hack out the second <ul> element:
int startIndex = content.indexOf("</ul>");
String mainCategories = content.substring(content.indexOf("<ul>", startIndex), content.indexOf("</ul>", startIndex + 5) + 5);
Document doc = builder.parse(new ByteArrayInputStream(mainCategories.getBytes(StandardCharsets.UTF_8)));
XPathExpression expr = xpath.compile("/ul/li/span");
NodeList nl = (NodeList) expr.evaluate(doc, XPathConstants.NODESET);
for (int i=0; i < nl.getLength(); i++) {
String categoryId = nl.item(i).getAttributes().getNamedItem("catId").getTextContent();
String categoryName = nl.item(i).getTextContent();
tree.root.children.add(new TreeNode<String, String>(categoryId, categoryName));
System.out.println(categoryId + ", " + categoryName);
}
}
public static void main(String[] args) {
// TODO Auto-generated method stub
new AHTreeReader();
}
}
@@ -10,7 +10,7 @@ import org.apache.http.impl.client.CloseableHttpClient;
import org.apache.http.impl.client.HttpClients;
import org.apache.http.util.EntityUtils;
import de.tilman.AHTreeReader.Tree.Node;
import de.tilman.AHTreeReader.Tree.TreeNode;
public class PlainApp {
@@ -52,10 +52,10 @@ public class PlainApp {
mainCategories = mainCategories.substring(mainCategories.indexOf('\n') + 1);
System.out.println(key + ", " + value);
tr.root.children.add(new Node<String, String>(key, value));
tr.root.children.add(new TreeNode<String, String>(key, value));
}
for (Node<String, String> n : tr.root.children) {
for (TreeNode<String, String> n : tr.root.children) {
httpPost = new HttpPost("https://media-assistant.animalhealth.bayer.com/assets/php/ajax_getCategoryFolderSubnavi.php");
// httpPost.setConfig(config);
httpPost.setHeader("Cookie", cookie);
@@ -4,26 +4,26 @@ import java.util.ArrayList;
import java.util.List;
public class Tree<K,V> {
public Node<K,V> root;
public TreeNode<K,V> root;
public Tree() {
root = new Node<K,V>();
root.children = new ArrayList<Node<K,V>>();
root = new TreeNode<K,V>();
root.children = new ArrayList<TreeNode<K,V>>();
}
public static class Node<K,V> {
public static class TreeNode<K,V> {
public Node() { }
public TreeNode() { }
public Node(K key, V value) {
public TreeNode(K key, V value) {
this.key = key;
this.value = value;
}
public K key;
public V value;
public Node<K,V> parent;
public List<Node<K,V>> children = new ArrayList<Node<K,V>>();
public TreeNode<K,V> parent;
public List<TreeNode<K,V>> children = new ArrayList<TreeNode<K,V>>();
}
}