2 * Copyright (C) 2001, 2002 The Mir-coders group
4 * This file is part of Mir.
6 * Mir is free software; you can redistribute it and/or modify
7 * it under the terms of the GNU General Public License as published by
8 * the Free Software Foundation; either version 2 of the License, or
9 * (at your option) any later version.
11 * Mir is distributed in the hope that it will be useful,
12 * but WITHOUT ANY WARRANTY; without even the implied warranty of
13 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
14 * GNU General Public License for more details.
16 * You should have received a copy of the GNU General Public License
17 * along with Mir; if not, write to the Free Software
18 * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA
20 * In addition, as a special exception, The Mir-coders gives permission to link
21 * the code of this program with any library licensed under the Apache Software License,
22 * The Sun (tm) Java Advanced Imaging library (JAI), The Sun JIMI library
23 * (or with modified versions of the above that use the same license as the above),
24 * and distribute linked combinations including the two. You must obey the
25 * GNU General Public License in all respects for all of the code used other than
26 * the above mentioned libraries. If you modify this file, you may extend this
27 * exception to your version of the file, but you are not obligated to do so.
28 * If you do not wish to do so, delete this exception statement from your version.
32 import java.io.InputStream;
33 import java.util.ArrayList;
34 import java.util.HashMap;
35 import java.util.List;
38 import mir.util.DateTimeFunctions;
39 import mir.util.HTTPClientHelper;
40 import mir.util.xml.XMLParserEngine;
41 import mir.util.xml.XMLParserExc;
42 import mir.util.xml.XMLParserFailure;
44 public class RSSReader {
45 public static final String RDF_NAMESPACE_URI = "http://www.w3.org/1999/02/22-rdf-syntax-ns#";
46 public static final String RSS_1_0_NAMESPACE_URI = "http://purl.org/rss/1.0/";
47 public static final String RSS_0_9_NAMESPACE_URI = "http://my.netscape.com/rdf/simple/0.9/";
48 public static final String DUBLINCORE_NAMESPACE_URI = "http://purl.org/dc/elements/1.1/";
49 public static final String EVENT_NAMESPACE_URI = "http://purl.org/rss/1.0/modules/event/";
50 public static final String TAXONOMY_NAMESPACE_URI = "http://web.resource.org/rss/1.0/modules/taxonomy/";
51 public static final String DUBLINCORE_TERMS_NAMESPACE_URI = "http://purl.org/dc/terms/";
52 public static final String CONTENT_NAMESPACE_URI = "http://purl.org/rss/1.0/modules/content/";
54 // ML: to be localized:
55 public static final String V2V_NAMESPACE_URI = "http://v2v.cc/rss/";
57 private static final mir.util.xml.XMLName RDF_ABOUT_PARAMETER = new mir.util.xml.XMLName(RDF_NAMESPACE_URI, "about");
58 private static final mir.util.xml.XMLName RDF_SEQUENCE_TAG = new mir.util.xml.XMLName(RDF_NAMESPACE_URI, "Seq");
59 private static final mir.util.xml.XMLName RDF_BAG_PARAMETER = new mir.util.xml.XMLName(RDF_NAMESPACE_URI, "Bag");
61 private static final mir.util.xml.XMLName RSS_CHANNEL_TAG = new mir.util.xml.XMLName(RSS_1_0_NAMESPACE_URI, "channel");
62 private static final mir.util.xml.XMLName RSS_ITEM_TAG = new mir.util.xml.XMLName(RSS_1_0_NAMESPACE_URI, "item");
63 private static final mir.util.xml.XMLName RSS_ITEMS_TAG = new mir.util.xml.XMLName(RSS_1_0_NAMESPACE_URI, "items");
66 private Map namespaceURItoModule;
67 private Map moduleToPrefix;
70 modules = new ArrayList();
71 namespaceURItoModule = new HashMap();
72 moduleToPrefix = new HashMap();
74 registerModule(new RSSBasicModule(RDF_NAMESPACE_URI, "RDF module"), "rdf");
75 registerModule(new RSSBasicModule(RSS_1_0_NAMESPACE_URI, "RSS 1.0 module"), "rss");
76 registerModule(new RSSBasicModule(RSS_0_9_NAMESPACE_URI, "RSS 0.9 module"), "rss");
78 RSSBasicModule dcModule = new RSSBasicModule(DUBLINCORE_NAMESPACE_URI, "RSS Dublin Core 1.1");
79 dcModule.addProperty("date", RSSModule.W3CDTF_PROPERTY_TYPE);
80 registerModule(dcModule, "dc");
82 RSSBasicModule dcTermsModule = new RSSBasicModule(DUBLINCORE_TERMS_NAMESPACE_URI, "RSS Qualified Dublin core");
83 dcTermsModule.addProperty("created", RSSModule.W3CDTF_PROPERTY_TYPE);
84 dcTermsModule.addProperty("issued", RSSModule.W3CDTF_PROPERTY_TYPE);
85 dcTermsModule.addProperty("modified", RSSModule.W3CDTF_PROPERTY_TYPE);
86 dcTermsModule.addProperty("dateAccepted", RSSModule.W3CDTF_PROPERTY_TYPE);
87 dcTermsModule.addProperty("dateCopyrighted", RSSModule.W3CDTF_PROPERTY_TYPE);
88 dcTermsModule.addProperty("dateSubmitted", RSSModule.W3CDTF_PROPERTY_TYPE);
89 registerModule(dcTermsModule, "dcterms");
91 RSSBasicModule v2vTermsModule = new RSSBasicModule(V2V_NAMESPACE_URI, "indymedia v2v RSS module");
92 v2vTermsModule.addMultiValuedProperty("topic", RSSModule.PCDATA_PROPERTY_TYPE);
93 v2vTermsModule.addMultiValuedProperty("genre", RSSModule.PCDATA_PROPERTY_TYPE);
94 v2vTermsModule.addMultiValuedProperty("link", RSSModule.PCDATA_PROPERTY_TYPE);
95 registerModule(v2vTermsModule, "v2v");
97 registerModule(new RSSBasicModule(EVENT_NAMESPACE_URI, "Event RSS module"), "ev");
98 registerModule(new RSSBasicModule(TAXONOMY_NAMESPACE_URI, "Taxonomy RSS module"), "taxo");
99 registerModule(new RSSBasicModule(CONTENT_NAMESPACE_URI , "Content RSS module"), "content");
102 public void registerModule(RSSModule aModule, String aPrefix) {
103 modules.add(aModule);
104 namespaceURItoModule.put(aModule.getNamespaceURI(), aModule);
105 moduleToPrefix.put(aModule, aPrefix);
108 public RSSData parseInputStream(InputStream aStream) throws RSSExc, RSSFailure {
110 RSSData result = new RSSData();
111 XMLParserEngine.getInstance().parse("xml", aStream, new RootSectionHandler(result));
115 catch (Throwable t) {
116 throw new RSSFailure(t);
120 public RSSData parseInputStream(InputStream aStream, String anEncoding) throws RSSExc, RSSFailure {
122 RSSData result = new RSSData();
123 XMLParserEngine.getInstance().parse("xml", aStream, anEncoding, new RootSectionHandler(result));
127 catch (Throwable t) {
128 throw new RSSFailure(t);
132 public RSSData parseUrl(String anUrl) throws RSSExc, RSSFailure {
134 HTTPClientHelper httpClientHelper = new HTTPClientHelper();
135 InputStream inputStream = httpClientHelper.getUrl(anUrl);
136 if (inputStream==null)
137 throw new RSSExc("RSSChannel.parseUrl: Can't get url content");
139 RSSData theRSSData = parseInputStream(inputStream);
140 httpClientHelper.releaseHTTPConnection();
143 catch (Throwable t) {
144 throw new RSSFailure(t);
148 public RSSData parseUrl(String anUrl, String anEncoding) throws RSSExc, RSSFailure {
150 HTTPClientHelper httpClientHelper = new HTTPClientHelper();
151 InputStream inputStream = httpClientHelper.getUrl(anUrl);
152 if (inputStream==null)
153 throw new RSSExc("RSSChannel.parseUrl: Can't get url content");
155 RSSData theRSSData = parseInputStream(inputStream, anEncoding);
156 httpClientHelper.releaseHTTPConnection();
159 catch (Throwable t) {
160 throw new RSSFailure(t);
164 private class RootSectionHandler extends mir.util.xml.AbstractSectionHandler {
165 private RSSData data;
167 public RootSectionHandler(RSSData aData) {
171 public mir.util.xml.SectionHandler startElement(mir.util.xml.XMLName aTag, Map anAttributes) throws XMLParserExc {
172 if (aTag.getLocalName().equals("RDF")) {
173 return new RDFSectionHandler(data);
176 throw new XMLParserFailure(new RSSExc("'RDF' tag expected"));
179 public void endElement(mir.util.xml.SectionHandler aHandler) throws XMLParserExc {
182 public void characters(String aCharacters) throws XMLParserExc {
183 if (aCharacters.trim().length()>0)
184 throw new XMLParserExc("No character data allowed here");
187 public void finishSection() throws XMLParserExc {
191 private class RDFSectionHandler extends mir.util.xml.AbstractSectionHandler {
192 private RSSData data;
195 public RDFSectionHandler(RSSData aData) {
199 public mir.util.xml.SectionHandler startElement(mir.util.xml.XMLName aTag, Map anAttributes) throws XMLParserExc {
200 String identifier = (String) anAttributes.get(RDF_ABOUT_PARAMETER);
201 String rdfClass = makeQualifiedName(aTag);
203 return new RDFResourceSectionHandler(rdfClass, identifier);
206 public void endElement(mir.util.xml.SectionHandler aHandler) throws XMLParserExc {
207 if (aHandler instanceof RDFResourceSectionHandler) {
208 data.addResource(((RDFResourceSectionHandler) aHandler).getResource());
212 public void characters(String aCharacters) throws XMLParserExc {
213 if (aCharacters.trim().length()>0)
214 throw new XMLParserExc("No character data allowed here");
217 public void finishSection() throws XMLParserExc {
221 private mir.util.xml.SectionHandler makePropertyValueSectionHandler(mir.util.xml.XMLName aTag, Map anAttributes) {
222 RSSModule module = (RSSModule) namespaceURItoModule.get(aTag.getNamespaceURI());
225 RSSModule.RSSModuleProperty property = module.getPropertyForName(aTag.getLocalName());
227 if (property!=null) {
228 switch (property.getType()) {
230 RSSModule.PCDATA_PROPERTY_TYPE:
231 return new PCDATASectionHandler();
233 RSSModule.RDFCOLLECTION_PROPERTY_TYPE:
234 return new RDFCollectionSectionHandler();
236 // RSSModule.RDF_PROPERTY_TYPE:
237 // return new RDFValueSectionHandler();
239 RSSModule.W3CDTF_PROPERTY_TYPE:
240 return new DateSectionHandler();
245 return new FlexiblePropertyValueSectionHandler();
248 private void usePropertyValueSectionHandler(RDFResource aResource, PropertyValueSectionHandler aHandler, mir.util.xml.XMLName aTag) {
249 RSSModule module = (RSSModule) namespaceURItoModule.get(aTag.getNamespaceURI());
252 RSSModule.RSSModuleProperty property = module.getPropertyForName(aTag.getLocalName());
254 if (property!=null && property.getIsMultiValued()) {
255 List value = (List) aResource.get(makeQualifiedName(aTag));
258 value = new ArrayList();
259 aResource.set(makeQualifiedName(aTag), value);
262 value.add(aHandler.getValue());
268 aResource.set(makeQualifiedName(aTag), aHandler.getValue());
271 private String makeQualifiedName(mir.util.xml.XMLName aName) {
272 String result=aName.getLocalName();
273 RSSModule module = (RSSModule) namespaceURItoModule.get(aName.getNamespaceURI());
275 String prefix = (String) moduleToPrefix.get(module);
277 if (prefix!=null && prefix.length()>0)
278 result = prefix+":"+result;
284 private class RDFResourceSectionHandler extends mir.util.xml.AbstractSectionHandler {
285 private String image;
286 private mir.util.xml.XMLName currentTag;
287 private RDFResource resource;
289 public RDFResourceSectionHandler(String anRDFClass, String anIdentifier) {
290 resource = new RDFResource(anRDFClass, anIdentifier);
293 public mir.util.xml.SectionHandler startElement(mir.util.xml.XMLName aTag, Map anAttributes) throws XMLParserExc {
296 return makePropertyValueSectionHandler(aTag, anAttributes);
299 public void endElement(mir.util.xml.SectionHandler aHandler) throws XMLParserExc {
300 if (aHandler instanceof PropertyValueSectionHandler) {
301 usePropertyValueSectionHandler(resource, (PropertyValueSectionHandler) aHandler, currentTag);
302 // resource.set(makeQualifiedName(currentTag), ( (PropertyValueSectionHandler) aHandler).getFieldValue());
306 public void characters(String aCharacters) throws XMLParserExc {
307 if (aCharacters.trim().length()>0)
308 throw new XMLParserExc("No character data allowed here");
311 public void finishSection() throws XMLParserExc {
314 public RDFResource getResource() {
315 if ((resource.getIdentifier()==null || resource.getIdentifier().length()==0) && resource.get("rss:link")!=null) {
316 resource.setIdentifier(resource.get("rss:link").toString());
323 private abstract class PropertyValueSectionHandler extends mir.util.xml.AbstractSectionHandler {
324 public abstract Object getValue();
327 private class FlexiblePropertyValueSectionHandler extends PropertyValueSectionHandler {
328 private StringBuffer stringData;
329 private Object structuredData;
331 public FlexiblePropertyValueSectionHandler() {
332 stringData = new StringBuffer();
336 public mir.util.xml.SectionHandler startElement(String aTag, Map anAttributes) throws XMLParserExc {
337 if (aTag.equals(RDF_SEQUENCE_TAG))
338 return new RDFSequenceSectionHandler();
340 return new DiscardingSectionHandler();
343 public void endElement(mir.util.xml.SectionHandler aHandler) throws XMLParserExc {
344 if (aHandler instanceof RDFSequenceSectionHandler) {
345 structuredData= ((RDFSequenceSectionHandler) aHandler).getItems();
349 public void characters(String aCharacters) throws XMLParserExc {
350 stringData.append(aCharacters);
353 public void finishSection() throws XMLParserExc {
356 public String getData() {
357 return stringData.toString();
360 public Object getValue() {
361 if (structuredData==null)
362 return stringData.toString();
364 return structuredData;
368 private class RDFCollectionSectionHandler extends PropertyValueSectionHandler {
371 public RDFCollectionSectionHandler() {
372 items = new ArrayList();
375 public mir.util.xml.SectionHandler startElement(String aTag, Map anAttributes) throws XMLParserExc {
376 if (aTag.equals(RDF_SEQUENCE_TAG))
377 return new RDFSequenceSectionHandler();
379 return new DiscardingSectionHandler();
382 public void endElement(mir.util.xml.SectionHandler aHandler) throws XMLParserExc {
383 if (aHandler instanceof RDFSequenceSectionHandler) {
384 items.addAll(((RDFSequenceSectionHandler) aHandler).getItems());
388 public void characters(String aCharacters) throws XMLParserExc {
389 if (aCharacters.trim().length()>0)
390 throw new XMLParserExc("No character data allowed here");
393 public void finishSection() throws XMLParserExc {
396 public List getItems() {
400 public Object getValue() {
405 private class PCDATASectionHandler extends PropertyValueSectionHandler {
406 private StringBuffer data;
408 public PCDATASectionHandler() {
409 data = new StringBuffer();
412 public mir.util.xml.SectionHandler startElement(String aTag, Map anAttributes) throws XMLParserExc {
413 throw new XMLParserFailure(new RSSExc("No subtags allowed here"));
416 public void endElement(mir.util.xml.SectionHandler aHandler) throws XMLParserExc {
419 public void characters(String aCharacters) throws XMLParserExc {
420 data.append(aCharacters);
423 public void finishSection() throws XMLParserExc {
426 public String getData() {
427 return data.toString();
430 public Object getValue() {
431 return data.toString();
435 private class DateSectionHandler extends PropertyValueSectionHandler {
436 private StringBuffer data;
438 public DateSectionHandler() {
439 data = new StringBuffer();
442 public mir.util.xml.SectionHandler startElement(String aTag, Map anAttributes) throws XMLParserExc {
443 throw new XMLParserFailure(new RSSExc("No subtags allowed here"));
446 public void endElement(mir.util.xml.SectionHandler aHandler) throws XMLParserExc {
449 public void characters(String aCharacters) throws XMLParserExc {
450 data.append(aCharacters);
453 public void finishSection() throws XMLParserExc {
456 public Object getValue() {
458 String expression = data.toString().trim();
460 return DateTimeFunctions.parseW3CDTFString(expression);
462 catch (Throwable t) {
470 private class RDFSequenceSectionHandler extends mir.util.xml.AbstractSectionHandler {
473 public RDFSequenceSectionHandler() {
474 items = new ArrayList();
477 public mir.util.xml.SectionHandler startElement(String aTag, Map anAttributes) throws XMLParserExc {
478 if (aTag.equals("rdf:li")) {
479 String item = (String) anAttributes.get("rdf:resource");
485 return new DiscardingSectionHandler();
488 public void endElement(mir.util.xml.SectionHandler aHandler) throws XMLParserExc {
491 public void characters(String aCharacters) throws XMLParserExc {
494 public void finishSection() throws XMLParserExc {
497 public List getItems() {
502 private class RDFLiteralSectionHandler extends PropertyValueSectionHandler {
503 private StringBuffer data;
506 public RDFLiteralSectionHandler() {
507 data = new StringBuffer();
510 protected StringBuffer getData() {
514 public mir.util.xml.SectionHandler startElement(String aTag, Map anAttributes) throws XMLParserExc {
516 data.append("<"+tag+">");
518 return new RDFLiteralSectionHandler();
521 public void endElement(mir.util.xml.SectionHandler aHandler) throws XMLParserExc {
522 data.append(((RDFLiteralSectionHandler) aHandler).getData());
523 data.append("</"+tag+">");
526 public void characters(String aCharacters) throws XMLParserExc {
527 data.append(aCharacters);
530 public void finishSection() throws XMLParserExc {
533 public Object getValue() {
534 return data.toString();
538 private class DiscardingSectionHandler extends mir.util.xml.AbstractSectionHandler {
539 public mir.util.xml.SectionHandler startElement(String aTag, Map anAttributes) throws XMLParserExc {
543 public void endElement(mir.util.xml.SectionHandler aHandler) throws XMLParserExc {
546 public void characters(String aCharacters) throws XMLParserExc {
549 public void finishSection() throws XMLParserExc {