rss.go 8.6 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383
  1. // Copyright 2017 Frédéric Guillot. All rights reserved.
  2. // Use of this source code is governed by the Apache 2.0
  3. // license that can be found in the LICENSE file.
  4. package rss // import "miniflux.app/reader/rss"
  5. import (
  6. "encoding/xml"
  7. "html"
  8. "path"
  9. "strconv"
  10. "strings"
  11. "time"
  12. "miniflux.app/crypto"
  13. "miniflux.app/logger"
  14. "miniflux.app/model"
  15. "miniflux.app/reader/date"
  16. "miniflux.app/reader/media"
  17. "miniflux.app/reader/sanitizer"
  18. "miniflux.app/url"
  19. )
  20. // Specs: https://cyber.harvard.edu/rss/rss.html
  21. type rssFeed struct {
  22. XMLName xml.Name `xml:"rss"`
  23. Version string `xml:"version,attr"`
  24. Title string `xml:"channel>title"`
  25. Links []rssLink `xml:"channel>link"`
  26. Language string `xml:"channel>language"`
  27. Description string `xml:"channel>description"`
  28. PubDate string `xml:"channel>pubDate"`
  29. ManagingEditor string `xml:"channel>managingEditor"`
  30. Webmaster string `xml:"channel>webMaster"`
  31. Items []rssItem `xml:"channel>item"`
  32. PodcastFeedElement
  33. }
  34. func (r *rssFeed) Transform(baseURL string) *model.Feed {
  35. var err error
  36. feed := new(model.Feed)
  37. siteURL := r.siteURL()
  38. feed.SiteURL, err = url.AbsoluteURL(baseURL, siteURL)
  39. if err != nil {
  40. feed.SiteURL = siteURL
  41. }
  42. feedURL := r.feedURL()
  43. feed.FeedURL, err = url.AbsoluteURL(baseURL, feedURL)
  44. if err != nil {
  45. feed.FeedURL = feedURL
  46. }
  47. feed.Title = html.UnescapeString(strings.TrimSpace(r.Title))
  48. if feed.Title == "" {
  49. feed.Title = feed.SiteURL
  50. }
  51. for _, item := range r.Items {
  52. entry := item.Transform()
  53. if entry.Author == "" {
  54. entry.Author = r.feedAuthor()
  55. }
  56. if entry.URL == "" {
  57. entry.URL = feed.SiteURL
  58. } else {
  59. entryURL, err := url.AbsoluteURL(feed.SiteURL, entry.URL)
  60. if err == nil {
  61. entry.URL = entryURL
  62. }
  63. }
  64. if entry.Title == "" {
  65. entry.Title = entry.URL
  66. }
  67. feed.Entries = append(feed.Entries, entry)
  68. }
  69. return feed
  70. }
  71. func (r *rssFeed) siteURL() string {
  72. for _, element := range r.Links {
  73. if element.XMLName.Space == "" {
  74. return strings.TrimSpace(element.Data)
  75. }
  76. }
  77. return ""
  78. }
  79. func (r *rssFeed) feedURL() string {
  80. for _, element := range r.Links {
  81. if element.XMLName.Space == "http://www.w3.org/2005/Atom" {
  82. return strings.TrimSpace(element.Href)
  83. }
  84. }
  85. return ""
  86. }
  87. func (r rssFeed) feedAuthor() string {
  88. author := r.PodcastAuthor()
  89. switch {
  90. case r.ManagingEditor != "":
  91. author = r.ManagingEditor
  92. case r.Webmaster != "":
  93. author = r.Webmaster
  94. }
  95. return sanitizer.StripTags(strings.TrimSpace(author))
  96. }
  97. type rssLink struct {
  98. XMLName xml.Name
  99. Data string `xml:",chardata"`
  100. Href string `xml:"href,attr"`
  101. Rel string `xml:"rel,attr"`
  102. }
  103. type rssCommentLink struct {
  104. XMLName xml.Name
  105. Data string `xml:",chardata"`
  106. }
  107. type rssAuthor struct {
  108. XMLName xml.Name
  109. Data string `xml:",chardata"`
  110. Name string `xml:"name"`
  111. Email string `xml:"email"`
  112. Inner string `xml:",innerxml"`
  113. }
  114. type rssTitle struct {
  115. XMLName xml.Name
  116. Data string `xml:",chardata"`
  117. Inner string `xml:",innerxml"`
  118. }
  119. type rssEnclosure struct {
  120. URL string `xml:"url,attr"`
  121. Type string `xml:"type,attr"`
  122. Length string `xml:"length,attr"`
  123. }
  124. func (enclosure *rssEnclosure) Size() int64 {
  125. if enclosure.Length == "" {
  126. return 0
  127. }
  128. size, _ := strconv.ParseInt(enclosure.Length, 10, 0)
  129. return size
  130. }
  131. type rssItem struct {
  132. GUID string `xml:"guid"`
  133. Title []rssTitle `xml:"title"`
  134. Links []rssLink `xml:"link"`
  135. Description string `xml:"description"`
  136. PubDate string `xml:"pubDate"`
  137. Authors []rssAuthor `xml:"author"`
  138. CommentLinks []rssCommentLink `xml:"comments"`
  139. EnclosureLinks []rssEnclosure `xml:"enclosure"`
  140. DublinCoreElement
  141. FeedBurnerElement
  142. PodcastEntryElement
  143. media.Element
  144. }
  145. func (r *rssItem) Transform() *model.Entry {
  146. entry := new(model.Entry)
  147. entry.URL = r.entryURL()
  148. entry.CommentsURL = r.entryCommentsURL()
  149. entry.Date = r.entryDate()
  150. entry.Author = r.entryAuthor()
  151. entry.Hash = r.entryHash()
  152. entry.Content = r.entryContent()
  153. entry.Title = r.entryTitle()
  154. entry.Enclosures = r.entryEnclosures()
  155. return entry
  156. }
  157. func (r *rssItem) entryDate() time.Time {
  158. value := r.PubDate
  159. if r.DublinCoreDate != "" {
  160. value = r.DublinCoreDate
  161. }
  162. if value != "" {
  163. result, err := date.Parse(value)
  164. if err != nil {
  165. logger.Error("rss: %v (entry GUID = %s)", err, r.GUID)
  166. return time.Now()
  167. }
  168. return result
  169. }
  170. return time.Now()
  171. }
  172. func (r *rssItem) entryAuthor() string {
  173. author := ""
  174. for _, rssAuthor := range r.Authors {
  175. switch rssAuthor.XMLName.Space {
  176. case "http://www.itunes.com/dtds/podcast-1.0.dtd", "http://www.google.com/schemas/play-podcasts/1.0":
  177. author = rssAuthor.Data
  178. case "http://www.w3.org/2005/Atom":
  179. if rssAuthor.Name != "" {
  180. author = rssAuthor.Name
  181. } else if rssAuthor.Email != "" {
  182. author = rssAuthor.Email
  183. }
  184. default:
  185. if rssAuthor.Name != "" {
  186. author = rssAuthor.Name
  187. } else if strings.Contains(rssAuthor.Inner, "<![CDATA[") {
  188. author = rssAuthor.Data
  189. } else {
  190. author = rssAuthor.Inner
  191. }
  192. }
  193. }
  194. if author == "" {
  195. author = r.DublinCoreCreator
  196. }
  197. return sanitizer.StripTags(strings.TrimSpace(author))
  198. }
  199. func (r *rssItem) entryHash() string {
  200. for _, value := range []string{r.GUID, r.entryURL()} {
  201. if value != "" {
  202. return crypto.Hash(value)
  203. }
  204. }
  205. return ""
  206. }
  207. func (r *rssItem) entryTitle() string {
  208. var title string
  209. for _, rssTitle := range r.Title {
  210. switch rssTitle.XMLName.Space {
  211. case "http://search.yahoo.com/mrss/":
  212. // Ignore title in media namespace
  213. case "http://purl.org/dc/elements/1.1/":
  214. title = rssTitle.Data
  215. default:
  216. title = rssTitle.Data
  217. }
  218. if title != "" {
  219. break
  220. }
  221. }
  222. return html.UnescapeString(strings.TrimSpace(title))
  223. }
  224. func (r *rssItem) entryContent() string {
  225. for _, value := range []string{r.DublinCoreContent, r.Description, r.PodcastDescription()} {
  226. if value != "" {
  227. return value
  228. }
  229. }
  230. return ""
  231. }
  232. func (r *rssItem) entryURL() string {
  233. if r.FeedBurnerLink != "" {
  234. return r.FeedBurnerLink
  235. }
  236. for _, link := range r.Links {
  237. if link.XMLName.Space == "http://www.w3.org/2005/Atom" && link.Href != "" && isValidLinkRelation(link.Rel) {
  238. return strings.TrimSpace(link.Href)
  239. }
  240. if link.Data != "" {
  241. return strings.TrimSpace(link.Data)
  242. }
  243. }
  244. return ""
  245. }
  246. func (r *rssItem) entryEnclosures() model.EnclosureList {
  247. enclosures := make(model.EnclosureList, 0)
  248. duplicates := make(map[string]bool)
  249. for _, mediaThumbnail := range r.AllMediaThumbnails() {
  250. if _, found := duplicates[mediaThumbnail.URL]; !found {
  251. duplicates[mediaThumbnail.URL] = true
  252. enclosures = append(enclosures, &model.Enclosure{
  253. URL: mediaThumbnail.URL,
  254. MimeType: mediaThumbnail.MimeType(),
  255. Size: mediaThumbnail.Size(),
  256. })
  257. }
  258. }
  259. for _, enclosure := range r.EnclosureLinks {
  260. enclosureURL := enclosure.URL
  261. if r.FeedBurnerEnclosureLink != "" {
  262. filename := path.Base(r.FeedBurnerEnclosureLink)
  263. if strings.Contains(enclosureURL, filename) {
  264. enclosureURL = r.FeedBurnerEnclosureLink
  265. }
  266. }
  267. if enclosureURL == "" {
  268. continue
  269. }
  270. if _, found := duplicates[enclosureURL]; !found {
  271. duplicates[enclosureURL] = true
  272. enclosures = append(enclosures, &model.Enclosure{
  273. URL: enclosureURL,
  274. MimeType: enclosure.Type,
  275. Size: enclosure.Size(),
  276. })
  277. }
  278. }
  279. for _, mediaContent := range r.AllMediaContents() {
  280. if _, found := duplicates[mediaContent.URL]; !found {
  281. duplicates[mediaContent.URL] = true
  282. enclosures = append(enclosures, &model.Enclosure{
  283. URL: mediaContent.URL,
  284. MimeType: mediaContent.MimeType(),
  285. Size: mediaContent.Size(),
  286. })
  287. }
  288. }
  289. for _, mediaPeerLink := range r.AllMediaPeerLinks() {
  290. if _, found := duplicates[mediaPeerLink.URL]; !found {
  291. duplicates[mediaPeerLink.URL] = true
  292. enclosures = append(enclosures, &model.Enclosure{
  293. URL: mediaPeerLink.URL,
  294. MimeType: mediaPeerLink.MimeType(),
  295. Size: mediaPeerLink.Size(),
  296. })
  297. }
  298. }
  299. return enclosures
  300. }
  301. func (r *rssItem) entryCommentsURL() string {
  302. for _, commentLink := range r.CommentLinks {
  303. if commentLink.XMLName.Space == "" {
  304. commentsURL := strings.TrimSpace(commentLink.Data)
  305. // The comments URL is supposed to be absolute (some feeds publishes incorrect comments URL)
  306. // See https://cyber.harvard.edu/rss/rss.html#ltcommentsgtSubelementOfLtitemgt
  307. if url.IsAbsoluteURL(commentsURL) {
  308. return commentsURL
  309. }
  310. }
  311. }
  312. return ""
  313. }
  314. func isValidLinkRelation(rel string) bool {
  315. switch rel {
  316. case "", "alternate", "enclosure", "related", "self", "via":
  317. return true
  318. default:
  319. if strings.HasPrefix(rel, "http") {
  320. return true
  321. }
  322. return false
  323. }
  324. }