Readability.js 88 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430143114321433143414351436143714381439144014411442144314441445144614471448144914501451145214531454145514561457145814591460146114621463146414651466146714681469147014711472147314741475147614771478147914801481148214831484148514861487148814891490149114921493149414951496149714981499150015011502150315041505150615071508150915101511151215131514151515161517151815191520152115221523152415251526152715281529153015311532153315341535153615371538153915401541154215431544154515461547154815491550155115521553155415551556155715581559156015611562156315641565156615671568156915701571157215731574157515761577157815791580158115821583158415851586158715881589159015911592159315941595159615971598159916001601160216031604160516061607160816091610161116121613161416151616161716181619162016211622162316241625162616271628162916301631163216331634163516361637163816391640164116421643164416451646164716481649165016511652165316541655165616571658165916601661166216631664166516661667166816691670167116721673167416751676167716781679168016811682168316841685168616871688168916901691169216931694169516961697169816991700170117021703170417051706170717081709171017111712171317141715171617171718171917201721172217231724172517261727172817291730173117321733173417351736173717381739174017411742174317441745174617471748174917501751175217531754175517561757175817591760176117621763176417651766176717681769177017711772177317741775177617771778177917801781178217831784178517861787178817891790179117921793179417951796179717981799180018011802180318041805180618071808180918101811181218131814181518161817181818191820182118221823182418251826182718281829183018311832183318341835183618371838183918401841184218431844184518461847184818491850185118521853185418551856185718581859186018611862186318641865186618671868186918701871187218731874187518761877187818791880188118821883188418851886188718881889189018911892189318941895189618971898189919001901190219031904190519061907190819091910191119121913191419151916191719181919192019211922192319241925192619271928192919301931193219331934193519361937193819391940194119421943194419451946194719481949195019511952195319541955195619571958195919601961196219631964196519661967196819691970197119721973197419751976197719781979198019811982198319841985198619871988198919901991199219931994199519961997199819992000200120022003200420052006200720082009201020112012201320142015201620172018201920202021202220232024202520262027202820292030203120322033203420352036203720382039204020412042204320442045204620472048204920502051205220532054205520562057205820592060206120622063206420652066206720682069207020712072207320742075207620772078207920802081208220832084208520862087208820892090209120922093209420952096209720982099210021012102210321042105210621072108210921102111211221132114211521162117211821192120212121222123212421252126212721282129213021312132213321342135213621372138213921402141214221432144214521462147214821492150215121522153215421552156215721582159216021612162216321642165216621672168216921702171217221732174217521762177217821792180218121822183218421852186218721882189219021912192219321942195219621972198219922002201220222032204220522062207220822092210221122122213221422152216221722182219222022212222222322242225222622272228222922302231223222332234223522362237223822392240224122422243224422452246224722482249225022512252225322542255225622572258225922602261226222632264226522662267226822692270227122722273227422752276227722782279228022812282228322842285228622872288228922902291229222932294229522962297229822992300230123022303230423052306230723082309231023112312231323142315231623172318231923202321232223232324232523262327232823292330233123322333233423352336233723382339234023412342234323442345234623472348234923502351235223532354235523562357235823592360236123622363236423652366236723682369237023712372237323742375237623772378237923802381238223832384238523862387238823892390239123922393239423952396239723982399240024012402240324042405240624072408240924102411241224132414241524162417241824192420242124222423242424252426242724282429243024312432243324342435243624372438243924402441244224432444244524462447244824492450245124522453245424552456245724582459246024612462246324642465246624672468246924702471247224732474247524762477247824792480248124822483248424852486248724882489249024912492249324942495249624972498249925002501250225032504250525062507250825092510251125122513251425152516251725182519252025212522252325242525252625272528252925302531253225332534253525362537253825392540254125422543254425452546254725482549255025512552255325542555255625572558255925602561256225632564256525662567256825692570257125722573257425752576257725782579258025812582258325842585258625872588258925902591259225932594259525962597259825992600260126022603260426052606260726082609261026112612261326142615261626172618261926202621262226232624262526262627262826292630263126322633263426352636263726382639264026412642264326442645264626472648264926502651265226532654265526562657265826592660266126622663266426652666266726682669267026712672267326742675267626772678267926802681268226832684268526862687268826892690269126922693269426952696269726982699270027012702270327042705270627072708270927102711271227132714271527162717271827192720272127222723272427252726272727282729273027312732273327342735273627372738273927402741274227432744274527462747274827492750275127522753275427552756275727582759276027612762276327642765276627672768276927702771277227732774277527762777277827792780278127822783278427852786278727882789
  1. (function() {
  2. /*
  3. * Copyright (c) 2010 Arc90 Inc
  4. *
  5. * Licensed under the Apache License, Version 2.0 (the "License");
  6. * you may not use this file except in compliance with the License.
  7. * You may obtain a copy of the License at
  8. *
  9. * http://www.apache.org/licenses/LICENSE-2.0
  10. *
  11. * Unless required by applicable law or agreed to in writing, software
  12. * distributed under the License is distributed on an "AS IS" BASIS,
  13. * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
  14. * See the License for the specific language governing permissions and
  15. * limitations under the License.
  16. */
  17. /*
  18. * This code is heavily based on Arc90's readability.js (1.7.1) script
  19. * available at: http://code.google.com/p/arc90labs-readability
  20. */
  21. /**
  22. * Public constructor.
  23. * @param {HTMLDocument} doc The document to parse.
  24. * @param {Object} options The options object.
  25. */
  26. function Readability(doc, options) {
  27. // In some older versions, people passed a URI as the first argument. Cope:
  28. if (options && options.documentElement) {
  29. doc = options;
  30. options = arguments[2];
  31. } else if (!doc || !doc.documentElement) {
  32. throw new Error(
  33. "First argument to Readability constructor should be a document object."
  34. );
  35. }
  36. options = options || {};
  37. this._doc = doc;
  38. this._docJSDOMParser = this._doc.firstChild.__JSDOMParser__;
  39. this._articleTitle = null;
  40. this._articleByline = null;
  41. this._articleDir = null;
  42. this._articleSiteName = null;
  43. this._attempts = [];
  44. this._metadata = {};
  45. // Configurable options
  46. this._debug = !!options.debug;
  47. this._maxElemsToParse =
  48. options.maxElemsToParse || this.DEFAULT_MAX_ELEMS_TO_PARSE;
  49. this._nbTopCandidates =
  50. options.nbTopCandidates || this.DEFAULT_N_TOP_CANDIDATES;
  51. this._charThreshold = options.charThreshold || this.DEFAULT_CHAR_THRESHOLD;
  52. this._classesToPreserve = this.CLASSES_TO_PRESERVE.concat(
  53. options.classesToPreserve || []
  54. );
  55. this._keepClasses = !!options.keepClasses;
  56. this._serializer =
  57. options.serializer ||
  58. function (el) {
  59. return el.innerHTML;
  60. };
  61. this._disableJSONLD = !!options.disableJSONLD;
  62. this._allowedVideoRegex = options.allowedVideoRegex || this.REGEXPS.videos;
  63. this._linkDensityModifier = options.linkDensityModifier || 0;
  64. // Start with all flags set
  65. this._flags =
  66. this.FLAG_STRIP_UNLIKELYS |
  67. this.FLAG_WEIGHT_CLASSES |
  68. this.FLAG_CLEAN_CONDITIONALLY;
  69. // Control whether log messages are sent to the console
  70. if (this._debug) {
  71. let logNode = function (node) {
  72. if (node.nodeType == node.TEXT_NODE) {
  73. return `${node.nodeName} ("${node.textContent}")`;
  74. }
  75. let attrPairs = Array.from(node.attributes || [], function (attr) {
  76. return `${attr.name}="${attr.value}"`;
  77. }).join(" ");
  78. return `<${node.localName} ${attrPairs}>`;
  79. };
  80. this.log = function () {
  81. if (typeof console !== "undefined") {
  82. let args = Array.from(arguments, arg => {
  83. if (arg && arg.nodeType == this.ELEMENT_NODE) {
  84. return logNode(arg);
  85. }
  86. return arg;
  87. });
  88. args.unshift("Reader: (Readability)");
  89. // eslint-disable-next-line no-console
  90. console.log(...args);
  91. } else if (typeof dump !== "undefined") {
  92. /* global dump */
  93. var msg = Array.prototype.map
  94. .call(arguments, function (x) {
  95. return x && x.nodeName ? logNode(x) : x;
  96. })
  97. .join(" ");
  98. dump("Reader: (Readability) " + msg + "\n");
  99. }
  100. };
  101. } else {
  102. this.log = function () {};
  103. }
  104. }
  105. Readability.prototype = {
  106. FLAG_STRIP_UNLIKELYS: 0x1,
  107. FLAG_WEIGHT_CLASSES: 0x2,
  108. FLAG_CLEAN_CONDITIONALLY: 0x4,
  109. // https://developer.mozilla.org/en-US/docs/Web/API/Node/nodeType
  110. ELEMENT_NODE: 1,
  111. TEXT_NODE: 3,
  112. // Max number of nodes supported by this parser. Default: 0 (no limit)
  113. DEFAULT_MAX_ELEMS_TO_PARSE: 0,
  114. // The number of top candidates to consider when analysing how
  115. // tight the competition is among candidates.
  116. DEFAULT_N_TOP_CANDIDATES: 5,
  117. // Element tags to score by default.
  118. DEFAULT_TAGS_TO_SCORE: "section,h2,h3,h4,h5,h6,p,td,pre"
  119. .toUpperCase()
  120. .split(","),
  121. // The default number of chars an article must have in order to return a result
  122. DEFAULT_CHAR_THRESHOLD: 500,
  123. // All of the regular expressions in use within readability.
  124. // Defined up here so we don't instantiate them repeatedly in loops.
  125. REGEXPS: {
  126. // NOTE: These two regular expressions are duplicated in
  127. // Readability-readerable.js. Please keep both copies in sync.
  128. unlikelyCandidates:
  129. /-ad-|ai2html|banner|breadcrumbs|combx|comment|community|cover-wrap|disqus|extra|footer|gdpr|header|legends|menu|related|remark|replies|rss|shoutbox|sidebar|skyscraper|social|sponsor|supplemental|ad-break|agegate|pagination|pager|popup|yom-remote/i,
  130. okMaybeItsACandidate: /and|article|body|column|content|main|shadow/i,
  131. positive:
  132. /article|body|content|entry|hentry|h-entry|main|page|pagination|post|text|blog|story/i,
  133. negative:
  134. /-ad-|hidden|^hid$| hid$| hid |^hid |banner|combx|comment|com-|contact|footer|gdpr|masthead|media|meta|outbrain|promo|related|scroll|share|shoutbox|sidebar|skyscraper|sponsor|shopping|tags|widget/i,
  135. extraneous:
  136. /print|archive|comment|discuss|e[\-]?mail|share|reply|all|login|sign|single|utility/i,
  137. byline: /byline|author|dateline|writtenby|p-author/i,
  138. replaceFonts: /<(\/?)font[^>]*>/gi,
  139. normalize: /\s{2,}/g,
  140. videos:
  141. /\/\/(www\.)?((dailymotion|youtube|youtube-nocookie|player\.vimeo|v\.qq)\.com|(archive|upload\.wikimedia)\.org|player\.twitch\.tv)/i,
  142. shareElements: /(\b|_)(share|sharedaddy)(\b|_)/i,
  143. nextLink: /(next|weiter|continue|>([^\|]|$)|»([^\|]|$))/i,
  144. prevLink: /(prev|earl|old|new|<|«)/i,
  145. tokenize: /\W+/g,
  146. whitespace: /^\s*$/,
  147. hasContent: /\S$/,
  148. hashUrl: /^#.+/,
  149. srcsetUrl: /(\S+)(\s+[\d.]+[xw])?(\s*(?:,|$))/g,
  150. b64DataUrl: /^data:\s*([^\s;,]+)\s*;\s*base64\s*,/i,
  151. // Commas as used in Latin, Sindhi, Chinese and various other scripts.
  152. // see: https://en.wikipedia.org/wiki/Comma#Comma_variants
  153. commas: /\u002C|\u060C|\uFE50|\uFE10|\uFE11|\u2E41|\u2E34|\u2E32|\uFF0C/g,
  154. // See: https://schema.org/Article
  155. jsonLdArticleTypes:
  156. /^Article|AdvertiserContentArticle|NewsArticle|AnalysisNewsArticle|AskPublicNewsArticle|BackgroundNewsArticle|OpinionNewsArticle|ReportageNewsArticle|ReviewNewsArticle|Report|SatiricalArticle|ScholarlyArticle|MedicalScholarlyArticle|SocialMediaPosting|BlogPosting|LiveBlogPosting|DiscussionForumPosting|TechArticle|APIReference$/,
  157. // used to see if a node's content matches words commonly used for ad blocks or loading indicators
  158. adWords:
  159. /^(ad(vertising|vertisement)?|pub(licité)?|werb(ung)?|广告|Реклама|Anuncio)$/iu,
  160. loadingWords:
  161. /^((loading|正在加载|Загрузка|chargement|cargando)(…|\.\.\.)?)$/iu,
  162. },
  163. UNLIKELY_ROLES: [
  164. "menu",
  165. "menubar",
  166. "complementary",
  167. "navigation",
  168. "alert",
  169. "alertdialog",
  170. "dialog",
  171. ],
  172. DIV_TO_P_ELEMS: new Set([
  173. "BLOCKQUOTE",
  174. "DL",
  175. "DIV",
  176. "IMG",
  177. "OL",
  178. "P",
  179. "PRE",
  180. "TABLE",
  181. "UL",
  182. ]),
  183. ALTER_TO_DIV_EXCEPTIONS: ["DIV", "ARTICLE", "SECTION", "P", "OL", "UL"],
  184. PRESENTATIONAL_ATTRIBUTES: [
  185. "align",
  186. "background",
  187. "bgcolor",
  188. "border",
  189. "cellpadding",
  190. "cellspacing",
  191. "frame",
  192. "hspace",
  193. "rules",
  194. "style",
  195. "valign",
  196. "vspace",
  197. ],
  198. DEPRECATED_SIZE_ATTRIBUTE_ELEMS: ["TABLE", "TH", "TD", "HR", "PRE"],
  199. // The commented out elements qualify as phrasing content but tend to be
  200. // removed by readability when put into paragraphs, so we ignore them here.
  201. PHRASING_ELEMS: [
  202. // "CANVAS", "IFRAME", "SVG", "VIDEO",
  203. "ABBR",
  204. "AUDIO",
  205. "B",
  206. "BDO",
  207. "BR",
  208. "BUTTON",
  209. "CITE",
  210. "CODE",
  211. "DATA",
  212. "DATALIST",
  213. "DFN",
  214. "EM",
  215. "EMBED",
  216. "I",
  217. "IMG",
  218. "INPUT",
  219. "KBD",
  220. "LABEL",
  221. "MARK",
  222. "MATH",
  223. "METER",
  224. "NOSCRIPT",
  225. "OBJECT",
  226. "OUTPUT",
  227. "PROGRESS",
  228. "Q",
  229. "RUBY",
  230. "SAMP",
  231. "SCRIPT",
  232. "SELECT",
  233. "SMALL",
  234. "SPAN",
  235. "STRONG",
  236. "SUB",
  237. "SUP",
  238. "TEXTAREA",
  239. "TIME",
  240. "VAR",
  241. "WBR",
  242. ],
  243. // These are the classes that readability sets itself.
  244. CLASSES_TO_PRESERVE: ["page"],
  245. // These are the list of HTML entities that need to be escaped.
  246. HTML_ESCAPE_MAP: {
  247. lt: "<",
  248. gt: ">",
  249. amp: "&",
  250. quot: '"',
  251. apos: "'",
  252. },
  253. /**
  254. * Run any post-process modifications to article content as necessary.
  255. *
  256. * @param Element
  257. * @return void
  258. **/
  259. _postProcessContent(articleContent) {
  260. // Readability cannot open relative uris so we convert them to absolute uris.
  261. this._fixRelativeUris(articleContent);
  262. this._simplifyNestedElements(articleContent);
  263. if (!this._keepClasses) {
  264. // Remove classes.
  265. this._cleanClasses(articleContent);
  266. }
  267. },
  268. /**
  269. * Iterates over a NodeList, calls `filterFn` for each node and removes node
  270. * if function returned `true`.
  271. *
  272. * If function is not passed, removes all the nodes in node list.
  273. *
  274. * @param NodeList nodeList The nodes to operate on
  275. * @param Function filterFn the function to use as a filter
  276. * @return void
  277. */
  278. _removeNodes(nodeList, filterFn) {
  279. // Avoid ever operating on live node lists.
  280. if (this._docJSDOMParser && nodeList._isLiveNodeList) {
  281. throw new Error("Do not pass live node lists to _removeNodes");
  282. }
  283. for (var i = nodeList.length - 1; i >= 0; i--) {
  284. var node = nodeList[i];
  285. var parentNode = node.parentNode;
  286. if (parentNode) {
  287. if (!filterFn || filterFn.call(this, node, i, nodeList)) {
  288. parentNode.removeChild(node);
  289. }
  290. }
  291. }
  292. },
  293. /**
  294. * Iterates over a NodeList, and calls _setNodeTag for each node.
  295. *
  296. * @param NodeList nodeList The nodes to operate on
  297. * @param String newTagName the new tag name to use
  298. * @return void
  299. */
  300. _replaceNodeTags(nodeList, newTagName) {
  301. // Avoid ever operating on live node lists.
  302. if (this._docJSDOMParser && nodeList._isLiveNodeList) {
  303. throw new Error("Do not pass live node lists to _replaceNodeTags");
  304. }
  305. for (const node of nodeList) {
  306. this._setNodeTag(node, newTagName);
  307. }
  308. },
  309. /**
  310. * Iterate over a NodeList, which doesn't natively fully implement the Array
  311. * interface.
  312. *
  313. * For convenience, the current object context is applied to the provided
  314. * iterate function.
  315. *
  316. * @param NodeList nodeList The NodeList.
  317. * @param Function fn The iterate function.
  318. * @return void
  319. */
  320. _forEachNode(nodeList, fn) {
  321. Array.prototype.forEach.call(nodeList, fn, this);
  322. },
  323. /**
  324. * Iterate over a NodeList, and return the first node that passes
  325. * the supplied test function
  326. *
  327. * For convenience, the current object context is applied to the provided
  328. * test function.
  329. *
  330. * @param NodeList nodeList The NodeList.
  331. * @param Function fn The test function.
  332. * @return void
  333. */
  334. _findNode(nodeList, fn) {
  335. return Array.prototype.find.call(nodeList, fn, this);
  336. },
  337. /**
  338. * Iterate over a NodeList, return true if any of the provided iterate
  339. * function calls returns true, false otherwise.
  340. *
  341. * For convenience, the current object context is applied to the
  342. * provided iterate function.
  343. *
  344. * @param NodeList nodeList The NodeList.
  345. * @param Function fn The iterate function.
  346. * @return Boolean
  347. */
  348. _someNode(nodeList, fn) {
  349. return Array.prototype.some.call(nodeList, fn, this);
  350. },
  351. /**
  352. * Iterate over a NodeList, return true if all of the provided iterate
  353. * function calls return true, false otherwise.
  354. *
  355. * For convenience, the current object context is applied to the
  356. * provided iterate function.
  357. *
  358. * @param NodeList nodeList The NodeList.
  359. * @param Function fn The iterate function.
  360. * @return Boolean
  361. */
  362. _everyNode(nodeList, fn) {
  363. return Array.prototype.every.call(nodeList, fn, this);
  364. },
  365. _getAllNodesWithTag(node, tagNames) {
  366. if (node.querySelectorAll) {
  367. return node.querySelectorAll(tagNames.join(","));
  368. }
  369. return [].concat.apply(
  370. [],
  371. tagNames.map(function (tag) {
  372. var collection = node.getElementsByTagName(tag);
  373. return Array.isArray(collection) ? collection : Array.from(collection);
  374. })
  375. );
  376. },
  377. /**
  378. * Removes the class="" attribute from every element in the given
  379. * subtree, except those that match CLASSES_TO_PRESERVE and
  380. * the classesToPreserve array from the options object.
  381. *
  382. * @param Element
  383. * @return void
  384. */
  385. _cleanClasses(node) {
  386. var classesToPreserve = this._classesToPreserve;
  387. var className = (node.getAttribute("class") || "")
  388. .split(/\s+/)
  389. .filter(cls => classesToPreserve.includes(cls))
  390. .join(" ");
  391. if (className) {
  392. node.setAttribute("class", className);
  393. } else {
  394. node.removeAttribute("class");
  395. }
  396. for (node = node.firstElementChild; node; node = node.nextElementSibling) {
  397. this._cleanClasses(node);
  398. }
  399. },
  400. /**
  401. * Tests whether a string is a URL or not.
  402. *
  403. * @param {string} str The string to test
  404. * @return {boolean} true if str is a URL, false if not
  405. */
  406. _isUrl(str) {
  407. try {
  408. new URL(str);
  409. return true;
  410. } catch {
  411. return false;
  412. }
  413. },
  414. /**
  415. * Converts each <a> and <img> uri in the given element to an absolute URI,
  416. * ignoring #ref URIs.
  417. *
  418. * @param Element
  419. * @return void
  420. */
  421. _fixRelativeUris(articleContent) {
  422. var baseURI = this._doc.baseURI;
  423. var documentURI = this._doc.documentURI;
  424. function toAbsoluteURI(uri) {
  425. // Leave hash links alone if the base URI matches the document URI:
  426. if (baseURI == documentURI && uri.charAt(0) == "#") {
  427. return uri;
  428. }
  429. // Otherwise, resolve against base URI:
  430. try {
  431. return new URL(uri, baseURI).href;
  432. } catch (ex) {
  433. // Something went wrong, just return the original:
  434. }
  435. return uri;
  436. }
  437. var links = this._getAllNodesWithTag(articleContent, ["a"]);
  438. this._forEachNode(links, function (link) {
  439. var href = link.getAttribute("href");
  440. if (href) {
  441. // Remove links with javascript: URIs, since
  442. // they won't work after scripts have been removed from the page.
  443. if (href.indexOf("javascript:") === 0) {
  444. // if the link only contains simple text content, it can be converted to a text node
  445. if (
  446. link.childNodes.length === 1 &&
  447. link.childNodes[0].nodeType === this.TEXT_NODE
  448. ) {
  449. var text = this._doc.createTextNode(link.textContent);
  450. link.parentNode.replaceChild(text, link);
  451. } else {
  452. // if the link has multiple children, they should all be preserved
  453. var container = this._doc.createElement("span");
  454. while (link.firstChild) {
  455. container.appendChild(link.firstChild);
  456. }
  457. link.parentNode.replaceChild(container, link);
  458. }
  459. } else {
  460. link.setAttribute("href", toAbsoluteURI(href));
  461. }
  462. }
  463. });
  464. var medias = this._getAllNodesWithTag(articleContent, [
  465. "img",
  466. "picture",
  467. "figure",
  468. "video",
  469. "audio",
  470. "source",
  471. ]);
  472. this._forEachNode(medias, function (media) {
  473. var src = media.getAttribute("src");
  474. var poster = media.getAttribute("poster");
  475. var srcset = media.getAttribute("srcset");
  476. if (src) {
  477. media.setAttribute("src", toAbsoluteURI(src));
  478. }
  479. if (poster) {
  480. media.setAttribute("poster", toAbsoluteURI(poster));
  481. }
  482. if (srcset) {
  483. var newSrcset = srcset.replace(
  484. this.REGEXPS.srcsetUrl,
  485. function (_, p1, p2, p3) {
  486. return toAbsoluteURI(p1) + (p2 || "") + p3;
  487. }
  488. );
  489. media.setAttribute("srcset", newSrcset);
  490. }
  491. });
  492. },
  493. _simplifyNestedElements(articleContent) {
  494. var node = articleContent;
  495. while (node) {
  496. if (
  497. node.parentNode &&
  498. ["DIV", "SECTION"].includes(node.tagName) &&
  499. !(node.id && node.id.startsWith("readability"))
  500. ) {
  501. if (this._isElementWithoutContent(node)) {
  502. node = this._removeAndGetNext(node);
  503. continue;
  504. } else if (
  505. this._hasSingleTagInsideElement(node, "DIV") ||
  506. this._hasSingleTagInsideElement(node, "SECTION")
  507. ) {
  508. var child = node.children[0];
  509. for (var i = 0; i < node.attributes.length; i++) {
  510. child.setAttributeNode(node.attributes[i].cloneNode());
  511. }
  512. node.parentNode.replaceChild(child, node);
  513. node = child;
  514. continue;
  515. }
  516. }
  517. node = this._getNextNode(node);
  518. }
  519. },
  520. /**
  521. * Get the article title as an H1.
  522. *
  523. * @return string
  524. **/
  525. _getArticleTitle() {
  526. var doc = this._doc;
  527. var curTitle = "";
  528. var origTitle = "";
  529. try {
  530. curTitle = origTitle = doc.title.trim();
  531. // If they had an element with id "title" in their HTML
  532. if (typeof curTitle !== "string") {
  533. curTitle = origTitle = this._getInnerText(
  534. doc.getElementsByTagName("title")[0]
  535. );
  536. }
  537. } catch (e) {
  538. /* ignore exceptions setting the title. */
  539. }
  540. var titleHadHierarchicalSeparators = false;
  541. function wordCount(str) {
  542. return str.split(/\s+/).length;
  543. }
  544. // If there's a separator in the title, first remove the final part
  545. if (/ [\|\-\\\/>»] /.test(curTitle)) {
  546. titleHadHierarchicalSeparators = / [\\\/>»] /.test(curTitle);
  547. let allSeparators = Array.from(origTitle.matchAll(/ [\|\-\\\/>»] /gi));
  548. curTitle = origTitle.substring(0, allSeparators.pop().index);
  549. // If the resulting title is too short, remove the first part instead:
  550. if (wordCount(curTitle) < 3) {
  551. curTitle = origTitle.replace(/^[^\|\-\\\/>»]*[\|\-\\\/>»]/gi, "");
  552. }
  553. } else if (curTitle.includes(": ")) {
  554. // Check if we have an heading containing this exact string, so we
  555. // could assume it's the full title.
  556. var headings = this._getAllNodesWithTag(doc, ["h1", "h2"]);
  557. var trimmedTitle = curTitle.trim();
  558. var match = this._someNode(headings, function (heading) {
  559. return heading.textContent.trim() === trimmedTitle;
  560. });
  561. // If we don't, let's extract the title out of the original title string.
  562. if (!match) {
  563. curTitle = origTitle.substring(origTitle.lastIndexOf(":") + 1);
  564. // If the title is now too short, try the first colon instead:
  565. if (wordCount(curTitle) < 3) {
  566. curTitle = origTitle.substring(origTitle.indexOf(":") + 1);
  567. // But if we have too many words before the colon there's something weird
  568. // with the titles and the H tags so let's just use the original title instead
  569. } else if (wordCount(origTitle.substr(0, origTitle.indexOf(":"))) > 5) {
  570. curTitle = origTitle;
  571. }
  572. }
  573. } else if (curTitle.length > 150 || curTitle.length < 15) {
  574. var hOnes = doc.getElementsByTagName("h1");
  575. if (hOnes.length === 1) {
  576. curTitle = this._getInnerText(hOnes[0]);
  577. }
  578. }
  579. curTitle = curTitle.trim().replace(this.REGEXPS.normalize, " ");
  580. // If we now have 4 words or fewer as our title, and either no
  581. // 'hierarchical' separators (\, /, > or ») were found in the original
  582. // title or we decreased the number of words by more than 1 word, use
  583. // the original title.
  584. var curTitleWordCount = wordCount(curTitle);
  585. if (
  586. curTitleWordCount <= 4 &&
  587. (!titleHadHierarchicalSeparators ||
  588. curTitleWordCount !=
  589. wordCount(origTitle.replace(/[\|\-\\\/>»]+/g, "")) - 1)
  590. ) {
  591. curTitle = origTitle;
  592. }
  593. return curTitle;
  594. },
  595. /**
  596. * Prepare the HTML document for readability to scrape it.
  597. * This includes things like stripping javascript, CSS, and handling terrible markup.
  598. *
  599. * @return void
  600. **/
  601. _prepDocument() {
  602. var doc = this._doc;
  603. // Remove all style tags in head
  604. this._removeNodes(this._getAllNodesWithTag(doc, ["style"]));
  605. if (doc.body) {
  606. this._replaceBrs(doc.body);
  607. }
  608. this._replaceNodeTags(this._getAllNodesWithTag(doc, ["font"]), "SPAN");
  609. },
  610. /**
  611. * Finds the next node, starting from the given node, and ignoring
  612. * whitespace in between. If the given node is an element, the same node is
  613. * returned.
  614. */
  615. _nextNode(node) {
  616. var next = node;
  617. while (
  618. next &&
  619. next.nodeType != this.ELEMENT_NODE &&
  620. this.REGEXPS.whitespace.test(next.textContent)
  621. ) {
  622. next = next.nextSibling;
  623. }
  624. return next;
  625. },
  626. /**
  627. * Replaces 2 or more successive <br> elements with a single <p>.
  628. * Whitespace between <br> elements are ignored. For example:
  629. * <div>foo<br>bar<br> <br><br>abc</div>
  630. * will become:
  631. * <div>foo<br>bar<p>abc</p></div>
  632. */
  633. _replaceBrs(elem) {
  634. this._forEachNode(this._getAllNodesWithTag(elem, ["br"]), function (br) {
  635. var next = br.nextSibling;
  636. // Whether 2 or more <br> elements have been found and replaced with a
  637. // <p> block.
  638. var replaced = false;
  639. // If we find a <br> chain, remove the <br>s until we hit another node
  640. // or non-whitespace. This leaves behind the first <br> in the chain
  641. // (which will be replaced with a <p> later).
  642. while ((next = this._nextNode(next)) && next.tagName == "BR") {
  643. replaced = true;
  644. var brSibling = next.nextSibling;
  645. next.remove();
  646. next = brSibling;
  647. }
  648. // If we removed a <br> chain, replace the remaining <br> with a <p>. Add
  649. // all sibling nodes as children of the <p> until we hit another <br>
  650. // chain.
  651. if (replaced) {
  652. var p = this._doc.createElement("p");
  653. br.parentNode.replaceChild(p, br);
  654. next = p.nextSibling;
  655. while (next) {
  656. // If we've hit another <br><br>, we're done adding children to this <p>.
  657. if (next.tagName == "BR") {
  658. var nextElem = this._nextNode(next.nextSibling);
  659. if (nextElem && nextElem.tagName == "BR") {
  660. break;
  661. }
  662. }
  663. if (!this._isPhrasingContent(next)) {
  664. break;
  665. }
  666. // Otherwise, make this node a child of the new <p>.
  667. var sibling = next.nextSibling;
  668. p.appendChild(next);
  669. next = sibling;
  670. }
  671. while (p.lastChild && this._isWhitespace(p.lastChild)) {
  672. p.lastChild.remove();
  673. }
  674. if (p.parentNode.tagName === "P") {
  675. this._setNodeTag(p.parentNode, "DIV");
  676. }
  677. }
  678. });
  679. },
  680. _setNodeTag(node, tag) {
  681. this.log("_setNodeTag", node, tag);
  682. if (this._docJSDOMParser) {
  683. node.localName = tag.toLowerCase();
  684. node.tagName = tag.toUpperCase();
  685. return node;
  686. }
  687. var replacement = node.ownerDocument.createElement(tag);
  688. while (node.firstChild) {
  689. replacement.appendChild(node.firstChild);
  690. }
  691. node.parentNode.replaceChild(replacement, node);
  692. if (node.readability) {
  693. replacement.readability = node.readability;
  694. }
  695. for (var i = 0; i < node.attributes.length; i++) {
  696. replacement.setAttributeNode(node.attributes[i].cloneNode());
  697. }
  698. return replacement;
  699. },
  700. /**
  701. * Prepare the article node for display. Clean out any inline styles,
  702. * iframes, forms, strip extraneous <p> tags, etc.
  703. *
  704. * @param Element
  705. * @return void
  706. **/
  707. _prepArticle(articleContent) {
  708. this._cleanStyles(articleContent);
  709. // Check for data tables before we continue, to avoid removing items in
  710. // those tables, which will often be isolated even though they're
  711. // visually linked to other content-ful elements (text, images, etc.).
  712. this._markDataTables(articleContent);
  713. this._fixLazyImages(articleContent);
  714. // Clean out junk from the article content
  715. this._cleanConditionally(articleContent, "form");
  716. this._cleanConditionally(articleContent, "fieldset");
  717. this._clean(articleContent, "object");
  718. this._clean(articleContent, "embed");
  719. this._clean(articleContent, "footer");
  720. this._clean(articleContent, "link");
  721. this._clean(articleContent, "aside");
  722. // Clean out elements with little content that have "share" in their id/class combinations from final top candidates,
  723. // which means we don't remove the top candidates even they have "share".
  724. var shareElementThreshold = this.DEFAULT_CHAR_THRESHOLD;
  725. this._forEachNode(articleContent.children, function (topCandidate) {
  726. this._cleanMatchedNodes(topCandidate, function (node, matchString) {
  727. return (
  728. this.REGEXPS.shareElements.test(matchString) &&
  729. node.textContent.length < shareElementThreshold
  730. );
  731. });
  732. });
  733. this._clean(articleContent, "iframe");
  734. this._clean(articleContent, "input");
  735. this._clean(articleContent, "textarea");
  736. this._clean(articleContent, "select");
  737. this._clean(articleContent, "button");
  738. this._cleanHeaders(articleContent);
  739. // Do these last as the previous stuff may have removed junk
  740. // that will affect these
  741. this._cleanConditionally(articleContent, "table");
  742. this._cleanConditionally(articleContent, "ul");
  743. this._cleanConditionally(articleContent, "div");
  744. // replace H1 with H2 as H1 should be only title that is displayed separately
  745. this._replaceNodeTags(
  746. this._getAllNodesWithTag(articleContent, ["h1"]),
  747. "h2"
  748. );
  749. // Remove extra paragraphs
  750. this._removeNodes(
  751. this._getAllNodesWithTag(articleContent, ["p"]),
  752. function (paragraph) {
  753. // At this point, nasty iframes have been removed; only embedded video
  754. // ones remain.
  755. var contentElementCount = this._getAllNodesWithTag(paragraph, [
  756. "img",
  757. "embed",
  758. "object",
  759. "iframe",
  760. ]).length;
  761. return (
  762. contentElementCount === 0 && !this._getInnerText(paragraph, false)
  763. );
  764. }
  765. );
  766. this._forEachNode(
  767. this._getAllNodesWithTag(articleContent, ["br"]),
  768. function (br) {
  769. var next = this._nextNode(br.nextSibling);
  770. if (next && next.tagName == "P") {
  771. br.remove();
  772. }
  773. }
  774. );
  775. // Remove single-cell tables
  776. this._forEachNode(
  777. this._getAllNodesWithTag(articleContent, ["table"]),
  778. function (table) {
  779. var tbody = this._hasSingleTagInsideElement(table, "TBODY")
  780. ? table.firstElementChild
  781. : table;
  782. if (this._hasSingleTagInsideElement(tbody, "TR")) {
  783. var row = tbody.firstElementChild;
  784. if (this._hasSingleTagInsideElement(row, "TD")) {
  785. var cell = row.firstElementChild;
  786. cell = this._setNodeTag(
  787. cell,
  788. this._everyNode(cell.childNodes, this._isPhrasingContent)
  789. ? "P"
  790. : "DIV"
  791. );
  792. table.parentNode.replaceChild(cell, table);
  793. }
  794. }
  795. }
  796. );
  797. },
  798. /**
  799. * Initialize a node with the readability object. Also checks the
  800. * className/id for special names to add to its score.
  801. *
  802. * @param Element
  803. * @return void
  804. **/
  805. _initializeNode(node) {
  806. node.readability = { contentScore: 0 };
  807. switch (node.tagName) {
  808. case "DIV":
  809. node.readability.contentScore += 5;
  810. break;
  811. case "PRE":
  812. case "TD":
  813. case "BLOCKQUOTE":
  814. node.readability.contentScore += 3;
  815. break;
  816. case "ADDRESS":
  817. case "OL":
  818. case "UL":
  819. case "DL":
  820. case "DD":
  821. case "DT":
  822. case "LI":
  823. case "FORM":
  824. node.readability.contentScore -= 3;
  825. break;
  826. case "H1":
  827. case "H2":
  828. case "H3":
  829. case "H4":
  830. case "H5":
  831. case "H6":
  832. case "TH":
  833. node.readability.contentScore -= 5;
  834. break;
  835. }
  836. node.readability.contentScore += this._getClassWeight(node);
  837. },
  838. _removeAndGetNext(node) {
  839. var nextNode = this._getNextNode(node, true);
  840. node.remove();
  841. return nextNode;
  842. },
  843. /**
  844. * Traverse the DOM from node to node, starting at the node passed in.
  845. * Pass true for the second parameter to indicate this node itself
  846. * (and its kids) are going away, and we want the next node over.
  847. *
  848. * Calling this in a loop will traverse the DOM depth-first.
  849. *
  850. * @param {Element} node
  851. * @param {boolean} ignoreSelfAndKids
  852. * @return {Element}
  853. */
  854. _getNextNode(node, ignoreSelfAndKids) {
  855. // First check for kids if those aren't being ignored
  856. if (!ignoreSelfAndKids && node.firstElementChild) {
  857. return node.firstElementChild;
  858. }
  859. // Then for siblings...
  860. if (node.nextElementSibling) {
  861. return node.nextElementSibling;
  862. }
  863. // And finally, move up the parent chain *and* find a sibling
  864. // (because this is depth-first traversal, we will have already
  865. // seen the parent nodes themselves).
  866. do {
  867. node = node.parentNode;
  868. } while (node && !node.nextElementSibling);
  869. return node && node.nextElementSibling;
  870. },
  871. // compares second text to first one
  872. // 1 = same text, 0 = completely different text
  873. // works the way that it splits both texts into words and then finds words that are unique in second text
  874. // the result is given by the lower length of unique parts
  875. _textSimilarity(textA, textB) {
  876. var tokensA = textA
  877. .toLowerCase()
  878. .split(this.REGEXPS.tokenize)
  879. .filter(Boolean);
  880. var tokensB = textB
  881. .toLowerCase()
  882. .split(this.REGEXPS.tokenize)
  883. .filter(Boolean);
  884. if (!tokensA.length || !tokensB.length) {
  885. return 0;
  886. }
  887. var uniqTokensB = tokensB.filter(token => !tokensA.includes(token));
  888. var distanceB = uniqTokensB.join(" ").length / tokensB.join(" ").length;
  889. return 1 - distanceB;
  890. },
  891. /**
  892. * Checks whether an element node contains a valid byline
  893. *
  894. * @param node {Element}
  895. * @param matchString {string}
  896. * @return boolean
  897. */
  898. _isValidByline(node, matchString) {
  899. var rel = node.getAttribute("rel");
  900. var itemprop = node.getAttribute("itemprop");
  901. var bylineLength = node.textContent.trim().length;
  902. return (
  903. (rel === "author" ||
  904. (itemprop && itemprop.includes("author")) ||
  905. this.REGEXPS.byline.test(matchString)) &&
  906. !!bylineLength &&
  907. bylineLength < 100
  908. );
  909. },
  910. _getNodeAncestors(node, maxDepth) {
  911. maxDepth = maxDepth || 0;
  912. var i = 0,
  913. ancestors = [];
  914. while (node.parentNode) {
  915. ancestors.push(node.parentNode);
  916. if (maxDepth && ++i === maxDepth) {
  917. break;
  918. }
  919. node = node.parentNode;
  920. }
  921. return ancestors;
  922. },
  923. /***
  924. * grabArticle - Using a variety of metrics (content score, classname, element types), find the content that is
  925. * most likely to be the stuff a user wants to read. Then return it wrapped up in a div.
  926. *
  927. * @param page a document to run upon. Needs to be a full document, complete with body.
  928. * @return Element
  929. **/
  930. /* eslint-disable-next-line complexity */
  931. _grabArticle(page) {
  932. this.log("**** grabArticle ****");
  933. var doc = this._doc;
  934. var isPaging = page !== null;
  935. page = page ? page : this._doc.body;
  936. // We can't grab an article if we don't have a page!
  937. if (!page) {
  938. this.log("No body found in document. Abort.");
  939. return null;
  940. }
  941. var pageCacheHtml = page.innerHTML;
  942. while (true) {
  943. this.log("Starting grabArticle loop");
  944. var stripUnlikelyCandidates = this._flagIsActive(
  945. this.FLAG_STRIP_UNLIKELYS
  946. );
  947. // First, node prepping. Trash nodes that look cruddy (like ones with the
  948. // class name "comment", etc), and turn divs into P tags where they have been
  949. // used inappropriately (as in, where they contain no other block level elements.)
  950. var elementsToScore = [];
  951. var node = this._doc.documentElement;
  952. let shouldRemoveTitleHeader = true;
  953. while (node) {
  954. if (node.tagName === "HTML") {
  955. this._articleLang = node.getAttribute("lang");
  956. }
  957. var matchString = node.className + " " + node.id;
  958. if (!this._isProbablyVisible(node)) {
  959. this.log("Removing hidden node - " + matchString);
  960. node = this._removeAndGetNext(node);
  961. continue;
  962. }
  963. // User is not able to see elements applied with both "aria-modal = true" and "role = dialog"
  964. if (
  965. node.getAttribute("aria-modal") == "true" &&
  966. node.getAttribute("role") == "dialog"
  967. ) {
  968. node = this._removeAndGetNext(node);
  969. continue;
  970. }
  971. // If we don't have a byline yet check to see if this node is a byline; if it is store the byline and remove the node.
  972. if (
  973. !this._articleByline &&
  974. !this._metadata.byline &&
  975. this._isValidByline(node, matchString)
  976. ) {
  977. // Find child node matching [itemprop="name"] and use that if it exists for a more accurate author name byline
  978. var endOfSearchMarkerNode = this._getNextNode(node, true);
  979. var next = this._getNextNode(node);
  980. var itemPropNameNode = null;
  981. while (next && next != endOfSearchMarkerNode) {
  982. var itemprop = next.getAttribute("itemprop");
  983. if (itemprop && itemprop.includes("name")) {
  984. itemPropNameNode = next;
  985. break;
  986. } else {
  987. next = this._getNextNode(next);
  988. }
  989. }
  990. this._articleByline = (itemPropNameNode ?? node).textContent.trim();
  991. node = this._removeAndGetNext(node);
  992. continue;
  993. }
  994. if (shouldRemoveTitleHeader && this._headerDuplicatesTitle(node)) {
  995. this.log(
  996. "Removing header: ",
  997. node.textContent.trim(),
  998. this._articleTitle.trim()
  999. );
  1000. shouldRemoveTitleHeader = false;
  1001. node = this._removeAndGetNext(node);
  1002. continue;
  1003. }
  1004. // Remove unlikely candidates
  1005. if (stripUnlikelyCandidates) {
  1006. if (
  1007. this.REGEXPS.unlikelyCandidates.test(matchString) &&
  1008. !this.REGEXPS.okMaybeItsACandidate.test(matchString) &&
  1009. !this._hasAncestorTag(node, "table") &&
  1010. !this._hasAncestorTag(node, "code") &&
  1011. node.tagName !== "BODY" &&
  1012. node.tagName !== "A"
  1013. ) {
  1014. this.log("Removing unlikely candidate - " + matchString);
  1015. node = this._removeAndGetNext(node);
  1016. continue;
  1017. }
  1018. if (this.UNLIKELY_ROLES.includes(node.getAttribute("role"))) {
  1019. this.log(
  1020. "Removing content with role " +
  1021. node.getAttribute("role") +
  1022. " - " +
  1023. matchString
  1024. );
  1025. node = this._removeAndGetNext(node);
  1026. continue;
  1027. }
  1028. }
  1029. // Remove DIV, SECTION, and HEADER nodes without any content(e.g. text, image, video, or iframe).
  1030. if (
  1031. (node.tagName === "DIV" ||
  1032. node.tagName === "SECTION" ||
  1033. node.tagName === "HEADER" ||
  1034. node.tagName === "H1" ||
  1035. node.tagName === "H2" ||
  1036. node.tagName === "H3" ||
  1037. node.tagName === "H4" ||
  1038. node.tagName === "H5" ||
  1039. node.tagName === "H6") &&
  1040. this._isElementWithoutContent(node)
  1041. ) {
  1042. node = this._removeAndGetNext(node);
  1043. continue;
  1044. }
  1045. if (this.DEFAULT_TAGS_TO_SCORE.includes(node.tagName)) {
  1046. elementsToScore.push(node);
  1047. }
  1048. // Turn all divs that don't have children block level elements into p's
  1049. if (node.tagName === "DIV") {
  1050. // Put phrasing content into paragraphs.
  1051. var p = null;
  1052. var childNode = node.firstChild;
  1053. while (childNode) {
  1054. var nextSibling = childNode.nextSibling;
  1055. if (this._isPhrasingContent(childNode)) {
  1056. if (p !== null) {
  1057. p.appendChild(childNode);
  1058. } else if (!this._isWhitespace(childNode)) {
  1059. p = doc.createElement("p");
  1060. node.replaceChild(p, childNode);
  1061. p.appendChild(childNode);
  1062. }
  1063. } else if (p !== null) {
  1064. while (p.lastChild && this._isWhitespace(p.lastChild)) {
  1065. p.lastChild.remove();
  1066. }
  1067. p = null;
  1068. }
  1069. childNode = nextSibling;
  1070. }
  1071. // Sites like http://mobile.slate.com encloses each paragraph with a DIV
  1072. // element. DIVs with only a P element inside and no text content can be
  1073. // safely converted into plain P elements to avoid confusing the scoring
  1074. // algorithm with DIVs with are, in practice, paragraphs.
  1075. if (
  1076. this._hasSingleTagInsideElement(node, "P") &&
  1077. this._getLinkDensity(node) < 0.25
  1078. ) {
  1079. var newNode = node.children[0];
  1080. node.parentNode.replaceChild(newNode, node);
  1081. node = newNode;
  1082. elementsToScore.push(node);
  1083. } else if (!this._hasChildBlockElement(node)) {
  1084. node = this._setNodeTag(node, "P");
  1085. elementsToScore.push(node);
  1086. }
  1087. }
  1088. node = this._getNextNode(node);
  1089. }
  1090. /**
  1091. * Loop through all paragraphs, and assign a score to them based on how content-y they look.
  1092. * Then add their score to their parent node.
  1093. *
  1094. * A score is determined by things like number of commas, class names, etc. Maybe eventually link density.
  1095. **/
  1096. var candidates = [];
  1097. this._forEachNode(elementsToScore, function (elementToScore) {
  1098. if (
  1099. !elementToScore.parentNode ||
  1100. typeof elementToScore.parentNode.tagName === "undefined"
  1101. ) {
  1102. return;
  1103. }
  1104. // If this paragraph is less than 25 characters, don't even count it.
  1105. var innerText = this._getInnerText(elementToScore);
  1106. if (innerText.length < 25) {
  1107. return;
  1108. }
  1109. // Exclude nodes with no ancestor.
  1110. var ancestors = this._getNodeAncestors(elementToScore, 5);
  1111. if (ancestors.length === 0) {
  1112. return;
  1113. }
  1114. var contentScore = 0;
  1115. // Add a point for the paragraph itself as a base.
  1116. contentScore += 1;
  1117. // Add points for any commas within this paragraph.
  1118. contentScore += innerText.split(this.REGEXPS.commas).length;
  1119. // For every 100 characters in this paragraph, add another point. Up to 3 points.
  1120. contentScore += Math.min(Math.floor(innerText.length / 100), 3);
  1121. // Initialize and score ancestors.
  1122. this._forEachNode(ancestors, function (ancestor, level) {
  1123. if (
  1124. !ancestor.tagName ||
  1125. !ancestor.parentNode ||
  1126. typeof ancestor.parentNode.tagName === "undefined"
  1127. ) {
  1128. return;
  1129. }
  1130. if (typeof ancestor.readability === "undefined") {
  1131. this._initializeNode(ancestor);
  1132. candidates.push(ancestor);
  1133. }
  1134. // Node score divider:
  1135. // - parent: 1 (no division)
  1136. // - grandparent: 2
  1137. // - great grandparent+: ancestor level * 3
  1138. if (level === 0) {
  1139. var scoreDivider = 1;
  1140. } else if (level === 1) {
  1141. scoreDivider = 2;
  1142. } else {
  1143. scoreDivider = level * 3;
  1144. }
  1145. ancestor.readability.contentScore += contentScore / scoreDivider;
  1146. });
  1147. });
  1148. // After we've calculated scores, loop through all of the possible
  1149. // candidate nodes we found and find the one with the highest score.
  1150. var topCandidates = [];
  1151. for (var c = 0, cl = candidates.length; c < cl; c += 1) {
  1152. var candidate = candidates[c];
  1153. // Scale the final candidates score based on link density. Good content
  1154. // should have a relatively small link density (5% or less) and be mostly
  1155. // unaffected by this operation.
  1156. var candidateScore =
  1157. candidate.readability.contentScore *
  1158. (1 - this._getLinkDensity(candidate));
  1159. candidate.readability.contentScore = candidateScore;
  1160. this.log("Candidate:", candidate, "with score " + candidateScore);
  1161. for (var t = 0; t < this._nbTopCandidates; t++) {
  1162. var aTopCandidate = topCandidates[t];
  1163. if (
  1164. !aTopCandidate ||
  1165. candidateScore > aTopCandidate.readability.contentScore
  1166. ) {
  1167. topCandidates.splice(t, 0, candidate);
  1168. if (topCandidates.length > this._nbTopCandidates) {
  1169. topCandidates.pop();
  1170. }
  1171. break;
  1172. }
  1173. }
  1174. }
  1175. var topCandidate = topCandidates[0] || null;
  1176. var neededToCreateTopCandidate = false;
  1177. var parentOfTopCandidate;
  1178. // If we still have no top candidate, just use the body as a last resort.
  1179. // We also have to copy the body node so it is something we can modify.
  1180. if (topCandidate === null || topCandidate.tagName === "BODY") {
  1181. // Move all of the page's children into topCandidate
  1182. topCandidate = doc.createElement("DIV");
  1183. neededToCreateTopCandidate = true;
  1184. // Move everything (not just elements, also text nodes etc.) into the container
  1185. // so we even include text directly in the body:
  1186. while (page.firstChild) {
  1187. this.log("Moving child out:", page.firstChild);
  1188. topCandidate.appendChild(page.firstChild);
  1189. }
  1190. page.appendChild(topCandidate);
  1191. this._initializeNode(topCandidate);
  1192. } else if (topCandidate) {
  1193. // Find a better top candidate node if it contains (at least three) nodes which belong to `topCandidates` array
  1194. // and whose scores are quite closed with current `topCandidate` node.
  1195. var alternativeCandidateAncestors = [];
  1196. for (var i = 1; i < topCandidates.length; i++) {
  1197. if (
  1198. topCandidates[i].readability.contentScore /
  1199. topCandidate.readability.contentScore >=
  1200. 0.75
  1201. ) {
  1202. alternativeCandidateAncestors.push(
  1203. this._getNodeAncestors(topCandidates[i])
  1204. );
  1205. }
  1206. }
  1207. var MINIMUM_TOPCANDIDATES = 3;
  1208. if (alternativeCandidateAncestors.length >= MINIMUM_TOPCANDIDATES) {
  1209. parentOfTopCandidate = topCandidate.parentNode;
  1210. while (parentOfTopCandidate.tagName !== "BODY") {
  1211. var listsContainingThisAncestor = 0;
  1212. for (
  1213. var ancestorIndex = 0;
  1214. ancestorIndex < alternativeCandidateAncestors.length &&
  1215. listsContainingThisAncestor < MINIMUM_TOPCANDIDATES;
  1216. ancestorIndex++
  1217. ) {
  1218. listsContainingThisAncestor += Number(
  1219. alternativeCandidateAncestors[ancestorIndex].includes(
  1220. parentOfTopCandidate
  1221. )
  1222. );
  1223. }
  1224. if (listsContainingThisAncestor >= MINIMUM_TOPCANDIDATES) {
  1225. topCandidate = parentOfTopCandidate;
  1226. break;
  1227. }
  1228. parentOfTopCandidate = parentOfTopCandidate.parentNode;
  1229. }
  1230. }
  1231. if (!topCandidate.readability) {
  1232. this._initializeNode(topCandidate);
  1233. }
  1234. // Because of our bonus system, parents of candidates might have scores
  1235. // themselves. They get half of the node. There won't be nodes with higher
  1236. // scores than our topCandidate, but if we see the score going *up* in the first
  1237. // few steps up the tree, that's a decent sign that there might be more content
  1238. // lurking in other places that we want to unify in. The sibling stuff
  1239. // below does some of that - but only if we've looked high enough up the DOM
  1240. // tree.
  1241. parentOfTopCandidate = topCandidate.parentNode;
  1242. var lastScore = topCandidate.readability.contentScore;
  1243. // The scores shouldn't get too low.
  1244. var scoreThreshold = lastScore / 3;
  1245. while (parentOfTopCandidate.tagName !== "BODY") {
  1246. if (!parentOfTopCandidate.readability) {
  1247. parentOfTopCandidate = parentOfTopCandidate.parentNode;
  1248. continue;
  1249. }
  1250. var parentScore = parentOfTopCandidate.readability.contentScore;
  1251. if (parentScore < scoreThreshold) {
  1252. break;
  1253. }
  1254. if (parentScore > lastScore) {
  1255. // Alright! We found a better parent to use.
  1256. topCandidate = parentOfTopCandidate;
  1257. break;
  1258. }
  1259. lastScore = parentOfTopCandidate.readability.contentScore;
  1260. parentOfTopCandidate = parentOfTopCandidate.parentNode;
  1261. }
  1262. // If the top candidate is the only child, use parent instead. This will help sibling
  1263. // joining logic when adjacent content is actually located in parent's sibling node.
  1264. parentOfTopCandidate = topCandidate.parentNode;
  1265. while (
  1266. parentOfTopCandidate.tagName != "BODY" &&
  1267. parentOfTopCandidate.children.length == 1
  1268. ) {
  1269. topCandidate = parentOfTopCandidate;
  1270. parentOfTopCandidate = topCandidate.parentNode;
  1271. }
  1272. if (!topCandidate.readability) {
  1273. this._initializeNode(topCandidate);
  1274. }
  1275. }
  1276. // Now that we have the top candidate, look through its siblings for content
  1277. // that might also be related. Things like preambles, content split by ads
  1278. // that we removed, etc.
  1279. var articleContent = doc.createElement("DIV");
  1280. if (isPaging) {
  1281. articleContent.id = "readability-content";
  1282. }
  1283. var siblingScoreThreshold = Math.max(
  1284. 10,
  1285. topCandidate.readability.contentScore * 0.2
  1286. );
  1287. // Keep potential top candidate's parent node to try to get text direction of it later.
  1288. parentOfTopCandidate = topCandidate.parentNode;
  1289. var siblings = parentOfTopCandidate.children;
  1290. for (var s = 0, sl = siblings.length; s < sl; s++) {
  1291. var sibling = siblings[s];
  1292. var append = false;
  1293. this.log(
  1294. "Looking at sibling node:",
  1295. sibling,
  1296. sibling.readability
  1297. ? "with score " + sibling.readability.contentScore
  1298. : ""
  1299. );
  1300. this.log(
  1301. "Sibling has score",
  1302. sibling.readability ? sibling.readability.contentScore : "Unknown"
  1303. );
  1304. if (sibling === topCandidate) {
  1305. append = true;
  1306. } else {
  1307. var contentBonus = 0;
  1308. // Give a bonus if sibling nodes and top candidates have the example same classname
  1309. if (
  1310. sibling.className === topCandidate.className &&
  1311. topCandidate.className !== ""
  1312. ) {
  1313. contentBonus += topCandidate.readability.contentScore * 0.2;
  1314. }
  1315. if (
  1316. sibling.readability &&
  1317. sibling.readability.contentScore + contentBonus >=
  1318. siblingScoreThreshold
  1319. ) {
  1320. append = true;
  1321. } else if (sibling.nodeName === "P") {
  1322. var linkDensity = this._getLinkDensity(sibling);
  1323. var nodeContent = this._getInnerText(sibling);
  1324. var nodeLength = nodeContent.length;
  1325. if (nodeLength > 80 && linkDensity < 0.25) {
  1326. append = true;
  1327. } else if (
  1328. nodeLength < 80 &&
  1329. nodeLength > 0 &&
  1330. linkDensity === 0 &&
  1331. nodeContent.search(/\.( |$)/) !== -1
  1332. ) {
  1333. append = true;
  1334. }
  1335. }
  1336. }
  1337. if (append) {
  1338. this.log("Appending node:", sibling);
  1339. if (!this.ALTER_TO_DIV_EXCEPTIONS.includes(sibling.nodeName)) {
  1340. // We have a node that isn't a common block level element, like a form or td tag.
  1341. // Turn it into a div so it doesn't get filtered out later by accident.
  1342. this.log("Altering sibling:", sibling, "to div.");
  1343. sibling = this._setNodeTag(sibling, "DIV");
  1344. }
  1345. articleContent.appendChild(sibling);
  1346. // Fetch children again to make it compatible
  1347. // with DOM parsers without live collection support.
  1348. siblings = parentOfTopCandidate.children;
  1349. // siblings is a reference to the children array, and
  1350. // sibling is removed from the array when we call appendChild().
  1351. // As a result, we must revisit this index since the nodes
  1352. // have been shifted.
  1353. s -= 1;
  1354. sl -= 1;
  1355. }
  1356. }
  1357. if (this._debug) {
  1358. this.log("Article content pre-prep: " + articleContent.innerHTML);
  1359. }
  1360. // So we have all of the content that we need. Now we clean it up for presentation.
  1361. this._prepArticle(articleContent);
  1362. if (this._debug) {
  1363. this.log("Article content post-prep: " + articleContent.innerHTML);
  1364. }
  1365. if (neededToCreateTopCandidate) {
  1366. // We already created a fake div thing, and there wouldn't have been any siblings left
  1367. // for the previous loop, so there's no point trying to create a new div, and then
  1368. // move all the children over. Just assign IDs and class names here. No need to append
  1369. // because that already happened anyway.
  1370. topCandidate.id = "readability-page-1";
  1371. topCandidate.className = "page";
  1372. } else {
  1373. var div = doc.createElement("DIV");
  1374. div.id = "readability-page-1";
  1375. div.className = "page";
  1376. while (articleContent.firstChild) {
  1377. div.appendChild(articleContent.firstChild);
  1378. }
  1379. articleContent.appendChild(div);
  1380. }
  1381. if (this._debug) {
  1382. this.log("Article content after paging: " + articleContent.innerHTML);
  1383. }
  1384. var parseSuccessful = true;
  1385. // Now that we've gone through the full algorithm, check to see if
  1386. // we got any meaningful content. If we didn't, we may need to re-run
  1387. // grabArticle with different flags set. This gives us a higher likelihood of
  1388. // finding the content, and the sieve approach gives us a higher likelihood of
  1389. // finding the -right- content.
  1390. var textLength = this._getInnerText(articleContent, true).length;
  1391. if (textLength < this._charThreshold) {
  1392. parseSuccessful = false;
  1393. // eslint-disable-next-line no-unsanitized/property
  1394. page.innerHTML = pageCacheHtml;
  1395. this._attempts.push({
  1396. articleContent,
  1397. textLength,
  1398. });
  1399. if (this._flagIsActive(this.FLAG_STRIP_UNLIKELYS)) {
  1400. this._removeFlag(this.FLAG_STRIP_UNLIKELYS);
  1401. } else if (this._flagIsActive(this.FLAG_WEIGHT_CLASSES)) {
  1402. this._removeFlag(this.FLAG_WEIGHT_CLASSES);
  1403. } else if (this._flagIsActive(this.FLAG_CLEAN_CONDITIONALLY)) {
  1404. this._removeFlag(this.FLAG_CLEAN_CONDITIONALLY);
  1405. } else {
  1406. // No luck after removing flags, just return the longest text we found during the different loops
  1407. this._attempts.sort(function (a, b) {
  1408. return b.textLength - a.textLength;
  1409. });
  1410. // But first check if we actually have something
  1411. if (!this._attempts[0].textLength) {
  1412. return null;
  1413. }
  1414. articleContent = this._attempts[0].articleContent;
  1415. parseSuccessful = true;
  1416. }
  1417. }
  1418. if (parseSuccessful) {
  1419. // Find out text direction from ancestors of final top candidate.
  1420. var ancestors = [parentOfTopCandidate, topCandidate].concat(
  1421. this._getNodeAncestors(parentOfTopCandidate)
  1422. );
  1423. this._someNode(ancestors, function (ancestor) {
  1424. if (!ancestor.tagName) {
  1425. return false;
  1426. }
  1427. var articleDir = ancestor.getAttribute("dir");
  1428. if (articleDir) {
  1429. this._articleDir = articleDir;
  1430. return true;
  1431. }
  1432. return false;
  1433. });
  1434. return articleContent;
  1435. }
  1436. }
  1437. },
  1438. /**
  1439. * Converts some of the common HTML entities in string to their corresponding characters.
  1440. *
  1441. * @param str {string} - a string to unescape.
  1442. * @return string without HTML entity.
  1443. */
  1444. _unescapeHtmlEntities(str) {
  1445. if (!str) {
  1446. return str;
  1447. }
  1448. var htmlEscapeMap = this.HTML_ESCAPE_MAP;
  1449. return str
  1450. .replace(/&(quot|amp|apos|lt|gt);/g, function (_, tag) {
  1451. return htmlEscapeMap[tag];
  1452. })
  1453. .replace(/&#(?:x([0-9a-f]+)|([0-9]+));/gi, function (_, hex, numStr) {
  1454. var num = parseInt(hex || numStr, hex ? 16 : 10);
  1455. // these character references are replaced by a conforming HTML parser
  1456. if (num == 0 || num > 0x10ffff || (num >= 0xd800 && num <= 0xdfff)) {
  1457. num = 0xfffd;
  1458. }
  1459. return String.fromCodePoint(num);
  1460. });
  1461. },
  1462. /**
  1463. * Try to extract metadata from JSON-LD object.
  1464. * For now, only Schema.org objects of type Article or its subtypes are supported.
  1465. * @return Object with any metadata that could be extracted (possibly none)
  1466. */
  1467. _getJSONLD(doc) {
  1468. var scripts = this._getAllNodesWithTag(doc, ["script"]);
  1469. var metadata;
  1470. this._forEachNode(scripts, function (jsonLdElement) {
  1471. if (
  1472. !metadata &&
  1473. jsonLdElement.getAttribute("type") === "application/ld+json"
  1474. ) {
  1475. try {
  1476. // Strip CDATA markers if present
  1477. var content = jsonLdElement.textContent.replace(
  1478. /^\s*<!\[CDATA\[|\]\]>\s*$/g,
  1479. ""
  1480. );
  1481. var parsed = JSON.parse(content);
  1482. if (Array.isArray(parsed)) {
  1483. parsed = parsed.find(it => {
  1484. return (
  1485. it["@type"] &&
  1486. it["@type"].match(this.REGEXPS.jsonLdArticleTypes)
  1487. );
  1488. });
  1489. if (!parsed) {
  1490. return;
  1491. }
  1492. }
  1493. var schemaDotOrgRegex = /^https?\:\/\/schema\.org\/?$/;
  1494. var matches =
  1495. (typeof parsed["@context"] === "string" &&
  1496. parsed["@context"].match(schemaDotOrgRegex)) ||
  1497. (typeof parsed["@context"] === "object" &&
  1498. typeof parsed["@context"]["@vocab"] == "string" &&
  1499. parsed["@context"]["@vocab"].match(schemaDotOrgRegex));
  1500. if (!matches) {
  1501. return;
  1502. }
  1503. if (!parsed["@type"] && Array.isArray(parsed["@graph"])) {
  1504. parsed = parsed["@graph"].find(it => {
  1505. return (it["@type"] || "").match(this.REGEXPS.jsonLdArticleTypes);
  1506. });
  1507. }
  1508. if (
  1509. !parsed ||
  1510. !parsed["@type"] ||
  1511. !parsed["@type"].match(this.REGEXPS.jsonLdArticleTypes)
  1512. ) {
  1513. return;
  1514. }
  1515. metadata = {};
  1516. if (
  1517. typeof parsed.name === "string" &&
  1518. typeof parsed.headline === "string" &&
  1519. parsed.name !== parsed.headline
  1520. ) {
  1521. // we have both name and headline element in the JSON-LD. They should both be the same but some websites like aktualne.cz
  1522. // put their own name into "name" and the article title to "headline" which confuses Readability. So we try to check if either
  1523. // "name" or "headline" closely matches the html title, and if so, use that one. If not, then we use "name" by default.
  1524. var title = this._getArticleTitle();
  1525. var nameMatches = this._textSimilarity(parsed.name, title) > 0.75;
  1526. var headlineMatches =
  1527. this._textSimilarity(parsed.headline, title) > 0.75;
  1528. if (headlineMatches && !nameMatches) {
  1529. metadata.title = parsed.headline;
  1530. } else {
  1531. metadata.title = parsed.name;
  1532. }
  1533. } else if (typeof parsed.name === "string") {
  1534. metadata.title = parsed.name.trim();
  1535. } else if (typeof parsed.headline === "string") {
  1536. metadata.title = parsed.headline.trim();
  1537. }
  1538. if (parsed.author) {
  1539. if (typeof parsed.author.name === "string") {
  1540. metadata.byline = parsed.author.name.trim();
  1541. } else if (
  1542. Array.isArray(parsed.author) &&
  1543. parsed.author[0] &&
  1544. typeof parsed.author[0].name === "string"
  1545. ) {
  1546. metadata.byline = parsed.author
  1547. .filter(function (author) {
  1548. return author && typeof author.name === "string";
  1549. })
  1550. .map(function (author) {
  1551. return author.name.trim();
  1552. })
  1553. .join(", ");
  1554. }
  1555. }
  1556. if (typeof parsed.description === "string") {
  1557. metadata.excerpt = parsed.description.trim();
  1558. }
  1559. if (parsed.publisher && typeof parsed.publisher.name === "string") {
  1560. metadata.siteName = parsed.publisher.name.trim();
  1561. }
  1562. if (typeof parsed.datePublished === "string") {
  1563. metadata.datePublished = parsed.datePublished.trim();
  1564. }
  1565. } catch (err) {
  1566. this.log(err.message);
  1567. }
  1568. }
  1569. });
  1570. return metadata ? metadata : {};
  1571. },
  1572. /**
  1573. * Attempts to get excerpt and byline metadata for the article.
  1574. *
  1575. * @param {Object} jsonld — object containing any metadata that
  1576. * could be extracted from JSON-LD object.
  1577. *
  1578. * @return Object with optional "excerpt" and "byline" properties
  1579. */
  1580. _getArticleMetadata(jsonld) {
  1581. var metadata = {};
  1582. var values = {};
  1583. var metaElements = this._doc.getElementsByTagName("meta");
  1584. // property is a space-separated list of values
  1585. var propertyPattern =
  1586. /\s*(article|dc|dcterm|og|twitter)\s*:\s*(author|creator|description|published_time|title|site_name)\s*/gi;
  1587. // name is a single value
  1588. var namePattern =
  1589. /^\s*(?:(dc|dcterm|og|twitter|parsely|weibo:(article|webpage))\s*[-\.:]\s*)?(author|creator|pub-date|description|title|site_name)\s*$/i;
  1590. // Find description tags.
  1591. this._forEachNode(metaElements, function (element) {
  1592. var elementName = element.getAttribute("name");
  1593. var elementProperty = element.getAttribute("property");
  1594. var content = element.getAttribute("content");
  1595. if (!content) {
  1596. return;
  1597. }
  1598. var matches = null;
  1599. var name = null;
  1600. if (elementProperty) {
  1601. matches = elementProperty.match(propertyPattern);
  1602. if (matches) {
  1603. // Convert to lowercase, and remove any whitespace
  1604. // so we can match below.
  1605. name = matches[0].toLowerCase().replace(/\s/g, "");
  1606. // multiple authors
  1607. values[name] = content.trim();
  1608. }
  1609. }
  1610. if (!matches && elementName && namePattern.test(elementName)) {
  1611. name = elementName;
  1612. if (content) {
  1613. // Convert to lowercase, remove any whitespace, and convert dots
  1614. // to colons so we can match below.
  1615. name = name.toLowerCase().replace(/\s/g, "").replace(/\./g, ":");
  1616. values[name] = content.trim();
  1617. }
  1618. }
  1619. });
  1620. // get title
  1621. metadata.title =
  1622. jsonld.title ||
  1623. values["dc:title"] ||
  1624. values["dcterm:title"] ||
  1625. values["og:title"] ||
  1626. values["weibo:article:title"] ||
  1627. values["weibo:webpage:title"] ||
  1628. values.title ||
  1629. values["twitter:title"] ||
  1630. values["parsely-title"];
  1631. if (!metadata.title) {
  1632. metadata.title = this._getArticleTitle();
  1633. }
  1634. const articleAuthor =
  1635. typeof values["article:author"] === "string" &&
  1636. !this._isUrl(values["article:author"])
  1637. ? values["article:author"]
  1638. : undefined;
  1639. // get author
  1640. metadata.byline =
  1641. jsonld.byline ||
  1642. values["dc:creator"] ||
  1643. values["dcterm:creator"] ||
  1644. values.author ||
  1645. values["parsely-author"] ||
  1646. articleAuthor;
  1647. // get description
  1648. metadata.excerpt =
  1649. jsonld.excerpt ||
  1650. values["dc:description"] ||
  1651. values["dcterm:description"] ||
  1652. values["og:description"] ||
  1653. values["weibo:article:description"] ||
  1654. values["weibo:webpage:description"] ||
  1655. values.description ||
  1656. values["twitter:description"];
  1657. // get site name
  1658. metadata.siteName = jsonld.siteName || values["og:site_name"];
  1659. // get article published time
  1660. metadata.publishedTime =
  1661. jsonld.datePublished ||
  1662. values["article:published_time"] ||
  1663. values["parsely-pub-date"] ||
  1664. null;
  1665. // in many sites the meta value is escaped with HTML entities,
  1666. // so here we need to unescape it
  1667. metadata.title = this._unescapeHtmlEntities(metadata.title);
  1668. metadata.byline = this._unescapeHtmlEntities(metadata.byline);
  1669. metadata.excerpt = this._unescapeHtmlEntities(metadata.excerpt);
  1670. metadata.siteName = this._unescapeHtmlEntities(metadata.siteName);
  1671. metadata.publishedTime = this._unescapeHtmlEntities(metadata.publishedTime);
  1672. return metadata;
  1673. },
  1674. /**
  1675. * Check if node is image, or if node contains exactly only one image
  1676. * whether as a direct child or as its descendants.
  1677. *
  1678. * @param Element
  1679. **/
  1680. _isSingleImage(node) {
  1681. while (node) {
  1682. if (node.tagName === "IMG") {
  1683. return true;
  1684. }
  1685. if (node.children.length !== 1 || node.textContent.trim() !== "") {
  1686. return false;
  1687. }
  1688. node = node.children[0];
  1689. }
  1690. return false;
  1691. },
  1692. /**
  1693. * Find all <noscript> that are located after <img> nodes, and which contain only one
  1694. * <img> element. Replace the first image with the image from inside the <noscript> tag,
  1695. * and remove the <noscript> tag. This improves the quality of the images we use on
  1696. * some sites (e.g. Medium).
  1697. *
  1698. * @param Element
  1699. **/
  1700. _unwrapNoscriptImages(doc) {
  1701. // Find img without source or attributes that might contains image, and remove it.
  1702. // This is done to prevent a placeholder img is replaced by img from noscript in next step.
  1703. var imgs = Array.from(doc.getElementsByTagName("img"));
  1704. this._forEachNode(imgs, function (img) {
  1705. for (var i = 0; i < img.attributes.length; i++) {
  1706. var attr = img.attributes[i];
  1707. switch (attr.name) {
  1708. case "src":
  1709. case "srcset":
  1710. case "data-src":
  1711. case "data-srcset":
  1712. return;
  1713. }
  1714. if (/\.(jpg|jpeg|png|webp)/i.test(attr.value)) {
  1715. return;
  1716. }
  1717. }
  1718. img.remove();
  1719. });
  1720. // Next find noscript and try to extract its image
  1721. var noscripts = Array.from(doc.getElementsByTagName("noscript"));
  1722. this._forEachNode(noscripts, function (noscript) {
  1723. // Parse content of noscript and make sure it only contains image
  1724. if (!this._isSingleImage(noscript)) {
  1725. return;
  1726. }
  1727. var tmp = doc.createElement("div");
  1728. // We're running in the document context, and using unmodified
  1729. // document contents, so doing this should be safe.
  1730. // (Also we heavily discourage people from allowing script to
  1731. // run at all in this document...)
  1732. // eslint-disable-next-line no-unsanitized/property
  1733. tmp.innerHTML = noscript.innerHTML;
  1734. // If noscript has previous sibling and it only contains image,
  1735. // replace it with noscript content. However we also keep old
  1736. // attributes that might contains image.
  1737. var prevElement = noscript.previousElementSibling;
  1738. if (prevElement && this._isSingleImage(prevElement)) {
  1739. var prevImg = prevElement;
  1740. if (prevImg.tagName !== "IMG") {
  1741. prevImg = prevElement.getElementsByTagName("img")[0];
  1742. }
  1743. var newImg = tmp.getElementsByTagName("img")[0];
  1744. for (var i = 0; i < prevImg.attributes.length; i++) {
  1745. var attr = prevImg.attributes[i];
  1746. if (attr.value === "") {
  1747. continue;
  1748. }
  1749. if (
  1750. attr.name === "src" ||
  1751. attr.name === "srcset" ||
  1752. /\.(jpg|jpeg|png|webp)/i.test(attr.value)
  1753. ) {
  1754. if (newImg.getAttribute(attr.name) === attr.value) {
  1755. continue;
  1756. }
  1757. var attrName = attr.name;
  1758. if (newImg.hasAttribute(attrName)) {
  1759. attrName = "data-old-" + attrName;
  1760. }
  1761. newImg.setAttribute(attrName, attr.value);
  1762. }
  1763. }
  1764. noscript.parentNode.replaceChild(tmp.firstElementChild, prevElement);
  1765. }
  1766. });
  1767. },
  1768. /**
  1769. * Removes script tags from the document.
  1770. *
  1771. * @param Element
  1772. **/
  1773. _removeScripts(doc) {
  1774. this._removeNodes(this._getAllNodesWithTag(doc, ["script", "noscript"]));
  1775. },
  1776. /**
  1777. * Check if this node has only whitespace and a single element with given tag
  1778. * Returns false if the DIV node contains non-empty text nodes
  1779. * or if it contains no element with given tag or more than 1 element.
  1780. *
  1781. * @param Element
  1782. * @param string tag of child element
  1783. **/
  1784. _hasSingleTagInsideElement(element, tag) {
  1785. // There should be exactly 1 element child with given tag
  1786. if (element.children.length != 1 || element.children[0].tagName !== tag) {
  1787. return false;
  1788. }
  1789. // And there should be no text nodes with real content
  1790. return !this._someNode(element.childNodes, function (node) {
  1791. return (
  1792. node.nodeType === this.TEXT_NODE &&
  1793. this.REGEXPS.hasContent.test(node.textContent)
  1794. );
  1795. });
  1796. },
  1797. _isElementWithoutContent(node) {
  1798. return (
  1799. node.nodeType === this.ELEMENT_NODE &&
  1800. !node.textContent.trim().length &&
  1801. (!node.children.length ||
  1802. node.children.length ==
  1803. node.getElementsByTagName("br").length +
  1804. node.getElementsByTagName("hr").length)
  1805. );
  1806. },
  1807. /**
  1808. * Determine whether element has any children block level elements.
  1809. *
  1810. * @param Element
  1811. */
  1812. _hasChildBlockElement(element) {
  1813. return this._someNode(element.childNodes, function (node) {
  1814. return (
  1815. this.DIV_TO_P_ELEMS.has(node.tagName) ||
  1816. this._hasChildBlockElement(node)
  1817. );
  1818. });
  1819. },
  1820. /***
  1821. * Determine if a node qualifies as phrasing content.
  1822. * https://developer.mozilla.org/en-US/docs/Web/Guide/HTML/Content_categories#Phrasing_content
  1823. **/
  1824. _isPhrasingContent(node) {
  1825. return (
  1826. node.nodeType === this.TEXT_NODE ||
  1827. this.PHRASING_ELEMS.includes(node.tagName) ||
  1828. ((node.tagName === "A" ||
  1829. node.tagName === "DEL" ||
  1830. node.tagName === "INS") &&
  1831. this._everyNode(node.childNodes, this._isPhrasingContent))
  1832. );
  1833. },
  1834. _isWhitespace(node) {
  1835. return (
  1836. (node.nodeType === this.TEXT_NODE &&
  1837. node.textContent.trim().length === 0) ||
  1838. (node.nodeType === this.ELEMENT_NODE && node.tagName === "BR")
  1839. );
  1840. },
  1841. /**
  1842. * Get the inner text of a node - cross browser compatibly.
  1843. * This also strips out any excess whitespace to be found.
  1844. *
  1845. * @param Element
  1846. * @param Boolean normalizeSpaces (default: true)
  1847. * @return string
  1848. **/
  1849. _getInnerText(e, normalizeSpaces) {
  1850. normalizeSpaces =
  1851. typeof normalizeSpaces === "undefined" ? true : normalizeSpaces;
  1852. var textContent = e.textContent.trim();
  1853. if (normalizeSpaces) {
  1854. return textContent.replace(this.REGEXPS.normalize, " ");
  1855. }
  1856. return textContent;
  1857. },
  1858. /**
  1859. * Get the number of times a string s appears in the node e.
  1860. *
  1861. * @param Element
  1862. * @param string - what to split on. Default is ","
  1863. * @return number (integer)
  1864. **/
  1865. _getCharCount(e, s) {
  1866. s = s || ",";
  1867. return this._getInnerText(e).split(s).length - 1;
  1868. },
  1869. /**
  1870. * Remove the style attribute on every e and under.
  1871. * TODO: Test if getElementsByTagName(*) is faster.
  1872. *
  1873. * @param Element
  1874. * @return void
  1875. **/
  1876. _cleanStyles(e) {
  1877. if (!e || e.tagName.toLowerCase() === "svg") {
  1878. return;
  1879. }
  1880. // Remove `style` and deprecated presentational attributes
  1881. for (var i = 0; i < this.PRESENTATIONAL_ATTRIBUTES.length; i++) {
  1882. e.removeAttribute(this.PRESENTATIONAL_ATTRIBUTES[i]);
  1883. }
  1884. if (this.DEPRECATED_SIZE_ATTRIBUTE_ELEMS.includes(e.tagName)) {
  1885. e.removeAttribute("width");
  1886. e.removeAttribute("height");
  1887. }
  1888. var cur = e.firstElementChild;
  1889. while (cur !== null) {
  1890. this._cleanStyles(cur);
  1891. cur = cur.nextElementSibling;
  1892. }
  1893. },
  1894. /**
  1895. * Get the density of links as a percentage of the content
  1896. * This is the amount of text that is inside a link divided by the total text in the node.
  1897. *
  1898. * @param Element
  1899. * @return number (float)
  1900. **/
  1901. _getLinkDensity(element) {
  1902. var textLength = this._getInnerText(element).length;
  1903. if (textLength === 0) {
  1904. return 0;
  1905. }
  1906. var linkLength = 0;
  1907. // XXX implement _reduceNodeList?
  1908. this._forEachNode(element.getElementsByTagName("a"), function (linkNode) {
  1909. var href = linkNode.getAttribute("href");
  1910. var coefficient = href && this.REGEXPS.hashUrl.test(href) ? 0.3 : 1;
  1911. linkLength += this._getInnerText(linkNode).length * coefficient;
  1912. });
  1913. return linkLength / textLength;
  1914. },
  1915. /**
  1916. * Get an elements class/id weight. Uses regular expressions to tell if this
  1917. * element looks good or bad.
  1918. *
  1919. * @param Element
  1920. * @return number (Integer)
  1921. **/
  1922. _getClassWeight(e) {
  1923. if (!this._flagIsActive(this.FLAG_WEIGHT_CLASSES)) {
  1924. return 0;
  1925. }
  1926. var weight = 0;
  1927. // Look for a special classname
  1928. if (typeof e.className === "string" && e.className !== "") {
  1929. if (this.REGEXPS.negative.test(e.className)) {
  1930. weight -= 25;
  1931. }
  1932. if (this.REGEXPS.positive.test(e.className)) {
  1933. weight += 25;
  1934. }
  1935. }
  1936. // Look for a special ID
  1937. if (typeof e.id === "string" && e.id !== "") {
  1938. if (this.REGEXPS.negative.test(e.id)) {
  1939. weight -= 25;
  1940. }
  1941. if (this.REGEXPS.positive.test(e.id)) {
  1942. weight += 25;
  1943. }
  1944. }
  1945. return weight;
  1946. },
  1947. /**
  1948. * Clean a node of all elements of type "tag".
  1949. * (Unless it's a youtube/vimeo video. People love movies.)
  1950. *
  1951. * @param Element
  1952. * @param string tag to clean
  1953. * @return void
  1954. **/
  1955. _clean(e, tag) {
  1956. var isEmbed = ["object", "embed", "iframe"].includes(tag);
  1957. this._removeNodes(this._getAllNodesWithTag(e, [tag]), function (element) {
  1958. // Allow youtube and vimeo videos through as people usually want to see those.
  1959. if (isEmbed) {
  1960. // First, check the elements attributes to see if any of them contain youtube or vimeo
  1961. for (var i = 0; i < element.attributes.length; i++) {
  1962. if (this._allowedVideoRegex.test(element.attributes[i].value)) {
  1963. return false;
  1964. }
  1965. }
  1966. // For embed with <object> tag, check inner HTML as well.
  1967. if (
  1968. element.tagName === "object" &&
  1969. this._allowedVideoRegex.test(element.innerHTML)
  1970. ) {
  1971. return false;
  1972. }
  1973. }
  1974. return true;
  1975. });
  1976. },
  1977. /**
  1978. * Check if a given node has one of its ancestor tag name matching the
  1979. * provided one.
  1980. * @param HTMLElement node
  1981. * @param String tagName
  1982. * @param Number maxDepth
  1983. * @param Function filterFn a filter to invoke to determine whether this node 'counts'
  1984. * @return Boolean
  1985. */
  1986. _hasAncestorTag(node, tagName, maxDepth, filterFn) {
  1987. maxDepth = maxDepth || 3;
  1988. tagName = tagName.toUpperCase();
  1989. var depth = 0;
  1990. while (node.parentNode) {
  1991. if (maxDepth > 0 && depth > maxDepth) {
  1992. return false;
  1993. }
  1994. if (
  1995. node.parentNode.tagName === tagName &&
  1996. (!filterFn || filterFn(node.parentNode))
  1997. ) {
  1998. return true;
  1999. }
  2000. node = node.parentNode;
  2001. depth++;
  2002. }
  2003. return false;
  2004. },
  2005. /**
  2006. * Return an object indicating how many rows and columns this table has.
  2007. */
  2008. _getRowAndColumnCount(table) {
  2009. var rows = 0;
  2010. var columns = 0;
  2011. var trs = table.getElementsByTagName("tr");
  2012. for (var i = 0; i < trs.length; i++) {
  2013. var rowspan = trs[i].getAttribute("rowspan") || 0;
  2014. if (rowspan) {
  2015. rowspan = parseInt(rowspan, 10);
  2016. }
  2017. rows += rowspan || 1;
  2018. // Now look for column-related info
  2019. var columnsInThisRow = 0;
  2020. var cells = trs[i].getElementsByTagName("td");
  2021. for (var j = 0; j < cells.length; j++) {
  2022. var colspan = cells[j].getAttribute("colspan") || 0;
  2023. if (colspan) {
  2024. colspan = parseInt(colspan, 10);
  2025. }
  2026. columnsInThisRow += colspan || 1;
  2027. }
  2028. columns = Math.max(columns, columnsInThisRow);
  2029. }
  2030. return { rows, columns };
  2031. },
  2032. /**
  2033. * Look for 'data' (as opposed to 'layout') tables, for which we use
  2034. * similar checks as
  2035. * https://searchfox.org/mozilla-central/rev/f82d5c549f046cb64ce5602bfd894b7ae807c8f8/accessible/generic/TableAccessible.cpp#19
  2036. */
  2037. _markDataTables(root) {
  2038. var tables = root.getElementsByTagName("table");
  2039. for (var i = 0; i < tables.length; i++) {
  2040. var table = tables[i];
  2041. var role = table.getAttribute("role");
  2042. if (role == "presentation") {
  2043. table._readabilityDataTable = false;
  2044. continue;
  2045. }
  2046. var datatable = table.getAttribute("datatable");
  2047. if (datatable == "0") {
  2048. table._readabilityDataTable = false;
  2049. continue;
  2050. }
  2051. var summary = table.getAttribute("summary");
  2052. if (summary) {
  2053. table._readabilityDataTable = true;
  2054. continue;
  2055. }
  2056. var caption = table.getElementsByTagName("caption")[0];
  2057. if (caption && caption.childNodes.length) {
  2058. table._readabilityDataTable = true;
  2059. continue;
  2060. }
  2061. // If the table has a descendant with any of these tags, consider a data table:
  2062. var dataTableDescendants = ["col", "colgroup", "tfoot", "thead", "th"];
  2063. var descendantExists = function (tag) {
  2064. return !!table.getElementsByTagName(tag)[0];
  2065. };
  2066. if (dataTableDescendants.some(descendantExists)) {
  2067. this.log("Data table because found data-y descendant");
  2068. table._readabilityDataTable = true;
  2069. continue;
  2070. }
  2071. // Nested tables indicate a layout table:
  2072. if (table.getElementsByTagName("table")[0]) {
  2073. table._readabilityDataTable = false;
  2074. continue;
  2075. }
  2076. var sizeInfo = this._getRowAndColumnCount(table);
  2077. if (sizeInfo.columns == 1 || sizeInfo.rows == 1) {
  2078. // single colum/row tables are commonly used for page layout purposes.
  2079. table._readabilityDataTable = false;
  2080. continue;
  2081. }
  2082. if (sizeInfo.rows >= 10 || sizeInfo.columns > 4) {
  2083. table._readabilityDataTable = true;
  2084. continue;
  2085. }
  2086. // Now just go by size entirely:
  2087. table._readabilityDataTable = sizeInfo.rows * sizeInfo.columns > 10;
  2088. }
  2089. },
  2090. /* convert images and figures that have properties like data-src into images that can be loaded without JS */
  2091. _fixLazyImages(root) {
  2092. this._forEachNode(
  2093. this._getAllNodesWithTag(root, ["img", "picture", "figure"]),
  2094. function (elem) {
  2095. // In some sites (e.g. Kotaku), they put 1px square image as base64 data uri in the src attribute.
  2096. // So, here we check if the data uri is too short, just might as well remove it.
  2097. if (elem.src && this.REGEXPS.b64DataUrl.test(elem.src)) {
  2098. // Make sure it's not SVG, because SVG can have a meaningful image in under 133 bytes.
  2099. var parts = this.REGEXPS.b64DataUrl.exec(elem.src);
  2100. if (parts[1] === "image/svg+xml") {
  2101. return;
  2102. }
  2103. // Make sure this element has other attributes which contains image.
  2104. // If it doesn't, then this src is important and shouldn't be removed.
  2105. var srcCouldBeRemoved = false;
  2106. for (var i = 0; i < elem.attributes.length; i++) {
  2107. var attr = elem.attributes[i];
  2108. if (attr.name === "src") {
  2109. continue;
  2110. }
  2111. if (/\.(jpg|jpeg|png|webp)/i.test(attr.value)) {
  2112. srcCouldBeRemoved = true;
  2113. break;
  2114. }
  2115. }
  2116. // Here we assume if image is less than 100 bytes (or 133 after encoded to base64)
  2117. // it will be too small, therefore it might be placeholder image.
  2118. if (srcCouldBeRemoved) {
  2119. var b64starts = parts[0].length;
  2120. var b64length = elem.src.length - b64starts;
  2121. if (b64length < 133) {
  2122. elem.removeAttribute("src");
  2123. }
  2124. }
  2125. }
  2126. // also check for "null" to work around https://github.com/jsdom/jsdom/issues/2580
  2127. if (
  2128. (elem.src || (elem.srcset && elem.srcset != "null")) &&
  2129. !elem.className.toLowerCase().includes("lazy")
  2130. ) {
  2131. return;
  2132. }
  2133. for (var j = 0; j < elem.attributes.length; j++) {
  2134. attr = elem.attributes[j];
  2135. if (
  2136. attr.name === "src" ||
  2137. attr.name === "srcset" ||
  2138. attr.name === "alt"
  2139. ) {
  2140. continue;
  2141. }
  2142. var copyTo = null;
  2143. if (/\.(jpg|jpeg|png|webp)\s+\d/.test(attr.value)) {
  2144. copyTo = "srcset";
  2145. } else if (/^\s*\S+\.(jpg|jpeg|png|webp)\S*\s*$/.test(attr.value)) {
  2146. copyTo = "src";
  2147. }
  2148. if (copyTo) {
  2149. //if this is an img or picture, set the attribute directly
  2150. if (elem.tagName === "IMG" || elem.tagName === "PICTURE") {
  2151. elem.setAttribute(copyTo, attr.value);
  2152. } else if (
  2153. elem.tagName === "FIGURE" &&
  2154. !this._getAllNodesWithTag(elem, ["img", "picture"]).length
  2155. ) {
  2156. //if the item is a <figure> that does not contain an image or picture, create one and place it inside the figure
  2157. //see the nytimes-3 testcase for an example
  2158. var img = this._doc.createElement("img");
  2159. img.setAttribute(copyTo, attr.value);
  2160. elem.appendChild(img);
  2161. }
  2162. }
  2163. }
  2164. }
  2165. );
  2166. },
  2167. _getTextDensity(e, tags) {
  2168. var textLength = this._getInnerText(e, true).length;
  2169. if (textLength === 0) {
  2170. return 0;
  2171. }
  2172. var childrenLength = 0;
  2173. var children = this._getAllNodesWithTag(e, tags);
  2174. this._forEachNode(
  2175. children,
  2176. child => (childrenLength += this._getInnerText(child, true).length)
  2177. );
  2178. return childrenLength / textLength;
  2179. },
  2180. /**
  2181. * Clean an element of all tags of type "tag" if they look fishy.
  2182. * "Fishy" is an algorithm based on content length, classnames, link density, number of images & embeds, etc.
  2183. *
  2184. * @return void
  2185. **/
  2186. _cleanConditionally(e, tag) {
  2187. if (!this._flagIsActive(this.FLAG_CLEAN_CONDITIONALLY)) {
  2188. return;
  2189. }
  2190. // Gather counts for other typical elements embedded within.
  2191. // Traverse backwards so we can remove nodes at the same time
  2192. // without effecting the traversal.
  2193. //
  2194. // TODO: Consider taking into account original contentScore here.
  2195. this._removeNodes(this._getAllNodesWithTag(e, [tag]), function (node) {
  2196. // First check if this node IS data table, in which case don't remove it.
  2197. var isDataTable = function (t) {
  2198. return t._readabilityDataTable;
  2199. };
  2200. var isList = tag === "ul" || tag === "ol";
  2201. if (!isList) {
  2202. var listLength = 0;
  2203. var listNodes = this._getAllNodesWithTag(node, ["ul", "ol"]);
  2204. this._forEachNode(
  2205. listNodes,
  2206. list => (listLength += this._getInnerText(list).length)
  2207. );
  2208. isList = listLength / this._getInnerText(node).length > 0.9;
  2209. }
  2210. if (tag === "table" && isDataTable(node)) {
  2211. return false;
  2212. }
  2213. // Next check if we're inside a data table, in which case don't remove it as well.
  2214. if (this._hasAncestorTag(node, "table", -1, isDataTable)) {
  2215. return false;
  2216. }
  2217. if (this._hasAncestorTag(node, "code")) {
  2218. return false;
  2219. }
  2220. // keep element if it has a data tables
  2221. if (
  2222. [...node.getElementsByTagName("table")].some(
  2223. tbl => tbl._readabilityDataTable
  2224. )
  2225. ) {
  2226. return false;
  2227. }
  2228. var weight = this._getClassWeight(node);
  2229. this.log("Cleaning Conditionally", node);
  2230. var contentScore = 0;
  2231. if (weight + contentScore < 0) {
  2232. return true;
  2233. }
  2234. if (this._getCharCount(node, ",") < 10) {
  2235. // If there are not very many commas, and the number of
  2236. // non-paragraph elements is more than paragraphs or other
  2237. // ominous signs, remove the element.
  2238. var p = node.getElementsByTagName("p").length;
  2239. var img = node.getElementsByTagName("img").length;
  2240. var li = node.getElementsByTagName("li").length - 100;
  2241. var input = node.getElementsByTagName("input").length;
  2242. var headingDensity = this._getTextDensity(node, [
  2243. "h1",
  2244. "h2",
  2245. "h3",
  2246. "h4",
  2247. "h5",
  2248. "h6",
  2249. ]);
  2250. var embedCount = 0;
  2251. var embeds = this._getAllNodesWithTag(node, [
  2252. "object",
  2253. "embed",
  2254. "iframe",
  2255. ]);
  2256. for (var i = 0; i < embeds.length; i++) {
  2257. // If this embed has attribute that matches video regex, don't delete it.
  2258. for (var j = 0; j < embeds[i].attributes.length; j++) {
  2259. if (this._allowedVideoRegex.test(embeds[i].attributes[j].value)) {
  2260. return false;
  2261. }
  2262. }
  2263. // For embed with <object> tag, check inner HTML as well.
  2264. if (
  2265. embeds[i].tagName === "object" &&
  2266. this._allowedVideoRegex.test(embeds[i].innerHTML)
  2267. ) {
  2268. return false;
  2269. }
  2270. embedCount++;
  2271. }
  2272. var innerText = this._getInnerText(node);
  2273. // toss any node whose inner text contains nothing but suspicious words
  2274. if (
  2275. this.REGEXPS.adWords.test(innerText) ||
  2276. this.REGEXPS.loadingWords.test(innerText)
  2277. ) {
  2278. return true;
  2279. }
  2280. var contentLength = innerText.length;
  2281. var linkDensity = this._getLinkDensity(node);
  2282. var textishTags = ["SPAN", "LI", "TD"].concat(
  2283. Array.from(this.DIV_TO_P_ELEMS)
  2284. );
  2285. var textDensity = this._getTextDensity(node, textishTags);
  2286. var isFigureChild = this._hasAncestorTag(node, "figure");
  2287. // apply shadiness checks, then check for exceptions
  2288. const shouldRemoveNode = () => {
  2289. const errs = [];
  2290. if (!isFigureChild && img > 1 && p / img < 0.5) {
  2291. errs.push(`Bad p to img ratio (img=${img}, p=${p})`);
  2292. }
  2293. if (!isList && li > p) {
  2294. errs.push(`Too many li's outside of a list. (li=${li} > p=${p})`);
  2295. }
  2296. if (input > Math.floor(p / 3)) {
  2297. errs.push(`Too many inputs per p. (input=${input}, p=${p})`);
  2298. }
  2299. if (
  2300. !isList &&
  2301. !isFigureChild &&
  2302. headingDensity < 0.9 &&
  2303. contentLength < 25 &&
  2304. (img === 0 || img > 2) &&
  2305. linkDensity > 0
  2306. ) {
  2307. errs.push(
  2308. `Suspiciously short. (headingDensity=${headingDensity}, img=${img}, linkDensity=${linkDensity})`
  2309. );
  2310. }
  2311. if (
  2312. !isList &&
  2313. weight < 25 &&
  2314. linkDensity > 0.2 + this._linkDensityModifier
  2315. ) {
  2316. errs.push(
  2317. `Low weight and a little linky. (linkDensity=${linkDensity})`
  2318. );
  2319. }
  2320. if (weight >= 25 && linkDensity > 0.5 + this._linkDensityModifier) {
  2321. errs.push(
  2322. `High weight and mostly links. (linkDensity=${linkDensity})`
  2323. );
  2324. }
  2325. if ((embedCount === 1 && contentLength < 75) || embedCount > 1) {
  2326. errs.push(
  2327. `Suspicious embed. (embedCount=${embedCount}, contentLength=${contentLength})`
  2328. );
  2329. }
  2330. if (img === 0 && textDensity === 0) {
  2331. errs.push(
  2332. `No useful content. (img=${img}, textDensity=${textDensity})`
  2333. );
  2334. }
  2335. if (errs.length) {
  2336. this.log("Checks failed", errs);
  2337. return true;
  2338. }
  2339. return false;
  2340. };
  2341. var haveToRemove = shouldRemoveNode();
  2342. // Allow simple lists of images to remain in pages
  2343. if (isList && haveToRemove) {
  2344. for (var x = 0; x < node.children.length; x++) {
  2345. let child = node.children[x];
  2346. // Don't filter in lists with li's that contain more than one child
  2347. if (child.children.length > 1) {
  2348. return haveToRemove;
  2349. }
  2350. }
  2351. let li_count = node.getElementsByTagName("li").length;
  2352. // Only allow the list to remain if every li contains an image
  2353. if (img == li_count) {
  2354. return false;
  2355. }
  2356. }
  2357. return haveToRemove;
  2358. }
  2359. return false;
  2360. });
  2361. },
  2362. /**
  2363. * Clean out elements that match the specified conditions
  2364. *
  2365. * @param Element
  2366. * @param Function determines whether a node should be removed
  2367. * @return void
  2368. **/
  2369. _cleanMatchedNodes(e, filter) {
  2370. var endOfSearchMarkerNode = this._getNextNode(e, true);
  2371. var next = this._getNextNode(e);
  2372. while (next && next != endOfSearchMarkerNode) {
  2373. if (filter.call(this, next, next.className + " " + next.id)) {
  2374. next = this._removeAndGetNext(next);
  2375. } else {
  2376. next = this._getNextNode(next);
  2377. }
  2378. }
  2379. },
  2380. /**
  2381. * Clean out spurious headers from an Element.
  2382. *
  2383. * @param Element
  2384. * @return void
  2385. **/
  2386. _cleanHeaders(e) {
  2387. let headingNodes = this._getAllNodesWithTag(e, ["h1", "h2"]);
  2388. this._removeNodes(headingNodes, function (node) {
  2389. let shouldRemove = this._getClassWeight(node) < 0;
  2390. if (shouldRemove) {
  2391. this.log("Removing header with low class weight:", node);
  2392. }
  2393. return shouldRemove;
  2394. });
  2395. },
  2396. /**
  2397. * Check if this node is an H1 or H2 element whose content is mostly
  2398. * the same as the article title.
  2399. *
  2400. * @param Element the node to check.
  2401. * @return boolean indicating whether this is a title-like header.
  2402. */
  2403. _headerDuplicatesTitle(node) {
  2404. if (node.tagName != "H1" && node.tagName != "H2") {
  2405. return false;
  2406. }
  2407. var heading = this._getInnerText(node, false);
  2408. this.log("Evaluating similarity of header:", heading, this._articleTitle);
  2409. return this._textSimilarity(this._articleTitle, heading) > 0.75;
  2410. },
  2411. _flagIsActive(flag) {
  2412. return (this._flags & flag) > 0;
  2413. },
  2414. _removeFlag(flag) {
  2415. this._flags = this._flags & ~flag;
  2416. },
  2417. _isProbablyVisible(node) {
  2418. // Have to null-check node.style and node.className.includes to deal with SVG and MathML nodes.
  2419. return (
  2420. (!node.style || node.style.display != "none") &&
  2421. (!node.style || node.style.visibility != "hidden") &&
  2422. !node.hasAttribute("hidden") &&
  2423. //check for "fallback-image" so that wikimedia math images are displayed
  2424. (!node.hasAttribute("aria-hidden") ||
  2425. node.getAttribute("aria-hidden") != "true" ||
  2426. (node.className &&
  2427. node.className.includes &&
  2428. node.className.includes("fallback-image")))
  2429. );
  2430. },
  2431. /**
  2432. * Runs readability.
  2433. *
  2434. * Workflow:
  2435. * 1. Prep the document by removing script tags, css, etc.
  2436. * 2. Build readability's DOM tree.
  2437. * 3. Grab the article content from the current dom tree.
  2438. * 4. Replace the current DOM tree with the new one.
  2439. * 5. Read peacefully.
  2440. *
  2441. * @return void
  2442. **/
  2443. parse() {
  2444. // Avoid parsing too large documents, as per configuration option
  2445. if (this._maxElemsToParse > 0) {
  2446. var numTags = this._doc.getElementsByTagName("*").length;
  2447. if (numTags > this._maxElemsToParse) {
  2448. throw new Error(
  2449. "Aborting parsing document; " + numTags + " elements found"
  2450. );
  2451. }
  2452. }
  2453. // Unwrap image from noscript
  2454. this._unwrapNoscriptImages(this._doc);
  2455. // Extract JSON-LD metadata before removing scripts
  2456. var jsonLd = this._disableJSONLD ? {} : this._getJSONLD(this._doc);
  2457. // Remove script tags from the document.
  2458. this._removeScripts(this._doc);
  2459. this._prepDocument();
  2460. var metadata = this._getArticleMetadata(jsonLd);
  2461. this._metadata = metadata;
  2462. this._articleTitle = metadata.title;
  2463. var articleContent = this._grabArticle();
  2464. if (!articleContent) {
  2465. return null;
  2466. }
  2467. this.log("Grabbed: " + articleContent.innerHTML);
  2468. this._postProcessContent(articleContent);
  2469. // If we haven't found an excerpt in the article's metadata, use the article's
  2470. // first paragraph as the excerpt. This is used for displaying a preview of
  2471. // the article's content.
  2472. if (!metadata.excerpt) {
  2473. var paragraphs = articleContent.getElementsByTagName("p");
  2474. if (paragraphs.length) {
  2475. metadata.excerpt = paragraphs[0].textContent.trim();
  2476. }
  2477. }
  2478. var textContent = articleContent.textContent;
  2479. return {
  2480. title: this._articleTitle,
  2481. byline: metadata.byline || this._articleByline,
  2482. dir: this._articleDir,
  2483. lang: this._articleLang,
  2484. content: this._serializer(articleContent),
  2485. textContent,
  2486. length: textContent.length,
  2487. excerpt: metadata.excerpt,
  2488. siteName: metadata.siteName || this._articleSiteName,
  2489. publishedTime: metadata.publishedTime,
  2490. };
  2491. },
  2492. };
  2493. if (typeof module === "object") {
  2494. /* eslint-disable-next-line no-redeclare */
  2495. /* global module */
  2496. window.Readability = Readability;
  2497. }
  2498. })();