scrape.mjs 4.8 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155
  1. import * as cheerio from "cheerio";
  2. import { mkdir, writeFile } from "node:fs/promises";
  3. const BASE = "https://scientia.ru";
  4. const UA =
  5. "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36";
  6. const PAGES = [
  7. "/",
  8. "/about",
  9. "/prorock",
  10. "/magicore",
  11. "/prorockfeatures",
  12. "/digger",
  13. "/nur",
  14. "/emergency_care",
  15. "/joint_explorer",
  16. "/geodesy",
  17. "/prorock_updates",
  18. "/personal_data",
  19. "/education",
  20. "/scientiajoints",
  21. "/subscribe",
  22. "/diggerslope_updates",
  23. ];
  24. const clean = (s) => s.replace(/\s+/g, " ").trim();
  25. function extract(html, path) {
  26. const $ = cheerio.load(html);
  27. const meta = {
  28. title: clean($("title").first().text()),
  29. description: $('meta[name="description"]').attr("content")?.trim() || "",
  30. ogImage: $('meta[property="og:image"]').attr("content") || "",
  31. h1: clean($("h1").first().text()),
  32. };
  33. const blocks = [];
  34. const seenImages = new Set();
  35. $(".t-rec").each((_, rec) => {
  36. const $rec = $(rec);
  37. if ($rec.parents(".t-rec").length) return; // top-level records only
  38. const block = { type: "", headings: [], texts: [], images: [], links: [], form: null };
  39. $rec.find("h1,h2,h3,h4,h5,h6").each((__, h) => {
  40. const t = clean($(h).text());
  41. if (t) block.headings.push({ level: Number(h.tagName[1]), text: t });
  42. });
  43. $rec
  44. .find(
  45. ".t-descr, .t-text, .t-uptitle, .t-name, .t-heading, .t-subtitle, .t-section__title, .t-section__descr, .tn-atom",
  46. )
  47. .each((__, el) => {
  48. const $el = $(el);
  49. if ($el.parents("form").length) return;
  50. if ($el.find(".tn-atom").length) return; // nested duplicates in zero blocks
  51. const t = clean($el.text());
  52. if (t && !block.texts.includes(t)) block.texts.push(t);
  53. });
  54. $rec.find("li").each((__, li) => {
  55. const t = clean($(li).text());
  56. if (t && !block.texts.some((x) => x.includes(t))) block.texts.push("• " + t);
  57. });
  58. $rec.find("img").each((__, img) => {
  59. const src = ($(img).attr("data-original") || $(img).attr("src") || "").split("?")[0];
  60. if (src && /tildacdn|scientia\.ru/.test(src) && !seenImages.has(src)) {
  61. seenImages.add(src);
  62. block.images.push(src);
  63. }
  64. });
  65. $rec.find('[style*="background-image"]').each((__, el) => {
  66. const m = ($(el).attr("style") || "").match(/url\(['"]?([^'")]+)['"]?\)/);
  67. if (m && /tildacdn|scientia\.ru/.test(m[1]) && !seenImages.has(m[1])) {
  68. seenImages.add(m[1]);
  69. block.images.push(m[1].split("?")[0]);
  70. }
  71. });
  72. $rec.find("a[href]").each((__, a) => {
  73. const href = $(a).attr("href") || "";
  74. if (href && !href.startsWith("#") && !href.startsWith("javascript") && !block.links.includes(href)) {
  75. block.links.push(href);
  76. }
  77. });
  78. const form = $rec.find("form").first();
  79. if (form.length) {
  80. block.form = {
  81. action: form.attr("action") || "",
  82. fields: form
  83. .find("input,textarea,select")
  84. .map((__, f) => ({
  85. tag: f.tagName,
  86. name: $(f).attr("name") || "",
  87. type: $(f).attr("type") || "",
  88. placeholder: $(f).attr("placeholder") || "",
  89. }))
  90. .get(),
  91. };
  92. }
  93. if (block.headings.length || block.texts.length || block.images.length || block.form) {
  94. blocks.push(block);
  95. }
  96. });
  97. // Global navigation
  98. const nav = [];
  99. $(".t228__list_item a, .tmenu a, nav a, .t396__elem a[href^='/']").each((__, a) => {
  100. const text = clean($(a).text());
  101. const href = $(a).attr("href") || "";
  102. if (text && href && !nav.some((n) => n.href === href)) nav.push({ text, href });
  103. });
  104. return { path, meta, nav, blocks };
  105. }
  106. async function main() {
  107. await mkdir(".scratch/raw", { recursive: true });
  108. await mkdir(".scratch/content", { recursive: true });
  109. const allImages = new Map();
  110. for (const path of PAGES) {
  111. const slug = path === "/" ? "home" : path.replace(/^\//, "");
  112. process.stdout.write(`fetch ${path} ... `);
  113. try {
  114. const res = await fetch(BASE + path, { headers: { "User-Agent": UA } });
  115. if (!res.ok) {
  116. console.log(`HTTP ${res.status}`);
  117. continue;
  118. }
  119. const html = await res.text();
  120. await writeFile(`.scratch/raw/${slug}.html`, html);
  121. const data = extract(html, path);
  122. await writeFile(`.scratch/content/${slug}.json`, JSON.stringify(data, null, 2));
  123. for (const b of data.blocks) for (const img of b.images) allImages.set(img, slug);
  124. console.log(`${data.blocks.length} blocks, meta: "${data.meta.title}"`);
  125. } catch (e) {
  126. console.log(`ERROR ${e.message}`);
  127. }
  128. }
  129. await writeFile(
  130. ".scratch/images.json",
  131. JSON.stringify([...allImages.entries()].map(([url, page]) => ({ url, page })), null, 2),
  132. );
  133. console.log(`\n${allImages.size} unique images found`);
  134. }
  135. main();