ASSISTANT
Certainly! Here is the entire script with the necessary modifications to use the new regex pattern for extracting the title, matching alphanumeric and structured text:
```javascript
// ==UserScript==
// @name JYNGMN Image Downloader
// @namespace http://tampermonkey.net/
// @version 1.26
// @description Download images from specified WebURLs as a ZIP file, log image URLs, and add DOWNLOAD buttons correctly, without leaving the page. Compatible with Chrome and Edge browsers, supports multiple pages.
// @author
// @match https://www.xinggan5.top/*
// @connect *
// @grant GM_xmlhttpRequest
// @grant GM_download
// ==/UserScript==
(function () {
'use strict';
// **Initialization Log**
console.log('JYNGMN Image Downloader script initializing...');
/**
* Utility function to convert relative URLs to absolute URLs.
* @param {string} url - The URL to convert.
* @param {string} base - The base URL for conversion.
* @returns {string} - The absolute URL.
*/
function makeAbsoluteURL(url, base) {
try {
return new URL(url, base || window.location.origin).href;
} catch (error) {
console.error(`Invalid URL: ${url} - Error: ${error.message}`);
return url;
}
}
/**
* Function to inject external scripts (e.g., JSZip).
* @param {string} url - The URL of the script to inject.
* @returns {Promise} - Resolves when the script is loaded.
*/
async function injectScript(url) {
return new Promise((resolve, reject) => {
if (document.querySelector(`script[src="${url}"]`)) {
console.log(`Script already injected: ${url}`);
resolve();
return;
}
const script = document.createElement('script');
script.src = url;
script.onload = () => {
console.log(`Injected script loaded: ${url}`);
resolve();
};
script.onerror = () => {
console.error(`Failed to load script: ${url}`);
reject(new Error(`Failed to load script: ${url}`));
};
document.head.appendChild(script);
});
}
/**
* Function to fetch and parse HTML from a given URL using GM_xmlhttpRequest.
* @param {string} url - The URL to fetch.
* @returns {Promise<Document>} - Resolves with the parsed HTML document.
*/
function fetchHTML(url) {
return new Promise((resolve, reject) => {
GM_xmlhttpRequest({
method: 'GET',
url: url,
onload: function (response) {
if (response.status >= 200 && response.status < 400) {
const parser = new DOMParser();
const doc = parser.parseFromString(response.responseText, 'text/html');
resolve(doc);
} else {
reject(new Error(`Failed to fetch page: ${url} - Status: ${response.status}`));
}
},
onerror: function () {
reject(new Error(`Network error while fetching page: ${url}`));
}
});
});
}
/**
* Function to fetch an image blob from a URL using the Fetch API.
* @param {string} url - The image URL.
* @returns {Promise<Blob|null>} - Resolves with the image blob or null if failed.
*/
async function fetchImage(url) {
try {
console.log(`Fetching image: ${url}`);
const response = await fetch(url, { mode: 'cors', credentials: 'omit' });
if (!response.ok) throw new Error(`Failed to fetch image: ${url} - Status: ${response.status}`);
return await response.blob();
} catch (error) {
console.error(`Error fetching image: ${error.message}`);
return null;
}
}
/**
* Utility function to extract the file extension from a URL.
* @param {string} url - The URL to extract the extension from.
* @returns {string} - The file extension (default: 'jpg').
*/
function getFileExtension(url) {
const match = url.match(/\.([a-zA-Z0-9]+)(?:\?.*)?$/);
return match ? match[1].toLowerCase() : 'jpg';
}
/**
* Function to trigger the download of a ZIP blob using an anchor element.
* @param {string} blobUrl - The Object URL of the ZIP blob.
* @param {string} fileName - The desired name for the downloaded ZIP file.
*/
function downloadUsingAnchor(blobUrl, fileName) {
const a = document.createElement('a');
a.href = blobUrl;
a.download = fileName;
a.style.display = 'none';
document.body.appendChild(a);
a.click();
document.body.removeChild(a);
console.log(`Download initiated for: ${fileName}`);
}
/**
* Function to create and download the ZIP file using JSZip.
* @param {JSZip} zip - The JSZip instance containing files.
* @param {string} [zipName='images.zip'] - The desired name for the ZIP file.
*/
async function createAndDownloadZip(zip, zipName = 'images.zip') {
try {
console.log('Generating ZIP Blob...');
const zipBlob = await zip.generateAsync({ type: 'blob', compression: 'STORE' });
console.log('ZIP Blob generated. Size:', zipBlob.size, 'bytes');
const blobUrl = URL.createObjectURL(zipBlob);
downloadUsingAnchor(blobUrl, zipName);
} catch (error) {
console.error(`Error creating or downloading ZIP: ${error.message}`);
alert(`An error occurred while creating the ZIP file: ${error.message}`);
}
}
/**
* Function to add a DOWNLOAD button to a target element with an associated linkUrl.
* @param {HTMLElement} targetElement - The element to which the button will be appended or after which it will be inserted.
* @param {string} linkUrl - The URL to be processed when the button is clicked.
* @param {boolean} useCurrentPage - If true, processes the current page without fetching linkUrl.
* @param {string} insertMode - Where to insert the button: 'after' or 'append'.
*/
function addDownloadButton(targetElement, linkUrl, useCurrentPage = false, insertMode = 'after') {
// Check if the button already exists based on the insertMode
if (insertMode === 'append') {
if (targetElement.querySelector('button.download-button')) {
return; // Button already exists inside the target
}
} else if (insertMode === 'after') {
// Check the next sibling if insertMode is 'after'
if (targetElement.nextElementSibling && targetElement.nextElementSibling.classList.contains('download-button')) {
return; // Button already exists after the target
}
}
const button = document.createElement('button');
button.innerText = 'DOWNLOAD';
button.classList.add('download-button');
// **Styling the Button**
button.style.marginLeft = '10px';
button.style.padding = '5px 10px';
button.style.cursor = 'pointer';
button.style.backgroundColor = '#28a745';
button.style.color = '#fff';
button.style.border = 'none';
button.style.borderRadius = '4px';
button.style.fontSize = '14px';
button.style.transition = 'background-color 0.3s';
button.style.boxShadow = '0 2px 4px rgba(0,0,0,0.2)';
// **Hover Effects for the Button**
button.addEventListener('mouseenter', () => {
button.style.backgroundColor = '#218838';
});
button.addEventListener('mouseleave', () => {
button.style.backgroundColor = '#28a745';
});
// **Click Event to Initiate the Download Process**
button.addEventListener('click', async (event) => {
event.preventDefault(); // Prevent default behavior if applicable
button.disabled = true;
const originalText = button.innerText;
button.innerText = 'Processing...';
console.log(`DOWNLOAD button clicked for URL: ${linkUrl}, useCurrentPage: ${useCurrentPage}`);
await downloadImagesFromUrl(linkUrl, useCurrentPage);
button.innerText = originalText;
button.disabled = false;
});
// **Inserting the Button at the Correct Location**
if (insertMode === 'append') {
targetElement.appendChild(button);
console.log(`DOWNLOAD button appended to`, targetElement);
} else { // 'after'
targetElement.insertAdjacentElement('afterend', button);
console.log(`DOWNLOAD button added after`, targetElement);
}
}
/**
* Main function to download images from a given URL.
* @param {string} linkUrl - The URL from which to download images.
* @param {boolean} useCurrentPage - If true, use the current document instead of fetching linkUrl.
*/
async function downloadImagesFromUrl(linkUrl, useCurrentPage = false) {
try {
console.log(`Starting image download process for URL: ${useCurrentPage ? window.location.href : linkUrl}`);
let doc;
if (useCurrentPage) {
doc = document;
} else {
doc = await fetchHTML(linkUrl);
}
// **Extracting the Title for the ZIP Filename**
const titleText = doc.body.textContent || ''; // Assuming the whole body as potential source
console.log(`Extracted text for title matching: ${titleText.substring(0, 100)}...`); // Sample for log
const titleMatch = titleText.match(/\[[^\]]+\] .+? \[[^\]]+P\]/);
if (!titleMatch || !titleMatch[0]) {
console.error('TITLE not found in extracted text.');
alert('Title pattern not matched. Please check the page structure.');
return;
}
const TITLE = titleMatch[0].trim().replace(/[\\/:*?"<>|]/g, '_');
console.log(`Extracted TITLE for ZIP: ${TITLE}`);
// **Collecting All Image Page URLs from div.pagination**
const paginationDiv = doc.querySelector('div.pagination');
let pageUrls = [];
if (paginationDiv) {
const pageLinks = paginationDiv.querySelectorAll('a[href]');
pageUrls = Array.from(pageLinks)
.filter(link => /^\d+$/.test(link.textContent.trim()))
.map(link => makeAbsoluteURL(link.getAttribute('href'), linkUrl));
} else {
console.warn('div.pagination not found. Proceeding with the current page only.');
}
// **Handling Button 2 (Current Page)**
if (useCurrentPage && pageUrls.length === 0) {
pageUrls.push(window.location.href);
}
console.log(`Total related pages found: ${pageUrls.length}`);
// **Storing Unique Image URLs**
const images = new Set();
/**
* Function to process a single page and collect image URLs.
* @param {string} pageUrl - The URL of the page to process.
*/
async function processPage(pageUrl) {
try {
console.log(`Processing page: ${pageUrl}`);
const pageDoc = useCurrentPage && pageUrl === window.location.href ? document : await fetchHTML(pageUrl);
const contentDivs = pageDoc.querySelectorAll('article.article-content p');
if (contentDivs.length > 0) {
let imagesInPage = [];
contentDivs.forEach(div => {
const imgs = div.querySelectorAll('img[src]');
imagesInPage.push(...imgs);
});
console.log(`Found ${imagesInPage.length} images in ${pageUrl}`);
imagesInPage.forEach(img => {
const src = img.getAttribute('src');
if (src) {
images.add(makeAbsoluteURL(src, pageUrl));
}
});
} else {
console.warn(`No article.article-content p found in ${pageUrl}`);
}
} catch (e) {
console.warn(`Failed to process ${pageUrl}: ${e.message}`);
}
}
// **Processing All Pages with Concurrency Control**
const CONCURRENT_LIMIT = 5;
let index = 0;
async function processNext() {
while (index < pageUrls.length) {
const currentIndex = index++;
const pageUrl = pageUrls[currentIndex];
await processPage(pageUrl);
}
}
const workers = [];
for (let i = 0; i < CONCURRENT_LIMIT; i++) {
workers.push(processNext());
}
await Promise.all(workers);
if (images.size === 0) {
console.error('No images found.');
alert('No images found on the target page(s).');
return;
}
const imageArray = Array.from(images);
console.log(`Total images collected: ${imageArray.length}`);
// **Initializing JSZip**
const zip = new JSZip();
let counter = 0;
const totalImages = imageArray.length;
// **Loop Through Each Image to Add to ZIP**
for (let i = 0; i < imageArray.length; i++) {
const imgUrl = imageArray[i];
const isLastImage = (i === imageArray.length - 1); // Check if it's the last image
try {
const blob = await fetchImage(imgUrl);
if (!blob) {
console.warn(`Skipping image due to fetch failure: ${imgUrl}`);
continue;
}
const extension = getFileExtension(imgUrl);
const validExtensions = ['jpg', 'jpeg', 'png', 'webp', 'bmp', 'tiff'];
const finalExtension = validExtensions.includes(extension) ? extension : 'jpg';
let filename;
if (isLastImage) {
// **Rename the Last Image to '000.ext'**
filename = `000.${finalExtension}`;
} else {
// **Name Other Images Sequentially: '001.ext', '002.ext', etc.**
// Note: Counter starts at 0, so add 1 and pad with zeros
filename = `${String(counter + 1).padStart(3, '0')}.${finalExtension}`;
}
zip.file(filename, blob);
counter++;
console.log(`Added ${filename} to ZIP (${counter}/${totalImages})`);
} catch (e) {
console.warn(`Failed to fetch/add image ${imgUrl}: ${e.message}`);
}
}
if (counter === 0) {
console.error('No images were successfully added to the ZIP.');
alert('No images were successfully downloaded.');
return;
}
console.log(`Successfully added ${counter} images to the ZIP.`);
await createAndDownloadZip(zip, `${TITLE}.zip`);
} catch (error) {
console.error(`Error during image download process: ${error.message}`);
alert('An error occurred during the image download process.');
}
}
/**
* Function to find relevant elements and add DOWNLOAD buttons accordingly.
* Adjusted to match the provided HTML structures.
*/
function findAndAddButtons() {
// **Adding DOWNLOAD Button 1 to div.node p a b**
const titleLinks = document.querySelectorAll('div.node p a b');
titleLinks.forEach(link => {
const aTag = link.parentElement; // Get the <a> tag
if (aTag && !aTag.nextElementSibling?.classList.contains('download-button')) {
const href = aTag.getAttribute('href');
if (href) {
const linkUrl = makeAbsoluteURL(href, window.location.href);
addDownloadButton(link, linkUrl, false, 'after'); // Insert after the <b> tag
} else {
console.warn('No href found for div.node p a b element.');
}
}
});
// **Adding DOWNLOAD Button 2 to div.toptip**
const toptipDivs = document.querySelectorAll('div.toptip');
toptipDivs.forEach(div => {
if (!div.querySelector('button.download-button')) {
// Button 2 processes the current page
addDownloadButton(div, window.location.href, true, 'append'); // Insert inside div.toptip
}
});
}
/**
* Function to initialize the script: inject JSZip and add buttons.
*/
async function initializeScript() {
console.log('Initializing JYNGMN Image Downloader script...');
try {
// **Injecting JSZip**
await injectScript('https://cdn.jsdelivr.net/npm/jszip@3.10.0/dist/jszip.min.js');
if (typeof JSZip === 'undefined') {
throw new Error('JSZip failed to load after injection.');
}
console.log('JSZip loaded successfully. Version:', JSZip.version);
// **Adding DOWNLOAD Buttons**
findAndAddButtons();
// **Setting Up MutationObserver to Handle Dynamic Content**
const observer = new MutationObserver((mutations) => {
for (const mutation of mutations) {
if (mutation.type === 'childList') {
findAndAddButtons();
}
}
});
observer.observe(document.body, { childList: true, subtree: true });
console.log('MutationObserver set to watch for new elements to add DOWNLOAD buttons.');
} catch (error) {
console.error('Failed to inject JSZip:', error.message);
alert('Failed to load JSZip library. Please check the console for details.');
}
}
/**
* Ensuring the script runs after the DOM is fully loaded.
*/
if (document.readyState === 'loading') {
document.addEventListener('DOMContentLoaded', initializeScript);
} else {
initializeScript();
}
})();
```
### Key Modifications:
- **Title Extraction**: I updated the part where the script extracts the title using the new regex pattern `\[[^\]]+\] .+? \[[^\]]+P\]`, scanning the entire document’s text content.
- **Logging**: Added logging to provide a sample of extracted text for debugging purposes, and to alert or log when the title pattern match fails.
- **Pre-checks**: Includes further checks and alerts to maintain robustness in scenarios where expected text patterns do not exist.
This script should be suitable for adapting scenarios where you're trying to find sections matching the new flexible pattern in the HTML document's text. Adjustments might be needed depending on the actual HTML structure it interacts with.