diff --git a/extract-images-from-website/add-error-handling-missing-src-attributes-log-warnings-without-interrupting-workflow.cs b/extract-images-from-website/add-error-handling-missing-src-attributes-log-warnings-without-interrupting-workflow.cs new file mode 100644 index 0000000..198ee89 --- /dev/null +++ b/extract-images-from-website/add-error-handling-missing-src-attributes-log-warnings-without-interrupting-workflow.cs @@ -0,0 +1,69 @@ +// Add error handling for missing src attributes and log warnings without interrupting workflow. + +using System; +using System.IO; +using Aspose.Html; +using Aspose.Html.Accessibility; +using Aspose.Html.Accessibility.Results; +using Aspose.Html.Dom; +using Aspose.Html.Dom.XPath; + +class Program +{ + static void Main() + { + try + { + // Sample HTML content with an missing the src attribute + string htmlContent = "no src"; + string baseUri = "about:blank"; + + // Load the document using the two‑argument constructor (content, baseUri) + Aspose.Html.HTMLDocument doc = new Aspose.Html.HTMLDocument(htmlContent, baseUri); + + // Log file path + string logPath = "validation_warnings.log"; + File.WriteAllText(logPath, "Validation warnings:" + Environment.NewLine); + + // ----------------------------------------------------------------- + // Detect elements without a src attribute using XPath + // ----------------------------------------------------------------- + Aspose.Html.Dom.XPath.IXPathResult xpathResult = doc.Evaluate( + "//img[not(@src)]", + doc, + doc.CreateNSResolver(doc), + Aspose.Html.Dom.XPath.XPathResultType.Any, + null); + + Aspose.Html.Dom.Node node; + while ((node = xpathResult.IterateNext()) != null) + { + Aspose.Html.HTMLImageElement img = node as Aspose.Html.HTMLImageElement; + if (img != null) + { + string warning = $"Warning: element missing src attribute. OuterHTML: {img.OuterHTML}"; + File.AppendAllText(logPath, warning + Environment.NewLine); + } + } + + // ----------------------------------------------------------------- + // Run the accessibility validator (optional – logs all rule results) + // ----------------------------------------------------------------- + Aspose.Html.Accessibility.WebAccessibility webAccessibility = new Aspose.Html.Accessibility.WebAccessibility(); + Aspose.Html.Accessibility.AccessibilityValidator validator = webAccessibility.CreateValidator(); + Aspose.Html.Accessibility.Results.ValidationResult validationResult = validator.Validate(doc); + + foreach (Aspose.Html.Accessibility.Results.RuleValidationResult detail in validationResult.Details) + { + // Log each validation detail using its default string representation + File.AppendAllText(logPath, detail.ToString() + Environment.NewLine); + } + + Console.WriteLine("Validation completed. See log at " + Path.GetFullPath(logPath)); + } + catch (Exception ex) + { + Console.WriteLine("Error: " + ex.Message); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/add_retry_logic_for_transient_network_failures_when_downloading_images_or_icons.cs b/extract-images-from-website/add-retry-logic-transient-network-failures-downloading-images-icons.cs similarity index 56% rename from extract-images-from-website/add_retry_logic_for_transient_network_failures_when_downloading_images_or_icons.cs rename to extract-images-from-website/add-retry-logic-transient-network-failures-downloading-images-icons.cs index 8547176..663cf0e 100644 --- a/extract-images-from-website/add_retry_logic_for_transient_network_failures_when_downloading_images_or_icons.cs +++ b/extract-images-from-website/add-retry-logic-transient-network-failures-downloading-images-icons.cs @@ -1,11 +1,16 @@ // Add retry logic for transient network failures when downloading images or icons. using System; +using System.IO; +using Aspose.Html.Net; public sealed class RetryHandler : Aspose.Html.Net.MessageHandler { private readonly int _maxRetries; - public RetryHandler(int maxRetries) { _maxRetries = maxRetries; } + public RetryHandler(int maxRetries) + { + _maxRetries = maxRetries; + } public override void Invoke(Aspose.Html.Net.INetworkOperationContext context) { @@ -13,29 +18,38 @@ public override void Invoke(Aspose.Html.Net.INetworkOperationContext context) { Next(context); if (context.Response != null && (int)context.Response.StatusCode < 500) + { break; + } } } } -class Program +public class Program { - static void Main() + public static void Main() { try { + // Input HTML URL (can be any reachable page with images/icons) + string inputUrl = "https://example.com/sample.html"; + // Output image file path + string outputPath = Path.Combine(Directory.GetCurrentDirectory(), "output.png"); + + // Configure Aspose.HTML with retry handler Aspose.Html.Configuration configuration = new Aspose.Html.Configuration(); Aspose.Html.Services.INetworkService networkService = configuration.GetService(); networkService.MessageHandlers.Add(new RetryHandler(3)); - string html = "

Sample

sample"; - using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(html, configuration)) + // Load the HTML document from the URL using the configured network service + using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(inputUrl, configuration)) { - Aspose.Html.Saving.ImageSaveOptions options = new Aspose.Html.Saving.ImageSaveOptions(); - Aspose.Html.Converters.Converter.ConvertHTML(document, options, "output.png"); + // Convert the HTML document to an image (PNG) + Aspose.Html.Saving.ImageSaveOptions saveOptions = new Aspose.Html.Saving.ImageSaveOptions(); + Aspose.Html.Converters.Converter.ConvertHTML(document, saveOptions, outputPath); } - Console.WriteLine("Conversion completed successfully."); + Console.WriteLine("Conversion completed successfully. Output saved to: " + outputPath); } catch (Exception ex) { diff --git a/extract-images-from-website/add_error_handling_for_missing_src_attributes_and_log_warnings_without_interrupting_workflow.cs b/extract-images-from-website/add_error_handling_for_missing_src_attributes_and_log_warnings_without_interrupting_workflow.cs deleted file mode 100644 index e25f9f6..0000000 --- a/extract-images-from-website/add_error_handling_for_missing_src_attributes_and_log_warnings_without_interrupting_workflow.cs +++ /dev/null @@ -1,50 +0,0 @@ -// Add error handling for missing src attributes and log warnings without interrupting workflow. - -using System; -using System.IO; - -class Program -{ - static void Main() - { - try - { - string htmlPath = "sample.html"; - string logPath = "log.txt"; - - if (!File.Exists(htmlPath)) - { - string sampleHtml = "" + - "" + - "" + - "" + - ""; - File.WriteAllText(htmlPath, sampleHtml); - } - - Aspose.Html.HTMLDocument doc = new Aspose.Html.HTMLDocument(htmlPath); - - Aspose.Html.Dom.XPath.IXPathResult result = doc.Evaluate("//img", doc, doc.CreateNSResolver(doc), Aspose.Html.Dom.XPath.XPathResultType.Any, null); - - Aspose.Html.Dom.Node node; - while ((node = result.IterateNext()) != null) - { - Aspose.Html.HTMLImageElement img = node as Aspose.Html.HTMLImageElement; - if (img != null) - { - if (string.IsNullOrEmpty(img.Src)) - { - string warning = "Warning: element missing src attribute."; - File.AppendAllText(logPath, warning + Environment.NewLine); - } - } - } - - Console.WriteLine("Processing completed. Check log for warnings."); - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/apply_naming_pattern_prefixes_saved_file_source_domain_name.cs b/extract-images-from-website/apply-naming-pattern-prefixes-each-saved-file-with-source-domain-name.cs similarity index 65% rename from extract-images-from-website/apply_naming_pattern_prefixes_saved_file_source_domain_name.cs rename to extract-images-from-website/apply-naming-pattern-prefixes-each-saved-file-with-source-domain-name.cs index 8c0501c..7fa8922 100644 --- a/extract-images-from-website/apply_naming_pattern_prefixes_saved_file_source_domain_name.cs +++ b/extract-images-from-website/apply-naming-pattern-prefixes-each-saved-file-with-source-domain-name.cs @@ -5,7 +5,7 @@ class Program { - static void Main() + static void Main(string[] args) { try { @@ -16,33 +16,36 @@ static void Main() // Input files string[] inputs = new string[] { "sample1.html", "sample2.html" }; - // Create minimal sample HTML files if they do not exist - if (!File.Exists(inputs[0])) - File.WriteAllText(inputs[0], "

Sample 1

"); - if (!File.Exists(inputs[1])) - File.WriteAllText(inputs[1], "

Sample 2

"); - for (int i = 0; i < inputs.Length; i++) { string inputPath = inputs[i]; - using (var document = new Aspose.Html.HTMLDocument(inputPath, Directory.GetCurrentDirectory())) + // Ensure sample input file exists (minimal content) + if (!File.Exists(inputPath)) + { + File.WriteAllText(inputPath, "

Sample

"); + } + + using (var document = new Aspose.Html.HTMLDocument(inputPath)) { // Configure image options var options = new Aspose.Html.Saving.ImageSaveOptions(Aspose.Html.Rendering.Image.ImageFormat.Jpeg); options.HorizontalResolution = 300; options.VerticalResolution = 300; - // Determine domain name for prefix + // Determine domain prefix string domain; - Uri uri = new Uri(Path.GetFullPath(inputPath), UriKind.Absolute); - if (uri.IsAbsoluteUri && !string.IsNullOrEmpty(uri.Host)) + if (Uri.TryCreate(inputPath, UriKind.Absolute, out Uri uri) && uri.IsAbsoluteUri && !string.IsNullOrEmpty(uri.Host)) + { domain = uri.Host; + } else + { domain = "local"; + } - string outputFileName = $"{domain}_{Path.GetFileNameWithoutExtension(inputPath)}.jpg"; - string outputPath = Path.Combine(outputDir, outputFileName); + string fileName = $"{domain}_{Path.GetFileNameWithoutExtension(inputPath)}.jpeg"; + string outputPath = Path.Combine(outputDir, fileName); // Convert Aspose.Html.Converters.Converter.ConvertHTML(document, options, outputPath); diff --git a/extract-images-from-website/archive-all-extracted-images-icons-zip-file-easy-distribution.cs b/extract-images-from-website/archive-all-extracted-images-icons-zip-file-easy-distribution.cs new file mode 100644 index 0000000..0efc661 --- /dev/null +++ b/extract-images-from-website/archive-all-extracted-images-icons-zip-file-easy-distribution.cs @@ -0,0 +1,123 @@ +// Archive all extracted images and icons into a ZIP file for easy distribution. + +using System; +using System.IO; +using System.IO.Compression; +using System.Collections.Generic; +using Aspose.Html; + +class Program +{ + static void Main() + { + try + { + // Define paths + string sourceZipPath = Path.Combine(Path.GetTempPath(), "source.zip"); + string extractDir = Path.Combine(Path.GetTempPath(), "extracted_assets"); + string outputZipPath = Path.Combine(Path.GetTempPath(), "ImagesAndIcons.zip"); + + // Ensure clean directories + if (Directory.Exists(extractDir)) + Directory.Delete(extractDir, true); + Directory.CreateDirectory(extractDir); + + // Create a sample source zip with dummy image/icon files if it does not exist + if (!File.Exists(sourceZipPath)) + { + using (FileStream srcStream = new FileStream(sourceZipPath, FileMode.Create)) + using (ZipArchive srcArchive = new ZipArchive(srcStream, ZipArchiveMode.Create)) + { + // Dummy PNG + ZipArchiveEntry pngEntry = srcArchive.CreateEntry("images/sample.png"); + using (Stream entryStream = pngEntry.Open()) + using (StreamWriter writer = new StreamWriter(entryStream)) + { + writer.Write("dummy png content"); + } + + // Dummy JPG + ZipArchiveEntry jpgEntry = srcArchive.CreateEntry("icons/icon.jpg"); + using (Stream entryStream = jpgEntry.Open()) + using (StreamWriter writer = new StreamWriter(entryStream)) + { + writer.Write("dummy jpg content"); + } + + // Non-image file (should be ignored) + ZipArchiveEntry txtEntry = srcArchive.CreateEntry("docs/readme.txt"); + using (Stream entryStream = txtEntry.Open()) + using (StreamWriter writer = new StreamWriter(entryStream)) + { + writer.Write("just a text file"); + } + } + } + + // Extract only image and icon files from the source zip + using (FileStream zipStream = File.OpenRead(sourceZipPath)) + using (ZipArchive archive = new ZipArchive(zipStream, ZipArchiveMode.Read)) + { + foreach (ZipArchiveEntry entry in archive.Entries) + { + if (entry.FullName.EndsWith(".png", StringComparison.OrdinalIgnoreCase) || + entry.FullName.EndsWith(".jpg", StringComparison.OrdinalIgnoreCase) || + entry.FullName.EndsWith(".jpeg", StringComparison.OrdinalIgnoreCase) || + entry.FullName.EndsWith(".gif", StringComparison.OrdinalIgnoreCase) || + entry.FullName.EndsWith(".ico", StringComparison.OrdinalIgnoreCase) || + entry.FullName.EndsWith(".svg", StringComparison.OrdinalIgnoreCase)) + { + string entryPath = Path.Combine(extractDir, entry.FullName); + Directory.CreateDirectory(Path.GetDirectoryName(entryPath)); + entry.ExtractToFile(entryPath, true); + } + } + } + + // Optional: Load an HTML file from the extracted assets and render to PDF using Aspose.HTML + // (Demonstrates usage of Aspose.HTML without affecting the zip operation) + string[] htmlFiles = Directory.GetFiles(extractDir, "*.html", SearchOption.AllDirectories); + if (htmlFiles.Length > 0) + { + Aspose.Html.Configuration configuration = new Aspose.Html.Configuration(); + using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(htmlFiles[0], configuration)) + { + string pdfPath = Path.ChangeExtension(htmlFiles[0], ".pdf"); + using (Aspose.Html.Rendering.Pdf.PdfDevice device = new Aspose.Html.Rendering.Pdf.PdfDevice(pdfPath)) + { + document.RenderTo(device); + } + } + } + + // Create a new zip containing only the extracted images and icons + if (File.Exists(outputZipPath)) + File.Delete(outputZipPath); + using (FileStream outStream = new FileStream(outputZipPath, FileMode.Create)) + using (ZipArchive outArchive = new ZipArchive(outStream, ZipArchiveMode.Create)) + { + foreach (string filePath in Directory.GetFiles(extractDir, "*.*", SearchOption.AllDirectories)) + { + if (filePath.EndsWith(".png", StringComparison.OrdinalIgnoreCase) || + filePath.EndsWith(".jpg", StringComparison.OrdinalIgnoreCase) || + filePath.EndsWith(".jpeg", StringComparison.OrdinalIgnoreCase) || + filePath.EndsWith(".gif", StringComparison.OrdinalIgnoreCase) || + filePath.EndsWith(".ico", StringComparison.OrdinalIgnoreCase) || + filePath.EndsWith(".svg", StringComparison.OrdinalIgnoreCase)) + { + string entryName = Path.GetRelativePath(extractDir, filePath); + outArchive.CreateEntryFromFile(filePath, entryName); + } + } + } + + Console.WriteLine("Extraction and archiving completed successfully."); + Console.WriteLine($"Extracted assets directory: {extractDir}"); + Console.WriteLine($"Output ZIP file: {outputZipPath}"); + } + catch (Exception ex) + { + Console.Error.WriteLine($"Error: {ex.Message}"); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/archive_all_extracted_images_and_icons_zip_for_easy_distribution.cs b/extract-images-from-website/archive_all_extracted_images_and_icons_zip_for_easy_distribution.cs deleted file mode 100644 index aaba674..0000000 --- a/extract-images-from-website/archive_all_extracted_images_and_icons_zip_for_easy_distribution.cs +++ /dev/null @@ -1,75 +0,0 @@ -// Archive all extracted images and icons into a ZIP file for easy distribution. - -using System; -using System.IO; -using System.IO.Compression; - -class Program -{ - static void Main() - { - try - { - // Input ZIP containing HTML, images, icons - string zipPath = "sample.zip"; - // Directory to extract files - string extractDir = Path.Combine(Path.GetTempPath(), "ExtractedContent"); - Directory.CreateDirectory(extractDir); - - // Extract images, icons, and HTML files - using (FileStream zipStream = File.OpenRead(zipPath)) - using (ZipArchive archive = new ZipArchive(zipStream, ZipArchiveMode.Read)) - { - foreach (ZipArchiveEntry entry in archive.Entries) - { - if (entry.FullName.EndsWith(".png", StringComparison.OrdinalIgnoreCase) || - entry.FullName.EndsWith(".ico", StringComparison.OrdinalIgnoreCase) || - entry.FullName.EndsWith(".html", StringComparison.OrdinalIgnoreCase) || - entry.FullName.EndsWith(".htm", StringComparison.OrdinalIgnoreCase)) - { - string entryPath = Path.Combine(extractDir, entry.FullName); - Directory.CreateDirectory(Path.GetDirectoryName(entryPath)); - entry.ExtractToFile(entryPath, true); - } - } - } - - // If an HTML file was extracted, render it to PDF using Aspose.HTML - string[] htmlFiles = Directory.GetFiles(extractDir, "*.html", SearchOption.AllDirectories); - if (htmlFiles.Length == 0) - htmlFiles = Directory.GetFiles(extractDir, "*.htm", SearchOption.AllDirectories); - - if (htmlFiles.Length > 0) - { - string htmlPath = htmlFiles[0]; - Aspose.Html.Configuration configuration = new Aspose.Html.Configuration(); - using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(htmlPath, configuration)) - { - string pdfPath = Path.Combine(extractDir, "output.pdf"); - using (Aspose.Html.Rendering.Pdf.PdfDevice device = new Aspose.Html.Rendering.Pdf.PdfDevice(pdfPath)) - { - document.RenderTo(device); - } - } - } - - // Create a ZIP archive containing all extracted images, icons, and the generated PDF - string outputZipPath = "images_archive.zip"; - using (FileStream zipToCreate = new FileStream(outputZipPath, FileMode.Create)) - using (ZipArchive outArchive = new ZipArchive(zipToCreate, ZipArchiveMode.Create)) - { - foreach (string file in Directory.GetFiles(extractDir, "*.*", SearchOption.AllDirectories)) - { - string entryName = Path.GetRelativePath(extractDir, file); - outArchive.CreateEntryFromFile(file, entryName); - } - } - - Console.WriteLine("Images, icons, and PDF have been archived to: " + outputZipPath); - } - catch (Exception ex) - { - Console.Error.WriteLine("An error occurred: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/implement_batch_extraction_process_multiple_web_pages_sequentially_single_run.cs b/extract-images-from-website/batch-extraction-process-multiple-web-pages-sequentially-single-run.cs similarity index 84% rename from extract-images-from-website/implement_batch_extraction_process_multiple_web_pages_sequentially_single_run.cs rename to extract-images-from-website/batch-extraction-process-multiple-web-pages-sequentially-single-run.cs index 67de075..5e21ec7 100644 --- a/extract-images-from-website/implement_batch_extraction_process_multiple_web_pages_sequentially_single_run.cs +++ b/extract-images-from-website/batch-extraction-process-multiple-web-pages-sequentially-single-run.cs @@ -19,8 +19,8 @@ static void Main() { try { - string inputFolder = "InputHtml"; - string outputFolder = "OutputJson"; + string inputFolder = "input_html"; + string outputFolder = "output_json"; if (!Directory.Exists(inputFolder)) Directory.CreateDirectory(inputFolder); @@ -28,10 +28,20 @@ static void Main() Directory.CreateDirectory(outputFolder); // Create a sample HTML file if none exist - if (Directory.GetFiles(inputFolder, "*.html").Length == 0) + string[] existingFiles = Directory.GetFiles(inputFolder, "*.html"); + if (existingFiles.Length == 0) { string samplePath = Path.Combine(inputFolder, "sample.html"); - File.WriteAllText(samplePath, "

Title

Section

Subsection

"); + File.WriteAllText(samplePath, +@" +Sample + +

Title

+

Section 1

+

Subsection 1.1

+

Section 2

+ +"); } foreach (string htmlPath in Directory.GetFiles(inputFolder, "*.html")) @@ -76,7 +86,7 @@ static void Main() } catch (Exception ex) { - Console.WriteLine($"Error: {ex.Message}"); + Console.WriteLine("Error: " + ex.Message); } } } \ No newline at end of file diff --git a/extract-images-from-website/collect-all-link-elements-with-rel-icon-attribute-from-loaded-html-document.cs b/extract-images-from-website/collect-all-link-elements-with-rel-icon-attribute-from-loaded-html-document.cs new file mode 100644 index 0000000..837ffea --- /dev/null +++ b/extract-images-from-website/collect-all-link-elements-with-rel-icon-attribute-from-loaded-html-document.cs @@ -0,0 +1,32 @@ +// Collect all elements with rel='icon' attribute from the loaded HTML document. + +using System; + +public class Program +{ + public static void Main() + { + try + { + string html = ""; + using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(html, "about:blank")) + { + Aspose.Html.Collections.HTMLCollection linkElements = document.GetElementsByTagName("link"); + for (int i = 0; i < linkElements.Length; i++) + { + Aspose.Html.Dom.Element link = linkElements[i]; + string rel = link.GetAttribute("rel"); + if (!string.IsNullOrEmpty(rel) && rel.Equals("icon", StringComparison.OrdinalIgnoreCase)) + { + string href = link.GetAttribute("href"); + System.Console.WriteLine($"Icon href: {href}"); + } + } + } + } + catch (Exception ex) + { + System.Console.WriteLine($"Error: {ex.Message}"); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/collect_all_link_elements_with_rel_icon_attribute_from_loaded_html_document.cs b/extract-images-from-website/collect_all_link_elements_with_rel_icon_attribute_from_loaded_html_document.cs deleted file mode 100644 index 57f6d33..0000000 --- a/extract-images-from-website/collect_all_link_elements_with_rel_icon_attribute_from_loaded_html_document.cs +++ /dev/null @@ -1,30 +0,0 @@ -// Collect all elements with rel='icon' attribute from the loaded HTML document. - -class Program -{ - static void Main() - { - try - { - string htmlContent = ""; - string filePath = System.IO.Path.Combine(System.IO.Path.GetTempPath(), "sample.html"); - System.IO.File.WriteAllText(filePath, htmlContent); - Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(filePath); - Aspose.Html.Collections.HTMLCollection links = document.GetElementsByTagName("link"); - for (int i = 0; i < links.Length; i++) - { - Aspose.Html.Dom.Element link = links[i]; - string rel = link.GetAttribute("rel"); - if (!string.IsNullOrEmpty(rel) && rel.Equals("icon", System.StringComparison.OrdinalIgnoreCase)) - { - string href = link.GetAttribute("href"); - System.Console.WriteLine($"Icon link: href='{href}'"); - } - } - } - catch (System.Exception ex) - { - System.Console.WriteLine($"Error: {ex.Message}"); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/configure-httpclient-proxy-settings-extraction-corporate-firewalls.cs b/extract-images-from-website/configure-httpclient-proxy-settings-extraction-corporate-firewalls.cs new file mode 100644 index 0000000..eaf93e9 --- /dev/null +++ b/extract-images-from-website/configure-httpclient-proxy-settings-extraction-corporate-firewalls.cs @@ -0,0 +1,40 @@ +// Configure HttpClient with proxy settings to support extraction behind corporate firewalls. + +using System; +using System.Net; +using System.Net.Http; + +class Program +{ + static void Main() + { + try + { + // Configure proxy + var proxy = new WebProxy("http://proxy.example.com:8080"); + var httpHandler = new HttpClientHandler + { + Proxy = proxy, + UseProxy = true + }; + + using (var httpClient = new HttpClient(httpHandler)) + { + string url = "https://example.com"; + string htmlContent = httpClient.GetStringAsync(url).Result; + + // Load HTML content with Aspose.HTML + var configuration = new Aspose.Html.Configuration(); + using (var document = new Aspose.Html.HTMLDocument(htmlContent, url, configuration)) + { + // Save the document to a file + document.Save("output.html"); + } + } + } + catch (Exception ex) + { + Console.WriteLine($"Error: {ex.Message}"); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/configure_httpclient_proxy_settings_support_extraction_behind_corporate_firewalls.cs b/extract-images-from-website/configure_httpclient_proxy_settings_support_extraction_behind_corporate_firewalls.cs deleted file mode 100644 index f008064..0000000 --- a/extract-images-from-website/configure_httpclient_proxy_settings_support_extraction_behind_corporate_firewalls.cs +++ /dev/null @@ -1,58 +0,0 @@ -// Configure HttpClient with proxy settings to support extraction behind corporate firewalls. - -using System; -using System.Net; -using Aspose.Html; -using Aspose.Html.Net; -using Aspose.Html.Services; - -class Program -{ - static void Main(string[] args) - { - try - { - // Configure proxy settings - var proxy = new WebProxy("http://proxy.example.com:8080", false); - WebRequest.DefaultWebProxy = proxy; - - // Create Aspose.HTML configuration - var configuration = new Aspose.Html.Configuration(); - - // Optionally, add a message handler for credentials if needed - // var network = configuration.GetService(); - // network.MessageHandlers.Add(new ProxyCredentialsHandler(new NetworkCredential("user", "password"))); - - // Create request message with target URL - var request = new Aspose.Html.Net.RequestMessage("https://example.com"); - - // Load the document using the configured proxy - using (var document = new Aspose.Html.HTMLDocument(request, configuration)) - { - // Save the extracted HTML to a local file - document.Save("output.html"); - } - - Console.WriteLine("HTML document saved successfully."); - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} - -// Example of a custom message handler to set credentials (optional) -class ProxyCredentialsHandler : Aspose.Html.Net.MessageHandler -{ - private readonly ICredentials _credentials; - public ProxyCredentialsHandler(ICredentials credentials) - { - _credentials = credentials; - } - public override void Invoke(Aspose.Html.Net.INetworkOperationContext context) - { - context.Request.Credentials = _credentials; - Next(context); - } -} \ No newline at end of file diff --git a/extract-images-from-website/convert-downloaded-image-bytes-into-base64-strings-json-payload-embedding.cs b/extract-images-from-website/convert-downloaded-image-bytes-into-base64-strings-json-payload-embedding.cs new file mode 100644 index 0000000..cb43212 --- /dev/null +++ b/extract-images-from-website/convert-downloaded-image-bytes-into-base64-strings-json-payload-embedding.cs @@ -0,0 +1,107 @@ +// Convert downloaded image bytes to Base64 strings for embedding into JSON payloads. + +using System; +using System.IO; +using System.Drawing; +using System.Drawing.Imaging; +using System.Collections.Generic; +using Aspose.Html.IO; + +class MemoryStreamProvider : Aspose.Html.IO.ICreateStreamProvider, IDisposable +{ + private readonly List _streams = new List(); + public IReadOnlyList Streams => _streams; + + public Stream GetStream(string name, string extension) + { + var ms = new MemoryStream(); + _streams.Add(ms); + return ms; + } + + public Stream GetStream(string name, string extension, int page) + { + var ms = new MemoryStream(); + _streams.Add(ms); + return ms; + } + + public void ReleaseStream(Stream stream) + { + // No action needed for in-memory streams + } + + public void Dispose() + { + foreach (var ms in _streams) + { + ms.Dispose(); + } + _streams.Clear(); + } +} + +class Program +{ + static void Main() + { + try + { + // Prepare a sample PNG image (1x1 red pixel) if it does not exist + string sampleImagePath = "sample.png"; + if (!File.Exists(sampleImagePath)) + { + using (var bmp = new Bitmap(1, 1)) + { + bmp.SetPixel(0, 0, Color.Red); + bmp.Save(sampleImagePath, ImageFormat.Png); + } + } + + // Read image bytes and convert to Base64 + byte[] imageBytes = File.ReadAllBytes(sampleImagePath); + string base64Image = Convert.ToBase64String(imageBytes); + + // Create HTML content embedding the image + string htmlContent = $""; + + // Initialize stream provider + using (var provider = new MemoryStreamProvider()) + { + // Create HTML document from inline content + var document = new Aspose.Html.HTMLDocument(htmlContent, "about:blank"); + + // Set image save options (PNG format) + var options = new Aspose.Html.Saving.ImageSaveOptions(Aspose.Html.Rendering.Image.ImageFormat.Png); + + // Convert HTML to PNG using the provider + Aspose.Html.Converters.Converter.ConvertHTML(document, options, provider); + + // Retrieve the generated PNG stream + if (provider.Streams.Count == 0) + throw new InvalidOperationException("No output stream was generated."); + + var outputStream = provider.Streams[0]; + outputStream.Position = 0; + + // Save the PNG to a file + string outputPath = "output.png"; + using (var fileStream = File.Create(outputPath)) + { + outputStream.CopyTo(fileStream); + } + + // Convert the resulting PNG to Base64 for JSON payload + string outputBase64 = Convert.ToBase64String(File.ReadAllBytes(outputPath)); + string jsonPayload = $"{{\"image\":\"{outputBase64}\"}}"; + + Console.WriteLine("JSON payload with embedded image:"); + Console.WriteLine(jsonPayload); + } + } + catch (Exception ex) + { + Console.WriteLine($"Error: {ex.Message}"); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/convert_downloaded_image_bytes_to_base64_strings_for_json_payload_embedding.cs b/extract-images-from-website/convert_downloaded_image_bytes_to_base64_strings_for_json_payload_embedding.cs deleted file mode 100644 index 16622ff..0000000 --- a/extract-images-from-website/convert_downloaded_image_bytes_to_base64_strings_for_json_payload_embedding.cs +++ /dev/null @@ -1,30 +0,0 @@ -// Convert downloaded image bytes to Base64 strings for embedding into JSON payloads. - -using System; -using System.Text.Json; - -class Program -{ - static void Main() - { - try - { - // Simulate downloaded image bytes (e.g., a minimal JPEG header) - byte[] imageBytes = new byte[] { 0xFF, 0xD8, 0xFF, 0xE0, 0x00, 0x10, 0x4A, 0x46, 0x49, 0x46, 0x00, 0x01 }; - - // Convert to Base64 string - string base64String = Convert.ToBase64String(imageBytes); - - // Create JSON payload with the Base64 image - var payload = new { image = base64String }; - string json = JsonSerializer.Serialize(payload); - - // Output the JSON - Console.WriteLine(json); - } - catch (Exception ex) - { - Console.Error.WriteLine($"Error: {ex.Message}"); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/create-reusable-method-accepts-url-returns-list-absolute-image-urls.cs b/extract-images-from-website/create-reusable-method-accepts-url-returns-list-absolute-image-urls.cs new file mode 100644 index 0000000..6e7cde0 --- /dev/null +++ b/extract-images-from-website/create-reusable-method-accepts-url-returns-list-absolute-image-urls.cs @@ -0,0 +1,46 @@ +// Create a reusable method that accepts a URL and returns a list of absolute image URLs. + +using System; +using System.Collections.Generic; +using Aspose.Html; +using Aspose.Html.Collections; +using Aspose.Html.Dom; +using Aspose.Html.Net; + +class Program +{ + static void Main() + { + try + { + string url = "https://example.com"; + List imageUrls = GetImageUrls(url); + Console.WriteLine("Found " + imageUrls.Count + " image(s):"); + foreach (string imgUrl in imageUrls) + { + Console.WriteLine(imgUrl); + } + } + catch (Exception ex) + { + Console.WriteLine("Error: " + ex.Message); + } + } + + static List GetImageUrls(string url) + { + var result = new List(); + HTMLDocument document = new HTMLDocument(url); + HTMLCollection images = document.GetElementsByTagName("img"); + for (int i = 0; i < images.Length; i++) + { + Element imgElement = (Element)images[i]; + string src = imgElement.GetAttribute("src"); + if (string.IsNullOrEmpty(src)) + continue; + Url absoluteUrl = new Url(src, document.BaseURI); + result.Add(absoluteUrl.ToString()); + } + return result; + } +} \ No newline at end of file diff --git a/extract-images-from-website/create_reusable_method_accepts_url_returns_list_of_absolute_image_urls.cs b/extract-images-from-website/create_reusable_method_accepts_url_returns_list_of_absolute_image_urls.cs deleted file mode 100644 index 5587afe..0000000 --- a/extract-images-from-website/create_reusable_method_accepts_url_returns_list_of_absolute_image_urls.cs +++ /dev/null @@ -1,57 +0,0 @@ -// Create a reusable method that accepts a URL and returns a list of absolute image URLs. - -using System; -using System.Collections.Generic; - -class Program -{ - static void Main() - { - try - { - string pageUrl = "https://example.com"; - List imageUrls = GetAbsoluteImageUrls(pageUrl); - foreach (string url in imageUrls) - { - Console.WriteLine(url); - } - } - catch (Exception ex) - { - Console.WriteLine($"Error: {ex.Message}"); - } - } - - static List GetAbsoluteImageUrls(string pageUrl) - { - var result = new List(); - var document = new Aspose.Html.HTMLDocument(pageUrl); - var images = document.GetElementsByTagName("img"); - for (int i = 0; i < images.Length; i++) - { - var imgElement = (Aspose.Html.Dom.Element)images[i]; - string src = imgElement.GetAttribute("src"); - if (string.IsNullOrEmpty(src)) - continue; - - Uri baseUri; - try - { - baseUri = new Uri(document.BaseURI); - } - catch - { - baseUri = new Uri(pageUrl); - } - - Uri absoluteUri; - if (Uri.IsWellFormedUriString(src, UriKind.Absolute)) - absoluteUri = new Uri(src); - else - absoluteUri = new Uri(baseUri, src); - - result.Add(absoluteUri.ToString()); - } - return result; - } -} \ No newline at end of file diff --git a/extract-images-from-website/document-extraction-workflow-inline-code-comments-generate-xml-documentation-public-methods.cs b/extract-images-from-website/document-extraction-workflow-inline-code-comments-generate-xml-documentation-public-methods.cs new file mode 100644 index 0000000..e76a09b --- /dev/null +++ b/extract-images-from-website/document-extraction-workflow-inline-code-comments-generate-xml-documentation-public-methods.cs @@ -0,0 +1,168 @@ +// Document the extraction workflow with inline code comments and generate XML documentation for public methods. + +using System; +using System.IO; +using System.Collections.Generic; + +namespace AsposeHtmlExample +{ + /// + /// Demonstrates accessibility validation, XPath extraction, and network header capture using Aspose.HTML for .NET. + /// + class Program + { + static void Main() + { + const string inputHtmlPath = "sample.html"; + const string outputHtmlPath = "output.html"; + const string logPath = "log.txt"; + + try + { + // Prepare a minimal HTML file. + if (!File.Exists(inputHtmlPath)) + { + string sampleHtml = @" + +Sample + +

Test Document

+ Image 1 + Image 2 + +"; + File.WriteAllText(inputHtmlPath, sampleHtml); + } + + // Clear previous log. + if (File.Exists(logPath)) + { + File.Delete(logPath); + } + + ValidateDocument(inputHtmlPath, logPath); + ExtractImageSources(inputHtmlPath, logPath); + ProcessDocumentWithNetworkHandler(inputHtmlPath, outputHtmlPath, logPath); + } + catch (Exception ex) + { + Console.Error.WriteLine($"Error: {ex.Message}"); + } + } + + /// + /// Performs accessibility validation on the specified HTML document and logs the result in XML format. + /// + /// Path to the HTML file to validate. + /// Path to the log file where validation output will be appended. + public static void ValidateDocument(string htmlPath, string logPath) + { + // Load the document. + using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(htmlPath)) + { + // Create the validator. + Aspose.Html.Accessibility.AccessibilityValidator validator = + new Aspose.Html.Accessibility.WebAccessibility().CreateValidator(); + + // Run validation. + Aspose.Html.Accessibility.Results.ValidationResult validationResult = validator.Validate(document); + + // Save validation result as XML to a string. + using (StringWriter sw = new StringWriter()) + { + validationResult.SaveTo(sw, Aspose.Html.Accessibility.Saving.ValidationResultSaveFormat.XML); + File.AppendAllText(logPath, sw.ToString() + Environment.NewLine); + } + } + } + + /// + /// Extracts the src attribute of all <img> elements using XPath and logs each value. + /// + /// Path to the HTML file to process. + /// Path to the log file where image sources will be appended. + public static void ExtractImageSources(string htmlPath, string logPath) + { + using (Aspose.Html.HTMLDocument doc = new Aspose.Html.HTMLDocument(htmlPath)) + { + // Evaluate XPath to select all img elements. + Aspose.Html.Dom.XPath.IXPathResult result = doc.Evaluate( + "//img", + doc, + doc.CreateNSResolver(doc), + Aspose.Html.Dom.XPath.XPathResultType.Any, + null); + + Aspose.Html.Dom.Node node; + while ((node = result.IterateNext()) != null) + { + Aspose.Html.HTMLImageElement img = (Aspose.Html.HTMLImageElement)node; + File.AppendAllText(logPath, img.Src + Environment.NewLine); + } + } + } + + /// + /// Loads the HTML document with a custom network message handler that captures response headers, + /// inserts them as a comment node at the top of the document, and saves the modified document. + /// + /// Path to the HTML file to load. + /// Path where the processed HTML document will be saved. + /// Path to the log file where captured headers will be appended. + public static void ProcessDocumentWithNetworkHandler(string htmlPath, string outputPath, string logPath) + { + // Configure Aspose.HTML with a custom network service. + Aspose.Html.Configuration configuration = new Aspose.Html.Configuration(); + Aspose.Html.Services.INetworkService networkService = configuration.GetService(); + networkService.MessageHandlers.Add(new NetworkMessageHandler()); + + // Load the document using the custom configuration. + using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(htmlPath, configuration)) + { + // If any headers were captured, insert them as a comment node. + if (NetworkMessageHandler.CapturedHeaders.Count > 0) + { + string commentText = "\n" + string.Join("\n", NetworkMessageHandler.CapturedHeaders) + "\n"; + var commentNode = document.CreateComment(commentText); + document.InsertBefore(commentNode, document.DocumentElement); + } + + // Save the modified document. + document.Save(outputPath); + } + + // Log captured headers. + if (NetworkMessageHandler.CapturedHeaders.Count > 0) + { + File.AppendAllText(logPath, "Captured Headers:" + Environment.NewLine); + foreach (string header in NetworkMessageHandler.CapturedHeaders) + { + File.AppendAllText(logPath, header + Environment.NewLine); + } + } + } + } + + /// + /// Custom message handler that records all response headers from network operations. + /// + class NetworkMessageHandler : Aspose.Html.Net.MessageHandler + { + public static List CapturedHeaders = new List(); + + public override void Invoke(Aspose.Html.Net.INetworkOperationContext context) + { + // Continue the request pipeline. + Next(context); + + // Capture response headers. + foreach (object headerItem in context.Response.Headers) + { + if (headerItem != null) + { + CapturedHeaders.Add(headerItem.ToString()); + } + } + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/document_extraction_workflow_inline_code_comments_generate_xml_documentation_public_methods.cs b/extract-images-from-website/document_extraction_workflow_inline_code_comments_generate_xml_documentation_public_methods.cs deleted file mode 100644 index 301d6c0..0000000 --- a/extract-images-from-website/document_extraction_workflow_inline_code_comments_generate_xml_documentation_public_methods.cs +++ /dev/null @@ -1,141 +0,0 @@ -// Document the extraction workflow with inline code comments and generate XML documentation for public methods. - -using System; -using System.IO; -using System.Collections.Generic; -using Aspose.Html; -using Aspose.Html.Accessibility; -using Aspose.Html.Accessibility.Results; -using Aspose.Html.Accessibility.Saving; -using Aspose.Html.Dom; -using Aspose.Html.Dom.XPath; -using Aspose.Html.Net; -using Aspose.Html.Services; - -public class Program -{ - /// - /// Entry point of the console application. - /// - public static void Main() - { - try - { - // Define file paths - string inputPath = "sample.html"; - string outputPath = "output.html"; - string logPath = "extraction.log"; - - // Ensure a minimal HTML file exists - if (!File.Exists(inputPath)) - { - File.WriteAllText(inputPath, ""); - } - - // Create configuration and attach custom network message handler - Aspose.Html.Configuration configuration = new Aspose.Html.Configuration(); - INetworkService networkService = configuration.GetService(); - networkService.MessageHandlers.Add(new MyMessageHandler()); - - // Load the HTML document with the custom configuration - using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(inputPath, configuration)) - { - // If any HTTP headers were captured, insert them as a comment node at the top of the document - if (MyMessageHandler.CapturedHeaders.Count > 0) - { - string commentText = "\n" + string.Join("\n", MyMessageHandler.CapturedHeaders) + "\n"; - var commentNode = document.CreateComment(commentText); - document.InsertBefore(commentNode, document.DocumentElement); - } - - // Perform accessibility validation and log the result - ValidateAccessibility(document, logPath); - - // Extract image sources using XPath and log them - ExtractImageSources(document, logPath); - - // Save the modified document to the output file - document.Save(outputPath); - } - - Console.WriteLine("Processing completed successfully."); - } - catch (Exception ex) - { - Console.Error.WriteLine($"Error: {ex.Message}"); - } - } - - /// - /// Validates the accessibility of the provided HTML document and appends the validation result to the log file. - /// - /// The HTML document to validate. - /// The path of the log file where the validation result will be appended. - public static void ValidateAccessibility(Aspose.Html.HTMLDocument document, string logPath) - { - // Create an accessibility validator - Aspose.Html.Accessibility.AccessibilityValidator validator = new Aspose.Html.Accessibility.WebAccessibility().CreateValidator(); - - // Perform validation - ValidationResult validationResult = validator.Validate(document); - - // Save validation result as XML to a string writer - using (StringWriter sw = new StringWriter()) - { - validationResult.SaveTo(sw, ValidationResultSaveFormat.XML); - // Append the XML result to the log file - File.AppendAllText(logPath, sw.ToString() + Environment.NewLine); - } - } - - /// - /// Extracts the 'src' attribute of all elements in the document using XPath and logs each source to the specified log file. - /// - /// The HTML document to process. - /// The path of the log file where image sources will be recorded. - public static void ExtractImageSources(Aspose.Html.HTMLDocument document, string logPath) - { - // Evaluate XPath to select all image elements - IXPathResult result = document.Evaluate("//img", document, document.CreateNSResolver(document), XPathResultType.Any, null); - - // Iterate over the result set - Node node; - while ((node = result.IterateNext()) != null) - { - // Cast the node to HTMLImageElement to access the Src property - Aspose.Html.HTMLImageElement img = (Aspose.Html.HTMLImageElement)node; - // Append the image source to the log file - File.AppendAllText(logPath, img.Src + Environment.NewLine); - } - } -} - -/// -/// Custom network message handler that captures response headers for later inspection. -/// -public class MyMessageHandler : Aspose.Html.Net.MessageHandler -{ - /// - /// List that stores captured header strings. - /// - public static List CapturedHeaders = new List(); - - /// - /// Invoked for each network operation; captures response headers after the operation proceeds. - /// - /// The network operation context. - public override void Invoke(Aspose.Html.Net.INetworkOperationContext context) - { - // Continue with the next handler in the pipeline - Next(context); - - // Capture all response headers - foreach (object headerItem in context.Response.Headers) - { - if (headerItem != null) - { - CapturedHeaders.Add(headerItem.ToString()); - } - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/download-icon-data-synchronously-using-webclient-for-each-resolved-icon-url-resource.cs b/extract-images-from-website/download-icon-data-synchronously-using-webclient-for-each-resolved-icon-url-resource.cs new file mode 100644 index 0000000..48ece5a --- /dev/null +++ b/extract-images-from-website/download-icon-data-synchronously-using-webclient-for-each-resolved-icon-url-resource.cs @@ -0,0 +1,44 @@ +// Download icon data synchronously using WebClient for each resolved icon URL resource. + +class Program +{ + static void Main(string[] args) + { + try + { + string html = ""; + string outputDir = "Icons"; + System.IO.Directory.CreateDirectory(outputDir); + using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(html, "about:blank")) + { + Aspose.Html.Collections.HTMLCollection images = document.GetElementsByTagName("img"); + for (int i = 0; i < images.Length; i++) + { + Aspose.Html.Dom.Element imgElement = (Aspose.Html.Dom.Element)images[i]; + string src = imgElement.GetAttribute("src"); + if (string.IsNullOrEmpty(src)) + continue; + Aspose.Html.Url imageUrl = new Aspose.Html.Url(src, document.BaseURI); + string urlString = imageUrl.ToString(); + string extension = System.IO.Path.GetExtension(urlString); + if (!extension.Equals(".png", System.StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".jpg", System.StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".jpeg", System.StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".gif", System.StringComparison.OrdinalIgnoreCase)) + continue; + using (System.Net.WebClient webClient = new System.Net.WebClient()) + { + byte[] imageBytes = webClient.DownloadData(urlString); + string fileName = System.IO.Path.GetFileName(urlString); + string savePath = System.IO.Path.Combine(outputDir, fileName); + System.IO.File.WriteAllBytes(savePath, imageBytes); + } + } + } + } + catch (System.Exception ex) + { + System.Console.WriteLine("Error: " + ex.Message); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/download_image_data_asynchronously_httpclient_each_resolved_image_url_resource.cs b/extract-images-from-website/download-image-data-asynchronously-httpclient-resolved-image-url-resource.cs similarity index 54% rename from extract-images-from-website/download_image_data_asynchronously_httpclient_each_resolved_image_url_resource.cs rename to extract-images-from-website/download-image-data-asynchronously-httpclient-resolved-image-url-resource.cs index b8b4c48..aba76da 100644 --- a/extract-images-from-website/download_image_data_asynchronously_httpclient_each_resolved_image_url_resource.cs +++ b/extract-images-from-website/download-image-data-asynchronously-httpclient-resolved-image-url-resource.cs @@ -15,9 +15,15 @@ static async Task Main(string[] args) { try { - string inputPath = "input.html"; - string outputPath = "output.html"; - string outputDir = "images"; + string inputPath = "sample.html"; + string outputDir = "downloaded_images"; + + // Create a minimal sample HTML file if it does not exist + if (!File.Exists(inputPath)) + { + string sampleHtml = ""; + File.WriteAllText(inputPath, sampleHtml); + } Directory.CreateDirectory(outputDir); @@ -29,40 +35,34 @@ static async Task Main(string[] args) { Element imgElement = (Element)images[i]; string src = imgElement.GetAttribute("src"); - if (string.IsNullOrWhiteSpace(src)) + if (string.IsNullOrEmpty(src)) continue; Url imageUrl = new Url(src, document.BaseURI); string urlString = imageUrl.ToString(); + string extension = Path.GetExtension(urlString); + if (!extension.Equals(".png", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".jpg", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".jpeg", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".gif", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".svg", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".webp", StringComparison.OrdinalIgnoreCase)) + { + continue; + } byte[] imageBytes = await httpClient.GetByteArrayAsync(urlString); string fileName = Path.GetFileName(urlString); string savePath = Path.Combine(outputDir, fileName); File.WriteAllBytes(savePath, imageBytes); - - imgElement.SetAttribute("src", Path.Combine(outputDir, fileName)); } - - document.Save(outputPath); } + + Console.WriteLine("Image download completed."); } catch (Exception ex) { Console.WriteLine("Error: " + ex.Message); } } - - static string GetMimeType(string extension) - { - switch (extension.ToLowerInvariant()) - { - case ".png": return "image/png"; - case ".jpg": - case ".jpeg": return "image/jpeg"; - case ".gif": return "image/gif"; - case ".svg": return "image/svg+xml"; - case ".webp": return "image/webp"; - default: return "application/octet-stream"; - } - } } \ No newline at end of file diff --git a/extract-images-from-website/download_icon_data_synchronously_using_webclient_for_each_resolved_icon_url_resource.cs b/extract-images-from-website/download_icon_data_synchronously_using_webclient_for_each_resolved_icon_url_resource.cs deleted file mode 100644 index 7ad7112..0000000 --- a/extract-images-from-website/download_icon_data_synchronously_using_webclient_for_each_resolved_icon_url_resource.cs +++ /dev/null @@ -1,53 +0,0 @@ -// Download icon data synchronously using WebClient for each resolved icon URL resource. - -using System; -using System.IO; -using System.Net; -using Aspose.Html; -using Aspose.Html.Collections; -using Aspose.Html.Dom; - -class Program -{ - static void Main(string[] args) - { - try - { - string htmlContent = "" + - "" + - "" + - ""; - using (HTMLDocument document = new HTMLDocument(htmlContent)) - { - string outputDir = "DownloadedIcons"; - Directory.CreateDirectory(outputDir); - - HTMLCollection images = document.GetElementsByTagName("img"); - for (int i = 0; i < images.Length; i++) - { - Element imgElement = (Element)images[i]; - string src = imgElement.GetAttribute("src"); - if (string.IsNullOrWhiteSpace(src)) - continue; - - Url imageUrl = new Url(src, document.BaseURI); - string urlString = imageUrl.ToString(); - - using (WebClient webClient = new WebClient()) - { - byte[] imageBytes = webClient.DownloadData(urlString); - string fileName = Path.GetFileName(urlString); - string savePath = Path.Combine(outputDir, fileName); - File.WriteAllBytes(savePath, imageBytes); - } - } - } - - Console.WriteLine("Icon download completed."); - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/extract-image-dimensions-from-html-attributes-store-alongside-file-metadata.cs b/extract-images-from-website/extract-image-dimensions-from-html-attributes-store-alongside-file-metadata.cs new file mode 100644 index 0000000..a132a8c --- /dev/null +++ b/extract-images-from-website/extract-image-dimensions-from-html-attributes-store-alongside-file-metadata.cs @@ -0,0 +1,72 @@ +// Extract image dimensions from HTML attributes and store them alongside file metadata. + +using System; +using System.IO; +using System.Collections.Generic; + +class Program +{ + static void Main() + { + try + { + // Define paths + string htmlPath = "sample.html"; + string outputCsvPath = "image_metadata.csv"; + + // Create a minimal HTML file if it does not exist + if (!File.Exists(htmlPath)) + { + string sampleHtml = @" + +Sample + + + + +"; + File.WriteAllText(htmlPath, sampleHtml); + } + + // Load the HTML document + var document = new Aspose.Html.HTMLDocument(htmlPath); + + // Query all img elements + var imgNodeList = document.QuerySelectorAll("img"); + + var records = new List(); + records.Add("Src,Width,Height"); + + for (int i = 0; i < imgNodeList.Length; i++) + { + var element = imgNodeList[i] as Aspose.Html.Dom.Element; + if (element == null) + continue; + + var img = element as Aspose.Html.HTMLImageElement; + if (img == null) + continue; + + string src = img.GetAttribute("src") ?? ""; + string widthAttr = img.GetAttribute("width") ?? "0"; + string heightAttr = img.GetAttribute("height") ?? "0"; + + int width = 0; + int height = 0; + Int32.TryParse(widthAttr, out width); + Int32.TryParse(heightAttr, out height); + + records.Add($"{src},{width},{height}"); + } + + // Write metadata to CSV file + File.WriteAllLines(outputCsvPath, records); + + Console.WriteLine($"Image metadata extracted to '{outputCsvPath}'."); + } + catch (Exception ex) + { + Console.WriteLine(ex.Message); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/filter_extracted_images_by_file_extension_downloading_png_and_jpeg.cs b/extract-images-from-website/filter-extracted-images-by-file-extension-downloading-only-png-and-jpeg-formats.cs similarity index 54% rename from extract-images-from-website/filter_extracted_images_by_file_extension_downloading_png_and_jpeg.cs rename to extract-images-from-website/filter-extracted-images-by-file-extension-downloading-only-png-and-jpeg-formats.cs index 15c9fef..adf97e2 100644 --- a/extract-images-from-website/filter_extracted_images_by_file_extension_downloading_png_and_jpeg.cs +++ b/extract-images-from-website/filter-extracted-images-by-file-extension-downloading-only-png-and-jpeg-formats.cs @@ -1,32 +1,28 @@ // Filter extracted images by file extension, downloading only PNG and JPEG formats. using System; -using System.IO; using System.Net.Http; +using System.IO; +using Aspose.Html; +using Aspose.Html.Collections; +using Aspose.Html.Dom; +using Aspose.Html.Net; class Program { - static void Main() + static void Main(string[] args) { try { - // Sample HTML content with image references - string htmlContent = @" - - - - - - - "; - - // Create HTML document - Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(htmlContent, "."); + string htmlContent = "" + + "" + + "" + + "" + + ""; - // Get all elements - Aspose.Html.Collections.HTMLCollection images = document.GetElementsByTagName("img"); + HTMLDocument document = new HTMLDocument(htmlContent, "about:blank"); + HTMLCollection images = document.GetElementsByTagName("img"); - // Output directory string outputDir = "ExtractedImages"; Directory.CreateDirectory(outputDir); @@ -34,29 +30,21 @@ static void Main() { for (int i = 0; i < images.Length; i++) { - Aspose.Html.Dom.Element imgElement = (Aspose.Html.Dom.Element)images[i]; + Element imgElement = (Element)images[i]; string src = imgElement.GetAttribute("src"); if (string.IsNullOrEmpty(src)) continue; - // Resolve absolute URL - Uri baseUri = new Uri(document.BaseURI ?? "http://localhost/"); - Uri imageUri = new Uri(baseUri, src); - string urlString = imageUri.ToString(); - - // Filter by extension + Url imageUrl = new Url(src, document.BaseURI); + string urlString = imageUrl.ToString(); string extension = Path.GetExtension(urlString); + if (!extension.Equals(".png", StringComparison.OrdinalIgnoreCase) && !extension.Equals(".jpg", StringComparison.OrdinalIgnoreCase) && !extension.Equals(".jpeg", StringComparison.OrdinalIgnoreCase)) - { continue; - } - // Download image bytes byte[] imageBytes = httpClient.GetByteArrayAsync(urlString).GetAwaiter().GetResult(); - - // Save to file string fileName = Path.GetFileName(urlString); string savePath = Path.Combine(outputDir, fileName); File.WriteAllBytes(savePath, imageBytes); diff --git a/extract-images-from-website/implement_configurable_timeout_for_httpclient_requests_to_avoid_hanging_on_slow_resources.cs b/extract-images-from-website/implement-configurable-timeout-httpclient-requests-avoid-hanging-slow-resources.cs similarity index 93% rename from extract-images-from-website/implement_configurable_timeout_for_httpclient_requests_to_avoid_hanging_on_slow_resources.cs rename to extract-images-from-website/implement-configurable-timeout-httpclient-requests-avoid-hanging-slow-resources.cs index c4c3a7c..75e5d70 100644 --- a/extract-images-from-website/implement_configurable_timeout_for_httpclient_requests_to_avoid_hanging_on_slow_resources.cs +++ b/extract-images-from-website/implement-configurable-timeout-httpclient-requests-avoid-hanging-slow-resources.cs @@ -10,7 +10,7 @@ static void Main() { string url = "https://example.com"; Aspose.Html.Net.RequestMessage request = new Aspose.Html.Net.RequestMessage(url); - request.Timeout = System.TimeSpan.FromSeconds(5); + request.Timeout = System.TimeSpan.FromSeconds(10); using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(request)) { string html = document.DocumentElement != null ? document.DocumentElement.OuterHTML : string.Empty; diff --git a/extract-images-from-website/implement-parallel-image-download-using-taskwhenall-improve-overall-extraction-performance.cs b/extract-images-from-website/implement-parallel-image-download-using-taskwhenall-improve-overall-extraction-performance.cs new file mode 100644 index 0000000..22f1d3f --- /dev/null +++ b/extract-images-from-website/implement-parallel-image-download-using-taskwhenall-improve-overall-extraction-performance.cs @@ -0,0 +1,102 @@ +// Implement parallel image download using Task.WhenAll to improve overall extraction performance. + +using System; +using System.IO; +using System.Net.Http; +using System.Threading.Tasks; +using System.Collections.Generic; +using Aspose.Html; +using Aspose.Html.Collections; +using Aspose.Html.Dom; +using Aspose.Html.Net; + +class Program +{ + static async Task Main(string[] args) + { + try + { + // Prepare sample HTML file with image tags + string htmlPath = "sample.html"; + if (!File.Exists(htmlPath)) + { + string sampleHtml = @" + + +Sample + + + + + + +"; + File.WriteAllText(htmlPath, sampleHtml); + } + + // Load HTML document + HTMLDocument document = new HTMLDocument(htmlPath); + + // Get all elements + HTMLCollection images = document.GetElementsByTagName("img"); + + // Output directory for downloaded images + string outputDir = "Images"; + Directory.CreateDirectory(outputDir); + + // Prepare HttpClient + using (HttpClient httpClient = new HttpClient()) + { + List downloadTasks = new List(); + + for (int i = 0; i < images.Length; i++) + { + Element imgElement = (Element)images[i]; + string src = imgElement.GetAttribute("src"); + if (string.IsNullOrEmpty(src)) + continue; + + // Resolve URL against document base URI + Url imageUrl = new Url(src, document.BaseURI); + string urlString = imageUrl.ToString(); + + string extension = Path.GetExtension(urlString); + if (!extension.Equals(".jpg", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".jpeg", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".png", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".gif", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".bmp", StringComparison.OrdinalIgnoreCase)) + continue; + + string fileName = Path.GetFileName(urlString); + string savePath = Path.Combine(outputDir, fileName); + + // Create a download task + Task downloadTask = httpClient.GetByteArrayAsync(urlString).ContinueWith(t => + { + if (t.Exception == null) + { + File.WriteAllBytes(savePath, t.Result); + Console.WriteLine($"Downloaded: {fileName}"); + } + else + { + Console.WriteLine($"Failed to download {urlString}: {t.Exception.GetBaseException().Message}"); + } + }); + + downloadTasks.Add(downloadTask); + } + + // Wait for all downloads to complete + await Task.WhenAll(downloadTasks); + } + + Console.WriteLine("Image extraction completed."); + } + catch (Exception ex) + { + Console.WriteLine($"Error: {ex.Message}"); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/implement_progress_reporting_callback_reports_number_of_images_downloaded_versus_total.cs b/extract-images-from-website/implement-progress-reporting-callback-reports-number-of-images-downloaded-versus-total.cs similarity index 61% rename from extract-images-from-website/implement_progress_reporting_callback_reports_number_of_images_downloaded_versus_total.cs rename to extract-images-from-website/implement-progress-reporting-callback-reports-number-of-images-downloaded-versus-total.cs index 8fd4bd4..8bb6e39 100644 --- a/extract-images-from-website/implement_progress_reporting_callback_reports_number_of_images_downloaded_versus_total.cs +++ b/extract-images-from-website/implement-progress-reporting-callback-reports-number-of-images-downloaded-versus-total.cs @@ -14,33 +14,25 @@ static void Main() { try { - // Prepare sample HTML file - string htmlPath = "sample.html"; - if (!File.Exists(htmlPath)) - { - string sampleHtml = @" - - -Sample - - - - -"; - File.WriteAllText(htmlPath, sampleHtml); - } - - // Output directory for downloaded images + string inputHtml = "sample.html"; string outputDir = "downloaded_images"; + if (!Directory.Exists(outputDir)) Directory.CreateDirectory(outputDir); - // Load HTML document - using (HTMLDocument document = new HTMLDocument(htmlPath)) + if (!File.Exists(inputHtml)) + { + string htmlContent = "" + + "" + + "" + + ""; + File.WriteAllText(inputHtml, htmlContent); + } + + using (HTMLDocument document = new HTMLDocument(inputHtml)) { HTMLCollection images = document.GetElementsByTagName("img"); int total = images.Length; - int downloaded = 0; using (HttpClient httpClient = new HttpClient()) { @@ -51,29 +43,23 @@ static void Main() if (string.IsNullOrEmpty(src)) continue; - // Resolve relative URLs against the document base URI Url imageUrl = new Url(src, document.BaseURI); string urlString = imageUrl.ToString(); - // Download image bytes byte[] imageBytes = httpClient.GetByteArrayAsync(urlString).GetAwaiter().GetResult(); - // Save to file string fileName = Path.GetFileName(urlString); string savePath = Path.Combine(outputDir, fileName); File.WriteAllBytes(savePath, imageBytes); - downloaded++; - Console.WriteLine($"Downloaded {downloaded}/{total} images - {fileName}"); + Console.WriteLine($"Downloaded {i + 1}/{total}: {fileName}"); } } } - - Console.WriteLine("Image download completed."); } catch (Exception ex) { - Console.WriteLine($"Error: {ex.Message}"); + Console.WriteLine("Error: " + ex.Message); } } } \ No newline at end of file diff --git a/extract-images-from-website/implement_parallel_image_download_using_task_whenall_to_improve_extraction_performance.cs b/extract-images-from-website/implement_parallel_image_download_using_task_whenall_to_improve_extraction_performance.cs deleted file mode 100644 index d73cbdb..0000000 --- a/extract-images-from-website/implement_parallel_image_download_using_task_whenall_to_improve_extraction_performance.cs +++ /dev/null @@ -1,81 +0,0 @@ -// Implement parallel image download using Task.WhenAll to improve overall extraction performance. - -using System; -using System.IO; -using System.Collections.Generic; -using System.Threading.Tasks; -using System.Net.Http; -using Aspose.Html; -using Aspose.Html.Collections; -using Aspose.Html.Dom; -using Aspose.Html.Net; - -class Program -{ - static void Main() - { - try - { - // Prepare sample HTML file with image references - string inputHtmlPath = "sample.html"; - string htmlContent = "" + - "" + - "" + - ""; - File.WriteAllText(inputHtmlPath, htmlContent); - - // Load HTML document - HTMLDocument document = new HTMLDocument(inputHtmlPath); - - // Get all elements - HTMLCollection images = document.GetElementsByTagName("img"); - - // Output directory for downloaded images - string outputDir = "downloaded_images"; - Directory.CreateDirectory(outputDir); - - // HttpClient for downloading images - using (HttpClient httpClient = new HttpClient()) - { - List downloadTasks = new List(); - - for (int i = 0; i < images.Length; i++) - { - Element imgElement = (Element)images[i]; - string src = imgElement.GetAttribute("src"); - if (string.IsNullOrEmpty(src)) - continue; - - Url imageUrl = new Url(src, document.BaseURI); - string urlString = imageUrl.ToString(); - string extension = Path.GetExtension(urlString); - - if (!extension.Equals(".jpg", StringComparison.OrdinalIgnoreCase) && - !extension.Equals(".jpeg", StringComparison.OrdinalIgnoreCase) && - !extension.Equals(".png", StringComparison.OrdinalIgnoreCase)) - continue; - - // Create a task for each image download - Task downloadTask = Task.Run(async () => - { - byte[] imageBytes = await httpClient.GetByteArrayAsync(urlString); - string fileName = Path.GetFileName(urlString); - string savePath = Path.Combine(outputDir, fileName); - await File.WriteAllBytesAsync(savePath, imageBytes); - }); - - downloadTasks.Add(downloadTask); - } - - // Wait for all downloads to complete - Task.WhenAll(downloadTasks).GetAwaiter().GetResult(); - } - - Console.WriteLine("Image download completed."); - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/integrate-extraction-routine-aspnet-core-controller-endpoint-on-demand-usage.cs b/extract-images-from-website/integrate-extraction-routine-aspnet-core-controller-endpoint-on-demand-usage.cs new file mode 100644 index 0000000..3b87f49 --- /dev/null +++ b/extract-images-from-website/integrate-extraction-routine-aspnet-core-controller-endpoint-on-demand-usage.cs @@ -0,0 +1,41 @@ +// Integrate the extraction routine into an ASP.NET Core controller endpoint for on‑demand usage. + +using System; +using System.IO; + +class Program +{ + static void Main() + { + try + { + const string inputMhtmlPath = "sample.mhtml"; + const string outputDocxPath = "output.docx"; + + // Create a minimal MHTML file if it does not exist + if (!File.Exists(inputMhtmlPath)) + { + File.WriteAllText(inputMhtmlPath, string.Empty); + } + + using (Stream inputStream = File.OpenRead(inputMhtmlPath)) + { + byte[] docxBytes = ConvertMhtmlToDocxBytes(inputStream); + File.WriteAllBytes(outputDocxPath, docxBytes); + Console.WriteLine($"DOCX file saved to: {outputDocxPath}"); + } + } + catch (Exception ex) + { + Console.Error.WriteLine($"Error: {ex.Message}"); + } + } + + private static byte[] ConvertMhtmlToDocxBytes(Stream inputStream) + { + string tempDocxPath = Path.Combine(Path.GetTempPath(), Guid.NewGuid().ToString() + ".docx"); + var options = new Aspose.Html.Saving.DocSaveOptions(); + Aspose.Html.Converters.Converter.ConvertMHTML(inputStream, options, tempDocxPath); + return File.ReadAllBytes(tempDocxPath); + } +} \ No newline at end of file diff --git a/extract-images-from-website/integrate_extraction_routine_into_aspnet_core_controller_endpoint_for_on_demand_usage.cs b/extract-images-from-website/integrate_extraction_routine_into_aspnet_core_controller_endpoint_for_on_demand_usage.cs deleted file mode 100644 index 102a4c1..0000000 --- a/extract-images-from-website/integrate_extraction_routine_into_aspnet_core_controller_endpoint_for_on_demand_usage.cs +++ /dev/null @@ -1,49 +0,0 @@ -// Integrate the extraction routine into an ASP.NET Core controller endpoint for on‑demand usage. - -using System; -using System.IO; - -public class Program -{ - public static void Main() - { - try - { - string inputPath = "sample.mhtml"; - if (!File.Exists(inputPath)) - { - // Create a minimal placeholder MHTML file - File.WriteAllText(inputPath, string.Empty); - } - - using (FileStream inputStream = File.OpenRead(inputPath)) - { - byte[] docxBytes = ConvertMhtmlToDocxBytes(inputStream); - string outputPath = "output.docx"; - File.WriteAllBytes(outputPath, docxBytes); - Console.WriteLine("Conversion completed. Output saved to " + outputPath); - } - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } - - private static byte[] ConvertMhtmlToDocxBytes(Stream inputStream) - { - string tempDocxPath = Path.Combine(Path.GetTempPath(), Guid.NewGuid().ToString() + ".docx"); - Aspose.Html.Saving.DocSaveOptions options = new Aspose.Html.Saving.DocSaveOptions(); - Aspose.Html.Converters.Converter.ConvertMHTML(inputStream, options, tempDocxPath); - byte[] result = File.ReadAllBytes(tempDocxPath); - try - { - File.Delete(tempDocxPath); - } - catch - { - // Ignore any errors during cleanup - } - return result; - } -} \ No newline at end of file diff --git a/extract-images-from-website/log-detailed-extraction-steps-file-using-serilog-for-troubleshooting.cs b/extract-images-from-website/log-detailed-extraction-steps-file-using-serilog-for-troubleshooting.cs new file mode 100644 index 0000000..0b23240 --- /dev/null +++ b/extract-images-from-website/log-detailed-extraction-steps-file-using-serilog-for-troubleshooting.cs @@ -0,0 +1,92 @@ +// Log detailed extraction steps to a file using Serilog for troubleshooting purposes. + +using System; +using System.IO; + +class Program +{ + static void Main() + { + string logPath = "extraction.log"; + try + { + // Initialize log file + File.AppendAllText(logPath, "=== Extraction started at " + DateTime.Now + " ===" + Environment.NewLine); + + // Create a minimal HTML file with images + string htmlPath = "sample.html"; + string htmlContent = @" + +Sample + + + + +"; + File.WriteAllText(htmlPath, htmlContent); + File.AppendAllText(logPath, "Created sample HTML file: " + htmlPath + Environment.NewLine); + + // Configure Aspose.HTML with a custom network logger (optional for file loading) + Aspose.Html.Configuration configuration = new Aspose.Html.Configuration(); + Aspose.Html.Services.INetworkService networkService = configuration.GetService(); + networkService.MessageHandlers.Add(new TimeLoggerMessageHandler(logPath)); + File.AppendAllText(logPath, "Configured network service with custom logger." + Environment.NewLine); + + // Load the HTML document + Aspose.Html.HTMLDocument doc = new Aspose.Html.HTMLDocument(htmlPath, configuration); + File.AppendAllText(logPath, "Loaded HTML document from: " + htmlPath + Environment.NewLine); + + // Evaluate XPath to select all img elements + Aspose.Html.Dom.XPath.IXPathResult xpathResult = doc.Evaluate( + "//img", + doc, + doc.CreateNSResolver(doc), + Aspose.Html.Dom.XPath.XPathResultType.Any, + null); + File.AppendAllText(logPath, "Executed XPath query: //img" + Environment.NewLine); + + // Iterate over results and log each image source + Aspose.Html.Dom.Node node; + while ((node = xpathResult.IterateNext()) != null) + { + Aspose.Html.HTMLImageElement img = (Aspose.Html.HTMLImageElement)node; + File.AppendAllText(logPath, "Found image src: " + img.Src + Environment.NewLine); + } + + // Optionally save the document to a new file + string outputPath = "output.html"; + doc.Save(outputPath); + File.AppendAllText(logPath, "Saved document to: " + outputPath + Environment.NewLine); + + File.AppendAllText(logPath, "=== Extraction completed successfully ===" + Environment.NewLine); + } + catch (Exception ex) + { + File.AppendAllText(logPath, "Error: " + ex.Message + Environment.NewLine); + Console.WriteLine("An error occurred: " + ex.Message); + } + } +} + +// Custom message handler for detailed network logging +public sealed class TimeLoggerMessageHandler : Aspose.Html.Net.MessageHandler +{ + private readonly string _logFile; + public TimeLoggerMessageHandler(string logFilePath) + { + _logFile = logFilePath; + } + + public override void Invoke(Aspose.Html.Net.INetworkOperationContext context) + { + using (var writer = new StreamWriter(_logFile, true)) + { + writer.WriteLine("Request URI: " + context.Request.RequestUri); + writer.WriteLine("Request Headers: " + Convert.ToString(context.Request.Headers)); + Next(context); + writer.WriteLine("Response Status: " + context.Response.StatusCode); + writer.WriteLine("Response Headers: " + Convert.ToString(context.Response.Headers)); + writer.WriteLine(new string('-', 40)); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/log_detailed_extraction_steps_to_file_using_serilog_for_troubleshooting.cs b/extract-images-from-website/log_detailed_extraction_steps_to_file_using_serilog_for_troubleshooting.cs deleted file mode 100644 index ebe0314..0000000 --- a/extract-images-from-website/log_detailed_extraction_steps_to_file_using_serilog_for_troubleshooting.cs +++ /dev/null @@ -1,48 +0,0 @@ -// Log detailed extraction steps to a file using Serilog for troubleshooting purposes. - -using System; -using System.IO; - -class Program -{ - static void Main() - { - try - { - // Define log file path - string logPath = "extraction.log"; - File.AppendAllText(logPath, "=== Extraction started ===" + Environment.NewLine); - - // Create a minimal HTML file with images - string htmlPath = "sample.html"; - string htmlContent = @"" + - "" + - "" + - ""; - File.WriteAllText(htmlPath, htmlContent); - File.AppendAllText(logPath, "Created sample HTML file: " + htmlPath + Environment.NewLine); - - // Load the HTML document - Aspose.Html.HTMLDocument doc = new Aspose.Html.HTMLDocument(htmlPath); - File.AppendAllText(logPath, "Loaded HTML document." + Environment.NewLine); - - // Evaluate XPath to select all img elements - Aspose.Html.Dom.XPath.IXPathResult result = doc.Evaluate("//img", doc, doc.CreateNSResolver(doc), Aspose.Html.Dom.XPath.XPathResultType.Any, null); - File.AppendAllText(logPath, "Evaluated XPath '//img'." + Environment.NewLine); - - // Iterate over the result nodes and log each image src - Aspose.Html.Dom.Node node; - while ((node = result.IterateNext()) != null) - { - Aspose.Html.HTMLImageElement img = (Aspose.Html.HTMLImageElement)node; - File.AppendAllText(logPath, "Found image src: " + img.Src + Environment.NewLine); - } - - File.AppendAllText(logPath, "=== Extraction completed ===" + Environment.NewLine); - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/parse-remote-web-page-url-retrieve-all-img-elements-using-htmldocument.cs b/extract-images-from-website/parse-remote-web-page-url-retrieve-all-img-elements-using-htmldocument.cs new file mode 100644 index 0000000..411228b --- /dev/null +++ b/extract-images-from-website/parse-remote-web-page-url-retrieve-all-img-elements-using-htmldocument.cs @@ -0,0 +1,27 @@ +// Parse a remote web page URL and retrieve all elements using HtmlDocument. + +public class Program +{ + public static void Main(string[] args) + { + try + { + string url = "https://example.com"; + Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(url); + Aspose.Html.Collections.HTMLCollection images = document.GetElementsByTagName("img"); + for (int i = 0; i < images.Length; i++) + { + Aspose.Html.Dom.Element imgElement = (Aspose.Html.Dom.Element)images[i]; + string src = imgElement.GetAttribute("src"); + if (!string.IsNullOrEmpty(src)) + { + System.Console.WriteLine(src); + } + } + } + catch (System.Exception ex) + { + System.Console.WriteLine("Error: " + ex.Message); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/provide-option-overwrite-existing-files-skip-them-based-on-user-defined-flag.cs b/extract-images-from-website/provide-option-overwrite-existing-files-skip-them-based-on-user-defined-flag.cs new file mode 100644 index 0000000..7affea9 --- /dev/null +++ b/extract-images-from-website/provide-option-overwrite-existing-files-skip-them-based-on-user-defined-flag.cs @@ -0,0 +1,57 @@ +// Provide an option to overwrite existing files or skip them based on a user‑defined flag. + +using System; + +class Program +{ + static void Main() + { + try + { + string inputFolder = "InputHtml"; + string outputFolder = "OutputPdf"; + + System.IO.Directory.CreateDirectory(inputFolder); + System.IO.Directory.CreateDirectory(outputFolder); + + // Create a sample HTML file if none exists + string sampleHtmlPath = System.IO.Path.Combine(inputFolder, "sample.html"); + if (!System.IO.File.Exists(sampleHtmlPath)) + { + string htmlContent = "

Hello Aspose.HTML

"; + System.IO.File.WriteAllText(sampleHtmlPath, htmlContent); + } + + // User‑defined flag: set to true to overwrite existing PDFs, false to skip them + bool overwriteExisting = false; + + string[] htmlFiles = System.IO.Directory.GetFiles(inputFolder, "*.html"); + foreach (string htmlPath in htmlFiles) + { + string fileNameWithoutExt = System.IO.Path.GetFileNameWithoutExtension(htmlPath); + string outputPath = System.IO.Path.Combine(outputFolder, fileNameWithoutExt + ".pdf"); + + if (System.IO.File.Exists(outputPath) && !overwriteExisting) + { + System.Console.WriteLine($"Skipping existing file: {outputPath}"); + continue; + } + + // Load the HTML document from file + var document = new Aspose.Html.HTMLDocument(htmlPath); + + // Configure PDF save options (default options are sufficient for this example) + var pdfOptions = new Aspose.Html.Saving.PdfSaveOptions(); + + // Perform the conversion + Aspose.Html.Converters.Converter.ConvertHTML(document, pdfOptions, outputPath); + + System.Console.WriteLine($"Converted: {htmlPath} -> {outputPath}"); + } + } + catch (System.Exception ex) + { + System.Console.WriteLine($"Error: {ex.Message}"); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/provide_option_overwrite_existing_files_skip_based_on_user_defined_flag.cs b/extract-images-from-website/provide_option_overwrite_existing_files_skip_based_on_user_defined_flag.cs deleted file mode 100644 index 115c5b4..0000000 --- a/extract-images-from-website/provide_option_overwrite_existing_files_skip_based_on_user_defined_flag.cs +++ /dev/null @@ -1,55 +0,0 @@ -// Provide an option to overwrite existing files or skip them based on a user‑defined flag. - -using System; -using System.IO; - -class Program -{ - static void Main() - { - try - { - string inputFolder = "InputHtml"; - bool overwrite = true; // Set to false to skip existing PDF files - - if (!Directory.Exists(inputFolder)) - { - Directory.CreateDirectory(inputFolder); - string samplePath = Path.Combine(inputFolder, "sample.html"); - File.WriteAllText(samplePath, "

Hello Aspose

"); - } - - ConvertHtmlFilesInFolder(inputFolder, overwrite); - } - catch (Exception ex) - { - Console.WriteLine($"Error: {ex.Message}"); - } - } - - static void ConvertHtmlFilesInFolder(string folderPath, bool overwrite) - { - string[] htmlFiles = Directory.GetFiles(folderPath, "*.html"); - foreach (string htmlPath in htmlFiles) - { - string outputPath = Path.ChangeExtension(htmlPath, ".pdf"); - - if (File.Exists(outputPath) && !overwrite) - { - Console.WriteLine($"Skipping existing file: {outputPath}"); - continue; - } - - Aspose.Html.Configuration config = new Aspose.Html.Configuration(); - config.Security |= Aspose.Html.Sandbox.Scripts; - - Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(htmlPath, config); - - Aspose.Html.Saving.PdfSaveOptions options = new Aspose.Html.Saving.PdfSaveOptions(); - - Aspose.Html.Converters.Converter.ConvertHTML(document, options, outputPath); - - Console.WriteLine($"Converted: {outputPath}"); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/register_html_document_and_http_client_services_dependency_injection_testability.cs b/extract-images-from-website/register-htmldocument-and-httpclient-services-via-dependency-injection-for-testability.cs similarity index 57% rename from extract-images-from-website/register_html_document_and_http_client_services_dependency_injection_testability.cs rename to extract-images-from-website/register-htmldocument-and-httpclient-services-via-dependency-injection-for-testability.cs index 04dbb37..10ec96f 100644 --- a/extract-images-from-website/register_html_document_and_http_client_services_dependency_injection_testability.cs +++ b/extract-images-from-website/register-htmldocument-and-httpclient-services-via-dependency-injection-for-testability.cs @@ -1,41 +1,32 @@ // Register HtmlDocument and HttpClient services via dependency injection for testability. using System; -using System.Net.Http; class Program { + static Aspose.Html.HTMLDocument CreateDocument(string htmlContent, string baseUrl) + { + return new Aspose.Html.HTMLDocument(htmlContent, baseUrl); + } + + static string LoadHtml() + { + return "

Hello, Aspose!

"; + } + static void Main() { try { - var httpClient = new HttpClient(); - httpClient.BaseAddress = new Uri("http://example.com"); - - string htmlContent = LoadHtml(); - - using (Aspose.Html.HTMLDocument document = CreateDocument(htmlContent, "http://example.com")) + using (Aspose.Html.HTMLDocument document = CreateDocument(LoadHtml(), "about:blank")) { string outerHtml = ((Aspose.Html.HTMLElement)document.DocumentElement).OuterHTML; - Console.WriteLine("Outer HTML:"); Console.WriteLine(outerHtml); } - - Console.WriteLine("HttpClient BaseAddress: " + (httpClient.BaseAddress?.ToString() ?? "null")); } catch (Exception ex) { - Console.WriteLine("Error: " + ex.Message); + Console.WriteLine($"Error: {ex.Message}"); } } - - static Aspose.Html.HTMLDocument CreateDocument(string htmlContent, string baseUrl) - { - return new Aspose.Html.HTMLDocument(htmlContent, baseUrl); - } - - static string LoadHtml() - { - return "

Hello World

"; - } } \ No newline at end of file diff --git a/extract-images-from-website/resolve-each-image-src-attribute-to-absolute-url-using-url-class-and-document-baseuri.cs b/extract-images-from-website/resolve-each-image-src-attribute-to-absolute-url-using-url-class-and-document-baseuri.cs new file mode 100644 index 0000000..7df2eff --- /dev/null +++ b/extract-images-from-website/resolve-each-image-src-attribute-to-absolute-url-using-url-class-and-document-baseuri.cs @@ -0,0 +1,48 @@ +// Resolve each image src attribute to an absolute URL using Url class and document BaseURI. + +using System; +using Aspose.Html; +using Aspose.Html.Collections; +using Aspose.Html.Dom; + +class Program +{ + static void Main() + { + try + { + // Sample HTML with relative image sources + string htmlContent = @" + + + + + + + + "; + + // Load the document with a base URI (about:blank is fine when a tag is present) + HTMLDocument document = new HTMLDocument(htmlContent, "about:blank"); + + // Get all elements + HTMLCollection images = document.GetElementsByTagName("img"); + + for (int i = 0; i < images.Length; i++) + { + Element imageElement = (Element)images[i]; + string src = imageElement.GetAttribute("src"); + if (string.IsNullOrEmpty(src)) + continue; + + // Resolve to absolute URL using document's BaseURI + Url absoluteUrl = new Url(src, document.BaseURI); + Console.WriteLine($"Image {i + 1}: {absoluteUrl}"); + } + } + catch (Exception ex) + { + Console.WriteLine($"Error: {ex.Message}"); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/resolve-icon-href-attribute-to-absolute-url-using-url-class-and-document-baseuri.cs b/extract-images-from-website/resolve-icon-href-attribute-to-absolute-url-using-url-class-and-document-baseuri.cs new file mode 100644 index 0000000..3f9968e --- /dev/null +++ b/extract-images-from-website/resolve-icon-href-attribute-to-absolute-url-using-url-class-and-document-baseuri.cs @@ -0,0 +1,39 @@ +// Resolve each icon href attribute to an absolute URL using Url class and document BaseURI. + +using System; +using Aspose.Html; +using Aspose.Html.Collections; +using Aspose.Html.Dom; + +class Program +{ + static void Main() + { + try + { + string htmlContent = ""; + // Load HTML with a base URI for relative URL resolution + HTMLDocument document = new HTMLDocument(htmlContent, "http://example.com/"); + + HTMLCollection links = document.GetElementsByTagName("link"); + for (int i = 0; i < links.Length; i++) + { + Element linkElement = (Element)links[i]; + string rel = linkElement.GetAttribute("rel"); + if (string.IsNullOrEmpty(rel) || !rel.Equals("icon", StringComparison.OrdinalIgnoreCase)) + continue; + + string href = linkElement.GetAttribute("href"); + if (string.IsNullOrEmpty(href)) + continue; + + Url absoluteUrl = new Url(href, document.BaseURI); + Console.WriteLine(absoluteUrl); + } + } + catch (Exception ex) + { + Console.WriteLine("Error: " + ex.Message); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/resolve_each_image_src_attribute_to_absolute_url_using_url_class_and_document_baseuri.cs b/extract-images-from-website/resolve_each_image_src_attribute_to_absolute_url_using_url_class_and_document_baseuri.cs deleted file mode 100644 index cf6c0b5..0000000 --- a/extract-images-from-website/resolve_each_image_src_attribute_to_absolute_url_using_url_class_and_document_baseuri.cs +++ /dev/null @@ -1,29 +0,0 @@ -// Resolve each image src attribute to an absolute URL using Url class and document BaseURI. - -using System; - -class Program -{ - static void Main() - { - try - { - string html = ""; - Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(html); - Aspose.Html.Collections.HTMLCollection images = document.GetElementsByTagName("img"); - for (int i = 0; i < images.Length; i++) - { - Aspose.Html.Dom.Element imageElement = (Aspose.Html.Dom.Element)images[i]; - string src = imageElement.GetAttribute("src"); - if (string.IsNullOrEmpty(src)) - continue; - Aspose.Html.Url absoluteUrl = new Aspose.Html.Url(src, document.BaseURI); - Console.WriteLine(absoluteUrl.ToString()); - } - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/resolve_icon_href_attribute_to_absolute_url_using_url_class_and_document_baseuri.cs b/extract-images-from-website/resolve_icon_href_attribute_to_absolute_url_using_url_class_and_document_baseuri.cs deleted file mode 100644 index 9553b28..0000000 --- a/extract-images-from-website/resolve_icon_href_attribute_to_absolute_url_using_url_class_and_document_baseuri.cs +++ /dev/null @@ -1,37 +0,0 @@ -// Resolve each icon href attribute to an absolute URL using Url class and document BaseURI. - -using System; - -class Program -{ - static void Main() - { - try - { - // Sample HTML with icon link - string html = ""; - - // Load HTML document - Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(html); - - // Get all elements - Aspose.Html.Collections.HTMLCollection links = document.GetElementsByTagName("link"); - - // Iterate and resolve href attributes - for (int i = 0; i < links.Length; i++) - { - Aspose.Html.Dom.Element linkElement = (Aspose.Html.Dom.Element)links[i]; - string href = linkElement.GetAttribute("href"); - if (string.IsNullOrEmpty(href)) - continue; - - Aspose.Html.Url absoluteUrl = new Aspose.Html.Url(href, document.BaseURI); - Console.WriteLine(absoluteUrl.ToString()); - } - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/save-downloaded-icons-to-dedicated-icons-directory-using-custom-naming-convention.cs b/extract-images-from-website/save-downloaded-icons-to-dedicated-icons-directory-using-custom-naming-convention.cs new file mode 100644 index 0000000..1030c9e --- /dev/null +++ b/extract-images-from-website/save-downloaded-icons-to-dedicated-icons-directory-using-custom-naming-convention.cs @@ -0,0 +1,51 @@ +// Save downloaded icons to a dedicated icons directory using a custom naming convention. + +using System; +using System.IO; +using Aspose.Html.Dom.Svg; +using Aspose.Html.Saving.ResourceHandlers; + +class Program +{ + static void Main() + { + try + { + // Create output directory for icons + string iconsDirectory = System.IO.Path.Combine(System.IO.Directory.GetCurrentDirectory(), "icons"); + System.IO.Directory.CreateDirectory(iconsDirectory); + + // Prepare a simple SVG that references an external icon + string svgContent = "" + + "" + + ""; + + // Save SVG content to a temporary file + string svgFilePath = System.IO.Path.Combine(System.IO.Directory.GetCurrentDirectory(), "sample.svg"); + System.IO.File.WriteAllText(svgFilePath, svgContent); + + // Load SVG document and save external resources (icons) to the icons directory + using (Aspose.Html.Dom.Svg.SVGDocument doc = new Aspose.Html.Dom.Svg.SVGDocument(svgFilePath)) + { + doc.Save(new Aspose.Html.Saving.ResourceHandlers.FileSystemResourceHandler(iconsDirectory)); + } + + // Apply custom naming convention to downloaded icons + string[] files = System.IO.Directory.GetFiles(iconsDirectory); + int index = 1; + foreach (string filePath in files) + { + string extension = System.IO.Path.GetExtension(filePath); + string newFileName = System.IO.Path.Combine(iconsDirectory, $"custom_icon_{index}{extension}"); + System.IO.File.Move(filePath, newFileName); + index++; + } + + System.Console.WriteLine("Icons have been saved to: " + iconsDirectory); + } + catch (Exception ex) + { + System.Console.WriteLine("Error: " + ex.Message); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/save-downloaded-images-specified-local-folder-preserving-original-file-names.cs b/extract-images-from-website/save-downloaded-images-specified-local-folder-preserving-original-file-names.cs new file mode 100644 index 0000000..f35aecc --- /dev/null +++ b/extract-images-from-website/save-downloaded-images-specified-local-folder-preserving-original-file-names.cs @@ -0,0 +1,64 @@ +// Save downloaded images to a specified local folder preserving original file names. + +using System; +using System.IO; +using System.Net.Http; +using Aspose.Html; +using Aspose.Html.Collections; +using Aspose.Html.Dom; + +class Program +{ + static void Main(string[] args) + { + try + { + // Prepare input HTML file + string inputHtmlPath = Path.Combine(Directory.GetCurrentDirectory(), "sample.html"); + if (!File.Exists(inputHtmlPath)) + { + string htmlContent = "" + + "" + + "" + + ""; + File.WriteAllText(inputHtmlPath, htmlContent); + } + + // Prepare output directory + string outputDir = Path.Combine(Directory.GetCurrentDirectory(), "DownloadedImages"); + Directory.CreateDirectory(outputDir); + + // Load HTML document + using (HTMLDocument document = new HTMLDocument(inputHtmlPath)) + using (HttpClient httpClient = new HttpClient()) + { + HTMLCollection images = document.GetElementsByTagName("img"); + for (int i = 0; i < images.Length; i++) + { + Element imageElement = (Element)images[i]; + string src = imageElement.GetAttribute("src"); + if (string.IsNullOrWhiteSpace(src)) + continue; + + Uri baseUri = new Uri(document.BaseURI.ToString(), UriKind.Absolute); + Uri resolvedUri = new Uri(baseUri, src); + + byte[] imageBytes = httpClient.GetByteArrayAsync(resolvedUri).GetAwaiter().GetResult(); + + string fileName = Path.GetFileName(resolvedUri.LocalPath); + if (string.IsNullOrEmpty(fileName)) + fileName = Guid.NewGuid().ToString(); + + string savePath = Path.Combine(outputDir, fileName); + File.WriteAllBytes(savePath, imageBytes); + } + } + + Console.WriteLine("Images have been downloaded to: " + outputDir); + } + catch (Exception ex) + { + Console.WriteLine("Error: " + ex.Message); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/save_downloaded_icons_to_dedicated_icons_directory_using_custom_naming_convention.cs b/extract-images-from-website/save_downloaded_icons_to_dedicated_icons_directory_using_custom_naming_convention.cs deleted file mode 100644 index ad4bff2..0000000 --- a/extract-images-from-website/save_downloaded_icons_to_dedicated_icons_directory_using_custom_naming_convention.cs +++ /dev/null @@ -1,52 +0,0 @@ -// Save downloaded icons to a dedicated icons directory using a custom naming convention. - -using System; -using System.IO; -using System.Collections.Generic; -using Aspose.Html.Saving; -using Aspose.Html.Converters; - -class Program -{ - static void Main() - { - try - { - // Define the icons output directory - string iconsDirectory = Path.Combine(Directory.GetCurrentDirectory(), "icons"); - Directory.CreateDirectory(iconsDirectory); - - // Sample SVG icon contents - List svgContents = new List - { - @"", - @"" - }; - - for (int i = 0; i < svgContents.Count; i++) - { - // Create a temporary SVG file - string tempSvgPath = Path.Combine(Path.GetTempPath(), $"icon_{i}.svg"); - File.WriteAllText(tempSvgPath, svgContents[i]); - - // Define output PNG path with custom naming convention - string outputPngPath = Path.Combine(iconsDirectory, $"custom_icon_{i + 1}.png"); - - // Set image save options - ImageSaveOptions options = new ImageSaveOptions(); - - // Convert SVG to PNG - Aspose.Html.Converters.Converter.ConvertSVG(tempSvgPath, "", options, outputPngPath); - - // Clean up temporary SVG file - File.Delete(tempSvgPath); - } - - Console.WriteLine("Icons have been saved to: " + iconsDirectory); - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/skip-downloading-icons-larger-than-specified-byte-size-threshold-conserve-bandwidth.cs b/extract-images-from-website/skip-downloading-icons-larger-than-specified-byte-size-threshold-conserve-bandwidth.cs new file mode 100644 index 0000000..8e6e449 --- /dev/null +++ b/extract-images-from-website/skip-downloading-icons-larger-than-specified-byte-size-threshold-conserve-bandwidth.cs @@ -0,0 +1,85 @@ +// Skip downloading icons larger than a specified byte size threshold to conserve bandwidth. + +using System; +using System.IO; +using System.Net.Http; +using Aspose.Html; +using Aspose.Html.Collections; +using Aspose.Html.Dom; + +class Program +{ + static void Main() + { + try + { + // Define sample HTML with image references + string htmlContent = "" + + "" + + "" + + ""; + + // Prepare input and output paths + string inputHtmlPath = Path.Combine(Directory.GetCurrentDirectory(), "sample.html"); + string outputDir = Path.Combine(Directory.GetCurrentDirectory(), "downloaded_images"); + Directory.CreateDirectory(outputDir); + + // Write sample HTML to file + File.WriteAllText(inputHtmlPath, htmlContent); + + // Load the HTML document + using (HTMLDocument document = new HTMLDocument(inputHtmlPath)) + { + // Get all elements + HTMLCollection images = document.GetElementsByTagName("img"); + + // Size threshold in bytes (e.g., 100 KB) + const long maxSizeBytes = 100 * 1024; + + using (HttpClient httpClient = new HttpClient()) + { + for (int i = 0; i < images.Length; i++) + { + Element imgElement = (Element)images[i]; + string src = imgElement.GetAttribute("src"); + if (string.IsNullOrEmpty(src)) + continue; + + // Resolve the image URL relative to the document base URI + Url imageUrl = new Url(src, document.BaseURI); + string urlString = imageUrl.ToString(); + + // Perform a HEAD request to check Content-Length + using (HttpRequestMessage headRequest = new HttpRequestMessage(HttpMethod.Head, urlString)) + using (HttpResponseMessage headResponse = httpClient.Send(headRequest)) + { + if (!headResponse.IsSuccessStatusCode) + continue; + + long contentLength = headResponse.Content.Headers.ContentLength ?? -1; + if (contentLength > maxSizeBytes && contentLength != -1) + { + Console.WriteLine($"Skipping large image ({contentLength} bytes): {urlString}"); + continue; + } + } + + // Download the image + byte[] imageBytes = httpClient.GetByteArrayAsync(urlString).GetAwaiter().GetResult(); + string fileName = Path.GetFileName(new Uri(urlString).LocalPath); + if (string.IsNullOrEmpty(fileName)) + fileName = Guid.NewGuid().ToString() + ".img"; + + string savePath = Path.Combine(outputDir, fileName); + File.WriteAllBytes(savePath, imageBytes); + Console.WriteLine($"Saved image: {savePath}"); + } + } + } + } + catch (Exception ex) + { + Console.WriteLine($"Error: {ex.Message}"); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/skip_downloading_icons_larger_than_specified_byte_size_threshold_conserve_bandwidth.cs b/extract-images-from-website/skip_downloading_icons_larger_than_specified_byte_size_threshold_conserve_bandwidth.cs deleted file mode 100644 index 5eb043e..0000000 --- a/extract-images-from-website/skip_downloading_icons_larger_than_specified_byte_size_threshold_conserve_bandwidth.cs +++ /dev/null @@ -1,90 +0,0 @@ -// Skip downloading icons larger than a specified byte size threshold to conserve bandwidth. - -using System; -using System.IO; -using System.Net.Http; -using Aspose.Html; -using Aspose.Html.Collections; -using Aspose.Html.Dom; - -class Program -{ - static void Main() - { - try - { - // Define input HTML file and output directory - string htmlPath = "sample.html"; - string outputDir = "downloaded_icons"; - const long maxSizeBytes = 100 * 1024; // 100 KB threshold - - // Create a minimal sample HTML file if it does not exist - if (!File.Exists(htmlPath)) - { - string sampleHtml = @" - - -"; - File.WriteAllText(htmlPath, sampleHtml); - } - - // Load the HTML document - HTMLDocument document = new HTMLDocument(htmlPath); - - // Get all elements - HTMLCollection images = document.GetElementsByTagName("img"); - - // Ensure output directory exists - Directory.CreateDirectory(outputDir); - - using (HttpClient httpClient = new HttpClient()) - { - for (int i = 0; i < images.Length; i++) - { - Element imgElement = (Element)images[i]; - string src = imgElement.GetAttribute("src"); - if (string.IsNullOrEmpty(src)) - continue; - - // Resolve relative URLs against the document base URI - Url imageUrl = new Url(src, document.BaseURI); - string urlString = imageUrl.ToString(); - - // Filter supported image extensions - string extension = Path.GetExtension(urlString); - if (!extension.Equals(".png", StringComparison.OrdinalIgnoreCase) && - !extension.Equals(".jpg", StringComparison.OrdinalIgnoreCase) && - !extension.Equals(".jpeg", StringComparison.OrdinalIgnoreCase) && - !extension.Equals(".ico", StringComparison.OrdinalIgnoreCase) && - !extension.Equals(".svg", StringComparison.OrdinalIgnoreCase)) - { - continue; - } - - // Download the image bytes - byte[] imageBytes = httpClient.GetByteArrayAsync(urlString).GetAwaiter().GetResult(); - - // Skip saving if the image exceeds the size threshold - if (imageBytes.Length > maxSizeBytes) - { - Console.WriteLine($"Skipped downloading large icon: {urlString} ({imageBytes.Length} bytes)"); - continue; - } - - // Save the image to the output directory - string fileName = Path.GetFileName(urlString); - string savePath = Path.Combine(outputDir, fileName); - File.WriteAllBytes(savePath, imageBytes); - Console.WriteLine($"Downloaded and saved: {savePath} ({imageBytes.Length} bytes)"); - } - } - - // Cleanup - document.Dispose(); - } - catch (Exception ex) - { - Console.WriteLine($"Error: {ex.Message}"); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/store-extracted-images-memory-streams-further-processing-writing-to-disk.cs b/extract-images-from-website/store-extracted-images-memory-streams-further-processing-writing-to-disk.cs new file mode 100644 index 0000000..16a9ea8 --- /dev/null +++ b/extract-images-from-website/store-extracted-images-memory-streams-further-processing-writing-to-disk.cs @@ -0,0 +1,76 @@ +// Store extracted images in memory streams for further processing before writing to disk. + +using System; + +class MemoryStreamProvider : Aspose.Html.IO.ICreateStreamProvider, System.IDisposable +{ + public System.Collections.Generic.List Streams { get; } = new System.Collections.Generic.List(); + + public System.IO.Stream GetStream(string name, string extension) + { + var ms = new System.IO.MemoryStream(); + Streams.Add(ms); + return ms; + } + + public System.IO.Stream GetStream(string name, string extension, int page) + { + var ms = new System.IO.MemoryStream(); + Streams.Add(ms); + return ms; + } + + public void ReleaseStream(System.IO.Stream stream) + { + if (stream != null) + { + stream.Flush(); + } + } + + public void Dispose() + { + foreach (var ms in Streams) + { + ms.Dispose(); + } + } +} + +class Program +{ + static void Main() + { + try + { + string inputPath = "sample.epub"; + string outputDir = "output_images"; + + System.IO.Directory.CreateDirectory(outputDir); + + using (System.IO.Stream epubStream = System.IO.File.OpenRead(inputPath)) + { + var options = new Aspose.Html.Saving.ImageSaveOptions(Aspose.Html.Rendering.Image.ImageFormat.Png); + var provider = new MemoryStreamProvider(); + + Aspose.Html.Converters.Converter.ConvertEPUB(epubStream, options, provider); + + int index = 0; + foreach (var memoryStream in provider.Streams) + { + memoryStream.Position = 0; + string outputPath = System.IO.Path.Combine(outputDir, $"image_{index}.png"); + using (System.IO.FileStream fileStream = new System.IO.FileStream(outputPath, System.IO.FileMode.Create, System.IO.FileAccess.Write)) + { + memoryStream.CopyTo(fileStream); + } + index++; + } + } + } + catch (System.Exception ex) + { + System.Console.WriteLine($"Error: {ex.Message}"); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/store_extracted_images_memory_streams_further_processing_writing_to_disk.cs b/extract-images-from-website/store_extracted_images_memory_streams_further_processing_writing_to_disk.cs deleted file mode 100644 index 25f3ba5..0000000 --- a/extract-images-from-website/store_extracted_images_memory_streams_further_processing_writing_to_disk.cs +++ /dev/null @@ -1,98 +0,0 @@ -// Store extracted images in memory streams for further processing before writing to disk. - -using System; -using System.Collections.Generic; -using System.IO; -using Aspose.Html.IO; -using Aspose.Html.Saving; -using Aspose.Html.Rendering.Image; -using Aspose.Html.Converters; - -class MemoryStreamProvider : Aspose.Html.IO.ICreateStreamProvider, IDisposable -{ - public List Streams { get; } = new List(); - - public Stream GetStream(string name, string extension) - { - MemoryStream ms = new MemoryStream(); - Streams.Add(ms); - return ms; - } - - public Stream GetStream(string name, string extension, int page) - { - MemoryStream ms = new MemoryStream(); - Streams.Add(ms); - return ms; - } - - public void ReleaseStream(Stream stream) - { - if (stream != null) - { - stream.Flush(); - } - } - - public void Dispose() - { - foreach (MemoryStream ms in Streams) - { - ms.Dispose(); - } - } -} - -class Program -{ - static void Main() - { - try - { - string inputPath = "sample.epub"; - string outputDirectory = "output_images"; - - // Ensure output directory exists - Directory.CreateDirectory(outputDirectory); - - // Open the EPUB file - using (Stream inputStream = File.OpenRead(inputPath)) - { - // Configure image save options (PNG format) - ImageSaveOptions options = new ImageSaveOptions(ImageFormat.Png); - - // Create the custom stream provider - using (MemoryStreamProvider provider = new MemoryStreamProvider()) - { - // Convert EPUB to images, storing each image in a memory stream - Aspose.Html.Converters.Converter.ConvertEPUB(inputStream, options, provider); - - // Process each generated image stream - int index = 0; - foreach (MemoryStream ms in provider.Streams) - { - ms.Position = 0; // Rewind before reading - - // Example further processing: obtain byte array - byte[] imageBytes = ms.ToArray(); - - // Write the image to disk - string outputPath = Path.Combine(outputDirectory, $"image_{index}.png"); - using (FileStream fileStream = new FileStream(outputPath, FileMode.Create, FileAccess.Write)) - { - ms.CopyTo(fileStream); - } - - index++; - } - } - } - - Console.WriteLine("Image extraction completed successfully."); - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/support-extraction-favicon-ico-files-handling-link-elements-rel-shortcut-icon.cs b/extract-images-from-website/support-extraction-favicon-ico-files-handling-link-elements-rel-shortcut-icon.cs new file mode 100644 index 0000000..540a510 --- /dev/null +++ b/extract-images-from-website/support-extraction-favicon-ico-files-handling-link-elements-rel-shortcut-icon.cs @@ -0,0 +1,29 @@ +// Support extraction of favicon.ico files by handling link elements with rel='shortcut icon'. + +using System; + +class Program +{ + static void Main() + { + try + { + string html = ""; + var doc = new Aspose.Html.HTMLDocument(html, "about:blank"); + Aspose.Html.Dom.XPath.IXPathResult result = doc.Evaluate("//link[@rel='shortcut icon']", doc, doc.CreateNSResolver(doc), Aspose.Html.Dom.XPath.XPathResultType.Any, null); + Aspose.Html.Dom.Node node; + while ((node = result.IterateNext()) != null) + { + var link = node as Aspose.Html.HTMLLinkElement; + if (link != null) + { + Console.WriteLine("Favicon href: " + link.Href); + } + } + } + catch (Exception ex) + { + Console.WriteLine("Error: " + ex.Message); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/use-async-streams-write-image-files-directly-to-disk-while-downloading-reduce-memory-usage.cs b/extract-images-from-website/use-async-streams-write-image-files-directly-to-disk-while-downloading-reduce-memory-usage.cs new file mode 100644 index 0000000..92c7a78 --- /dev/null +++ b/extract-images-from-website/use-async-streams-write-image-files-directly-to-disk-while-downloading-reduce-memory-usage.cs @@ -0,0 +1,89 @@ +// Use async streams to write image files directly to disk while downloading to reduce memory usage. + +using System; +using System.IO; +using System.Net.Http; +using System.Collections.Generic; +using System.Threading.Tasks; +using Aspose.Html.IO; + +class FileStreamProvider : Aspose.Html.IO.ICreateStreamProvider, IDisposable +{ + private readonly string _outputDir; + public List FilePaths { get; } = new List(); + + public FileStreamProvider(string outputDir) + { + _outputDir = outputDir; + Directory.CreateDirectory(_outputDir); + } + + public Stream GetStream(string name, string extension) + { + string filePath = Path.Combine(_outputDir, $"{name}{extension}"); + var fs = new FileStream(filePath, FileMode.Create, FileAccess.Write, FileShare.None, 8192, useAsync: true); + FilePaths.Add(filePath); + return fs; + } + + public Stream GetStream(string name, string extension, int page) + { + string filePath = Path.Combine(_outputDir, $"{name}_page{page}{extension}"); + var fs = new FileStream(filePath, FileMode.Create, FileAccess.Write, FileShare.None, 8192, useAsync: true); + FilePaths.Add(filePath); + return fs; + } + + public void ReleaseStream(Stream stream) + { + if (stream != null) + { + stream.Flush(); + } + } + + public void Dispose() + { + // No unmanaged resources to clean up in this example. + } +} + +class Program +{ + static async Task Main(string[] args) + { + try + { + // URL of the EPUB file to download (replace with a valid URL for real testing) + string epubUrl = "https://example.com/sample.epub"; + string tempEpubPath = Path.Combine(Path.GetTempPath(), "sample.epub"); + + // Download EPUB using async streams + using (var httpClient = new HttpClient()) + using (var response = await httpClient.GetAsync(epubUrl, HttpCompletionOption.ResponseHeadersRead)) + { + response.EnsureSuccessStatusCode(); + await using (Stream downloadStream = await response.Content.ReadAsStreamAsync()) + await using (FileStream fileStream = new FileStream(tempEpubPath, FileMode.Create, FileAccess.Write, FileShare.None, 8192, useAsync: true)) + { + await downloadStream.CopyToAsync(fileStream); + } + } + + // Convert EPUB pages to images, writing each image directly to disk via async streams + using (FileStream epubStream = File.OpenRead(tempEpubPath)) + { + var options = new Aspose.Html.Saving.ImageSaveOptions(Aspose.Html.Rendering.Image.ImageFormat.Png); + string outputDir = Path.Combine(Directory.GetCurrentDirectory(), "output_images"); + using var provider = new FileStreamProvider(outputDir); + Aspose.Html.Converters.Converter.ConvertEPUB(epubStream, options, provider); + } + + Console.WriteLine("EPUB conversion completed successfully."); + } + catch (Exception ex) + { + Console.WriteLine($"Error: {ex.Message}"); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/use-cancellation-token-graceful-termination-image-extraction-process.cs b/extract-images-from-website/use-cancellation-token-graceful-termination-image-extraction-process.cs new file mode 100644 index 0000000..d7fcb2f --- /dev/null +++ b/extract-images-from-website/use-cancellation-token-graceful-termination-image-extraction-process.cs @@ -0,0 +1,77 @@ +// Use CancellationToken to allow graceful termination of the image extraction process. + +using System; +using System.IO; +using System.Net.Http; +using System.Threading; +using Aspose.Html; +using Aspose.Html.Collections; +using Aspose.Html.Dom; + +class Program +{ + static void Main(string[] args) + { + try + { + string baseDir = AppDomain.CurrentDomain.BaseDirectory; + string inputPath = Path.Combine(baseDir, "sample.html"); + string outputDir = Path.Combine(baseDir, "output"); + Directory.CreateDirectory(outputDir); + + if (!File.Exists(inputPath)) + { + string htmlContent = ""; + File.WriteAllText(inputPath, htmlContent); + } + + using (HTMLDocument document = new HTMLDocument(inputPath)) + { + HTMLCollection images = document.GetElementsByTagName("img"); + using (HttpClient httpClient = new HttpClient()) + { + using (CancellationTokenSource cts = new CancellationTokenSource()) + { + // Optional timeout (e.g., 30 seconds) + cts.CancelAfter(TimeSpan.FromSeconds(30)); + CancellationToken token = cts.Token; + + for (int i = 0; i < images.Length; i++) + { + if (token.IsCancellationRequested) + break; + + Element imgElement = (Element)images[i]; + string src = imgElement.GetAttribute("src"); + if (string.IsNullOrEmpty(src)) + continue; + + string absoluteUrl; + if (Uri.IsWellFormedUriString(src, UriKind.Absolute)) + { + absoluteUrl = src; + } + else + { + Uri baseUri = new Uri(document.BaseURI, UriKind.Absolute); + Uri resolved = new Uri(baseUri, src); + absoluteUrl = resolved.ToString(); + } + + byte[] imageBytes = httpClient.GetByteArrayAsync(absoluteUrl, token).GetAwaiter().GetResult(); + string fileName = Path.GetFileName(absoluteUrl); + string savePath = Path.Combine(outputDir, fileName); + File.WriteAllBytes(savePath, imageBytes); + } + } + } + } + + Console.WriteLine("Image extraction completed successfully."); + } + catch (Exception ex) + { + Console.WriteLine("Error: " + ex.Message); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/use_custom_http_message_handler_to_simulate_network_latency_during_performance_testing.cs b/extract-images-from-website/use-custom-httpmessagehandler-simulate-network-latency-performance-testing.cs similarity index 66% rename from extract-images-from-website/use_custom_http_message_handler_to_simulate_network_latency_during_performance_testing.cs rename to extract-images-from-website/use-custom-httpmessagehandler-simulate-network-latency-performance-testing.cs index f29b36f..7451cd7 100644 --- a/extract-images-from-website/use_custom_http_message_handler_to_simulate_network_latency_during_performance_testing.cs +++ b/extract-images-from-website/use-custom-httpmessagehandler-simulate-network-latency-performance-testing.cs @@ -1,20 +1,19 @@ // Use custom HttpMessageHandler to simulate network latency during performance testing. using System; -using System.Diagnostics; public sealed class LatencyMessageHandler : Aspose.Html.Net.MessageHandler { public override void Invoke(Aspose.Html.Net.INetworkOperationContext context) { - DateTime startTime = DateTime.UtcNow; + System.DateTime startTime = System.DateTime.UtcNow; Next(context); - DateTime endTime = DateTime.UtcNow; - TimeSpan elapsed = endTime - startTime; - Debug.WriteLine("Request: " + context.Request.RequestUri); - Debug.WriteLine("Start: " + startTime.ToString("O")); - Debug.WriteLine("End: " + endTime.ToString("O")); - Debug.WriteLine("Elapsed: " + elapsed.TotalMilliseconds + " ms"); + System.DateTime endTime = System.DateTime.UtcNow; + System.TimeSpan elapsed = endTime - startTime; + System.Diagnostics.Debug.WriteLine("Request: " + context.Request.RequestUri); + System.Diagnostics.Debug.WriteLine("Start: " + startTime.ToString("O")); + System.Diagnostics.Debug.WriteLine("End: " + endTime.ToString("O")); + System.Diagnostics.Debug.WriteLine("Elapsed: " + elapsed.TotalMilliseconds + " ms"); } } diff --git a/extract-images-from-website/use-htmldocument-queryselectorall-css-selector-img-data-important-true-target-specific-images.cs b/extract-images-from-website/use-htmldocument-queryselectorall-css-selector-img-data-important-true-target-specific-images.cs new file mode 100644 index 0000000..de0dccf --- /dev/null +++ b/extract-images-from-website/use-htmldocument-queryselectorall-css-selector-img-data-important-true-target-specific-images.cs @@ -0,0 +1,40 @@ +// Use HtmlDocument.QuerySelectorAll with CSS selector "img[data-important='true']" to target specific images. + +using System; + +class Program +{ + static void Main() + { + try + { + string htmlContent = "" + + "" + + "" + + "

Sample paragraph.

" + + ""; + + using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(htmlContent, "about:blank")) + { + Aspose.Html.Collections.NodeList selectedNodes = document.QuerySelectorAll("img[data-important='true']"); + + foreach (Aspose.Html.Dom.Node node in selectedNodes) + { + Aspose.Html.Dom.Element element = node as Aspose.Html.Dom.Element; + if (element != null && element.ParentNode != null) + { + element.ParentNode.RemoveChild(element); + } + } + + string outputPath = "output.html"; + document.Save(outputPath); + Console.WriteLine("Modified document saved to: " + outputPath); + } + } + catch (Exception ex) + { + Console.WriteLine("Error: " + ex.Message); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/use-retry-after-header-http-response-schedule-delayed-retries-rate-limited-resources.cs b/extract-images-from-website/use-retry-after-header-http-response-schedule-delayed-retries-rate-limited-resources.cs new file mode 100644 index 0000000..a8bdf23 --- /dev/null +++ b/extract-images-from-website/use-retry-after-header-http-response-schedule-delayed-retries-rate-limited-resources.cs @@ -0,0 +1,65 @@ +// Use a retry‑after header from HTTP response to schedule delayed retries for rate‑limited resources. + +namespace Example +{ + class Program + { + static void Main() + { + try + { + Aspose.Html.Configuration configuration = new Aspose.Html.Configuration(); + Aspose.Html.Services.INetworkService network = configuration.GetService(); + network.MessageHandlers.Add(new RetryAfterHandler(3)); + + string url = "https://example.com/rate-limited-resource"; + using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(url, configuration)) + { + string html = document.DocumentElement != null ? document.DocumentElement.OuterHTML : string.Empty; + System.Console.WriteLine(html); + } + } + catch (System.Exception ex) + { + System.Console.WriteLine("Error: " + ex.Message); + } + } + } + + class RetryAfterHandler : Aspose.Html.Net.MessageHandler + { + private readonly int _maxRetries; + public RetryAfterHandler(int maxRetries) + { + _maxRetries = maxRetries; + } + + public override void Invoke(Aspose.Html.Net.INetworkOperationContext context) + { + for (int attempt = 0; attempt <= _maxRetries; attempt++) + { + Next(context); + int status = (int)context.Response.StatusCode; + if (status != 429) + { + break; + } + + string retryAfter = null; + try + { + retryAfter = context.Response.Headers["Retry-After"]; + } + catch { } + + int delaySeconds = 5; + if (!string.IsNullOrEmpty(retryAfter) && int.TryParse(retryAfter, out int secs)) + { + delaySeconds = secs; + } + + System.Threading.Thread.Sleep(System.TimeSpan.FromSeconds(delaySeconds)); + } + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/use_async_streams_write_image_files_directly_to_disk_while_downloading_reduce_memory_usage.cs b/extract-images-from-website/use_async_streams_write_image_files_directly_to_disk_while_downloading_reduce_memory_usage.cs deleted file mode 100644 index 321933e..0000000 --- a/extract-images-from-website/use_async_streams_write_image_files_directly_to_disk_while_downloading_reduce_memory_usage.cs +++ /dev/null @@ -1,106 +0,0 @@ -// Use async streams to write image files directly to disk while downloading to reduce memory usage. - -using System; -using System.Collections.Generic; -using System.IO; -using System.Threading.Tasks; -using Aspose.Html.IO; -using Aspose.Html.Saving; -using Aspose.Html.Rendering.Image; -using Aspose.Html.Converters; -using Aspose.Html; - -class MemoryStreamProvider : Aspose.Html.IO.ICreateStreamProvider, IDisposable -{ - public List Streams { get; } = new List(); - - public Stream GetStream(string name, string extension) - { - var ms = new MemoryStream(); - Streams.Add(ms); - return ms; - } - - public Stream GetStream(string name, string extension, int page) - { - var ms = new MemoryStream(); - Streams.Add(ms); - return ms; - } - - public void ReleaseStream(Stream stream) - { - if (stream != null) - { - stream.Flush(); - } - } - - public void Dispose() - { - foreach (var ms in Streams) - { - ms.Dispose(); - } - } -} - -class Program -{ - static async Task Main(string[] args) - { - try - { - // Prepare sample HTML file - string inputPath = Path.Combine(Directory.GetCurrentDirectory(), "sample.html"); - string htmlContent = "

Hello, Aspose.HTML!

"; - await File.WriteAllTextAsync(inputPath, htmlContent); - - // Output image path - string outputPath = Path.Combine(Directory.GetCurrentDirectory(), "output.jpg"); - - // Open input stream - using (Stream inputStream = File.OpenRead(inputPath)) - { - // Load HTML document - using (HTMLDocument document = new HTMLDocument(inputStream, "")) - { - // Set image save options - ImageSaveOptions options = new ImageSaveOptions(ImageFormat.Jpeg); - - // Create custom stream provider - using (MemoryStreamProvider provider = new MemoryStreamProvider()) - { - // Convert HTML to image using provider - Aspose.Html.Converters.Converter.ConvertHTML(document, options, provider); - - // Write each generated image stream directly to disk asynchronously - int index = 0; - foreach (MemoryStream ms in provider.Streams) - { - ms.Position = 0; - string filePath = outputPath; - if (provider.Streams.Count > 1) - { - filePath = Path.Combine(Directory.GetCurrentDirectory(), $"output_{index}.jpg"); - } - - using (FileStream fileStream = new FileStream(filePath, FileMode.Create, FileAccess.Write, FileShare.None, 8192, true)) - { - await ms.CopyToAsync(fileStream); - } - - index++; - } - } - } - } - - Console.WriteLine("Conversion completed successfully."); - } - catch (Exception ex) - { - Console.WriteLine($"Error: {ex.Message}"); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/use_cancellationtoken_graceful_termination_image_extraction.cs b/extract-images-from-website/use_cancellationtoken_graceful_termination_image_extraction.cs deleted file mode 100644 index de6434b..0000000 --- a/extract-images-from-website/use_cancellationtoken_graceful_termination_image_extraction.cs +++ /dev/null @@ -1,81 +0,0 @@ -// Use CancellationToken to allow graceful termination of the image extraction process. - -using System; -using System.IO; -using System.Net.Http; -using System.Text; -using System.Threading; - -class Program -{ - static void Main() - { - try - { - string dataDir = "Data"; - string inputFile = Path.Combine(dataDir, "sample.html"); - string outputDir = "OutputImages"; - - // Ensure input file exists - if (!File.Exists(inputFile)) - { - Directory.CreateDirectory(dataDir); - File.WriteAllText(inputFile, "", Encoding.UTF8); - } - - Directory.CreateDirectory(outputDir); - - using (CancellationTokenSource cts = new CancellationTokenSource()) - { - cts.CancelAfter(TimeSpan.FromSeconds(30)); - CancellationToken token = cts.Token; - - using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(new Aspose.Html.Url(inputFile))) - { - Aspose.Html.Collections.HTMLCollection images = document.GetElementsByTagName("img"); - using (HttpClient httpClient = new HttpClient()) - { - for (int i = 0; i < images.Length; i++) - { - if (token.IsCancellationRequested) - { - Console.WriteLine("Operation cancelled."); - break; - } - - Aspose.Html.Dom.Element imgElement = (Aspose.Html.Dom.Element)images[i]; - string src = imgElement.GetAttribute("src"); - if (string.IsNullOrEmpty(src)) - continue; - - string baseUri = document.BaseURI; - Uri imageUri; - if (Uri.IsWellFormedUriString(src, UriKind.Absolute)) - imageUri = new Uri(src); - else - imageUri = new Uri(new Uri(baseUri), src); - - string urlString = imageUri.ToString(); - string extension = Path.GetExtension(urlString); - if (!extension.Equals(".png", StringComparison.OrdinalIgnoreCase) && - !extension.Equals(".jpg", StringComparison.OrdinalIgnoreCase) && - !extension.Equals(".jpeg", StringComparison.OrdinalIgnoreCase)) - continue; - - byte[] imageBytes = httpClient.GetByteArrayAsync(urlString, token).GetAwaiter().GetResult(); - string fileName = Path.GetFileName(urlString); - string savePath = Path.Combine(outputDir, fileName); - File.WriteAllBytes(savePath, imageBytes); - } - } - } - } - - Console.WriteLine("Image extraction completed."); - } - catch (Exception ex) - { - Console.WriteLine($"Error: {ex.Message}"); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/use_htmldocument_queryselectorall_css_selector_img_data_important_true_target_specific_images.cs b/extract-images-from-website/use_htmldocument_queryselectorall_css_selector_img_data_important_true_target_specific_images.cs deleted file mode 100644 index eab0941..0000000 --- a/extract-images-from-website/use_htmldocument_queryselectorall_css_selector_img_data_important_true_target_specific_images.cs +++ /dev/null @@ -1,39 +0,0 @@ -// Use HtmlDocument.QuerySelectorAll with CSS selector "img[data-important='true']" to target specific images. - -using System; -using System.IO; - -class Program -{ - static void Main() - { - try - { - string htmlContent = "" + - "" + - "" + - "" + - ""; - - string inputPath = Path.Combine(Path.GetTempPath(), "sample.html"); - File.WriteAllText(inputPath, htmlContent); - - using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(inputPath)) - { - Aspose.Html.Collections.NodeList images = document.QuerySelectorAll("img[data-important='true']"); - - foreach (Aspose.Html.Dom.Element img in images) - { - string src = img.GetAttribute("src"); - Console.WriteLine("Important image src: " + src); - } - } - - File.Delete(inputPath); - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/use_retry_after_header_from_http_response_to_schedule_delayed_retries_for_rate_limited_resource_requests.cs b/extract-images-from-website/use_retry_after_header_from_http_response_to_schedule_delayed_retries_for_rate_limited_resource_requests.cs deleted file mode 100644 index 4a2a974..0000000 --- a/extract-images-from-website/use_retry_after_header_from_http_response_to_schedule_delayed_retries_for_rate_limited_resource_requests.cs +++ /dev/null @@ -1,81 +0,0 @@ -// Use a retry‑after header from HTTP response to schedule delayed retries for rate‑limited resources. - -using System; -using System.Threading; -using Aspose.Html; -using Aspose.Html.Net; -using Aspose.Html.Services; - -class RetryAfterHandler : Aspose.Html.Net.MessageHandler -{ - private readonly int maxRetries; - public RetryAfterHandler(int maxRetries) - { - this.maxRetries = maxRetries; - } - - public override void Invoke(Aspose.Html.Net.INetworkOperationContext context) - { - for (int attempt = 0; attempt <= maxRetries; attempt++) - { - // Perform the request - Next(context); - - int statusCode = (int)context.Response.StatusCode; - - // If request succeeded or client error other than rate limiting, exit loop - if (statusCode < 500 && statusCode != 429 && statusCode != 503) - break; - - // Handle rate limiting responses - if (statusCode == 429 || statusCode == 503) - { - string retryAfterValue = context.Response.Headers["Retry-After"]; - int seconds; - if (!string.IsNullOrEmpty(retryAfterValue) && int.TryParse(retryAfterValue, out seconds)) - { - Thread.Sleep(TimeSpan.FromSeconds(seconds)); - } - else - { - // Default wait time if header is missing or invalid - Thread.Sleep(TimeSpan.FromSeconds(1)); - } - - // Continue to next attempt - continue; - } - - // For other server errors, do not retry - break; - } - } -} - -class Program -{ - static void Main() - { - try - { - // Create configuration and attach the custom message handler - Aspose.Html.Configuration configuration = new Aspose.Html.Configuration(); - Aspose.Html.Services.INetworkService network = configuration.GetService(); - network.MessageHandlers.Add(new RetryAfterHandler(3)); - - // Define the URL of the rate‑limited resource - string url = "https://example.com/rate-limited-resource.html"; - - // Load the document using the configuration with the handler - using (Aspose.Html.HTMLDocument document = new Aspose.Html.HTMLDocument(url, configuration)) - { - string html = document.DocumentElement != null ? document.DocumentElement.OuterHTML : string.Empty; - Console.WriteLine(html); - } - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/validate-resolved-url-returns-successful-http-status-before-attempting-download.cs b/extract-images-from-website/validate-resolved-url-returns-successful-http-status-before-attempting-download.cs new file mode 100644 index 0000000..4d5ade3 --- /dev/null +++ b/extract-images-from-website/validate-resolved-url-returns-successful-http-status-before-attempting-download.cs @@ -0,0 +1,86 @@ +// Validate that each resolved URL returns a successful HTTP status before attempting download. + +using System; +using System.IO; +using Aspose.Html; +using Aspose.Html.Net; +using Aspose.Html.Dom; +using Aspose.Html.Services; + +class Program +{ + static void Main() + { + try + { + // Prepare sample HTML file + string htmlFilePath = Path.Combine(Path.GetTempPath(), "sample.html"); + string htmlContent = @"" + + @"Valid Link" + + @"Invalid Link" + + @""; + File.WriteAllText(htmlFilePath, htmlContent); + + // Load the HTML document + using (HTMLDocument document = new HTMLDocument(htmlFilePath)) + { + // Ensure only local file resources are allowed (optional security handler) + Configuration configuration = new Configuration(); + INetworkService networkService = configuration.GetService(); + networkService.MessageHandlers.Insert(0, new LocalOnlyMessageHandler()); + + // Output directory for downloaded resources + string downloadDir = Path.Combine(Path.GetTempPath(), "downloaded_resources"); + Directory.CreateDirectory(downloadDir); + + foreach (Element linkElement in document.Links) + { + string href = linkElement.GetAttribute("href"); + if (string.IsNullOrEmpty(href)) + continue; + + // Resolve the URL against the document's base URI + Url resolvedUrl = new Url(href, document.BaseURI); + + // Send a HEAD request to validate the URL + RequestMessage request = new RequestMessage(resolvedUrl); + ResponseMessage response = document.Context.Network.Send(request); + + if (!response.IsSuccess) + { + Console.WriteLine($"Link validation failed: {resolvedUrl} (Status: {(int)response.StatusCode})"); + continue; + } + + // Download the content + byte[] contentBytes = response.Content.ReadAsByteArray(); + + // Determine a safe file name + string fileName = Path.GetFileName(resolvedUrl.Pathname); + if (string.IsNullOrEmpty(fileName)) + fileName = "downloaded_content_" + Guid.NewGuid().ToString() + ".bin"; + + string outputPath = Path.Combine(downloadDir, fileName); + File.WriteAllBytes(outputPath, contentBytes); + Console.WriteLine($"Successfully downloaded: {resolvedUrl} -> {outputPath}"); + } + } + } + catch (Exception ex) + { + Console.WriteLine("Error: " + ex.Message); + } + } + + // Message handler that allows only local file resources + public sealed class LocalOnlyMessageHandler : Aspose.Html.Net.MessageHandler + { + public override void Invoke(Aspose.Html.Net.INetworkOperationContext context) + { + string requestUri = context.Request.RequestUri == null ? string.Empty : context.Request.RequestUri.ToString(); + if (!string.IsNullOrEmpty(requestUri) && !requestUri.StartsWith("file:", StringComparison.OrdinalIgnoreCase)) + throw new InvalidOperationException("Only local file resources are allowed."); + Next(context); + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/validate-saved-image-files-not-corrupted-by-checking-file-signatures-after-write-operation.cs b/extract-images-from-website/validate-saved-image-files-not-corrupted-by-checking-file-signatures-after-write-operation.cs new file mode 100644 index 0000000..8dc0894 --- /dev/null +++ b/extract-images-from-website/validate-saved-image-files-not-corrupted-by-checking-file-signatures-after-write-operation.cs @@ -0,0 +1,93 @@ +// Validate that saved image files are not corrupted by checking file signatures after write operation. + +using System; +using System.IO; +using Aspose.Html.Saving; +using Aspose.Html.Rendering.Image; + +class Program +{ + static void Main() + { + try + { + // Prepare folders + string inputFolder = "Input"; + string outputFolder = "Output"; + Directory.CreateDirectory(inputFolder); + Directory.CreateDirectory(outputFolder); + + // Create a simple SVG file + string svgPath = Path.Combine(inputFolder, "sample.svg"); + string svgContent = @" + + +"; + File.WriteAllText(svgPath, svgContent); + + // Output paths + string highOutputPath = Path.Combine(outputFolder, "high.jpg"); + string lowOutputPath = Path.Combine(outputFolder, "low.jpg"); + + // High quality options (higher resolution) + ImageSaveOptions highOptions = new ImageSaveOptions(ImageFormat.Jpeg); + highOptions.HorizontalResolution = 300; + highOptions.VerticalResolution = 300; + + // Low quality options (lower resolution) + ImageSaveOptions lowOptions = new ImageSaveOptions(ImageFormat.Jpeg); + lowOptions.HorizontalResolution = 72; + lowOptions.VerticalResolution = 72; + + // Convert SVG to JPEG with both options + Aspose.Html.Converters.Converter.ConvertSVG(svgPath, highOptions, highOutputPath); + Aspose.Html.Converters.Converter.ConvertSVG(svgPath, lowOptions, lowOutputPath); + + // Compare file sizes + long highSize = new FileInfo(highOutputPath).Length; + long lowSize = new FileInfo(lowOutputPath).Length; + if (highSize > lowSize) + { + Console.WriteLine("High resolution image is larger than low resolution image."); + } + else + { + Console.WriteLine("Low resolution image is larger than high resolution image."); + } + + // Validate JPEG signatures + byte[] jpegSignature = new byte[] { 0xFF, 0xD8, 0xFF }; + + bool highValid = CheckSignature(highOutputPath, jpegSignature); + bool lowValid = CheckSignature(lowOutputPath, jpegSignature); + + Console.WriteLine(highValid ? "High image file is valid." : "High image file is corrupted."); + Console.WriteLine(lowValid ? "Low image file is valid." : "Low image file is corrupted."); + } + catch (Exception ex) + { + Console.WriteLine("Error: " + ex.Message); + } + } + + static bool CheckSignature(string filePath, byte[] expectedSignature) + { + if (!File.Exists(filePath)) + return false; + + byte[] buffer = new byte[expectedSignature.Length]; + using (FileStream fs = File.OpenRead(filePath)) + { + int read = fs.Read(buffer, 0, buffer.Length); + if (read != buffer.Length) + return false; + } + + for (int i = 0; i < expectedSignature.Length; i++) + { + if (buffer[i] != expectedSignature[i]) + return false; + } + return true; + } +} \ No newline at end of file diff --git a/extract-images-from-website/validate_each_resolved_url_successful_http_status_before_download.cs b/extract-images-from-website/validate_each_resolved_url_successful_http_status_before_download.cs deleted file mode 100644 index 6bd96ce..0000000 --- a/extract-images-from-website/validate_each_resolved_url_successful_http_status_before_download.cs +++ /dev/null @@ -1,72 +0,0 @@ -// Validate that each resolved URL returns a successful HTTP status before attempting download. - -using System; -using System.IO; -using System.Collections.Generic; -using Aspose.Html; -using Aspose.Html.Net; -using Aspose.Html.Dom; - -class Program -{ - static void Main() - { - try - { - string htmlPath = "sample.html"; - - if (!File.Exists(htmlPath)) - { - string sampleHtml = "" + - "Example" + - "Broken" + - ""; - File.WriteAllText(htmlPath, sampleHtml); - } - - using (HTMLDocument document = new HTMLDocument(htmlPath)) - { - List brokenLinks = new List(); - - foreach (Element linkElement in document.Links) - { - string href = linkElement.GetAttribute("href"); - if (string.IsNullOrEmpty(href)) - continue; - - Url resolvedUrl = new Url(href, document.BaseURI); - RequestMessage request = new RequestMessage(resolvedUrl); - ResponseMessage response = document.Context.Network.Send(request); - - if (!response.IsSuccess) - { - brokenLinks.Add(resolvedUrl.ToString()); - Console.WriteLine($"Link failed: {resolvedUrl} Status: {(int)response.StatusCode}"); - } - else - { - byte[] contentBytes = response.Content.ReadAsByteArray(); - Console.WriteLine($"Link succeeded: {resolvedUrl} Size: {contentBytes.Length} bytes"); - } - } - - if (brokenLinks.Count == 0) - { - Console.WriteLine("All links are valid."); - } - else - { - Console.WriteLine("Broken links detected:"); - foreach (string url in brokenLinks) - { - Console.WriteLine(url); - } - } - } - } - catch (Exception ex) - { - Console.WriteLine($"Error: {ex.Message}"); - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/validate_image_files_not_corrupted_file_signatures_after_write.cs b/extract-images-from-website/validate_image_files_not_corrupted_file_signatures_after_write.cs deleted file mode 100644 index ff6b4aa..0000000 --- a/extract-images-from-website/validate_image_files_not_corrupted_file_signatures_after_write.cs +++ /dev/null @@ -1,70 +0,0 @@ -// Validate that saved image files are not corrupted by checking file signatures after write operation. - -using System; -using System.IO; - -class Program -{ - static void Main() - { - try - { - // Prepare directories - string dataDir = Path.Combine(Directory.GetCurrentDirectory(), "Data"); - Directory.CreateDirectory(dataDir); - string outputDir = Path.Combine(Directory.GetCurrentDirectory(), "Output"); - Directory.CreateDirectory(outputDir); - - // Create a minimal SVG file - string svgPath = Path.Combine(dataDir, "sample.svg"); - if (!File.Exists(svgPath)) - { - string svgContent = @" - -"; - File.WriteAllText(svgPath, svgContent); - } - - // Define output paths - string highOutputPath = Path.Combine(outputDir, "high.jpg"); - string lowOutputPath = Path.Combine(outputDir, "low.jpg"); - - // Convert SVG to JPEG with default options - var highOptions = new Aspose.Html.Saving.ImageSaveOptions(Aspose.Html.Rendering.Image.ImageFormat.Jpeg); - var lowOptions = new Aspose.Html.Saving.ImageSaveOptions(Aspose.Html.Rendering.Image.ImageFormat.Jpeg); - - Aspose.Html.Converters.Converter.ConvertSVG(svgPath, highOptions, highOutputPath); - Aspose.Html.Converters.Converter.ConvertSVG(svgPath, lowOptions, lowOutputPath); - - // Validate JPEG signatures - bool highValid = IsJpeg(highOutputPath); - bool lowValid = IsJpeg(lowOutputPath); - - if (highValid && lowValid) - { - Console.WriteLine("Both images are valid JPEG files."); - } - else - { - if (!highValid) - Console.WriteLine("High-quality image is corrupted or not a JPEG."); - if (!lowValid) - Console.WriteLine("Low-quality image is corrupted or not a JPEG."); - } - } - catch (Exception ex) - { - Console.WriteLine($"Error: {ex.Message}"); - } - } - - static bool IsJpeg(string path) - { - using (FileStream fs = new FileStream(path, FileMode.Open, FileAccess.Read)) - { - byte[] header = new byte[3]; - int read = fs.Read(header, 0, 3); - return read == 3 && header[0] == 0xFF && header[1] == 0xD8 && header[2] == 0xFF; - } - } -} \ No newline at end of file diff --git a/extract-images-from-website/write-unit-tests-mock-httpclient-responses-verify-url-resolution-download-logic.cs b/extract-images-from-website/write-unit-tests-mock-httpclient-responses-verify-url-resolution-download-logic.cs new file mode 100644 index 0000000..824d199 --- /dev/null +++ b/extract-images-from-website/write-unit-tests-mock-httpclient-responses-verify-url-resolution-download-logic.cs @@ -0,0 +1,125 @@ +// Write unit tests that mock HttpClient responses to verify URL resolution and download logic. + +using System; +using System.IO; +using System.Net; +using System.Net.Http; +using System.Threading; +using System.Threading.Tasks; +using Aspose.Html; +using Aspose.Html.Collections; +using Aspose.Html.Dom; +using Aspose.Html.Net; + +class MockHttpMessageHandler : HttpMessageHandler +{ + private readonly byte[] _responseContent; + private readonly string _expectedUrl; + + public MockHttpMessageHandler(string expectedUrl, byte[] responseContent) + { + _expectedUrl = expectedUrl; + _responseContent = responseContent; + } + + protected override Task SendAsync(HttpRequestMessage request, CancellationToken cancellationToken) + { + HttpResponseMessage response; + if (request.RequestUri != null && request.RequestUri.ToString().Equals(_expectedUrl, StringComparison.OrdinalIgnoreCase)) + { + response = new HttpResponseMessage(HttpStatusCode.OK) + { + Content = new System.Net.Http.ByteArrayContent(_responseContent) + }; + } + else + { + response = new HttpResponseMessage(HttpStatusCode.NotFound); + } + return Task.FromResult(response); + } +} + +class Program +{ + static void Main() + { + try + { + bool result = TestImageDownload(); + Console.WriteLine(result ? "Test passed." : "Test failed."); + } + catch (Exception ex) + { + Console.WriteLine("Error: " + ex.Message); + } + } + + static bool TestImageDownload() + { + // Prepare sample HTML with a relative image URL + string html = ""; + string baseUri = "http://example.com/folder/"; + string expectedUrl = "http://example.com/folder/image.png"; + byte[] imageBytes = new byte[] { 1, 2, 3, 4 }; + + // Create a temporary output directory + string outputDir = Path.Combine(Path.GetTempPath(), "AsposeHtmlImageTest"); + Directory.CreateDirectory(outputDir); + + // Set up mocked HttpClient + using (HttpClient httpClient = new HttpClient(new MockHttpMessageHandler(expectedUrl, imageBytes))) + { + // Execute the download logic + DownloadImages(html, baseUri, outputDir, httpClient); + } + + // Verify that the image file was saved correctly + string savedFilePath = Path.Combine(outputDir, "image.png"); + if (!File.Exists(savedFilePath)) + return false; + + byte[] savedBytes = File.ReadAllBytes(savedFilePath); + return savedBytes.Length == imageBytes.Length && savedBytes[0] == imageBytes[0] && savedBytes[1] == imageBytes[1]; + } + + static void DownloadImages(string htmlContent, string baseUri, string outputDir, HttpClient httpClient) + { + // Ensure output directory exists + Directory.CreateDirectory(outputDir); + + // Load HTML document from inline content + using (HTMLDocument document = new HTMLDocument(htmlContent, baseUri)) + { + // Get all elements + HTMLCollection images = document.GetElementsByTagName("img"); + for (int i = 0; i < images.Length; i++) + { + Element imgElement = (Element)images[i]; + string src = imgElement.GetAttribute("src"); + if (string.IsNullOrEmpty(src)) + continue; + + // Resolve the image URL against the document's base URI + Url imageUrl = new Url(src, document.BaseURI); + string urlString = imageUrl.ToString(); + string extension = Path.GetExtension(urlString); + if (!extension.Equals(".png", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".jpg", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".jpeg", StringComparison.OrdinalIgnoreCase) && + !extension.Equals(".gif", StringComparison.OrdinalIgnoreCase)) + { + continue; + } + + // Download image bytes + byte[] imageBytes = httpClient.GetByteArrayAsync(urlString).GetAwaiter().GetResult(); + + // Save to output directory + string fileName = Path.GetFileName(urlString); + string savePath = Path.Combine(outputDir, fileName); + File.WriteAllBytes(savePath, imageBytes); + } + } + } +} \ No newline at end of file diff --git a/extract-images-from-website/write_unit_tests_mock_httpclient_responses_verify_url_resolution_download_logic.cs b/extract-images-from-website/write_unit_tests_mock_httpclient_responses_verify_url_resolution_download_logic.cs deleted file mode 100644 index baef487..0000000 --- a/extract-images-from-website/write_unit_tests_mock_httpclient_responses_verify_url_resolution_download_logic.cs +++ /dev/null @@ -1,94 +0,0 @@ -// Write unit tests that mock HttpClient responses to verify URL resolution and download logic. - -using System; -using System.IO; -using System.Net; -using System.Net.Http; -using System.Threading; -using System.Threading.Tasks; -using Aspose.Html; -using Aspose.Html.Collections; -using Aspose.Html.Dom; -using Aspose.Html.Rendering; - -class MockHttpMessageHandler : HttpMessageHandler -{ - protected override Task SendAsync(HttpRequestMessage request, CancellationToken cancellationToken) - { - var response = new HttpResponseMessage(HttpStatusCode.OK); - if (request.RequestUri.ToString() == "http://example.com/images/pic.png") - { - byte[] bytes = new byte[] { 1, 2, 3, 4 }; - response.Content = new ByteArrayContent(bytes); - } - else - { - response.StatusCode = HttpStatusCode.NotFound; - } - return Task.FromResult(response); - } -} - -class Program -{ - static void Main() - { - try - { - RunTest(); - } - catch (Exception ex) - { - Console.WriteLine("Error: " + ex.Message); - } - } - - static void RunTest() - { - string html = ""; - string baseUri = "http://example.com/"; - HTMLDocument document = new HTMLDocument(html, baseUri); - - MockHttpMessageHandler handler = new MockHttpMessageHandler(); - HttpClient httpClient = new HttpClient(handler); - - string outputDir = Path.Combine(Path.GetTempPath(), "AsposeHtmlTest"); - DownloadImages(document, httpClient, outputDir); - - string expectedPath = Path.Combine(outputDir, "pic.png"); - if (!File.Exists(expectedPath)) - throw new Exception("Image file was not created."); - - byte[] data = File.ReadAllBytes(expectedPath); - if (data.Length != 4 || data[0] != 1 || data[1] != 2 || data[2] != 3 || data[3] != 4) - throw new Exception("Image content does not match expected bytes."); - - Console.WriteLine("Test passed."); - } - - static void DownloadImages(HTMLDocument document, HttpClient httpClient, string outputDir) - { - Directory.CreateDirectory(outputDir); - HTMLCollection images = document.GetElementsByTagName("img"); - for (int i = 0; i < images.Length; i++) - { - Element imgElement = (Element)images[i]; - string src = imgElement.GetAttribute("src"); - if (string.IsNullOrEmpty(src)) - continue; - - Url imageUrl = new Url(src, document.BaseURI); - string urlString = imageUrl.ToString(); - string extension = Path.GetExtension(urlString); - if (!extension.Equals(".png", StringComparison.OrdinalIgnoreCase) && - !extension.Equals(".jpg", StringComparison.OrdinalIgnoreCase) && - !extension.Equals(".jpeg", StringComparison.OrdinalIgnoreCase)) - continue; - - byte[] imageBytes = httpClient.GetByteArrayAsync(urlString).GetAwaiter().GetResult(); - string fileName = Path.GetFileName(urlString); - string savePath = Path.Combine(outputDir, fileName); - File.WriteAllBytes(savePath, imageBytes); - } - } -} \ No newline at end of file