public fileunit toNewOutput(String content, String name, String extension = "txt") { fileunit funit = getNewOutput(name, extension); funit.setContent(content); return(funit); }
private void reportTarget(spiderTarget t, folderNode fn, int c) { string pageFolder = "P" + c.ToString("D3") + "_" + t.IsRelevant.ToString(); folderNode pfn = fn.Add(pageFolder, "Page " + c.ToString(), "Report on page " + t.url + " crawled by " + name + ". Target.IsRelevant: " + t.IsRelevant + ".".addLine(pageDescription)); fileunit content = new fileunit(pfn.pathFor("content.txt"), false); fileunit links = new fileunit(pfn.pathFor("links.txt"), false); if (t.evaluation != null) { t.evaluation.saveObjectToXML(pfn.pathFor("relevance.xml")); } content.setContent(t.pageText); //t.page.relationship.outflowLinks if (t.page != null) { foreach (spiderLink ln in t.page.relationship.outflowLinks.items.Values) { string rl = ln.url; links.Append(ln.url); } //t.page.webpage.links.ForEach(x => links.Append(x.nature + " | " + x.name + " | " + x.url)); } content.Save(); links.Save(); // marks.Save(); }
public void reportCrawler(modelSpiderTestRecord tRecord) { folderNode fn = folder[DRFolderEnum.crawler]; string fileprefix = tRecord.instance.name.getCleanFilePath(); //tRecord.name.getCleanFilepath(); if (REPORT_TIMELINE) { DataTable timeline = timeSeries.GetAggregatedTable("frontier_stats", dataPointAggregationAspect.overlapMultiTable); //.GetSumTable("timeline_" + fileprefix.Replace(" ", "")); timeline.GetReportAndSave(folder[DRFolderEnum.crawler], notation, "frontier_stats" + fileprefix); } if (REPORT_ITERATION_URLS) { tRecord.allUrls = urlsLoaded.GetAllUnique(); tRecord.allDetectedUrls = urlsDetected.GetAllUnique(); saveOutput(tRecord.allDetectedUrls, folder[DRFolderEnum.crawler].pathFor("urls_detected.txt")); saveOutput(tRecord.allUrls, folder[DRFolderEnum.crawler].pathFor("urls_loaded.txt")); saveOutput(tRecord.relevantPages, folder[DRFolderEnum.crawler].pathFor("urls_relevant_loaded.txt")); } // Int32 iterations = tRecord.instance.settings.limitIterations; DataTable cpuTable = tRecord.cpuTaker.GetDataTableBase("cpuMetrics").GetReportAndSave(folder[DRFolderEnum.crawler], notation, "cpu_" + fileprefix); DataTable dataTable = tRecord.dataLoadTaker.GetDataTableBase("dataLoadMetrics").GetReportAndSave(folder[DRFolderEnum.crawler], notation, "dataload_" + fileprefix); DataTable resourcesTable = tRecord.measureTaker.GetDataTableBase("resourceMetrics").GetReportAndSave(folder[DRFolderEnum.crawler], notation, "resource_" + fileprefix); if (imbWEMManager.settings.directReportEngine.doPublishPerformance) { tRecord.performance.folderName = folder.name; tRecord.performance.deploy(tRecord); tRecord.performance.saveObjectToXML(folder[DRFolderEnum.crawler].pathFor("performance.xml")); DataTable pTable = tRecord.performance.GetDataTable(true).GetReportAndSave(folder, notation, "crawler_performance" + fileprefix); } tRecord.lastDomainIterationTable.GetDataTable(null, imbWEMManager.index.experimentEntry.CrawlID).GetReportAndSave(folder, notation, "DLCs_performance_" + fileprefix); tRecord.reporter = this; signature.deployReport(tRecord); //signature.notation = notation; signature.saveObjectToXML(folder.pathFor("signature.xml")); folder.generateReadmeFiles(notation); fileunit tLog = new fileunit(folder[DRFolderEnum.logs].pathFor(fileprefix + ".txt"), false); tLog.setContent(tRecord.logBuilder.ContentToString(true)); tLog.Save(); tRecord.instance.reportCrawlFinished(this, tRecord); aceLog.consoleControl.setLogFileWriter(); }
/// <summary> /// Runs when a DLC is finished /// </summary> /// <param name="wRecord">The w record.</param> public void reportDomainFinished(modelSpiderSiteRecord wRecord) { folderNode fn = null; string fileprefix = wRecord.domainInfo.domainRootName.getCleanFilepath(); if (imbWEMManager.settings.directReportEngine.doDomainReport) { fn = folder[DRFolderEnum.sites].Add(wRecord.domainInfo.domainRootName.getCleanFilepath(), "Report on " + wRecord.domainInfo.domainName, "Records on domain " + wRecord.domainInfo.domainName + " crawled by " + name); if (REPORT_DOMAIN_TERMS) { if (wRecord.tRecord.instance.settings.doEnableDLC_TFIDF) { if (wRecord.context.targets.dlTargetPageTokens != null) { wRecord.context.targets.dlTargetPageTokens.GetDataSet(true).serializeDataSet("token_ptkn", fn, dataTableExportEnum.excel, notation); } } if (wRecord.context.targets.dlTargetLinkTokens != null) { wRecord.context.targets.dlTargetLinkTokens.GetDataSet(true).serializeDataSet("token_ltkn", fn, dataTableExportEnum.excel, notation); } } if (REPORT_DOMAIN_PAGES) { int c = 1; foreach (spiderTarget t in wRecord.context.targets.GetLoadedInOrderOfLoad()) { reportTarget(t, fn, c); c++; } } fileunit wLog = new fileunit(folder[DRFolderEnum.logs].pathFor(fileprefix + ".txt"), false); wLog.setContent(wRecord.logBuilder.ContentToString(true)); wLog.Save(); if (REPORT_ITERATION_URLS) { textByIteration url_loaded = urlsLoaded[wRecord]; //.GetOrAdd(wRecord, new textByIteration()); textByIteration url_detected = urlsDetected[wRecord]; //, new textByIteration()); fileunit url_ld_out = new fileunit(folder[DRFolderEnum.sites].pathFor(fileprefix + "_url_ld.txt"), false); fileunit url_dt_out = new fileunit(folder[DRFolderEnum.sites].pathFor(fileprefix + "_url_dt.txt"), false); fileunit url_srb_out = new fileunit(folder[DRFolderEnum.sites].pathFor(fileprefix + "_url_srb_ld.txt"), false); url_ld_out.setContentLines(url_loaded.GetAllUnique()); url_dt_out.setContentLines(url_detected.GetAllUnique()); url_srb_out.setContentLines(wRecord.relevantPages); url_ld_out.Save(); url_dt_out.Save(); url_srb_out.Save(); } //terms_out.Save(); //sentence_out.Save(); } if (REPORT_MODULES) { if (wRecord.tRecord.instance is spiderModularEvaluatorBase) { wRecord.frontierDLC.reportDomainOut(wRecord, fn, fileprefix); } } if (REPORT_TIMELINE) { wRecord.iterationTableRecord.GetDataTable(null, "iteration_performace_" + fileprefix).GetReportAndSave(folder[DRFolderEnum.it], notation, "iteration_performace_" + fileprefix); //, notation); } //if (REPORT_TIMELINE) //{ // DataTable dt = wRecord.GetTimeSeriesPerformance(); // timeSeries.Add(dt); // dt.GetReportAndSave(folder[DRFolderEnum.it], notation, "iteration_frontier_stats_" + fileprefix); //} wRecord.tRecord.lastDomainIterationTable.Add(wRecord.iterationTableRecord.GetLastEntryTouched()); wRecord.tRecord.instance.reportDomainFinished(this, wRecord); wRecord.Dispose(); }