removing PDFLibrary code (kept at separate fork called NaskbanService)

This commit is contained in:
Hamid Reza Mohammadi 2023-11-23 19:21:41 +03:30
parent 9b01a1bf35
commit 9f4164d25d
12 changed files with 9 additions and 7735 deletions

File diff suppressed because it is too large Load Diff

View File

@ -26,7 +26,6 @@
<IncludeAssets>runtime; build; native; contentfiles; analyzers; buildtransitive</IncludeAssets>
</PackageReference>
<PackageReference Include="NAudio" Version="2.1.0" />
<PackageReference Include="PDFtoImage" Version="2.3.0" />
<PackageReference Include="RSecurityBackend" Version="1.2.1" />
<PackageReference Include="Swashbuckle.AspNetCore" Version="6.5.0" />
<PackageReference Include="Swashbuckle.AspNetCore.Annotations" Version="6.5.0" />

File diff suppressed because it is too large Load Diff

View File

@ -1,475 +0,0 @@
using RMuseum.Models.Artifact;
using RMuseum.Models.GanjoorIntegration;
using RMuseum.Models.GanjoorIntegration.ViewModels;
using RMuseum.Models.PDFLibrary;
using RMuseum.Models.PDFLibrary.ViewModels;
using RSecurityBackend.Models.Generic;
using System;
using System.Threading.Tasks;
namespace RMuseum.Services
{
/// <summary>
/// PDF Library Services
/// </summary>
public interface IPDFLibraryService
{
/// <summary>
/// import from known sources
/// </summary>
/// <param name="srcUrl"></param>
/// <returns></returns>
void StartImportingKnownSourceAsync(string srcUrl);
/// <summary>
/// get pdf book by id
/// </summary>
/// <param name="id"></param>
/// <param name="statusArray"></param>
/// <param name="includePages"></param>
/// <param name="includeBookText"></param>
/// <param name="includePageText"></param>
/// <returns></returns>
Task<RServiceResult<PDFBook>> GetPDFBookByIdAsync(int id, PublishStatus[] statusArray, bool includePages, bool includeBookText, bool includePageText);
/// <summary>
/// get all pdfbooks (including CoverImage info but not pages or tagibutes info)
/// </summary>
/// <param name="paging"></param>
/// <param name="statusArray"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, PDFBook[] Books)>> GetAllPDFBooksAsync(PagingParameterModel paging, PublishStatus[] statusArray);
/// <summary>
/// an incomplete prototype for removing PDF books
/// </summary>
/// <param name="pdfBookId"></param>
/// <returns></returns>
Task<RServiceResult<bool>> RemovePDFBookAsync(int pdfBookId);
/// <summary>
/// add pdf book tag value
/// </summary>
/// <param name="pdfBookId"></param>
/// <param name="rTag"></param>
/// <returns></returns>
Task<RServiceResult<RTagValue>> TagPDFBookAsync(int pdfBookId, RTag rTag);
/// <summary>
/// remove pdf book tag value
/// </summary>
/// <param name="pdfBookId"></param>
/// <param name="tagValueId"></param>
/// <returns></returns>
Task<RServiceResult<bool>> UnTagPDFBookAsync(int pdfBookId, Guid tagValueId);
/// <summary>
/// edit pdf book tag value
/// </summary>
/// <param name="pdfBookId"></param>
/// <param name="edited"></param>
/// <param name="global">apply on all same value tags</param>
/// <returns></returns>
Task<RServiceResult<RTagValue>> EditPDFBookTagValueAsync(int pdfBookId, RTagValue edited, bool global);
/// <summary>
/// get tagged publish pdfbooks (including CoverImage info but not pages or tagibutes info)
/// </summary>
/// <param name="tagUrl"></param>
/// <param name="valueUrl"></param>
/// <param name="statusArray"></param>
/// <returns></returns>
Task<RServiceResult<PDFBook[]>> GetPDFBookByTagValueAsync(string tagUrl, string valueUrl, PublishStatus[] statusArray);
/// <summary>
/// add author
/// </summary>
/// <param name="author"></param>
/// <returns></returns>
Task<RServiceResult<Author>> AddAuthorAsync(Author author);
/// <summary>
/// get author by id
/// </summary>
/// <param name="id"></param>
/// <returns></returns>
Task<RServiceResult<Author>> GetAuthorByIdAsync(int id);
/// <summary>
/// get authors
/// </summary>
/// <param name="paging"></param>
/// <param name="authorName"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, Author[] Authors)>> GetAuthorsAsync(PagingParameterModel paging, string authorName);
/// <summary>
/// update author
/// </summary>
/// <param name="model"></param>
/// <returns></returns>
Task<RServiceResult<Author>> UpdateAuthorAsync(Author model);
/// <summary>
/// delete author by id
/// </summary>
/// <param name="id"></param>
/// <returns></returns>
Task<RServiceResult<bool>> DeleteAuthorAsync(int id);
/// <summary>
/// add pdf book contributer
/// </summary>
/// <param name="pdfBookId"></param>
/// <param name="authorId"></param>
/// <param name="role"></param>
/// <returns></returns>
Task<RServiceResult<bool>> AddPDFBookContributerAsync(int pdfBookId, int authorId, string role);
/// <summary>
/// remove contribution from pdf book
/// </summary>
/// <param name="pdfBookId"></param>
/// <param name="contributionId"></param>
/// <returns></returns>
Task<RServiceResult<bool>> DeletePDFBookContributerAsync(int pdfBookId, int contributionId);
/// <summary>
/// get published pdf books by author
/// </summary>
/// <param name="paging"></param>
/// <param name="authorId"></param>
/// <param name="role"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, PDFBook[] Books)>> GetPublishedPDFBooksByAuthorAsync(PagingParameterModel paging, int authorId, string role);
/// <summary>
/// get published pdf books by author stats (group by role)
/// </summary>
/// <param name="authorId"></param>
/// <returns></returns>
Task<RServiceResult<AuthorRoleCount[]>> GetPublishedPDFBookbyAuthorGroupedByRoleAsync(int authorId);
/// <summary>
/// get all books
/// </summary>
/// <param name="paging"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, Book[] Books)>> GetAllBooksAsync(PagingParameterModel paging);
/// <summary>
/// add book
/// </summary>
/// <param name="book"></param>
/// <returns></returns>
Task<RServiceResult<Book>> AddBookAsync(Book book);
/// <summary>
/// update book
/// </summary>
/// <param name="model"></param>
/// <returns></returns>
Task<RServiceResult<Book>> UpdateBookAsync(Book model);
/// <summary>
/// delete book
/// </summary>
/// <param name="id"></param>
/// <returns></returns>
Task<RServiceResult<bool>> DeleteBookAsync(int id);
/// <summary>
/// add book author
/// </summary>
/// <param name="bookId"></param>
/// <param name="authorId"></param>
/// <param name="role"></param>
/// <returns></returns>
Task<RServiceResult<bool>> AddBookAuthorAsync(int bookId, int authorId, string role);
/// <summary>
/// remove author from book
/// </summary>
/// <param name="bookId"></param>
/// <param name="contributionId"></param>
/// <returns></returns>
Task<RServiceResult<bool>> DeleteBookAuthorAsync(int bookId, int contributionId);
/// <summary>
/// book by id
/// </summary>
/// <param name="id"></param>
/// <returns></returns>
Task<RServiceResult<Book>> GetBookByIdAsync(int id);
/// <summary>
/// get books by author
/// </summary>
/// <param name="paging"></param>
/// <param name="authorId"></param>
/// <param name="role"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, Book[] Books)>> GetBooksByAuthorAsync(PagingParameterModel paging, int authorId, string role);
/// <summary>
/// get books by author stats (group by role)
/// </summary>
/// <param name="authorId"></param>
/// <returns></returns>
Task<RServiceResult<AuthorRoleCount[]>> GetBookbyAuthorGroupedByRoleAsync(int authorId);
/// <summary>
/// get book related pdf books
/// </summary>
/// <param name="paging"></param>
/// <param name="bookId"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, PDFBook[] Books)>> GetBookRelatedPDFBooksAsync(PagingParameterModel paging, int bookId);
/// <summary>
/// add multi volume pdf collection
/// </summary>
/// <param name="multiVolumePDFCollection"></param>
/// <returns></returns>
Task<RServiceResult<MultiVolumePDFCollection>> AddMultiVolumePDFCollectionAsync(MultiVolumePDFCollection multiVolumePDFCollection);
/// <summary>
/// update multi volume pdf collection
/// </summary>
/// <param name="model"></param>
/// <returns></returns>
Task<RServiceResult<MultiVolumePDFCollection>> UpdateMultiVolumePDFCollectionAsync(MultiVolumePDFCollection model);
/// <summary>
/// delete multi volume pdf collection
/// </summary>
/// <param name="id"></param>
/// <returns></returns>
Task<RServiceResult<bool>> DeleteMultiVolumePDFCollectionAsync(int id);
/// <summary>
/// start importing local pdf file
/// </summary>
/// <param name="model"></param>
/// <returns></returns>
Task<RServiceResult<bool>> StartImportingLocalPDFAsync(NewPDFBookViewModel model);
/// <summary>
/// edit pdf book master record
/// </summary>
/// <param name="model"></param>
/// <param name="canChangeStatusToAwaiting"></param>
/// <param name="canPublish"></param>
/// <returns></returns>
Task<RServiceResult<PDFBook>> EditPDFBookMasterRecordAsync(PDFBook model, bool canChangeStatusToAwaiting, bool canPublish);
/// <summary>
/// Copy PDF Book Cover Image From Page Thumbnail image
/// </summary>
/// <param name="pdfBookId"></param>
/// <param name="pdfpageId"></param>
/// <returns></returns>
Task<RServiceResult<bool>> SetPDFBookCoverImageFromPageAsync(int pdfBookId, int pdfpageId);
/// <summary>
/// get volumes pdf books
/// </summary>
/// <param name="volumeId"></param>
/// <returns></returns>
Task<RServiceResult<PDFBook[]>> GetVolumesPDFBooks(int volumeId);
/// <summary>
/// get volumes by id
/// </summary>
/// <param name="id"></param>
/// <returns></returns>
Task<RServiceResult<MultiVolumePDFCollection>> GetMultiVolumePDFCollectionByIdAsync(int id);
/// <summary>
/// get pdf source by id
/// </summary>
/// <param name="id"></param>
/// <returns></returns>
Task<RServiceResult<PDFSource>> GetPDFSourceByIdAsync(int id);
/// <summary>
/// Get All PDF Sources
/// </summary>
/// <returns></returns>
Task<RServiceResult<PDFSource[]>> GetPDFSourcesAsync();
/// <summary>
/// Add PDF Source
/// </summary>
/// <param name="source"></param>
/// <returns></returns>
Task<RServiceResult<PDFSource>> AddPDFSourceAsync(PDFSource source);
/// <summary>
/// update PDF Source
/// </summary>
/// <param name="model"></param>
/// <returns></returns>
Task<RServiceResult<PDFSource>> UpdatePDFSourceAsync(PDFSource model);
/// <summary>
/// delete PDF Source
/// </summary>
/// <param name="id"></param>
/// <returns></returns>
Task<RServiceResult<bool>> DeletePDFSourceAsync(int id);
/// <summary>
/// get source pdf books
/// </summary>
/// <param name="paging"></param>
/// <param name="sourceId"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, PDFBook[] Books)>> GetSourceRelatedPDFBooksAsync(PagingParameterModel paging, int sourceId);
/// <summary>
/// batch import soha library
/// </summary>
/// <param name="start"></param>
/// <param name="end"></param>
/// <param name="finalizeDownload"></param>
void BatchImportSohaLibraryAsync(int start, int end, bool finalizeDownload);
/// <summary>
/// batch import eliteraturebook.com library
/// </summary>
/// <param name="ajaxPageIndexStart">from 0</param>
/// <param name="ajaxPageIndexEnd"></param>
/// <param name="finalizeDownload"></param>
void BatchImportELiteratureBookLibraryAsync(int ajaxPageIndexStart, int ajaxPageIndexEnd, bool finalizeDownload);
/// <summary>
/// search pdf books
/// </summary>
/// <param name="paging"></param>
/// <param name="term"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, PDFBook[] Items)>> SearchPDFBooksAsync(PagingParameterModel paging, string term);
/// <summary>
/// suggest ganjoor link
/// </summary>
/// <param name="userId"></param>
/// <param name="link"></param>
/// <returns></returns>
Task<RServiceResult<bool>> SuggestGanjoorLinkAsync(Guid userId, PDFGanjoorLinkSuggestion link);
/// <summary>
/// finds what the method name suggests
/// </summary>
/// <param name="skip"></param>
/// <returns></returns>
Task<RServiceResult<GanjoorLinkViewModel>> GetNextUnreviewedGanjoorLinkAsync(int skip);
/// <summary>
/// get unreviewed image count
/// </summary>
/// <returns></returns>
Task<RServiceResult<int>> GetUnreviewedGanjoorLinksCountAsync();
/// <summary>
/// Review Suggested Link
/// </summary>
/// <param name="linkId"></param>
/// <param name="userId"></param>
/// <param name="result"></param>
/// <returns></returns>
Task<RServiceResult<bool>> ReviewSuggestedLinkAsync(Guid linkId, Guid userId, ReviewResult result);
/// <summary>
/// get unsynced approved pdf ganjoor links
/// </summary>
/// <returns></returns>
Task<RServiceResult<PDFGanjoorLink[]>> GetUnsyncedPDFGanjoorLinksAsync();
/// <summary>
/// synchronize ganjoor link
/// </summary>
/// <param name="linkId"></param>
/// <returns></returns>
Task<RServiceResult<bool>> SynchronizePDFGanjoorLinkAsync(Guid linkId);
/// <summary>
/// get next un-ocred PDF Book
/// </summary>
/// <returns></returns>
Task<RServiceResult<PDFBook>> GetNextUnOCRedPDFBookAsync();
/// <summary>
/// reset OCR Queue (remove queued items)
/// </summary>
/// <returns></returns>
Task<RServiceResult<bool>> ResetOCRQueueAsync();
/// <summary>
/// set pdf page ocr info (and if a book whole pages are ocred the book ocred flag is set to true)
/// </summary>
/// <param name="model"></param>
/// <returns></returns>
Task<RServiceResult<bool>> SetPDFPageOCRInfoAsync(PDFPageOCRDataViewModel model);
/// <summary>
/// search pdf books pages for a text
/// </summary>
/// <param name="paging"></param>
/// <param name="term"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, PDFBook[] Books)>> SearchPDFBookForPDFPagesTextAsync(PagingParameterModel paging, string term);
/// <summary>
/// search pdf pages
/// </summary>
/// <param name="paging"></param>
/// <param name="bookId">0 for all pdf books</param>
/// <param name="term"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, PDFPage[] Items)>> SearchPDFPagesTextAsync(PagingParameterModel paging, int bookId, string term);
/// <summary>
/// get page by page number
/// </summary>
/// <param name="pdfBookId"></param>
/// <param name="pageNumber"></param>
/// <returns></returns>
Task<RServiceResult<PDFPage>> GetPDFPageAsync(int pdfBookId, int pageNumber);
/// <summary>
/// fill missing book texts
/// </summary>
void StartFillingMissingBookTextsAsync();
/// <summary>
/// queued downloding pdf books
/// </summary>
/// <param name="paging"></param>
/// <returns></returns>
Task<RServiceResult<(PaginationMetadata PagingMeta, QueuedPDFBook[] Books)>> GetQueuedPDFBooksAsync(PagingParameterModel paging);
/// <summary>
/// delete queued books
/// </summary>
/// <param name="id"></param>
/// <returns></returns>
Task<RServiceResult<bool>> DeleteQueuedPDFBookAsync(Guid id);
/// <summary>
/// mix queued pdf books
/// </summary>
/// <param name="step"></param>
/// <returns></returns>
Task<RServiceResult<bool>> MixQueuedPDFBooksAsync(int step);
/// <summary>
/// start processing queue pdf books
/// </summary>
/// <param name="count"></param>
void StartProcessingQueuedPDFBooks(int count);
}
}

View File

@ -1,4 +1,7 @@
using ganjoor;
/*
* removed
using ganjoor;
using Microsoft.EntityFrameworkCore;
using RMuseum.DbContext;
using RMuseum.Models.Artifact;
@ -309,3 +312,4 @@ namespace RMuseum.Services.Implementation
}
}
*/

View File

@ -1170,8 +1170,8 @@ namespace RMuseum.Services.Implementation
public async Task<RServiceResult<bool>> Import(string srcType, string resourceNumber, string friendlyUrl, string resourcePrefix, bool skipUpload)
{
return
srcType == "pdf" ?
await StartImportingLocalPDFFile(resourceNumber, friendlyUrl, resourcePrefix, skipUpload) :
/*srcType == "pdf" ?
await StartImportingLocalPDFFile(resourceNumber, friendlyUrl, resourcePrefix, skipUpload) :*/
srcType == "princeton" ?
await StartImportingFromPrinceton(resourceNumber, friendlyUrl, skipUpload)
:
@ -1387,9 +1387,9 @@ namespace RMuseum.Services.Implementation
scheduled.Add(job.ResourceNumber);
RServiceResult<bool> rescheduled =
job.JobType == JobType.Pdf ?
/*job.JobType == JobType.Pdf ?
await StartImportingLocalPDFFile(job.ResourceNumber, job.FriendlyUrl, job.SrcUrl, skipUpload)
:
:*/
job.JobType == JobType.Princeton ?
await StartImportingFromPrinceton(job.ResourceNumber, job.FriendlyUrl, skipUpload)
:

View File

@ -1,531 +0,0 @@
using ganjoor;
using Microsoft.EntityFrameworkCore;
using RMuseum.DbContext;
using RMuseum.Models.Artifact;
using RMuseum.Models.ImportJob;
using RMuseum.Models.PDFLibrary;
using RMuseum.Models.PDFLibrary.ViewModels;
using RSecurityBackend.Models.Generic;
using RSecurityBackend.Models.Image;
using System;
using System.Collections.Generic;
using System.Drawing;
using System.Drawing.Imaging;
using System.IO;
using System.Linq;
using System.Threading.Tasks;
namespace RMuseum.Services.Implementation
{
public partial class PDFLibraryService
{
/// <summary>
/// start importing local pdf file
/// </summary>
/// <param name="model"></param>
/// <returns></returns>
public async Task<RServiceResult<bool>> StartImportingLocalPDFAsync(NewPDFBookViewModel model)
{
try
{
if (model == null)
{
return new RServiceResult<bool>(false, "model == null");
}
if (!File.Exists(model.LocalImportingPDFFilePath))
{
return new RServiceResult<bool>(false, $"file does not exist! : {model.LocalImportingPDFFilePath}");
}
string fileChecksum = PoemAudio.ComputeCheckSum(model.LocalImportingPDFFilePath);
if (
(
await _context.ImportJobs
.Where(j => j.JobType == JobType.Pdf && j.SrcContent == fileChecksum && !(j.Status == ImportJobStatus.Failed || j.Status == ImportJobStatus.Aborted))
.SingleOrDefaultAsync()
)
!=
null
)
{
return new RServiceResult<bool>(false, $"Job is already scheduled or running for importing pdf file {model.LocalImportingPDFFilePath} (duplicated checksum: {fileChecksum})");
}
_backgroundTaskQueue.QueueBackgroundWorkItem
(
async token =>
{
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>()))
{
await ImportLocalPDFFileAsync(context, model, fileChecksum);
}
}
);
return new RServiceResult<bool>(true);
}
catch (Exception exp)
{
return new RServiceResult<bool>(false, exp.ToString());
}
}
private async Task<RServiceResult<PDFBook>> ImportLocalPDFFileAsync(RMuseumDbContext context, NewPDFBookViewModel model, string fileChecksum)
{
try
{
var pdfRes = await ImportLocalPDFFileAsync(context, model.BookId, model.MultiVolumePDFCollectionId, model.VolumeOrder, model.LocalImportingPDFFilePath, model.OriginalSourceUrl, model.SkipUpload, fileChecksum);
if (pdfRes.Result != null)
{
var pdfBook = pdfRes.Result;
pdfBook.Title = model.Title;
pdfBook.SubTitle = model.SubTitle;
pdfBook.AuthorsLine = model.AuthorsLine;
pdfBook.ISBN = model.ISBN;
pdfBook.Description = model.Description;
pdfBook.IsTranslation = model.IsTranslation;
pdfBook.TranslatorsLine = model.TranslatorsLine;
pdfBook.TitleInOriginalLanguage = model.TitleInOriginalLanguage;
pdfBook.PublisherLine = model.PublisherLine;
pdfBook.PublishingDate = model.PublishingDate;
pdfBook.PublishingLocation = model.PublishingLocation;
pdfBook.PublishingNumber = model.PublishingNumber == 0 ? null : model.PublishingNumber;
pdfBook.ClaimedPageCount = model.ClaimedPageCount == 0 ? null : model.ClaimedPageCount;
pdfBook.OriginalSourceName = model.OriginalSourceName;
pdfBook.OriginalFileUrl = model.OriginalFileUrl;
pdfBook.PDFSourceId = model.PDFSourceId;
pdfBook.Language = model.Language;
pdfBook.BookScriptType = model.BookScriptType;
List<AuthorRole> roles = new List<AuthorRole>();
if (model.WriterId != null)
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.WriterId).SingleAsync(),
Role = "نویسنده",
});
}
if (model.Writer2Id != null)
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.Writer2Id).SingleAsync(),
Role = "نویسنده",
});
}
if (model.Writer3Id != null)
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.Writer3Id).SingleAsync(),
Role = "نویسنده",
});
}
if (model.Writer4Id != null)
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.Writer4Id).SingleAsync(),
Role = "نویسنده",
});
}
if (model.TranslatorId != null)
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.TranslatorId).SingleAsync(),
Role = "مترجم",
});
}
if (model.Translator2Id != null)
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.Translator2Id).SingleAsync(),
Role = "مترجم",
});
}
if (model.Translator3Id != null)
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.Translator3Id).SingleAsync(),
Role = "مترجم",
});
}
if (model.Translator4Id != null)
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.Translator4Id).SingleAsync(),
Role = "مترجم",
});
}
if (model.CollectorId != null)
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.CollectorId).SingleAsync(),
Role = "مصحح",
});
}
if (model.Collector2Id != null)
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.Collector2Id).SingleAsync(),
Role = "مصحح",
});
}
if (model.OtherContributerId != null && !string.IsNullOrEmpty(model.OtherContributerRole))
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.OtherContributerId).SingleAsync(),
Role = model.OtherContributerRole,
});
}
if (model.OtherContributer2Id != null && !string.IsNullOrEmpty(model.OtherContributer2Role))
{
roles.Add(new AuthorRole()
{
Author = await context.Authors.Where(a => a.Id == model.OtherContributer2Id).SingleAsync(),
Role = model.OtherContributer2Role,
});
}
if (roles.Count > 0)
{
pdfBook.Contributers = roles;
}
context.Update(pdfBook);
await context.SaveChangesAsync();
return new RServiceResult<PDFBook>(pdfBook);
}
return pdfRes;
}
catch (Exception exp)
{
return new RServiceResult<PDFBook>(null, exp.ToString());
}
}
private async Task<RServiceResult<PDFBook>> ImportLocalPDFFileAsync(RMuseumDbContext context, int bookId, int? volumeId, int volumeOrder, string filePath, string srcUrl, bool skipUpload, string fileChecksum)
{
try
{
ImportJob job = new ImportJob()
{
JobType = JobType.Pdf,
SrcContent = fileChecksum,
ResourceNumber = filePath,
FriendlyUrl = "",
SrcUrl = srcUrl,
QueueTime = DateTime.Now,
ProgressPercent = 0,
Status = ImportJobStatus.NotStarted
};
await context.ImportJobs.AddAsync
(
job
);
await context.SaveChangesAsync();
if (volumeId == 0) volumeId = null;
if (!string.IsNullOrEmpty(srcUrl))
{
if (
(await context.PDFBooks.Where(a => a.OriginalSourceUrl == srcUrl).SingleOrDefaultAsync())
!=
null
)
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"duplicated srcUrl '{srcUrl}'";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<PDFBook>(null, job.Exception);
}
}
if (
(await context.PDFBooks.Where(a => a.FileMD5CheckSum == fileChecksum).SingleOrDefaultAsync())
!=
null
)
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"duplicated pdf with checksum '{fileChecksum}'";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<PDFBook>(null, job.Exception);
}
try
{
//this code fails on empty database, but it is not important for me!
string folderNumber = (1 + await context.PDFBooks.MaxAsync(p => p.Id)).ToString().PadLeft(8, '0');
Directory.CreateDirectory(Path.Combine(_imageFileService.ImageStoragePath, folderNumber));
PDFBook pdfBook = new PDFBook()
{
Status = PublishStatus.Draft,
DateTime = DateTime.Now,
LastModified = DateTime.Now,
FileMD5CheckSum = fileChecksum,
OriginalSourceUrl = srcUrl,
OriginalFileName = Path.GetFileName(filePath),
StorageFolderName = folderNumber,
BookId = bookId,
VolumeOrder = volumeOrder,
MultiVolumePDFCollectionId = volumeId,
};
job.FriendlyUrl = pdfBook.StorageFolderName;
List<RTagValue> meta = new List<RTagValue>();
job.StartTime = DateTime.Now;
job.Status = ImportJobStatus.Running;
job.SrcContent = "";
context.Update(job);
await context.SaveChangesAsync();
List<PDFPage> pages = await _ImportAndReturnPDFJobImages(pdfBook, job, 0);
pdfBook.Tags = meta.ToArray();
pdfBook.Pages = pages.ToArray();
pdfBook.PageCount = pages.Count;
if (pages.Count == 0)
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = "Pages.Count == 0";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<PDFBook>(null, job.Exception);
}
using (FileStream fs = new FileStream(filePath, FileMode.Open))
{
var pdfStorageResult = await _imageFileService.Add(null, fs, pdfBook.OriginalFileName, pdfBook.StorageFolderName, false, "application/pdf");
if (!string.IsNullOrEmpty(pdfStorageResult.ExceptionString))
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"pdfStorageResult.ExceptionString: {pdfStorageResult.ExceptionString}";
job.EndTime = DateTime.Now;
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<PDFBook>(null, job.Exception);
}
pdfBook.PDFFile = pdfStorageResult.Result;
}
await context.PDFBooks.AddAsync(pdfBook);
await context.SaveChangesAsync();
var resFTPUpload = await _UploadPDFBookToExternalServer(pdfBook, context, skipUpload);
if (!string.IsNullOrEmpty(resFTPUpload.ExceptionString))
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"UploadArtifactToExternalServer: {resFTPUpload.ExceptionString}";
job.EndTime = DateTime.Now;
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<PDFBook>(null, job.Exception);
}
var book = await context.Books.Where(b => b.Id == bookId).SingleAsync();
if (book.CoverImageId == null)
{
book.CoverImage = RImage.DuplicateExcludingId(pdfBook.CoverImage);
book.ExtenalCoverImageUrl = pdfBook.ExtenalCoverImageUrl;
context.Update(book);
await context.SaveChangesAsync();
}
job.ProgressPercent = 100;
job.Status = ImportJobStatus.Succeeded;
job.EndTime = DateTime.Now;
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<PDFBook>(pdfBook);
}
catch (Exception exp)
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = exp.ToString();
job.EndTime = DateTime.Now;
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<PDFBook>(null, job.Exception);
}
}
catch(Exception e)
{
return new RServiceResult<PDFBook>(null, e.ToString());
}
}
private async Task<List<PDFPage>> _ImportAndReturnPDFJobImages(PDFBook pdfBook, ImportJob job, int order)
{
List<PDFPage> pages = new List<PDFPage>();
string pdfFilePath = job.ResourceNumber;
string intermediateFolder = Path.Combine(Path.GetDirectoryName(pdfFilePath), Path.GetFileNameWithoutExtension(pdfFilePath));
try
{
Directory.CreateDirectory(intermediateFolder);
}
catch
{
intermediateFolder = Path.Combine(Path.GetDirectoryName(pdfFilePath), $"{pdfBook.Id}-catch");
}
List<string> fileNames = new List<string>();
int imageOrder = 1;
using (FileStream fs = File.OpenRead(pdfFilePath))
{
var skBitmaps = PDFtoImage.Conversion.ToImages(fs);
foreach (var skBitmap in skBitmaps)
{
string outFileName = Path.Combine(intermediateFolder, $"{imageOrder}".PadLeft(4, '0') + ".jpg");
using (FileStream fsOut = File.OpenWrite(outFileName))
{
skBitmap.Encode(fsOut, SkiaSharp.SKEncodedImageFormat.Jpeg, 90);
}
fileNames.Add(outFileName);
imageOrder++;
}
}
foreach (string fileName in fileNames)
{
using (RMuseumDbContext importJobUpdaterDb = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>()))
{
job.ProgressPercent = order * 100 / (decimal)fileNames.Count;
importJobUpdaterDb.Update(job);
await importJobUpdaterDb.SaveChangesAsync();
}
order++;
PDFPage page = new PDFPage()
{
PageNumber = order,
Description = "",
Tags = new RTagValue[] { },
LastModified = DateTime.Now
};
string thumbnailImageName = $"{page.PageNumber}".PadLeft(5, '0') + ".jpg";
string thumbnailPath = Path.Combine(Path.Combine(_imageFileService.ImageStoragePath, pdfBook.StorageFolderName), thumbnailImageName);
if (File.Exists(thumbnailPath))
{
File.Delete(thumbnailPath);
}
using (Image img = Image.FromFile(fileName))
{
page.FullResolutionImageWidth = img.Width;
page.FullResolutionImageHeight = img.Height;
int imageWidth = img.Width;
int imageHeight = img.Height;
int thumbnailImageWidth = ThumbnailImageWidth;
int thumbnailImageHeight = ThumbnailImageWidth * imageHeight / imageWidth;
if (thumbnailImageHeight > ThumbnailImageMaxHeight)
{
thumbnailImageHeight = ThumbnailImageMaxHeight;
thumbnailImageWidth = thumbnailImageHeight * imageWidth / imageHeight;
}
//روی خود img تأثیر می گذارد
// به احتمال قوی داریم روی تصویر resize کار می‌کنیم
Image thumbnail = new Bitmap(thumbnailImageWidth, thumbnailImageHeight);
using (Graphics gThumbnail = Graphics.FromImage(thumbnail))
{
gThumbnail.DrawImage(img, 0, 0, thumbnailImageWidth, thumbnailImageHeight);
}
using (MemoryStream msThumbnail = new MemoryStream())
{
thumbnail.Save(msThumbnail, ImageFormat.Jpeg);
msThumbnail.Seek(0, SeekOrigin.Begin);
RServiceResult<RImage> picture = await _imageFileService.Add(null, msThumbnail, thumbnailImageName, pdfBook.StorageFolderName);
if (picture.Result == null)
{
throw new Exception($"_imageFileService.Add : {picture.ExceptionString}");
}
page.ThumbnailImage = picture.Result;
if (page.PageNumber == 1)
{
pdfBook.CoverImage = RImage.DuplicateExcludingId(picture.Result);
}
}
}
pages.Add(page);
}
foreach (var fileName in fileNames)
{
File.Delete(fileName);
}
Directory.Delete(intermediateFolder, true);
return pages;
}
private async Task<RServiceResult<bool>> _UploadPDFBookToExternalServer(PDFBook book, RMuseumDbContext context, bool skipUpload)
{
try
{
var localPDFFilePath = _imageFileService.GetImagePath(book.PDFFile).Result;
var remotePDFFilePath = $"{Configuration.GetSection("ExternalFTPServer")["RootPath"]}/pdf/{book.StorageFolderName}/{Path.GetFileName(localPDFFilePath)}";
if (!skipUpload)
{
var res = await _ftpService.AddAsync(context, localPDFFilePath, remotePDFFilePath, true);
if(!string.IsNullOrEmpty(res.ExceptionString))
{
return new RServiceResult<bool>(false, $"ftp {localPDFFilePath} => {remotePDFFilePath} {res.ExceptionString}");
}
}
var localCoverImageFilePath = _imageFileService.GetImagePath(book.CoverImage).Result;
var remoteCoverImageFilePath = $"{Configuration.GetSection("ExternalFTPServer")["RootPath"]}/pdf/{book.StorageFolderName}/{Path.GetFileName(localCoverImageFilePath)}";
if (!skipUpload)
{
var res = await _ftpService.AddAsync(context, localCoverImageFilePath, remoteCoverImageFilePath, true);
if (!string.IsNullOrEmpty(res.ExceptionString))
{
return new RServiceResult<bool>(false, $"ftp {localCoverImageFilePath} => {remoteCoverImageFilePath} {res.ExceptionString}");
}
}
book.ExternalPDFFileUrl = $"{Configuration.GetSection("ExternalFTPServer")["RootUrl"]}/pdf/{book.StorageFolderName}/{Path.GetFileName(localPDFFilePath)}";
book.ExtenalCoverImageUrl = $"{Configuration.GetSection("ExternalFTPServer")["RootUrl"]}/pdf/{book.StorageFolderName}/{Path.GetFileName(localCoverImageFilePath)}";
context.Update(book);
foreach (var item in book.Pages)
{
var localFilePath = _imageFileService.GetImagePath(item.ThumbnailImage).Result;
item.ExtenalThumbnailImageUrl = $"{Configuration.GetSection("ExternalFTPServer")["RootUrl"]}/pdf/{book.StorageFolderName}/{Path.GetFileName(localFilePath)}";
context.Update(item);
if (!skipUpload)
{
var remoteFilePath = $"{Configuration.GetSection("ExternalFTPServer")["RootPath"]}/pdf/{book.CoverImage.FolderName}/{Path.GetFileName(localFilePath)}";
var res = await _ftpService.AddAsync(context, localFilePath, remoteFilePath, true);
if (!string.IsNullOrEmpty(res.ExceptionString))
{
return new RServiceResult<bool>(false, $"ftp {localFilePath} => {remoteFilePath} {res.ExceptionString}");
}
}
}
if (!skipUpload)
{
if(false == await context.QueuedFTPUploads.AsNoTracking().Where(p => p.Processing).AnyAsync())
{
var res = await _ftpService.ProcessQueueAsync(context);
if (!string.IsNullOrEmpty(res.ExceptionString))
{
return new RServiceResult<bool>(false, $"FTP ProcessQueueAsync : {res.ExceptionString}");
}
}
}
return new RServiceResult<bool>(true);
}
catch (Exception exp)
{
return new RServiceResult<bool>(false, exp.ToString());
}
}
/// <summary>
/// عرض تصویر بندانگشتی
/// </summary>
protected int ThumbnailImageWidth { get { return int.Parse($"{Configuration.GetSection("PictureFileService")["ThumbnailImageWidth"]}"); } }
/// <summary>
/// طول تصویر بندانگشتی
/// </summary>
protected int ThumbnailImageMaxHeight { get { return int.Parse($"{Configuration.GetSection("PictureFileService")["ThumbnailMaxHeight"]}"); } }
}
}

View File

@ -1,142 +0,0 @@
using Microsoft.EntityFrameworkCore;
using RMuseum.DbContext;
using RMuseum.Models.PDFLibrary;
using RSecurityBackend.Models.Generic;
using RSecurityBackend.Services.Implementation;
using System;
using System.Linq;
using System.Threading.Tasks;
namespace RMuseum.Services.Implementation
{
public partial class PDFLibraryService
{
/// <summary>
/// queued downloding pdf books
/// </summary>
/// <param name="paging"></param>
/// <returns></returns>
public async Task<RServiceResult<(PaginationMetadata PagingMeta, QueuedPDFBook[] Books)>> GetQueuedPDFBooksAsync(PagingParameterModel paging)
{
try
{
var source =
_context.QueuedPDFBooks.AsNoTracking()
.OrderBy(t => t.DownloadOrder)
.AsQueryable();
(PaginationMetadata PagingMeta, QueuedPDFBook[] Books) paginatedResult =
await QueryablePaginator<QueuedPDFBook>.Paginate(source, paging);
return new RServiceResult<(PaginationMetadata PagingMeta, QueuedPDFBook[] Books)>(paginatedResult);
}
catch (Exception exp)
{
return new RServiceResult<(PaginationMetadata PagingMeta, QueuedPDFBook[] Books)>((null, null), exp.ToString());
}
}
/// <summary>
/// delete queued books
/// </summary>
/// <param name="id"></param>
/// <returns></returns>
public async Task<RServiceResult<bool>> DeleteQueuedPDFBookAsync(Guid id)
{
try
{
var qb = await _context.QueuedPDFBooks.Where(t => t.Id == id).SingleAsync();
_context.Remove(qb);
await _context.SaveChangesAsync();
return new RServiceResult<bool>(true);
}
catch (Exception exp)
{
return new RServiceResult<bool>(false, exp.ToString());
}
}
/// <summary>
/// mix queued pdf books
/// </summary>
/// <param name="step"></param>
/// <returns></returns>
public async Task<RServiceResult<bool>> MixQueuedPDFBooksAsync(int step)
{
try
{
var processed = await _context.QueuedPDFBooks.Where(t => t.ResultId != 0).ToListAsync();
if(processed.Count > 0)
{
_context.RemoveRange(processed);
await _context.SaveChangesAsync();
}
var qSoha = await _context.QueuedPDFBooks.Where(t => t.OriginalSourceUrl.Contains("https://sohalibrary.com")).OrderBy(t => t.DownloadOrder).ToListAsync();
var qElit = await _context.QueuedPDFBooks.Where(t => !t.OriginalSourceUrl.Contains("https://sohalibrary.com")).OrderBy(t => t.DownloadOrder).ToListAsync();
int downloadOrder = 0;
int e = 0;
for (var i = 0; i < qSoha.Count; i++)
{
qSoha[i].DownloadOrder = downloadOrder;
qSoha[i].Processed = false;
downloadOrder++;
if (i % step == 0)
{
if (e < qElit.Count)
{
qElit[e].DownloadOrder = downloadOrder;
qElit[e].Processed = false;
downloadOrder++;
e++;
}
}
}
for (var i = e; e < qElit.Count; e++)
{
qElit[e].DownloadOrder = downloadOrder;
qElit[e].Processed = false;
downloadOrder++;
}
_context.UpdateRange(qSoha);
_context.UpdateRange(qElit);
await _context.SaveChangesAsync();
return new RServiceResult<bool>(true);
}
catch (Exception exp)
{
return new RServiceResult<bool>(false, exp.ToString());
}
}
/// <summary>
/// start processing queue pdf books
/// </summary>
/// <param name="count"></param>
public void StartProcessingQueuedPDFBooks(int count)
{
_backgroundTaskQueue.QueueBackgroundWorkItem
(
async token =>
{
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>()))
{
var jobs = await context.ImportJobs.ToListAsync();
context.RemoveRange(jobs);
await context.SaveChangesAsync();
var q = await context.QueuedPDFBooks.Where(i => i.Processed == false).OrderBy(i => i.DownloadOrder).Take(count).ToListAsync();
foreach (var item in q)
{
var res = await ImportfFromKnownSourceAsync(context, item.OriginalSourceUrl);
item.Processed = true;
item.ProcessResult = res.ExceptionString;
item.ResultId = res.Result;
context.Update(item);
await context.SaveChangesAsync();
}
}
}
);
}
}
}

View File

@ -1,674 +0,0 @@
using DNTPersianUtils.Core;
using ganjoor;
using Microsoft.EntityFrameworkCore;
using RMuseum.DbContext;
using RMuseum.Models.Artifact;
using RMuseum.Models.ImportJob;
using RMuseum.Models.PDFLibrary;
using RMuseum.Models.PDFLibrary.ViewModels;
using RSecurityBackend.Models.Generic;
using System;
using System.Collections.Generic;
using System.IO;
using System.Linq;
using System.Net.Http;
using System.Text.RegularExpressions;
using System.Threading.Tasks;
namespace RMuseum.Services.Implementation
{
public partial class PDFLibraryService
{
private async Task<RServiceResult<int>> StartImportingSohaLibraryUrlAsync(RMuseumDbContext ctx, string srcUrl)
{
try
{
if (
(await ctx.PDFBooks.Where(a => a.OriginalSourceUrl == srcUrl).SingleOrDefaultAsync())
!=
null
)
{
return new RServiceResult<int>(-1, $"duplicated srcUrl '{srcUrl}'");
}
if (
(
await ctx.ImportJobs
.Where(j => j.JobType == JobType.Pdf && j.SrcContent == ("scrapping ..." + srcUrl) && !(j.Status == ImportJobStatus.Failed || j.Status == ImportJobStatus.Aborted))
.SingleOrDefaultAsync()
)
!=
null
)
{
return new RServiceResult<int>(0, $"Job is already scheduled or running for importing source url: {srcUrl}");
}
return await ImportSohaLibraryUrlAsync(srcUrl, ctx, true);
}
catch (Exception exp)
{
return new RServiceResult<int>(0, exp.ToString());
}
}
private async Task<RServiceResult<int>> ImportSohaLibraryUrlAsync(string srcUrl, RMuseumDbContext context, bool finalizeDownload)
{
/*
var oldJobs = await context.ImportJobs.ToArrayAsync();
context.RemoveRange(oldJobs);
await context.SaveChangesAsync();
*/
ImportJob job = new ImportJob()
{
JobType = JobType.Pdf,
SrcContent = "",
ResourceNumber = "scrapping ..." + srcUrl,
FriendlyUrl = "",
SrcUrl = "",
QueueTime = DateTime.Now,
ProgressPercent = 0,
Status = ImportJobStatus.NotStarted
};
await context.ImportJobs.AddAsync
(
job
);
await context.SaveChangesAsync();
try
{
var pdfSource = await context.PDFSources.Where(s => s.Name == "سها").SingleOrDefaultAsync();
if (pdfSource == null)
{
PDFSource newSource = new PDFSource()
{
Name = "سها",
Url = "https://sohalibrary.com",
Description = "سامانهٔ جستجوی یکپارچهٔ نرم‌افزار سنا"
};
context.PDFSources.Add(newSource);
await context.SaveChangesAsync();
pdfSource = await context.PDFSources.Where(s => s.Name == "سها").SingleAsync();
}
NewPDFBookViewModel model = new NewPDFBookViewModel();
model.PDFSourceId = pdfSource.Id;
model.OriginalSourceName = pdfSource.Name;
model.OriginalSourceUrl = srcUrl;
model.BookScriptType = BookScriptType.Printed;
model.Language = "فارسی";
model.SkipUpload = true;
string html = "";
using (var client = new HttpClient())
{
using (var result = await client.GetAsync(srcUrl))
{
if (result.IsSuccessStatusCode)
{
html = await result.Content.ReadAsStringAsync();
}
else
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"Http result is not ok ({result.StatusCode}) for {srcUrl}";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
}
}
if (html.IndexOf("/item/download/") == -1)
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"/item/download/ not found in html source.";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
List<RTagValue> meta = new List<RTagValue>();
int idxStart;
int idx = html.IndexOf("branch-link");
string firstHandSource = "";
if (idx != -1)
{
idxStart = html.IndexOf(">", idx);
if (idxStart != -1)
{
int idxEnd = html.IndexOf("<", idxStart);
if (idxEnd != -1)
{
firstHandSource = html.Substring(idxStart + 1, idxEnd - idxStart - 1);
meta.Add
(
await TagHandler.PrepareAttribute(context, "First Hand Source", firstHandSource, 1)
);
}
}
}
idx = html.IndexOf("title-normal-for-book-name");
if (idx == -1)
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"title-normal-for-book-name not found in {srcUrl}";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
idxStart = html.IndexOf(">", idx);
if (idxStart != -1)
{
int idxEnd = html.IndexOf("<", idxStart);
if (idxEnd != -1)
{
model.Title = html.Substring(idxStart + 1, idxEnd - idxStart - 1);
//we can try to extract volume information from title here
model.Title = model.Title.Replace("\n", "").Replace("\r", "").Trim();
}
}
idxStart = html.IndexOf("width-150");
while (idxStart != -1)
{
idxStart = html.IndexOf(">", idxStart);
if (idxStart == -1) break;
int idxEnd = html.IndexOf("<", idxStart);
if (idxEnd == -1) break;
string tagName = html.Substring(idxStart + 1, idxEnd - idxStart - 1).ToPersianNumbers().ApplyCorrectYeKe().Trim();
idxStart = html.IndexOf("value-name", idxEnd);
if (idxStart == -1) break;
idxStart = html.IndexOf(">", idxStart);
if (idxStart == -1) break;
idxEnd = html.IndexOf("</span>", idxStart);
if (idxEnd == -1) break;
string tagValue = html.Substring(idxStart + 1, idxEnd - idxStart - 1).Trim();
tagValue = Regex.Replace(tagValue, "<.*?>", string.Empty).Replace("\n", "").Replace("\r", "").Trim();
string tagValueCleaned = tagValue.Replace("\n", "").Replace("\r", "").Trim().ToPersianNumbers().ApplyCorrectYeKe();
if (tagName == "نویسنده")
{
model.AuthorsLine = tagValueCleaned;
tagName = "Author";
string[] authors = tagValueCleaned.Split(',', StringSplitOptions.RemoveEmptyEntries);
for(int i = 0; i < authors.Length; i++)
{
string authorTrimmed = authors[i].Trim();
var existingAuthor = await context.Authors.AsNoTracking().Where(a => a.Name == authorTrimmed).FirstOrDefaultAsync();
if (existingAuthor != null)
{
if(i == 0)
{
model.WriterId = existingAuthor.Id;
}
if(i == 1)
{
model.Writer2Id = existingAuthor.Id;
}
if(i == 2)
{
model.Writer3Id = existingAuthor.Id;
}
if(i == 3)
{
model.Writer4Id = existingAuthor.Id;
}
}
else
{
var newAuthor = new Author()
{
Name = authorTrimmed
};
context.Authors.Add(newAuthor);
await context.SaveChangesAsync();
if(i == 0)
{
model.WriterId = newAuthor.Id;
}
if(i == 1)
{
model.Writer2Id = newAuthor.Id;
}
if(i == 2)
{
model.Writer3Id = newAuthor.Id;
}
if(i == 3)
{
model.Writer4Id = newAuthor.Id;
}
}
}
}
if (tagName == "مترجم")
{
model.TranslatorsLine = tagValueCleaned;
model.IsTranslation = true;
tagName = "Translator";
string[] authors = tagValueCleaned.Split(',', StringSplitOptions.RemoveEmptyEntries);
for(int i = 0; i < authors.Length; i++)
{
string authorTrimmed = authors[i].Trim();
var existingAuthor = await context.Authors.AsNoTracking().Where(a => a.Name == authorTrimmed).FirstOrDefaultAsync();
if (existingAuthor != null)
{
if( i == 0 )
{
model.TranslatorId = existingAuthor.Id;
}
if( i == 1)
{
model.Translator2Id = existingAuthor.Id;
}
if( i == 2)
{
model.Translator3Id = existingAuthor.Id;
}
if( i == 3)
{
model.Translator4Id = existingAuthor.Id;
}
}
else
{
var newAuthor = new Author()
{
Name = authorTrimmed
};
context.Authors.Add(newAuthor);
await context.SaveChangesAsync();
if(i == 0)
{
model.TranslatorId = newAuthor.Id;
}
if(i == 1)
{
model.Translator2Id = newAuthor.Id;
}
if(i == 2)
{
model.Translator3Id = newAuthor.Id;
}
if(i == 3)
{
model.Translator4Id = newAuthor.Id;
}
}
}
}
if (tagName == "مصحح")
{
tagName = "Collector";
string[] authors = tagValueCleaned.Split(',', StringSplitOptions.RemoveEmptyEntries);
for(int i=0; i<authors.Length; i++)
{
string authorTrimmed = authors[i].Trim();
var existingAuthor = await context.Authors.AsNoTracking().Where(a => a.Name == authorTrimmed).FirstOrDefaultAsync();
if (existingAuthor != null)
{
if(i == 0)
{
model.CollectorId = existingAuthor.Id;
}
if(i == 1)
{
model.Collector2Id = existingAuthor.Id;
}
}
else
{
var newAuthor = new Author()
{
Name = authorTrimmed
};
context.Authors.Add(newAuthor);
await context.SaveChangesAsync();
if (i == 0)
{
model.CollectorId = newAuthor.Id;
}
if (i == 1)
{
model.Collector2Id = newAuthor.Id;
}
}
}
}
if (tagName == "زبان")
{
model.Language = tagValueCleaned;
tagName = "Language";
if (!tagValueCleaned.Contains("فارسی"))
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = "Language is not فارسی";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
}
if (tagName == "شماره جلد")
{
if (int.TryParse(tagValue, out int v))
{
model.VolumeOrder = v;
}
tagName = "Volume";
}
if (tagName == "ناشر")
{
model.PublisherLine = tagValueCleaned;
tagName = "Publisher";
}
if (tagName == "محل نشر")
{
model.PublishingLocation = tagValueCleaned;
tagName = "Publishing Location";
}
if (tagName == "تاریخ انتشار")
{
model.PublishingDate = tagValueCleaned;
tagName = "Publishing Date";
}
if (tagName == "موضوع")
{
tagName = "Subject";
}
if (tagName == "تعداد صفحات")
{
if (tagValue.Contains("ـ"))
{
tagValue = tagValue.Substring(tagValue.IndexOf("ـ") + 1).Trim();
}
if (int.TryParse(tagValue, out int v))
{
model.ClaimedPageCount = v;
}
tagName = "Page Count";
}
meta.Add
(
await TagHandler.PrepareAttribute(context, tagName, tagValueCleaned, 1)
);
idxStart = html.IndexOf("width-150", idxEnd);
}
string bookTitle = model.Title;
int volumeNumber = 0;
if (bookTitle.Contains("ـ ج"))
{
bookTitle = bookTitle.Substring(0, model.Title.IndexOf("ـ ج") - 1);
int.TryParse(model.Title.Substring(model.Title.IndexOf("ـ ج") + "ـ ج".Length).Trim(), out volumeNumber);
}
bookTitle = bookTitle.ToPersianNumbers().ApplyCorrectYeKe().Trim();
model.Title = model.Title.ToPersianNumbers().ApplyCorrectYeKe().Trim();
var book = await context.Books.AsNoTracking().Where(b => b.Name == bookTitle).FirstOrDefaultAsync();
if (book != null)
{
model.BookId = book.Id;
}
else
{
Book newBook = new Book()
{
Name = bookTitle,
Description = "",
LastModified = DateTime.Now,
};
context.Books.Add(newBook);
await context.SaveChangesAsync();
model.BookId = newBook.Id;
}
if (volumeNumber != 0)
{
MultiVolumePDFCollection collection = await context.MultiVolumePDFCollections.Where(v => v.Name == bookTitle && v.BookId == model.BookId).SingleOrDefaultAsync();
if (collection != null)
{
model.MultiVolumePDFCollectionId = collection.Id;
collection.VolumeCount += 1;
context.Update(collection);
await context.SaveChangesAsync();
}
else
{
MultiVolumePDFCollection newCollection = new MultiVolumePDFCollection()
{
Name = bookTitle,
BookId = model.BookId,
Description = "",
VolumeCount = 1,
};
context.MultiVolumePDFCollections.Add(newCollection);
await context.SaveChangesAsync();
model.MultiVolumePDFCollectionId = newCollection.Id;
}
}
idx = html.IndexOf("/item/download/");
int idxQuote = html.IndexOf('"', idx);
string downloadUrl = html.Substring(idx, idxQuote - idx);
downloadUrl = "https://sohalibrary.com" + downloadUrl;
model.OriginalFileUrl = downloadUrl;
if(!finalizeDownload)
{
context.QueuedPDFBooks.Add
(
new QueuedPDFBook()
{
Title = model.Title,
DownloadOrder = await context.QueuedPDFBooks.CountAsync(),
AuthorsLine = model.AuthorsLine,
Description = model.Description,
Language = model.Language,
IsTranslation = model.IsTranslation,
TranslatorsLine = model.TranslatorsLine,
OriginalSourceName = model.OriginalSourceName,
OriginalSourceUrl = model.OriginalSourceUrl,
OriginalFileUrl = model.OriginalFileUrl,
Processed = false,
}
);
await context.SaveChangesAsync();
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Succeeded;
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
using (var client = new HttpClient())
{
using (var result = await client.GetAsync(downloadUrl))
{
if (result.IsSuccessStatusCode)
{
string fileName = (1 + await context.PDFBooks.MaxAsync(p => p.Id)).ToString().PadLeft(8, '0') + "-soha.pdf";
/*
string fileName = string.Empty;
if (result.Content.Headers.ContentDisposition != null)
{
fileName = System.Net.WebUtility.UrlDecode(result.Content.Headers.ContentDisposition.FileName).Replace("\"", "");
}
if (string.IsNullOrEmpty(fileName) || File.Exists(Path.Combine(_imageFileService.ImageStoragePath, fileName)))
{
fileName = downloadUrl.Substring(downloadUrl.LastIndexOf('/') + 1) + ".pdf";
}
if(fileName.Length > 70)
{
fileName = Path.GetFileNameWithoutExtension(fileName).Substring(0, 64) + ".pdf";
}
*/
model.LocalImportingPDFFilePath = Path.Combine(_imageFileService.ImageStoragePath, fileName);
if (File.Exists(model.LocalImportingPDFFilePath))
File.Delete(model.LocalImportingPDFFilePath);
using (Stream pdfStream = await result.Content.ReadAsStreamAsync())
{
pdfStream.Seek(0, SeekOrigin.Begin);
using (FileStream fs = File.OpenWrite(model.LocalImportingPDFFilePath))
{
pdfStream.CopyTo(fs);
}
}
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Succeeded;
context.Update(job);
await context.SaveChangesAsync();
string fileChecksum = PoemAudio.ComputeCheckSum(model.LocalImportingPDFFilePath);
var res = await ImportLocalPDFFileAsync(context, model, fileChecksum);
File.Delete(model.LocalImportingPDFFilePath);
if (res.Result != null)
{
var pdf = await context.PDFBooks.Include(p => p.Tags).Where(p => p.Id == res.Result.Id).SingleAsync();
if (pdf.Tags.Count > 0)
{
//not working
foreach (var tag in meta)
{
pdf.Tags.Add(tag);
}
}
else
{
pdf.Tags = meta;
}
context.Update(pdf);
await context.SaveChangesAsync();
return new RServiceResult<int>(res.Result.Id);
}
else
{
if(string.IsNullOrEmpty(res.ExceptionString))
{
return new RServiceResult<int>(0, "ImportLocalPDFFileAsync result was null");
}
if (res.ExceptionString.Contains("duplicated pdf with checksum"))
{
return new RServiceResult<int>(-2, res.ExceptionString);
}
return new RServiceResult<int>(0, res.ExceptionString);
}
}
else
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"Http result is not ok ({result.StatusCode}) for {downloadUrl}";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
}
}
}
catch (Exception e)
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = e.ToString();
job.EndTime = DateTime.Now;
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
}
/// <summary>
/// batch import soha library
/// </summary>
/// <param name="start"></param>
/// <param name="end"></param>
/// <param name="finalizeDownload"></param>
public void BatchImportSohaLibraryAsync(int start, int end, bool finalizeDownload)
{
_backgroundTaskQueue.QueueBackgroundWorkItem
(
async token =>
{
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>()))
{
for (int nUrlIndex = start; nUrlIndex <= end; nUrlIndex++)
{
string srcUrl = $"https://sohalibrary.com/item/view/{nUrlIndex}";
if (
(await context.PDFBooks.Where(a => a.OriginalSourceUrl == srcUrl).SingleOrDefaultAsync())
!=
null
)
{
continue;
}
try
{
await ImportSohaLibraryUrlAsync(srcUrl, context, finalizeDownload);
}
catch
{
//ignore
}
}
}
}
);
}
}
}

View File

@ -1,755 +0,0 @@
using DNTPersianUtils.Core;
using ganjoor;
using Microsoft.EntityFrameworkCore;
using RMuseum.DbContext;
using RMuseum.Models.Artifact;
using RMuseum.Models.ImportJob;
using RMuseum.Models.PDFLibrary;
using RMuseum.Models.PDFLibrary.ViewModels;
using RSecurityBackend.Models.Generic;
using System;
using System.Collections.Generic;
using System.IO;
using System.Linq;
using System.Net.Http;
using System.Text.RegularExpressions;
using System.Threading.Tasks;
namespace RMuseum.Services.Implementation
{
public partial class PDFLibraryService
{
private async Task<RServiceResult<int>> StartImportingELiteratureBookUrlAsync(RMuseumDbContext ctx, string srcUrl)
{
try
{
if (
(await ctx.PDFBooks.Where(a => a.OriginalSourceUrl == srcUrl).SingleOrDefaultAsync())
!=
null
)
{
return new RServiceResult<int>(-1, $"duplicated srcUrl '{srcUrl}'");
}
if (
(
await ctx.ImportJobs
.Where(j => j.JobType == JobType.Pdf && j.SrcContent == ("scrapping ..." + srcUrl) && !(j.Status == ImportJobStatus.Failed || j.Status == ImportJobStatus.Aborted))
.SingleOrDefaultAsync()
)
!=
null
)
{
return new RServiceResult<int>(0, $"Job is already scheduled or running for importing source url: {srcUrl}");
}
return await ImportELiteratureBookLibraryUrlAsync(srcUrl, ctx, true);
}
catch (Exception exp)
{
return new RServiceResult<int>(0, exp.ToString());
}
}
private async Task<RServiceResult<int>> ImportELiteratureBookLibraryUrlAsync(string srcUrl, RMuseumDbContext context, bool finalizeDownload)
{
/*
var oldJobs = await context.ImportJobs.ToArrayAsync();
context.RemoveRange(oldJobs);
await context.SaveChangesAsync();
*/
ImportJob job = new ImportJob()
{
JobType = JobType.Pdf,
SrcContent = "",
ResourceNumber = "scrapping ..." + srcUrl,
FriendlyUrl = "",
SrcUrl = "",
QueueTime = DateTime.Now,
ProgressPercent = 0,
Status = ImportJobStatus.NotStarted
};
await context.ImportJobs.AddAsync
(
job
);
await context.SaveChangesAsync();
try
{
var pdfSource = await context.PDFSources.Where(s => s.Name == "کتابخانهٔ مجازی ادبیات").SingleOrDefaultAsync();
if (pdfSource == null)
{
PDFSource newSource = new PDFSource()
{
Name = "کتابخانهٔ مجازی ادبیات",
Url = "https://eliteraturebook.com",
Description = "کتابخانهٔ مجازی ادبیات"
};
context.PDFSources.Add(newSource);
await context.SaveChangesAsync();
pdfSource = await context.PDFSources.Where(s => s.Name == "کتابخانهٔ مجازی ادبیات").SingleAsync();
}
NewPDFBookViewModel model = new NewPDFBookViewModel();
model.PDFSourceId = pdfSource.Id;
model.OriginalSourceName = pdfSource.Name;
model.OriginalSourceUrl = srcUrl;
model.BookScriptType = BookScriptType.Printed;
model.Language = "فارسی";
model.SkipUpload = true;
string html = "";
using (var client = new HttpClient())
{
using (var result = await client.GetAsync(srcUrl))
{
if (result.IsSuccessStatusCode)
{
html = await result.Content.ReadAsStringAsync();
}
else
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"Http result is not ok ({result.StatusCode}) for {srcUrl}";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
}
}
if (html.IndexOf("/download") == -1)
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"/download/ not found in html source.";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
List<RTagValue> meta = new List<RTagValue>();
int idxStart;
meta.Add
(
await TagHandler.PrepareAttribute(context, "First Hand Source", "کتابخانه تخصصی ادبیات", 1)
);
int idx = html.IndexOf("\"book-title\"");
if (idx == -1)
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"\"book-title\" not found in {srcUrl}";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
idxStart = html.IndexOf(">", idx);
if (idxStart != -1)
{
int idxEnd = html.IndexOf("<", idxStart);
if (idxEnd != -1)
{
model.Title = html.Substring(idxStart + 1, idxEnd - idxStart - 1);
//we can try to extract volume information from title here
model.Title = model.Title.Replace("\n", "").Replace("\r", "").Trim();
}
}
idxStart = html.IndexOf("box summary-box");
if(idxStart != -1)
{
idxStart = html.IndexOf("<p>", idxStart);
if(idxStart != -1)
{
idxStart += "<p>".Length;
int idxEnd = html.IndexOf("<", idxStart);
if (idxEnd != -1)
{
model.Description = html.Substring(idxStart + 1, idxEnd - idxStart - 1).ApplyCorrectYeKe();
model.Description = model.Description.Trim();
}
}
}
string tagValue;
string tagName;
idx = html.IndexOf("\"author-name\"");
if (idx != -1)
{
idx = html.IndexOf("ref=", idx);
idxStart = html.IndexOf(">", idx);
if (idxStart != -1)
{
int idxEnd = html.IndexOf("<", idxStart);
if (idxEnd != -1)
{
tagValue = html.Substring(idxStart + 1, idxEnd - idxStart - 1);
tagValue = Regex.Replace(tagValue, "<.*?>", string.Empty).Trim();
string tagValueCleaned = tagValue.ToPersianNumbers().ApplyCorrectYeKe();
tagName = "نویسنده";
if (tagName == "نویسنده")
{
model.AuthorsLine = tagValueCleaned;
tagName = "Author";
string[] authors = tagValueCleaned.Split(',', StringSplitOptions.RemoveEmptyEntries);
for (int i = 0; i < authors.Length; i++)
{
string authorTrimmed = authors[i].Trim();
var existingAuthor = await context.Authors.AsNoTracking().Where(a => a.Name == authorTrimmed).FirstOrDefaultAsync();
if (existingAuthor != null)
{
if (i == 0)
{
model.WriterId = existingAuthor.Id;
}
if (i == 1)
{
model.Writer2Id = existingAuthor.Id;
}
if (i == 2)
{
model.Writer3Id = existingAuthor.Id;
}
if (i == 3)
{
model.Writer4Id = existingAuthor.Id;
}
}
else
{
var newAuthor = new Author()
{
Name = authorTrimmed
};
context.Authors.Add(newAuthor);
await context.SaveChangesAsync();
if (i == 0)
{
model.WriterId = newAuthor.Id;
}
if (i == 1)
{
model.Writer2Id = newAuthor.Id;
}
if (i == 2)
{
model.Writer3Id = newAuthor.Id;
}
if (i == 3)
{
model.Writer4Id = newAuthor.Id;
}
}
}
}
}
}
}
idxStart = html.IndexOf("\"part view\"");
while (idxStart != -1)
{
idxStart = html.IndexOf(">", idxStart);
if (idxStart == -1) break;
int idxEnd = html.IndexOf("<", idxStart);
if (idxEnd == -1) break;
tagName = html.Substring(idxStart + 1, idxEnd - idxStart - 1).Replace(":", "").ToPersianNumbers().ApplyCorrectYeKe().Trim();
idxStart = html.IndexOf(">", idxEnd);
if (idxStart == -1) break;
idxEnd = html.IndexOf("<", idxStart);
if (idxEnd == -1) break;
tagValue = html.Substring(idxStart + 1, idxEnd - idxStart - 1);
tagValue = Regex.Replace(tagValue, "<.*?>", string.Empty).Replace("\n", "").Replace("\r", "").Trim();
string tagValueCleaned = tagValue.Replace("\n", "").Replace("\r", "").Trim().ToPersianNumbers().ApplyCorrectYeKe();
if (tagName == "مترجم")
{
model.TranslatorsLine = tagValueCleaned;
model.IsTranslation = true;
tagName = "Translator";
string[] authors = tagValueCleaned.Split(',', StringSplitOptions.RemoveEmptyEntries);
for (int i = 0; i < authors.Length; i++)
{
string authorTrimmed = authors[i].Trim();
var existingAuthor = await context.Authors.AsNoTracking().Where(a => a.Name == authorTrimmed).FirstOrDefaultAsync();
if (existingAuthor != null)
{
if (i == 0)
{
model.TranslatorId = existingAuthor.Id;
}
if (i == 1)
{
model.Translator2Id = existingAuthor.Id;
}
if (i == 2)
{
model.Translator3Id = existingAuthor.Id;
}
if (i == 3)
{
model.Translator4Id = existingAuthor.Id;
}
}
else
{
var newAuthor = new Author()
{
Name = authorTrimmed
};
context.Authors.Add(newAuthor);
await context.SaveChangesAsync();
if (i == 0)
{
model.TranslatorId = newAuthor.Id;
}
if (i == 1)
{
model.Translator2Id = newAuthor.Id;
}
if (i == 2)
{
model.Translator3Id = newAuthor.Id;
}
if (i == 3)
{
model.Translator4Id = newAuthor.Id;
}
}
}
}
if (tagName == "مصحح")
{
tagName = "Collector";
string[] authors = tagValueCleaned.Split(',', StringSplitOptions.RemoveEmptyEntries);
for (int i = 0; i < authors.Length; i++)
{
string authorTrimmed = authors[i].Trim();
var existingAuthor = await context.Authors.AsNoTracking().Where(a => a.Name == authorTrimmed).FirstOrDefaultAsync();
if (existingAuthor != null)
{
if (i == 0)
{
model.CollectorId = existingAuthor.Id;
}
if (i == 1)
{
model.Collector2Id = existingAuthor.Id;
}
}
else
{
var newAuthor = new Author()
{
Name = authorTrimmed
};
context.Authors.Add(newAuthor);
await context.SaveChangesAsync();
if (i == 0)
{
model.CollectorId = newAuthor.Id;
}
if (i == 1)
{
model.Collector2Id = newAuthor.Id;
}
}
}
}
if (tagName == "زبان")
{
model.Language = tagValueCleaned;
tagName = "Language";
if (!tagValueCleaned.Contains("فارسی"))
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = "Language is not فارسی";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
}
if (tagName == "شماره جلد")
{
if (int.TryParse(tagValue, out int v))
{
model.VolumeOrder = v;
}
tagName = "Volume";
}
if (tagName == "ناشر")
{
model.PublisherLine = tagValueCleaned;
tagName = "Publisher";
}
if (tagName == "محل نشر")
{
model.PublishingLocation = tagValueCleaned;
tagName = "Publishing Location";
}
if (tagName == "تاریخ انتشار")
{
model.PublishingDate = tagValueCleaned;
tagName = "Publishing Date";
}
if (tagName == "موضوع")
{
tagName = "Subject";
}
if (tagName == "تعداد صفحات")
{
if (tagValue.Contains("ـ"))
{
tagValue = tagValue.Substring(tagValue.IndexOf("ـ") + 1).Trim();
}
if (int.TryParse(tagValue, out int v))
{
model.ClaimedPageCount = v;
}
tagName = "Page Count";
}
meta.Add
(
await TagHandler.PrepareAttribute(context, tagName, tagValueCleaned, 1)
);
idxStart = html.IndexOf("\"part view\"", idxEnd);
}
idx = html.IndexOf("\"tag\"");
while(idx != -1)
{
idxStart = html.IndexOf(">", idx);
if (idxStart != -1)
{
int idxEnd = html.IndexOf("<", idxStart);
if (idxEnd != -1)
{
var tv = html.Substring(idxStart + 1, idxEnd - idxStart - 1).Replace("\n", "").Replace("\r", "").Trim().ApplyCorrectYeKe();
meta.Add
(
await TagHandler.PrepareAttribute(context, "Subject", tv, 1)
);
}
}
idx = html.IndexOf("\"tag\"", idxStart);
}
string bookTitle = model.Title;
int volumeNumber = 0;
if (bookTitle.Contains("ـ ج"))
{
bookTitle = bookTitle.Substring(0, model.Title.IndexOf("ـ ج") - 1);
int.TryParse(model.Title.Substring(model.Title.IndexOf("ـ ج") + "ـ ج".Length).Trim(), out volumeNumber);
}
bookTitle = bookTitle.ToPersianNumbers().ApplyCorrectYeKe().Trim();
model.Title = model.Title.ToPersianNumbers().ApplyCorrectYeKe().Trim();
var book = await context.Books.AsNoTracking().Where(b => b.Name == bookTitle).FirstOrDefaultAsync();
if (book != null)
{
model.BookId = book.Id;
}
else
{
Book newBook = new Book()
{
Name = bookTitle,
Description = model.Description,
LastModified = DateTime.Now,
};
context.Books.Add(newBook);
await context.SaveChangesAsync();
model.BookId = newBook.Id;
}
if (volumeNumber != 0)
{
MultiVolumePDFCollection collection = await context.MultiVolumePDFCollections.Where(v => v.Name == bookTitle && v.BookId == model.BookId).SingleOrDefaultAsync();
if (collection != null)
{
model.MultiVolumePDFCollectionId = collection.Id;
collection.VolumeCount += 1;
context.Update(collection);
await context.SaveChangesAsync();
}
else
{
MultiVolumePDFCollection newCollection = new MultiVolumePDFCollection()
{
Name = bookTitle,
BookId = model.BookId,
Description = model.Description,
VolumeCount = 1,
};
context.MultiVolumePDFCollections.Add(newCollection);
await context.SaveChangesAsync();
model.MultiVolumePDFCollectionId = newCollection.Id;
}
}
idx = html.IndexOf("https://eliteraturebook.com/books/download");
int idxQuote = html.IndexOf('"', idx);
string downloadUrl = html.Substring(idx, idxQuote - idx);
model.OriginalFileUrl = downloadUrl;
if (!finalizeDownload)
{
context.QueuedPDFBooks.Add
(
new QueuedPDFBook()
{
Title = model.Title,
DownloadOrder = await context.QueuedPDFBooks.CountAsync(),
AuthorsLine = model.AuthorsLine,
Description = model.Description,
Language = model.Language,
IsTranslation = model.IsTranslation,
TranslatorsLine = model.TranslatorsLine,
OriginalSourceName = model.OriginalSourceName,
OriginalSourceUrl = model.OriginalSourceUrl,
OriginalFileUrl = model.OriginalFileUrl,
Processed = false,
}
);
await context.SaveChangesAsync();
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Succeeded;
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
using (var client = new HttpClient())
{
using (var result = await client.GetAsync(downloadUrl))
{
if (result.IsSuccessStatusCode)
{
string fileName = (1 + await context.PDFBooks.MaxAsync(p => p.Id)).ToString().PadLeft(8, '0') + "-eliteraturebook.pdf";
model.LocalImportingPDFFilePath = Path.Combine(_imageFileService.ImageStoragePath, fileName);
if (File.Exists(model.LocalImportingPDFFilePath))
File.Delete(model.LocalImportingPDFFilePath);
using (Stream pdfStream = await result.Content.ReadAsStreamAsync())
{
pdfStream.Seek(0, SeekOrigin.Begin);
using (FileStream fs = File.OpenWrite(model.LocalImportingPDFFilePath))
{
pdfStream.CopyTo(fs);
}
}
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Succeeded;
context.Update(job);
await context.SaveChangesAsync();
string fileChecksum = PoemAudio.ComputeCheckSum(model.LocalImportingPDFFilePath);
var res = await ImportLocalPDFFileAsync(context, model, fileChecksum);
File.Delete(model.LocalImportingPDFFilePath);
if (res.Result != null)
{
var pdf = await context.PDFBooks.Include(p => p.Tags).Where(p => p.Id == res.Result.Id).SingleAsync();
if (pdf.Tags.Count > 0)
{
//not working
foreach (var tag in meta)
{
pdf.Tags.Add(tag);
}
}
else
{
pdf.Tags = meta;
}
context.Update(pdf);
await context.SaveChangesAsync();
return new RServiceResult<int>(res.Result.Id);
}
else
{
if (string.IsNullOrEmpty(res.ExceptionString))
{
return new RServiceResult<int>(0, "ImportLocalPDFFileAsync result was null");
}
if(res.ExceptionString.Contains("duplicated pdf with checksum"))
{
return new RServiceResult<int>(-2, res.ExceptionString);
}
return new RServiceResult<int>(0, res.ExceptionString);
}
}
else
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"Http result is not ok ({result.StatusCode}) for {downloadUrl}";
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
}
}
}
catch (Exception e)
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = e.ToString();
job.EndTime = DateTime.Now;
context.Update(job);
await context.SaveChangesAsync();
return new RServiceResult<int>(0, job.Exception);
}
}
/// <summary>
/// batch import eliteraturebook.com library
/// </summary>
/// <param name="ajaxPageIndexStart">from 0</param>
/// <param name="ajaxPageIndexEnd"></param>
/// <param name="finalizeDownload"></param>
public void BatchImportELiteratureBookLibraryAsync(int ajaxPageIndexStart, int ajaxPageIndexEnd, bool finalizeDownload)
{
_backgroundTaskQueue.QueueBackgroundWorkItem
(
async token =>
{
using (RMuseumDbContext context = new RMuseumDbContext(new DbContextOptions<RMuseumDbContext>()))
{
string ajaxPageUrl = "https://eliteraturebook.com/index";
for (int nAjaxPageIndex = ajaxPageIndexEnd; nAjaxPageIndex >= ajaxPageIndexStart; nAjaxPageIndex--)
{
try
{
ImportJob job = new ImportJob()
{
JobType = JobType.Pdf,
SrcContent = "",
ResourceNumber = $"scrapping ajax page ... {nAjaxPageIndex}",
FriendlyUrl = "",
SrcUrl = "",
QueueTime = DateTime.Now,
ProgressPercent = 0,
Status = ImportJobStatus.NotStarted
};
await context.ImportJobs.AddAsync
(
job
);
await context.SaveChangesAsync();
string html = string.Empty;
using (var client = new HttpClient())
{
using (var result = await client.PostAsync(ajaxPageUrl,
new StringContent($"ajaxPage={nAjaxPageIndex}")
))
{
if (result.IsSuccessStatusCode)
{
html = await result.Content.ReadAsStringAsync();
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Succeeded;
context.Update(job);
await context.SaveChangesAsync();
}
else
{
job.EndTime = DateTime.Now;
job.Status = ImportJobStatus.Failed;
job.Exception = $"Http result is not ok ({result.StatusCode}) for ajaxPage={nAjaxPageIndex}";
context.Update(job);
await context.SaveChangesAsync();
return;
}
}
}
if(!string.IsNullOrEmpty(html))
{
int idx = html.IndexOf("https://eliteraturebook.com/books/view/");
while(idx != -1)
{
int endIdx = html.IndexOf("\"", idx);
string srcUrl = html.Substring(idx, endIdx - idx);
idx = html.IndexOf("https://eliteraturebook.com/books/view/", idx + 1);
if (
(await context.PDFBooks.Where(a => a.OriginalSourceUrl == srcUrl).SingleOrDefaultAsync())
==
null
)
{
await ImportELiteratureBookLibraryUrlAsync(srcUrl, context, finalizeDownload);
}
}
}
}
catch
{
//ignore
}
}
}
}
);
}
}
}

File diff suppressed because it is too large Load Diff

View File

@ -303,8 +303,6 @@ namespace RMuseum
//faq service
services.AddTransient<IFAQService, FAQService>();
//PDF library service
services.AddTransient<IPDFLibraryService, PDFLibraryService>();
//Queued FTP Upload Service
services.AddTransient<IQueuedFTPUploadService, QueuedFTPUploadService>();