I have a input file as below which has digitized OCR text
https://drive.google.com/drive/folders/1mAzjcHKX1tsKhNvTtF8InhkXFZbmdbKz?usp=sharing
This PDF has person details starting from page3
The expected output is to based on the certain locations I am trying to accomplish using the array of zones of PDF locations since all of pages have text at same zones
The expected output array is "1","1YO6963656","व वनिता सुनील भोईर',"तीचे नाव. सुनील भोईर","घर क्रमांक", "32 लिंग:महिला" ... ...
I have tried with
<script src="https://cdnjs.cloudflare.com/ajax/libs/pdf.js/2.12.313/pdf.min.js" integrity="sha512-qa1o08MA0596eSNsnkRv5vuGloSKUhY09O31MY2OJpODjUVlaL0GOJJcyt7J7Z61FiEgHMgBkH04ZJ+vcuLs/w==" crossorigin="anonymous" referrerpolicy="no-referrer"></script>
<script src="https://cdnjs.cloudflare.com/ajax/libs/pdf.js/2.12.313/pdf.worker.min.js" integrity="sha512-S9Dwzi4TCjPQkxlaXsqQLj2gXUjPZk4HBzE7zWU6Itc1r2RNmlBrVLH4EsYQrdnzLgvkN8P7l9SCru+2I4rZwg==" crossorigin="anonymous" referrerpolicy="no-referrer"></script>
<label for="myfile">Select a file:</label>
<input type="file" id="file" name="file">
<script>
/*Offical release of the pdfjs worker*/
pdfjsLib.GlobalWorkerOptions.workerSrc = 'https://cdnjs.cloudflare.com/ajax/libs/pdf.js/2.5.207/pdf.worker.js';
document.getElementById('file').onchange = function(event) {
separator = ' '
var file = event.target.files[0];
var fileReader = new FileReader();
fileReader.onload = function() {
var typedarray = new Uint8Array(this.result);
console.log(typedarray);
let pdf = pdfjsLib.getDocument(typedarray);
return pdf.promise.then(function(pdf) { // get all pages text
let maxPages = pdf._pdfInfo.numPages;
let countPromises = []; // collecting all page promises
for (let i = 1; i <= maxPages; i++) {
let page = pdf.getPage(i);
countPromises.push(page.then(function(page) { // add page promise
let textContent = page.getTextContent();
return textContent.then(function(text) { // return content promise
return text.items.map(function(obj) {
return obj.str;
}).join(separator); // value page text
});
}));
};
return Promise.all(countPromises).then(function(texts) {
for(let i = 0; i < texts.length; i++){
texts[i] = texts[i].replace(/\s+/g, ' ').trim();
};
console.log(texts)
alert(texts)
return texts;
});
});
}
fileReader.readAsArrayBuffer(file);
}
</script>