Parse From URL Asynchronously - Java
Document Parser sample in Java demonstrating ‘Parse From URL Asynchronously’
MultiPageTable-template1.yml
templateName: Multipage Table Test
templateVersion: 4
templatePriority: 0
detectionRules:
keywords:
- Sample document with multi-page table
objects:
- name: total
objectType: field
fieldProperties:
fieldType: macros
expression: TOTAL{{Spaces}}({{Number}})
regex: true
dataType: decimal
- name: table1
objectType: table
tableProperties:
start:
expression: Item{{Spaces}}Description{{Spaces}}Price
regex: true
end:
expression: TOTAL{{Spaces}}{{Number}}
regex: true
row:
expression: '{{LineStart}}{{Spaces}}(?<itemNo>{{Digits}}){{Spaces}}(?<description>{{SentenceWithSingleSpaces}}){{Spaces}}(?<price>{{Number}}){{Spaces}}(?<qty>{{Digits}}){{Spaces}}(?<extPrice>{{Number}})'
regex: true
columns:
- name: itemNo
dataType: integer
- name: description
dataType: string
- name: price
dataType: decimal
- name: qty
dataType: integer
- name: extPrice
dataType: decimal
multipage: true
result.json
{
"objects": [
{
"name": "total",
"objectType": "field",
"value": 450.00,
"pageIndex": 1,
"rectangle": [
0.0,
0.0,
0.0,
0.0
]
},
{
"objectType": "table",
"name": "table1",
"rows": [
{
"itemNo": {
"pageIndex": 0,
"value": 1
},
"description": {
"pageIndex": 0,
"value": "Item 1"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 2
},
"description": {
"pageIndex": 0,
"value": "Item 2"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 3
},
"description": {
"pageIndex": 0,
"value": "Item 3"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 4
},
"description": {
"pageIndex": 0,
"value": "Item 4"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 5
},
"description": {
"pageIndex": 0,
"value": "Item 5"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 6
},
"description": {
"pageIndex": 0,
"value": "Item 6"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 7
},
"description": {
"pageIndex": 0,
"value": "Item 7"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 8
},
"description": {
"pageIndex": 0,
"value": "Item 8"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 9
},
"description": {
"pageIndex": 0,
"value": "Item 9"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 10
},
"description": {
"pageIndex": 0,
"value": "Item 10"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 11
},
"description": {
"pageIndex": 0,
"value": "Item 11"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 12
},
"description": {
"pageIndex": 0,
"value": "Item 12"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 13
},
"description": {
"pageIndex": 0,
"value": "Item 13"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 14
},
"description": {
"pageIndex": 0,
"value": "Item 14"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 15
},
"description": {
"pageIndex": 0,
"value": "Item 15"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 16
},
"description": {
"pageIndex": 0,
"value": "Item 16"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 17
},
"description": {
"pageIndex": 0,
"value": "Item 17"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 18
},
"description": {
"pageIndex": 0,
"value": "Item 18"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 19
},
"description": {
"pageIndex": 0,
"value": "Item 19"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 20
},
"description": {
"pageIndex": 0,
"value": "Item 20"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 21
},
"description": {
"pageIndex": 0,
"value": "Item 21"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 22
},
"description": {
"pageIndex": 0,
"value": "Item 22"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 23
},
"description": {
"pageIndex": 0,
"value": "Item 23"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 24
},
"description": {
"pageIndex": 0,
"value": "Item 24"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 25
},
"description": {
"pageIndex": 0,
"value": "Item 25"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 26
},
"description": {
"pageIndex": 0,
"value": "Item 26"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 27
},
"description": {
"pageIndex": 0,
"value": "Item 27"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 28
},
"description": {
"pageIndex": 0,
"value": "Item 28"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 29
},
"description": {
"pageIndex": 0,
"value": "Item 29"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 0,
"value": 30
},
"description": {
"pageIndex": 0,
"value": "Item 30"
},
"price": {
"pageIndex": 0,
"value": 10.00
},
"qty": {
"pageIndex": 0,
"value": 1
},
"extPrice": {
"pageIndex": 0,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 31
},
"description": {
"pageIndex": 1,
"value": "Item 31"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 32
},
"description": {
"pageIndex": 1,
"value": "Item 32"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 33
},
"description": {
"pageIndex": 1,
"value": "Item 33"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 34
},
"description": {
"pageIndex": 1,
"value": "Item 34"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 35
},
"description": {
"pageIndex": 1,
"value": "Item 35"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 36
},
"description": {
"pageIndex": 1,
"value": "Item 36"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 37
},
"description": {
"pageIndex": 1,
"value": "Item 37"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 38
},
"description": {
"pageIndex": 1,
"value": "Item 38"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 39
},
"description": {
"pageIndex": 1,
"value": "Item 39"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 40
},
"description": {
"pageIndex": 1,
"value": "Item 40"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 41
},
"description": {
"pageIndex": 1,
"value": "Item 41"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 42
},
"description": {
"pageIndex": 1,
"value": "Item 42"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 43
},
"description": {
"pageIndex": 1,
"value": "Item 43"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 44
},
"description": {
"pageIndex": 1,
"value": "Item 44"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
},
{
"itemNo": {
"pageIndex": 1,
"value": 45
},
"description": {
"pageIndex": 1,
"value": "Item 45"
},
"price": {
"pageIndex": 1,
"value": 10.00
},
"qty": {
"pageIndex": 1,
"value": 1
},
"extPrice": {
"pageIndex": 1,
"value": 10.00
}
}
]
}
],
"templateName": "Multipage Table Test",
"templateVersion": "4",
"timestamp": "2020-05-18T12:00:37"
}
Main.java
package com.company;
import com.google.gson.JsonObject;
import com.google.gson.JsonParser;
import com.google.gson.JsonPrimitive;
import okhttp3.*;
import java.io.File;
import java.io.FileOutputStream;
import java.io.IOException;
import java.io.OutputStream;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.time.LocalDateTime;
import java.time.format.DateTimeFormatter;
public class Main {
// The authentication key (API Key).
// Get your own by registering at https://app.pdf.co
final static String API_KEY = "********************************";
// (!) Make asynchronous job
final static boolean Async = true;
public static void main(String[] args) throws IOException {
// Source PDF file
// You can also upload your own file into PDF.co and use it as url. Check "Upload File" samples for code snippets: https://github.com/bytescout/pdf-co-api-samples/tree/master/File%20Upload/
final String SourceFileUrl = "https://bytescout-com.s3.amazonaws.com/files/demo-files/cloud-api/document-parser/MultiPageTable.pdf";
// PDF document password. Leave empty for unprotected documents.
final String Password = "";
// Destination JSON file name
final Path DestinationFile = Paths.get(".\\result.json");
// Template text. Use Document Parser (https://pdf.co/document-parser, https://app.pdf.co/document-parser)
// to create templates.
// Read template from file:
String templateText = new String(Files.readAllBytes(Paths.get(".\\MultiPageTable-template1.yml")), StandardCharsets.UTF_8);
// Create HTTP client instance
OkHttpClient webClient = new OkHttpClient();
// PARSE PDF DOCUMENT
ParseDocument(webClient, DestinationFile, Password, SourceFileUrl, templateText);
}
public static void ParseDocument(OkHttpClient webClient, Path destinationFile,
String password, String uploadedFileUrl, String templateText) throws IOException {
// Prepare POST request body in JSON format
JsonObject jsonBody = new JsonObject();
jsonBody.add("url", new JsonPrimitive(uploadedFileUrl));
jsonBody.add("template", new JsonPrimitive(templateText));
RequestBody body = RequestBody.create(MediaType.parse("application/json"), jsonBody.toString());
// Prepare URL for Document parser API call.
// See documentation: https://apidocs.pdf.co/?#1-pdfdocumentparser
String query = String.format("https://api.pdf.co/v1/pdf/documentparser?async=%s", Async);
DateTimeFormatter dtf = DateTimeFormatter.ofPattern("MM/dd/yyyy HH:mm:ss");
// Prepare request to `Document Parser` API
Request request = new Request.Builder()
.url(query)
.addHeader("x-api-key", API_KEY) // (!) Set API Key
.addHeader("Content-Type", "application/json")
.post(body)
.build();
// Execute request
Response response = webClient.newCall(request).execute();
if (response.code() == 200) {
// Parse JSON response
JsonObject json = new JsonParser().parse(response.body().string()).getAsJsonObject();
boolean error = json.get("error").getAsBoolean();
if (!error) {
// Asynchronous job ID
String jobId = json.get("jobId").getAsString();
System.out.println("Job#" + jobId + ": has been created. - " + dtf.format(LocalDateTime.now()));
// URL of generated json file that will available after the job completion
String resultFileUrl = json.get("url").getAsString();
// Check the job status in a loop.
// If you don't want to pause the main thread you can rework the code
// to use a separate thread for the status checking and completion.
do {
String status = CheckJobStatus(webClient, jobId); // Possible statuses: "working", "failed", "aborted", "success"
System.out.println("Job#" + jobId + ": " + status + " - " + dtf.format(LocalDateTime.now()));
if (status.compareToIgnoreCase("success") == 0) {
// Download JSON file
downloadFile(webClient, resultFileUrl, destinationFile.toFile());
System.out.printf("Generated JSON file saved as \"%s\" file.", destinationFile.toString());
break;
} else if (status.compareToIgnoreCase("working") == 0) {
// Pause for a few seconds
try {
Thread.sleep(3000);
} catch (InterruptedException ex) {
Thread.currentThread().interrupt(); // restore interrupted status
}
} else {
System.out.println(status);
break;
}
} while (true);
} else {
// Display service reported error
System.out.println(json.get("message").getAsString());
}
} else {
// Display request error
System.out.println(response.code() + " " + response.message());
}
}
// Check Job Status
private static String CheckJobStatus(OkHttpClient webClient, String jobId) throws IOException {
String url = "https://api.pdf.co/v1/job/check?jobid=" + jobId;
// Prepare request
Request request = new Request.Builder()
.url(url)
.addHeader("x-api-key", API_KEY) // (!) Set API Key
.build();
// Execute request
Response response = webClient.newCall(request).execute();
if (response.code() == 200) {
// Parse JSON response
JsonObject json = new JsonParser().parse(response.body().string()).getAsJsonObject();
return json.get("status").getAsString();
} else {
// Display request error
System.out.println(response.code() + " " + response.message());
}
return "Failed";
}
public static boolean uploadFile(OkHttpClient webClient, String apiKey, String url, Path sourceFile) throws IOException {
// Prepare request body
RequestBody body = RequestBody.create(MediaType.parse("application/octet-stream"), sourceFile.toFile());
// Prepare request
Request request = new Request.Builder()
.url(url)
.addHeader("x-api-key", apiKey) // (!) Set API Key
.addHeader("content-type", "application/octet-stream")
.put(body)
.build();
// Execute request
Response response = webClient.newCall(request).execute();
return (response.code() == 200);
}
public static void downloadFile(OkHttpClient webClient, String url, File destinationFile) throws IOException {
// Prepare request
Request request = new Request.Builder()
.url(url)
.build();
// Execute request
Response response = webClient.newCall(request).execute();
byte[] fileBytes = response.body().bytes();
// Save downloaded bytes to file
OutputStream output = new FileOutputStream(destinationFile);
output.write(fileBytes);
output.flush();
output.close();
response.close();
}
}
PDF.co Web API: the Web API with a set of tools for documents manipulation, data conversion, data extraction, splitting and merging of documents. Includes image recognition, built-in OCR, barcode generation and barcode decoders to decode bar codes from scans, pictures and pdf.
Download Source Code (.zip)
return to the previous page explore Document Parser endpoint
Copyright © 2016 - 2023 PDF.co