所有文档

          文字识别

          通用机打发票识别

          接口描述

          支持对国家/地方税务局发行的横/竖版通用机打发票的23个关键字段进行结构化识别,包括发票类型、发票号码、发票代码、开票日期、合计金额大写、合计金额小写、商品名称、商品单位、商品单价、商品数量、商品金额、机打代码、机打号码、校验码、销售方名称、销售方纳税人识别号、购买方名称、购买方纳税人识别号、合计税额等。

          请求说明

          请求示例

          HTTP 方法:POST

          请求URL: https://aip.baidubce.com/rest/2.0/ocr/v1/invoice

          URL参数:

          参数
          access_token 通过API Key和Secret Key获取的access_token,参考“Access Token获取

          Header如下:

          参数
          Content-Type application/x-www-form-urlencoded

          Body中放置请求参数,参数详情如下:

          请求参数

          参数 是否必选 类型 可选值范围 说明
          image string - 图像数据,base64编码后进行urlencode,要求base64编码和urlencode后大小不超过4M,最短边至少15px,最长边最大4096px,支持jpg/jpeg/png/bmp格式
          location string true/false 是否输出位置信息,true:输出位置信息,false:不输出位置信息,默认false

          请求代码示例

          提示一:使用示例代码前,请记得替换其中的示例Token、图片地址或Base64信息。

          提示二:部分语言依赖的类或库,请在代码注释中查看下载地址。

          curl -i -k 'https://aip.baidubce.com/rest/2.0/ocr/v1/invoice?access_token=【调用鉴权接口获取的token】' --data 'image=【图片Base64编码,需UrlEncode】' -H 'Content-Type:application/x-www-form-urlencoded'
          # encoding:utf-8
          
          import requests
          import base64
          
          '''
          通用机打发票识别
          '''
          
          request_url = "https://aip.baidubce.com/rest/2.0/ocr/v1/invoice"
          # 二进制方式打开图片文件
          f = open('[本地文件]', 'rb')
          img = base64.b64encode(f.read())
          
          params = {"image":img}
          access_token = '[调用鉴权接口获取的token]'
          request_url = request_url + "?access_token=" + access_token
          headers = {'content-type': 'application/x-www-form-urlencoded'}
          response = requests.post(request_url, data=params, headers=headers)
          if response:
              print (response.json())
          package com.baidu.ai.aip;
          
          import com.baidu.ai.aip.utils.Base64Util;
          import com.baidu.ai.aip.utils.FileUtil;
          import com.baidu.ai.aip.utils.HttpUtil;
          
          import java.net.URLEncoder;
          
          /**
          * 通用机打发票识别
          */
          public class Invoice {
          
              /**
              * 重要提示代码中所需工具类
              * FileUtil,Base64Util,HttpUtil,GsonUtils请从
              * https://ai.baidu.com/file/658A35ABAB2D404FBF903F64D47C1F72
              * https://ai.baidu.com/file/C8D81F3301E24D2892968F09AE1AD6E2
              * https://ai.baidu.com/file/544D677F5D4E4F17B4122FBD60DB82B3
              * https://ai.baidu.com/file/470B3ACCA3FE43788B5A963BF0B625F3
              * 下载
              */
              public static String invoice() {
                  // 请求url
                  String url = "https://aip.baidubce.com/rest/2.0/ocr/v1/invoice";
                  try {
                      // 本地文件路径
                      String filePath = "[本地文件路径]";
                      byte[] imgData = FileUtil.readFileByBytes(filePath);
                      String imgStr = Base64Util.encode(imgData);
                      String imgParam = URLEncoder.encode(imgStr, "UTF-8");
          
                      String param = "image=" + imgParam;
          
                      // 注意这里仅为了简化编码每一次请求都去获取access_token,线上环境access_token有过期时间, 客户端可自行缓存,过期后重新获取。
                      String accessToken = "[调用鉴权接口获取的token]";
          
                      String result = HttpUtil.post(url, accessToken, param);
                      System.out.println(result);
                      return result;
                  } catch (Exception e) {
                      e.printStackTrace();
                  }
                  return null;
              }
          
              public static void main(String[] args) {
                  Invoice.invoice();
              }
          }
          #include <iostream>
          #include <curl/curl.h>
          
          // libcurl库下载链接:https://curl.haxx.se/download.html
          // jsoncpp库下载链接:https://github.com/open-source-parsers/jsoncpp/
          const static std::string request_url = "https://aip.baidubce.com/rest/2.0/ocr/v1/invoice";
          static std::string invoice_result;
          /**
          * curl发送http请求调用的回调函数,回调函数中对返回的json格式的body进行了解析,解析结果储存在全局的静态变量当中
          * @param 参数定义见libcurl文档
          * @return 返回值定义见libcurl文档
          */
          static size_t callback(void *ptr, size_t size, size_t nmemb, void *stream) {
              // 获取到的body存放在ptr中,先将其转换为string格式
              invoice_result = std::string((char *) ptr, size * nmemb);
              return size * nmemb;
          }
          /**
          * 通用机打发票识别
          * @return 调用成功返回0,发生错误返回其他错误码
          */
          int invoice(std::string &json_result, const std::string &access_token) {
              std::string url = request_url + "?access_token=" + access_token;
              CURL *curl = NULL;
              CURLcode result_code;
              int is_success;
              curl = curl_easy_init();
              if (curl) {
                  curl_easy_setopt(curl, CURLOPT_URL, url.data());
                  curl_easy_setopt(curl, CURLOPT_POST, 1);
                  curl_httppost *post = NULL;
                  curl_httppost *last = NULL;
                  curl_formadd(&post, &last, CURLFORM_COPYNAME, "image", CURLFORM_COPYCONTENTS, "【base64_img】", CURLFORM_END);
          
                  curl_easy_setopt(curl, CURLOPT_HTTPPOST, post);
                  curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, callback);
                  result_code = curl_easy_perform(curl);
                  if (result_code != CURLE_OK) {
                      fprintf(stderr, "curl_easy_perform() failed: %s\n",
                              curl_easy_strerror(result_code));
                      is_success = 1;
                      return is_success;
                  }
                  json_result = invoice_result;
                  curl_easy_cleanup(curl);
                  is_success = 0;
              } else {
                  fprintf(stderr, "curl_easy_init() failed.");
                  is_success = 1;
              }
              return is_success;
          }
          <?php
          /**
          * 发起http post请求(REST API), 并获取REST请求的结果
          * @param string $url
          * @param string $param
          * @return - http response body if succeeds, else false.
          */
          function request_post($url = '', $param = '')
          {
              if (empty($url) || empty($param)) {
                  return false;
              }
          
              $postUrl = $url;
              $curlPost = $param;
              // 初始化curl
              $curl = curl_init();
              curl_setopt($curl, CURLOPT_URL, $postUrl);
              curl_setopt($curl, CURLOPT_HEADER, 0);
              // 要求结果为字符串且输出到屏幕上
              curl_setopt($curl, CURLOPT_RETURNTRANSFER, 1);
              curl_setopt($curl, CURLOPT_SSL_VERIFYPEER, false);
              // post提交方式
              curl_setopt($curl, CURLOPT_POST, 1);
              curl_setopt($curl, CURLOPT_POSTFIELDS, $curlPost);
              // 运行curl
              $data = curl_exec($curl);
              curl_close($curl);
          
              return $data;
          }
          
          $token = '[调用鉴权接口获取的token]';
          $url = 'https://aip.baidubce.com/rest/2.0/ocr/v1/invoice?access_token=' . $token;
          $img = file_get_contents('[本地文件路径]');
          $img = base64_encode($img);
          $bodys = array(
              'image' => $img
          );
          $res = request_post($url, $bodys);
          
          var_dump($res);
          using System;
          using System.IO;
          using System.Net;
          using System.Text;
          using System.Web;
          
          namespace com.baidu.ai
          {
              public class Invoice
              {
                  // 通用机打发票识别
                  public static string invoice()
                  {
                      string token = "[调用鉴权接口获取的token]";
                      string host = "https://aip.baidubce.com/rest/2.0/ocr/v1/invoice?access_token=" + token;
                      Encoding encoding = Encoding.Default;
                      HttpWebRequest request = (HttpWebRequest)WebRequest.Create(host);
                      request.Method = "post";
                      request.KeepAlive = true;
                      // 图片的base64编码
                      string base64 = getFileBase64("[本地图片文件]");
                      String str = "image=" + HttpUtility.UrlEncode(base64);
                      byte[] buffer = encoding.GetBytes(str);
                      request.ContentLength = buffer.Length;
                      request.GetRequestStream().Write(buffer, 0, buffer.Length);
                      HttpWebResponse response = (HttpWebResponse)request.GetResponse();
                      StreamReader reader = new StreamReader(response.GetResponseStream(), Encoding.Default);
                      string result = reader.ReadToEnd();
                      Console.WriteLine("通用机打发票识别:");
                      Console.WriteLine(result);
                      return result;
                  }
          
                  public static String getFileBase64(String fileName) {
                      FileStream filestream = new FileStream(fileName, FileMode.Open);
                      byte[] arr = new byte[filestream.Length];
                      filestream.Read(arr, 0, (int)filestream.Length);
                      string baser64 = Convert.ToBase64String(arr);
                      filestream.Close();
                      return baser64;
                  }
              }
          }

          返回说明

          返回参数

          字段 是否必选 类型 说明
          log_id uint64 唯一的log id,用于问题定位
          words_result_num uint32 识别结果数,表示words_result的元素个数
          words_result object{} 识别结果
          InvoiceType string 发票类型
          InvoiceCode string 发票代码
          InvoiceNum string 发票号码
          InvoiceDate string 开票日期
          AmountInFiguers string 合计金额小写
          AmountInWords string 合计金额大写
          CommodityName string 商品名称
          CommodityUnit string 商品单位
          CommodityPrice string 商品单价
          CommodityNum string 商品数量
          CommodityAmount string 商品金额
          MachineCode string 机打代码
          MachineNum string 机打号码
          CheckCode string 校验码
          SellerName string 销售方名称
          SellerRegisterNum string 销售方纳税人识别号
          PurchaserName string 购买方名称
          PurchaserRegisterNum string 购买方纳税人识别号
          TotalTax string 合计税额
          Province string
          City string
          Time string 时间
          SheetNum string 联次

          返回示例

          {
              "log_id": 4423022131715883558,
              "direction": 0,
              "words_result_num": 22,
              "words_result": {
                  "City": "",
                  "InvoiceNum": "01445096",
                  "SellerName": "百度餐饮店",
                  "IndustrSort": "生活服务",
                  "Province": "广东省",
                  "CommodityAmount": [
                      {
                          "word": "183.00",
                          "row": "1"
                      }
                  ],
                  "InvoiceDate": "2020年07月28日",
                  "PurchaserName": "中信建投证券股份有限公司",
                  "CommodityNum": [],
                  "InvoiceCode": "144001901511",
                  "CommodityUnit": [],
                  "SheetNum": "",
                  "PurchaserRegisterNum": "9144223008453480X9",
                  "Time": "",
                  "CommodityPrice": [],
                  "AmountInFiguers": "183.00",
                  "AmountInWords": "壹佰捌拾叁元整",
                  "CheckCode": "61042119820421061301",
                  "TotalTax": "183.00",
                  "InvoiceType": "广东通用机打发票",
                  "SellerRegisterNum": "61042119820421061301",
                  "CommodityName": [
                      {
                          "word": "餐费",
                          "row": "1"
                      }
                  ]
              }
          }
          上一篇
          定额发票识别
          下一篇
          火车票识别