[{"@type":"PropertyValue","name":"데이터 규모","value":"1,586,458세트, 기본 파싱 데이터 1,466,168세트, 정밀 파싱 데이터 15,289세트, 고정밀 파싱 데이터 5,001세트, 설명서 원본 문서 100,000세트 포함"},{"@type":"PropertyValue","name":"데이터 유형","value":"파싱 데이터(중국어 교재, 중국어 전자책, 중국어 학술지, 중국어 학습 보조 교재), 중국어 설명서 원본 문서"},{"@type":"PropertyValue","name":"데이터 형식","value":"원본 문서 파일 형식: .pdf, 문서 이미지 파일 형식: .png, OCR 어노테이션 파일 형식: .json, 구조화 파싱 파일 형식: Markdown(표 및 수식은 LaTeX 형식 또는 스크린샷 링크 사용)"}]
{"id":1749,"datatype":"1","titleimg":"https://ko.nexdata.ai/shujutang/static/image/index/datatang_tuxiang_default.webp","type1":"147","type1str":null,"type2":"150","type2str":null,"dataname":"1,586,458세트 문서 OCR 및 구조화 파싱 데이터","datazy":[{"title":"데이터 규모","content":"1,586,458세트, 기본 파싱 데이터 1,466,168세트, 정밀 파싱 데이터 15,289세트, 고정밀 파싱 데이터 5,001세트, 설명서 원본 문서 100,000세트 포함"},{"title":"데이터 유형","content":"파싱 데이터(중국어 교재, 중국어 전자책, 중국어 학술지, 중국어 학습 보조 교재), 중국어 설명서 원본 문서"},{"title":"데이터 형식","content":"원본 문서 파일 형식: .pdf, 문서 이미지 파일 형식: .png, OCR 어노테이션 파일 형식: .json, 구조화 파싱 파일 형식: Markdown(표 및 수식은 LaTeX 형식 또는 스크린샷 링크 사용)"}],"datatag":"OCR,Document,Structured parsing","technologydoc":null,"downurl":null,"datainfo":null,"standard":null,"dataylurl":null,"flag":null,"publishtime":null,"createby":null,"createtime":null,"ext1":null,"samplestoreloc":null,"hosturl":null,"datasize":null,"industryPlan":null,"keyInformation":null,"samplePresentation":[],"officialSummary":"1,586,458세트 문서 OCR 및 구조화 파싱 데이터로, 중국어 교재, 중국어 전자책, 중국어 학습 보조 교재 등을 포함하며, 어노테이션 파일에는 OCR 어노테이션과 구조화 파싱 데이터가 포함됩니다.","dataexampl":null,"datakeyword":["중국어","OCR","문서","구조화 파싱"],"isDelete":null,"ids":null,"idsList":null,"datasetCode":null,"productStatus":null,"tagTypeEn":"Data Type,Language","tagTypeZh":null,"website":null,"samplePresentationList":null,"datazyList":null,"keyInformationList":null,"dataexamplList":null,"bgimg":null,"datazyScriptList":null,"datakeywordListString":null,"sourceShowPage":"ocr","dataShowType":"[{\"code\":\"0\",\"language\":\"ZH\"},{\"code\":\"1\",\"language\":\"ZH\"},{\"code\":\"2\",\"language\":\"EN,JP,KO\"},{\"code\":\"3\",\"language\":\"EN\"},{\"code\":\"4\",\"language\":\"JP\"}]","productNameEn":"1,586,458 Sets-Document OCR&Parsing Data","BGimg":"","voiceBg":["/shujutang/static/image/comm/audio_bg.webp","/shujutang/static/image/comm/audio_bg2.webp","/shujutang/static/image/comm/audio_bg3.webp","/shujutang/static/image/comm/audio_bg4.webp","/shujutang/static/image/comm/audio_bg5.webp"]}
https://ko.nexdata.ai/shujutang/static/image/index/datatang_tuxiang_default.webp
[]
1,586,458세트 문서 OCR 및 구조화 파싱 데이터
중국어
OCR
문서
구조화 파싱
1,586,458세트 문서 OCR 및 구조화 파싱 데이터로, 중국어 교재, 중국어 전자책, 중국어 학습 보조 교재 등을 포함하며, 어노테이션 파일에는 OCR 어노테이션과 구조화 파싱 데이터가 포함됩니다.
이는 상업적 사용, 연구 목적 등을 위한 유료 데이터셋입니다.라이선스가 부여된 기성 데이터셋은 AI 프로젝트의 빠른 시작에 도움을 줍니다.
![사양]()
사양
데이터 규모
1,586,458세트, 기본 파싱 데이터 1,466,168세트, 정밀 파싱 데이터 15,289세트, 고정밀 파싱 데이터 5,001세트, 설명서 원본 문서 100,000세트 포함
데이터 유형
파싱 데이터(중국어 교재, 중국어 전자책, 중국어 학술지, 중국어 학습 보조 교재), 중국어 설명서 원본 문서
데이터 형식
원본 문서 파일 형식: .pdf, 문서 이미지 파일 형식: .png, OCR 어노테이션 파일 형식: .json, 구조화 파싱 파일 형식: Markdown(표 및 수식은 LaTeX 형식 또는 스크린샷 링크 사용)
![샘플]()
샘플
![추천 데이터셋]()
추천 데이터셋
8fd043f6-a696-4830-adb2-0e3909b5b39a