{"id":970106,"url":"https://alion.io/job/inetum-senior-python-data-engineer-ocr-document-processing-remote","title":"Senior Python Data Engineer (OCR & Document Processing)- remote","company":{"id":3561,"name":"Inetum","domain":"inetum.com","url":"https://alion.io/company/inetum","size_band":"5000+","is_staffing_agency":true,"is_intermediary":false,"listed_via":null,"ats_vendor":"SmartRecruiters","truth_index":{"grade":"A","score":87,"open_postings":632,"ghost_share":0,"stale_share":0.505,"repost_share":0,"time_to_fill_p50_days":35,"computed_at":"2026-09-24T05:45:00Z"}},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":"full_time","work_mode":"remote","remote_scope":"stated_countries","remote_scope_basis":"board_field","remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bucharest, Romania"],"countries":["RO"],"hiring_countries":["RO"],"hiring_countries_total":1,"salary":null,"salary_estimate":{"min_usd":19000,"max_usd":46000,"period":"year","method":"global_role_cell_scaled_by_country","sample_n":1254},"experience_years_min":5,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Amazon CloudWatch","optional":false},{"name":"Amazon S3","optional":false},{"name":"AWS","optional":false},{"name":"AWS Step Functions","optional":false},{"name":"CI/CD","optional":false},{"name":"Git","optional":false},{"name":"OCR","optional":false},{"name":"Python","optional":false},{"name":"RAG","optional":false},{"name":"SharePoint","optional":false},{"name":"SQL","optional":false},{"name":"Azure","optional":true},{"name":"Databricks","optional":true},{"name":"LLM","optional":true}],"status":"live","first_seen_at":"2026-09-16T11:54:44Z","employer_posted_date":"2026-09-16","last_verified_at":"2026-09-25T00:53:45Z","board_verified":true,"closed_at":null,"days_open":8,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":8},"description":"Inetum is a European leader in digital services. For businesses, public sector organizations, and society as a whole, the group’s 26,000 consultants and specialists work every day to create tangible digital impact: solutions that contribute to performance, innovation, and the common good.\nWith a presence in 19 countries, working closely with local communities, and alongside its major software developer partners, Inetum supports organizations in their digital transformation challenges with proximity, flexibility, and responsibility. Driven by its purpose, Inetum champions a vision of technology that is useful and well-managed, capable of unlocking the full potential of organizations and society: “Let’s make tech right.”\nIn 2025, the group generated revenue of 2.2 billion euros.\nMore information at: www.inetum.com\n Mission\nDesign, build, and optimize scalable data ingestion and document processing solutions that transform large volumes of unstructured insurance data into structured, AI-ready information. Enable downstream AI and retrieval systems by leveraging OCR, document intelligence, vector databases, and cloud-native data pipelines.\nResponsibilities:\nDesign and implement scalable data ingestion pipelines for processing high volumes of unstructured documents, including PDFs, scans, emails, and Office files.\nIntegrate, configure, and optimize OCR and document extraction technologies to maximize text extraction accuracy and document understanding.\nBuild automated workflows for document parsing, text cleaning, normalization, semantic chunking, and metadata enrichment.\nDevelop connectors and integrations for document sources such as SharePoint, email systems, and enterprise repositories.\nDesign and maintain vector database schemas and retrieval mechanisms to support Retrieval-Augmented Generation (RAG) solutions and AI applications.\nEnsure document processing pipelines meet enterprise security, compliance, performance, and availability requirements.\nImplement monitoring, validation, and quality-control mechanisms to identify and manage low-confidence OCR and extraction results.\nOptimize data processing workflows for scalability, reliability, and low-latency operations.\nCollaborate with AI Engineers, Backend Engineers, and Platform teams to deliver end-to-end AI-powered document processing solutions.\nDevelop and maintain cloud-native data ingestion solutions on public cloud platforms.\n Profile\nProfessional Experience\n5-10 years of experience in Data Engineering, Data Processing, Document Intelligence, or related fields.\nProven experience building scalable data ingestion and processing pipelines.\nExperience working with large volumes of unstructured and semi-structured data.\nExperience designing cloud-based data solutions.\nTechnical Skills\nStrong programming skills in Python.\nStrong SQL knowledge.\nHands-on experience with AWS services, including:S3\nStep Functions\nCloudWatch\n\nExperience processing unstructured documents such as:PDF\nWord\nExcel\nPowerPoint\nEmail content\n\nExperience building connectors and integrations with enterprise content repositories (e.g., SharePoint).\nExperience with OCR and document extraction tools (AWS Textract or equivalent).\nExperience designing and implementing data ingestion and transformation pipelines.\nFamiliarity with vector databases and Retrieval-Augmented Generation (RAG) concepts.\nExperience with software development best practices:Git\nCI/CD\nAutomated testing\n\nNice to Have\nExperience with Vector Databases.\nExperience with RAG architectures and AI/LLM-based applications.\nExperience with Azure cloud services.\nExperience with Databricks.\nExperience in Insurance, Banking, or other regulated industries.\n Benefits\nFull access to foreign language learning platform\nPersonalized access to tech learning platforms\nTailored workshops and trainings to sustain your growth\nMedical insurance\nMeal tickets\nMonthly budget to allocate on flexible benefit platform\nAccess to 7 Card services\nWellbeing activities and gatherings","description_format":"text","description_chars":4006,"description_truncated":false,"requirements":{"experience_years_min":5,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[{"language":"English","level":"All levels","optional":false}]},"benefits":["Health insurance"],"hiring_locations":[{"name":"Romania","iso":"RO","kind":"country"}],"hiring_excludes":[],"relocation_offered":false,"industries":["Government"],"lifecycle":[{"event":"open","at":"2026-09-16T17:18:37Z"}],"liveness":{"score":37,"band":"fade","label":"Fading","p_open":1,"p_active":0.372,"p_room":1,"age_days":7,"expected_fill_days":35,"reasons":["conf:10","agency","stale_co","velocity","win:early","comp:brand"],"computed_at":"2026-09-24T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/inetum-senior-python-data-engineer-ocr-document-processing-remote","json_url":"https://alion.io/job/inetum-senior-python-data-engineer-ocr-document-processing-remote.json","meta":{"generated_at":"2026-09-25T04:17:46Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4635,"day_limit":5000,"remaining_today":365,"minute_limit":60,"resets_at":"2026-09-26T00:00:00Z"}}}