diff --git a/.gitignore b/.gitignore index 428af9f0b..859ba378a 100644 --- a/.gitignore +++ b/.gitignore @@ -70,6 +70,7 @@ test_*.csv # Local debugging scripts apps/worker/scripts/ +apps/worker/experiments/ apps/worker/start_celery_worker.py apps/worker/start_celery_debug.sh apps/worker/clear_celery_queues.sh diff --git a/apps/api/app/api/dependencies/current_user.py b/apps/api/app/api/dependencies/current_user.py index 4c5593c1a..65448cd00 100644 --- a/apps/api/app/api/dependencies/current_user.py +++ b/apps/api/app/api/dependencies/current_user.py @@ -10,6 +10,9 @@ ) from app.services.rate_limit.job_admission_service import JobAdmissionService from fastapi import Depends +from sqlalchemy.ext.asyncio import AsyncSession + +from shared.core.database import get_db _job_admission_service = JobAdmissionService() @@ -17,9 +20,11 @@ async def with_current_user( route_context: RouteAdmissionContext = Depends(get_route_admission_context), user_id: str = Depends(get_current_user_id), + db: AsyncSession = Depends(get_db), ) -> AsyncGenerator[CurrentUser, None]: current_user = await _job_admission_service.resolve_current_user( route_context=route_context, user_id=user_id, + db=db, ) yield current_user diff --git a/apps/api/app/data/demo_documents/tsla-q4-2025/chunks.json b/apps/api/app/data/demo_documents/tsla-q4-2025/chunks.json index 177ae25c9..68d1d3111 100644 --- a/apps/api/app/data/demo_documents/tsla-q4-2025/chunks.json +++ b/apps/api/app/data/demo_documents/tsla-q4-2025/chunks.json @@ -25,7 +25,7 @@ "chunk_id": "68a6be7d-c587-5c73-abf2-56f4686e28e6", "type": "text", "content": "[tables/table-0 Tesla 2025 Results.html]", - "path": "TSLA-Q4-2025-Update.pdf-->HIGHLIGHTS", + "path": "TSLA-Q4-2025-Update.pdf/HIGHLIGHTS", "metadata": { "length": 83, "summary": "", @@ -138,7 +138,7 @@ "chunk_id": "60109008-6261-51e6-b202-093d904eb881", "type": "text", "content": "FINANCIAL SUMMARY\n(Unaudited)\n\n[tables/table-1 Q4 2025 Financials.html]\n\n(1) As a result of the adoption of the new crypto assets standard, the previously reported quarterly periods in 2024 have been recast.\n(2) Beginning in Q1'25, Adjusted EBITDA (non-GAAP) is presented net of digital assets gains and losses and all prior periods have been adjusted.\n(3) Beginning in Q1'25, Net income attributable to common stockholders (non-GAAP) is presented net of digital assets gains and losses and all prior periods have been adjusted.\n(4) Beginning in Q1'25, Capital expenditures is presented inclusive of purchases of energy generation and storage systems and all prior periods have been adjusted.\nFINANCIAL SUMMARY\n(Unaudited)\n\n[tables/table-2 Financial Data 2021-25.html]\n\n(1) Beginning in Q1'25, Adjusted EBITDA (non-GAAP) is presented net of digital assets gains and losses and all prior periods have been adjusted.\n(2) Beginning in Q1'25, Net income attributable to common stockholders (non-GAAP) is presented net of digital assets gains and losses and all prior periods have been adjusted.\n(3) Beginning in Q1'25, Capital expenditures is presented inclusive of purchases of energy generation and storage systems and all prior periods have been adjusted.\nOPERATIONAL SUMMARY\n(Unaudited)\n\n[tables/table-3 Tesla Q4-2025 Data.html]\n\nOPERATIONAL SUMMARY\n(Unaudited)\n\n[tables/table-4 Tesla 2021-2025 Data.html]", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY", "metadata": { "length": 1517, "summary": "The document presents unaudited financial and operational summaries for a company, covering quarterly data from Q4 2024 through Q4 2025 and annual data from 2021 to 2025. Key notes indicate significant accounting changes effective Q1 2025: Adjusted EBITDA and Net income attributable to common stockholders are now presented net of digital assets gains and losses, with all prior periods adjusted accordingly. Additionally, Capital expenditures now include purchases of energy generation and storage systems, requiring restatement of previous periods. The content references multiple tables detailing these metrics but does not display the specific numerical values.", @@ -240,7 +240,7 @@ "chunk_id": "60339310-5480-5ae2-8791-e6017bafb730", "type": "text", "content": "While automotive sales declined sequentially, gross margin (even when excluding the impact of regulatory credits) improved. The APAC region continued to show strength across multiple markets and set a record for deliveries in the quarter. We continued the rollout of Model Y variants across markets in Q4, including the standard and performance versions.\nPreparations continue in North America for the production ramps of Tesla Semi and Cybercab, both commencing 1H26, and production of the next-generation Roadster.", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->Automotive", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/Automotive", "metadata": { "length": 516, "summary": "", @@ -304,7 +304,7 @@ "chunk_id": "eb18e33e-b093-532e-b4df-bd05d63b0294", "type": "text", "content": "We achieved our highest quarterly energy storage deployments, driven by record Megapack deployments. Total gross profit rose, both sequentially and year-over-year, to a record \\$1.1 billion, marking the fifth consecutive record quarter. We plan to begin Megapack 3 and Megablock production at Megafactory Houston in 2026. In 2025, our global Powerwall network supported more than 89,000 Virtual Power Plant events across over 1 million installed units, allowing homeowners to save over \\$1 billion in electricity bills as Virtual Power Plant participation continues to scale rapidly.", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->Energy generation and storage", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/Energy generation and storage", "metadata": { "length": 583, "summary": "", @@ -395,7 +395,7 @@ "chunk_id": "60518074-bcbe-5b48-89a9-dfeb5d1b531b", "type": "text", "content": "We made further progress on the Optimus program in 2025. In Q1 of this year, we plan to unveil the Gen 3 version of Optimus, which will include major upgrades from version 2.5, including our latest hand design. The Gen 3 is our first design meant for mass production. Preparations are underway for the first production line, including supply chain readiness, with start of production planned before the end of 2026 and eventual planned capacity of 1 million robots per year.\nInstalled Annual Manufacturing Capacity\n\n[tables/table-5 Tesla Production.html]\n\nInstalled capacity ≠ current production rate and there may be limitations discovered as production rates approach capacity. Production rates depend on a variety of factors, including equipment uptime, component supply, downtime related to factory upgrades, regulatory considerations and other factors. Construction includes factory and infrastructure buildout as well as tool installation.", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->Robotics", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/Robotics", "metadata": { "length": 959, "summary": "", @@ -491,7 +491,7 @@ "chunk_id": "c49883a3-5834-5537-be53-35e58da70bf7", "type": "text", "content": "We are currently building Cortex 2 at Gigafactory Texas to further increase our AI training compute capacity. In the first half of 2026, we plan to more than double the size of onsite compute in Texas (in terms of H100 equivalents). We aim to maximize capital efficiency by scaling training compute judiciously, including when the training backlog gets too long or in anticipation of greater demand from our engineers to support our AI-related offerings.", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->AI Training Compute", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/AI Training Compute", "metadata": { "length": 454, "summary": "", @@ -545,7 +545,7 @@ "chunk_id": "fcbc3b02-dec6-5e91-849e-fccd3f22320e", "type": "text", "content": "Our lithium refinery commenced pilot production and is the first spodumene to lithium hydroxide refinery in North America, leveraging a simpler, cheaper and more environmentally friendly process. This refinery enables us to domestically produce critical minerals in support of energy storage, battery manufacturing and ultimately for EV growth.\nWe have begun to produce battery packs for certain Model Ys with our 4680 cells, unlocking an additional vector of supply to help navigate increasingly complex supply chain challenges caused by trade barriers and tariff risks. We now produce dry-electrode for 4680 cells with both anode and cathode made in Austin. We expect both domestic cathode material in Texas and LFP lines in Nevada to begin production in 2026.", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->Battery", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/Battery", "metadata": { "length": 762, "summary": "", @@ -667,7 +667,7 @@ "chunk_id": "34777c71-d716-5c39-b18a-bec0d0b64706", "type": "text", "content": "We continue to efficiently utilize our existing physical footprint in North America, with targeted augmentation to support the rollout of Robotaxi. While in the short-term, operational workstreams such as charging, cleaning and maintenance can be managed through our existing charging network, service centers and sales and delivery locations, we will have to add more capacity as the service expands. We added over 3,800 net new Supercharging stalls, growing the network 19% year-over-year.\nInstalled Annual Capacity\n\n[tables/table-6 Facility Status.html]\n\nInstalled capacity ≠ current production rate and there may be limitations discovered as production rates approach capacity. Production rates depend on a variety of factors, including equipment uptime, component supply, downtime related to factory upgrades, regulatory considerations and other factors. Construction includes factory and infrastructure buildout as well as tool installation.\n\n## Other Supporting Infrastructure Tesla AI Training Capacity Ramp (H100 equivalent GPUs)0\n[images/image-1-Capacity Growth Projection.jpg]\n\nTesla AI Training Capacity Ramp (H100 equivalent GPUs)", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->Other Supporting Infrastructure", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/Other Supporting Infrastructure", "metadata": { "length": 1142, "summary": "", @@ -786,7 +786,7 @@ "chunk_id": "778a63c2-7955-514c-84d0-9c2cfd99a489", "type": "text", "content": "We continue to enhance FSD (Supervised) $^{1}$ via our end-to-end foundation model trained on both customer and Robotaxi real-world data with our latest version, v14. FSD (Supervised) $^{1}$ increasingly provides safety and convenience functionality that can relieve drivers of many tedious and potentially dangerous aspects of road travel, including giving access to personal transport for those who otherwise have difficulty driving. V14 offers unparalleled driver assistance to safely drive to the customer's destination, find a free parking spot and park at that location. Our global fleet can collect the equivalent of over 500 years of continuous driving data per day $^{2}$ , allowing us to safely deploy and scale capabilities that can handle the long and fat tail of corner cases across diverse geographies and driving environments.", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->AI Software", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/AI Software", "metadata": { "length": 841, "summary": "", @@ -876,7 +876,7 @@ "chunk_id": "164babde-7a64-5b81-9a5a-41443b221c40", "type": "text", "content": "Development of our in-house, custom designed AI5 and AI6 inference chips for autonomy progressed during the quarter, with production planned for 2027 and 2028, respectively. We are targeting a 50x improvement in performance for AI5 relative to AI4 thanks to 10x raw compute, 9x memory capacity and 5x hardened block quantization and softmax function (the latter enabling efficient low-precision computing without sacrificing model accuracy).", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->AI Inference Compute", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/AI Inference Compute", "metadata": { "length": 441, "summary": "", @@ -970,7 +970,7 @@ "chunk_id": "32bfae6f-b2d4-51d3-94d9-82b9640828dd", "type": "text", "content": "The Robotaxi iOS app no longer has a waitlist in the areas we serve. Our vehicles keep getting better with our over-the-air updates, including: Grok (an AI companion) which now supports navigation commands (allowing users to find, add and edit navigation destinations hands-free); Tesla Photobooth which enables users to take photos in their car and download or share via the Tesla mobile app; Supercharger Site Maps which displays Supercharger layouts, nearby businesses and live availability details; Automatic HOV Lane Routing based on interior camera occupancy detection; Phone Left Behind Chime and SpaceX ISS Docking Simulator Game.\n\nWe continue to enhance FSD (Supervised) $^{1}$ via our end-to-end foundation model trained on both customer and Robotaxi real-world data with our latest version, v14. FSD (Supervised) $^{1}$ increasingly provides safety and convenience functionality that can relieve drivers of many tedious and potentially dangerous aspects of road travel, including giving access to personal transport for those who otherwise have difficulty driving. V14 offers unparalleled driver assistance to safely drive to the customer's destination, find a free parking spot and park at that location. Our global fleet can collect the equivalent of over 500 years of continuous driving data per day $^{2}$ , allowing us to safely deploy and scale capabilities that can handle the long and fat tail of corner cases across diverse geographies and driving environments. Cumulative Miles Driven with FSD (Supervised) $^{1}$ (billions)0\n[images/image-2-FSD Mileage Growth.jpg]\n\nCumulative Miles Driven with FSD (Supervised) $^{1}$ (billions)\n\nDevelopment of our in-house, custom designed AI5 and AI6 inference chips for autonomy progressed during the quarter, with production planned for 2027 and 2028, respectively. We are targeting a 50x improvement in performance for AI5 relative to AI4 thanks to 10x raw compute, 9x memory capacity and 5x hardened block quantization and softmax function (the latter enabling efficient low-precision computing without sacrificing model accuracy). Targeting Step-Function Improvement for our Next-Generation Inference Chip, AI50\n[images/image-3-Tesla Silicon Optimization.jpg]\n\nTargeting Step-Function Improvement for our Next-Generation Inference Chip, AI5\n(1) Active driver supervision required; does not make the vehicle autonomous\n(2) Calculated based on continuous hours of driving at an average of 30 miles per hour", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->Automotive and Other Software", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/Automotive and Other Software", "metadata": { "length": 2444, "summary": "The Robotaxi iOS app has removed its waitlist in served areas and now includes features like Grok navigation, Tesla Photobooth, Supercharger maps, automatic HOV routing, phone left-behind chimes, and a SpaceX game. FSD (Supervised) v14 uses an end-to-end foundation model trained on vast real-world data to assist drivers with navigation, parking, and safety, though active supervision remains required. The company is developing custom AI5 and AI6 inference chips for 2027 and 2028, targeting significant performance improvements over previous generations to handle complex driving scenarios globally.", @@ -1205,7 +1205,7 @@ "chunk_id": "41533301-58f0-554f-916d-8254cc2700df", "type": "text", "content": "We began testing driverless Robotaxis in Austin in December and began removing the safety monitor from customer rides in January on a limited basis, which will unlock further expansion of our Robotaxi fleet and coverage area in the Austin-metro. Our Bay Area ride-hailing service began serving the San Jose Airport in October, with plans to expand to other major airports in the Bay Area upon receiving required permitting.", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->Robotaxi", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/Robotaxi", "metadata": { "length": 423, "summary": "", @@ -1263,7 +1263,7 @@ "chunk_id": "3968a25f-933e-5a7f-a118-07db2e581153", "type": "text", "content": "We launched FSD (Supervised) $^{1}$ in South Korea, where customers drove over 1 million kilometers using the software in just one month. While we continue to pursue regulatory approval in China and Europe, we began offering ride-along experiences to consumers in Italy, Germany, France and Switzerland.\nMonthly subscriptions to FSD (Supervised) $^{1}$ continued to grow sequentially and more than doubled in 2025. Starting this quarter, we are transitioning access to FSD (Supervised) $^{1}$ to monthly subscriptions only as we begin to sunset the up-front payment option.", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->FSD (Supervised) $^{1}$", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/FSD (Supervised) $^{1}$", "metadata": { "length": 573, "summary": "", @@ -1364,7 +1364,7 @@ "chunk_id": "adf89909-d366-51e1-b68c-459f9024191c", "type": "text", "content": "Services and Other gross profit of approximately \\$300 million was partly driven by Part Sales and Supercharging. We now offer Tesla Insurance in Florida, as we continue to expand our insurance product to new states. In certain states, customers receive a discount on their insurance premiums when using FSD (Supervised) $^{1}$ . The more you drive with FSD (Supervised) $^{1}$ enabled, the bigger the discount is on your insurance premium – helping, in certain cases, to completely offset the monthly subscription cost for FSD (Supervised) $^{1}$ .\n\n## FSD (Supervised) $^{1}$ Cumulative Paid Robotaxi Miles0\n[images/image-4-Growth Trend 2025.jpg]\n\nCumulative Paid Robotaxi Miles\n\n[tables/table-7 Autonomous Driving Status.html]\n\nPlanned Robotaxi Coverage", - "path": "TSLA-Q4-2025-Update.pdf-->SUMMARY-->Automotive Services", + "path": "TSLA-Q4-2025-Update.pdf/SUMMARY/Automotive Services", "metadata": { "length": 742, "summary": "", @@ -1450,7 +1450,7 @@ "chunk_id": "cf91377e-176a-5768-855a-b859092f4695", "type": "text", "content": "On January 16, 2026, Tesla entered into an agreement to invest approximately \\$2 billion to acquire shares of Series E Preferred Stock of xAI as part of their recent publicly-disclosed financing round. Tesla’s investment was made on market terms consistent with those previously agreed to by other investors in the financing round. As set forth in Master Plan Part IV, Tesla is building products and services that bring AI into the physical world. Meanwhile, xAI is developing leading digital AI products and services, such as its large language model (Grok).\nIn that context, and as part of Tesla's broader strategy under Master Plan Part IV, Tesla and xAI also entered into a framework agreement in connection with the investment. Among other things, the framework agreement builds upon the existing relationship between Tesla and xAI by providing a framework for evaluating potential AI collaborations between the companies. Together, the investment and the related framework agreement are intended to enhance Tesla's ability to develop and deploy AI products and services into the physical world at scale. This investment is subject to customary regulatory conditions with the expectation to close in Q1'2026.", - "path": "TSLA-Q4-2025-Update.pdf-->OTHER UPDATES", + "path": "TSLA-Q4-2025-Update.pdf/OTHER UPDATES", "metadata": { "length": 1213, "summary": "", @@ -1552,7 +1552,7 @@ "chunk_id": "79b23d73-b353-5998-a646-93d6739ec465", "type": "text", "content": "", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK", "metadata": { "length": 0, "summary": "", @@ -1569,7 +1569,7 @@ "chunk_id": "8d4c49a6-c531-5fa6-a77d-ad028537fa52", "type": "text", "content": "We are focused on maximum capacity utilization at our factories. Deliveries and deployments will be impacted by aggregate demand for our products, supply chain readiness and allocation decisions between sale to customers or use for our owned and operated fleet.", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Volume", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Volume", "metadata": { "length": 261, "summary": "", @@ -1609,7 +1609,7 @@ "chunk_id": "426e5119-3ef6-5176-9119-fc044eb90be7", "type": "text", "content": "We will manage the businesses such that we ensure a strong balance sheet, maintaining sufficient liquidity to fund our product roadmap, long-term capacity expansion plans – including further vertical integration – and other expenses.", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Cash", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Cash", "metadata": { "length": 233, "summary": "", @@ -1649,7 +1649,7 @@ "chunk_id": "99f54247-8409-5df2-8299-39184863cd09", "type": "text", "content": "While we continue to execute on innovations to reduce the cost of manufacturing and operations, over time, we expect our hardware-related profits to be accompanied by an acceleration of AI, software and fleet-based profits.", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Profit", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Profit", "metadata": { "length": 223, "summary": "", @@ -1686,7 +1686,7 @@ "chunk_id": "29e89187-3dbe-5298-87c3-c38d3cbe6883", "type": "text", "content": "We continue to evolve and augment our product lineup with a focus on cost, scale and future monetization opportunities via services powered by our AI software. We remain focused on growing our sales volumes through a differentiated and efficiently managed product portfolio, which includes leveraging and optimizing our existing production capacity before building new factories and production lines.\nCybercab, Tesla Semi and Megapack 3 are on schedule for volume production starting in 2026. First generation production lines for Optimus are being installed in anticipation of volume production.\nPHOTOS & CHARTS", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Product", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Product", "metadata": { "length": 612, "summary": "", @@ -1772,7 +1772,7 @@ "chunk_id": "bbe9d89e-eebd-56b1-b7c0-7eda12dd3377", "type": "text", "content": "Cybercab, Tesla Semi and Megapack 3 are on schedule for volume production starting in 2026. First generation production lines for Optimus are being installed in anticipation of volume production. 0\n[images/image-5-Tesla Model Y Driving.jpg]", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Product-->MODEL Y - 2025 BEST IN CLASS EURO NCAP $^{(1)}$ SMALL SUV", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Product/MODEL Y - 2025 BEST IN CLASS EURO NCAP $^{(1)}$ SMALL SUV", "metadata": { "length": 245, "summary": "", @@ -1835,7 +1835,7 @@ "chunk_id": "02f7e756-798f-5675-9dd9-6ce9123f92d4", "type": "text", "content": " 0\n[images/image-6-Red Tesla on Coastal Road.jpg]", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Product-->MODEL 3 - 2025 BEST IN CLASS EURO NCAP $^{(1)}$ LARGE FAMILY CAR", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Product/MODEL 3 - 2025 BEST IN CLASS EURO NCAP $^{(1)}$ LARGE FAMILY CAR", "metadata": { "length": 66, "summary": "", @@ -1884,7 +1884,7 @@ "chunk_id": "494118dd-5788-526b-a139-407d1cfb0d20", "type": "text", "content": " 0\n[images/image-7-Tesla Interior Interface.jpg]", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Product-->FSD (SUPERVISED) $^{1}$ – V14 OFFERS UNPARALLELED DRIVER ASSISTANCE", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Product/FSD (SUPERVISED) $^{1}$ – V14 OFFERS UNPARALLELED DRIVER ASSISTANCE", "metadata": { "length": 66, "summary": "", @@ -1933,7 +1933,7 @@ "chunk_id": "6ac7765a-4da5-573a-963b-f35ab3796f6f", "type": "text", "content": " 0\n[images/image-8-Tesla Interior.jpg]", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Product-->DRIVERLESS ROBOTAXI - TESTING IN AUSTIN", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Product/DRIVERLESS ROBOTAXI - TESTING IN AUSTIN", "metadata": { "length": 66, "summary": "", @@ -2016,7 +2016,7 @@ "chunk_id": "4cfe7ce9-5f93-5816-9505-ba562683ee42", "type": "text", "content": " 0\n[images/image-9-Tesla Cybertruck in Snow.jpg]\n\n\n### DRIVERLESS ROBOTAXI - TESTING IN AUSTIN 0\n[images/image-10-Tesla Semi Trucks.jpg]\n\nTESLA SEMI - MEGACHARGER NETWORK PLANNED SITES FOR 2026\n\n### CYBERCAB - COLD WEATHER TESTING IN ALASKA 0\n[images/image-11-US Lightning Map.jpg]", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Product-->CYBERCAB - COLD WEATHER TESTING IN ALASKA", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Product/CYBERCAB - COLD WEATHER TESTING IN ALASKA", "metadata": { "length": 318, "summary": "", @@ -2102,7 +2102,7 @@ "chunk_id": "07535539-a2f9-5b79-ac6b-d41cf785ec88", "type": "text", "content": " 0\n[images/image-12-Tesla Factory Milestone.jpg]", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Product-->GIGAFACTORY SHANGHAI - 9 MILLIONTH VEHICLE PRODUCED (GLOBALLY)", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Product/GIGAFACTORY SHANGHAI - 9 MILLIONTH VEHICLE PRODUCED (GLOBALLY)", "metadata": { "length": 67, "summary": "", @@ -2202,7 +2202,7 @@ "chunk_id": "702bff82-1736-584d-956e-e7bf7eac4039", "type": "text", "content": " 0\n[images/image-13-Tesla Factory Milestone.jpg]\n\n\n### GIGAFACTORY SHANGHAI - 9 MILLIONTH VEHICLE PRODUCED (GLOBALLY) 0\n[images/image-14-Vehicle Delivery Trends.jpg]\n\n\n 0\n[images/image-15-Quarterly Cash Flow.jpg]\n\n\n 0\n[images/image-16-Financial Performance Chart.jpg]", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Product-->GIGAFACTORY NEVADA - 6 MILLIONTH DRIVE UNIT PRODUCED", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/Product/GIGAFACTORY NEVADA - 6 MILLIONTH DRIVE UNIT PRODUCED", "metadata": { "length": 327, "summary": "", @@ -2319,7 +2319,7 @@ "chunk_id": "5e17588f-ea71-56df-be53-135da93fb3a0", "type": "text", "content": " 0\n[images/image-17-Projected Vehicle Deliveries.jpg]\n\n\n 0\n[images/image-18-Cash Flow Trends.jpg]\n\n\n 0\n[images/image-19-Financial Performance Forecast.jpg]", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->KEY METRICS TRAILING 12 MONTHS (TTM) (Unaudited)", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/KEY METRICS TRAILING 12 MONTHS (TTM) (Unaudited)", "metadata": { "length": 207, "summary": "", @@ -2369,7 +2369,7 @@ "chunk_id": "c82a6fed-f085-5ec3-b10a-0381723fa3c9", "type": "text", "content": "Total quarterly revenue decreased 3% YoY to \\$24.9B. YoY, revenue was impacted by the following items $^{(1)}$ :\n- decrease in vehicle deliveries\n- lower regulatory credit revenue\n+ growth in Energy Generation and Storage\n+ growth in Services and Other\n+ positive FX impact of \\$0.3B $^{1}$\n\\+ growth in other automotive ancillary sales, partly driven by an increase in FSD subscriptions\n\\+ higher vehicle average selling price (ASP) (excl. FX impact $^{1}$ ), inclusive of mix impact", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->KEY METRICS TRAILING 12 MONTHS (TTM) (Unaudited)-->Revenue", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/KEY METRICS TRAILING 12 MONTHS (TTM) (Unaudited)/Revenue", "metadata": { "length": 484, "summary": "", @@ -2428,7 +2428,7 @@ "chunk_id": "946fabf6-000f-557a-9d81-0b8c1a25aafb", "type": "text", "content": "Our quarterly operating income decreased 11% YoY to \\$1.4B, resulting in a 5.7% operating margin. YoY, operating income was primarily impacted by the following items $^{(1)}$ :\n- increase in SBC and Restructuring and Other charges\n- increase in operating expenses (excl. SBC and Restructuring and Other) driven by AI and other R&D projects and SG&A\n- higher average cost per vehicle due to lower fixed cost absorption for certain models and an increase in tariffs\n- decrease in vehicle deliveries\n- lower regulatory credit revenue\n+ higher vehicle average gross profit due to mix and pricing impacts\n+ growth in Energy Generation and Storage gross profit\n+ growth in Services and Other gross profit\n+ growth in other automotive ancillary sales, partly driven by an increase in FSD subscriptions", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->KEY METRICS TRAILING 12 MONTHS (TTM) (Unaudited)-->Profitability", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/KEY METRICS TRAILING 12 MONTHS (TTM) (Unaudited)/Profitability", "metadata": { "length": 794, "summary": "", @@ -2502,7 +2502,7 @@ "chunk_id": "001d4b8a-b264-5cae-86a5-8c419eabbec0", "type": "text", "content": "Quarter-end cash, cash equivalents and investments was \\$44.1B. The sequential increase of \\$2.4B was primarily the result of positive free cash flow.", - "path": "TSLA-Q4-2025-Update.pdf-->OUTLOOK-->KEY METRICS TRAILING 12 MONTHS (TTM) (Unaudited)-->Cash", + "path": "TSLA-Q4-2025-Update.pdf/OUTLOOK/KEY METRICS TRAILING 12 MONTHS (TTM) (Unaudited)/Cash", "metadata": { "length": 150, "summary": "", @@ -2683,7 +2683,7 @@ "chunk_id": "15c18264-8e1e-5db6-b3c9-70cd181d1f39", "type": "text", "content": "STATEMENT OF OPERATIONS\n(Unaudited)\n\n[tables/table-8 Q4 2024-Q4 2025 Rev.html]\n\nBALANCE SHEET\n(Unaudited)\n\n[tables/table-9 Balance Sheet 2024-25.html]\n\nSTATEMENT OF CASH FLOWS\n(Unaudited)\n\n[tables/table-10 Cash Flow Q4-25.html]\n\nRECONCILIATION OF GAAP TO NON-GAAP FINANCIAL INFORMATION (Unaudited)\n\n[tables/table-11 Q4 2024-Q4 2025.html]\n\nRECONCILIATION OF GAAP TO NON-GAAP FINANCIAL INFORMATION\n(Unaudited)\n\n[tables/table-12 Financial Metrics 2021-25.html]\n\nRECONCILIATION OF GAAP TO NON-GAAP FINANCIAL INFORMATION\n(Unaudited)\n\n[tables/table-13 Financial Data 2022-25.html]\n\n\n[tables/table-14 Financial Metrics 2023-25.html]\n\nTTM = Trailing twelve months\n(1) Beginning in Q1'25, Capital expenditures is presented inclusive of purchases of energy generation and storage systems and all prior periods have been adjusted.\n(2) As a result of the adoption of the new crypto assets standard, the previously reported quarterly periods in 2024 have been recast.\n(3) Beginning in Q1'25, Adjusted EBITDA (non-GAAP) is presented net of digital assets gains and losses and all prior periods have been adjusted.", - "path": "TSLA-Q4-2025-Update.pdf-->FINANCIAL STATEMENTS", + "path": "TSLA-Q4-2025-Update.pdf/FINANCIAL STATEMENTS", "metadata": { "length": 1418, "summary": "", @@ -2816,7 +2816,7 @@ "chunk_id": "31babb3c-1abf-5f64-90c7-e8e8a5384bbd", "type": "text", "content": "Tesla will provide a live webcast of its fourth quarter 2025 financial results conference call beginning at 4:30 p.m. CT on January 28, 2026 at ir.tesla.com. This webcast will also be available for replay for approximately one year thereafter.", - "path": "TSLA-Q4-2025-Update.pdf-->WEBCAST INFORMATION", + "path": "TSLA-Q4-2025-Update.pdf/WEBCAST INFORMATION", "metadata": { "length": 243, "summary": "", @@ -2857,7 +2857,7 @@ "chunk_id": "19e3f236-d15a-5b2a-8589-43d28c5316fe", "type": "text", "content": "When used in this update, certain terms have the following meanings. Our vehicle deliveries include only vehicles that have been transferred to end customers with all paperwork correctly completed. Our energy product deployment volume includes both customer units when installed and equipment sales at time of delivery. \"Net income attributable to common stockholders (non-GAAP)\" is equal to (i) net income attributable to common stockholders before (ii)(a) stock-based compensation expense, net of tax, (b) digital assets (gain) loss, net of tax and (c) release of valuation allowance on deferred tax assets. \"Adjusted EBITDA (non-GAAP)\" is equal to (i) net income attributable to common stockholders before (ii)(a) interest expense, (b) provision for (benefit from) income taxes, (c) depreciation, amortization and impairment, (d) stock-based compensation expense and (e) digital assets loss (gain), net. \"Free cash flow\" is operating cash flow less capital expenditures. Average cost per vehicle is cost of automotive sales divided by new vehicle deliveries (excluding operating leases). \"Days sales outstanding\" is equal to (i) average accounts receivable, net for the period divided by (ii) total revenues and multiplied by (iii) the number of days in the period. \"Days payable outstanding\" is equal to (i) average accounts payable for the period divided by (ii) total cost of revenues and multiplied by (iii) the number of days in the period. \"Days of supply\" is calculated by dividing new car ending inventory by the relevant period's deliveries and using trading days. Constant currency impacts are calculated by comparing actuals against current results converted into USD using average exchange rates from the prior period.", - "path": "TSLA-Q4-2025-Update.pdf-->CERTAIN TERMS", + "path": "TSLA-Q4-2025-Update.pdf/CERTAIN TERMS", "metadata": { "length": 1733, "summary": "This passage defines key financial and operational terms used in a specific update. It clarifies that vehicle deliveries refer to units transferred to end customers with completed paperwork, while energy product deployment includes installed customer units and equipment sales. The text details non-GAAP measures: 'Net income attributable to common stockholders' adjusts for stock-based compensation, digital asset gains/losses, and valuation allowances; 'Adjusted EBITDA' excludes interest, taxes, depreciation, amortization, impairment, stock-based compensation, and digital asset impacts. 'Free cash flow' is defined as operating cash flow minus capital expenditures. Operational metrics include 'Average cost per vehicle' (automotive sales cost divided by new deliveries excluding leases), 'Days sales outstanding' (average receivables divided by revenue times days in period), 'Days payable outstanding' (average payables divided by cost of revenues times days in period), and 'Days of supply' (ending inventory divided by deliveries using trading days). Finally, constant currency impacts are calculated by comparing actuals against results converted to USD using prior period average exchange rates.", @@ -2982,7 +2982,7 @@ "chunk_id": "332edac0-294c-5e5b-8599-a0e52b25e53a", "type": "text", "content": "Consolidated financial information has been presented in accordance with GAAP as well as on a non-GAAP basis to supplement our consolidated financial results. Our non-GAAP financial measures include non-GAAP net income (loss) attributable to common stockholders, non-GAAP net income (loss) attributable to common stockholders on a diluted per share basis (calculated using weighted average shares for GAAP diluted net income (loss) attributable to common stockholders), Adjusted EBITDA margin, non-GAAP automotive gross margin and free cash flow. These non-GAAP financial measures also facilitate management's internal comparisons to Tesla's historical performance as well as comparisons to the operating results of other companies. Management believes that it is useful to supplement its GAAP financial statements with this non-GAAP information because management uses such information internally for its operating, budgeting and financial planning purposes. Management also believes that presentation of the non-GAAP financial measures provides useful information to our investors regarding our financial condition and results of operations, so that investors can see through the eyes of Tesla management regarding important financial metrics that Tesla uses to run the business and allowing investors to better understand Tesla's performance. Non-GAAP information is not prepared under a comprehensive set of accounting rules and therefore, should only be read in conjunction with financial information reported under U.S. GAAP when understanding Tesla's operating performance. A reconciliation between GAAP and non-GAAP financial information is provided above.", - "path": "TSLA-Q4-2025-Update.pdf-->NON-GAAP FINANCIAL INFORMATION", + "path": "TSLA-Q4-2025-Update.pdf/NON-GAAP FINANCIAL INFORMATION", "metadata": { "length": 1664, "summary": "Tesla presents consolidated financial information under both GAAP and non-GAAP standards to supplement its results. Key non-GAAP measures include net income attributable to common stockholders, diluted per share figures, Adjusted EBITDA margin, automotive gross margin, and free cash flow. These metrics aid internal management comparisons with historical performance and other companies, supporting operating, budgeting, and planning activities. Management believes these non-GAAP figures provide investors with a clearer view of Tesla's financial condition and operational results by reflecting the metrics used to run the business. However, since non-GAAP data is not prepared under comprehensive accounting rules, it should be read alongside U.S. GAAP information for a complete understanding of Tesla's performance. A reconciliation between GAAP and non-GAAP data is provided elsewhere.", @@ -3077,7 +3077,7 @@ "chunk_id": "34263854-525e-5a28-b9fd-a88d4813756e", "type": "text", "content": "Certain statements in this update, including, but not limited to, statements in the “Outlook” section; statements relating to the development, strategy, ramp, production and capacity, demand and market growth, cost, pricing and profitability, investment, deliveries, deployment, availability and other features and improvements and timing of existing and future Tesla products and services and supporting infrastructure; statements regarding operating margin, operating profits, spending and liquidity; and statements regarding expansion, improvements and/or ramp and related timing at our facilities are “forward-looking statements” within the meaning of the Private Securities Litigation Reform Act of 1995. Forward-looking statements are based on assumptions and management’s current expectations, involve certain risks and uncertainties, and are not guarantees. Future results may differ materially from those expressed in any forward-looking statement. The following important factors, without limitation, could cause actual results to differ materially from those in the forward-looking statements: the risk of delays in launching and/or manufacturing our products, services and features cost-effectively; our ability to build and/or grow our products and services, sales, delivery, installation, servicing and charging capabilities and effectively manage this growth; our ability to successfully and timely develop, introduce and scale, as well as our consumers’ demand for, products and services based on artificial intelligence, robotics and automation, electric vehicles, advanced driver assistance systems, and ride-hailing services generally and our vehicles and services specifically; the ability of suppliers to deliver components according to schedules, prices, quality and volumes acceptable to us, and our ability to manage such components effectively; any issues with lithium-ion cells or other components manufactured at our factories; our ability to ramp our factories in accordance with our plans; our ability to procure supply of battery cells, including through our own manufacturing; risks relating to international operations and expansion, including unfavorable and uncertain regulatory, political, economic, tax, tariff, export controls and labor conditions; any failures by Tesla products to perform as expected or if product recalls occur; the risk of product liability claims; competition in the automotive, transportation and energy product and services and robotics markets; our ability to maintain public credibility and confidence in our long-term business prospects; our ability to manage risks relating to our various product financing programs; the status of government and economic incentives for electric vehicles and energy products; our ability to attract, hire and retain key employees and qualified personnel; our ability to maintain the security of our information and production and product systems; our compliance with various regulations and laws applicable to our operations and products, which may evolve from time to time; risks relating to our indebtedness and financing strategies; and adverse foreign exchange movements. More information on potential factors that could affect our financial results is included from time to time in our Securities and Exchange Commission filings and reports, including the risks identified under the section captioned “Risk Factors” in our annual report on Form 10-K filed with the SEC on January 30, 2025 and subsequent quarterly reports on Form 10-Q. Tesla disclaims any obligation to update information contained in these forward-looking statements whether as a result of new information, future events or otherwise.\nTESLA", - "path": "TSLA-Q4-2025-Update.pdf-->FORWARD-LOOKING STATEMENTS", + "path": "TSLA-Q4-2025-Update.pdf/FORWARD-LOOKING STATEMENTS", "metadata": { "length": 3711, "summary": "This passage from Tesla outlines that various statements in the update, particularly regarding future outlooks, product development, production capacity, financial metrics, and facility expansions, constitute forward-looking statements under the Private Securities Litigation Reform Act of 1995. These statements reflect management's current expectations based on assumptions and are subject to risks and uncertainties, meaning actual results may differ materially. The text lists numerous specific risk factors that could impact outcomes, including manufacturing delays, supply chain challenges, regulatory hurdles, competition, product performance issues, employee retention, and foreign exchange fluctuations. Tesla advises investors to consult SEC filings, specifically the Form 10-K filed on January 30, 2025, for detailed risk disclosures. The company explicitly disclaims any obligation to update these forward-looking statements due to new information or future events.", diff --git a/apps/api/app/services/auth/api_key_authentication_service.py b/apps/api/app/services/auth/api_key_authentication_service.py index 67e677247..2625892d4 100644 --- a/apps/api/app/services/auth/api_key_authentication_service.py +++ b/apps/api/app/services/auth/api_key_authentication_service.py @@ -2,7 +2,6 @@ from __future__ import annotations -import asyncio import json from datetime import datetime, timezone @@ -11,11 +10,11 @@ from sqlalchemy.ext.asyncio import AsyncSession from shared.core.config import redis_pool_manager -from shared.core.database import get_db_context from shared.services.redis.redis_service import RedisService from shared.utils.api_keys import hash_api_key _API_KEY_USER_CACHE_TTL_SECONDS: int = 3600 +_LAST_USED_DEBOUNCE_SECONDS: int = 300 class APIKeyAuthenticationService: @@ -44,8 +43,12 @@ async def validate_api_key( if not api_key_record or not api_key_record.is_valid(): return None - self._schedule_last_used_update(str(api_key_record.id)) user_id = str(api_key_record.user_id) + await self._update_last_used_best_effort( + session, + redis_service, + str(api_key_record.id), + ) await self._set_cached_user_id( redis_service, key_hash, @@ -75,6 +78,10 @@ def _get_user_id_key(api_key_hash: str) -> str: def _get_user_api_keys_key(user_id: str) -> str: return f"api-key:user-hashes:{user_id}" + @staticmethod + def _get_last_used_debounce_key(api_key_id: str) -> str: + return f"api-key:last-used-debounce:{api_key_id}" + async def _get_cached_user_id( self, redis_service: RedisService, @@ -159,20 +166,44 @@ def _resolve_api_key_cache_ttl_seconds(self, expires_at: datetime | None) -> int remaining_seconds = int((expires_at_utc - now).total_seconds()) return max(1, min(_API_KEY_USER_CACHE_TTL_SECONDS, remaining_seconds)) - def _schedule_last_used_update(self, api_key_id: str) -> None: + async def _update_last_used_best_effort( + self, + session: AsyncSession, + redis_service: RedisService, + api_key_id: str, + ) -> None: + """Update last_used_at on the request session without a nested checkout. + + Redis SET NX debounce skips redundant writes within the debounce window + so job polls do not compete for QueuePool via create_task+get_db_context. + """ + debounce_key = self._get_last_used_debounce_key(api_key_id) try: - asyncio.create_task( - self._update_last_used_best_effort(api_key_id), - name=f"api_key_last_used:{api_key_id}", + acquired = await redis_service.set_nx( + debounce_key, + "1", + ex=_LAST_USED_DEBOUNCE_SECONDS, ) - except Exception as exc: + if not acquired: + return + except Exception: logger.warning( - f"Failed to schedule API key last-used update (ignored): {exc}" + "api_key_authentication: last-used debounce failed for api_key_id={}; " + "updating anyway", + api_key_id, ) - async def _update_last_used_best_effort(self, api_key_id: str) -> None: try: - async with get_db_context() as db: - await self._repository.update_last_used(db, api_key_id) + await self._repository.update_last_used(session, api_key_id) + # Request-scoped sessions do not auto-commit. + await session.commit() except Exception as exc: - logger.warning(f"Failed to update API key last-used time (ignored): {exc}") + logger.warning( + f"Failed to update API key last-used time (ignored): {exc}" + ) + try: + await session.rollback() + except Exception: + # Rollback itself can fail if the session is already closed; + # ignore so last-used remains best-effort. + pass diff --git a/apps/api/app/services/demo/source_catalog.py b/apps/api/app/services/demo/source_catalog.py index 054f48b92..534570abb 100644 --- a/apps/api/app/services/demo/source_catalog.py +++ b/apps/api/app/services/demo/source_catalog.py @@ -69,7 +69,7 @@ class DemoSourceDefinition: ), citations=( DemoCitationDefinition( - section_path=("TSLA-Q4-2025-Update.pdf-->OTHER UPDATES"), + section_path=("TSLA-Q4-2025-Update.pdf/OTHER UPDATES"), description="xAI investment", content=( "On January 16, 2026, Tesla entered into an agreement " @@ -92,7 +92,7 @@ class DemoSourceDefinition: citations=( DemoCitationDefinition( section_path=( - "TSLA-Q4-2025-Update.pdf-->SUMMARY-->" + "TSLA-Q4-2025-Update.pdf/SUMMARY/" "Energy generation and storage" ), description="Storage deployment growth", @@ -115,7 +115,7 @@ class DemoSourceDefinition: ), citations=( DemoCitationDefinition( - section_path=("TSLA-Q4-2025-Update.pdf-->OUTLOOK-->Product"), + section_path=("TSLA-Q4-2025-Update.pdf/OUTLOOK/Product"), description="2026 production plans", content=( "Cybercab, Tesla Semi and Megapack 3 are on schedule " diff --git a/apps/api/app/services/demo/source_projection.py b/apps/api/app/services/demo/source_projection.py index ce86b6a7a..8758067af 100644 --- a/apps/api/app/services/demo/source_projection.py +++ b/apps/api/app/services/demo/source_projection.py @@ -5,6 +5,11 @@ from typing import Any, Protocol from urllib.parse import quote +from shared.services.chunks.path_segments import ( + join_document_path, + split_escaped_document_path, +) + _SOURCE_FILE_EXTENSIONS = ( ".csv", ".doc", @@ -241,22 +246,18 @@ def _publication_path( if not raw: return prefix - if "-->" in raw: - sections = [part.strip() for part in raw.split("-->")[1:] if part.strip()] - return "/".join([prefix, *sections]) if sections else prefix - if raw.startswith("images/") or raw.startswith("tables/"): return f"{prefix}/Assets/{raw}" - parts = [part.strip() for part in raw.split("/") if part.strip()] + parts = split_escaped_document_path(raw) if parts and parts[0] == source.title: - return raw + return join_document_path(parts) if len(parts) >= 2 and parts[0] == "Default_Root": section_parts = parts[2:] if parts[1] == source.title else parts[1:] - return "/".join([prefix, *section_parts]) if section_parts else prefix + return join_document_path([prefix, *section_parts]) if section_parts else prefix if parts and _is_source_file_root(parts[0]): section_parts = parts[1:] - return "/".join([prefix, *section_parts]) if section_parts else prefix + return join_document_path([prefix, *section_parts]) if section_parts else prefix return prefix diff --git a/apps/api/app/services/jobs/result_projection.py b/apps/api/app/services/jobs/result_projection.py index 199625580..b39166c9d 100644 --- a/apps/api/app/services/jobs/result_projection.py +++ b/apps/api/app/services/jobs/result_projection.py @@ -15,6 +15,7 @@ JobStatusValue = Literal[ "pending", "waiting-file", "running", "converting", "done", "failed" ] +_CHARGED_BILLING_STATUS: str = "charged" def build_error_response( @@ -112,6 +113,17 @@ def _resolve_duration_seconds(job: Any) -> float | None: return None +def _resolve_credits_spent(job: Any) -> float: + if to_job_status_value(job.status) == "failed": + return 0.0 + + billing_status = getattr(job, "billing_status", None) + if billing_status != _CHARGED_BILLING_STATUS: + return 0.0 + + return MicroDollar(getattr(job, "credits_charged", 0) or 0).to_credit() + + async def _resolve_result_delivery( job: Any, ) -> tuple[dict[str, Any] | None, str | None, datetime]: @@ -162,9 +174,5 @@ async def build_job_result_response( model=parsing_params.get("model"), ocr_enabled=parsing_params.get("ocr_enabled"), duration_seconds=_resolve_duration_seconds(job), - credits_spent=( - MicroDollar(job.credits_charged).to_credit() - if hasattr(job, "credits_charged") - else 0 - ), + credits_spent=_resolve_credits_spent(job), ) diff --git a/apps/api/app/services/rate_limit/job_admission_service.py b/apps/api/app/services/rate_limit/job_admission_service.py index 130a880c3..1e46211da 100644 --- a/apps/api/app/services/rate_limit/job_admission_service.py +++ b/apps/api/app/services/rate_limit/job_admission_service.py @@ -33,8 +33,9 @@ async def resolve_current_user( *, route_context: RouteAdmissionContext, user_id: str, + db: AsyncSession, ) -> CurrentUser: - user_tier = await TierService.get_tier(user_id) + user_tier = await TierService.get_tier(user_id, session=db) self._route_policy_service.enforce_guest_api_key_scope( route_context=route_context, user_tier=user_tier, diff --git a/apps/api/app/services/rate_limit/tier_service.py b/apps/api/app/services/rate_limit/tier_service.py index 91bb73d00..e62a82124 100644 --- a/apps/api/app/services/rate_limit/tier_service.py +++ b/apps/api/app/services/rate_limit/tier_service.py @@ -31,9 +31,15 @@ class TierService: """Manages user tier lookup, caching, and refresh.""" @staticmethod - async def get_tier(user_id: str) -> str: + async def get_tier( + user_id: str, + session: AsyncSession | None = None, + ) -> str: """Return a user's tier from cache or database. + When ``session`` is provided, reuse it instead of opening a nested + ``get_db_context`` checkout (critical on the job-poll auth path). + Missing user tier state is treated as invalid data and raises directly; this method never falls back to a default tier for user lookup. """ @@ -42,18 +48,43 @@ async def get_tier(user_id: str) -> str: if cached_tier is not None: return cached_tier - async with get_db_context() as session: - try: - user_tier: str = await TierService._get_tier_from_db(session, user_id) - except NotFoundException: - user_tier = await TierService._initialize_missing_user_tier( - session, + if session is not None: + user_tier = await TierService._resolve_tier_from_db( + session, + user_id, + commit_on_initialize=True, + ) + else: + async with get_db_context() as owned_session: + user_tier = await TierService._resolve_tier_from_db( + owned_session, user_id, + commit_on_initialize=False, ) await TierService._set_cached_tier(redis_service, user_id, user_tier) return user_tier + @staticmethod + async def _resolve_tier_from_db( + session: AsyncSession, + user_id: str, + *, + commit_on_initialize: bool, + ) -> str: + """Load tier from DB, initializing missing first-use billing state.""" + try: + return await TierService._get_tier_from_db(session, user_id) + except NotFoundException: + user_tier = await TierService._initialize_missing_user_tier( + session, + user_id, + ) + # Request-scoped sessions do not auto-commit; get_db_context does. + if commit_on_initialize: + await session.commit() + return user_tier + @staticmethod async def refresh_tier(user_id: str, session: AsyncSession) -> str: """Called on payment success. diff --git a/apps/api/tests/contract/test_chunk_document_path_contract.py b/apps/api/tests/contract/test_chunk_document_path_contract.py index d10a99c46..76943f3fd 100644 --- a/apps/api/tests/contract/test_chunk_document_path_contract.py +++ b/apps/api/tests/contract/test_chunk_document_path_contract.py @@ -89,38 +89,6 @@ def test_should_not_treat_dotted_section_titles_as_legacy_document_files() -> No assert children[0]["path"] == chunk_path -def test_should_read_arrow_delimited_document_paths_as_section_paths() -> None: - chunk_path = "report.pdf-->Intro-->Subsection" - - doc_nav = ZipResultSchemaBuilder().build_doc_nav( - [ - { - "chunk_id": "chunk_arrow_delimited_path", - "type": "text", - "content": "arrow delimited content", - "path": chunk_path, - "metadata": {"summary": "arrow delimited summary"}, - } - ], - "report.pdf", - ) - - sections = cast(list[dict[str, object]], doc_nav["sections"]) - children = cast(list[dict[str, object]], sections[0]["children"]) - - assert ( - section_path_from_chunk_path( - chunk_path, - source_file_name="report.pdf", - ) - == "Intro / Subsection" - ) - assert sections[0]["title"] == "Intro" - assert sections[0]["path"] == "report.pdf/Intro" - assert children[0]["title"] == "Subsection" - assert children[0]["path"] == "report.pdf/Intro/Subsection" - - def test_should_preserve_literal_arrow_text_in_slash_paths() -> None: chunk_path = "report.pdf/Inputs --> Outputs/Details" diff --git a/apps/api/tests/contract/test_job_creation_contract.py b/apps/api/tests/contract/test_job_creation_contract.py index a375f5151..c5642f0e3 100644 --- a/apps/api/tests/contract/test_job_creation_contract.py +++ b/apps/api/tests/contract/test_job_creation_contract.py @@ -417,7 +417,7 @@ async def test_should_create_a_v2_page_memory_job_for_pdf_uploads( assert job_metadata["processing_generation"] == "page_memory" assert job_metadata["source_file_name"] == payload["file_name"] assert page_memory_config["max_pages"] == 1500 - assert page_memory_config["asset_extraction_enabled"] is False + assert page_memory_config["asset_extraction_enabled"] is True assert original_request["file_name"] == payload["file_name"] assert "parse_track" not in original_request diff --git a/apps/api/tests/contract/test_job_read_contract.py b/apps/api/tests/contract/test_job_read_contract.py index 7e52f3b68..ef6c68617 100644 --- a/apps/api/tests/contract/test_job_read_contract.py +++ b/apps/api/tests/contract/test_job_read_contract.py @@ -132,6 +132,46 @@ async def test_should_return_job_details_for_an_existing_waiting_file_job( assert response_json["credits_spent"] == 0.0 +@pytest.mark.asyncio +async def test_should_report_zero_credits_spent_for_refunded_failed_job( + developer_api_client_factory: Callable[ + [], AbstractAsyncContextManager[AsyncClient] + ], +) -> None: + async with developer_api_client_factory() as api_client: + created_job = await _create_waiting_file_job(api_client) + job_id = cast(str, created_job["job_id"]) + + await ContractDatabase.execute( + """ + UPDATE jobs + SET + status = 'failed', + error_code = 'INVALID_ARGUMENT', + error_message = 'Invalid file: the uploaded .docx file is not a valid Word document. Please check the file and upload again.', + credits_charged = 15000, + billing_status = 'refunded' + WHERE job_id = :job_id + """, + {"job_id": job_id}, + ) + + response = await api_client.get(f"/api/v1/jobs/{job_id}") + + assert response.status_code == 200 + + response_json = cast(dict[str, object], response.json()) + error = cast(dict[str, object], response_json["error"]) + + assert response_json["status"] == "failed" + assert error["code"] == "INVALID_ARGUMENT" + assert ( + error["message"] + == "Invalid file: the uploaded .docx file is not a valid Word document. Please check the file and upload again." + ) + assert response_json["credits_spent"] == 0.0 + + @pytest.mark.asyncio async def test_should_return_not_found_when_requesting_an_unknown_job( developer_api_client_factory: Callable[ diff --git a/apps/api/tests/unit/test_job_poll_session_hygiene.py b/apps/api/tests/unit/test_job_poll_session_hygiene.py new file mode 100644 index 000000000..3e8ce9b6a --- /dev/null +++ b/apps/api/tests/unit/test_job_poll_session_hygiene.py @@ -0,0 +1,358 @@ +"""Regression tests: job-poll auth must not open nested DB checkouts.""" + +from __future__ import annotations + +import sys +from pathlib import Path +from types import ModuleType, SimpleNamespace +from typing import Any +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest +from tests.support.import_environment import ( + configure_import_environment, + ensure_import_paths, +) + +# Defer importing apps/api `app` until test bodies run. Module-level imports +# would cache API's `app` package and break apps/worker contract collection in +# the same pytest process (both packages are named `app`). +configure_import_environment() +ensure_import_paths() + +_API_ROOT = str(Path(__file__).resolve().parents[2]) + + +def _prioritize_api_import_root() -> None: + """Keep apps/api ahead of apps/worker for the shared `app` package name.""" + ensure_import_paths() + if _API_ROOT in sys.path: + sys.path.remove(_API_ROOT) + sys.path.insert(0, _API_ROOT) + + +def _is_api_app_module(module: ModuleType | None) -> bool: + if module is None: + return False + module_file = getattr(module, "__file__", None) + if isinstance(module_file, str) and module_file.startswith(_API_ROOT): + return True + module_paths = getattr(module, "__path__", ()) + try: + return any(str(path).startswith(_API_ROOT) for path in module_paths) + except KeyError: + return False + + +def _drop_non_api_app_modules() -> None: + for module_name in sorted(sys.modules, key=len, reverse=True): + if module_name != "app" and not module_name.startswith("app."): + continue + if not _is_api_app_module(sys.modules.get(module_name)): + sys.modules.pop(module_name, None) + + +def _drop_api_app_modules() -> None: + for module_name in sorted(sys.modules, key=len, reverse=True): + if module_name != "app" and not module_name.startswith("app."): + continue + if _is_api_app_module(sys.modules.get(module_name)): + sys.modules.pop(module_name, None) + + +def _load_api_modules() -> tuple[ModuleType, ModuleType, ModuleType, ModuleType]: + _prioritize_api_import_root() + _drop_non_api_app_modules() + from app.services.auth import api_key_authentication_service + from app.services.rate_limit import ( + data_structures, + job_admission_service, + tier_service, + ) + + return ( + api_key_authentication_service, + data_structures, + job_admission_service, + tier_service, + ) + + +@pytest.fixture(autouse=True) +def _clear_api_app_modules_after_unit_test(): + """Avoid leaving API's `app` package cached for later worker contract tests.""" + yield + _drop_api_app_modules() + + + +@pytest.mark.asyncio +async def test_get_tier_reuses_provided_session_without_get_db_context() -> None: + _, _, _, tier_service = _load_api_modules() + TierService = tier_service.TierService + + session = AsyncMock() + redis_service = AsyncMock() + redis_service.get = AsyncMock(return_value=None) + redis_service.set = AsyncMock() + + with ( + patch( + "app.services.rate_limit.tier_service.redis_pool_manager.get_redis_service", + return_value=redis_service, + ), + patch( + "app.services.rate_limit.tier_service.get_db_context", + ) as get_db_context_mock, + patch.object( + TierService, + "_get_tier_from_db", + new=AsyncMock(return_value="pro"), + ) as get_tier_from_db, + ): + tier = await TierService.get_tier("user-1", session=session) + + assert tier == "pro" + get_tier_from_db.assert_awaited_once_with(session, "user-1") + get_db_context_mock.assert_not_called() + + +@pytest.mark.asyncio +async def test_resolve_current_user_passes_request_session_to_get_tier() -> None: + _, data_structures, job_admission_service, tier_service = _load_api_modules() + RouteAdmissionContext = data_structures.RouteAdmissionContext + JobAdmissionService = job_admission_service.JobAdmissionService + TierService = tier_service.TierService + + session = AsyncMock() + route_context = RouteAdmissionContext( + method="GET", + path="/v1/jobs/job_abc", + limit_identifier="GET:/v1/jobs/{job_id}", + ) + service = JobAdmissionService( + route_policy_service=MagicMock( + enforce_guest_api_key_scope=MagicMock(), + enforce_user_system_limit=AsyncMock(), + ), + ) + + with ( + patch.object( + TierService, + "get_tier", + new=AsyncMock(return_value="free"), + ) as get_tier, + patch( + "app.services.rate_limit.job_admission_service.RateLimitConfig.get_instance", + return_value=SimpleNamespace(is_enabled=False), + ), + ): + current_user = await service.resolve_current_user( + route_context=route_context, + user_id="user-1", + db=session, + ) + + assert current_user.user_id == "user-1" + assert current_user.user_tier == "free" + get_tier.assert_awaited_once_with("user-1", session=session) + + +@pytest.mark.asyncio +async def test_validate_api_key_updates_last_used_on_same_session() -> None: + api_key_authentication_service, _, _, _ = _load_api_modules() + APIKeyAuthenticationService = ( + api_key_authentication_service.APIKeyAuthenticationService + ) + + session = AsyncMock() + session.commit = AsyncMock() + redis_service = AsyncMock() + redis_service.get = AsyncMock(return_value=None) + redis_service.set_nx = AsyncMock(return_value=True) + redis_service.set = AsyncMock() + redis_service.sadd = AsyncMock() + redis_service.ttl = AsyncMock(return_value=-2) + redis_service.expire = AsyncMock() + + api_key_record = SimpleNamespace( + id="key-1", + user_id="user-1", + expires_at=None, + is_valid=lambda: True, + ) + repository = MagicMock() + repository.get_by_key_hash = AsyncMock(return_value=api_key_record) + repository.update_last_used = AsyncMock(return_value=True) + + service = APIKeyAuthenticationService(repository=repository) + + with ( + patch( + "app.services.auth.api_key_authentication_service.redis_pool_manager.get_redis_service", + return_value=redis_service, + ), + patch( + "app.services.auth.api_key_authentication_service.hash_api_key", + return_value="hash-1", + ), + patch("asyncio.create_task") as create_task_mock, + ): + user_id = await service.validate_api_key(session, "kw_test_key") + + assert user_id == "user-1" + repository.update_last_used.assert_awaited_once_with(session, "key-1") + session.commit.assert_awaited_once() + create_task_mock.assert_not_called() + redis_service.set_nx.assert_awaited_once_with( + "api-key:last-used-debounce:key-1", + "1", + ex=300, + ) + + +@pytest.mark.asyncio +async def test_validate_api_key_skips_last_used_when_debounced() -> None: + api_key_authentication_service, _, _, _ = _load_api_modules() + APIKeyAuthenticationService = ( + api_key_authentication_service.APIKeyAuthenticationService + ) + + session = AsyncMock() + session.commit = AsyncMock() + redis_service = AsyncMock() + redis_service.get = AsyncMock(return_value=None) + redis_service.set_nx = AsyncMock(return_value=False) + redis_service.set = AsyncMock() + redis_service.sadd = AsyncMock() + redis_service.ttl = AsyncMock(return_value=-2) + redis_service.expire = AsyncMock() + + api_key_record = SimpleNamespace( + id="key-1", + user_id="user-1", + expires_at=None, + is_valid=lambda: True, + ) + repository = MagicMock() + repository.get_by_key_hash = AsyncMock(return_value=api_key_record) + repository.update_last_used = AsyncMock(return_value=True) + + service = APIKeyAuthenticationService(repository=repository) + + with ( + patch( + "app.services.auth.api_key_authentication_service.redis_pool_manager.get_redis_service", + return_value=redis_service, + ), + patch( + "app.services.auth.api_key_authentication_service.hash_api_key", + return_value="hash-1", + ), + ): + user_id = await service.validate_api_key(session, "kw_test_key") + + assert user_id == "user-1" + repository.update_last_used.assert_not_called() + session.commit.assert_not_called() + + +@pytest.mark.asyncio +async def test_job_poll_auth_path_uses_single_session_factory_checkout() -> None: + """End-to-end hygiene: tier + last-used must not call get_db_context.""" + ( + api_key_authentication_service, + data_structures, + job_admission_service, + tier_service, + ) = _load_api_modules() + APIKeyAuthenticationService = ( + api_key_authentication_service.APIKeyAuthenticationService + ) + RouteAdmissionContext = data_structures.RouteAdmissionContext + JobAdmissionService = job_admission_service.JobAdmissionService + TierService = tier_service.TierService + + session = AsyncMock() + session.commit = AsyncMock() + checkout_count = {"n": 0} + + class _CountingContext: + async def __aenter__(self) -> Any: + checkout_count["n"] += 1 + return AsyncMock() + + async def __aexit__(self, *args: object) -> None: + return None + + redis_service = AsyncMock() + redis_service.get = AsyncMock(side_effect=[None, None]) # api-key miss, tier miss + redis_service.set_nx = AsyncMock(return_value=True) + redis_service.set = AsyncMock() + redis_service.sadd = AsyncMock() + redis_service.ttl = AsyncMock(return_value=-2) + redis_service.expire = AsyncMock() + + api_key_record = SimpleNamespace( + id="key-1", + user_id="user-1", + expires_at=None, + is_valid=lambda: True, + ) + repository = MagicMock() + repository.get_by_key_hash = AsyncMock(return_value=api_key_record) + repository.update_last_used = AsyncMock(return_value=True) + auth_service = APIKeyAuthenticationService(repository=repository) + + route_context = RouteAdmissionContext( + method="GET", + path="/v1/jobs/job_abc", + limit_identifier="GET:/v1/jobs/{job_id}", + ) + admission = JobAdmissionService( + route_policy_service=MagicMock( + enforce_guest_api_key_scope=MagicMock(), + enforce_user_system_limit=AsyncMock(), + ), + ) + + with ( + patch( + "app.services.auth.api_key_authentication_service.redis_pool_manager.get_redis_service", + return_value=redis_service, + ), + patch( + "app.services.rate_limit.tier_service.redis_pool_manager.get_redis_service", + return_value=redis_service, + ), + patch( + "app.services.auth.api_key_authentication_service.hash_api_key", + return_value="hash-1", + ), + patch( + "app.services.rate_limit.tier_service.get_db_context", + side_effect=_CountingContext, + ), + patch.object( + TierService, + "_get_tier_from_db", + new=AsyncMock(return_value="free"), + ), + patch( + "app.services.rate_limit.job_admission_service.RateLimitConfig.get_instance", + return_value=SimpleNamespace(is_enabled=False), + ), + patch("asyncio.create_task") as create_task_mock, + ): + user_id = await auth_service.validate_api_key(session, "kw_test_key") + current_user = await admission.resolve_current_user( + route_context=route_context, + user_id=user_id or "", + db=session, + ) + + assert current_user.user_id == "user-1" + assert checkout_count["n"] == 0 + create_task_mock.assert_not_called() + repository.update_last_used.assert_awaited_once_with(session, "key-1") diff --git a/apps/worker/app/services/connect_builder/summary_builder.py b/apps/worker/app/services/connect_builder/summary_builder.py index 960edb836..c02a57c4d 100644 --- a/apps/worker/app/services/connect_builder/summary_builder.py +++ b/apps/worker/app/services/connect_builder/summary_builder.py @@ -1,15 +1,18 @@ """ summary_builder: Bottom-up recursive summarization for document navigation. -Reads doc_nav.json + chunks.json for a file and generates ``summary`` fields -at every intermediate node via LLM aggregation. -The enriched doc_nav.json is written back to disk. +Reads doc_nav.json and in-memory chunks, generates ``summary`` at every +non-leaf via deterministic covers assembly or LLM aggregation (with self_only). +Persists document-level ``top_summary`` on doc_nav (default LLM); section-level +LLM remains opt-in via ``use_llm``. Usage (standalone): from app.services.connect_builder.summary_builder import enrich_doc_nav_summaries - enrich_doc_nav_summaries(document_workspace_dir, source_file="report.pdf") + enrich_doc_nav_summaries(document_workspace_dir, source_file="report.pdf", chunks=chunks) """ +from __future__ import annotations + import json import os from typing import Any, Dict, List, Optional, Tuple @@ -20,38 +23,78 @@ # ─── Constants ──────────────────────────────────────────────────────────────── -# Summary output max length for recursive LLM aggregation (≤100 chars per node) +# LLM trigger: sum of child contribution lengths + self_only must exceed this SUMMARY_MAX_LEN = 100 -# Navigation top-summary budget measured in semantic tokens (count_cn_en) +# Deterministic (and top-level LLM output) head…tail token budget +DETERMINISTIC_SUMMARY_HEAD = 100 +DETERMINISTIC_SUMMARY_TAIL = 100 +# Navigation top-summary LLM max_tokens NAVIGATION_TOP_SUMMARY_MAX_TOKENS = 200 -# TODO: revisit this cap after we collect more real-world prompt/token budget data. -NON_LLM_TOP_SUMMARY_MAX_SECTIONS = 20 -NON_LLM_TOP_SUMMARY_MAX_DEPTH = 2 + +SECTION_COVERS_PREFIX = "This section covers: " +DOCUMENT_INCLUDES_PREFIX = "This document includes: " # ─── LLM Interface ─────────────────────────────────────────────────────────── -def _llm_summarize(snippets_text: str, node_name: str, max_tokens: int = 100) -> str: - """ - Call LLM to produce a concise summary from aggregated child snippets. +def _build_scope_payload_text( + *, + node_name: str, + self_only: str, + child_rows: List[Tuple[str, str]], +) -> str: + """Flatten SCOPE fields for language detection / logging.""" + titles = [title for title, _ in child_rows] + covered = "\n".join(f"- [{title}] {contrib}" for title, contrib in child_rows) + return "\n".join( + [ + f"SCOPE_TITLE: {node_name}", + f"self_only: {'yes' if self_only.strip() else 'no'}", + f"children: {', '.join(titles)}", + f"SELF_ONLY_CONTENT:\n{self_only.strip() or '(none)'}", + f"COVERED_NODES:\n{covered or '(none)'}", + ] + ) - Returns plain text summary, or "" on failure. + +def _llm_summarize( + *, + node_name: str, + self_only: str, + child_rows: List[Tuple[str, str]], + max_tokens: int = 100, +) -> str: + """Call LLM to produce a concise summary for one scope. + + Returns plain text summary, or "" on failure / null. """ try: from shared.services.ai.prompt_service import build_prompt, _detect_text_language from shared.services.ai.llm_overrides import get_text_client - # Deterministic language lock — see prompt_service._language_directive - detected_lang = _detect_text_language(snippets_text) + payload_text = _build_scope_payload_text( + node_name=node_name, + self_only=self_only, + child_rows=child_rows, + ) + detected_lang = _detect_text_language(payload_text) + child_titles = [title for title, _ in child_rows] + covered_nodes = "\n".join( + f"- [{title}] {contrib}" for title, contrib in child_rows + ) prompt, temperature, top_p, _prompt_max_tokens = build_prompt( task="file-summary", - texts=snippets_text, + texts=payload_text, query="", paras={ "max_tokens": max_tokens, "node_name": node_name, "lang": detected_lang, + "has_self_only": bool(self_only.strip()), + "child_titles": child_titles, + "self_only_content": self_only.strip() or "(none)", + "covered_nodes": covered_nodes or "(none)", }, ) messages: list[ChatCompletionMessageParam] = [ @@ -78,11 +121,7 @@ def _llm_summarize(snippets_text: str, node_name: str, max_tokens: int = 100) -> return "" - -# -# Uses explicit children arrays: -# [{"title": "Section A", "summary": "...", "children": [{"title": "SubA1", ...}]}] -# ═══════════════════════════════════════════════════════════════════════════════ +# ─── doc_nav I/O ───────────────────────────────────────────────────────────── DOC_NAV_FILENAME = "doc_nav.json" @@ -130,6 +169,113 @@ def ensure_doc_nav_json( return nav_path +# ─── self_only + deterministic assembly ────────────────────────────────────── + + +def build_self_only_lookup( + chunks: List[Dict[str, Any]], + *, + source_file_name: str = "", +) -> Dict[str, str]: + """Map canonical section_path → concatenated exact-path chunk content. + + Exact path only (no descendants): same semantics as hydrate ``self_only``. + """ + from shared.services.retrieval.search.lexical_text import section_path_from_chunk_path + + by_path: Dict[str, List[str]] = {} + for chunk in chunks or []: + if not isinstance(chunk, dict): + continue + raw_path = str(chunk.get("path") or "").strip() + if not raw_path: + continue + section_path = section_path_from_chunk_path( + raw_path, + source_file_name=source_file_name, + ) + if not section_path or section_path == "Root": + continue + content = str(chunk.get("content") or chunk.get("text") or "").strip() + if not content: + metadata = chunk.get("metadata") or {} + if isinstance(metadata, dict): + content = str(metadata.get("summary") or "").strip() + if not content: + continue + by_path.setdefault(section_path, []).append(content) + return {path: "\n".join(parts) for path, parts in by_path.items()} + + +def _node_section_path(node: Dict[str, Any], source_file_name: str) -> str: + from shared.services.retrieval.search.lexical_text import section_path_from_chunk_path + + nav_path = str(node.get("path") or "").strip() + if not nav_path: + return "" + return section_path_from_chunk_path(nav_path, source_file_name=source_file_name) + + +def _deterministic_section_summary( + *, + is_top_level: bool, + self_only: str, + child_titles: List[str], +) -> str: + """covers/includes prefix → self_only → child titles; then head…tail whole string.""" + from shared.utils.text_utils import truncate_content_preview + + prefix = DOCUMENT_INCLUDES_PREFIX if is_top_level else SECTION_COVERS_PREFIX + segments: List[str] = [] + self_text = (self_only or "").strip() + if self_text: + segments.append(self_text) + if child_titles: + segments.append(", ".join(child_titles)) + assembled = prefix + " ".join(segments) if segments else prefix.rstrip() + return truncate_content_preview( + assembled, + head=DETERMINISTIC_SUMMARY_HEAD, + tail=DETERMINISTIC_SUMMARY_TAIL, + ) + + +def _child_title_list( + children: List[Dict[str, Any]], + *, + is_top_level: bool, +) -> List[str]: + """All direct child titles (empty titles omitted); top-level skips 'root'.""" + titles: List[str] = [] + for child in children: + title = str(child.get("title") or "").strip() + if not title: + continue + if is_top_level and title.lower() == "root": + continue + titles.append(title) + return titles + + +def _child_contribution_rows( + children: List[Dict[str, Any]], + *, + is_top_level: bool, +) -> List[Tuple[str, str]]: + """(title, contribution) where contribution = summary or title.""" + rows: List[Tuple[str, str]] = [] + for child in children: + title = str(child.get("title") or "").strip() + if is_top_level and title.lower() == "root": + continue + summary = str(child.get("summary") or "").strip() + contrib = summary or title + if not title and not contrib: + continue + rows.append((title or contrib, contrib)) + return rows + + # ─── Recursive summarization on doc_nav sections ───────────────────────────── @@ -137,92 +283,86 @@ def _recursive_summarize_nav( node: Dict[str, Any], use_llm: bool = True, is_top_level: bool = False, + *, + self_only_lookup: Optional[Dict[str, str]] = None, + source_file_name: str = "", ) -> str: """Bottom-up recursive summarization on a doc_nav section node. - Operates on the children-array tree structure of doc_nav.json. - - For each node: - - Leaf (children==[]) → keep existing summary (set during ZIP creation). - - Non-leaf → recursively summarize children, then aggregate. - - Writes summary in-place into ``node["summary"]``. - Returns the summary string. + - Leaf → keep existing summary. + - Non-leaf → recurse children, then: + - use_llm=False → always deterministic covers (titles ± self_only); ignore 100 + - use_llm=True → LLM only if sum(contrib lens)+len(self_only) > SUMMARY_MAX_LEN; + otherwise same deterministic covers path """ - children = node.get("children", []) - title = node.get("title", "") + children = list(node.get("children") or []) + title = str(node.get("title") or "") if not children: - # Leaf node — keep existing summary - existing = (node.get("summary") or "").strip() + existing = str(node.get("summary") or "").strip() if existing: node["summary"] = existing - return node.get("summary", "") + return str(node.get("summary") or "") - # Recurse into children - child_summaries: List[Tuple[str, str]] = [] + lookup = self_only_lookup or {} for child in children: - child_summary = _recursive_summarize_nav(child, use_llm, is_top_level=False) - if child_summary: - child_summaries.append((child.get("title", ""), child_summary)) - - if not child_summaries: - return node.get("summary", "") - - # Aggregate child summaries without hard truncation - aggregated_parts = [] - for name, summary in child_summaries: - aggregated_parts.append(f"[{name}] {summary}") - - aggregated_text = "\n".join(aggregated_parts) + _recursive_summarize_nav( + child, + use_llm=use_llm, + is_top_level=False, + self_only_lookup=lookup, + source_file_name=source_file_name, + ) - max_len = NAVIGATION_TOP_SUMMARY_MAX_TOKENS if is_top_level else SUMMARY_MAX_LEN + section_path = _node_section_path(node, source_file_name) + self_only = "" + if section_path and section_path != "Root": + self_only = str(lookup.get(section_path) or "") + if self_only.strip(): + node["self_summary"] = self_only.strip() + elif "self_summary" in node: + node.pop("self_summary", None) + + child_titles = _child_title_list(children, is_top_level=is_top_level) + child_rows = _child_contribution_rows(children, is_top_level=is_top_level) + contrib_len = sum(len(contrib) for _, contrib in child_rows) + len(self_only) + + deterministic = _deterministic_section_summary( + is_top_level=is_top_level, + self_only=self_only, + child_titles=child_titles, + ) - if len(child_summaries) <= 1 and not is_top_level: - result = child_summaries[0][1] + # use_llm=False → always title/covers path; 100-threshold only applies when LLM is on + if not use_llm: + result = deterministic + elif contrib_len > SUMMARY_MAX_LEN: + max_tokens = ( + NAVIGATION_TOP_SUMMARY_MAX_TOKENS if is_top_level else SUMMARY_MAX_LEN + ) + result = _llm_summarize( + node_name=title, + self_only=self_only, + child_rows=child_rows, + max_tokens=max_tokens, + ) else: - if is_top_level and not use_llm: - titles = [name for name, _ in child_summaries if name.lower() != "root"] - else: - titles = [name for name, _ in child_summaries] - - enum_prefix = "This document includes: " if is_top_level else "This section covers: " - title_enum = enum_prefix + ", ".join(titles) - - if not use_llm: - result = title_enum - else: - total_len = sum(len(s) for _, s in child_summaries) - if total_len > SUMMARY_MAX_LEN: - result = _llm_summarize(aggregated_text, title, max_tokens=max_len) - if not result: - result = title_enum - else: - result = title_enum + result = deterministic node["summary"] = result return result def _doc_nav_has_enriched_summaries(doc_nav: Dict[str, Any]) -> bool: - """Check if enrichment has already been run on this doc_nav. + """True iff every non-leaf has a non-empty summary.""" - Aligned with original enrichment logic: - In doc_nav.json, leaf nodes already have summary from ZIP creation, - so only non-leaf (parent) summaries are set by enrichment. - We recursively check that ALL non-leaf nodes across all depths - have a non-empty summary — if any is missing, enrichment is incomplete. - """ def _check_sections(sections: List[Dict[str, Any]]) -> bool: - """Returns True if all non-leaf nodes in sections have summaries.""" for section in sections: children = section.get("children", []) if not children: continue - # This is a non-leaf node — must have summary from enrichment if not section.get("summary"): return False - # Recurse into children to check deeper non-leaf nodes if not _check_sections(children): return False return True @@ -230,7 +370,6 @@ def _check_sections(sections: List[Dict[str, Any]]) -> bool: sections = doc_nav.get("sections", []) if not sections: return False - # Must have at least one non-leaf to be considered enriched has_non_leaf = any(s.get("children") for s in sections) if not has_non_leaf: return False @@ -238,38 +377,77 @@ def _check_sections(sections: List[Dict[str, Any]]) -> bool: def _build_nav_top_summary( - doc_nav: Dict[str, Any], - use_llm: bool = True + doc_nav: Dict[str, Any], + use_llm: bool = True, + *, + self_only_lookup: Optional[Dict[str, str]] = None, + source_file_name: str = "", ) -> str: - """Build navigation-facing top summary from enriched doc_nav.json. + """Build document-level top summary from already-enriched section nodes. - Strategy: - Treat all sections as children of a virtual Document node and recursively summarize. + Children are never re-summarized with LLM here — section LLM is controlled + only by ``enrich_doc_nav_summaries(use_llm=...)``. This path may optionally + LLM-aggregate the document overview from child contributions. """ - sections = doc_nav.get("sections", []) + sections = list(doc_nav.get("sections") or []) if not sections: return "" + file_name = source_file_name or str(doc_nav.get("file_name") or "") + # Fill any missing child summaries deterministically without enabling LLM. + for section in sections: + if isinstance(section, dict): + _recursive_summarize_nav( + section, + use_llm=False, + self_only_lookup=self_only_lookup, + source_file_name=file_name, + ) - virtual_doc_node = { - "title": "Document Overview", - "children": sections - } - - top_summary = _recursive_summarize_nav( - virtual_doc_node, - use_llm=use_llm, - is_top_level=True + child_titles = _child_title_list(sections, is_top_level=True) + child_rows = _child_contribution_rows(sections, is_top_level=True) + contrib_len = sum(len(contrib) for _, contrib in child_rows) + deterministic = _deterministic_section_summary( + is_top_level=True, + self_only="", + child_titles=child_titles, ) - - return top_summary + if not use_llm or contrib_len <= SUMMARY_MAX_LEN: + return deterministic + + llm_result = _llm_summarize( + node_name="Document Overview", + self_only="", + child_rows=child_rows, + max_tokens=NAVIGATION_TOP_SUMMARY_MAX_TOKENS, + ) + return llm_result or deterministic + + +def _persist_doc_nav_top_summary( + *, + file_dir: str, + doc_nav: Dict[str, Any], + top_summary: str, +) -> str: + """Write document-level top_summary once onto doc_nav and save.""" + cleaned = str(top_summary or "").strip() + if cleaned: + doc_nav["top_summary"] = cleaned + elif "top_summary" in doc_nav: + doc_nav.pop("top_summary", None) + _save_doc_nav(file_dir, doc_nav) + return cleaned def enrich_doc_nav_summaries( document_workspace_dir: str, source_file: Optional[str] = None, force: bool = False, - use_llm: bool = True, + use_llm: bool = False, + *, + top_summary_use_llm: bool = True, + chunks: Optional[List[Dict[str, Any]]] = None, ) -> Dict[str, str]: """Enrich doc_nav.json with bottom-up recursive summaries. @@ -277,13 +455,16 @@ def enrich_doc_nav_summaries( document_workspace_dir: Absolute path to the temporary document workspace. source_file: If given, only process this file. Otherwise process all. force: If True, regenerate even if summaries already exist. - use_llm: If True, use LLM for multi-child aggregation. + use_llm: Section-level LLM summaries (default off). + top_summary_use_llm: Document-level top summary LLM (default on). + chunks: In-memory parse chunks for exact-path self_only extraction. Returns: Dict mapping file_name → top-level summary string. """ results: Dict[str, str] = {} - mode_label = "LLM" if use_llm else "title-concat" + section_mode = "LLM" if use_llm else "title-concat" + top_mode = "LLM" if top_summary_use_llm else "title-concat" if source_file: targets = [source_file] @@ -302,35 +483,85 @@ def enrich_doc_nav_summaries( logger.debug(f"No {DOC_NAV_FILENAME} for {file_name}, skipping") continue + source_file_name = str(doc_nav.get("file_name") or file_name) + self_only_lookup = build_self_only_lookup( + list(chunks or []), + source_file_name=source_file_name, + ) + + existing_top = str(doc_nav.get("top_summary") or "").strip() + if ( + not force + and _doc_nav_has_enriched_summaries(doc_nav) + and existing_top + ): + logger.debug( + f"Summaries already exist in {DOC_NAV_FILENAME} for {file_name}, skipping" + ) + results[file_name] = existing_top + continue + if not force and _doc_nav_has_enriched_summaries(doc_nav): - logger.debug(f"Summaries already exist in {DOC_NAV_FILENAME} for {file_name}, skipping") - results[file_name] = _build_nav_top_summary(doc_nav, use_llm=use_llm) + logger.info( + f"📝 Building missing {DOC_NAV_FILENAME} top_summary for {file_name} " + f"(top_mode={top_mode})" + ) + top_summary = _build_nav_top_summary( + doc_nav, + use_llm=top_summary_use_llm, + self_only_lookup=self_only_lookup, + source_file_name=source_file_name, + ) + results[file_name] = _persist_doc_nav_top_summary( + file_dir=file_dir, + doc_nav=doc_nav, + top_summary=top_summary, + ) continue logger.info( f"📝 Enriching {DOC_NAV_FILENAME} summaries for {file_name} " - f"(mode={mode_label})" + f"(section_mode={section_mode}, top_mode={top_mode})" ) - # Recursively summarize each top-level section for section in doc_nav.get("sections", []): - _recursive_summarize_nav(section, use_llm=use_llm) + _recursive_summarize_nav( + section, + use_llm=use_llm, + self_only_lookup=self_only_lookup, + source_file_name=source_file_name, + ) - _save_doc_nav(file_dir, doc_nav) + top_summary = _build_nav_top_summary( + doc_nav, + use_llm=top_summary_use_llm, + self_only_lookup=self_only_lookup, + source_file_name=source_file_name, + ) + results[file_name] = _persist_doc_nav_top_summary( + file_dir=file_dir, + doc_nav=doc_nav, + top_summary=top_summary, + ) logger.info(f"✅ doc_nav summaries saved for {file_name}") - top_summary = _build_nav_top_summary(doc_nav, use_llm=use_llm) - results[file_name] = top_summary - return results def load_nav_top_summary(file_dir: str, file_name: str = "") -> str: - """Load doc_nav.json and extract the navigation top summary.""" + """Load persisted doc_nav top_summary, with deterministic fallback.""" doc_nav = _load_doc_nav(file_dir) - if doc_nav is not None: - return _build_nav_top_summary(doc_nav, use_llm=False) - return "" + if doc_nav is None: + return "" + existing = str(doc_nav.get("top_summary") or "").strip() + if existing: + return existing + source_file_name = file_name or str(doc_nav.get("file_name") or "") + return _build_nav_top_summary( + doc_nav, + use_llm=False, + source_file_name=source_file_name, + ) def build_section_summary_lookup(file_dir: str) -> Dict[str, str]: @@ -339,16 +570,6 @@ def build_section_summary_lookup(file_dir: str) -> Dict[str, str]: Keys use the DocumentSection.section_path format produced by ``section_path_from_chunk_path`` (strips the filename prefix, joins remaining parts with ``" / "``). - - Traverses the full section tree at all depths. Used by the publication - pipeline to backfill DocumentSection.summary rows. - - Args: - file_dir: Absolute path to the file-level directory - inside the task-scoped parse workspace. - - Returns: - Dict mapping section_path → summary string (empty dict on any error). """ from shared.services.retrieval.search.lexical_text import section_path_from_chunk_path @@ -375,11 +596,8 @@ def _walk(node: Dict[str, Any]) -> None: for section in doc_nav.get("sections", []): _walk(section) - # Populate Root with the document-level top_summary (tree preview). - # This mirrors GraphNode.properties.top_summary and ensures the - # DocumentSection Root row has a summary for data completeness. if "Root" not in lookup: - top_summary = _build_nav_top_summary(doc_nav, use_llm=False) + top_summary = load_nav_top_summary(file_dir, source_file_name) if top_summary: lookup["Root"] = top_summary diff --git a/apps/worker/app/services/document_agent/__init__.py b/apps/worker/app/services/document_agent/__init__.py index 59e0afc99..6baae446c 100644 --- a/apps/worker/app/services/document_agent/__init__.py +++ b/apps/worker/app/services/document_agent/__init__.py @@ -1,7 +1,5 @@ """Page anatomy agent for hierarchy-first PDF profiling.""" -from typing import TYPE_CHECKING - from app.services.document_agent.manifest import ( PageAnatomyMap, PageFeature, @@ -9,21 +7,9 @@ ShardPlan, ) -if TYPE_CHECKING: - from app.services.document_agent.profile_agent import ProfileAgent - - -def __getattr__(name: str): - if name == "ProfileAgent": - from app.services.document_agent.profile_agent import ProfileAgent - - return ProfileAgent - raise AttributeError(name) - __all__ = [ "PageAnatomyMap", "PageFeature", "PageLabel", - "ProfileAgent", "ShardPlan", ] diff --git a/apps/worker/app/services/document_agent/bootstrap/__init__.py b/apps/worker/app/services/document_agent/bootstrap/__init__.py index b8494b99f..2a490a4bd 100644 --- a/apps/worker/app/services/document_agent/bootstrap/__init__.py +++ b/apps/worker/app/services/document_agent/bootstrap/__init__.py @@ -2,6 +2,14 @@ from app.services.document_agent.bootstrap.aggregate_stats import aggregate_doc_stats from app.services.document_agent.bootstrap.classify import classify_page_kinds -from app.services.document_agent.bootstrap.probe import probe_page_features +from app.services.document_agent.bootstrap.probe import ( + probe_page_assets, + probe_page_features, +) -__all__ = ["aggregate_doc_stats", "classify_page_kinds", "probe_page_features"] +__all__ = [ + "aggregate_doc_stats", + "classify_page_kinds", + "probe_page_assets", + "probe_page_features", +] diff --git a/apps/worker/app/services/document_agent/bootstrap/aggregate_stats.py b/apps/worker/app/services/document_agent/bootstrap/aggregate_stats.py index f0739edfe..96a119421 100644 --- a/apps/worker/app/services/document_agent/bootstrap/aggregate_stats.py +++ b/apps/worker/app/services/document_agent/bootstrap/aggregate_stats.py @@ -8,28 +8,22 @@ from app.services.document_agent.manifest import PageFeature, ToolContext, ToolResult +# Coarse sampling extrema are text-only. Low text / density already proxy +# chart-heavy or asset-heavy pages; table/drawing counts are not used to +# nominate extrema pages for VLM coarse classification. PROFILE_METRICS = ( "raw_text_length", "text_density", - "image_coverage", - "table_count", - "drawings_count", ) EXTREMA_ROLES = { "raw_text_length": ("min", "max"), "text_density": ("min", "max"), - "image_coverage": ("max",), - "table_count": ("max",), - "drawings_count": ("max",), } EXTREMA_LABELS = { "raw_text_length": "text_length", "text_density": "text_density", - "image_coverage": "image_heavy", - "table_count": "table_heavy", - "drawings_count": "drawing_heavy", } @@ -97,38 +91,30 @@ def aggregate_doc_stats(ctx: ToolContext, _args: dict[str, Any]) -> ToolResult: deduped_extrema = sorted(set(extrema_pages)) landscape_pages = sum(1 for feature in features if feature.orientation == "landscape") - scan_like_pages = sum( - 1 - for feature in features - if feature.raw_text_length < 50 and feature.image_coverage >= 0.5 - ) - image_heavy_pages = sum( - 1 for feature in features if feature.image_coverage >= 0.35 - ) - table_signal_pages = sum( - 1 - for feature in features - if feature.table_count > 0 or feature.drawings_count >= 25 - ) - doc_shape = { + doc_shape: dict[str, Any] = { "page_count": page_count, "landscape_pages": landscape_pages, "landscape_ratio": round(landscape_pages / page_count, 4) if page_count else 0.0, - "scan_like_pages": scan_like_pages, - "scan_like_ratio": round(scan_like_pages / page_count, 4) - if page_count - else 0.0, - "image_heavy_pages": image_heavy_pages, - "image_heavy_ratio": round(image_heavy_pages / page_count, 4) - if page_count - else 0.0, - "table_signal_pages": table_signal_pages, - "table_signal_ratio": round(table_signal_pages / page_count, 4) - if page_count - else 0.0, } + if ctx.blackboard.global_signals.get("assets_probed"): + scan_like_pages = sum( + 1 + for feature in features + if feature.raw_text_length < 50 and feature.image_coverage >= 0.5 + ) + asset_pages = sum(1 for feature in features if feature.has_asset) + doc_shape.update( + { + "scan_like_pages": scan_like_pages, + "scan_like_ratio": round(scan_like_pages / page_count, 4) + if page_count + else 0.0, + "asset_pages": asset_pages, + "asset_ratio": round(asset_pages / page_count, 4) if page_count else 0.0, + } + ) ctx.blackboard.doc_stats = stats ctx.blackboard.extrema_pages = deduped_extrema ctx.blackboard.global_signals["doc_stats"] = stats diff --git a/apps/worker/app/services/document_agent/bootstrap/probe.py b/apps/worker/app/services/document_agent/bootstrap/probe.py index 14877b1ba..591dfbd94 100644 --- a/apps/worker/app/services/document_agent/bootstrap/probe.py +++ b/apps/worker/app/services/document_agent/bootstrap/probe.py @@ -1,5 +1,8 @@ -"""Bootstrap wrapper for deterministic page probing.""" +"""Bootstrap wrappers for deterministic page probing.""" -from app.services.document_agent.tools.probe_page_features import probe_page_features +from app.services.document_agent.tools.probe_page_features import ( + probe_page_assets, + probe_page_features, +) -__all__ = ["probe_page_features"] +__all__ = ["probe_page_assets", "probe_page_features"] diff --git a/apps/worker/app/services/document_agent/coordinator.py b/apps/worker/app/services/document_agent/coordinator.py index 4630ff43c..2f8c699fb 100644 --- a/apps/worker/app/services/document_agent/coordinator.py +++ b/apps/worker/app/services/document_agent/coordinator.py @@ -10,6 +10,7 @@ from app.services.document_agent.bootstrap import ( aggregate_doc_stats, classify_page_kinds, + probe_page_assets, probe_page_features, ) from app.services.document_agent.budget import BudgetTracker, StageEnvelope @@ -96,13 +97,6 @@ def __init__( self.round_index = 0 self._planner_cache: tuple[DocumentProfile, Any, ToolResult] | None = None - def run(self) -> PageAnatomyMap: - try: - return self._run_structural() - except Exception as exc: - self._record_failure(exc) - raise - def run_coarse(self) -> DocumentProfile: try: return self._run_coarse() @@ -110,9 +104,9 @@ def run_coarse(self) -> DocumentProfile: self._record_failure(exc) raise - def run_structural(self) -> PageAnatomyMap: + def run_structural(self, *, skip_shard_plan: bool = False) -> PageAnatomyMap: try: - return self._run_structural() + return self._run_structural(skip_shard_plan=skip_shard_plan) except Exception as exc: self._record_failure(exc) raise @@ -151,9 +145,10 @@ def _run_coarse(self) -> DocumentProfile: profile, _initial_decision, _planner_result = self._propose_profile( actor="planner:coarse" ) + self._ensure_asset_probe() return profile - def _run_structural(self) -> PageAnatomyMap: + def _run_structural(self, *, skip_shard_plan: bool = False) -> PageAnatomyMap: self.state = DocumentAgentState.RUNNING if not self.blackboard.page_features: self._run_bootstrap() @@ -164,14 +159,22 @@ def _run_structural(self) -> PageAnatomyMap: profile, initial_decision, _planner_result = self._propose_profile( actor="planner" ) - executor_result = ReActExecutor( - self.ctx, - registry=REGISTRY, - max_rounds=int(self.ctx.settings.get("max_rounds", 30)), - initial_decision=initial_decision, - ).run() - if executor_result.verdict.status != "success": - raise RuntimeError(f"profile aborted: {executor_result.verdict.rationale}") + self._ensure_asset_probe() + if skip_shard_plan: + # Page-memory oversized path never consumes shard_plan; only + # build_anatomy_map's invariant needs a non-empty plan. + self._apply_single_shard_placeholder() + else: + executor_result = ReActExecutor( + self.ctx, + registry=REGISTRY, + max_rounds=int(self.ctx.settings.get("max_rounds", 30)), + initial_decision=initial_decision, + ).run() + if executor_result.verdict.status != "success": + raise RuntimeError( + f"profile aborted: {executor_result.verdict.rationale}" + ) anatomy = build_anatomy_map(self.ctx) self._persist_ready_anatomy(anatomy) return anatomy @@ -200,6 +203,7 @@ def _run_lightweight_anatomy( self.state = DocumentAgentState.RUNNING if not self.blackboard.page_features: self._run_bootstrap() + self._ensure_asset_probe() if self.blackboard.toc_result is None: if self._toc_profile_enabled(): self.blackboard.toc_result = TocResult( @@ -214,9 +218,7 @@ def _run_lightweight_anatomy( # it. Populate a single-shard placeholder to skip the LLM shard # decision + H2 refinement (kept global for chunk-track oversized # MinerU sharding). - self.blackboard.shard_plan = single_shard_plan( - self.blackboard.page_count - ) + self._apply_single_shard_placeholder() else: result = REGISTRY.dispatch("propose.shard_plan", self.ctx, {}) self.trace.record_step( @@ -234,6 +236,9 @@ def _run_lightweight_anatomy( self._persist_ready_anatomy(anatomy) return anatomy + def _apply_single_shard_placeholder(self) -> None: + self.blackboard.shard_plan = single_shard_plan(self.blackboard.page_count) + def _persist_ready_anatomy(self, anatomy: PageAnatomyMap) -> None: persist_result = persist_anatomy_map(self.ctx, {}) self.trace.record_step( @@ -284,6 +289,28 @@ def _run_bootstrap(self) -> None: raise RuntimeError(result.error or f"{tool_name} failed") self.round_index += 1 + def _ensure_asset_probe(self) -> None: + if self.blackboard.global_signals.get("assets_probed"): + return + if not self.blackboard.page_features: + raise RuntimeError("page_features missing; run text bootstrap first") + for tool_name, handler in ( + ("probe.page_assets", probe_page_assets), + ("aggregate.doc_stats", aggregate_doc_stats), + ): + result = handler(self.ctx, {}) + self.trace.record_step( + round_index=self.round_index, + actor=f"bootstrap:{tool_name}", + action_type="bootstrap", + result=result, + tool_name=tool_name, + tool_args={}, + ) + if result.status != "ok": + raise RuntimeError(result.error or f"{tool_name} failed") + self.round_index += 1 + def _toc_result_requires_strict_retry(self) -> bool: toc_result = self.blackboard.toc_result return bool( diff --git a/apps/worker/app/services/document_agent/manifest.py b/apps/worker/app/services/document_agent/manifest.py index 8861f631e..d078f9d58 100644 --- a/apps/worker/app/services/document_agent/manifest.py +++ b/apps/worker/app/services/document_agent/manifest.py @@ -7,7 +7,7 @@ from typing import Any, Literal -PageKind = Literal["normal", "table_heavy", "image_heavy", "low_content", "landscape"] +PageKind = Literal["normal", "landscape"] TocFailureKind = Literal["none", "confirm_failed", "rejected_all", "degraded"] ReflexionAction = Literal["tool_call", "verdict_now"] @@ -26,8 +26,10 @@ class PageFeature: orientation: Literal["portrait", "landscape"] width: float height: float + has_asset: bool is_blank_like: bool - text_lines_preview: list[str] = field(default_factory=list) + # PDF-space boxes for detected assets; None when none were extracted. + asset_bboxes: list[dict[str, Any]] | None = None def to_dict(self) -> dict[str, Any]: return asdict(self) @@ -52,6 +54,10 @@ class DocumentProfile: category_rationale: str = "" language: str = "unknown" rationale: str = "" + # Content-band margins as fractions of page height (top origin, y down). + # header_y: lowest header line among sample pages; footer_y: highest footer. + header_y: float | None = None + footer_y: float | None = None def to_dict(self) -> dict[str, Any]: return asdict(self) @@ -229,6 +235,7 @@ class PageAnatomyMap: def to_dict(self) -> dict[str, Any]: return { "version": self.version, + "toc_hierarchies": self.toc_hierarchies, "job_id": self.job_id, "file_path": self.file_path, "page_count": self.page_count, @@ -240,7 +247,6 @@ def to_dict(self) -> dict[str, Any]: "document_profile": self.document_profile.to_dict() if self.document_profile else None, - "toc_hierarchies": self.toc_hierarchies, "toc_page_offset": self.toc_page_offset, "global_signals": dict(self.global_signals), "trace_summary": dict(self.trace_summary), diff --git a/apps/worker/app/services/document_agent/planner/planner.py b/apps/worker/app/services/document_agent/planner/planner.py index 533cc4736..7e5e1ce01 100644 --- a/apps/worker/app/services/document_agent/planner/planner.py +++ b/apps/worker/app/services/document_agent/planner/planner.py @@ -26,18 +26,6 @@ "Page with enough extractable native text and no dominant table/image " "structure." ), - "table_heavy": ( - "Page with detected tables or many vector drawings, often financial " - "tables or dense tabular layout." - ), - "image_heavy": ( - "Page dominated by image coverage with little extractable native text; " - "may be scanned, infographic, photo, or rendered page." - ), - "low_content": ( - "Page with very little extractable text and little visual/table content; " - "may be blank, separator, short heading page, or sparse transition page." - ), "landscape": "Landscape-oriented page, often wide tables, drawings, slides, or diagrams.", } @@ -56,11 +44,9 @@ def _feature_rows(ctx: ToolContext, pages: list[int]) -> list[dict[str, Any]]: "confidence": label.confidence if label else None, "raw_text_length": feature.raw_text_length, "text_density": feature.text_density, - "image_coverage": feature.image_coverage, - "image_count": feature.image_count, - "table_count": feature.table_count, - "drawings_count": feature.drawings_count, "orientation": feature.orientation, + "width": feature.width, + "height": feature.height, "is_blank_like": feature.is_blank_like, } ) @@ -78,6 +64,11 @@ def _segment_sample(candidates: list[int], count: int) -> list[int]: return [candidates[round(index * step)] for index in range(count)] +# Coarse VLM budget: extrema first, then front/mid/back fill, hard cap 10. +_COARSE_SAMPLE_CAP = 10 +_COARSE_SEGMENT_QUOTAS = (2, 2, 2) # front, middle, back + + def _sample_pages( page_count: int, extrema_pages: list[int], @@ -85,9 +76,16 @@ def _sample_pages( ) -> list[int]: """Select representative pages for VLM profiling. + Strategy (cap 10): + 1. Text extrema first (length/density min+max, ≤4 unique). + 2. Fill remaining slots with 2/2/2 stratified samples from + front/middle/back of the non-extrema pool. + 3. Hard truncate to ``_COARSE_SAMPLE_CAP``. + Args: page_count: Total number of pages. - extrema_pages: Pages with statistical extrema (min/max text, tables, etc.). + extrema_pages: Pages with text extrema (min/max raw_text_length / + text_density). Low-text extrema already surface chart/asset pages. exclude_pages: Pages to skip entirely (e.g. TOC pages already detected by the TOC pipeline). These inflate text-density metrics without adding profiling value. @@ -98,21 +96,34 @@ def _sample_pages( extrema = [page for page in extrema_pages if 1 <= page <= page_count and page not in skip] pool = [page for page in range(1, page_count + 1) if page not in set(extrema) and page not in skip] if not pool: - return sorted(set(extrema)) + return sorted(set(extrema))[:_COARSE_SAMPLE_CAP] third = max(len(pool) // 3, 1) front = pool[:third] middle = pool[third : third * 2] back = pool[third * 2 :] + front_n, middle_n, back_n = _COARSE_SEGMENT_QUOTAS sampled = ( - _segment_sample(front, 4) - + _segment_sample(middle or pool, 3) - + _segment_sample(back or pool, 3) + _segment_sample(front, front_n) + + _segment_sample(middle or pool, middle_n) + + _segment_sample(back or pool, back_n) ) ordered = [] for page in extrema + sampled: if page not in ordered: ordered.append(page) - return ordered[:20] + return ordered[:_COARSE_SAMPLE_CAP] + + +def _parse_margin_ratio(value: Any) -> float | None: + if value is None or value == "": + return None + try: + ratio = float(value) + except (TypeError, ValueError): + return None + if 0.0 <= ratio <= 1.0: + return ratio + return None def _parse_profile_and_decision(raw: str) -> tuple[DocumentProfile, ReflexionDecision]: @@ -130,6 +141,11 @@ def _parse_profile_and_decision(raw: str) -> tuple[DocumentProfile, ReflexionDec is_scanned = raw_is_scanned.strip().lower() in {"true", "yes", "1", "scanned"} else: is_scanned = bool(raw_is_scanned) + header_y = _parse_margin_ratio(data.get("header_y")) + footer_y = _parse_margin_ratio(data.get("footer_y")) + if header_y is not None and footer_y is not None and header_y >= footer_y: + header_y = None + footer_y = None profile = DocumentProfile( is_scanned=is_scanned, category=category or "unknown document", @@ -137,6 +153,8 @@ def _parse_profile_and_decision(raw: str) -> tuple[DocumentProfile, ReflexionDec category_rationale=str(data.get("category_rationale") or ""), language=str(data.get("language") or "unknown"), rationale=str(data.get("rationale") or ""), + header_y=header_y, + footer_y=footer_y, ) next_action = str(data.get("next_action") or "ready_to_shard") tool_name: str | None = None diff --git a/apps/worker/app/services/document_agent/planner/prompts.py b/apps/worker/app/services/document_agent/planner/prompts.py index 7bb1f3166..4ca93ef8e 100644 --- a/apps/worker/app/services/document_agent/planner/prompts.py +++ b/apps/worker/app/services/document_agent/planner/prompts.py @@ -1,20 +1,26 @@ """Prompts for the document profile planner.""" PLANNER_INSTRUCTIONS = ( - "You are a document profile agent. Use global page-feature statistics, " - "optional TOC/H1 evidence, and page screenshots to classify the PDF. Return strict " - "JSON only with keys: is_scanned, category, routing_category, " - "category_rationale, language, rationale, next_action, inspect_pages, grep_query. " - "category is a concise semantic document type, at most 5 English words, such " - "as Financial Prospectus, Technical Manual, Corporate Policy, Research Report, " - "Engineering Atlas, or Scanned Handbook. routing_category must be one of " - "atlas, scanned, slides, generic. Set routing_category=atlas only for " - "engineering drawing collections, construction standard atlases, or page sets " - "whose primary unit is a drawing/detail sheet rather than prose. next_action must be one of inspect_more, grep_text, " - "ready_to_shard, verdict_now. Use inspect_more only when specific extra " - "page screenshots are needed. Use grep_text only for native PDFs when a " - "global text search would clarify structure. Do not output a fixed step " - "plan." + "You are a document profile agent. Use page-feature statistics, optional " + "TOC/H1 evidence, and the provided page screenshots to classify the PDF. " + "Return strict JSON only with keys: is_scanned, category, routing_category, " + "category_rationale, language, rationale, header_y, footer_y, next_action, " + "inspect_pages, grep_query. " + "category is a concise semantic document type in at most 5 English words. " + "routing_category must be one of atlas, scanned, slides, generic. " + "Set routing_category=atlas only when pages are primarily drawing/detail " + "sheets rather than prose. " + "header_y and footer_y are document-level horizontal content-margin lines " + "as fractions of page height in [0, 1], origin at the top with y increasing " + "downward. From the sample pages shown: header_y is the lowest header line " + "you observe (largest y) when any header is present, otherwise null; " + "footer_y is the highest footer line you observe (smallest y) when any " + "footer is present, otherwise null. When both are set, require " + "header_y < footer_y. " + "next_action must be one of inspect_more, grep_text, ready_to_shard, " + "verdict_now. Use inspect_more only when extra page screenshots are needed. " + "Use grep_text only for native PDFs when a global text search would clarify " + "structure. Do not output a fixed step plan." ) __all__ = ["PLANNER_INSTRUCTIONS"] diff --git a/apps/worker/app/services/document_agent/profile_agent.py b/apps/worker/app/services/document_agent/profile_agent.py deleted file mode 100644 index c56e18c16..000000000 --- a/apps/worker/app/services/document_agent/profile_agent.py +++ /dev/null @@ -1,60 +0,0 @@ -"""Public entrypoint for document page anatomy profiling.""" - -from __future__ import annotations - -import os -from typing import Any - -from app.services.document_agent.coordinator import ProfileCoordinator -from app.services.document_agent.manifest import DocumentProfile, PageAnatomyMap - - -class ProfileAgent: - def __init__( - self, - *, - model: str | None = None, - settings: dict[str, Any] | None = None, - ) -> None: - self._model = model - self._settings = settings or {} - - def run( - self, - file_path: str, - job_id: str, - *, - output_dir: str | None = None, - db: Any | None = None, - ) -> PageAnatomyMap: - if not os.path.exists(file_path): - raise FileNotFoundError(file_path) - coordinator = ProfileCoordinator( - pdf_path=file_path, - job_id=job_id, - output_dir=output_dir, - db=db, - model=self._model, - settings=self._settings, - ) - return coordinator.run() - - def run_coarse( - self, - file_path: str, - job_id: str, - *, - output_dir: str | None = None, - db: Any | None = None, - ) -> DocumentProfile: - if not os.path.exists(file_path): - raise FileNotFoundError(file_path) - coordinator = ProfileCoordinator( - pdf_path=file_path, - job_id=job_id, - output_dir=output_dir, - db=db, - model=self._model, - settings=self._settings, - ) - return coordinator.run_coarse() diff --git a/apps/worker/app/services/document_agent/structure/hierarchy_locator.py b/apps/worker/app/services/document_agent/structure/hierarchy_locator.py index 28697c456..5815a4120 100644 --- a/apps/worker/app/services/document_agent/structure/hierarchy_locator.py +++ b/apps/worker/app/services/document_agent/structure/hierarchy_locator.py @@ -1,9 +1,9 @@ """Locate hierarchy titles on PDF pages and resolve page ranges. -This module is intentionally deterministic: it performs strict title anchoring, -candidate collection, and range assembly. The page-memory residual agent calls -into these primitives for grep-like tools and adds VLM verification outside this -module. +Deterministic title anchoring and range assembly. Leaf starts come from +offset-guided ``match_overrides`` or per-line strict exact. Null-page parents +are located upstream via compact-strict (cross-line) + optional VLM, then +resolved here including parent self-only spans for interstitial pages. """ from __future__ import annotations @@ -19,10 +19,6 @@ TitleMatchSource = Literal[ "anchored", - "page_compact", - "normalized", - "token", - "printed_prior", "h1_result", "agent_vlm", "agent_heuristic", @@ -80,47 +76,6 @@ class _LineHit: score: float -_STOPWORDS = { - "a", - "an", - "and", - "are", - "as", - "for", - "in", - "of", - "on", - "or", - "the", - "to", - "with", -} - - -def locate_title_start_page( - title: str, - *, - scope_pages: list[int], - page_texts: dict[int, str], - printed_page: int | None = None, - page_offset_hint: int | None = None, -) -> TitleMatch | None: - """Locate *title* using deterministic weak evidence. - - This is a candidate-gathering primitive. The C4 page-memory path should only - directly accept :func:`locate_title_strict_exact`; weak results from this - function are meant for residual agent/VLM arbitration. - """ - matches = collect_title_candidate_matches( - title, - scope_pages=scope_pages, - page_texts=page_texts, - printed_page=printed_page, - page_offset_hint=page_offset_hint, - ) - return matches[0] if matches else None - - def locate_title_strict_exact( title: str, *, @@ -135,84 +90,82 @@ def locate_title_strict_exact( return _choose_best_hit( hits, source="anchored", - printed_page=None, - page_offset_hint=None, extra_evidence={"accept": "strict_exact_unique"}, ) -def collect_title_candidate_matches( +def locate_title_compact_strict( title: str, *, scope_pages: list[int], page_texts: dict[int, str], - printed_page: int | None = None, - page_offset_hint: int | None = None, - limit: int | None = None, -) -> list[TitleMatch]: - """Collect grep-style candidate pages for a title without final arbitration.""" - normalized_title = normalize_heading_text(title) - if not normalized_title or not scope_pages: - return [] +) -> TitleMatch | None: + """Locate *title* after cross-line compact cleanup; accept only a unique page. - hits: list[_LineHit] = [] - for _source, finder in ( - ("anchored", _find_anchored_hits), - ("page_compact", _find_page_compact_hits), - ("normalized", _find_normalized_hits), - ("token", _find_token_hits), - ): - hits.extend(finder(normalized_title, scope_pages, page_texts)) - - by_page: dict[int, list[_LineHit]] = {} - for hit in hits: - by_page.setdefault(hit.page, []).append(hit) - - matches = [ - _choose_best_hit( - page_hits, - source=_preferred_source(page_hits), - printed_page=printed_page, - page_offset_hint=page_offset_hint, - ) - for page_hits in by_page.values() - ] + Pipeline: compact(page text) → contiguous strict match of compact(title) → + accept iff exactly one page in ``scope_pages`` hits. Handles PyMuPDF line + splits; does not use token/normalized weak matching. + """ + needle = _compact_match_text(clean_toc_title(title) or title) + if not needle or not scope_pages: + return None - prior_page = _resolve_printed_prior( - printed_page=printed_page, - page_offset_hint=page_offset_hint, - scope_pages=scope_pages, - ) - if prior_page is not None and prior_page not in by_page: - matches.append( - TitleMatch( - page=prior_page, - confidence=0.35, - source="printed_prior", - matched_line="", - score=0.35, - candidates=[prior_page], - evidence={ - "printed_page": printed_page, - "page_offset_hint": page_offset_hint, - }, - ) - ) + hit_pages: list[int] = [] + matched_preview = "" + for page in scope_pages: + haystack = _compact_match_text(page_texts.get(page, "")) + if not haystack or needle not in haystack: + continue + hit_pages.append(page) + if not matched_preview: + matched_preview = needle[:160] - matches.sort( - key=lambda match: ( - match.score, - match.confidence, - -abs(match.page - (printed_page + page_offset_hint)) - if printed_page is not None and page_offset_hint is not None - else 0, - -match.page, - ), - reverse=True, + unique_pages = sorted(set(hit_pages)) + if len(unique_pages) != 1: + return None + + page = unique_pages[0] + return TitleMatch( + page=page, + confidence=0.92, + source="anchored", + matched_line=matched_preview, + score=0.96, + candidates=[page], + evidence={"accept": "compact_strict_unique"}, ) - if limit is not None: - return matches[: max(int(limit), 0)] - return matches + + +def last_leaf_start_under( + node: TitleNode, + parent_titles: tuple[str, ...], + match_overrides: dict[tuple[str, ...], TitleMatch], +) -> int | None: + """Max start page among located leaves under *node*; None if none located.""" + max_page: int | None = None + for leaf_path, _leaf in iter_leaf_title_nodes([node], parent_titles=parent_titles): + match = match_overrides.get(leaf_path) + if match is None: + continue + if max_page is None or match.page > max_page: + max_page = match.page + return max_page + + +def first_leaf_start_under( + node: TitleNode, + parent_titles: tuple[str, ...], + match_overrides: dict[tuple[str, ...], TitleMatch], +) -> int | None: + """Min start page among located leaves under *node*; None if none located.""" + min_page: int | None = None + for leaf_path, _leaf in iter_leaf_title_nodes([node], parent_titles=parent_titles): + match = match_overrides.get(leaf_path) + if match is None: + continue + if min_page is None or match.page < min_page: + min_page = match.page + return min_page def resolve_hierarchy_page_ranges( @@ -221,15 +174,12 @@ def resolve_hierarchy_page_ranges( page_count: int, page_texts: dict[int, str], body_pages: list[int] | None = None, - page_offset_hint: int | None = None, match_overrides: dict[tuple[str, ...], TitleMatch] | None = None, - use_weak_fallback: bool = False, ) -> list[ResolvedHierarchyRange]: - """Resolve leaf hierarchy nodes into closed page ranges. + """Resolve hierarchy nodes into closed page ranges. - The emitted ranges are leaf-first and intentionally closed-closed: if the - next leaf starts on page N, the previous leaf may also include page N. This - preserves page-to-section many-to-many mapping for dense documents. + Emits leaf ranges and parent self-only spans when a parent start is strictly + before its first located descendant leaf. Ranges are closed-closed. """ if page_count <= 0 or not nodes: return [] @@ -248,9 +198,7 @@ def resolve_hierarchy_page_ranges( allowed_pages=allowed_pages, parent_titles=(), page_texts=page_texts, - page_offset_hint=page_offset_hint, match_overrides=match_overrides or {}, - use_weak_fallback=use_weak_fallback, resolved=resolved, ) return resolved @@ -274,9 +222,7 @@ def _resolve_siblings( allowed_pages: set[int], parent_titles: tuple[str, ...], page_texts: dict[int, str], - page_offset_hint: int | None, match_overrides: dict[tuple[str, ...], TitleMatch], - use_weak_fallback: bool, resolved: list[ResolvedHierarchyRange], ) -> None: located: list[tuple[TitleNode, int, TitleMatch | None]] = [] @@ -285,27 +231,13 @@ def _resolve_siblings( for index, node in enumerate(nodes): path_titles = (*parent_titles, node.title) pages = _allowed_pages_between(lower_bound, parent_scope.end, allowed_pages) - match = _match_override(path_titles, match_overrides, pages) - if match is None: - match = _match_physical_hint(node=node, scope_pages=pages) - if match is None: - match = locate_title_strict_exact( - node.title, - scope_pages=pages, - page_texts=page_texts, - ) - if match is None and use_weak_fallback: - match = locate_title_start_page( - node.title, - scope_pages=pages, - page_texts=page_texts, - printed_page=node.printed_page, - page_offset_hint=page_offset_hint, - ) - if match is None and node.children and match_overrides: - match = _infer_start_from_descendant_overrides( - node, parent_titles, match_overrides, pages, - ) + match = _locate_match_for_node( + node, + path_titles=path_titles, + scope_pages=pages, + page_texts=page_texts, + match_overrides=match_overrides, + ) if match is None: start_page = lower_bound else: @@ -321,9 +253,7 @@ def _resolve_siblings( parent_end=parent_scope.end, allowed_pages=allowed_pages, page_texts=page_texts, - page_offset_hint=page_offset_hint, match_overrides=match_overrides, - use_weak_fallback=use_weak_fallback, parent_titles=parent_titles, ) if next_match is not None: @@ -349,15 +279,32 @@ def _resolve_siblings( ) if node.children: + first_child_start = first_leaf_start_under( + node, parent_titles, match_overrides + ) + if ( + match is not None + and first_child_start is not None + and start_page < first_child_start + ): + resolved.append( + ResolvedHierarchyRange( + title=node.title, + level=node.level, + start_page=start_page, + end_page=first_child_start, + path_titles=path_titles, + match=match, + evidence={**evidence, "skeleton_kind": "parent_self_only"}, + ) + ) _resolve_siblings( node.children, parent_scope=PageRange(start_page, end_page), allowed_pages=allowed_pages, parent_titles=path_titles, page_texts=page_texts, - page_offset_hint=page_offset_hint, match_overrides=match_overrides, - use_weak_fallback=use_weak_fallback, resolved=resolved, ) continue @@ -375,6 +322,34 @@ def _resolve_siblings( ) +def _locate_match_for_node( + node: TitleNode, + *, + path_titles: tuple[str, ...], + scope_pages: list[int], + page_texts: dict[int, str], + match_overrides: dict[tuple[str, ...], TitleMatch], +) -> TitleMatch | None: + match = _match_override(path_titles, match_overrides, scope_pages) + if match is not None: + return match + match = _match_physical_hint(node=node, scope_pages=scope_pages) + if match is not None: + return match + if node.children: + # Parent active locate is upstream (compact-strict / visual). Wide-window + # strict_exact is intentionally not used here. + return _infer_start_from_descendant_overrides( + node, parent_titles=path_titles[:-1], match_overrides=match_overrides, + scope_pages=scope_pages, + ) + return locate_title_strict_exact( + node.title, + scope_pages=scope_pages, + page_texts=page_texts, + ) + + def _find_next_located_sibling( *, nodes: list[TitleNode], @@ -383,36 +358,19 @@ def _find_next_located_sibling( parent_end: int, allowed_pages: set[int], page_texts: dict[int, str], - page_offset_hint: int | None, match_overrides: dict[tuple[str, ...], TitleMatch], - use_weak_fallback: bool, parent_titles: tuple[str, ...], ) -> TitleMatch | None: pages = _allowed_pages_between(lower_bound, parent_end, allowed_pages) for sibling in nodes[start_index:]: path_titles = (*parent_titles, sibling.title) - match = _match_override(path_titles, match_overrides, pages) - if match is None: - match = _match_physical_hint(node=sibling, scope_pages=pages) - if match is not None: - return match - match = locate_title_strict_exact( - sibling.title, + match = _locate_match_for_node( + sibling, + path_titles=path_titles, scope_pages=pages, page_texts=page_texts, + match_overrides=match_overrides, ) - if match is None and use_weak_fallback: - match = locate_title_start_page( - sibling.title, - scope_pages=pages, - page_texts=page_texts, - printed_page=sibling.printed_page, - page_offset_hint=page_offset_hint, - ) - if match is None and sibling.children and match_overrides: - match = _infer_start_from_descendant_overrides( - sibling, parent_titles, match_overrides, pages, - ) if match is not None: return match return None @@ -424,13 +382,7 @@ def _infer_start_from_descendant_overrides( match_overrides: dict[tuple[str, ...], TitleMatch], scope_pages: list[int], ) -> TitleMatch | None: - """Infer a parent node's start page from its earliest located descendant leaf. - - When a non-leaf node cannot be directly located (no printed_page, no grep - match), its descendant leaves may already be in match_overrides from - offset-guided bulk anchoring. Use the minimum page among those descendants - as a synthetic match so the resolver can cap the previous sibling's end_page. - """ + """Final fallback: parent start = earliest located descendant leaf page.""" if not node.children or not match_overrides: return None leaves = iter_leaf_title_nodes([node], parent_titles=parent_titles) @@ -457,6 +409,7 @@ def _infer_start_from_descendant_overrides( evidence={ "inferred_from": "descendant_leaf_override", "original_confidence": min_match.confidence, + "status": "degraded", }, ) @@ -487,50 +440,6 @@ def iter_leaf_title_nodes( return leaves -def max_title_depth(nodes: list[TitleNode]) -> int: - if not nodes: - return 0 - return max( - max(node.level, max_title_depth(node.children)) - if node.children - else node.level - for node in nodes - ) - - -def prune_title_nodes_for_emit_depth( - nodes: list[TitleNode], - *, - emit_depth: int, -) -> list[TitleNode]: - pruned: list[TitleNode] = [] - for node in nodes: - if node.level >= emit_depth or not node.children: - pruned.append( - TitleNode( - title=node.title, - level=node.level, - printed_page=node.printed_page, - physical_page_hint=node.physical_page_hint, - children=[], - ) - ) - continue - pruned.append( - TitleNode( - title=node.title, - level=node.level, - printed_page=node.printed_page, - physical_page_hint=node.physical_page_hint, - children=prune_title_nodes_for_emit_depth( - node.children, - emit_depth=emit_depth, - ), - ) - ) - return pruned - - def _next_located_start( located: list[tuple[TitleNode, int, TitleMatch | None]], start_index: int, @@ -604,6 +513,10 @@ def _allowed_pages_between(start: int, end: int, allowed_pages: set[int]) -> lis return [page for page in range(start, end + 1) if page in allowed_pages] +def _compact_match_text(text: str) -> str: + return re.sub(r"\s+", "", normalize_heading_text(text)).casefold() + + def _find_anchored_hits( title: str, scope_pages: list[int], @@ -628,123 +541,24 @@ def _find_anchored_hits( return hits -def _find_page_compact_hits( - title: str, - scope_pages: list[int], - page_texts: dict[int, str], -) -> list[_LineHit]: - hits: list[_LineHit] = [] - needles = _compact_title_variants(title) - if not needles: - return hits - - for page in scope_pages: - raw_text = page_texts.get(page, "") - compact_text = _compact_match_text(raw_text) - if not compact_text: - continue - matched_needle = next((needle for needle in needles if needle in compact_text), None) - if matched_needle is None: - continue - line_index, evidence = _compact_match_evidence(raw_text, matched_needle) - hits.append( - _LineHit( - page=page, - line_index=line_index, - line=evidence, - source="page_compact", - score=_line_score(line=evidence, line_index=line_index, base=0.94), - ) - ) - return hits - - -def _find_normalized_hits( - title: str, - scope_pages: list[int], - page_texts: dict[int, str], -) -> list[_LineHit]: - hits: list[_LineHit] = [] - needle = normalize_heading_text(clean_toc_title(title)).casefold() - if len(needle) < 2: - return hits - for page, line_index, line in _iter_lines(scope_pages, page_texts): - cleaned_line = normalize_heading_text(clean_toc_title(line)).casefold() - if not cleaned_line: - continue - if needle in cleaned_line or _is_strong_reverse_match(cleaned_line, needle): - hits.append( - _LineHit( - page=page, - line_index=line_index, - line=line.strip(), - source="normalized", - score=_line_score(line=line, line_index=line_index, base=0.9), - ) - ) - return hits - - -def _find_token_hits( - title: str, - scope_pages: list[int], - page_texts: dict[int, str], -) -> list[_LineHit]: - title_tokens = _significant_tokens(clean_toc_title(title) or title) - if not title_tokens: - return [] - - hits: list[_LineHit] = [] - for page, line_index, line in _iter_lines(scope_pages, page_texts): - line_tokens = _significant_tokens(line) - if not line_tokens: - continue - coverage = len(title_tokens & line_tokens) / len(title_tokens) - if coverage < 0.8: - continue - hits.append( - _LineHit( - page=page, - line_index=line_index, - line=line.strip(), - source="token", - score=_line_score(line=line, line_index=line_index, base=0.78) - + coverage, - ) - ) - return hits - - def _choose_best_hit( hits: list[_LineHit], *, source: TitleMatchSource, - printed_page: int | None, - page_offset_hint: int | None, extra_evidence: dict[str, Any] | None = None, ) -> TitleMatch: - expected_page = ( - printed_page + page_offset_hint - if printed_page is not None and page_offset_hint is not None - else None + ordered = sorted( + hits, + key=lambda hit: (hit.score, -hit.line_index, -hit.page), + reverse=True, ) - - def sort_key(hit: _LineHit) -> tuple[float, int, int, int]: - printed_bonus = 0 - if expected_page is not None: - printed_bonus = -abs(hit.page - expected_page) - return (hit.score, printed_bonus, -hit.line_index, -hit.page) - - ordered = sorted(hits, key=sort_key, reverse=True) best = ordered[0] pages = sorted({hit.page for hit in ordered}) confidence_by_source = { "anchored": 0.92, - "page_compact": 0.9, - "normalized": 0.84, - "token": 0.72, - "printed_prior": 0.35, "h1_result": 0.88, + "agent_vlm": 0.75, + "agent_heuristic": 0.5, } return TitleMatch( page=best.page, @@ -756,41 +570,11 @@ def sort_key(hit: _LineHit) -> tuple[float, int, int, int]: evidence={ "line_index": best.line_index, "candidate_count": len(pages), - "printed_page": printed_page, - "page_offset_hint": page_offset_hint, **(extra_evidence or {}), }, ) -def _preferred_source(hits: list[_LineHit]) -> TitleMatchSource: - priority = { - "anchored": 50, - "page_compact": 40, - "normalized": 30, - "token": 20, - "printed_prior": 10, - "h1_result": 60, - "agent_vlm": 70, - "agent_heuristic": 15, - } - return max(hits, key=lambda hit: (priority.get(hit.source, 0), hit.score)).source - - -def _resolve_printed_prior( - *, - printed_page: int | None, - page_offset_hint: int | None, - scope_pages: list[int], -) -> int | None: - if printed_page is None or page_offset_hint is None or not scope_pages: - return None - page = printed_page + page_offset_hint - if page in scope_pages: - return page - return None - - def _line_score(*, line: str, line_index: int, base: float) -> float: stripped = normalize_heading_text(line) short_line_bonus = max(0.0, 1.0 - (len(stripped) / 140.0)) @@ -810,55 +594,6 @@ def _iter_lines( return rows -def _compact_title_variants(title: str) -> list[str]: - normalized = normalize_heading_text(clean_toc_title(title) or title).casefold() - compacted: list[str] = [] - compact = _compact_match_text(normalized) - if compact: - compacted.append(compact) - return compacted - - -def _compact_match_text(text: str) -> str: - return re.sub(r"\s+", "", normalize_heading_text(text)).casefold() - - -def _compact_match_evidence(raw_text: str, compact_needle: str) -> tuple[int, str]: - lines = [line.strip() for line in raw_text.splitlines() if line.strip()] - if not lines: - return 0, "" - - compact_so_far = "" - start_index = 0 - for index, line in enumerate(lines): - line_compact = _compact_match_text(line) - if not compact_so_far: - start_index = index - compact_so_far += line_compact - if compact_needle in compact_so_far: - return start_index, " ".join(lines[start_index : index + 1])[:160] - if len(compact_so_far) > len(compact_needle) * 3: - compact_so_far = line_compact - start_index = index - - return 0, " ".join(lines[:3])[:160] - - -def _is_strong_reverse_match(fragment: str, title: str) -> bool: - return len(fragment) >= 6 and fragment in title - - -def _significant_tokens(text: str) -> set[str]: - normalized = normalize_heading_text(clean_toc_title(text) or text).casefold() - latin = { - token - for token in re.findall(r"[a-z0-9][a-z0-9_-]+", normalized) - if token not in _STOPWORDS - } - cjk = set(re.findall(r"[\u4e00-\u9fff]", normalized)) - return latin | cjk - - def _extract_flat_entries(payload: Any) -> list[dict[str, Any]]: if isinstance(payload, list): return [ diff --git a/apps/worker/app/services/document_agent/structure/page_locate_agent.py b/apps/worker/app/services/document_agent/structure/page_locate_agent.py index 7a68f3625..f6aa312d4 100644 --- a/apps/worker/app/services/document_agent/structure/page_locate_agent.py +++ b/apps/worker/app/services/document_agent/structure/page_locate_agent.py @@ -1,4 +1,10 @@ -"""Residual page-location agent for hierarchy titles.""" +"""VLM page verification for page-memory offset calibration. + +This module provides the deterministic-input, VLM-arbitrated helper used by the +page-memory skeleton calibration to confirm which candidate page starts a given +section title. The former residual ReAct sub-agent has been removed; calibration +now drives offset-guided bulk anchoring directly and only needs this verifier. +""" from __future__ import annotations @@ -6,26 +12,12 @@ import json import os import time -from dataclasses import dataclass, field from typing import Any, cast from loguru import logger from app.services.document_agent.manifest import ToolContext -from app.services.document_agent.structure.hierarchy_locator import ( - TitleMatch, - TitleNode, - collect_title_candidate_matches, - iter_leaf_title_nodes, - locate_title_strict_exact, - max_title_depth, - prune_title_nodes_for_emit_depth, -) -from app.services.document_agent.structure.page_locate_subagent import ( - PageLocateSubAgent, - SubAgentConfig, -) -from shared.models.schemas.page_memory_config import PageMemoryConfig +from app.services.document_agent.structure.hierarchy_locator import TitleMatch VLM_CONFIRMED_DEFAULT_CONFIDENCE = 0.75 GREP_ONLY_CONFIDENCE_CAP = 0.62 @@ -34,224 +26,6 @@ VLM_FAILED_GREP_CONFIDENCE_CAP = 0.54 -@dataclass(frozen=True) -class PageLocateConfig: - residual_agent_limit: int = 50 - max_emit_depth: int = 5 - min_emit_depth: int = 2 - vlm_candidate_page_cap: int = 4 - full_leaf_sections: bool = False - - @classmethod - def from_page_memory_config( - cls, - page_memory_config: PageMemoryConfig, - ) -> "PageLocateConfig": - return cls( - residual_agent_limit=page_memory_config.page_locate_residual_agent_limit, - max_emit_depth=page_memory_config.page_locate_max_emit_depth, - min_emit_depth=page_memory_config.page_locate_min_emit_depth, - vlm_candidate_page_cap=( - page_memory_config.page_locate_vlm_candidate_page_cap - ), - full_leaf_sections=page_memory_config.page_locate_full_leaf_sections, - ) - - -@dataclass(frozen=True) -class ResidualRequest: - path_titles: tuple[str, ...] - node: TitleNode - - -@dataclass(frozen=True) -class PageLocatePrepareResult: - nodes: list[TitleNode] - match_overrides: dict[tuple[str, ...], TitleMatch] - summary: dict[str, Any] = field(default_factory=dict) - - -class PageLocateResidualAgent: - """Batch residual resolver for C4 title-to-page anchoring.""" - - def __init__( - self, - *, - ctx: ToolContext | None, - page_texts: dict[int, str], - body_pages: list[int], - page_count: int, - page_offset_hint: int | None, - config: PageLocateConfig | None = None, - ) -> None: - self.ctx = ctx - self.page_texts = page_texts - self.body_pages = sorted({page for page in body_pages if 1 <= page <= page_count}) - self.page_count = page_count - self.page_offset_hint = page_offset_hint - self.config = config or PageLocateConfig() - - def prepare(self, nodes: list[TitleNode]) -> PageLocatePrepareResult: - if not nodes: - return PageLocatePrepareResult(nodes=[], match_overrides={}, summary={}) - - actual_max_depth = max_title_depth(nodes) - emit_depth = min(actual_max_depth, self.config.max_emit_depth) - emit_depth = max(emit_depth, self.config.min_emit_depth) - direct_matches: dict[tuple[str, ...], TitleMatch] = {} - residuals: list[ResidualRequest] = [] - - while True: - selected_nodes = prune_title_nodes_for_emit_depth(nodes, emit_depth=emit_depth) - direct_matches, residuals = self._classify_residuals(selected_nodes) - if ( - self.config.full_leaf_sections - or len(residuals) <= self.config.residual_agent_limit - or emit_depth <= self.config.min_emit_depth - ): - break - emit_depth -= 1 - - residual_matches = self._resolve_residuals(residuals) - match_overrides = {**direct_matches, **residual_matches} - summary = { - "agent": "page_locate_residual", - "emit_depth": emit_depth, - "actual_max_depth": actual_max_depth, - "direct_exact_count": len(direct_matches), - "residual_count": len(residuals), - "resolved_residual_count": len(residual_matches), - "residual_limit": self.config.residual_agent_limit, - "vlm_candidate_page_cap": self.config.vlm_candidate_page_cap, - "full_leaf_sections": self.config.full_leaf_sections, - } - logger.info("[page_locate.agent] summary={}", summary) - return PageLocatePrepareResult( - nodes=selected_nodes, - match_overrides=match_overrides, - summary=summary, - ) - - def _entry_scope(self, node: TitleNode) -> list[int]: - """Narrow search scope for a node using its printed page + offset hint. - - When a node carries a printed_page and we have a page_offset_hint, - restrict grep to a small window around the expected physical page. - The window spans [printed_page, printed_page + offset] (inclusive, - order-independent) to cover the uncertainty between printed numbering - and physical page positions. - """ - if node.printed_page is None or self.page_offset_hint is None: - return self.body_pages - expected = node.printed_page + self.page_offset_hint - lo = min(node.printed_page, expected) - hi = max(node.printed_page, expected) - return [p for p in self.body_pages if lo <= p <= hi] - - def _classify_residuals( - self, - nodes: list[TitleNode], - ) -> tuple[dict[tuple[str, ...], TitleMatch], list[ResidualRequest]]: - direct: dict[tuple[str, ...], TitleMatch] = {} - residuals: list[ResidualRequest] = [] - for path_titles, node in iter_leaf_title_nodes(nodes): - scope = self._entry_scope(node) - match = locate_title_strict_exact( - node.title, - scope_pages=scope, - page_texts=self.page_texts, - ) - if match is not None: - direct[path_titles] = TitleMatch( - page=match.page, - confidence=match.confidence, - source=match.source, - matched_line=match.matched_line, - score=match.score, - candidates=match.candidates, - evidence={ - **match.evidence, - "page_locate_agent": { - "decision": "direct_strict_exact", - "path_titles": list(path_titles), - }, - }, - ) - else: - residuals.append(ResidualRequest(path_titles=path_titles, node=node)) - return direct, residuals - - def _resolve_residuals( - self, - residuals: list[ResidualRequest], - ) -> dict[tuple[str, ...], TitleMatch]: - matches: dict[tuple[str, ...], TitleMatch] = {} - if not residuals or self.ctx is None: - # No agent runtime (ctx/budget) → leave residuals for physical-hint / - # neighbor-boundary fallback in the resolver. The residual sub-agent - # only runs when a real ToolContext is available. - return matches - - sub_config = SubAgentConfig( - candidate_cap=max(self.config.vlm_candidate_page_cap * 2, 4), - verify_page_cap=max(self.config.vlm_candidate_page_cap, 1), - ) - for residual in residuals: - scope = self._entry_scope(residual.node) - agent = PageLocateSubAgent( - ctx=self.ctx, - scope_pages=scope, - page_count=self.page_count, - config=sub_config, - page_offset_hint=self.page_offset_hint, - ) - result = agent.locate( - title=residual.node.title, - printed_page=residual.node.printed_page, - ) - if result.match is None: - logger.warning( - "[page_locate.agent] unresolved title={!r} path_titles={} stop={}", - residual.node.title, - residual.path_titles, - result.stop_reason, - ) - continue - base = result.match - agent_evidence = dict(base.evidence.get("page_locate_agent", {})) - agent_evidence["path_titles"] = list(residual.path_titles) - agent_evidence["stop_reason"] = result.stop_reason - matches[residual.path_titles] = TitleMatch( - page=base.page, - confidence=base.confidence, - source=base.source, - matched_line=base.matched_line, - score=base.score, - candidates=base.candidates, - evidence={**base.evidence, "page_locate_agent": agent_evidence}, - ) - return matches - - -def grep_title_page_candidates( - *, - title: str, - scope_pages: list[int], - page_texts: dict[int, str], - printed_page: int | None = None, - page_offset_hint: int | None = None, - limit: int | None = None, -) -> list[TitleMatch]: - return collect_title_candidate_matches( - title, - scope_pages=scope_pages, - page_texts=page_texts, - printed_page=printed_page, - page_offset_hint=page_offset_hint, - limit=limit, - ) - - def verify_section_page_choice( *, ctx: ToolContext | None, diff --git a/apps/worker/app/services/document_agent/structure/page_locate_subagent.py b/apps/worker/app/services/document_agent/structure/page_locate_subagent.py deleted file mode 100644 index c3826cb2c..000000000 --- a/apps/worker/app/services/document_agent/structure/page_locate_subagent.py +++ /dev/null @@ -1,446 +0,0 @@ -"""Per-title ReAct sub-agent for residual hierarchy page location. - -This is intentionally NOT a fixed pipeline. For every title that strict anchoring -could not resolve, an LLM runs a bounded reasoning loop and decides — round by -round — which tool to call: - -* ``grep.title_pages`` — find candidate body pages for a (possibly rewritten) - query. The agent may shorten the title to a distinctive core when the full - title carries trailing document-reference codes / ``《》`` wrappers / version - notes that never appear contiguously in the body, or when the heading is split - across lines. -* ``verify.section_page`` — render candidate pages and ask a VLM which one truly - *starts* the section (vs a TOC row, header/footer, or a body citation). - -The agent then ``submit``s a confirmed start page (or gives up). Tools are -dispatched through the shared registry, so the agent reuses the exact same -grep/VLM primitives the rest of the profile agent uses. -""" - -from __future__ import annotations - -import json -from dataclasses import dataclass, field -from typing import Any, Callable, cast - -from loguru import logger - -from app.services.document_agent.manifest import ToolContext -from app.services.document_agent.registry import REGISTRY -from app.services.document_agent.structure.hierarchy_locator import TitleMatch - -DecideFn = Callable[..., dict[str, Any] | None] - -_ALLOWED_ACTIONS = {"grep", "verify", "submit", "give_up"} -VLM_MATCH_DEFAULT_CONFIDENCE = 0.7 -HEURISTIC_MATCH_DEFAULT_CONFIDENCE = 0.55 - - -@dataclass(frozen=True) -class SubAgentConfig: - max_rounds: int = 6 - candidate_cap: int = 8 - verify_page_cap: int = 4 - decide_max_tokens: int = 500 - - -@dataclass -class SubAgentResult: - match: TitleMatch | None - transcript: list[dict[str, Any]] = field(default_factory=list) - rounds: int = 0 - stop_reason: str = "exhausted" - - -class PageLocateSubAgent: - """Bounded ReAct loop that locates ONE residual title via grep + VLM tools.""" - - def __init__( - self, - *, - ctx: ToolContext | None, - scope_pages: list[int], - page_count: int, - config: SubAgentConfig | None = None, - page_offset_hint: int | None = None, - decide: DecideFn | None = None, - ) -> None: - self.ctx = ctx - self.page_count = page_count - self.scope_pages = sorted({p for p in scope_pages if 1 <= p <= page_count}) - self.config = config or SubAgentConfig() - self.page_offset_hint = page_offset_hint - self._decide = decide or decide_next_action - # Ensure the page-locate tools are registered. Importing the light - # ``structure`` module (not the heavy ``tools`` package) avoids pulling - # in S3/DB-dependent modules and avoids import cycles. - import app.services.document_agent.structure.page_locate_tools # noqa: F401 - - def locate(self, *, title: str, printed_page: int | None = None) -> SubAgentResult: - if not self.scope_pages: - return SubAgentResult(match=None, stop_reason="empty_scope") - - transcript: list[dict[str, Any]] = [] - seen_grep: dict[int, dict[str, Any]] = {} - last_verify: dict[str, Any] | None = None - - for round_index in range(self.config.max_rounds): - decision = self._decide( - ctx=self.ctx, - title=title, - scope_pages=self.scope_pages, - printed_page=printed_page, - page_offset_hint=self.page_offset_hint, - observations=transcript, - seen_pages=sorted(seen_grep.keys()), - last_verify=last_verify, - round_index=round_index, - max_rounds=self.config.max_rounds, - ) - if decision is None: - break - - action = str(decision.get("action") or "") - entry: dict[str, Any] = {"round": round_index, "decision": decision} - transcript.append(entry) - - if action == "grep": - query = str(decision.get("query_title") or title).strip() or title - pages = self._coerce_pages(decision.get("pages")) or self.scope_pages - result = self._dispatch( - "grep.title_pages", - { - "title": query, - "pages": pages, - "candidate_cap": self.config.candidate_cap, - }, - ) - candidates = [] - if result.status == "ok" and result.payload: - candidates = list(result.payload.get("candidates") or []) - for cand in candidates: - seen_grep[int(cand["page"])] = cand - entry["observation"] = { - "tool": "grep.title_pages", - "query": query, - "candidates": candidates[: self.config.candidate_cap], - } - - elif action == "verify": - pages = self._coerce_pages(decision.get("pages")) or sorted(seen_grep.keys()) - pages = pages[: max(self.config.verify_page_cap, 1)] - if not pages: - entry["observation"] = { - "tool": "verify.section_page", - "error": "no candidate pages to verify", - } - continue - result = self._dispatch("verify.section_page", {"title": title, "pages": pages}) - choice = result.payload if (result.status == "ok" and result.payload) else {} - last_verify = choice - entry["observation"] = { - "tool": "verify.section_page", - "pages": pages, - "choice": choice, - } - - elif action == "submit": - selected = decision.get("selected_page") - if selected is None: - return SubAgentResult(None, transcript, round_index + 1, "submit_null") - page = int(selected) - if page not in self.scope_pages: - entry["observation"] = {"error": f"submit page {page} outside scope"} - continue - match = self._build_match( - title=title, - page=page, - seen_grep=seen_grep, - last_verify=last_verify, - decision=decision, - transcript=transcript, - ) - return SubAgentResult(match, transcript, round_index + 1, "submit") - - elif action == "give_up": - return SubAgentResult(None, transcript, round_index + 1, "give_up") - - else: - entry["observation"] = {"error": f"unknown action {action!r}"} - - # Out of rounds / budget: accept a VLM-confirmed page if one exists. - if last_verify and last_verify.get("selected_page") is not None: - page = int(last_verify["selected_page"]) - if page in self.scope_pages: - match = self._build_match( - title=title, - page=page, - seen_grep=seen_grep, - last_verify=last_verify, - decision={"confidence": last_verify.get("confidence")}, - transcript=transcript, - ) - return SubAgentResult(match, transcript, self.config.max_rounds, "round_cap_with_verify") - return SubAgentResult(None, transcript, len(transcript), "exhausted") - - def _dispatch(self, name: str, args: dict[str, Any]): - assert self.ctx is not None, "dispatch requires a non-None ToolContext" - return REGISTRY.dispatch(name, self.ctx, args) - - def _coerce_pages(self, raw: Any) -> list[int]: - if not isinstance(raw, (list, tuple)): - return [] - pages: list[int] = [] - for value in raw: - try: - page = int(value) - except (TypeError, ValueError): - continue - if page in self.scope_pages and page not in pages: - pages.append(page) - return pages - - def _build_match( - self, - *, - title: str, - page: int, - seen_grep: dict[int, dict[str, Any]], - last_verify: dict[str, Any] | None, - decision: dict[str, Any], - transcript: list[dict[str, Any]], - ) -> TitleMatch: - grep_hit = seen_grep.get(page) or {} - vlm_confirmed = bool(last_verify and last_verify.get("selected_page") == page) - if vlm_confirmed: - assert last_verify is not None # guaranteed by vlm_confirmed check - source = cast(Any, last_verify.get("source") or "agent_vlm") - confidence = float(last_verify.get("confidence") or VLM_MATCH_DEFAULT_CONFIDENCE) - reason = last_verify.get("reason") - else: - source = cast(Any, "agent_heuristic") - confidence = float( - decision.get("confidence") - or grep_hit.get("confidence") - or HEURISTIC_MATCH_DEFAULT_CONFIDENCE - ) - reason = decision.get("reason") - return TitleMatch( - page=page, - confidence=confidence, - source=source, - matched_line=str(grep_hit.get("matched_line") or ""), - score=float(grep_hit.get("score") or confidence), - candidates=sorted(seen_grep.keys()), - evidence={ - "page_locate_agent": { - "decision": source, - "selected_page": page, - "reason": reason, - "vlm_confirmed": vlm_confirmed, - "rounds": len(transcript), - "seen_pages": sorted(seen_grep.keys()), - "transcript": transcript[-8:], - } - }, - ) - - -# ── Decision policy ──────────────────────────────────────────────────────── - - -def decide_next_action( - *, - ctx: ToolContext | None, - title: str, - scope_pages: list[int], - printed_page: int | None, - page_offset_hint: int | None, - observations: list[dict[str, Any]], - seen_pages: list[int], - last_verify: dict[str, Any] | None, - round_index: int, - max_rounds: int, -) -> dict[str, Any] | None: - """Pick the next action. - - Primary path is LLM-driven (the agent may rewrite the query, expand scope, - verify with a VLM, then submit). When no reasoning model is configured we - fall back to a minimal deterministic policy (strict grep → verify → submit) - that performs no query rewriting — only the LLM is allowed to relax the - query, so offline mode never silently mis-anchors. - """ - model = None - if ctx is not None: - model = ctx.settings.get("executor_model") or ctx.settings.get("model") - if ctx is None or not model or getattr(ctx, "budget", None) is None: - return _deterministic_decide( - title=title, - observations=observations, - seen_pages=seen_pages, - last_verify=last_verify, - ) - - from shared.utils.token_estimate import estimate_tokens - - prompt = _build_prompt( - title=title, - scope_pages=scope_pages, - printed_page=printed_page, - page_offset_hint=page_offset_hint, - observations=observations, - seen_pages=seen_pages, - last_verify=last_verify, - round_index=round_index, - max_rounds=max_rounds, - ) - est = estimate_tokens(prompt) - if not ctx.budget.try_reserve("plan", est): - logger.warning("[page_locate.subagent] planner budget exhausted for title={!r}", title) - return None - try: - from shared.services.ai.llm_overrides import get_text_client - - client, model = get_text_client(requested_model=model) - raw, usage = client.chat_completion_with_usage( - messages=[{"role": "user", "content": prompt}], - model=model, - temperature=0.0, - max_tokens=500, - response_format={"type": "json_object"}, - usage_task="page_memory.page_locate_decide", - ) - ctx.budget.commit("plan", actual=usage.get("total_tokens", est), est=est) - return _normalize_decision(json.loads(raw)) - except Exception as exc: - ctx.budget.refund("plan", est=est) - logger.warning("[page_locate.subagent] decide failed for title={!r}: {}", title, exc) - return None - - -def _deterministic_decide( - *, - title: str, - observations: list[dict[str, Any]], - seen_pages: list[int], - last_verify: dict[str, Any] | None, -) -> dict[str, Any]: - has_grep = any( - entry.get("observation", {}).get("tool") == "grep.title_pages" - for entry in observations - ) - if not has_grep: - return {"action": "grep", "query_title": title, "reason": "initial strict grep"} - if last_verify is None: - if seen_pages: - return {"action": "verify", "pages": seen_pages, "reason": "verify grep candidates"} - return {"action": "give_up", "reason": "strict grep found no candidates"} - selected = last_verify.get("selected_page") - if selected is not None: - return { - "action": "submit", - "selected_page": selected, - "confidence": last_verify.get("confidence", 0.6), - "reason": "submit VLM-confirmed page", - } - return {"action": "give_up", "reason": "verify returned no page"} - - -def _normalize_decision(data: dict[str, Any]) -> dict[str, Any]: - action = str(data.get("action") or "").strip().lower() - if action not in _ALLOWED_ACTIONS: - action = "give_up" - decision: dict[str, Any] = {"action": action, "reason": data.get("reason")} - if action == "grep": - decision["query_title"] = data.get("query_title") - decision["pages"] = data.get("pages") - elif action == "verify": - decision["pages"] = data.get("pages") - elif action == "submit": - selected = data.get("selected_page") - decision["selected_page"] = None if selected in (None, "", "null") else selected - decision["confidence"] = data.get("confidence") - return decision - - -def _build_prompt( - *, - title: str, - scope_pages: list[int], - printed_page: int | None, - page_offset_hint: int | None, - observations: list[dict[str, Any]], - seen_pages: list[int], - last_verify: dict[str, Any] | None, - round_index: int, - max_rounds: int, -) -> str: - scope_desc = ( - f"{scope_pages[0]}-{scope_pages[-1]} ({len(scope_pages)} pages)" - if scope_pages - else "none" - ) - hint = "" - if printed_page is not None: - guessed = printed_page + page_offset_hint if page_offset_hint is not None else None - hint = f"\nPrinted page number in the title's TOC entry: {printed_page}" + ( - f" (rough physical-page guess: {guessed})" if guessed is not None else "" - ) - compact_obs = json.dumps( - [_compact_obs(entry) for entry in observations[-4:]], - ensure_ascii=False, - ) - return ( - "You are a sub-agent that locates the START page of ONE section title " - "inside a PDF body. Decide the single next action and return strict JSON.\n\n" - f"Title to locate: {title!r}\n" - f"Allowed body pages: {scope_desc}{hint}\n" - f"Round {round_index + 1} of {max_rounds}. " - f"Candidate pages seen so far: {seen_pages}. " - f"Last VLM verification: {json.dumps(last_verify, ensure_ascii=False)}\n" - f"Observations (most recent last): {compact_obs}\n\n" - "Available actions (pick exactly ONE, as a JSON object):\n" - '1. {"action":"grep","query_title":"","reason":"..."}\n' - " Searches body pages for the query (exact line, whitespace-insensitive " - "compact text, and token overlap) and returns candidate pages with matched " - "lines. IMPORTANT: the title may carry a trailing document-reference code " - "that never appears contiguously in the body, and the heading may be split " - "across two lines. If a strict search of the full title returns nothing, " - "retry grep with a shortened, distinctive CORE of the title (drop the " - "trailing brackets/codes/notes).\n" - '2. {"action":"verify","pages":[,...],"reason":"..."}\n' - " Renders those pages as images and asks a vision model which one truly " - "STARTS the section (not a table-of-contents row, a running header/footer, " - "or a body mention/citation). Use this to disambiguate.\n" - '3. {"action":"submit","selected_page":,"confidence":<0..1>,"reason":"..."}\n' - " Finalize. selected_page must be a page you have already seen as a " - "candidate; use null only if the section is genuinely absent.\n" - '4. {"action":"give_up","reason":"..."}\n\n' - "Rules: when there is more than one candidate or any ambiguity, verify with " - "the vision model before submitting. Never invent page numbers. " - "Return ONLY the JSON object." - ) - - -def _compact_obs(entry: dict[str, Any]) -> dict[str, Any]: - obs = entry.get("observation") or {} - tool = obs.get("tool") - if tool == "grep.title_pages": - return { - "action": "grep", - "query": obs.get("query"), - "candidates": [ - {"page": c.get("page"), "source": c.get("source"), "line": (c.get("matched_line") or "")[:60]} - for c in (obs.get("candidates") or []) - ], - } - if tool == "verify.section_page": - choice = obs.get("choice") or {} - return { - "action": "verify", - "pages": obs.get("pages"), - "selected_page": choice.get("selected_page"), - "source": choice.get("source"), - "reason": (choice.get("reason") or "")[:80], - } - return {"action": entry.get("decision", {}).get("action"), "note": obs.get("error")} diff --git a/apps/worker/app/services/document_agent/structure/page_locate_tools.py b/apps/worker/app/services/document_agent/structure/page_locate_tools.py deleted file mode 100644 index 78f319b25..000000000 --- a/apps/worker/app/services/document_agent/structure/page_locate_tools.py +++ /dev/null @@ -1,158 +0,0 @@ -"""Registry tools for page-memory residual title location. - -These live under ``structure/`` (not ``tools/``) so the page-memory ReAct -sub-agent can register and dispatch them without importing the full profiling -``tools`` package (which pulls in S3/DB-heavy modules). ``tools/page_locate.py`` -re-exports these so the profile executor and ``tools/__init__`` keep working. -""" - -from __future__ import annotations - -import time -from typing import Any - -from app.services.document_agent.manifest import ToolContext, ToolResult -from app.services.document_agent.registry import register_tool -from app.services.document_agent.structure import page_locate_agent as _pla -from app.services.document_agent.structure.hierarchy_locator import TitleMatch - -SYNTHETIC_CANDIDATE_CONFIDENCE = 0.4 - - -@register_tool( - name="grep.title_pages", - description=( - "Find candidate body pages for a section title using strict heading, " - "normalized, compact, and token grep variants." - ), - parameters={ - "type": "object", - "properties": { - "title": {"type": "string"}, - "pages": {"type": "array", "items": {"type": "integer"}}, - "candidate_cap": {"type": "integer"}, - }, - "required": ["title", "pages"], - }, -) -def grep_title_pages(ctx: ToolContext, args: dict[str, Any]) -> ToolResult: - start = time.monotonic() - title = str(args.get("title") or "").strip() - pages = _valid_pages(ctx, args.get("pages") or []) - if not title or not pages: - return ToolResult( - status="error", - error="grep.title_pages requires title and pages", - latency_ms=int((time.monotonic() - start) * 1000), - ) - page_texts = _get_page_texts(ctx, pages) - matches = _pla.grep_title_page_candidates( - title=title, - scope_pages=pages, - page_texts=page_texts, - limit=int(args.get("candidate_cap") or 8), - ) - return ToolResult( - status="ok", - payload={ - "title": title, - "candidates": [_match_to_payload(match) for match in matches], - }, - latency_ms=int((time.monotonic() - start) * 1000), - ) - - -@register_tool( - name="verify.section_page", - description=( - "Use VLM page screenshots to choose which candidate page starts a section." - ), - parameters={ - "type": "object", - "properties": { - "title": {"type": "string"}, - "pages": {"type": "array", "items": {"type": "integer"}}, - }, - "required": ["title", "pages"], - }, -) -def verify_section_page(ctx: ToolContext, args: dict[str, Any]) -> ToolResult: - start = time.monotonic() - title = str(args.get("title") or "").strip() - pages = _valid_pages(ctx, args.get("pages") or []) - if not title or not pages: - return ToolResult( - status="error", - error="verify.section_page requires title and pages", - latency_ms=int((time.monotonic() - start) * 1000), - ) - page_texts = _get_page_texts(ctx, pages) - grep_by_page = { - match.page: match - for match in _pla.grep_title_page_candidates( - title=title, - scope_pages=pages, - page_texts=page_texts, - limit=len(pages), - ) - } - # Always verify the *requested* pages, even when grep found nothing on them - # (the heading may be split across lines / wrapped, so render + VLM decides). - candidates = [ - grep_by_page.get(page) - or TitleMatch( - page=page, - confidence=SYNTHETIC_CANDIDATE_CONFIDENCE, - source="agent_heuristic", - matched_line="", - score=SYNTHETIC_CANDIDATE_CONFIDENCE, - candidates=[page], - evidence={"synthesized": True}, - ) - for page in pages - ] - choice = _pla.verify_section_page_choice( - ctx=ctx, - title=title, - candidate_matches=candidates, - candidate_page_cap=len(pages), - ) - return ToolResult( - status="ok", - payload=choice, - latency_ms=int((time.monotonic() - start) * 1000), - tokens_used=int(choice.get("tokens_used") or 0), - ) - - -def _valid_pages(ctx: ToolContext, raw_pages: Any) -> list[int]: - page_count = max(int(ctx.blackboard.page_count or 0), 0) - return sorted( - { - int(page) - for page in raw_pages - if 1 <= int(page) <= page_count - } - ) - - -def _get_page_texts(ctx: ToolContext, pages: list[int]) -> dict[int, str]: - cache = getattr(ctx.blackboard, "page_full_text_cache", {}) - missing = [page for page in pages if page not in cache] - if missing: - from app.services.document_agent.pdf_text import read_page_texts - - cache.update(read_page_texts(ctx.pdf_path, missing, timeout=300)) - return {page: cache.get(page, "") for page in pages} - - -def _match_to_payload(match: Any) -> dict[str, Any]: - return { - "page": match.page, - "confidence": match.confidence, - "source": match.source, - "matched_line": match.matched_line, - "score": match.score, - "candidates": match.candidates, - "evidence": match.evidence, - } diff --git a/apps/worker/app/services/document_agent/tools/__init__.py b/apps/worker/app/services/document_agent/tools/__init__.py index ac5c29909..177ff879f 100644 --- a/apps/worker/app/services/document_agent/tools/__init__.py +++ b/apps/worker/app/services/document_agent/tools/__init__.py @@ -6,7 +6,6 @@ from . import find_toc_anchor_pages as find_toc_anchor_pages # noqa: F401 from . import grep_text as grep_text # noqa: F401 from . import inspect_pages as inspect_pages # noqa: F401 -from . import page_locate as page_locate # noqa: F401 from . import propose_shard_plan as propose_shard_plan # noqa: F401 from . import validate_anatomy_map as validate_anatomy_map # noqa: F401 from . import verdict as verdict # noqa: F401 diff --git a/apps/worker/app/services/document_agent/tools/classify_page_kinds.py b/apps/worker/app/services/document_agent/tools/classify_page_kinds.py index fb60765d7..d3a11d52b 100644 --- a/apps/worker/app/services/document_agent/tools/classify_page_kinds.py +++ b/apps/worker/app/services/document_agent/tools/classify_page_kinds.py @@ -11,17 +11,6 @@ def _label_feature(feature: PageFeature) -> PageLabel: page = feature.page - if ( - feature.raw_text_length < 80 - and feature.image_coverage < 0.02 - and feature.drawings_count < 5 - ): - return PageLabel( - page=page, - kind="low_content", - confidence=0.78, - evidence={"signal": "low_text_image_drawings"}, - ) if feature.orientation == "landscape": return PageLabel( page=page, @@ -29,23 +18,6 @@ def _label_feature(feature: PageFeature) -> PageLabel: confidence=0.78, evidence={"width": feature.width, "height": feature.height}, ) - if feature.image_coverage >= 0.35 and feature.raw_text_length < 250: - return PageLabel( - page=page, - kind="image_heavy", - confidence=0.84, - evidence={"image_coverage": feature.image_coverage}, - ) - if feature.table_count > 0 or feature.drawings_count >= 80: - return PageLabel( - page=page, - kind="table_heavy", - confidence=0.72, - evidence={ - "table_count": feature.table_count, - "drawings_count": feature.drawings_count, - }, - ) return PageLabel(page=page, kind="normal", confidence=0.65, evidence={}) @@ -67,10 +39,6 @@ def classify_page_kinds(ctx: ToolContext, _args: dict[str, Any]) -> ToolResult: "confidence": label.confidence, "evidence": label.evidence, "raw_text_length": feature.raw_text_length if feature else None, - "image_coverage": feature.image_coverage if feature else None, - "table_count": feature.table_count if feature else None, - "drawings_count": feature.drawings_count if feature else None, - "text_preview": (feature.text_lines_preview[:4] if feature else []), } ) return ToolResult( diff --git a/apps/worker/app/services/document_agent/tools/extract_toc_with_boundaries.py b/apps/worker/app/services/document_agent/tools/extract_toc_with_boundaries.py index af2379013..a2de3d295 100644 --- a/apps/worker/app/services/document_agent/tools/extract_toc_with_boundaries.py +++ b/apps/worker/app/services/document_agent/tools/extract_toc_with_boundaries.py @@ -447,16 +447,6 @@ def extract_toc_with_boundaries( "batch_meta": batch_meta, } - # Persist toc_hierarchies to disk for inspection / downstream reuse - if toc_hierarchies and ctx.output_dir: - toc_json_path = os.path.join(ctx.output_dir, "toc_hierarchies.json") - try: - with open(toc_json_path, "w", encoding="utf-8") as f: - json.dump(toc_hierarchies, f, ensure_ascii=False, indent=2) - logger.info("[extract.toc] wrote toc_hierarchies to {}", toc_json_path) - except Exception as exc: - logger.warning("[extract.toc] failed to write toc_hierarchies: {}", exc) - # Build toc_ranges from confirmed TOC pages for summary toc_ranges_out: list[list[int]] = [] if toc_hierarchies: diff --git a/apps/worker/app/services/document_agent/tools/page_locate.py b/apps/worker/app/services/document_agent/tools/page_locate.py deleted file mode 100644 index 0e67f1435..000000000 --- a/apps/worker/app/services/document_agent/tools/page_locate.py +++ /dev/null @@ -1,16 +0,0 @@ -"""Page-memory residual title-location tools. - -The implementations live in ``structure/page_locate_tools.py`` so the -page-memory sub-agent can register/dispatch them without importing this heavy -``tools`` package. Importing this module (e.g. via ``tools/__init__``) ensures -the tools are registered for the profile executor as well. -""" - -from __future__ import annotations - -from app.services.document_agent.structure.page_locate_tools import ( # noqa: F401 - grep_title_pages, - verify_section_page, -) - -__all__ = ["grep_title_pages", "verify_section_page"] diff --git a/apps/worker/app/services/document_agent/tools/probe_page_features.py b/apps/worker/app/services/document_agent/tools/probe_page_features.py index 678557fad..be92b853c 100644 --- a/apps/worker/app/services/document_agent/tools/probe_page_features.py +++ b/apps/worker/app/services/document_agent/tools/probe_page_features.py @@ -1,19 +1,30 @@ -"""Full-page structural probing.""" +"""Full-page structural probing: text pass, then optional asset pass.""" from __future__ import annotations import gc +import math import time +from collections import defaultdict from typing import Any from app.services.document_agent.manifest import PageFeature, ToolContext, ToolResult -from app.services.document_agent.pdf_text import top_lines from app.services.document_parser.formats.pdf.pymupdf_subprocess import ( run_in_child_process, worker, ) from loguru import logger +# Cluster nearby drawing strokes into one figure region (PDF points). +_FIGURE_CLUSTER_GAP = 18.0 +# Drop clusters smaller than this many strokes (noise). +_FIGURE_CLUSTER_MIN_PATHS = 3 +# Drop aggregated figures smaller than this area (PDF pt²); catches most logos. +_FIGURE_MIN_AREA = 5000.0 +# Near-full-page sparse stroke frames are treated as borders, not figures. +_FIGURE_FULLPAGE_AREA_RATIO = 0.92 +_FIGURE_FULLPAGE_MAX_PATHS = 25 + def _rect_area(rect: Any) -> float: width = max(float(getattr(rect, "width", 0.0) or 0.0), 0.0) @@ -21,12 +32,185 @@ def _rect_area(rect: Any) -> float: return width * height -def _image_coverage(page: Any, page_area: float) -> tuple[float, int]: - if page_area <= 0: - return 0.0, 0 - area = 0.0 +def _clip_bbox( + x0: float, + y0: float, + x1: float, + y1: float, + *, + clip: Any, +) -> list[float]: + return [ + round(max(x0, float(clip.x0)), 2), + round(max(y0, float(clip.y0)), 2), + round(min(x1, float(clip.x1)), 2), + round(min(y1, float(clip.y1)), 2), + ] + + +def _valid_bbox(box: list[float]) -> bool: + return len(box) == 4 and box[2] > box[0] and box[3] > box[1] + + +def _outside_content_band( + y0: float, + y1: float, + *, + page_h: float, + header_y: float | None, + footer_y: float | None, +) -> bool: + """True if any vertical edge of an aggregated figure leaves the content band. + + ``header_y`` / ``footer_y`` are fractions of page height (top origin, y down). + """ + if header_y is not None and y0 < header_y * page_h: + return True + if footer_y is not None and y1 > footer_y * page_h: + return True + return False + + +def _rects_near( + a: tuple[float, float, float, float], + b: tuple[float, float, float, float], + gap: float, +) -> bool: + ax0, ay0, ax1, ay1 = a + bx0, by0, bx1, by1 = b + return not ( + ax1 + gap < bx0 + or bx1 + gap < ax0 + or ay1 + gap < by0 + or by1 + gap < ay0 + ) + + +def _cluster_indices(rects: list[tuple[float, float, float, float]], gap: float) -> list[list[int]]: + """Union-find cluster of nearby AABBs via spatial grid (avg ~O(n)).""" + n = len(rects) + if n == 0: + return [] + parent = list(range(n)) + + def find(i: int) -> int: + while parent[i] != i: + parent[i] = parent[parent[i]] + i = parent[i] + return i + + def union(i: int, j: int) -> None: + ri, rj = find(i), find(j) + if ri != rj: + parent[rj] = ri + + cell = max(gap, 1.0) + grid: dict[tuple[int, int], list[int]] = defaultdict(list) + + def cell_range( + x0: float, y0: float, x1: float, y1: float + ) -> tuple[int, int, int, int]: + return ( + math.floor(x0 / cell), + math.floor(y0 / cell), + math.floor(x1 / cell), + math.floor(y1 / cell), + ) + + for i, (x0, y0, x1, y1) in enumerate(rects): + ix0, iy0, ix1, iy1 = cell_range(x0, y0, x1, y1) + for ix in range(ix0, ix1 + 1): + for iy in range(iy0, iy1 + 1): + grid[(ix, iy)].append(i) + + for i, (x0, y0, x1, y1) in enumerate(rects): + ix0, iy0, ix1, iy1 = cell_range(x0 - gap, y0 - gap, x1 + gap, y1 + gap) + seen: set[int] = set() + for ix in range(ix0, ix1 + 1): + for iy in range(iy0, iy1 + 1): + for j in grid.get((ix, iy), ()): + if j <= i or j in seen: + continue + seen.add(j) + if _rects_near(rects[i], rects[j], gap): + union(i, j) + + groups: dict[int, list[int]] = defaultdict(list) + for i in range(n): + groups[find(i)].append(i) + return list(groups.values()) + +def _figure_bboxes_from_drawings( + drawings: list[Any], + *, + page_rect: Any, + page_area: float, + header_y: float | None, + footer_y: float | None, +) -> list[dict[str, Any]]: + """Cluster drawing strokes, then filter aggregated figure bboxes only.""" + page_h = float(getattr(page_rect, "height", 0.0) or 0.0) + kept_rects: list[tuple[float, float, float, float]] = [] + for drawing in drawings: + if not isinstance(drawing, dict): + continue + rect = drawing.get("rect") + if rect is None: + continue + x0 = float(getattr(rect, "x0", 0.0) or 0.0) + y0 = float(getattr(rect, "y0", 0.0) or 0.0) + x1 = float(getattr(rect, "x1", 0.0) or 0.0) + y1 = float(getattr(rect, "y1", 0.0) or 0.0) + if x1 <= x0 or y1 <= y0: + continue + kept_rects.append((x0, y0, x1, y1)) + + if not kept_rects: + return [] + + figures: list[dict[str, Any]] = [] + for idxs in _cluster_indices(kept_rects, _FIGURE_CLUSTER_GAP): + if len(idxs) < _FIGURE_CLUSTER_MIN_PATHS: + continue + xs0 = [kept_rects[i][0] for i in idxs] + ys0 = [kept_rects[i][1] for i in idxs] + xs1 = [kept_rects[i][2] for i in idxs] + ys1 = [kept_rects[i][3] for i in idxs] + box = _clip_bbox(min(xs0), min(ys0), max(xs1), max(ys1), clip=page_rect) + if not _valid_bbox(box): + continue + area = max(box[2] - box[0], 0.0) * max(box[3] - box[1], 0.0) + if area < _FIGURE_MIN_AREA: + continue + if _outside_content_band( + box[1], box[3], page_h=page_h, header_y=header_y, footer_y=footer_y + ): + continue + area_ratio = area / max(page_area, 1.0) + if ( + area_ratio > _FIGURE_FULLPAGE_AREA_RATIO + and len(idxs) < _FIGURE_FULLPAGE_MAX_PATHS + ): + continue + figures.append({"kind": "figure", "bbox": box}) + return figures + + +def _probe_visual_assets( + page: Any, + page_area: float, + *, + header_y: float | None, + footer_y: float | None, +) -> dict[str, Any]: + """Collect counts + coarse asset bboxes from images / tables / drawings.""" + image_area = 0.0 + bboxes: list[dict[str, Any]] = [] + seen_image_rects: set[tuple[float, float, float, float]] = set() + page_rect = page.rect + images = page.get_images(full=True) or [] - seen: set[tuple[float, float, float, float]] = set() + image_count = len(images) for image in images: xref = image[0] try: @@ -34,56 +218,87 @@ def _image_coverage(page: Any, page_area: float) -> tuple[float, int]: except Exception: rects = [] for rect in rects: - key = ( - round(float(getattr(rect, "x0", 0.0) or 0.0), 2), - round(float(getattr(rect, "y0", 0.0) or 0.0), 2), - round(float(getattr(rect, "x1", 0.0) or 0.0), 2), - round(float(getattr(rect, "y1", 0.0) or 0.0), 2), + box = _clip_bbox( + float(getattr(rect, "x0", 0.0) or 0.0), + float(getattr(rect, "y0", 0.0) or 0.0), + float(getattr(rect, "x1", 0.0) or 0.0), + float(getattr(rect, "y1", 0.0) or 0.0), + clip=page_rect, ) - if key in seen: + if not _valid_bbox(box): continue - seen.add(key) - area += _rect_area(rect) - return min(area / page_area, 1.0), len(images) - + key = (box[0], box[1], box[2], box[3]) + if key in seen_image_rects: + continue + seen_image_rects.add(key) + image_area += max(box[2] - box[0], 0.0) * max(box[3] - box[1], 0.0) + bboxes.append({"kind": "image", "bbox": box}) -def _table_count(page: Any) -> int: + table_count = 0 try: finder = page.find_tables() - return len(getattr(finder, "tables", []) or []) + tables = getattr(finder, "tables", []) or [] + table_count = len(tables) + for table in tables: + raw = getattr(table, "bbox", None) + if raw is None: + continue + box = _clip_bbox( + float(raw[0]), + float(raw[1]), + float(raw[2]), + float(raw[3]), + clip=page_rect, + ) + if _valid_bbox(box): + bboxes.append({"kind": "table", "bbox": box}) except Exception: - return 0 + table_count = 0 + + try: + drawings = page.get_drawings() or [] + except Exception: + drawings = [] + drawings_count = len(drawings) + bboxes.extend( + _figure_bboxes_from_drawings( + drawings, + page_rect=page_rect, + page_area=page_area, + header_y=header_y, + footer_y=footer_y, + ) + ) + + coverage = min(image_area / page_area, 1.0) if page_area > 0 else 0.0 + return { + "image_coverage": round(coverage, 4), + "image_count": image_count, + "table_count": table_count, + "drawings_count": drawings_count, + "asset_bboxes": bboxes or None, + } -def _probe_one(page: Any, page_number: int) -> dict[str, Any]: +def _probe_text_one(page: Any, page_number: int) -> dict[str, Any]: rect = page.rect area = max(_rect_area(rect), 1.0) text = page.get_text() or "" raw_text_length = len(text.strip()) - image_coverage, image_count = _image_coverage(page, area) - try: - drawings_count = len(page.get_drawings() or []) - except Exception: - drawings_count = 0 orientation = "landscape" if float(rect.width) > float(rect.height) else "portrait" return { "page": page_number, "raw_text_length": raw_text_length, "text_density": round(raw_text_length / area * 10000, 4), - "image_coverage": round(image_coverage, 4), - "image_count": image_count, - "table_count": _table_count(page), - "drawings_count": drawings_count, "orientation": orientation, "width": round(float(rect.width), 2), "height": round(float(rect.height), 2), - "is_blank_like": raw_text_length < 20 and image_coverage < 0.02 and drawings_count < 5, - "text_lines_preview": top_lines(text, max_lines=30), + "is_blank_like": raw_text_length < 50, } @worker -def _probe_worker(queue, pdf_path: str) -> None: +def _probe_text_worker(queue, pdf_path: str) -> None: import pymupdf # type: ignore[import] features: list[dict[str, Any]] = [] @@ -92,7 +307,7 @@ def _probe_worker(queue, pdf_path: str) -> None: doc = pymupdf.open(pdf_path) page_count = int(doc.page_count) for idx in range(page_count): - features.append(_probe_one(doc[idx], idx + 1)) + features.append(_probe_text_one(doc[idx], idx + 1)) finally: try: doc.close() @@ -102,31 +317,62 @@ def _probe_worker(queue, pdf_path: str) -> None: queue.put({"ok": True, "page_count": page_count, "features": features}) +@worker +def _probe_assets_worker( + queue, + pdf_path: str, + header_y: float | None, + footer_y: float | None, +) -> None: + import pymupdf # type: ignore[import] + + assets: list[dict[str, Any]] = [] + try: + doc = pymupdf.open(pdf_path) + for idx in range(int(doc.page_count)): + page = doc[idx] + area = max(_rect_area(page.rect), 1.0) + visual = _probe_visual_assets( + page, area, header_y=header_y, footer_y=footer_y + ) + assets.append({"page": idx + 1, **visual}) + finally: + try: + doc.close() + except Exception as exc: + logger.warning("[document_agent] failed to close PDF document in assets probe worker: {}", exc) + gc.collect() + queue.put({"ok": True, "assets": assets}) + + def probe_page_features(ctx: ToolContext, _args: dict[str, Any]) -> ToolResult: + """Text-only page probe used before coarse VLM classification.""" start = time.monotonic() try: - result = run_in_child_process(_probe_worker, ctx.pdf_path, timeout=300) + result = run_in_child_process(_probe_text_worker, ctx.pdf_path, timeout=300) features = [ PageFeature( page=int(item["page"]), raw_text_length=int(item.get("raw_text_length") or 0), text_density=float(item.get("text_density") or 0.0), - image_coverage=float(item.get("image_coverage") or 0.0), - image_count=int(item.get("image_count") or 0), - table_count=int(item.get("table_count") or 0), - drawings_count=int(item.get("drawings_count") or 0), + image_coverage=0.0, + image_count=0, + table_count=0, + drawings_count=0, orientation=str(item.get("orientation") or "portrait"), # type: ignore[arg-type] width=float(item.get("width") or 0.0), height=float(item.get("height") or 0.0), + has_asset=False, is_blank_like=bool(item.get("is_blank_like")), - text_lines_preview=list(item.get("text_lines_preview") or []), + asset_bboxes=None, ) for item in (result.get("features") or []) ] ctx.blackboard.page_features = sorted(features, key=lambda f: f.page) ctx.blackboard.page_count = int(result.get("page_count") or len(features)) ctx.blackboard.global_signals["total_pages"] = ctx.blackboard.page_count - logger.info("[document_agent] probed {} pages", ctx.blackboard.page_count) + ctx.blackboard.global_signals["assets_probed"] = False + logger.info("[document_agent] probed text on {} pages", ctx.blackboard.page_count) return ToolResult( status="ok", payload={"page_count": ctx.blackboard.page_count}, @@ -138,3 +384,80 @@ def probe_page_features(ctx: ToolContext, _args: dict[str, Any]) -> ToolResult: error=str(exc), latency_ms=int((time.monotonic() - start) * 1000), ) + + +def probe_page_assets(ctx: ToolContext, _args: dict[str, Any]) -> ToolResult: + """Asset coarse extraction after coarse VLM; drawings honor margin band.""" + start = time.monotonic() + if not ctx.blackboard.page_features: + return ToolResult( + status="error", + error="page_features missing; run text probe first", + latency_ms=int((time.monotonic() - start) * 1000), + ) + profile = ctx.blackboard.document_profile + header_y = getattr(profile, "header_y", None) if profile else None + footer_y = getattr(profile, "footer_y", None) if profile else None + try: + result = run_in_child_process( + _probe_assets_worker, + ctx.pdf_path, + header_y, + footer_y, + timeout=300, + ) + by_page = { + int(item["page"]): item for item in (result.get("assets") or []) + } + updated: list[PageFeature] = [] + for feature in ctx.blackboard.page_features: + visual = by_page.get(feature.page) or {} + has_asset = visual.get("asset_bboxes") is not None + updated.append( + PageFeature( + page=feature.page, + raw_text_length=feature.raw_text_length, + text_density=feature.text_density, + image_coverage=float(visual.get("image_coverage") or 0.0), + image_count=int(visual.get("image_count") or 0), + table_count=int(visual.get("table_count") or 0), + drawings_count=int(visual.get("drawings_count") or 0), + orientation=feature.orientation, + width=feature.width, + height=feature.height, + has_asset=has_asset, + is_blank_like=feature.raw_text_length < 50 and not has_asset, + asset_bboxes=( + list(visual["asset_bboxes"]) + if isinstance(visual.get("asset_bboxes"), list) + else None + ), + ) + ) + ctx.blackboard.page_features = sorted(updated, key=lambda f: f.page) + ctx.blackboard.global_signals["assets_probed"] = True + ctx.blackboard.global_signals["content_margins"] = { + "header_y": header_y, + "footer_y": footer_y, + } + logger.info( + "[document_agent] probed assets on {} pages (header_y={}, footer_y={})", + ctx.blackboard.page_count, + header_y, + footer_y, + ) + return ToolResult( + status="ok", + payload={ + "page_count": ctx.blackboard.page_count, + "header_y": header_y, + "footer_y": footer_y, + }, + latency_ms=int((time.monotonic() - start) * 1000), + ) + except Exception as exc: + return ToolResult( + status="error", + error=str(exc), + latency_ms=int((time.monotonic() - start) * 1000), + ) diff --git a/apps/worker/app/services/document_agent/tools/propose_shard_plan.py b/apps/worker/app/services/document_agent/tools/propose_shard_plan.py index 714ba9999..1c2737db7 100644 --- a/apps/worker/app/services/document_agent/tools/propose_shard_plan.py +++ b/apps/worker/app/services/document_agent/tools/propose_shard_plan.py @@ -651,20 +651,20 @@ def _deterministic_no_toc_plan( *, page_count: int, max_pages: int, - low_content_pages: list[int], + blank_pages: list[int], ) -> tuple[list[tuple[int, str, str, float]], str]: - """Deterministic shard plan using low-content pages as split candidates.""" + """Deterministic shard plan using blank-like pages as split candidates.""" cuts: list[tuple[int, str, str, float]] = [] previous = 0 while page_count - previous > max_pages: target = previous + max_pages - # Look for a low-content page near the max boundary + # Look for a blank-like page near the max boundary eligible = [ - p for p in low_content_pages if previous + (max_pages - 20) < p <= target + p for p in blank_pages if previous + (max_pages - 20) < p <= target ] if eligible: chosen = max(eligible) - cuts.append((chosen, "blank_separator", f"low-content page at {chosen}", 0.5)) + cuts.append((chosen, "blank_separator", f"blank-like page at {chosen}", 0.5)) previous = chosen else: cut_page = previous + max_pages @@ -673,10 +673,10 @@ def _deterministic_no_toc_plan( return cuts, "too_large" -def _get_low_content_pages(ctx: ToolContext) -> list[int]: - """Extract low-content page numbers from page labels.""" - labels = ctx.blackboard.page_labels or [] - return sorted(label.page for label in labels if label.kind == "low_content") +def _get_blank_pages(ctx: ToolContext) -> list[int]: + """Extract blank-like page numbers from page features.""" + features = ctx.blackboard.page_features or [] + return sorted(feature.page for feature in features if feature.is_blank_like) @register_tool( @@ -781,14 +781,14 @@ def propose_shard_plan(ctx: ToolContext, _args: dict[str, Any]) -> ToolResult: ) rationale = "Deterministic chapter plan (no LLM)." else: - # Path B: No TOC — purely deterministic using low-content pages - low_content_pages = _get_low_content_pages(ctx) + # Path B: No TOC — purely deterministic using blank-like pages + blank_pages = _get_blank_pages(ctx) cuts, reason = _deterministic_no_toc_plan( page_count=page_count, max_pages=max_pages, - low_content_pages=low_content_pages, + blank_pages=blank_pages, ) - rationale = "Deterministic plan from low-content page boundaries (no TOC)." + rationale = "Deterministic plan from blank-like page boundaries (no TOC)." shards = _cuts_to_shards(cuts, page_count) enabled = len(shards) > 1 diff --git a/apps/worker/app/services/document_ingestion/success_finalization.py b/apps/worker/app/services/document_ingestion/success_finalization.py index d900acc34..9acc0e3ab 100644 --- a/apps/worker/app/services/document_ingestion/success_finalization.py +++ b/apps/worker/app/services/document_ingestion/success_finalization.py @@ -48,7 +48,6 @@ def finalize_parse_success( job_context=job_context, source_file_name=source_file_name, ) - _attach_document_top_summary(result_package.chunks, document_top_summary) _refresh_processing_stages(job_context) lifecycle_service.update_progress( @@ -92,6 +91,7 @@ def finalize_parse_success( stored_count=stored_count, delivery_mode="url", section_summaries=section_summaries, + document_top_summary=document_top_summary, ) if finalization_response.get("status") != "success": logger.error( @@ -137,6 +137,7 @@ def _enrich_document_navigation( ) -> tuple[str, dict[str, str]]: document_top_summary = "" section_summaries: dict[str, str] = {} + enrich_results: dict[str, str] = {} add_dir = artifact.add_dir parsed_contents_df = artifact.dataframe if add_dir and source_file_name: @@ -153,15 +154,27 @@ def _enrich_document_navigation( "summary_use_llm", False, ) - enrich_doc_nav_summaries( + top_summary_use_llm = JobMetadataHelper.get_parsing_param( + job_context.job_metadata, + "top_summary_use_llm", + True, + ) + enrich_results = enrich_doc_nav_summaries( document_root_for_enrich, source_file=source_file_name, use_llm=summary_use_llm, + top_summary_use_llm=top_summary_use_llm, + chunks=chunks, ) section_summaries = build_section_summary_lookup(str(add_dir)) except Exception as exc: logger.warning(f"doc_nav enrichment failed (non-fatal): {exc}") - document_top_summary = load_nav_top_summary(str(add_dir), source_file_name) + enrich_results = {} + document_top_summary = str( + enrich_results.get(source_file_name) or "" + ).strip() + if not document_top_summary: + document_top_summary = load_nav_top_summary(str(add_dir), source_file_name) return document_top_summary, section_summaries @@ -180,21 +193,6 @@ def _refresh_processing_stages(job_context: ParseJobContext) -> None: job_context.job_metadata["stages"] = stages -def _attach_document_top_summary( - chunks: list[dict[str, Any]], - document_top_summary: str, -) -> None: - if not document_top_summary: - return - - for chunk in chunks: - metadata = chunk.get("metadata") - if not isinstance(metadata, dict): - metadata = {} - chunk["metadata"] = metadata - metadata["document_top_summary"] = document_top_summary - - def _record_processing_completion( *, job_id: str, diff --git a/apps/worker/app/services/document_parser/assets/image_size_filter.py b/apps/worker/app/services/document_parser/assets/image_size_filter.py new file mode 100644 index 000000000..3c2c59dc7 --- /dev/null +++ b/apps/worker/app/services/document_parser/assets/image_size_filter.py @@ -0,0 +1,46 @@ +"""Shared minimum image size gate used across parsers. + +Discard before rename / VLM summary so undersized icons never enter the +asset pipeline. +""" + +from __future__ import annotations + +from pathlib import Path + +from loguru import logger + +from shared.core.constants.processing import ProcessingConstants + + +def is_below_img_min_size(byte_size: int) -> bool: + """Return True when ``byte_size`` is under ``IMG_MIN_SIZE`` (10KB).""" + return byte_size < ProcessingConstants.IMG_MIN_SIZE + + +def discard_undersized_image_file( + path: Path | str, + *, + label: str = "image", +) -> bool: + """Delete ``path`` when it exists and is undersized. + + Returns True when the file was discarded (caller should skip rename/LLM). + Returns False when the file is large enough to keep, or does not exist. + """ + image_path = Path(path) + if not image_path.exists(): + return False + + file_size = image_path.stat().st_size + if not is_below_img_min_size(file_size): + return False + + logger.debug( + f"Skipping {label} (too small: {file_size / 1024:.1f} KB): {image_path}" + ) + try: + image_path.unlink() + except OSError as exc: + logger.debug(f"Failed to remove undersized {label} {image_path}: {exc}") + return True diff --git a/apps/worker/app/services/document_parser/assets/inline_asset.py b/apps/worker/app/services/document_parser/assets/inline_asset.py index c39917d61..27348a19b 100644 --- a/apps/worker/app/services/document_parser/assets/inline_asset.py +++ b/apps/worker/app/services/document_parser/assets/inline_asset.py @@ -38,12 +38,18 @@ def build_table_asset_row( addtime: str, entities: str = "", asset_title: str = "", + image_refs: list[str] | None = None, ) -> ParsedRow: row_content = relative_path + # Multiline type channel carries table→image embeds + # (same pattern as PTXT\n[tables/...] for text rows). + type_value = "table" + if image_refs: + type_value = "\n".join(["table", *image_refs]) return ParsedRow( content=row_content, path=relative_path, - type="table", + type=type_value, keywords=keywords, summary=summary, know_id=know_id, diff --git a/apps/worker/app/services/document_parser/formats/docx/block_stream.py b/apps/worker/app/services/document_parser/formats/docx/block_stream.py index dafe28aba..3d8ddacf0 100644 --- a/apps/worker/app/services/document_parser/formats/docx/block_stream.py +++ b/apps/worker/app/services/document_parser/formats/docx/block_stream.py @@ -5,6 +5,7 @@ import zipfile from app.services.document_parser.formats.docx.toc import detect_doc_tocs, detect_sdt_toc +from app.services.document_parser.assets.image_size_filter import is_below_img_min_size from docx import Document from docx.oxml.table import CT_Tbl from docx.oxml.text.paragraph import CT_P @@ -134,6 +135,8 @@ def iter_block_items(doc_data): continue seen_rids.add(rid) data = docx.read("word/" + target) + if is_below_img_min_size(len(data)): + continue yield ( ele_num, None, @@ -234,9 +237,7 @@ def iter_block_items(doc_data): continue cell_seen_rids.add(rid) data = docx.read("word/" + target) - if ( - len(data) < 10 * 1024 - ): # Skip small images (<10KB, likely icons) + if is_below_img_min_size(len(data)): continue imgs_in_cell.append( { @@ -271,7 +272,7 @@ def iter_block_items(doc_data): f"Failed to convert VML cell image to PNG: {e}" ) continue - if len(png_data) < 10 * 1024: + if is_below_img_min_size(len(png_data)): continue orig_name = target.split("/")[-1] png_name = os.path.splitext(orig_name)[0] + ".png" diff --git a/apps/worker/app/services/document_parser/formats/docx/parser.py b/apps/worker/app/services/document_parser/formats/docx/parser.py index 6ec2d3d87..f978ab77d 100755 --- a/apps/worker/app/services/document_parser/formats/docx/parser.py +++ b/apps/worker/app/services/document_parser/formats/docx/parser.py @@ -43,6 +43,7 @@ from shared.core.config import settings from shared.core.exceptions.domain_exceptions import DocxParsingException from shared.core.exceptions.knowhere_exception import KnowhereException +from shared.services.chunks.path_segments import escape_path_segment from shared.utils.chunk_refs import build_chunk_ref, has_chunk_ref from app.services.common.file_loading import load_file_bytes from app.services.common.file_utils import path_handle @@ -583,9 +584,6 @@ def parse_docx( headings_stack[-1]["content"].append(text) elif label == "IMAGE": - if meta and meta.get("size", 0) < 10 * 1024: - continue - headings_stack = asset_accumulator.append_image( meta, headings_stack, @@ -670,7 +668,8 @@ def convert_doc2dics( continue # Build tentative path to check for duplicates - tentative_path = doc_name + split_char + key + escaped_doc_name = escape_path_segment(doc_name) + tentative_path = escaped_doc_name + split_char + key # Deduplicate: if path already exists, add suffix if tentative_path in path_counter: @@ -680,7 +679,7 @@ def convert_doc2dics( else: path_counter[tentative_path] = 1 - path_keys.append((doc_name + split_char + key)) + path_keys.append((escaped_doc_name + split_char + key)) bottom_content = joined bottom_tokens = tokenize2stw_remove( [bottom_content], base_llm_paras["stopwords"] @@ -699,11 +698,14 @@ def convert_doc2dics( know_id = gen_str_codes(pure_text) # Use relative_root for path instead of the absolute output directory. path_suffix = key if key.strip() else "" - know_path = ( - split_char.join([relative_root, path_suffix]) - if relative_root and path_suffix - else (relative_root or path_suffix) - ) + if relative_root and path_suffix: + know_path = f"{escape_path_segment(relative_root)}/{path_suffix}" + else: + know_path = ( + escape_path_segment(relative_root) + if relative_root + else path_suffix + ) df_list.append( ParsedRow( content=bottom_content, diff --git a/apps/worker/app/services/document_parser/formats/excel/table_parser.py b/apps/worker/app/services/document_parser/formats/excel/table_parser.py index 8dc10101e..4aa4c5e78 100644 --- a/apps/worker/app/services/document_parser/formats/excel/table_parser.py +++ b/apps/worker/app/services/document_parser/formats/excel/table_parser.py @@ -27,6 +27,7 @@ from shared.core.exceptions.domain_exceptions import TableParsingException from shared.core.exceptions.knowhere_exception import KnowhereException +from shared.services.chunks.path_segments import join_document_path from app.services.common.file_loading import load_file_bytes from app.services.common.file_utils import path_handle from shared.utils.text_utils import tokenize2stw_remove @@ -296,11 +297,13 @@ def _build_excel_table_path( ] if subtable_title: parts.append(_clean_path_segment(subtable_title)) - return "/".join(part for part in parts if part) + return join_document_path(parts) def _clean_path_segment(value: str) -> str: - return str(value).strip().replace("/", "_").replace("\\", "_") + # Keep semantic ``/`` for join_document_path escaping; only neutralize + # filesystem-hostile backslashes in sheet/subtable labels. + return str(value).strip().replace("\\", "_") def _summarize_excel_table( diff --git a/apps/worker/app/services/document_parser/formats/image/parser.py b/apps/worker/app/services/document_parser/formats/image/parser.py index dd7fdd09d..52ff4df6d 100755 --- a/apps/worker/app/services/document_parser/formats/image/parser.py +++ b/apps/worker/app/services/document_parser/formats/image/parser.py @@ -87,7 +87,7 @@ def local_image_to_data_url(path, cut=True, min_size=None, max_size=None): if cut: file_size = path.stat().st_size # Bytes. - if file_size < min_size: # Smaller than 10 KB. + if file_size < min_size: logger.debug(f"Skipping {path} (too small: {file_size / 1024:.1f} KB)") return None if file_size >= max_size: # Larger than 5 MB. @@ -232,15 +232,13 @@ def parse_image( img_obj = Image.open(io.BytesIO(img_bytes)) img_obj.save(img_path) - # Early exit: skip images smaller than 10KB + # Early exit: skip images smaller than IMG_MIN_SIZE before VLM work. + from app.services.document_parser.assets.image_size_filter import ( + discard_undersized_image_file, + ) from shared.core.constants import ProcessingConstants - saved_size = os.path.getsize(img_path) - if saved_size < ProcessingConstants.IMG_MIN_SIZE: - logger.debug( - f"Skipping image {filename} (too small: {saved_size / 1024:.1f} KB)" - ) - os.remove(img_path) + if discard_undersized_image_file(img_path, label=f"image {filename}"): return pd.DataFrame(columns=list(PARSER_ROW_COLUMNS)) # Extract image content diff --git a/apps/worker/app/services/document_parser/formats/markdown/deferred_summary.py b/apps/worker/app/services/document_parser/formats/markdown/deferred_summary.py index d946db4f1..06ff1ad06 100644 --- a/apps/worker/app/services/document_parser/formats/markdown/deferred_summary.py +++ b/apps/worker/app/services/document_parser/formats/markdown/deferred_summary.py @@ -83,20 +83,46 @@ def apply_markdown_deferred_summaries( def replace_chunk_ref_in_rows( rows: list[list[str | int]], old_path: str, new_path: str ) -> None: + """Rewrite asset paths after deferred rename. + + Text/image rows store bracketed refs (``[tables/...]`` / ``[images/...]``). + Table rows store the bare relative path as ``content``. Both forms must move + with the renamed file; historically only the bracketed form was updated, + leaving table ``content`` stuck on the pre-rename path. + """ + if not old_path or old_path == new_path: + return + old_ref = build_chunk_ref(old_path) new_ref = build_chunk_ref(new_path) - if not old_ref or old_ref == new_ref: - return for row in rows: if len(row) > 0 and isinstance(row[0], str): - row[0] = row[0].replace(old_ref, new_ref) + updated = row[0] + if old_ref and new_ref and old_ref != new_ref: + updated = updated.replace(old_ref, new_ref) + # Table asset rows use the bare path as content (no brackets). + if updated == old_path: + updated = new_path + elif old_path in updated: + updated = updated.replace(old_path, new_path) + row[0] = updated if len(row) > 1 and row[1] == old_path: row[1] = new_path if len(row) > 2 and isinstance(row[2], str): - row[2] = row[2].replace(old_ref, new_ref) + updated_type = row[2] + if old_ref and new_ref and old_ref != new_ref: + updated_type = updated_type.replace(old_ref, new_ref) + if old_path in updated_type: + updated_type = updated_type.replace(old_path, new_path) + row[2] = updated_type if len(row) > 8 and isinstance(row[8], str): - row[8] = row[8].replace(old_ref, new_ref) + updated_connect = row[8] + if old_ref and new_ref and old_ref != new_ref: + updated_connect = updated_connect.replace(old_ref, new_ref) + if old_path in updated_connect: + updated_connect = updated_connect.replace(old_path, new_path) + row[8] = updated_connect def _run_deferred_summary_tasks( @@ -242,6 +268,11 @@ def _apply_image_summary_result( row = rows[row_index] _apply_asset_result_preserving_index(row, result) + # Table-embedded images keep a stable filename so in + # tables/*.html stays valid after summary generation. + if not original_task.rename_file: + return + img_title = result.title if not img_title: return @@ -285,10 +316,18 @@ def _apply_table_summary_result( table_dir = original_task.table_dir old_table_name = original_task.table_name - table_count = original_task.table_count + # Keep the original table-N index (same pattern as image rename). + table_num_match = re.match(r"table-(\d+)", str(old_table_name)) + table_num = ( + table_num_match.group(1) + if table_num_match + else str(old_table_name).split("-")[1] + if "-" in str(old_table_name) + else "0" + ) safe_title = sanitize_table_name_from_header(str(title)) new_table_name = path_handle( - f"table-{table_count} {safe_title}", mode="clean_single" + f"table-{table_num} {safe_title}", mode="clean_single" ) old_path = os.path.join(table_dir, f"{old_table_name}.html") new_path = os.path.join(table_dir, f"{new_table_name}.html") @@ -296,6 +335,8 @@ def _apply_table_summary_result( return os.rename(old_path, new_path) + old_relative_path = str(row[1]) if len(row) > 1 else f"tables/{old_table_name}.html" new_relative_path = f"tables/{new_table_name}.html" - replace_chunk_ref_in_rows(rows, str(row[1]), new_relative_path) + replace_chunk_ref_in_rows(rows, old_relative_path, new_relative_path) + row[0] = new_relative_path row[1] = new_relative_path diff --git a/apps/worker/app/services/document_parser/formats/markdown/deferred_task.py b/apps/worker/app/services/document_parser/formats/markdown/deferred_task.py index f2cb82409..c0467504e 100644 --- a/apps/worker/app/services/document_parser/formats/markdown/deferred_task.py +++ b/apps/worker/app/services/document_parser/formats/markdown/deferred_task.py @@ -11,6 +11,7 @@ class ImageDeferredSummaryTask: image_dir: str image_name: str image_suffix: str + rename_file: bool = True @dataclass(frozen=True) @@ -19,7 +20,6 @@ class TableDeferredSummaryTask: table_html: str table_dir: str table_name: str - table_count: int @dataclass(frozen=True) diff --git a/apps/worker/app/services/document_parser/formats/markdown/image_asset.py b/apps/worker/app/services/document_parser/formats/markdown/image_asset.py index ed5f51abf..52305689d 100644 --- a/apps/worker/app/services/document_parser/formats/markdown/image_asset.py +++ b/apps/worker/app/services/document_parser/formats/markdown/image_asset.py @@ -7,6 +7,9 @@ from app.services.document_parser.support.identifiers import gen_str_codes from app.services.document_parser.formats.image.parser import perceptual_hash +from app.services.document_parser.assets.image_size_filter import ( + discard_undersized_image_file, +) from app.services.document_parser.assets.inline_asset import build_image_asset_row from app.services.document_parser.formats.markdown.deferred_task import ( ImageDeferredSummaryTask, @@ -27,6 +30,8 @@ class MarkdownImageAsset: cache_entry: dict[str, str] | None deferred_task: MarkdownDeferredSummaryTask | None should_advance_image_count: bool + relative_path: str | None + discarded_undersized: bool = False @dataclass(frozen=True) @@ -42,6 +47,7 @@ class MarkdownImageAssetRequest: seen_images: dict[str, dict[str, str]] summary_image: bool row_index: int + rename_on_summary: bool = True def build_markdown_image_asset( @@ -56,6 +62,13 @@ def build_markdown_image_asset( logger.warning(f"Image file not found, skipping rename: {request.image_path}") return _empty_asset(should_advance_image_count=True) + # Gate before rename / perceptual hash / deferred VLM summary. + if discard_undersized_image_file(source_path, label="markdown image"): + return _empty_asset( + should_advance_image_count=False, + discarded_undersized=True, + ) + with open(source_path, "rb") as image_file: image_binary_hash = perceptual_hash(image_file.read()) @@ -105,6 +118,7 @@ def build_markdown_image_asset( image_dir=request.image_dir, image_name=request.image_name, image_suffix=image_suffix, + rename_file=request.rename_on_summary, ) return MarkdownImageAsset( @@ -114,12 +128,15 @@ def build_markdown_image_asset( cache_entry=cache_entry, deferred_task=deferred_task, should_advance_image_count=True, + relative_path=relative_image_path, ) def build_markdown_image_name(*, image_count: int, last_context: str) -> str: - image_name_context = path_handle(last_context[:10], mode="clean_single") - return f"image-{str(image_count)}-{image_name_context}" + image_name_context = path_handle(last_context.strip(), mode="clean_single") + if image_name_context: + return f"image-{image_count}-{image_name_context}" + return f"image-{image_count}" def resolve_workspace_image_path( @@ -164,9 +181,10 @@ def _build_duplicate_image_asset( cache_entry: dict[str, str], timestamp: str, ) -> MarkdownImageAsset: + relative_path = cache_entry["relative_img_path"] row_values = _build_image_row_values( content=cache_entry["img_content"], - relative_path=cache_entry["relative_img_path"], + relative_path=relative_path, summary=cache_entry["img_summary_field"], know_id=cache_entry["temp_uid"], timestamp=timestamp, @@ -183,6 +201,7 @@ def _build_duplicate_image_asset( cache_entry=None, deferred_task=None, should_advance_image_count=False, + relative_path=relative_path, ) @@ -211,7 +230,11 @@ def _build_image_row_values( return cast(ParserRowValues, image_row.to_list()) -def _empty_asset(*, should_advance_image_count: bool) -> MarkdownImageAsset: +def _empty_asset( + *, + should_advance_image_count: bool, + discarded_undersized: bool = False, +) -> MarkdownImageAsset: return MarkdownImageAsset( content_item=None, row_values=None, @@ -219,4 +242,6 @@ def _empty_asset(*, should_advance_image_count: bool) -> MarkdownImageAsset: cache_entry=None, deferred_task=None, should_advance_image_count=should_advance_image_count, + relative_path=None, + discarded_undersized=discarded_undersized, ) diff --git a/apps/worker/app/services/document_parser/formats/markdown/parse_state.py b/apps/worker/app/services/document_parser/formats/markdown/parse_state.py index 8afab023c..a8d709d38 100644 --- a/apps/worker/app/services/document_parser/formats/markdown/parse_state.py +++ b/apps/worker/app/services/document_parser/formats/markdown/parse_state.py @@ -13,6 +13,7 @@ TextDeferredSummaryTask, ) from app.services.document_parser.support.parser_rows import ParsedRow, ParsedRowsBuilder +from shared.services.chunks.path_segments import escape_path_segment ParserRowValues = list[str | int] @@ -98,11 +99,7 @@ def enter_heading(self, heading: str, level: int) -> None: if item_level < adjusted_level ] - current_heading = ( - heading.replace(self.split_char, "∕") - if self.split_char in heading - else heading - ) + current_heading = escape_path_segment(heading) tentative_names = [item_heading for item_heading, _ in self.path_stack] tentative_names.append(current_heading) tentative_path_parts = [self.relative_root] if self.relative_root else [] diff --git a/apps/worker/app/services/document_parser/formats/markdown/parser.py b/apps/worker/app/services/document_parser/formats/markdown/parser.py index 4f552cbb5..32b5094b1 100755 --- a/apps/worker/app/services/document_parser/formats/markdown/parser.py +++ b/apps/worker/app/services/document_parser/formats/markdown/parser.py @@ -19,6 +19,9 @@ MarkdownTableAssetRequest, build_markdown_table_asset, ) +from app.services.document_parser.formats.markdown.table_embedded_images import ( + extract_table_embedded_images, +) from app.services.document_parser.support.parser_rows import ParsedRow from app.services.document_parser.support.path_helpers import find_matches_parsing from app.services.document_parser.formats.html.parser import ( @@ -417,8 +420,6 @@ def parse_md( if image_asset.should_advance_image_count: parser_state.image_count += 1 - # TODO for large and dense tables, such as "Epstein flight logs", - # integrate tabula-py as an independent extraction path to solve VLM hallucinations and misplacement # b. handle lines containing tables tb_bool, form, _ = identify_tables(line) if tb_bool: @@ -447,14 +448,30 @@ def parse_md( else: continue # Unknown form, skip + embedded = extract_table_embedded_images( + table_html=tb_str, + parser_state=parser_state, + output_dir=output_dir, + image_dir=img_dir, + summary_image=bool(base_llm_paras["summary_image"]), + ) + for image_asset in embedded.image_assets: + if image_asset.row_values is not None: + parser_state.append_row(image_asset.row_values) + if image_asset.deferred_task is not None: + parser_state.schedule_deferred_task(image_asset.deferred_task) + for image_ref in embedded.image_refs: + parser_state.append_content_item(f"\n{image_ref}\n") + table_asset = build_markdown_table_asset( MarkdownTableAssetRequest( - table_html=tb_str, + table_html=embedded.rewritten_html, table_dir=tb_dir, table_count=parser_state.table_count, timestamp=parser_state.timestamp, summary_table=bool(base_llm_paras["summary_table"]), row_index=len(parser_state.rows), + image_refs=embedded.image_refs, ) ) parser_state.append_content_item(table_asset.content_item) diff --git a/apps/worker/app/services/document_parser/formats/markdown/table_asset.py b/apps/worker/app/services/document_parser/formats/markdown/table_asset.py index 89558cb1e..355dc7ecf 100644 --- a/apps/worker/app/services/document_parser/formats/markdown/table_asset.py +++ b/apps/worker/app/services/document_parser/formats/markdown/table_asset.py @@ -34,6 +34,7 @@ class MarkdownTableAssetRequest: timestamp: str summary_table: bool row_index: int + image_refs: list[str] | None = None def build_markdown_table_asset( @@ -60,6 +61,7 @@ def build_markdown_table_asset( keywords="", know_id=gen_str_codes((request.table_html + str(request.table_count))), addtime=request.timestamp, + image_refs=request.image_refs or [], ) deferred_task = None @@ -69,7 +71,6 @@ def build_markdown_table_asset( table_html=request.table_html, table_dir=request.table_dir, table_name=table_name, - table_count=request.table_count - 1, ) return MarkdownTableAsset( diff --git a/apps/worker/app/services/document_parser/formats/markdown/table_embedded_images.py b/apps/worker/app/services/document_parser/formats/markdown/table_embedded_images.py new file mode 100644 index 000000000..ab82543d5 --- /dev/null +++ b/apps/worker/app/services/document_parser/formats/markdown/table_embedded_images.py @@ -0,0 +1,173 @@ +"""Extract and rewrite tags embedded inside MinerU HTML tables.""" + +from __future__ import annotations + +import re +from dataclasses import dataclass + +from bs4 import BeautifulSoup, Tag +from loguru import logger + +from app.services.document_parser.formats.markdown.image_asset import ( + MarkdownImageAsset, + MarkdownImageAssetRequest, + build_markdown_image_asset, + build_markdown_image_name, +) +from app.services.document_parser.formats.markdown.parse_state import MarkdownParseState +from shared.utils.chunk_refs import build_chunk_ref + +_IMG_SRC_RE = re.compile( + r"""]*\bsrc\s*=\s*(?P["'])(?P[^"']+)(?P=quote)""", + re.IGNORECASE, +) +_IMG_TAG_WITH_SRC_RE = re.compile( + r"""]*\bsrc\s*=\s*(?P["'])(?P[^"']+)(?P=quote)[^>]*/?\s*>""", + re.IGNORECASE, +) + + +@dataclass(frozen=True) +class TableEmbeddedImagesResult: + rewritten_html: str + """HTML with rewritten to stable images/image-N-* paths.""" + + image_assets: list[MarkdownImageAsset] + """Newly created image assets that should be registered as rows.""" + + image_refs: list[str] + """Chunk refs ([images/...]) for text content_items and table type channel.""" + + +def extract_table_embedded_images( + *, + table_html: str, + parser_state: MarkdownParseState, + output_dir: str, + image_dir: str, + summary_image: bool, +) -> TableEmbeddedImagesResult: + """Pull assets out of a table, rename them, and rewrite HTML srcs. + + Duplicate perceptual hashes and repeated srcs reuse the first stable path + without creating an extra image row. Missing files leave the original src. + """ + srcs = _unique_img_srcs(table_html) + if not srcs: + return TableEmbeddedImagesResult( + rewritten_html=table_html, + image_assets=[], + image_refs=[], + ) + + naming_context = _first_cell_text(table_html) + rewritten_html = table_html + image_assets: list[MarkdownImageAsset] = [] + image_refs: list[str] = [] + src_to_relative: dict[str, str] = {} + + for src in srcs: + if src in src_to_relative: + continue + + image_name = build_markdown_image_name( + image_count=parser_state.image_count, + last_context=naming_context, + ) + image_asset = build_markdown_image_asset( + MarkdownImageAssetRequest( + output_dir=output_dir, + image_dir=image_dir, + image_path=src, + image_name=image_name, + image_count=parser_state.image_count, + last_context=naming_context, + image_summary=naming_context or None, + timestamp=parser_state.timestamp, + seen_images=parser_state.seen_images, + summary_image=summary_image, + row_index=len(parser_state.rows) + len(image_assets), + rename_on_summary=False, + ) + ) + if image_asset.discarded_undersized: + rewritten_html = _strip_img_tags(rewritten_html, src) + continue + if image_asset.relative_path is None: + logger.warning( + f"Table-embedded image not found, leaving original src: {src}" + ) + continue + + src_to_relative[src] = image_asset.relative_path + image_refs.append(build_chunk_ref(image_asset.relative_path)) + + if image_asset.should_advance_image_count: + image_assets.append(image_asset) + # Advance immediately so the next distinct src gets a new image-N. + parser_state.image_count += 1 + if ( + image_asset.cache_key is not None + and image_asset.cache_entry is not None + ): + parser_state.seen_images[image_asset.cache_key] = ( + image_asset.cache_entry + ) + + for old_src, new_relative in src_to_relative.items(): + rewritten_html = _rewrite_img_src(rewritten_html, old_src, new_relative) + + return TableEmbeddedImagesResult( + rewritten_html=rewritten_html, + image_assets=image_assets, + image_refs=image_refs, + ) + + +def _unique_img_srcs(table_html: str) -> list[str]: + seen: set[str] = set() + ordered: list[str] = [] + for match in _IMG_SRC_RE.finditer(table_html): + src = match.group("src").strip() + if not src or src in seen: + continue + seen.add(src) + ordered.append(src) + return ordered + + +def _first_cell_text(table_html: str) -> str: + soup = BeautifulSoup(table_html, "html.parser") + table = soup.find("table") + if not isinstance(table, Tag): + return "" + for cell in table.find_all(["td", "th"]): + if not isinstance(cell, Tag): + continue + text = cell.get_text(strip=True) + if text: + return text + return "" + + +def _rewrite_img_src(html: str, old_src: str, new_src: str) -> str: + pattern = re.compile( + r"""(]*\bsrc\s*=\s*)(["'])""" + + re.escape(old_src) + + r"""\2""", + re.IGNORECASE, + ) + + def _replace(match: re.Match[str]) -> str: + return f"{match.group(1)}{match.group(2)}{new_src}{match.group(2)}" + + return pattern.sub(_replace, html) + + +def _strip_img_tags(html: str, src: str) -> str: + def _drop(match: re.Match[str]) -> str: + if match.group("src") == src: + return "" + return match.group(0) + + return _IMG_TAG_WITH_SRC_RE.sub(_drop, html) diff --git a/apps/worker/app/services/document_parser/formats/text/parser.py b/apps/worker/app/services/document_parser/formats/text/parser.py index 70ac5ce6b..66ef843c4 100755 --- a/apps/worker/app/services/document_parser/formats/text/parser.py +++ b/apps/worker/app/services/document_parser/formats/text/parser.py @@ -8,6 +8,11 @@ from loguru import logger from shared.core.config import settings +from shared.services.chunks.path_segments import ( + append_document_path, + join_document_path, + split_escaped_document_path, +) from shared.utils.chunk_refs import CHUNK_REF_PATTERN from app.services.common.file_loading import load_file_bytes @@ -102,9 +107,8 @@ def postprocess_leaf_dics( summary_len = ProcessingConstants.POSTPROCESS_SUMMARY_LEN merged_dict = {} - split_char = settings.SPLIT_CHAR or "/" for identifier, d in dict_list: - identifier = split_char.join(identifier) + identifier = join_document_path(identifier) if identifier in merged_dict: merged_dict[identifier][content_key].extend(d[content_key]) @@ -116,7 +120,7 @@ def postprocess_leaf_dics( merged_list = [(identifier, v["content"]) for identifier, v in merged_dict.items()] merge_df = pd.DataFrame(merged_list, columns=["path_identifier", "content_lst"]) - merge_df["path"] = merge_df["path_identifier"].apply(lambda x: x.split(split_char)) + merge_df["path"] = merge_df["path_identifier"].apply(split_escaped_document_path) merge_df = merge_df[["path", "content_lst", "path_identifier"]] # TODO rough dividing of contents (need more smart dividing) @@ -139,16 +143,13 @@ def postprocess_leaf_dics( head = row["path_identifier"] if not head: head = "**Preface**" + head_parts = split_escaped_document_path(head) + leaf_title = head_parts[-1] if head_parts else head for k in range(num): - sub_head = ( - head - + split_char - + head.split(split_char)[-1] - + " part " - + str(k + 1) - ) + part_title = f"{leaf_title} part {k + 1}" + sub_head = append_document_path(head, part_title) df_with_divides.loc[len(df_with_divides)] = { - "path": sub_head.split(split_char), + "path": split_escaped_document_path(sub_head), "content_lst": sublists[k], "path_identifier": sub_head, } diff --git a/apps/worker/app/services/document_parser/orchestration/format_adapters.py b/apps/worker/app/services/document_parser/orchestration/format_adapters.py index 7b3446ec5..5a26a45ba 100644 --- a/apps/worker/app/services/document_parser/orchestration/format_adapters.py +++ b/apps/worker/app/services/document_parser/orchestration/format_adapters.py @@ -138,6 +138,10 @@ def parse(self, session: ParseSession) -> ParseOutput: @dataclass(frozen=True) class PptxParseAdapter: # Deprecated: prefer page_memory track for PPTX (parse_track="page_memory"). + # TODO(pptx): convert PPTX→PDF first, then reuse the standard PDF PROFILE + # path instead of a separate PPTX PROFILE; keep page_memory input schema + # stable (fields may expand later). page_memory/normalizer.py already + # converts before PROFILE. document_format: object def parse(self, session: ParseSession) -> ParseOutput: diff --git a/apps/worker/app/services/document_parser/orchestration/office_container_validator.py b/apps/worker/app/services/document_parser/orchestration/office_container_validator.py new file mode 100644 index 000000000..3009603c1 --- /dev/null +++ b/apps/worker/app/services/document_parser/orchestration/office_container_validator.py @@ -0,0 +1,98 @@ +from __future__ import annotations + +import zipfile +from dataclasses import dataclass + +from app.services.document_parser.orchestration.format_router import DocumentFormat + +from shared.core.exceptions.domain_exceptions import ValidationException + + +@dataclass(frozen=True) +class _OfficeContainerRequirement: + extension: str + document_label: str + required_member: str + + @property + def user_message(self) -> str: + return ( + f"Invalid file: the uploaded {self.extension} file is not a valid " + f"{self.document_label}. Please check the file and upload again." + ) + + @property + def violation_description(self) -> str: + return ( + f"Expected a valid {self.extension[1:].upper()} ZIP package containing " + f"{self.required_member}" + ) + + +_CONTENT_TYPES_MEMBER: str = "[Content_Types].xml" +_OFFICE_CONTAINER_REQUIREMENTS: dict[ + DocumentFormat, + _OfficeContainerRequirement, +] = { + DocumentFormat.DOCX: _OfficeContainerRequirement( + extension=".docx", + document_label="Word document", + required_member="word/document.xml", + ), + DocumentFormat.XLSX: _OfficeContainerRequirement( + extension=".xlsx", + document_label="Excel workbook", + required_member="xl/workbook.xml", + ), + DocumentFormat.PPTX: _OfficeContainerRequirement( + extension=".pptx", + document_label="PowerPoint presentation", + required_member="ppt/presentation.xml", + ), +} + + +def validate_office_container( + file_path: str, + document_format: DocumentFormat, +) -> None: + """Validate OOXML containers before dispatching to archive-based parsers.""" + requirement = _OFFICE_CONTAINER_REQUIREMENTS.get(document_format) + if requirement is None: + return + + if not zipfile.is_zipfile(file_path): + _raise_invalid_office_file(requirement) + + try: + with zipfile.ZipFile(file_path, "r") as archive: + member_names = set(archive.namelist()) + except zipfile.BadZipFile as exc: + raise _build_invalid_office_file_exception(requirement) from exc + + if ( + _CONTENT_TYPES_MEMBER not in member_names + or requirement.required_member not in member_names + ): + _raise_invalid_office_file(requirement) + + +def _raise_invalid_office_file(requirement: _OfficeContainerRequirement) -> None: + raise _build_invalid_office_file_exception(requirement) + + +def _build_invalid_office_file_exception( + requirement: _OfficeContainerRequirement, +) -> ValidationException: + return ValidationException( + user_message=requirement.user_message, + violations=[ + { + "field": "file", + "description": requirement.violation_description, + } + ], + ) + + +__all__ = ["validate_office_container"] diff --git a/apps/worker/app/services/document_parser/orchestration/postprocess.py b/apps/worker/app/services/document_parser/orchestration/postprocess.py index 0fbb9f5c8..28f30a966 100644 --- a/apps/worker/app/services/document_parser/orchestration/postprocess.py +++ b/apps/worker/app/services/document_parser/orchestration/postprocess.py @@ -11,6 +11,15 @@ from app.services.document_parser.support.stage_profiler import stage_timer from loguru import logger +_IMG_SRC_BASENAME_RE = re.compile( + r"""]*\bsrc\s*=\s*["']([^"']+)["']""", + re.IGNORECASE, +) +_HASH_IMAGE_RE = re.compile( + r"^[a-f0-9]{64}\.(?:jpg|jpeg|png|gif|webp)$", + re.IGNORECASE, +) + def apply_parse_postprocess( output_dir: str, @@ -39,19 +48,23 @@ def apply_parse_postprocess( def cleanup_unreferenced_images(output_dir: str) -> int: - """Remove UUID-named images that are not referenced by final parsed output.""" + """Remove hash-named MinerU images that are not referenced by table HTML. + + Images extracted as ``image-N-*`` are never matched by the hash pattern and + are always kept. Hash-named files still referenced from ``tables/*.html`` + (e.g. failed extraction) are also preserved. + """ image_dir = os.path.join(output_dir, "images") if not os.path.isdir(image_dir): return 0 - uuid_pattern = re.compile( - r"^[a-f0-9]{64}\.(?:jpg|jpeg|png|gif|webp)$", - re.IGNORECASE, - ) + protected_basenames = _collect_table_img_basenames(output_dir) removed_count = 0 for filename in os.listdir(image_dir): - if not uuid_pattern.match(filename): + if not _HASH_IMAGE_RE.match(filename): + continue + if filename in protected_basenames: continue file_path = os.path.join(image_dir, filename) @@ -68,3 +81,26 @@ def cleanup_unreferenced_images(output_dir: str) -> int: ) return removed_count + + +def _collect_table_img_basenames(output_dir: str) -> set[str]: + tables_dir = os.path.join(output_dir, "tables") + if not os.path.isdir(tables_dir): + return set() + + basenames: set[str] = set() + for filename in os.listdir(tables_dir): + if not filename.endswith(".html"): + continue + table_path = os.path.join(tables_dir, filename) + try: + with open(table_path, encoding="utf-8") as table_file: + html = table_file.read() + except OSError as exc: + logger.warning(f"Failed to read table HTML {table_path}: {exc}") + continue + for match in _IMG_SRC_BASENAME_RE.finditer(html): + src = match.group(1).strip() + if src: + basenames.add(os.path.basename(src)) + return basenames diff --git a/apps/worker/app/services/document_parser/orchestration/route_parse.py b/apps/worker/app/services/document_parser/orchestration/route_parse.py index 68bb57e73..1c05fd442 100644 --- a/apps/worker/app/services/document_parser/orchestration/route_parse.py +++ b/apps/worker/app/services/document_parser/orchestration/route_parse.py @@ -2,6 +2,9 @@ get_document_parse_adapter, resolve_document_format, ) +from app.services.document_parser.orchestration.office_container_validator import ( + validate_office_container, +) from app.services.document_parser.orchestration.parse_output import ParseOutput from app.services.document_parser.orchestration.parse_session import ParseSession @@ -9,5 +12,6 @@ def route_document_parse(session: ParseSession) -> ParseOutput: """Route a parser session to the correct adapter and return its output.""" document_format = resolve_document_format(session.file_full_path) + validate_office_container(session.file_full_path, document_format) adapter = get_document_parse_adapter(document_format) return adapter.parse(session) diff --git a/apps/worker/app/services/document_parser/profiling/doc_profiler.py b/apps/worker/app/services/document_parser/profiling/doc_profiler.py index 1dbf6dab6..a0a3077bd 100644 --- a/apps/worker/app/services/document_parser/profiling/doc_profiler.py +++ b/apps/worker/app/services/document_parser/profiling/doc_profiler.py @@ -42,13 +42,14 @@ def profile_document( filename: File name (used to infer type) job_id: Parse job id for profile trace artifacts output_dir: Parser output directory - skip_shard_plan: When True, the lightweight anatomy stage skips the - LLM shard decision (+ H2 refinement) and populates a single-shard - placeholder instead. Used by the page-memory track, which never - consumes the shard plan. Chunk-track keeps the default (False). + skip_shard_plan: When True, lightweight and structural anatomy skip + LLM/ReAct shard planning and populate a single-shard placeholder. + Used by the page-memory track, which never consumes the shard plan. + Chunk-track keeps the default (False) so oversized MinerU sharding + still receives a real plan. oversized_policy: Controls oversized PDF admission. ``chunk`` applies - the legacy MinerU shard gate, while ``page_memory`` lets the - page-memory track continue to structural profiling. + the MinerU shard gate, while ``page_memory`` lets the page-memory + track continue to structural profiling. Returns: ParserDocumentProfile @@ -112,7 +113,13 @@ def _profile_pdf_with_db( ) -> ParserDocumentProfile: profile_job_id = job_id or filename agent_output_dir = os.path.join(output_dir, "_doc_agent") if output_dir else None - page_toc_enabled = settings.PDF_PROFILE_TOC_ENABLED + # Page-memory sections are anchored on the TOC (page-based VLM TOC pipeline), + # so TOC profiling is mandatory for that track regardless of the global + # PDF_PROFILE_TOC_ENABLED flag (which only gates the optional chunk-track + # TOC profiling that can otherwise fall back to MinerU markdown headings). + page_toc_enabled = ( + oversized_policy == "page_memory" or settings.PDF_PROFILE_TOC_ENABLED + ) coordinator = ProfileCoordinator( pdf_path=file_path, job_id=profile_job_id, @@ -152,7 +159,9 @@ def _profile_pdf_with_db( raise_if_oversized_pdf_not_supported(page_count=profile.page_count) if not profile.is_atlas: try: - profile.anatomy = coordinator.run_structural() + profile.anatomy = coordinator.run_structural( + skip_shard_plan=skip_shard_plan + ) profile.toc = _map_toc_profile(coordinator) except Exception as exc: if oversized_policy == "page_memory": @@ -165,8 +174,12 @@ def _profile_pdf_with_db( original_exception=exc, ) from exc else: + # TODO(page_memory): oversized atlas skips anatomy, so coarse + # has_asset / page_features never reach page_memory (Root fallback). profile.toc = _map_toc_profile(coordinator) else: + # TODO(page_memory): non-oversized atlas skips anatomy; coarse + # has_asset / page_features never reach page_memory via profile.anatomy. if not profile.is_atlas: profile.anatomy = coordinator.run_lightweight_anatomy( skip_shard_plan=skip_shard_plan diff --git a/apps/worker/app/services/document_parser/structure/heading_tree.py b/apps/worker/app/services/document_parser/structure/heading_tree.py index dcad6024d..05f5919c7 100644 --- a/apps/worker/app/services/document_parser/structure/heading_tree.py +++ b/apps/worker/app/services/document_parser/structure/heading_tree.py @@ -3,6 +3,8 @@ import pandas as pd from loguru import logger +from shared.services.chunks.path_segments import append_document_path + def build_tree_from_dataframe( heading_preds: pd.DataFrame, @@ -33,9 +35,7 @@ def build_tree_from_dataframe( node_to_id[node_key] = row_id parent_dict[tree_node_key] = {} - current_path = ( - f"{parent_path}/{tree_node_key}" if parent_path else tree_node_key - ) + current_path = append_document_path(parent_path, tree_node_key) stack.append((level, parent_dict[tree_node_key], tree_node_key, current_path)) return root, node_to_id, id_to_row @@ -114,9 +114,7 @@ def _extract_headings_from_tree( ) if isinstance(children, dict) and children: - current_path = ( - f"{parent_path}/{tree_node_key}" if parent_path else tree_node_key - ) + current_path = append_document_path(parent_path, tree_node_key) results.extend( _extract_headings_from_tree( children, @@ -150,12 +148,12 @@ def _remove_isolated_nodes_recursive( else: result_dict[heading] = _remove_isolated_nodes_recursive( children, - parent_path=f"{parent_path}/{heading}" if parent_path else heading, + parent_path=append_document_path(parent_path, heading), ) elif isinstance(children, dict) and children: result_dict[heading] = _remove_isolated_nodes_recursive( children, - parent_path=f"{parent_path}/{heading}" if parent_path else heading, + parent_path=append_document_path(parent_path, heading), ) else: result_dict[heading] = children diff --git a/apps/worker/app/services/document_parser/support/path_helpers.py b/apps/worker/app/services/document_parser/support/path_helpers.py index c94cf3101..ad3a89612 100644 --- a/apps/worker/app/services/document_parser/support/path_helpers.py +++ b/apps/worker/app/services/document_parser/support/path_helpers.py @@ -5,6 +5,7 @@ from typing import Any from shared.utils.chunk_refs import extract_chunk_refs +from shared.services.chunks.path_segments import join_document_path from app.services.common.file_utils import path_handle SUMMARY_PATH_MARKERS: tuple[str, ...] = ("summary", "\u6458\u8981\u603b\u7ed3") @@ -56,8 +57,7 @@ def flatten_dic2paths( if isinstance(value, dict) and value: flatten_dic2paths(value, new_path, result) else: - split_char = os.getenv("SPLIT_CHAR", "/") - result.append(split_char.join(new_path)) + result.append(join_document_path(new_path)) return result diff --git a/apps/worker/app/services/page_memory/_serialization.py b/apps/worker/app/services/page_memory/_serialization.py index b09e34202..87caea104 100644 --- a/apps/worker/app/services/page_memory/_serialization.py +++ b/apps/worker/app/services/page_memory/_serialization.py @@ -9,8 +9,8 @@ from app.services.page_memory._utils import ( collapse_page_ranges, page_scope_info, - sort_skeletons, ) +from shared.services.chunks.path_segments import split_escaped_document_path def write_json(path: Path, value: Any) -> None: @@ -92,9 +92,17 @@ def serialize_skeletons(skeletons: list[Any]) -> list[dict[str, Any]]: def build_hierarchy_tree(skeletons: list[Any]) -> dict[str, Any]: + """Build a nested title tree preserving ``skeletons`` list order. + + Sibling key order follows first-seen order in ``skeletons`` (dict + insertion order). Callers that need cross-page ordering should + ``sort_skeletons`` first; same-page order must remain VLM/TOC order. + """ hierarchy: dict[str, Any] = {} - for skel in sort_skeletons(skeletons): - parts = str(getattr(skel, "section_path", "") or "").split("/") + for skel in skeletons: + parts = split_escaped_document_path( + getattr(skel, "section_path", "") or "" + ) section_parts = parts[1:] if len(parts) > 1 else parts current = hierarchy for part in section_parts: @@ -109,6 +117,7 @@ def serialize_hierarchy_artifact( *, scope_manifest_data: dict[str, Any] | None = None, ) -> dict[str, Any]: + # Keep nodes and HIERARCHY on the same list order — do not re-sort here. nodes = serialize_skeletons(skeletons) artifact: dict[str, Any] = { "HIERARCHY": build_hierarchy_tree(skeletons), @@ -140,6 +149,7 @@ def serialize_page_tags(tags: list[Any]) -> list[dict[str, Any]]: "keywords": list(item.keywords), "entities": list(getattr(item, "entities", []) or []), "strategy_used": item.strategy_used, + "observed_titles": list(getattr(item, "observed_titles", []) or []), } for item in tags ] @@ -169,23 +179,68 @@ def serialize_assets(assets_by_page: dict[int, list[Any]]) -> list[dict[str, Any return rows +def serialize_scope_skeletons( + *, + scope_id: str, + start_page: int, + end_page: int, + strategy: str, + skeletons: list[Any], +) -> dict[str, Any]: + """Coarse scope input artifact (Stage3 → Stage4 handoff). + + Closed-closed ``start_page``/``end_page`` plus coarse skeleton rows + (including ``evidence``). Downstream stages read this file; refined + hierarchy lives in ``fine_hierarchy.json``. + """ + start = max(1, int(start_page)) + end = max(start, int(end_page)) + rows = [ + { + "section_path": getattr(item, "section_path", ""), + "title": getattr(item, "title", ""), + "level": int(getattr(item, "level", 0) or 0), + "start_page": int(getattr(item, "start_page", 0) or 0), + "end_page": int(getattr(item, "end_page", 0) or 0), + "parent_path": getattr(item, "parent_path", None), + "evidence": dict(getattr(item, "evidence", {}) or {}), + } + for item in skeletons + ] + return { + "scope_id": scope_id, + "start_page": start, + "end_page": end, + "page_count": end - start + 1, + "strategy": strategy, + "skeleton_count": len(rows), + "skeletons": rows, + } + + def write_scope_artifacts( *, output_dir: str, scope_id: str, scope_manifest_data: dict[str, Any], hierarchy: list[Any], - tags: list[Any], + tags: list[Any] | None = None, assets_by_page: dict[int, list[Any]] | None = None, ) -> None: + """Write per-scope viewing artifacts. + + Always refreshes ``fine_hierarchy.json`` (embeds ``scope`` manifest). + ``tags`` / ``assets_by_page`` of ``None`` leave the existing file untouched + so later stages do not wipe earlier placeholders or results. + """ scope_dir = Path(output_dir) / "scopes" / scope_id - write_json(scope_dir / "scope.json", scope_manifest_data) write_json( scope_dir / "fine_hierarchy.json", serialize_hierarchy_artifact(hierarchy, scope_manifest_data=scope_manifest_data), ) - write_json(scope_dir / "page_tags.json", serialize_page_tags(tags)) - if assets_by_page: + if tags is not None: + write_json(scope_dir / "page_tags.json", serialize_page_tags(tags)) + if assets_by_page is not None: write_json(scope_dir / "assets.json", serialize_assets(assets_by_page)) @@ -199,7 +254,7 @@ def write_top_level_artifacts( root = Path(output_dir) write_json(root / "hierarchy.json", serialize_hierarchy_artifact(hierarchy)) write_json(root / "page_tags.json", serialize_page_tags(tags)) - if assets_by_page: + if assets_by_page is not None: write_json(root / "assets.json", serialize_assets(assets_by_page)) else: (root / "assets.json").unlink(missing_ok=True) diff --git a/apps/worker/app/services/page_memory/_utils.py b/apps/worker/app/services/page_memory/_utils.py index 219889bfc..23d89f7f9 100644 --- a/apps/worker/app/services/page_memory/_utils.py +++ b/apps/worker/app/services/page_memory/_utils.py @@ -2,26 +2,22 @@ from __future__ import annotations -import re from dataclasses import dataclass, replace from typing import Any -_NATURAL_RE = re.compile(r"(\d+)") - - -def _natural_key(path: str) -> list[int | str]: - """Split a string into a list of int/str segments for natural ordering.""" - return [int(c) if c.isdigit() else c.lower() for c in _NATURAL_RE.split(path)] - def sort_skeletons(skeletons: list[Any]) -> list[Any]: + """Order skeletons by start page only. + + Same-page relative order is preserved (Python ``sorted`` is stable). That + relative order is authoritative: coarse TOC emit order, or VLM observed + title order from fine hierarchy. Do **not** tie-break by ``section_path`` / + title — alphabetical path order reverses real reading order (e.g. Open + access before Separation on the same page). + """ return sorted( skeletons, - key=lambda item: ( - int(getattr(item, "start_page", 0) or 0), - int(getattr(item, "level", 0) or 0), - _natural_key(str(getattr(item, "section_path", "") or "")), - ), + key=lambda item: int(getattr(item, "start_page", 0) or 0), ) diff --git a/apps/worker/app/services/page_memory/fine_hierarchy.py b/apps/worker/app/services/page_memory/fine_hierarchy.py index 0e4466824..96fd18de4 100644 --- a/apps/worker/app/services/page_memory/fine_hierarchy.py +++ b/apps/worker/app/services/page_memory/fine_hierarchy.py @@ -5,6 +5,11 @@ row semantics are designed for raw text. This module calls a dedicated ``page-memory-hierarchy`` prompt and builds deeper ``SectionSkeleton`` entries under each coarse TOC leaf. + +Boundary trimming is code-side: VLM extracts all outline titles in reading +order; ``_collect_candidates`` drops titles at/before the coarse start anchor +and at/after the next-skeleton end anchor. If the end anchor is missing, +candidates on the scan boundary page are dropped (page-level fallback). """ from __future__ import annotations @@ -18,10 +23,11 @@ from app.services.page_memory.page_tagger import PageTagResult from app.services.page_memory.skeleton_extractor import SectionSkeleton -from app.services.page_memory._utils import page_scope_info +from app.services.page_memory._utils import page_scope_info, sort_skeletons from shared.services.ai.llm_overrides import get_text_client from shared.services.ai.prompt_service import build_prompt from shared.services.ai.response_process_service import eval_response +from shared.services.chunks.path_segments import append_document_path def refine_fat_leaf_skeletons( @@ -29,6 +35,7 @@ def refine_fat_leaf_skeletons( coarse_skeletons: list[SectionSkeleton], tag_results: list[PageTagResult], fat_leaf_pages: set[int], + next_title_by_path: dict[str, str | None] | None = None, model_name: str | None = None, max_tokens: int = 2000, max_depth: int = 6, @@ -37,8 +44,8 @@ def refine_fat_leaf_skeletons( """Refine coarse TOC leaf skeletons using VLM-observed title candidates. For each coarse skeleton that overlaps fat-leaf pages: - 1. Collect real ``observed_titles`` from that leaf using exclusive end - boundaries. + 1. Collect ``observed_titles`` in reading order and trim by start/end + coarse anchors (tail miss → drop the boundary page). 2. Ask the page-memory hierarchy prompt to assign relative levels. 3. Rebuild the nested tree and graft it under the coarse leaf. @@ -47,6 +54,7 @@ def refine_fat_leaf_skeletons( if not fat_leaf_pages: return coarse_skeletons + next_map = next_title_by_path or {} refined: list[SectionSkeleton] = [] for index, skeleton in enumerate(coarse_skeletons): @@ -58,19 +66,20 @@ def refine_fat_leaf_skeletons( refined.append(skeleton) continue - # Collect observed titles from tagged pages within this skeleton + end_title = next_map.get(skeleton.section_path) candidates = _collect_candidates( skeleton=skeleton, exclusive_end=exclusive_end, tag_results=tag_results, fat_leaf_pages=fat_leaf_pages, + start_title=skeleton.title, + end_title=end_title, ) if not candidates: refined.append(skeleton) continue - # Run hierarchy LLM and graft the resulting tree under the coarse leaf deeper = _run_hierarchy_on_candidates( candidates=candidates, skeleton=skeleton, @@ -103,10 +112,9 @@ def compute_fat_leaf_pages( A "fat leaf" is a coarse skeleton (TOC leaf node) whose page range exceeds ``min_pages``. Only these pages get VLM title detection. - Uses exclusive end boundaries: each skeleton's scan range ends at - ``next_skeleton.start_page - 1`` to avoid scanning pages that belong - to the next sibling skeleton (the raw ``end_page`` from the hierarchy - locator uses closed-closed intervals that can overlap at boundaries). + Uses exclusive end boundaries when sibling starts are visible in + ``skeletons``. For single-leaf scopes the closed ``end_page`` is used, + which includes the shared boundary page with the next leaf. """ fat_pages: set[int] = set() for idx, skel in enumerate(skeletons): @@ -117,6 +125,28 @@ def compute_fat_leaf_pages( return fat_pages +def build_next_title_by_path( + skeletons: list[SectionSkeleton], +) -> dict[str, str | None]: + """Map each skeleton ``section_path`` to the next title in reading order. + + Order matches ``sort_skeletons``: by ``start_page`` only, stable so + same-page relative order (TOC emit / parent-before-child) is preserved. + The next title is the immediately following skeleton — including a + same-page sibling or a ``parent_self_only`` node's first child — used as + the tail trim anchor in ``_trim_by_coarse_anchors``. + """ + ordered = sort_skeletons(skeletons) + next_map: dict[str, str | None] = {} + for index, skeleton in enumerate(ordered): + next_title: str | None = None + if index + 1 < len(ordered): + title = str(getattr(ordered[index + 1], "title", "") or "").strip() + next_title = title or None + next_map[str(skeleton.section_path)] = next_title + return next_map + + # ── Internal helpers ───────────────────────────────────────────────── @@ -126,17 +156,24 @@ def _collect_candidates( exclusive_end: int, tag_results: list[PageTagResult], fat_leaf_pages: set[int], + start_title: str, + end_title: str | None, ) -> list[dict[str, Any]]: """Gather VLM-observed titles within a skeleton's page range. - Returns list of ``{id, heading, page, prominence}`` sorted by page then - prominence (strongest first within page). + Preserves page ascending order and within-page VLM reading order + (``observed_titles`` as returned by the VLM; prominence is metadata only). + Then trims by coarse start/end anchors: + + - Drop the start anchor and everything before it. + - Drop the end anchor (next skeleton title) and everything after it. + - If the end anchor is missing, drop all candidates on ``exclusive_end`` + (page-level fallback so the next section cannot bleed in). + + Returns list of ``{id, heading, page, prominence}``. """ - candidates: list[dict[str, Any]] = [] + raw: list[dict[str, Any]] = [] tag_by_page = {t.page_index: t for t in tag_results} - candidate_id = 0 - seen: set[str] = set() - parent_key = _title_key(skeleton.title) for page in range(skeleton.start_page, exclusive_end + 1): if page not in fat_leaf_pages: @@ -144,31 +181,101 @@ def _collect_candidates( tag = tag_by_page.get(page) if tag is None or not tag.observed_titles: continue - - # Sort by prominence descending within page - sorted_titles = sorted( - tag.observed_titles, - key=lambda t: -(t.get("prominence") or 0.5), - ) - for title_entry in sorted_titles: - text = title_entry.get("text", "").strip() + for title_entry in tag.observed_titles: + text = str(title_entry.get("text", "") or "").strip() if not text or len(text) < 2: continue key = _title_key(text) - if not key or key == parent_key or key in seen: + if not key: continue - seen.add(key) - candidate_id += 1 - candidates.append({ - "id": candidate_id, + raw.append({ "heading": text, "page": page, "prominence": title_entry.get("prominence"), + "key": key, }) + trimmed = _trim_by_coarse_anchors( + raw, + start_title=start_title, + end_title=end_title, + section_path=skeleton.section_path, + boundary_page=exclusive_end, + ) + + candidates: list[dict[str, Any]] = [] + seen: set[str] = set() + candidate_id = 0 + for item in trimmed: + key = str(item["key"]) + if key in seen: + continue + seen.add(key) + candidate_id += 1 + candidates.append({ + "id": candidate_id, + "heading": item["heading"], + "page": item["page"], + "prominence": item.get("prominence"), + }) return candidates +def _trim_by_coarse_anchors( + raw: list[dict[str, Any]], + *, + start_title: str, + end_title: str | None, + section_path: str, + boundary_page: int, +) -> list[dict[str, Any]]: + """Hard-trim reading-order titles by start/end coarse anchors. + + Tail miss policy: if ``end_title`` is set but not found among candidates, + drop every entry on ``boundary_page`` (the scope's last scanned page). + Head miss stays lenient — only drop exact parent-title duplicates. + """ + start_key = _title_key(start_title) + end_key = _title_key(end_title) if end_title else "" + + trimmed = list(raw) + if start_key: + start_idx = next( + (i for i, item in enumerate(trimmed) if item.get("key") == start_key), + None, + ) + if start_idx is None: + logger.warning( + "[page_memory.fine_hierarchy] start anchor not found for {}; " + "keeping all candidates before end trim", + section_path, + ) + trimmed = [item for item in trimmed if item.get("key") != start_key] + else: + trimmed = trimmed[start_idx + 1 :] + + if end_key: + end_idx = next( + (i for i, item in enumerate(trimmed) if item.get("key") == end_key), + None, + ) + if end_idx is None: + logger.warning( + "[page_memory.fine_hierarchy] end anchor {!r} not found for {}; " + "dropping all candidates on boundary page {}", + end_title, + section_path, + boundary_page, + ) + trimmed = [ + item for item in trimmed if int(item.get("page") or 0) != boundary_page + ] + else: + trimmed = trimmed[:end_idx] + + return trimmed + + def _run_hierarchy_on_candidates( *, candidates: list[dict[str, Any]], @@ -285,7 +392,7 @@ def _run_hierarchy_on_candidates( while stack and stack[-1][0] >= rel_level: stack.pop() parent_path = stack[-1][1] if stack else skeleton.section_path - section_path = f"{parent_path}/{heading}" + section_path = append_document_path(parent_path, heading) stack.append((rel_level, section_path)) nodes.append({ "section_path": section_path, @@ -330,6 +437,18 @@ def _run_hierarchy_on_candidates( def _exclusive_end(skeletons: list[SectionSkeleton], index: int) -> int: + """Last page to *scan* for title candidates under ``skeletons[index]``. + + When later starts are visible in ``skeletons``, the span stops before the + next start (``next.start - 1``). For a single-leaf scope list the closed + ``end_page`` is returned, which may be the shared boundary page with the + next coarse leaf — that page is scanned here; candidate ownership on it is + decided later by ``_trim_by_coarse_anchors`` via the global next-title + anchor (not by this helper). + + Distinct from ``node_assembler._exclusive_end``, which decides body-text + ownership / summary slices at assembly time. + """ skeleton = skeletons[index] later_starts = [ skel.start_page @@ -341,7 +460,7 @@ def _exclusive_end(skeletons: list[SectionSkeleton], index: int) -> int: return min(skeleton.end_page, min(later_starts) - 1) -def _title_key(title: str) -> str: +def _title_key(title: str | None) -> str: normalized = re.sub(r"\s+", "", str(title or "")).casefold() normalized = re.sub(r"[^\w\u4e00-\u9fff]+", "", normalized) return normalized diff --git a/apps/worker/app/services/page_memory/memory_service.py b/apps/worker/app/services/page_memory/memory_service.py index f34cc42c0..0c09005ba 100644 --- a/apps/worker/app/services/page_memory/memory_service.py +++ b/apps/worker/app/services/page_memory/memory_service.py @@ -220,8 +220,8 @@ def _build_page_dataframe( C1 page_renderer → PageRenderResult[] C2 page_plan → PagePlan[] C3 page_tagger → PageTagResult[] - C3b title detection → observed_titles - C4b fine_hierarchy → refined SectionSkeleton[] + C3b title detection → observed_titles (reading-order, untrimmed) + C4b fine_hierarchy → start/end anchor trim + refined SectionSkeleton[] C5 page_assets → assets anchored to refined hierarchy pages C7 assemble node-granularity DataFrame """ @@ -230,6 +230,7 @@ def _build_page_dataframe( collapse_single_child_chains, extract_section_skeletons, ) + from shared.services.chunks.path_segments import join_document_path anatomy = getattr(profile, "anatomy", None) page_count = max(int(profile.page_count or 0), 0) if page_count <= 0: @@ -269,7 +270,7 @@ def _build_page_dataframe( if not skeletons: skeletons = [ SectionSkeleton( - section_path=f"{filename}/Root", + section_path=join_document_path([filename, "Root"]), level=1, start_page=1, end_page=page_count, @@ -279,6 +280,9 @@ def _build_page_dataframe( ) ] + from app.services.page_memory.fine_hierarchy import build_next_title_by_path + + next_title_by_path = build_next_title_by_path(skeletons) coarse_scopes = _build_hierarchy_scopes( skeletons=skeletons, filename=filename, @@ -340,6 +344,7 @@ def _build_page_dataframe( asset_extraction_enabled=asset_extraction_enabled, trace_recorder=trace_recorder, page_memory_config=page_memory_config, + next_title_by_path=next_title_by_path, ) if scope_concurrency <= 1 or len(coarse_scopes) <= 1: @@ -579,6 +584,7 @@ def _run_hierarchy_scope( asset_max_pages: int, trace_recorder: Any | None, page_memory_config: PageMemoryConfig, + next_title_by_path: dict[str, str | None] | None = None, ) -> _ScopeRunResult: from app.services.page_memory.fine_hierarchy import ( compute_fat_leaf_pages, @@ -651,6 +657,7 @@ def _run_hierarchy_scope( fat_leaf_pages=fat_leaf_pages, budget=None, vlm_model=vlm_model, + scan_direction=page_memory_config.scan_direction, max_concurrent=page_memory_config.title_detection_concurrency, ) _record_trace_stage( @@ -667,6 +674,7 @@ def _run_hierarchy_scope( coarse_skeletons=scope_skeletons, tag_results=title_tags, fat_leaf_pages=fat_leaf_pages, + next_title_by_path=next_title_by_path, model_name=_resolve_hierarchy_model(page_memory_config), max_tokens=page_memory_config.hierarchy_max_tokens, max_depth=page_memory_config.max_heading_depth, @@ -729,7 +737,6 @@ def _run_hierarchy_scope( page_count=page_count, page_labels=page_labels, page_features=page_features, - tag_mode=page_memory_config.tag_mode, ) final_page_set = set(final_pages) plans = [plan for plan in plans if plan.page_index in final_page_set] @@ -764,10 +771,13 @@ def _run_hierarchy_scope( assets_by_page: dict[int, list[Any]] = {} if asset_extraction_enabled and asset_max_pages > 0: + asset_rendered = _select_rendered_pages_with_assets( + rendered, page_features + ) with stage_timer("page_memory.assets", page_count=asset_max_pages): assets_by_page = extract_page_assets_from_renders( pdf_path=pdf_path, - rendered_pages=rendered, + rendered_pages=asset_rendered, output_dir=output_dir, model_name=page_memory_config.asset_model, budget=None, @@ -791,6 +801,7 @@ def _run_hierarchy_scope( }, }, ) + # Refresh hierarchy/tags and write assets without wiping unrelated slots. _write_scope_artifacts( output_dir=output_dir, scope_id=scope.scope_id, @@ -810,6 +821,22 @@ def _run_hierarchy_scope( ) +# ── helpers ─────────────────────────────────────────────────────────── + + +def _select_rendered_pages_with_assets( + rendered: list[Any], + page_features: list[Any], +) -> list[Any]: + """Keep only rendered pages that coarse profile marked ``has_asset``.""" + asset_pages = { + int(getattr(feature, "page", 0) or 0) + for feature in page_features + if getattr(feature, "has_asset", False) + } + return [item for item in rendered if item.page_index in asset_pages] + + # ── whole_doc builder (PR3, unchanged) ──────────────────────────────── @@ -870,34 +897,29 @@ def _record_trace_stage( def _cleanup_page_memory_artifacts(output_dir: str) -> None: root = Path(output_dir) - legacy_files = { + stale_files = { "assets.json", "chunks.json", - "coarse_tag_scope.json", + "coarse_scopes.json", "doc_nav.json", "hierarchy.json", "manifest.json", "node_rows.csv", "node_rows.json", - "page_memory_fine_hierarchy.json", "page_plans.json", "page_rendered.json", "page_tags.json", - "page_tags_after_titles.json", - "page_tags_pre_hierarchy.json", "report.md", - "skeletons.json", - "tag_scope.json", "trace.json", } - for name in legacy_files: + for name in stale_files: path = root / name try: if path.is_file(): path.unlink() except Exception: logger.debug("[page_memory] failed to cleanup artifact {}", path) - for name in ("asset_annotate", "debug", "fine_hierarchy", "images", "pages", "scopes", "tables"): + for name in ("asset_annotate", "debug", "images", "pages", "scopes", "tables"): path = root / name try: if path.is_dir(): diff --git a/apps/worker/app/services/page_memory/page_assets.py b/apps/worker/app/services/page_memory/page_assets.py index 7e067a71c..05c17fe60 100644 --- a/apps/worker/app/services/page_memory/page_assets.py +++ b/apps/worker/app/services/page_memory/page_assets.py @@ -52,7 +52,7 @@ class PageAsset: def page_asset_extraction_enabled() -> bool: - return False + return True def page_asset_summary_enabled() -> bool: diff --git a/apps/worker/app/services/page_memory/page_plan.py b/apps/worker/app/services/page_memory/page_plan.py index 0cbdb69ca..234e28075 100644 --- a/apps/worker/app/services/page_memory/page_plan.py +++ b/apps/worker/app/services/page_memory/page_plan.py @@ -1,14 +1,10 @@ """Page processing plan: maps page labels to tagging strategy. Reads ``PageLabel.kind`` + ``PageFeature`` and assigns each page -one of three strategies: +one of two strategies: - ``vlm_lite`` — send the page image to VLM for summary/entities -- ``text_only`` — use raw text via page-memory-text-tag (no VLM call) - ``skip_tagging`` — blank-like page, preserve image only - -Set ``tag_mode=text`` to force all non-skip pages to ``text_only``. -Default is ``vlm``. """ from __future__ import annotations @@ -23,7 +19,6 @@ class PageProcessingStrategy(str, Enum): """Tagging strategy for a single page.""" VLM_LITE = "vlm_lite" - TEXT_ONLY = "text_only" SKIP_TAGGING = "skip_tagging" @@ -41,15 +36,12 @@ def derive_page_processing_plan( page_count: int, page_labels: list[PageLabel], page_features: list[PageFeature], - tag_mode: str = "vlm", ) -> list[PagePlan]: """Assign a processing strategy to every page. Rules: - - ``low_content`` + ``is_blank_like`` → ``skip_tagging`` - - ``tag_mode=text`` → ``text_only`` for all remaining pages - - ``table_heavy`` with sufficient raw text → ``text_only`` - - everything else (normal / image_heavy / landscape) → ``vlm_lite`` + - ``is_blank_like`` → ``skip_tagging`` + - everything else → ``vlm_lite`` Returns one ``PagePlan`` per page (1-indexed), ordered by page_index. """ @@ -60,33 +52,21 @@ def derive_page_processing_plan( for page in range(1, page_count + 1): label = label_map.get(page) feature = feature_map.get(page) - strategy, reason = _classify_page(label, feature, tag_mode=tag_mode) + strategy, reason = _classify_page(label, feature) plans.append(PagePlan(page_index=page, strategy=strategy, reason=reason)) return plans -# ── minimum raw text length for table_heavy → text_only ────────────── -_TABLE_TEXT_THRESHOLD = 200 - def _classify_page( label: PageLabel | None, feature: PageFeature | None, - *, - tag_mode: str, ) -> tuple[PageProcessingStrategy, str]: """Determine the strategy for a single page.""" kind = label.kind if label else "normal" is_blank = feature.is_blank_like if feature else False - text_len = feature.raw_text_length if feature else 0 - - if kind == "low_content" and is_blank: - return PageProcessingStrategy.SKIP_TAGGING, "low_content + blank_like" - - if tag_mode.strip().lower() == "text": - return PageProcessingStrategy.TEXT_ONLY, f"text_mode_override kind={kind}" - if kind == "table_heavy" and text_len >= _TABLE_TEXT_THRESHOLD: - return PageProcessingStrategy.TEXT_ONLY, f"table_heavy with {text_len} chars" + if is_blank: + return PageProcessingStrategy.SKIP_TAGGING, "blank_like" return PageProcessingStrategy.VLM_LITE, f"kind={kind}" diff --git a/apps/worker/app/services/page_memory/page_tagger.py b/apps/worker/app/services/page_memory/page_tagger.py index 4476e6a69..2204ddbd7 100644 --- a/apps/worker/app/services/page_memory/page_tagger.py +++ b/apps/worker/app/services/page_memory/page_tagger.py @@ -2,14 +2,11 @@ For ``vlm_lite`` pages, sends the page PNG to the VLM and expects a JSON response with ``summary`` and ``keywords``. -For ``text_only`` pages, calls the existing ``summary-full`` LLM prompt -to extract summary + keywords from raw text. For ``skip_tagging`` pages, content is preserved but summary is omitted. -Step 2 of page-memory native hierarchy adds: -- Independent VLM title candidate extraction (``observed_titles``) -- Fat-leaf gating: only pages in TOC leaves with > N pages trigger title detection -- Title extraction uses a dedicated verbatim-only prompt (temp=0, small max_tokens) +Title detection (``tag_page_titles``) extracts outline-level headings in +reading order on fat-leaf pages. Coarse start/end trimming is applied later +in ``fine_hierarchy._collect_candidates``. """ from __future__ import annotations @@ -50,6 +47,8 @@ class PageTagResult: _MAX_JSON_RETRIES = 1 _DEFAULT_FINE_MIN_PAGES = 4 +# Dense section-start pages can emit long title JSON; escalate only on truncation. +_TITLE_TOKEN_BUDGETS: tuple[int, ...] = (300, 600, 1200) def tag_pages( @@ -101,15 +100,12 @@ def _tag_one(page: PageRenderResult) -> PageTagResult: if strategy == PageProcessingStrategy.SKIP_TAGGING: return _tag_skip(page) - if strategy == PageProcessingStrategy.TEXT_ONLY: - return _tag_text_only(page) - if not model: logger.warning( - "[page_tagger] no VLM model configured for page {}; using text_only", + "[page_tagger] no VLM model configured for page {}; skipping tag", page.page_index, ) - return _tag_text_only(page) + return _tag_skip(page) return _tag_vlm_lite(page, model=model) @@ -120,10 +116,9 @@ def _tag_one(page: PageRenderResult) -> PageTagResult: results = [cast(PageTagResult, g.value) for g in greenlets] vlm_calls = sum(1 for r in results if r.strategy_used == "vlm_lite") logger.info( - "[page_tagger] tagged {} pages ({} VLM calls, {} text_only, {} skipped, {} failed) concurrency={}", + "[page_tagger] tagged {} pages ({} VLM calls, {} skipped, {} failed) concurrency={}", len(results), vlm_calls, - sum(1 for r in results if r.strategy_used == "text_only"), sum(1 for r in results if r.strategy_used == "skip_tagging"), len(greenlets) - len(results), resolved_max_concurrent, @@ -145,85 +140,6 @@ def _tag_skip(page: PageRenderResult) -> PageTagResult: ) -def _tag_text_only( - page: PageRenderResult, -) -> PageTagResult: - """Extract summary + entities from raw page text via page-memory-text-tag. - - Uses the same output spec ({summary, entities}) as the VLM path so - downstream consumers need no special-casing. Returns an EMPTY-marked - result when the page has no extractable text. - """ - raw = page.raw_text.strip() - if not raw: - return PageTagResult( - page_index=page.page_index, - summary="EMPTY", - keywords=[], - strategy_used="text_only", - ) - - model = os.environ.get("NORMOL_MODEL", "deepseek-v4-flash") - text_input = raw[:4000] # cap to avoid token overflow - prompt, temperature, top_p, max_tokens = build_prompt( - "page-memory-text-tag", - "", - "", - paras={"max_tokens": 600, "page_text": text_input}, - ) - - try: - from shared.services.ai.llm_overrides import get_text_client - - client, model = get_text_client(requested_model=model) - raw_response, usage = client.chat_completion_with_usage( - messages=cast(Any, [{"role": "user", "content": prompt}]), - model=model, - temperature=temperature, - top_p=top_p, - max_tokens=max_tokens, - response_format={"type": "json_object"}, - usage_task="page_memory.text_tag", - ) - except UnavailableException: - raise - except Exception as exc: - logger.warning( - "[page_tagger] text_only LLM failed for page {}: {}", - page.page_index, - exc, - ) - return PageTagResult( - page_index=page.page_index, - summary="", - keywords=[], - strategy_used="text_only", - ) - - try: - data = json.loads(raw_response) - except json.JSONDecodeError: - return PageTagResult( - page_index=page.page_index, - summary="", - keywords=[], - strategy_used="text_only", - ) - - entities_raw = data.get("entities") or [] - entities = [ - e for e in entities_raw - if isinstance(e, dict) and e.get("text") - ] - return PageTagResult( - page_index=page.page_index, - summary=str(data.get("summary") or "").strip(), - keywords=[str(e["text"]) for e in entities], - entities=entities, - strategy_used="text_only", - ) - - def _tag_vlm_lite( page: PageRenderResult, *, @@ -232,10 +148,10 @@ def _tag_vlm_lite( """Send page PNG to the VLM via the unified engine.""" if not page.image_path or not os.path.exists(page.image_path): logger.warning( - "[page_tagger] no PNG for page {}; text_only fallback", + "[page_tagger] no PNG for page {}; skipping tag", page.page_index, ) - return _tag_text_only(page) + return _tag_skip(page) result = summarize( mode="page", @@ -267,10 +183,14 @@ def tag_page_titles( fat_leaf_pages: set[int], budget: Any | None = None, vlm_model: str | None = None, + scan_direction: str = "top_to_bottom_left_to_right", max_concurrent: int | None = None, ) -> list[PageTagResult]: """Run independent VLM title detection on fat-leaf pages. + Extracts all outline-level headings in reading order. Coarse start/end + trimming happens later in ``fine_hierarchy._collect_candidates``. + Parameters ---------- pages: @@ -284,6 +204,8 @@ def tag_page_titles( Deprecated, ignored. Kept for call-site compatibility. vlm_model: VLM model name; falls back to ``$IMAGE_MODEL``. + scan_direction: + Reading-order wording for the title prompt. max_concurrent: Maximum concurrent title-detection calls. @@ -291,6 +213,7 @@ def tag_page_titles( ------- list[PageTagResult] Updated tag results with ``observed_titles`` populated for fat-leaf pages. + Title lists preserve VLM reading order (no re-sort). """ if not fat_leaf_pages: return tag_results @@ -323,7 +246,11 @@ def _detect_one( page_idx: int, page: PageRenderResult, ) -> tuple[int, list[dict[str, Any]]]: - return page_idx, _tag_vlm_titles(page, model=model) + return page_idx, _tag_vlm_titles( + page, + model=model, + scan_direction=scan_direction, + ) import gevent from gevent.pool import Pool as GeventPool @@ -357,17 +284,94 @@ def _detect_one( return tag_results +def _completion_tokens(usage: Any) -> int: + if isinstance(usage, dict): + return int(usage.get("completion_tokens") or 0) + return int(getattr(usage, "completion_tokens", 0) or 0) + + +def _title_response_truncated( + raw_response: str, + *, + usage: Any, + max_tokens: int, +) -> bool: + """True when the completion likely hit the budget mid-JSON.""" + if max_tokens > 0 and _completion_tokens(usage) >= max_tokens: + return True + stripped = (raw_response or "").rstrip() + if not stripped: + return False + return not stripped.endswith("}") + + +def _parse_observed_titles( + raw_response: str, + *, + page_index: int, +) -> list[dict[str, Any]]: + data = json.loads(raw_response) + titles_raw = data.get("titles", []) + if not isinstance(titles_raw, list): + return [] + + observed: list[dict[str, Any]] = [] + for item in titles_raw: + if not isinstance(item, dict) or not item.get("text"): + continue + text = str(item["text"]).strip() + is_table = item.get("is_in_table") is True + is_header = item.get("is_in_header_footer") is True + if is_table or is_header: + logger.debug( + "[page_tagger] filtered CoT title on page {}: '{}' (table={}, header={})", + page_index, + text, + is_table, + is_header, + ) + continue + if not text: + continue + prominence = None + try: + prominence = float(item.get("prominence", 0.5)) + except (TypeError, ValueError) as exc: + logger.debug( + "[page_tagger] ignored non-numeric title prominence {}: {}", + item.get("prominence"), + exc, + ) + observed.append( + { + "text": text, + "prominence": prominence, + "is_in_table": is_table, + "is_in_header_footer": is_header, + } + ) + return observed + + def _tag_vlm_titles( page: PageRenderResult, *, model: str, + scan_direction: str = "top_to_bottom_left_to_right", ) -> list[dict[str, Any]]: - """Send page PNG to VLM with the title-only prompt and parse results.""" - prompt, temperature, _top_p, max_tokens = build_prompt( + """Send page PNG to VLM with the title-only prompt and parse results. + + Starts at a small completion budget and escalates only when the response + is truncated (budget hit / incomplete JSON that fails to parse). + """ + prompt, temperature, _top_p, _default_max_tokens = build_prompt( "page-memory-vlm-title", "", "", - paras={"max_tokens": 300}, + paras={ + "max_tokens": _TITLE_TOKEN_BUDGETS[0], + "scan_direction": scan_direction, + }, ) try: @@ -393,70 +397,71 @@ def _tag_vlm_titles( client, resolved_model = get_vision_client(requested_model=model) model = resolved_model or model - for attempt in range(_MAX_JSON_RETRIES + 1): - try: - raw_response, usage = client.chat_completion_with_usage( - messages=cast(Any, [{"role": "user", "content": content_parts}]), - model=model, - temperature=temperature, - max_tokens=max_tokens, - response_format={"type": "json_object"}, - usage_task="page_memory.title_detection", - ) + last_truncated = False + for budget_index, max_tokens in enumerate(_TITLE_TOKEN_BUDGETS): + for attempt in range(_MAX_JSON_RETRIES + 1): + try: + raw_response, usage = client.chat_completion_with_usage( + messages=cast(Any, [{"role": "user", "content": content_parts}]), + model=model, + temperature=temperature, + max_tokens=max_tokens, + response_format={"type": "json_object"}, + usage_task="page_memory.title_detection", + ) + except UnavailableException: + raise + except Exception as exc: + logger.warning( + "[page_tagger] title VLM failed for page {}: {}", + page.page_index, + exc, + ) + return [] - data = json.loads(raw_response) - titles_raw = data.get("titles", []) - if not isinstance(titles_raw, list): + try: + observed = _parse_observed_titles( + raw_response, + page_index=page.page_index, + ) + except json.JSONDecodeError: + truncated = _title_response_truncated( + raw_response, + usage=usage, + max_tokens=max_tokens, + ) + if truncated and budget_index + 1 < len(_TITLE_TOKEN_BUDGETS): + last_truncated = True + logger.info( + "[page_tagger] title JSON truncated on page {} " + "(budget={}, completion_tokens={}); escalating", + page.page_index, + max_tokens, + _completion_tokens(usage), + ) + break # next budget + if attempt < _MAX_JSON_RETRIES: + continue + logger.warning( + "[page_tagger] title JSON retry exhausted for page {}", + page.page_index, + ) return [] - observed: list[dict[str, Any]] = [] - for item in titles_raw: - if isinstance(item, dict) and item.get("text"): - text = str(item["text"]).strip() - - is_table = item.get("is_in_table") is True - is_header = item.get("is_in_header_footer") is True - - if is_table or is_header: - logger.debug( - "[page_tagger] filtered CoT title on page {}: '{}' (table={}, header={})", - page.page_index, text, is_table, is_header - ) - continue - - if text: - prominence = None - try: - prominence = float(item.get("prominence", 0.5)) - except (TypeError, ValueError) as exc: - logger.debug( - "[page_tagger] ignored non-numeric title prominence {}: {}", - item.get("prominence"), - exc, - ) - observed.append({ - "text": text, - "prominence": prominence, - "is_in_table": is_table, - "is_in_header_footer": is_header - }) + if last_truncated: + logger.info( + "[page_tagger] title detection recovered on page {} with budget={}", + page.page_index, + max_tokens, + ) return observed + else: + continue - except json.JSONDecodeError: - if attempt < _MAX_JSON_RETRIES: - continue - logger.warning( - "[page_tagger] title JSON retry exhausted for page {}", - page.page_index, - ) - return [] - except UnavailableException: - raise - except Exception as exc: - logger.warning( - "[page_tagger] title VLM failed for page {}: {}", - page.page_index, exc, - ) - return [] - + if last_truncated: + logger.warning( + "[page_tagger] title JSON still truncated on page {} after budgets {}", + page.page_index, + list(_TITLE_TOKEN_BUDGETS), + ) return [] diff --git a/apps/worker/app/services/page_memory/skeleton_extractor.py b/apps/worker/app/services/page_memory/skeleton_extractor.py index 4b917cc27..43f664029 100644 --- a/apps/worker/app/services/page_memory/skeleton_extractor.py +++ b/apps/worker/app/services/page_memory/skeleton_extractor.py @@ -22,13 +22,20 @@ TitleMatch, TitleNode, extract_toc_nodes, + first_leaf_start_under, iter_leaf_title_nodes, + last_leaf_start_under, + locate_title_compact_strict, resolve_hierarchy_page_ranges, ) from app.services.document_parser.structure.body_boundary import ( clean_toc_title, ) from loguru import logger +from shared.services.chunks.path_segments import ( + append_document_path, + join_document_path, +) _FRONT_TOC_REGION_GAP_PAGES = 5 @@ -188,7 +195,7 @@ def extract_section_skeletons( ) ] - # Phase A3: try offset-guided bulk anchoring before expensive residual agent. + # Phase A3: offset-guided bulk anchoring for printed-page leaves. offset_matches: dict[tuple[str, ...], TitleMatch] | None = None if offset_hint is not None and ctx is not None: offset_matches = _offset_guided_anchoring( @@ -217,13 +224,30 @@ def extract_section_skeletons( } resolve_nodes = nodes + match_overrides, null_page_report = locate_null_page_parent_overrides( + nodes=resolve_nodes, + match_overrides=match_overrides, + page_texts=page_texts, + body_pages=primary_body_pages, + ctx=ctx, + ) + locate_summary["null_page_parent_locate"] = { + "attempted": len(null_page_report), + "located": sum(1 for row in null_page_report if row.get("page") is not None), + "unresolved": sum( + 1 for row in null_page_report if row.get("result") == "unresolved" + ), + "visual_verify_calls": sum( + int(row.get("visual_verify_calls") or 0) for row in null_page_report + ), + "entries": null_page_report, + } ranges = resolve_hierarchy_page_ranges( resolve_nodes, page_count=primary_page_count, page_texts=page_texts, body_pages=primary_body_pages, - page_offset_hint=offset_hint, match_overrides=match_overrides, ) if not ranges: @@ -275,8 +299,12 @@ def _range_to_skeleton( start_page = _clamp_page(item.start_page, page_count) end_page = _clamp_page(item.end_page, page_count) path_titles = [clean_toc_title(title) or title for title in item.path_titles] - section_path = "/".join([filename, *path_titles]) - parent_path = "/".join([filename, *path_titles[:-1]]) if len(path_titles) > 1 else filename + section_path = join_document_path([filename, *path_titles]) + parent_path = ( + join_document_path([filename, *path_titles[:-1]]) + if len(path_titles) > 1 + else filename + ) evidence = { **item.evidence, "resolver": "hierarchy_locator", @@ -473,8 +501,207 @@ def _body_pages(*, anatomy: Any | None, page_count: int) -> list[int]: return [page for page in range(1, page_count + 1) if page not in excluded] -# ── VLM offset calibration (Phase A1) ─────────────────────────────────────── +# ── Null-page parent locate (compact-strict + RTL visual) ─────────────────── + +_NULL_PARENT_VISUAL_CONFIDENCE = 0.6 + +def locate_null_page_parent_overrides( + *, + nodes: list[TitleNode], + match_overrides: dict[tuple[str, ...], TitleMatch], + page_texts: dict[int, str], + body_pages: list[int], + ctx: ToolContext | None, +) -> tuple[dict[tuple[str, ...], TitleMatch], list[dict[str, Any]]]: + """Locate TOC parents with ``printed_page=None`` into ``match_overrides``. + + Window for parent P: ``[last leaf start under previous same-level sibling, + first leaf start under P]``. Text path is compact→strict unique page; on + miss/ambiguity, scan right→left with ``verify_section_page_choice``. + + Returns ``(overrides, report)`` where *report* lists every null-page parent + attempt (for debug / LLM-call accounting). + """ + if not nodes or not body_pages: + return dict(match_overrides), [] + + out = dict(match_overrides) + body_set = set(body_pages) + parent_scope_start = body_pages[0] + report: list[dict[str, Any]] = [] + + def walk( + sibling_nodes: list[TitleNode], + parent_titles: tuple[str, ...], + scope_start: int, + ) -> None: + for index, node in enumerate(sibling_nodes): + path_titles = (*parent_titles, node.title) + if ( + node.children + and node.printed_page is None + and path_titles not in out + ): + if index > 0: + left = last_leaf_start_under( + sibling_nodes[index - 1], parent_titles, out + ) + if left is None: + left = scope_start + else: + left = scope_start + right = first_leaf_start_under(node, parent_titles, out) + entry: dict[str, Any] = { + "path_titles": list(path_titles), + "title": node.title, + "printed_page": None, + "window": None, + "result": "skipped_no_right", + "page": None, + "accept": None, + "visual_verify_calls": 0, + } + if right is None or right < left: + report.append(entry) + logger.info( + "[page_memory.skeleton] null-page parent skipped: " + "title={!r} reason=no_located_first_child left={}", + node.title, + left, + ) + else: + entry["window"] = [left, right] + scope_pages = [ + page for page in body_pages if left <= page <= right + ] + match = locate_title_compact_strict( + node.title, + scope_pages=scope_pages, + page_texts=page_texts, + ) + visual_calls = 0 + if match is None and ctx is not None: + match, visual_calls = _visual_rtl_locate_parent( + title=node.title, + left=left, + right=right, + body_set=body_set, + ctx=ctx, + ) + entry["visual_verify_calls"] = visual_calls + if match is not None and match.page in body_set: + out[path_titles] = match + entry["result"] = str(match.evidence.get("accept") or match.source) + entry["page"] = match.page + entry["accept"] = match.evidence.get("accept") + logger.info( + "[page_memory.skeleton] null-page parent located: " + "title={!r} page={} window={} accept={} visual_calls={}", + node.title, + match.page, + [left, right], + match.evidence.get("accept"), + visual_calls, + ) + else: + entry["result"] = "unresolved" + logger.info( + "[page_memory.skeleton] null-page parent unresolved: " + "title={!r} window={} visual_calls={}", + node.title, + [left, right], + visual_calls, + ) + report.append(entry) + if node.children: + child_scope_start = ( + out[path_titles].page if path_titles in out else scope_start + ) + walk(node.children, path_titles, child_scope_start) + + walk(nodes, (), parent_scope_start) + logger.info( + "[page_memory.skeleton] null-page parent locate summary: " + "attempted={} located={} unresolved={} visual_verify_calls={}", + len(report), + sum(1 for row in report if row.get("page") is not None), + sum(1 for row in report if row.get("result") == "unresolved"), + sum(int(row.get("visual_verify_calls") or 0) for row in report), + ) + return out, report + + +def _visual_rtl_locate_parent( + *, + title: str, + left: int, + right: int, + body_set: set[int], + ctx: ToolContext, +) -> tuple[TitleMatch | None, int]: + """Confirm parent title from right boundary toward left via VLM verify.""" + visual_calls = 0 + for page in range(right, left - 1, -1): + if page not in body_set: + continue + candidate = TitleMatch( + page=page, + confidence=0.4, + source="agent_heuristic", + matched_line="", + score=0.4, + candidates=[page], + evidence={"null_page_parent_probe": True}, + ) + visual_calls += 1 + result = verify_section_page_choice( + ctx=ctx, + title=title, + candidate_matches=[candidate], + candidate_page_cap=1, + ) + selected = result.get("selected_page") + confidence = float(result.get("confidence") or 0.0) + if selected != page or confidence < _NULL_PARENT_VISUAL_CONFIDENCE: + continue + if result.get("source") == "agent_vlm": + return ( + TitleMatch( + page=page, + confidence=confidence, + source="agent_vlm", + matched_line="", + score=confidence, + candidates=[page], + evidence={ + "accept": "visual_rtl", + "reason": result.get("reason", ""), + "visual_verify_calls": visual_calls, + }, + ), + visual_calls, + ) + return ( + TitleMatch( + page=page, + confidence=confidence, + source="agent_heuristic", + matched_line="", + score=confidence, + candidates=[page], + evidence={ + "accept": "visual_rtl", + "reason": result.get("reason", ""), + "visual_verify_calls": visual_calls, + }, + ), + visual_calls, + ) + return None, visual_calls + + +# ── VLM offset calibration (Phase A1) ─────────────────────────────────────── _CALIBRATION_WINDOW_PAGES = 10 _CALIBRATION_LEAF_PROBE_COUNT = 3 @@ -974,12 +1201,30 @@ def _resolve_pending_tocs( "reason": "offset_guided_anchoring_skipped_or_empty", } + match_overrides, null_page_report = locate_null_page_parent_overrides( + nodes=nodes, + match_overrides=match_overrides, + page_texts=page_texts, + body_pages=toc_body_pages, + ctx=ctx, + ) + locate_summary["null_page_parent_locate"] = { + "attempted": len(null_page_report), + "located": sum(1 for row in null_page_report if row.get("page") is not None), + "unresolved": sum( + 1 for row in null_page_report if row.get("result") == "unresolved" + ), + "visual_verify_calls": sum( + int(row.get("visual_verify_calls") or 0) for row in null_page_report + ), + "entries": null_page_report, + } + ranges = resolve_hierarchy_page_ranges( nodes, page_count=toc_scope_end, page_texts=page_texts, body_pages=toc_body_pages, - page_offset_hint=offset, match_overrides=match_overrides, ) @@ -1158,7 +1403,7 @@ def _collapse_node(path: str) -> None: new_children: list[str] = [] for gc_path in grandchild_paths: gc = by_path[gc_path] - new_path = f"{node.section_path}/{gc.title}" + new_path = append_document_path(node.section_path, gc.title) promoted = SectionSkeleton( section_path=new_path, level=gc.level - 1, diff --git a/apps/worker/experiments/chart_asset_probe.py b/apps/worker/experiments/chart_asset_probe.py deleted file mode 100644 index 016e08ca5..000000000 --- a/apps/worker/experiments/chart_asset_probe.py +++ /dev/null @@ -1,470 +0,0 @@ -"""Experimental: VLM-driven asset bbox detection + cropping. - -Goal of this experiment ------------------------ -Test whether a VLM can directly locate table/chart/figure bounding boxes on a -*rendered page image* (the same way PAGE-TRACK renders pages), so we can -**crop** those regions out and later hand the crop to a dedicated table -model (e.g. tabular / table-transformer) instead of asking the VLM to -transcribe the whole table verbatim (error-prone + expensive output). - -This script does NOT touch production code. It only: - - 1. Renders selected PDF pages to PNG at a fixed DPI (mirrors PAGE-TRACK, - default 144 DPI) so the pixel<->point mapping is fully controlled. - 2. (VLM) Asks the model only for table/chart/figure regions as normalized - [0,1000] boxes, maps them to pixels, crops, and draws an annotated overlay. - -Outputs (under --out): - pages/page-N.png full page render - crops/page-N_vlm-K_.png VLM crops - crops/page-N_ref-K_.png reference crops from existing chunks - asset_annotate/page_N.png page with VLM(red) + reference(green) boxes - results.json all regions + metadata - report.md human-readable summary - -Run: - cd apps/worker - uv run python experiments/chart_asset_probe.py \ - --pdf "/path/to/doc.pdf" \ - --pages all - # Uses $IMAGE_MODEL (default qwen3.6-flash) unless --model is set. - # Alternate: --model qwen3-vl-32b-instruct (open-weights, local-deployable). - -VLM model guidance (bbox grounding) ------------------------------------------------- -* **Default (cloud):** ``qwen3.6-flash`` via ``$IMAGE_MODEL`` — cheapest, - strong bbox quality on our probe PDFs. -* **Alternate (cloud or self-hosted):** ``qwen3-vl-32b-instruct`` — open - weights, can be deployed locally (vLLM / SGLang / Ollama); DashScope - China pricing (2026-04): input ¥2 / output ¥8 per 1M tokens (see - https://help.aliyun.com/zh/model-studio/model-pricing ). -* Coordinates are requested in a normalized 0-1000 space to be robust - to whatever internal resize the API performs. -""" - -from __future__ import annotations - -import argparse -import base64 -import json -import os -import re -import sys -from dataclasses import dataclass, field, asdict -from pathlib import Path -from typing import Any - -import fitz # PyMuPDF -from PIL import Image, ImageDraw, ImageFont - - -# ── coordinate convention ───────────────────────────────────────────── -# We ask the VLM for boxes in a normalized integer space [0, 1000] for -# BOTH axes, with origin at the top-left of the page image. This is robust -# to whatever internal resize the API performs. -NORM = 1000 -_VALID_KINDS = {"table", "figure"} - -_PROMPT = ( - "You are a precise document layout detector. The attached image is a single " - "rendered PDF page.\n\n" - "Find visually distinct tables and figures that should become reusable " - "document assets. Locate them only — do NOT summarize, transcribe full " - "content, extract keywords, or read data values. Return strict JSON:\n" - "{{\n" - ' "regions": [\n' - " {{\n" - ' "kind": "table|figure",\n' - ' "bbox": [x1, y1, x2, y2],\n' - ' "title": "",\n' - ' "confidence": 0.0\n' - " }}\n" - " ]\n" - "}}\n\n" - "Coordinate system:\n" - "- Treat the page image as a {n}x{n} grid.\n" - "- Origin is the top-left corner.\n" - "- bbox values must be integers in [0, {n}].\n" - "- bbox must tightly include the whole asset: its title, caption, legend, " - "axes, labels, table headers, and footnotes that belong to that asset.\n" - "- Exclude surrounding body paragraphs, page headers, page footers, and " - "page numbers.\n\n" - "Rules:\n" - "- \"table\": data arranged in clear rows and columns — grid lines, cell " - "borders, or strongly aligned cells (data tables, forms, financial tables, " - "appendix tables).\n" - "- \"figure\": any non-table visual asset — bar/line/pie/scatter charts, " - "plots, diagrams, flowcharts, architecture drawings, schematics, or " - "embedded images.\n" - "- Do not mark ordinary paragraphs, bullet lists, title blocks, or loose " - "multi-line text as tables.\n" - "- Do not split a single coherent table or figure into sub-parts.\n" - "- \"title\" is one short label only (single line). Do not duplicate it into " - "other fields and do not write a summary. Use an empty string when there is " - "no visible title or caption.\n" - "- Use confidence 0.0-1.0. Only include assets you can localize.\n" - "- If there are no qualifying assets, return {{\"regions\":[]}}.\n" - "- Return ONLY the JSON object, no markdown fences or explanations." -).format(n=NORM) - - -@dataclass -class Region: - source: str # "vlm" | "reference" - page: int - kind: str # table | figure - bbox_px: list[int] # [x1,y1,x2,y2] in rendered-image pixels - title: str = "" # short asset title / caption (single field, no duplication) - confidence: float = 0.0 - crop_path: str = "" - - -@dataclass -class PageResult: - page: int - width_px: int - height_px: int - width_pt: float - height_pt: float - image_path: str - vlm_regions: list[Region] = field(default_factory=list) - reference_regions: list[Region] = field(default_factory=list) - vlm_error: str = "" - - -# ── rendering ───────────────────────────────────────────────────────── - - -def render_page(page: fitz.Page, dpi: int, out_path: Path) -> tuple[int, int]: - zoom = dpi / 72.0 - mat = fitz.Matrix(zoom, zoom) - pix = page.get_pixmap(matrix=mat, alpha=False) - pix.save(str(out_path)) - return pix.width, pix.height - - -# ── VLM call ────────────────────────────────────────────────────────── - - -def call_vlm(image_path: Path, model: str) -> dict[str, Any]: - from shared.services.ai.openai_compatible_client_sync import get_openai_client - - with open(image_path, "rb") as f: - img_b64 = base64.b64encode(f.read()).decode() - - content_parts = [ - {"type": "text", "text": _PROMPT}, - { - "type": "image_url", - "image_url": {"url": f"data:image/png;base64,{img_b64}"}, - }, - ] - # For this standalone experiment we pass a direct Qwen key when available. - # The production Ali token pool depends on local Redis; direct mode keeps - # the probe runnable on a laptop without changing production behavior. - client = get_openai_client(model=model, api_key=_direct_api_key_for_model(model)) - raw, usage = client.chat_completion_with_usage( - messages=[{"role": "user", "content": content_parts}], - model=model, - temperature=0.0, - max_tokens=1200, - response_format={"type": "json_object"}, - usage_task="experiment.chart_asset_probe", - ) - data = json.loads(raw) - data["_usage"] = usage - return data - - -def _direct_api_key_for_model(model: str) -> str | None: - model_lower = model.lower() - if "qwen" not in model_lower: - return None - single = os.environ.get("ALI_API_KEY", "").strip() - if single: - return single - keys = os.environ.get("ALI_API_KEYS", "").strip() - if not keys: - return None - for item in re.split(r"[,;\s]+", keys): - if item.strip(): - return item.strip() - return None - - -def norm_to_px(box: list[float], w: int, h: int) -> list[int]: - x1, y1, x2, y2 = box - px = [ - int(round(x1 / NORM * w)), - int(round(y1 / NORM * h)), - int(round(x2 / NORM * w)), - int(round(y2 / NORM * h)), - ] - # normalize ordering + clamp - x1, x2 = sorted((px[0], px[2])) - y1, y2 = sorted((px[1], px[3])) - x1 = max(0, min(x1, w)) - x2 = max(0, min(x2, w)) - y1 = max(0, min(y1, h)) - y2 = max(0, min(y2, h)) - return [x1, y1, x2, y2] - - -def load_reference_regions(chunks_path: Path) -> dict[int, list[Region]]: - if not chunks_path.exists(): - raise FileNotFoundError(f"reference chunks not found: {chunks_path}") - payload = json.loads(chunks_path.read_text(encoding="utf-8")) - chunks = payload.get("chunks") if isinstance(payload, dict) else None - if not isinstance(chunks, list): - raise ValueError(f"reference chunks must contain a chunks[] array: {chunks_path}") - - by_page: dict[int, list[Region]] = {} - for chunk in chunks: - if not isinstance(chunk, dict): - continue - metadata = chunk.get("metadata") - if not isinstance(metadata, dict): - continue - bbox = metadata.get("bbox_px") - page_index = metadata.get("page_index") - if not isinstance(bbox, list) or len(bbox) != 4 or page_index is None: - continue - try: - page = int(page_index) - bbox_px = [int(round(float(item))) for item in bbox] - except (TypeError, ValueError): - continue - kind = str(metadata.get("asset_kind") or chunk.get("type") or "asset") - by_page.setdefault(page, []).append( - Region( - source="reference", - page=page, - kind=kind, - bbox_px=bbox_px, - title=str(metadata.get("title") or metadata.get("caption") or ""), - confidence=float(metadata.get("confidence") or 0.0), - ) - ) - return by_page - - -# ── cropping + overlay ──────────────────────────────────────────────── - - -def crop_region(img: Image.Image, region: Region, margin: int, out_path: Path) -> None: - x1, y1, x2, y2 = region.bbox_px - x1 = max(0, x1 - margin) - y1 = max(0, y1 - margin) - x2 = min(img.width, x2 + margin) - y2 = min(img.height, y2 + margin) - if x2 - x1 < 4 or y2 - y1 < 4: - return - img.crop((x1, y1, x2, y2)).save(out_path) - region.crop_path = str(out_path) - - -def draw_overlay(img: Image.Image, page_res: PageResult, out_path: Path) -> None: - canvas = img.convert("RGB").copy() - draw = ImageDraw.Draw(canvas) - try: - font = ImageFont.truetype("/System/Library/Fonts/Supplemental/Arial.ttf", 22) - except Exception: # noqa: BLE001 - font = ImageFont.load_default() - - for r in page_res.reference_regions: - x1, y1, x2, y2 = r.bbox_px - draw.rectangle([x1, y1, x2, y2], outline=(0, 170, 0), width=3) - draw.text((x1 + 2, y1 + 2), f"ref:{r.kind}", fill=(0, 120, 0), font=font) - - for i, r in enumerate(page_res.vlm_regions): - x1, y1, x2, y2 = r.bbox_px - draw.rectangle([x1, y1, x2, y2], outline=(220, 0, 0), width=3) - label = f"vlm:{r.kind} {r.confidence:.2f}" - if r.title: - label += f" | {r.title[:24]}" - draw.text((x1 + 2, max(0, y1 - 24)), label, fill=(200, 0, 0), font=font) - - canvas.save(out_path) - - -# ── page range parsing ──────────────────────────────────────────────── - - -def parse_pages(spec: str, total: int) -> list[int]: - if spec.strip().lower() == "all": - return list(range(1, total + 1)) - pages: set[int] = set() - for part in spec.split(","): - part = part.strip() - if not part: - continue - if "-" in part: - a, b = part.split("-", 1) - pages.update(range(int(a), int(b) + 1)) - else: - pages.add(int(part)) - return sorted(p for p in pages if 1 <= p <= total) - - -# ── main ────────────────────────────────────────────────────────────── - - -def main() -> int: - ap = argparse.ArgumentParser(description=__doc__) - ap.add_argument("--pdf", required=True) - ap.add_argument("--pages", default="all", help="e.g. 'all', '1-10', '1,3,5'") - ap.add_argument("--dpi", type=int, default=144, help="render DPI (PAGE-TRACK uses 144)") - ap.add_argument( - "--model", - default=os.environ.get("IMAGE_MODEL", ""), - help=( - "VLM model for bbox-only detection. Defaults to $IMAGE_MODEL " - "(qwen3.6-flash). Alternate: qwen3-vl-32b-instruct " - "(open-weights, local-deployable; DashScope ¥2/¥8 per 1M in/out)." - ), - ) - ap.add_argument("--margin", type=int, default=8, help="crop padding px") - ap.add_argument("--no-vlm", action="store_true", help="skip VLM (reference only)") - ap.add_argument( - "--reference-chunks", - default="", - help="existing chunks.json with old asset bbox metadata to overlay in green", - ) - ap.add_argument("--out", default="") - args = ap.parse_args() - - pdf_path = Path(args.pdf) - if not pdf_path.exists(): - print(f"PDF not found: {pdf_path}", file=sys.stderr) - return 2 - if not args.no_vlm and not args.model: - print("No --model and IMAGE_MODEL unset; pass --model or use --no-vlm", - file=sys.stderr) - return 2 - - out_dir = Path(args.out) if args.out else ( - Path.home() / ".knowhere" / "_debug_parse" / pdf_path.stem / "asset_probe" - ) - (out_dir / "pages").mkdir(parents=True, exist_ok=True) - (out_dir / "crops").mkdir(parents=True, exist_ok=True) - (out_dir / "asset_annotate").mkdir(parents=True, exist_ok=True) - - doc = fitz.open(str(pdf_path)) - page_nums = parse_pages(args.pages, doc.page_count) - reference_by_page = ( - load_reference_regions(Path(args.reference_chunks)) - if args.reference_chunks - else {} - ) - print(f"PDF: {pdf_path.name} | {doc.page_count} pages | probing {len(page_nums)} " - f"| dpi={args.dpi} | model={args.model or '(none)'} " - f"| reference={sum(len(items) for items in reference_by_page.values())}") - - results: list[PageResult] = [] - total_tokens = 0 - - for pno in page_nums: - page = doc[pno - 1] - img_path = out_dir / "pages" / f"page-{pno}.png" - w, h = render_page(page, args.dpi, img_path) - pr = PageResult( - page=pno, width_px=w, height_px=h, - width_pt=page.rect.width, height_pt=page.rect.height, - image_path=str(img_path), - ) - pr.reference_regions = list(reference_by_page.get(pno, [])) - print(f"\n[page {pno}] {w}x{h}px") - - if not args.no_vlm: - try: - data = call_vlm(img_path, args.model) - usage = data.pop("_usage", {}) - total_tokens += int(usage.get("total_tokens", 0) or 0) - for k, reg in enumerate(data.get("regions", [])): - box = reg.get("bbox") or reg.get("bbox_norm") - if not box or len(box) != 4: - continue - kind = str(reg.get("kind") or reg.get("type") or "").strip().lower() - if kind not in _VALID_KINDS: - continue - region = Region( - source="vlm", page=pno, - kind=kind, - bbox_px=norm_to_px([float(v) for v in box], w, h), - title=str(reg.get("title") or reg.get("caption") or ""), - confidence=float(reg.get("confidence", 0.0) or 0.0), - ) - pr.vlm_regions.append(region) - print(f" [vlm] {len(pr.vlm_regions)} region(s); " - f"tokens={usage.get('total_tokens', '?')}") - except Exception as exc: # noqa: BLE001 - pr.vlm_error = str(exc) - print(f" [vlm] ERROR: {exc}", file=sys.stderr) - - # crops + overlay - with Image.open(img_path) as im: - for k, r in enumerate(pr.vlm_regions): - crop_region(im, r, args.margin, - out_dir / "crops" / f"page-{pno}_vlm-{k}_{r.kind}.png") - for k, r in enumerate(pr.reference_regions): - crop_region(im, r, args.margin, - out_dir / "crops" / f"page-{pno}_ref-{k}_{r.kind}.png") - if pr.vlm_regions or pr.reference_regions: - draw_overlay(im, pr, out_dir / "asset_annotate" / f"page_{pno}.png") - results.append(pr) - - doc.close() - - # results.json - (out_dir / "results.json").write_text( - json.dumps([asdict(r) for r in results], ensure_ascii=False, indent=2), - encoding="utf-8", - ) - - # report.md - _write_report(out_dir, pdf_path, args, results, total_tokens) - print(f"\nDone. Output: {out_dir}") - print(f" - annotated overlays: {out_dir/'asset_annotate'}") - print(f" - crops: {out_dir/'crops'}") - print(" - results.json / report.md") - return 0 - - -def _write_report(out_dir: Path, pdf_path: Path, args: Any, - results: list[PageResult], total_tokens: int) -> None: - n_vlm = sum(len(r.vlm_regions) for r in results) - n_ref = sum(len(r.reference_regions) for r in results) - lines = [ - f"# Chart/Table Asset Probe — {pdf_path.name}", - "", - f"- pages probed: **{len(results)}**", - f"- dpi: **{args.dpi}**, model: **{args.model or '(no-vlm)'}**", - f"- VLM regions: **{n_vlm}**, reference regions: **{n_ref}**", - f"- total VLM tokens: **{total_tokens}**", - "", - "| page | px | vlm | ref | vlm kinds | vlm titles | vlm error |", - "|-----:|----|----:|----:|-----------|------------|-----------|", - ] - for r in results: - kinds = ",".join(sorted({x.kind for x in r.vlm_regions})) or "-" - titles = "; ".join(x.title for x in r.vlm_regions if x.title) or "-" - err = (r.vlm_error[:40] + "…") if r.vlm_error else "" - lines.append( - f"| {r.page} | {r.width_px}x{r.height_px} | {len(r.vlm_regions)} " - f"| {len(r.reference_regions)} | {kinds} | {titles} | {err} |" - ) - lines += [ - "", - "## How to read", - "- `asset_annotate/page_N.png`: red = new VLM boxes, green = reference boxes from existing chunks.", - "- Judge VLM by: does the red box tightly enclose the table/chart " - + "(incl. caption, excl. body text)? Compare against the green reference.", - "- `crops/`: the actual extracted assets to feed a table model next.", - ] - (out_dir / "report.md").write_text("\n".join(lines), encoding="utf-8") - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/apps/worker/experiments/vlm_bbox_tabula_probe.py b/apps/worker/experiments/vlm_bbox_tabula_probe.py deleted file mode 100644 index 53938bb55..000000000 --- a/apps/worker/experiments/vlm_bbox_tabula_probe.py +++ /dev/null @@ -1,294 +0,0 @@ -"""Experimental: VLM bbox + Tabula PDF table extraction. - -This probe tests the next step after ``chart_asset_probe.py``: - -1. Read only VLM-detected ``kind=table`` regions from ``results.json``. -2. Convert rendered-image pixel boxes back to PDF point coordinates. -3. Pass each area to Tabula (`tabula-py`) against the original PDF text layer. -4. Export candidate DataFrames as HTML/CSV for manual inspection. - -Important: Tabula does not read PNG crops. It needs a text-based PDF, so the -VLM box is used only as a precise `area=[top,left,bottom,right]` constraint. - -Run: - cd apps/worker - uv run --with tabula-py python experiments/vlm_bbox_tabula_probe.py \ - --pdf "/path/to/doc.pdf" \ - --results "/path/to/asset_probe/results.json" -""" - -from __future__ import annotations - -import argparse -import json -import shutil -import subprocess -from dataclasses import asdict, dataclass -from pathlib import Path -from typing import Any - -import pandas as pd - - -@dataclass -class TabulaCandidate: - page: int - region_index: int - mode: str - caption: str - bbox_px: list[int] - area_pt: list[float] - ok: bool - rows: int = 0 - cols: int = 0 - html_path: str = "" - csv_path: str = "" - error: str = "" - - -def _px_box_to_tabula_area( - bbox_px: list[int], - *, - width_px: int, - height_px: int, - width_pt: float, - height_pt: float, - margin_pt: float, -) -> list[float]: - """Convert [x1,y1,x2,y2] px box into Tabula [top,left,bottom,right] points.""" - x1, y1, x2, y2 = bbox_px - left = x1 / width_px * width_pt - right = x2 / width_px * width_pt - top = y1 / height_px * height_pt - bottom = y2 / height_px * height_pt - return [ - round(max(0.0, top - margin_pt), 2), - round(max(0.0, left - margin_pt), 2), - round(min(height_pt, bottom + margin_pt), 2), - round(min(width_pt, right + margin_pt), 2), - ] - - -def _load_vlm_table_regions(results_path: Path) -> list[dict[str, Any]]: - pages = json.loads(results_path.read_text(encoding="utf-8")) - table_regions: list[dict[str, Any]] = [] - for page in pages: - for region_index, region in enumerate(page.get("vlm_regions", [])): - if str(region.get("kind", "")).lower() != "table": - continue - table_regions.append({ - "page": int(page["page"]), - "region_index": region_index, - "caption": str(region.get("caption", "")), - "bbox_px": list(region["bbox_px"]), - "width_px": int(page["width_px"]), - "height_px": int(page["height_px"]), - "width_pt": float(page["width_pt"]), - "height_pt": float(page["height_pt"]), - }) - return table_regions - - -def _is_meaningful_frame(df: pd.DataFrame) -> bool: - if df.empty or df.shape[1] == 0: - return False - non_empty = df.fillna("").astype(str).map(lambda s: bool(s.strip())) - return bool(non_empty.values.sum() >= 2) - - -def _safe_name(page: int, region_index: int, mode: str, candidate_index: int) -> str: - return f"page-{page}_vlm-{region_index}_{mode}-{candidate_index}" - - -def _has_working_java() -> bool: - if shutil.which("java") is None: - return False - try: - result = subprocess.run( - ["java", "-version"], - check=False, - capture_output=True, - text=True, - timeout=10, - ) - except Exception: - return False - combined = f"{result.stdout}\n{result.stderr}" - return result.returncode == 0 and "Unable to locate a Java Runtime" not in combined - - -def _extract_one( - *, - tabula: Any, - pdf_path: Path, - out_dir: Path, - region: dict[str, Any], - mode: str, - area_pt: list[float], -) -> list[TabulaCandidate]: - lattice = mode == "lattice" - stream = mode == "stream" - try: - frames = tabula.read_pdf( - str(pdf_path), - pages=region["page"], - area=area_pt, - guess=False, - lattice=lattice, - stream=stream, - multiple_tables=True, - pandas_options={"header": None}, - ) - except Exception as exc: # noqa: BLE001 - return [ - TabulaCandidate( - page=region["page"], - region_index=region["region_index"], - mode=mode, - caption=region["caption"], - bbox_px=region["bbox_px"], - area_pt=area_pt, - ok=False, - error=str(exc), - ) - ] - - candidates: list[TabulaCandidate] = [] - if not frames: - return [ - TabulaCandidate( - page=region["page"], - region_index=region["region_index"], - mode=mode, - caption=region["caption"], - bbox_px=region["bbox_px"], - area_pt=area_pt, - ok=False, - error="tabula returned no tables", - ) - ] - - for candidate_index, df in enumerate(frames): - name = _safe_name(region["page"], region["region_index"], mode, candidate_index) - candidate = TabulaCandidate( - page=region["page"], - region_index=region["region_index"], - mode=mode, - caption=region["caption"], - bbox_px=region["bbox_px"], - area_pt=area_pt, - ok=_is_meaningful_frame(df), - rows=int(df.shape[0]), - cols=int(df.shape[1]), - ) - if candidate.ok: - html_path = out_dir / "html" / f"{name}.html" - csv_path = out_dir / "csv" / f"{name}.csv" - df.to_html(html_path, index=False, header=False, na_rep="") - df.to_csv(csv_path, index=False, header=False) - candidate.html_path = str(html_path) - candidate.csv_path = str(csv_path) - else: - candidate.error = "empty or near-empty dataframe" - candidates.append(candidate) - return candidates - - -def _write_report(out_dir: Path, candidates: list[TabulaCandidate]) -> None: - ok_count = sum(1 for item in candidates if item.ok) - lines = [ - "# VLM BBox + Tabula Probe", - "", - f"- candidates: **{len(candidates)}**", - f"- successful non-empty tables: **{ok_count}**", - "", - "| page | vlm | mode | ok | shape | caption | error |", - "|-----:|----:|------|----|-------|---------|-------|", - ] - for item in candidates: - error = item.error.replace("\n", " ")[:80] - lines.append( - f"| {item.page} | {item.region_index} | {item.mode} | " - f"{'yes' if item.ok else 'no'} | {item.rows}x{item.cols} | " - f"{item.caption} | {error} |" - ) - lines.append("") - lines.append("Only VLM `kind=table` regions were used; baseline regions were ignored.") - (out_dir / "report.md").write_text("\n".join(lines), encoding="utf-8") - - -def main() -> int: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--pdf", required=True) - parser.add_argument("--results", required=True) - parser.add_argument("--out", default="") - parser.add_argument("--margin-pt", type=float, default=2.0) - args = parser.parse_args() - - pdf_path = Path(args.pdf) - results_path = Path(args.results) - out_dir = ( - Path(args.out) - if args.out - else results_path.parent / "tabula_from_vlm_bbox" - ) - (out_dir / "html").mkdir(parents=True, exist_ok=True) - (out_dir / "csv").mkdir(parents=True, exist_ok=True) - - if not pdf_path.exists(): - raise FileNotFoundError(pdf_path) - if not results_path.exists(): - raise FileNotFoundError(results_path) - if not _has_working_java(): - raise RuntimeError( - "Java runtime is required by tabula-py but was not found. " - "Install a JRE (e.g. OpenJDK/Temurin) and rerun this probe." - ) - - try: - import tabula - except ImportError as exc: - raise RuntimeError( - "tabula-py is required. Run with: uv run --with tabula-py python ..." - ) from exc - - regions = _load_vlm_table_regions(results_path) - print(f"Loaded {len(regions)} VLM table regions from {results_path}") - - candidates: list[TabulaCandidate] = [] - for region in regions: - area_pt = _px_box_to_tabula_area( - region["bbox_px"], - width_px=region["width_px"], - height_px=region["height_px"], - width_pt=region["width_pt"], - height_pt=region["height_pt"], - margin_pt=args.margin_pt, - ) - for mode in ("lattice", "stream"): - extracted = _extract_one( - tabula=tabula, - pdf_path=pdf_path, - out_dir=out_dir, - region=region, - mode=mode, - area_pt=area_pt, - ) - candidates.extend(extracted) - for item in extracted: - print( - f"page={item.page} vlm={item.region_index} " - f"mode={item.mode} ok={item.ok} shape={item.rows}x{item.cols}" - ) - - (out_dir / "results.json").write_text( - json.dumps([asdict(item) for item in candidates], ensure_ascii=False, indent=2), - encoding="utf-8", - ) - _write_report(out_dir, candidates) - print(f"Done. Output: {out_dir}") - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/apps/worker/tests/contract/conftest.py b/apps/worker/tests/contract/conftest.py index f9738370f..1a9ae74a9 100644 --- a/apps/worker/tests/contract/conftest.py +++ b/apps/worker/tests/contract/conftest.py @@ -37,7 +37,14 @@ def _module_loaded_from(module_name: str, root: Path) -> bool: return True module_paths = getattr(module, "__path__", ()) - return any(str(module_path).startswith(root_value) for module_path in module_paths) + try: + return any( + str(module_path).startswith(root_value) for module_path in module_paths + ) + except KeyError: + # Namespace path iteration can raise if a parent package was already + # removed from sys.modules mid-eviction. + return False def _ensure_worker_import_context() -> None: @@ -46,11 +53,16 @@ def _ensure_worker_import_context() -> None: sys.path.remove(worker_root_value) sys.path.insert(0, worker_root_value) - cached_module_names = list(sys.modules) - for module_name in cached_module_names: - if module_name == "app" or module_name.startswith("app."): - if _module_loaded_from(module_name, _API_ROOT): - sys.modules.pop(module_name, None) + cached_module_names = [ + module_name + for module_name in sys.modules + if module_name == "app" or module_name.startswith("app.") + ] + # Evict deepest modules first so namespace __path__ checks never observe a + # missing parent `app` entry while walking API-shadowed packages. + for module_name in sorted(cached_module_names, key=len, reverse=True): + if _module_loaded_from(module_name, _API_ROOT): + sys.modules.pop(module_name, None) @pytest.fixture(autouse=True) diff --git a/apps/worker/tests/contract/test_doc_profile_anatomy_contract.py b/apps/worker/tests/contract/test_doc_profile_anatomy_contract.py index b45a26af1..839b61654 100644 --- a/apps/worker/tests/contract/test_doc_profile_anatomy_contract.py +++ b/apps/worker/tests/contract/test_doc_profile_anatomy_contract.py @@ -56,11 +56,36 @@ def _page_feature(page: int = 1) -> PageFeature: orientation="portrait", width=72.0, height=72.0, + has_asset=False, is_blank_like=False, - text_lines_preview=["Section 1"], + asset_bboxes=None, ) +def _seed_preprobed_pages( + coordinator: ProfileCoordinator, + *, + page_count: int, + pages: list[int] | None = None, +) -> None: + """Seed text-bootstrap state for coordinator tests without a real PDF. + + Marks assets as already probed so `_ensure_asset_probe` does not spawn a + PyMuPDF child against a missing fixture path. + """ + probed_pages = pages or list(range(1, page_count + 1)) + coordinator.blackboard.page_count = page_count + coordinator.blackboard.page_features = [_page_feature(page) for page in probed_pages] + coordinator.blackboard.page_labels = [ + PageLabel(page=page, kind="normal", confidence=1.0) for page in probed_pages + ] + coordinator.blackboard.doc_stats = {"page_count": page_count} + coordinator.blackboard.global_signals["page_kind_counts"] = { + "normal": page_count + } + coordinator.blackboard.global_signals["assets_probed"] = True + + def test_toc_anchor_text_scan_matches_full_page_and_cross_line_keywords() -> None: late_lines = [f"body line {idx}" for idx in range(60)] + ["目录"] split_lines = ["Table of", "Con", "tents"] @@ -112,14 +137,7 @@ def test_run_lightweight_anatomy_builds_single_shard_without_planner_llm( output_dir=str(output_dir), settings={"shard_threshold": 200}, ) - coordinator.blackboard.page_count = 2 - coordinator.blackboard.page_features = [_page_feature(1), _page_feature(2)] - coordinator.blackboard.page_labels = [ - PageLabel(page=1, kind="normal", confidence=1.0), - PageLabel(page=2, kind="normal", confidence=1.0), - ] - coordinator.blackboard.doc_stats = {"page_count": 2} - coordinator.blackboard.global_signals["page_kind_counts"] = {"normal": 2} + _seed_preprobed_pages(coordinator, page_count=2) coordinator.blackboard.document_profile = DocumentProfile( is_scanned=False, category="Research Report", @@ -135,10 +153,65 @@ def test_run_lightweight_anatomy_builds_single_shard_without_planner_llm( assert anatomy.shard_plan.shards[0].page_end == 2 assert anatomy.toc_result.method == "none" assert (output_dir / "anatomy_map.json").exists() + anatomy_data = json.loads( + (output_dir / "anatomy_map.json").read_text(encoding="utf-8") + ) + assert list(anatomy_data)[:2] == ["version", "toc_hierarchies"] + assert "text_lines_preview" not in anatomy_data["page_features"][0] trace_data = json.loads((output_dir / "trace.json").read_text(encoding="utf-8")) assert "visual_stages" in trace_data["summary"]["budget"] +def test_run_coarse_runs_asset_probe_after_planner(monkeypatch, tmp_path: Path) -> None: + coordinator = ProfileCoordinator( + pdf_path=str(tmp_path / "doc.pdf"), + job_id="job-asset-probe-after-coarse", + output_dir=str(tmp_path / "profile"), + ) + (tmp_path / "profile").mkdir() + coordinator.blackboard.page_count = 2 + coordinator.blackboard.page_features = [_page_feature(1), _page_feature(2)] + coordinator.blackboard.page_labels = [ + PageLabel(page=1, kind="normal", confidence=1.0), + PageLabel(page=2, kind="normal", confidence=1.0), + ] + coordinator.blackboard.doc_stats = {"page_count": 2} + coordinator.blackboard.global_signals["page_kind_counts"] = {"normal": 2} + + calls: list[str] = [] + + def fake_propose(_self): + calls.append("planner") + return ( + DocumentProfile( + is_scanned=False, + category="Research Report", + routing_category=PdfRoutingCategory.GENERIC.value, + ), + None, + ToolResult(status="ok", payload={}), + ) + + def fake_probe_page_assets(ctx, _args): + calls.append("probe.page_assets") + ctx.blackboard.global_signals["assets_probed"] = True + return ToolResult(status="ok", payload={"page_count": 2}) + + def fake_aggregate(ctx, _args): + calls.append("aggregate.doc_stats") + return ToolResult(status="ok", payload={}) + + monkeypatch.setattr(coordinator_module.ProfilePlanner, "propose", fake_propose) + monkeypatch.setattr(coordinator_module, "probe_page_assets", fake_probe_page_assets) + monkeypatch.setattr(coordinator_module, "aggregate_doc_stats", fake_aggregate) + + profile = coordinator.run_coarse() + + assert profile.category == "Research Report" + assert calls == ["planner", "probe.page_assets", "aggregate.doc_stats"] + assert coordinator.blackboard.global_signals["assets_probed"] is True + + def test_parse_run_recorder_doc_profile_uses_final_anatomy_toc() -> None: recorder = ParseRunRecorder(job_id="job-doc-profile") recorder.set_doc_profile( @@ -234,14 +307,7 @@ def test_run_structural_retries_transient_confirm_failed_toc_result( output_dir=str(tmp_path / "profile"), ) (tmp_path / "profile").mkdir() - coordinator.blackboard.page_count = 3 - coordinator.blackboard.page_features = [_page_feature(1), _page_feature(2)] - coordinator.blackboard.page_labels = [ - PageLabel(page=1, kind="normal", confidence=1.0), - PageLabel(page=2, kind="normal", confidence=1.0), - ] - coordinator.blackboard.doc_stats = {"page_count": 3} - coordinator.blackboard.global_signals["page_kind_counts"] = {"normal": 3} + _seed_preprobed_pages(coordinator, page_count=3, pages=[1, 2]) coordinator.blackboard.document_profile = DocumentProfile( is_scanned=False, category="Prospectus", @@ -323,6 +389,71 @@ def run(self): assert anatomy.toc_result.toc_pages == [17] +def test_run_structural_skip_shard_plan_uses_placeholder_without_executor( + monkeypatch, + tmp_path: Path, +) -> None: + coordinator = ProfileCoordinator( + pdf_path=str(tmp_path / "oversized.pdf"), + job_id="job-structural-skip-shard", + output_dir=str(tmp_path / "profile"), + ) + (tmp_path / "profile").mkdir() + _seed_preprobed_pages(coordinator, page_count=4) + coordinator.blackboard.document_profile = DocumentProfile( + is_scanned=False, + category="Prospectus", + routing_category=PdfRoutingCategory.GENERIC.value, + ) + coordinator.blackboard.toc_result = TocResult( + toc_pages=[1], + method="vlm_batch", + notes="ok", + ) + coordinator.blackboard.toc_hierarchies = [ + {"toc_range": [1, 1], "toc_range_unit": "page", "toc_tree": {}} + ] + + monkeypatch.setattr( + coordinator, + "_run_toc_extraction_pipeline", + lambda: (_ for _ in ()).throw( + AssertionError("existing TOC should not be re-extracted") + ), + ) + monkeypatch.setattr( + coordinator, + "_persist_ready_anatomy", + lambda _anatomy: None, + ) + monkeypatch.setattr( + coordinator_module.ProfilePlanner, + "propose", + lambda self: ( + coordinator.blackboard.document_profile, + None, + ToolResult(status="ok", payload={}), + ), + ) + + class BoomExecutor: + def __init__(self, *_args, **_kwargs) -> None: + raise AssertionError("ReActExecutor must not run when skip_shard_plan") + + def run(self): # pragma: no cover + raise AssertionError("unreachable") + + monkeypatch.setattr(coordinator_module, "ReActExecutor", BoomExecutor) + + anatomy = coordinator.run_structural(skip_shard_plan=True) + + assert anatomy.shard_plan.enabled is False + assert len(anatomy.shard_plan.shards) == 1 + assert anatomy.shard_plan.shards[0].page_start == 1 + assert anatomy.shard_plan.shards[0].page_end == 4 + assert anatomy.toc_result.toc_pages == [1] + + def test_run_structural_trusts_rejected_all_toc_and_fails_open( monkeypatch, tmp_path: Path, @@ -333,14 +464,7 @@ def test_run_structural_trusts_rejected_all_toc_and_fails_open( output_dir=str(tmp_path / "profile"), ) (tmp_path / "profile").mkdir() - coordinator.blackboard.page_count = 3 - coordinator.blackboard.page_features = [_page_feature(1), _page_feature(2)] - coordinator.blackboard.page_labels = [ - PageLabel(page=1, kind="normal", confidence=1.0), - PageLabel(page=2, kind="normal", confidence=1.0), - ] - coordinator.blackboard.doc_stats = {"page_count": 3} - coordinator.blackboard.global_signals["page_kind_counts"] = {"normal": 3} + _seed_preprobed_pages(coordinator, page_count=3, pages=[1, 2]) coordinator.blackboard.document_profile = DocumentProfile( is_scanned=False, category="Prospectus", @@ -426,14 +550,7 @@ def test_run_coarse_runs_toc_before_planner_for_oversized_and_reuses_planner( settings={"toc_before_coarse": True}, ) (tmp_path / "profile").mkdir() - coordinator.blackboard.page_count = 3 - coordinator.blackboard.page_features = [_page_feature(1), _page_feature(2)] - coordinator.blackboard.page_labels = [ - PageLabel(page=1, kind="normal", confidence=1.0), - PageLabel(page=2, kind="normal", confidence=1.0), - ] - coordinator.blackboard.doc_stats = {"page_count": 3} - coordinator.blackboard.global_signals["page_kind_counts"] = {"normal": 3} + _seed_preprobed_pages(coordinator, page_count=3, pages=[1, 2]) calls: list[str] = [] @@ -658,6 +775,57 @@ def run_lightweight_anatomy(self, *, skip_shard_plan: bool = False): assert profile.anatomy is fake_anatomy +def test_page_memory_forces_toc_profiling_despite_kill_switch( + monkeypatch, + tmp_path: Path, +) -> None: + fake_anatomy = object() + init_settings: list[dict[str, object]] = [] + + class FakeCoordinator: + def __init__(self, **kwargs) -> None: + self.calls: list[str] = [] + init_settings.append(kwargs["settings"]) + self.blackboard = SimpleNamespace( + page_count=2, + doc_stats={"page_count": 2}, + global_signals={}, + toc_result=None, + toc_hierarchies=None, + ) + + def run_coarse(self) -> DocumentProfile: + self.calls.append("run_coarse") + self.blackboard.toc_result = TocResult(method="none") + return DocumentProfile( + is_scanned=False, + category="Research Report", + routing_category=PdfRoutingCategory.GENERIC.value, + ) + + def run_lightweight_anatomy(self, *, skip_shard_plan: bool = False): + self.calls.append("run_lightweight_anatomy") + return fake_anatomy + + monkeypatch.setattr(doc_profiler, "ProfileCoordinator", FakeCoordinator) + monkeypatch.setattr(doc_profiler.settings, "MAX_PDF_PAGE_LIMIT", 200) + # Global kill switch OFF; page_memory track must still enable TOC. + monkeypatch.setattr(doc_profiler.settings, "PDF_PROFILE_TOC_ENABLED", False) + + profile = profile_document( + str(tmp_path / "standard.pdf"), + "standard.pdf", + job_id="job-page-memory-toc-forced", + output_dir=str(tmp_path), + skip_shard_plan=True, + oversized_policy="page_memory", + ) + + assert init_settings[0]["toc_profile_enabled"] is True + assert init_settings[0]["toc_before_coarse"] is True + assert profile.anatomy is fake_anatomy + + def test_page_memory_profile_bypasses_chunk_oversized_gate( monkeypatch, tmp_path: Path, @@ -685,8 +853,9 @@ def run_coarse(self) -> DocumentProfile: routing_category=PdfRoutingCategory.GENERIC.value, ) - def run_structural(self): + def run_structural(self, *, skip_shard_plan: bool = False): self.calls.append("run_structural") + self.skip_shard_plan = skip_shard_plan return fake_anatomy monkeypatch.setattr(doc_profiler, "ProfileCoordinator", FakeCoordinator) @@ -698,11 +867,13 @@ def run_structural(self): "oversized.pdf", job_id="job-page-memory-oversized", output_dir=str(tmp_path), + skip_shard_plan=True, oversized_policy="page_memory", ) assert profile.anatomy is fake_anatomy assert fake_instances[0].calls == ["run_coarse", "run_structural"] + assert fake_instances[0].skip_shard_plan is True def test_standard_pdf_profile_maps_page_toc_evidence( @@ -820,10 +991,10 @@ def run_coarse(self) -> DocumentProfile: routing_category=PdfRoutingCategory.ATLAS.value, ) - def run_structural(self): + def run_structural(self, *, skip_shard_plan: bool = False): raise AssertionError("oversized atlas should not run structural anatomy") - def run_lightweight_anatomy(self): + def run_lightweight_anatomy(self, *, skip_shard_plan: bool = False): raise AssertionError("oversized atlas should not run lightweight anatomy") monkeypatch.setattr(doc_profiler, "ProfileCoordinator", FakeCoordinator) diff --git a/apps/worker/tests/contract/test_image_size_filter_contract.py b/apps/worker/tests/contract/test_image_size_filter_contract.py new file mode 100644 index 000000000..1b04bdb22 --- /dev/null +++ b/apps/worker/tests/contract/test_image_size_filter_contract.py @@ -0,0 +1,35 @@ +from __future__ import annotations + +import os +from pathlib import Path + +os.environ.setdefault("DATABASE_URL", "postgresql+asyncpg://test:test@localhost/test") +os.environ.setdefault("TMP_PATH", "/tmp/knowhere-test") +os.environ.setdefault("S3_BUCKET_NAME", "test-uploads") +os.environ.setdefault("S3_ACCESS_KEY_ID", "test") +os.environ.setdefault("S3_SECRET_ACCESS_KEY", "test") +os.environ.setdefault("S3_TEMP_PATH", "/tmp") + +from app.services.document_parser.assets.image_size_filter import ( # noqa: E402 + discard_undersized_image_file, + is_below_img_min_size, +) +from shared.core.constants.processing import ProcessingConstants # noqa: E402 + + +def test_is_below_img_min_size_boundary() -> None: + assert is_below_img_min_size(ProcessingConstants.IMG_MIN_SIZE - 1) + assert not is_below_img_min_size(ProcessingConstants.IMG_MIN_SIZE) + assert not is_below_img_min_size(ProcessingConstants.IMG_MIN_SIZE + 1) + + +def test_discard_undersized_image_file(tmp_path: Path) -> None: + small = tmp_path / "small.jpg" + small.write_bytes(b"x" * (ProcessingConstants.IMG_MIN_SIZE - 1)) + assert discard_undersized_image_file(small, label="unit") is True + assert not small.exists() + + large = tmp_path / "large.jpg" + large.write_bytes(b"y" * ProcessingConstants.IMG_MIN_SIZE) + assert discard_undersized_image_file(large, label="unit") is False + assert large.exists() diff --git a/apps/worker/tests/contract/test_markdown_table_rename_contract.py b/apps/worker/tests/contract/test_markdown_table_rename_contract.py new file mode 100644 index 000000000..6360181b9 --- /dev/null +++ b/apps/worker/tests/contract/test_markdown_table_rename_contract.py @@ -0,0 +1,109 @@ +from __future__ import annotations + +import os +from pathlib import Path + +os.environ.setdefault("DATABASE_URL", "postgresql+asyncpg://test:test@localhost/test") +os.environ.setdefault("TMP_PATH", "/tmp/knowhere-test") +os.environ.setdefault("S3_BUCKET_NAME", "test-uploads") +os.environ.setdefault("S3_ACCESS_KEY_ID", "test") +os.environ.setdefault("S3_SECRET_ACCESS_KEY", "test") +os.environ.setdefault("S3_TEMP_PATH", "/tmp") + +from app.services.document_parser.formats.markdown.deferred_summary import ( # noqa: E402 + replace_chunk_ref_in_rows, + _apply_table_summary_result, +) +from app.services.document_parser.formats.markdown.deferred_task import ( # noqa: E402 + TableDeferredSummaryTask, +) +from app.services.document_parser.formats.markdown.table_asset import ( # noqa: E402 + MarkdownTableAssetRequest, + build_markdown_table_asset, +) +from shared.services.ai.summary.model import AssetSummary, Entity # noqa: E402 +from shared.utils.chunk_refs import build_chunk_ref # noqa: E402 + + +def test_replace_chunk_ref_updates_bare_table_content_and_bracketed_text() -> None: + old_path = "tables/table-12 流程名称 招标文件.html" + new_path = "tables/table-12 招标文件及议标项目评审流程图.html" + rows: list[list[str | int]] = [ + [old_path, old_path, "table", 10, "", "table-12", "id", "", ""], + [ + f"see {build_chunk_ref(old_path)}", + "doc/section", + "ptxt", + 20, + "", + "", + "text-id", + "", + "", + ], + ] + + replace_chunk_ref_in_rows(rows, old_path, new_path) + + assert rows[0][0] == new_path + assert rows[0][1] == new_path + assert rows[1][0] == f"see {build_chunk_ref(new_path)}" + + +def test_table_deferred_task_has_no_legacy_count_field(tmp_path: Path) -> None: + asset = build_markdown_table_asset( + MarkdownTableAssetRequest( + table_html="
流程名称招标文件
", + table_dir=str(tmp_path), + table_count=12, + timestamp="2026-07-21 00:00:00", + summary_table=True, + row_index=0, + ) + ) + + assert asset.deferred_task is not None + assert isinstance(asset.deferred_task, TableDeferredSummaryTask) + assert not hasattr(asset.deferred_task, "table_count") + assert asset.deferred_task.table_name.startswith("table-12 ") + + +def test_apply_table_summary_rename_keeps_index_and_syncs_content( + tmp_path: Path, +) -> None: + old_stem = "table-12 流程名称 招标文件及议标项目评审流程图 流程编号" + old_relative = f"tables/{old_stem}.html" + old_file = tmp_path / f"{old_stem}.html" + old_file.write_text("
x
", encoding="utf-8") + + text_content = f"\n{build_chunk_ref(old_relative)}\n" + rows: list[list[str | int]] = [ + [old_relative, old_relative, "table", len(old_relative), "", "table-12", "t", "", ""], + [text_content, "doc/3、招标文件评审流程运行图", "ptxt", len(text_content), "", "", "x", "", ""], + ] + task = TableDeferredSummaryTask( + row_index=0, + table_html="
x
", + table_dir=str(tmp_path), + table_name=old_stem, + ) + + _apply_table_summary_result( + rows, + task, + 0, + AssetSummary( + title="招标文件及议标项目评审流程图", + summary="三类风险招标文件评审流程", + entities=[Entity(text="市场开发部", type="org")], + kind="table", + ), + ) + + new_relative = str(rows[0][1]) + assert new_relative.startswith("tables/table-12 ") + assert "招标文件及议标项目评审流程图" in new_relative + assert rows[0][0] == new_relative + assert build_chunk_ref(new_relative) in str(rows[1][0]) + assert not old_file.exists() + assert (tmp_path / Path(new_relative).name).exists() diff --git a/apps/worker/tests/contract/test_page_memory_asset_java_contract.py b/apps/worker/tests/contract/test_page_memory_asset_java_contract.py index 78f13946b..3d809c3d8 100644 --- a/apps/worker/tests/contract/test_page_memory_asset_java_contract.py +++ b/apps/worker/tests/contract/test_page_memory_asset_java_contract.py @@ -2,6 +2,7 @@ import json import os +from types import SimpleNamespace os.environ.setdefault("DATABASE_URL", "postgresql+asyncpg://test:test@localhost/test") os.environ.setdefault("TMP_PATH", "/tmp/knowhere-test") @@ -175,3 +176,45 @@ def test_page_assets_adds_java_home_to_path(monkeypatch, tmp_path) -> None: assert page_assets._has_working_java() is True # noqa: SLF001 assert os.environ["PATH"].split(os.pathsep)[0] == str(java_bin) + + +def test_select_rendered_pages_with_assets_keeps_only_has_asset_pages() -> None: + from app.services.page_memory.memory_service import ( + _select_rendered_pages_with_assets, + ) + + rendered = [ + PageRenderResult( + page_index=1, + image_path="/tmp/1.png", + raw_text="", + width=10, + height=10, + is_landscape=False, + ), + PageRenderResult( + page_index=2, + image_path="/tmp/2.png", + raw_text="", + width=10, + height=10, + is_landscape=False, + ), + PageRenderResult( + page_index=3, + image_path="/tmp/3.png", + raw_text="", + width=10, + height=10, + is_landscape=False, + ), + ] + page_features = [ + SimpleNamespace(page=1, has_asset=False), + SimpleNamespace(page=2, has_asset=True), + SimpleNamespace(page=3, has_asset=True), + ] + + selected = _select_rendered_pages_with_assets(rendered, page_features) + + assert [item.page_index for item in selected] == [2, 3] diff --git a/apps/worker/tests/contract/test_page_memory_collapse_single_child_contract.py b/apps/worker/tests/contract/test_page_memory_collapse_single_child_contract.py index 0401b2f0a..dec0f13c4 100644 --- a/apps/worker/tests/contract/test_page_memory_collapse_single_child_contract.py +++ b/apps/worker/tests/contract/test_page_memory_collapse_single_child_contract.py @@ -213,8 +213,8 @@ def test_generic_label_merges_regardless_of_content() -> None: # ── Test 5: result is sorted consistently ──────────────────────────── -def test_result_sorted_by_start_page_level_path() -> None: - """Output respects ``_sort_skeletons`` ordering so downstream is stable.""" +def test_result_sorted_by_start_page_preserving_same_page_order() -> None: + """Output is ordered by start_page; same-page relative order is preserved.""" skeletons = [ _skel( "doc.pdf/Z", diff --git a/apps/worker/tests/contract/test_page_memory_fine_hierarchy_contract.py b/apps/worker/tests/contract/test_page_memory_fine_hierarchy_contract.py index 4eac1d249..c78ba9dc5 100644 --- a/apps/worker/tests/contract/test_page_memory_fine_hierarchy_contract.py +++ b/apps/worker/tests/contract/test_page_memory_fine_hierarchy_contract.py @@ -187,3 +187,162 @@ def test_refine_fat_leaf_skeletons_uses_page_memory_prompt_without_demoting_sibl assert refined[1].end_page == 227 assert refined[2].parent_path.endswith("/2 术语") assert all(item.title != skeleton.title for item in refined) + + +def test_refine_fat_leaf_keeps_slash_in_title_as_single_path_segment( + monkeypatch, +) -> None: + skeleton = SectionSkeleton( + section_path="manual.pdf/Index", + level=1, + start_page=10, + end_page=14, + title="Index", + parent_path="manual.pdf", + ) + tags = [ + PageTagResult( + page_index=10, + observed_titles=[ + {"text": "Index", "prominence": 1.0}, + {"text": "Symbols/Numbers", "prominence": 1.0}, + ], + ), + PageTagResult( + page_index=12, + observed_titles=[{"text": "A entries", "prominence": 1.0}], + ), + ] + + monkeypatch.setattr( + fine_hierarchy, + "get_text_client", + lambda requested_model=None: ( + _FakeClient( + [ + {"id": 1, "level": 1}, + {"id": 2, "level": 1}, + ] + ), + requested_model, + ), + ) + + refined = fine_hierarchy.refine_fat_leaf_skeletons( + coarse_skeletons=[skeleton], + tag_results=tags, + fat_leaf_pages={10, 11, 12, 13, 14}, + model_name="test-model", + ) + + assert [item.title for item in refined] == ["Symbols/Numbers", "A entries"] + assert refined[0].section_path == "manual.pdf/Index/Symbols\u2215Numbers" + assert refined[1].parent_path == "manual.pdf/Index" + assert refined[0].section_path.count("/") == 2 + + +def test_build_next_title_by_path_preserves_parent_before_same_page_child() -> None: + parent = SectionSkeleton( + section_path="demo.pdf/Section Z Parent", + level=1, + start_page=10, + end_page=12, + title="Section Z Parent", + parent_path="demo.pdf", + evidence={"skeleton_kind": "parent_self_only"}, + ) + # Alphabetically sorts before the parent path; stable start_page order + # must still keep emit order (parent first). + child = SectionSkeleton( + section_path="demo.pdf/Section Z Parent/A First Child", + level=2, + start_page=10, + end_page=12, + title="A First Child", + parent_path="demo.pdf/Section Z Parent", + ) + sibling = SectionSkeleton( + section_path="demo.pdf/Later", + level=1, + start_page=13, + end_page=15, + title="Later", + parent_path="demo.pdf", + ) + + next_map = fine_hierarchy.build_next_title_by_path([parent, child, sibling]) + + assert next_map[parent.section_path] == "A First Child" + assert next_map[child.section_path] == "Later" + assert next_map[sibling.section_path] is None + + +def test_trim_drops_boundary_page_when_end_anchor_missing() -> None: + raw = [ + {"heading": "Keep", "page": 10, "key": "keep"}, + {"heading": "Boundary Leftover", "page": 12, "key": "boundaryleftover"}, + {"heading": "Also Boundary", "page": 12, "key": "alsoboundary"}, + ] + + trimmed = fine_hierarchy._trim_by_coarse_anchors( + raw, + start_title="", + end_title="Next Section Title", + section_path="demo.pdf/Current", + boundary_page=12, + ) + + assert [item["heading"] for item in trimmed] == ["Keep"] + + +def test_refine_fat_leaf_drops_boundary_page_on_tail_miss(monkeypatch) -> None: + previous = SectionSkeleton( + section_path="demo.pdf/History", + level=2, + start_page=14, + end_page=23, + title="History", + parent_path="demo.pdf", + ) + next_section = SectionSkeleton( + section_path="demo.pdf/List of amendments", + level=2, + start_page=23, + end_page=30, + title="List of amendments", + parent_path="demo.pdf", + ) + tags = [ + PageTagResult( + page_index=22, + observed_titles=[{"text": "NCC 2022", "prominence": 1.0}], + ), + PageTagResult( + page_index=23, + observed_titles=[ + {"text": "Not The Next Title", "prominence": 1.0}, + {"text": "Another Boundary Title", "prominence": 0.9}, + ], + ), + ] + + monkeypatch.setattr( + fine_hierarchy, + "get_text_client", + lambda requested_model=None: ( + _FakeClient([{"id": 1, "level": 1}]), + requested_model, + ), + ) + + refined = fine_hierarchy.refine_fat_leaf_skeletons( + coarse_skeletons=[previous], + tag_results=tags, + fat_leaf_pages={22, 23}, + next_title_by_path={previous.section_path: next_section.title}, + model_name="test-model", + ) + + assert [item.title for item in refined] == ["NCC 2022"] + assert all("amendment" not in item.title.casefold() for item in refined) + assert all(item.start_page != 23 for item in refined) diff --git a/apps/worker/tests/contract/test_page_memory_navigation_contract.py b/apps/worker/tests/contract/test_page_memory_navigation_contract.py index 23e60c289..47913989c 100644 --- a/apps/worker/tests/contract/test_page_memory_navigation_contract.py +++ b/apps/worker/tests/contract/test_page_memory_navigation_contract.py @@ -44,6 +44,7 @@ def test_page_doc_nav_uses_path_based_leaf_summaries_and_page_counts() -> None: def test_zip_result_service_builds_navigation_from_chunk_paths() -> None: doc_nav, hierarchy = ZipResultService()._build_navigation_outputs( # noqa: SLF001 + add_dir="", formatted_chunks=_page_chunks(), source_file_name="demo.pdf", ) @@ -57,6 +58,41 @@ def test_zip_result_service_builds_navigation_from_chunk_paths() -> None: } +def test_zip_result_service_prefers_enriched_on_disk_doc_nav(tmp_path) -> None: + enriched = { + "version": "1.0", + "file_name": "demo.pdf", + "top_summary": "Document overview from enrich", + "stats": {}, + "sections": [ + { + "title": "Kept From Disk", + "path": "demo.pdf/Kept From Disk", + "summary": "enriched section", + "chunk_count": 1, + "children": [], + } + ], + "resources": {"images": [], "tables": []}, + } + (tmp_path / "doc_nav.json").write_text( + __import__("json").dumps(enriched, ensure_ascii=False), + encoding="utf-8", + ) + + doc_nav, hierarchy = ZipResultService()._build_navigation_outputs( # noqa: SLF001 + add_dir=str(tmp_path), + formatted_chunks=_page_chunks(), + source_file_name="demo.pdf", + ) + + assert doc_nav is not None + assert doc_nav["top_summary"] == "Document overview from enrich" + assert doc_nav["sections"][0]["title"] == "Kept From Disk" + assert hierarchy == {"Kept From Disk": {}} + + + def _page_chunks() -> list[dict[str, object]]: return [ { diff --git a/apps/worker/tests/contract/test_page_memory_page_tagger_contract.py b/apps/worker/tests/contract/test_page_memory_page_tagger_contract.py index 8eeba6360..ae45c0a56 100644 --- a/apps/worker/tests/contract/test_page_memory_page_tagger_contract.py +++ b/apps/worker/tests/contract/test_page_memory_page_tagger_contract.py @@ -12,7 +12,13 @@ os.environ.setdefault("S3_TEMP_PATH", "/tmp") from app.services.page_memory.page_renderer import PageRenderResult -from app.services.page_memory.page_tagger import PageTagResult, tag_page_titles +from app.services.page_memory.page_tagger import ( + PageTagResult, + _completion_tokens, + _tag_vlm_titles, + _title_response_truncated, + tag_page_titles, +) from shared.core.exceptions.domain_exceptions import UnavailableException @@ -22,6 +28,81 @@ def _write_page_image(tmp_path, page_index: int) -> str: return str(image_path) +def test_title_response_truncated_detects_budget_hit_and_incomplete_json() -> None: + assert _title_response_truncated( + '{"titles":[', + usage={"completion_tokens": 300}, + max_tokens=300, + ) + assert _title_response_truncated( + '{"titles":[{"text":"A"', + usage={"completion_tokens": 120}, + max_tokens=300, + ) + assert not _title_response_truncated( + '{"titles":[]}', + usage={"completion_tokens": 40}, + max_tokens=300, + ) + assert _completion_tokens({"completion_tokens": 12}) == 12 + + +def test_title_detection_escalates_token_budget_on_truncated_json( + monkeypatch, + tmp_path, +) -> None: + truncated = ( + '{\n "titles": [\n' + " {\n" + ' "text": "Section A Governing requirements",\n' + ' "prominence": 1.0,\n' + ' "is_in_table": false,\n' + ) + complete = ( + '{\n "titles": [\n' + " {\n" + ' "text": "Section A Governing requirements",\n' + ' "prominence": 1.0,\n' + ' "is_in_table": false,\n' + ' "is_in_header_footer": false\n' + " }\n" + " ]\n" + "}" + ) + calls: list[int] = [] + + class _FakeClient: + def chat_completion_with_usage(self, **kwargs): + max_tokens = int(kwargs["max_tokens"]) + calls.append(max_tokens) + if max_tokens == 300: + return truncated, {"completion_tokens": 300, "prompt_tokens": 10} + return complete, {"completion_tokens": 180, "prompt_tokens": 10} + + monkeypatch.setattr( + "shared.services.ai.llm_overrides.get_vision_client", + lambda requested_model=None: (_FakeClient(), requested_model or "fake-vlm"), + ) + monkeypatch.setattr( + "app.services.page_memory.page_tagger.build_prompt", + lambda *args, **kwargs: ("prompt", 0.0, 0.01, 300), + ) + + page = PageRenderResult( + page_index=38, + image_path=_write_page_image(tmp_path, 38), + raw_text="", + width=100, + height=200, + is_landscape=False, + ) + observed = _tag_vlm_titles(page, model="fake-vlm") + assert calls == [300, 600] + assert [item["text"] for item in observed] == [ + "Section A Governing requirements" + ] + + def test_title_detection_preserves_page_index_assignment_under_concurrency( monkeypatch, tmp_path, @@ -32,6 +113,7 @@ def _fake_tag_vlm_titles( page: PageRenderResult, *, model: str, + scan_direction: str = "top_to_bottom_left_to_right", ) -> list[dict[str, object]]: gevent.sleep(0.01 * (4 - page.page_index)) return [{"text": f"title-{page.page_index}", "prominence": 0.8}] @@ -82,6 +164,7 @@ def _fake_tag_vlm_titles( page: PageRenderResult, *, model: str, + scan_direction: str = "top_to_bottom_left_to_right", ) -> list[dict[str, object]]: if page.page_index == 2: raise RuntimeError("title detection failed") @@ -123,6 +206,7 @@ def _fake_tag_vlm_titles( page: PageRenderResult, *, model: str, + scan_direction: str = "top_to_bottom_left_to_right", ) -> list[dict[str, object]]: raise UnavailableException( internal_message="capacity busy", diff --git a/apps/worker/tests/contract/test_parse_task_contract.py b/apps/worker/tests/contract/test_parse_task_contract.py index 59cde7310..696c5e766 100644 --- a/apps/worker/tests/contract/test_parse_task_contract.py +++ b/apps/worker/tests/contract/test_parse_task_contract.py @@ -483,6 +483,58 @@ def test_parse_task_should_refund_charged_job_when_uploaded_file_cannot_be_parse ] +def test_parse_task_should_report_invalid_docx_as_client_file_error( + worker_contract_environment: None, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + contract = WorkerParseContract.create() + contract.use_workspace_root(monkeypatch, tmp_path) + contract.use_billing(monkeypatch, is_enabled=True) + + invalid_docx_path = tmp_path / "invalid.docx" + invalid_docx_path.write_bytes(b"this is not a docx package") + job = contract.create_file_job( + source_file_name="contract-invalid.docx", + job_id_prefix="job_invalid_docx", + ) + contract.upload_source_file( + local_file_path=invalid_docx_path, + s3_key=job["s3_key"], + ) + + celery_result = contract.enqueue_parse_task( + job_id=job["job_id"], + user_id=job["user_id"], + ) + + assert celery_result.failed() + assert contract.find_task_workspaces(tmp_path, job["job_id"]) == [] + + job_row = contract.observe_job_status(job["job_id"]) + assert job_row["status"] == "failed" + assert job_row["billing_status"] == "refunded" + assert job_row["credits_charged"] == int(contract.settings.MICRO_DOLLARS_PER_PAGE) + assert job_row["error_code"] == "INVALID_ARGUMENT" + assert ( + job_row["error_message"] + == "Invalid file: the uploaded .docx file is not a valid Word document. Please check the file and upload again." + ) + assert contract.count_job_results(job["job_id"]) == 0 + + metadata = contract.get_job_metadata(job["job_id"]) + assert metadata["error_details"] == { + "violations": [ + { + "field": "file", + "description": ( + "Expected a valid DOCX ZIP package containing word/document.xml" + ), + } + ] + } + + def test_should_reject_pdf_when_page_count_exceeds_configured_limit( worker_contract_environment: None, monkeypatch: pytest.MonkeyPatch, diff --git a/apps/worker/tests/contract/test_table_embedded_images_contract.py b/apps/worker/tests/contract/test_table_embedded_images_contract.py new file mode 100644 index 000000000..0da70fad8 --- /dev/null +++ b/apps/worker/tests/contract/test_table_embedded_images_contract.py @@ -0,0 +1,234 @@ +from __future__ import annotations + +import os +from pathlib import Path + +os.environ.setdefault("DATABASE_URL", "postgresql+asyncpg://test:test@localhost/test") +os.environ.setdefault("TMP_PATH", "/tmp/knowhere-test") +os.environ.setdefault("S3_BUCKET_NAME", "test-uploads") +os.environ.setdefault("S3_ACCESS_KEY_ID", "test") +os.environ.setdefault("S3_SECRET_ACCESS_KEY", "test") +os.environ.setdefault("S3_TEMP_PATH", "/tmp") + +from PIL import Image # noqa: E402 + +from app.services.document_parser.formats.markdown.parse_state import ( # noqa: E402 + MarkdownParseState, +) +from app.services.document_parser.formats.markdown.parser import ( # noqa: E402 + update_df_list, +) +from app.services.document_parser.formats.markdown.table_asset import ( # noqa: E402 + MarkdownTableAssetRequest, + build_markdown_table_asset, +) +from app.services.document_parser.formats.markdown.table_embedded_images import ( # noqa: E402 + extract_table_embedded_images, +) +from app.services.document_parser.orchestration.postprocess import ( # noqa: E402 + cleanup_unreferenced_images, +) +from shared.core.constants.processing import ProcessingConstants # noqa: E402 +from shared.services.chunks.dataframe_chunk_converter import ( # noqa: E402 + dataframe_to_chunks, +) +from shared.services.retrieval.agentic.evidence.renderer import ( # noqa: E402 + render_table_chunk_lines, +) + + +def _write_jpeg( + path: Path, + *, + size: tuple[int, int] = (640, 640), + min_bytes: int | None = None, +) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + # Solid colors compress too well; patterned noise keeps file size realistic. + width, height = size + pixels = bytearray(width * height * 3) + for index in range(len(pixels)): + pixels[index] = (index * 37) % 256 + Image.frombytes("RGB", size, bytes(pixels)).save(path, format="JPEG", quality=95) + if min_bytes is not None: + current_size = path.stat().st_size + if current_size < min_bytes: + with path.open("ab") as handle: + handle.write(b"\x00" * (min_bytes - current_size)) + + +def _make_parser_state() -> MarkdownParseState: + return MarkdownParseState( + relative_root="doc.pdf", + split_char="/", + llm_parameters={ + "summary_image": False, + "summary_table": False, + "summary_txt": False, + "stopwords": set(), + }, + timestamp="2026-01-01 00:00:00", + row_updater=update_df_list, + ) + + +def test_table_embedded_images_are_extracted_rewritten_and_linked( + tmp_path: Path, +) -> None: + output_dir = tmp_path / "doc" + images_dir = output_dir / "images" + tables_dir = output_dir / "tables" + images_dir.mkdir(parents=True) + tables_dir.mkdir(parents=True) + + hash_name = "a" * 64 + ".jpg" + image_path = images_dir / hash_name + _write_jpeg(image_path, min_bytes=ProcessingConstants.IMG_MIN_SIZE) + assert image_path.stat().st_size >= ProcessingConstants.IMG_MIN_SIZE + + table_html = ( + "" + f'' + "
流程名称评审流程图
说明审批节点
" + ) + + parser_state = _make_parser_state() + + embedded = extract_table_embedded_images( + table_html=table_html, + parser_state=parser_state, + output_dir=str(output_dir), + image_dir=str(images_dir), + summary_image=False, + ) + + assert len(embedded.image_assets) == 1 + assert len(embedded.image_refs) == 1 + assert hash_name not in embedded.rewritten_html + assert " None: + output_dir = tmp_path / "doc" + images_dir = output_dir / "images" + images_dir.mkdir(parents=True) + + hash_name = "c" * 64 + ".jpg" + image_path = images_dir / hash_name + _write_jpeg(image_path, size=(16, 16)) + assert image_path.stat().st_size < ProcessingConstants.IMG_MIN_SIZE + + table_html = ( + "" + f'
Logo
' + ) + + embedded = extract_table_embedded_images( + table_html=table_html, + parser_state=_make_parser_state(), + output_dir=str(output_dir), + image_dir=str(images_dir), + summary_image=False, + ) + + assert embedded.image_assets == [] + assert embedded.image_refs == [] + assert " Dict[str, Any]: + node: Dict[str, Any] = {"title": title, "summary": summary, "children": []} + if path: + node["path"] = path + return node + + +def _parent( + title: str, + children: List[Dict[str, Any]], + *, + path: str = "", + summary: str = "", +) -> Dict[str, Any]: + node: Dict[str, Any] = { + "title": title, + "summary": summary, + "children": children, + } + if path: + node["path"] = path + return node + + +class TestDeterministicAssembly: + def test_order_covers_self_only_then_titles(self) -> None: + text = _deterministic_section_summary( + is_top_level=False, + self_only="intro paragraph here", + child_titles=["Alpha", "Beta"], + ) + assert text.startswith("This section covers: ") + assert "intro paragraph here" in text + assert "Alpha, Beta" in text + # self_only before titles + assert text.index("intro paragraph here") < text.index("Alpha, Beta") + + def test_all_child_titles_even_when_summary_empty(self) -> None: + parent = _parent( + "Parent", + [ + _leaf("HasText", summary="body"), + _leaf("NoSummary", summary=""), + ], + path="doc.pdf/Parent", + ) + result = _recursive_summarize_nav( + parent, + use_llm=False, + source_file_name="doc.pdf", + ) + assert "HasText" in result + assert "NoSummary" in result + assert result.startswith("This section covers: ") + + +class TestSelfOnlyLookup: + def test_exact_path_only_excludes_descendants(self) -> None: + chunks = [ + { + "path": "doc.pdf/2.4.4 隐患治理", + "content": "PARENT_INTRO_ONLY", + }, + { + "path": "doc.pdf/2.4.4 隐患治理/清单项A", + "content": "CHILD_BODY_SHOULD_NOT_APPEAR", + }, + ] + lookup = build_self_only_lookup(chunks, source_file_name="doc.pdf") + assert lookup["2.4.4 隐患治理"] == "PARENT_INTRO_ONLY" + assert "CHILD_BODY_SHOULD_NOT_APPEAR" not in lookup["2.4.4 隐患治理"] + assert "2.4.4 隐患治理 / 清单项A" in lookup + + def test_nonleaf_includes_self_only_in_deterministic(self) -> None: + parent = _parent( + "2.4.4 隐患治理", + [ + _leaf("清单项A", summary="a", path="doc.pdf/2.4.4 隐患治理/清单项A"), + _leaf("清单项B", summary="b", path="doc.pdf/2.4.4 隐患治理/清单项B"), + ], + path="doc.pdf/2.4.4 隐患治理", + ) + lookup = build_self_only_lookup( + [{"path": "doc.pdf/2.4.4 隐患治理", "content": "方案包括以下内容:"}], + source_file_name="doc.pdf", + ) + result = _recursive_summarize_nav( + parent, + use_llm=False, + self_only_lookup=lookup, + source_file_name="doc.pdf", + ) + assert "方案包括以下内容:" in result + assert "清单项A" in result + assert "清单项B" in result + assert parent.get("self_summary") == "方案包括以下内容:" + + +class TestLlmTrigger: + def test_short_contrib_skips_llm(self, monkeypatch: pytest.MonkeyPatch) -> None: + called = {"n": 0} + + def _boom(**kwargs: Any) -> str: + called["n"] += 1 + return "SHOULD_NOT_USE" + + monkeypatch.setattr( + "app.services.connect_builder.summary_builder._llm_summarize", + _boom, + ) + parent = _parent( + "P", + [_leaf("A", summary="x"), _leaf("B", summary="y")], + path="doc.pdf/P", + ) + result = _recursive_summarize_nav(parent, use_llm=True, source_file_name="doc.pdf") + assert called["n"] == 0 + assert result.startswith("This section covers: ") + assert "A" in result and "B" in result + + def test_long_contrib_calls_llm_with_title_for_empty_summary( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: + captured: Dict[str, Any] = {} + + def _fake_llm(**kwargs: Any) -> str: + captured.update(kwargs) + return "LLM_SUMMARY" + + monkeypatch.setattr( + "app.services.connect_builder.summary_builder._llm_summarize", + _fake_llm, + ) + long_a = "A" * (SUMMARY_MAX_LEN + 5) + parent = _parent( + "P", + [ + _leaf("HasSummary", summary=long_a), + _leaf("EmptySummary", summary=""), + ], + path="doc.pdf/P", + ) + lookup = {"P": "SELF_ONLY_INTRO"} + # section path from doc.pdf/P is "P" + result = _recursive_summarize_nav( + parent, + use_llm=True, + self_only_lookup=lookup, + source_file_name="doc.pdf", + ) + assert result == "LLM_SUMMARY" + assert captured["self_only"] == "SELF_ONLY_INTRO" + titles = [t for t, _ in captured["child_rows"]] + contribs = {t: c for t, c in captured["child_rows"]} + assert "EmptySummary" in titles + assert contribs["EmptySummary"] == "EmptySummary" + assert contribs["HasSummary"] == long_a + + def test_single_child_with_self_only_does_not_copy_child( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: + monkeypatch.setattr( + "app.services.connect_builder.summary_builder._llm_summarize", + lambda **kwargs: "MERGED", + ) + long_child = "C" * (SUMMARY_MAX_LEN + 1) + parent = _parent( + "P", + [_leaf("OnlyChild", summary=long_child)], + path="doc.pdf/P", + ) + result = _recursive_summarize_nav( + parent, + use_llm=True, + self_only_lookup={"P": "intro"}, + source_file_name="doc.pdf", + ) + assert result == "MERGED" + assert result != long_child + + +class TestPromptPayload: + def test_file_summary_prompt_contains_scope_blocks( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: + captured: Dict[str, Any] = {} + + def _fake_client(**_kwargs: Any) -> Any: + class _C: + def chat_completion(self, **kwargs: Any) -> str: + captured["messages"] = kwargs.get("messages") + return "ok" + + return _C() + + monkeypatch.setattr( + "shared.services.ai.openai_compatible_client_sync.get_openai_client", + _fake_client, + ) + # Ensure build_prompt path works + out = _llm_summarize( + node_name="Parent", + self_only="intro text", + child_rows=[("ChildA", "summary A"), ("ChildB", "ChildB")], + max_tokens=100, + ) + assert out == "ok" + user = captured["messages"][1]["content"] + assert "SCOPE_TITLE: Parent" in user + assert "SELF_ONLY_CONTENT:" in user + assert "intro text" in user + assert "COVERED_NODES:" in user + assert "[ChildA] summary A" in user + assert "[ChildB] ChildB" in user + # legacy flat blob prompt removed + assert "You will receive summaries of sub-sections" not in user + + +class TestDocNavTopSummaryPersistence: + def test_enrich_persists_top_summary_and_defaults_top_llm( + self, tmp_path, monkeypatch: pytest.MonkeyPatch + ) -> None: + import json + + from app.services.connect_builder.summary_builder import ( + enrich_doc_nav_summaries, + load_nav_top_summary, + ) + + captured: Dict[str, Any] = {} + + def _fake_llm(**kwargs: Any) -> str: + captured["is_top"] = kwargs.get("node_name") == "Document Overview" + captured["max_tokens"] = kwargs.get("max_tokens") + captured["calls"] = int(captured.get("calls") or 0) + 1 + return "LLM document overview" + + monkeypatch.setattr( + "app.services.connect_builder.summary_builder._llm_summarize", + _fake_llm, + ) + + file_dir = tmp_path / "report.pdf" + file_dir.mkdir() + long_leaf = "L" * (SUMMARY_MAX_LEN + 5) + doc_nav = { + "version": "1.0", + "file_name": "report.pdf", + "stats": {}, + "sections": [ + { + "title": "Chapter 1", + "path": "report.pdf/Chapter 1", + "summary": long_leaf, + "chunk_count": 1, + "children": [], + }, + { + "title": "Chapter 2", + "path": "report.pdf/Chapter 2", + "summary": long_leaf, + "chunk_count": 1, + "children": [], + }, + ], + "resources": {"images": [], "tables": []}, + } + (file_dir / "doc_nav.json").write_text( + json.dumps(doc_nav, ensure_ascii=False), + encoding="utf-8", + ) + + results = enrich_doc_nav_summaries( + str(tmp_path), + source_file="report.pdf", + use_llm=False, + top_summary_use_llm=True, + ) + assert results["report.pdf"] == "LLM document overview" + assert captured["calls"] == 1 + assert captured["is_top"] is True + + saved = json.loads((file_dir / "doc_nav.json").read_text(encoding="utf-8")) + assert saved["top_summary"] == "LLM document overview" + # Section leaves keep original summaries; top LLM must not rewrite them. + assert saved["sections"][0]["summary"] == long_leaf + assert load_nav_top_summary(str(file_dir), "report.pdf") == ( + "LLM document overview" + ) diff --git a/packages/shared-python/shared/core/config/ai.py b/packages/shared-python/shared/core/config/ai.py index 33d711fd2..b2f612d2c 100644 --- a/packages/shared-python/shared/core/config/ai.py +++ b/packages/shared-python/shared/core/config/ai.py @@ -44,10 +44,6 @@ class AIConfig(BaseModel): "Same alternates as IMAGE_MODEL (e.g. qwen3-vl-32b-instruct)." ), ) - RETRIEVAL_DECOMPOSITION_ENABLED: bool = Field( - default=False, - description="Enable query-decomposition workflow before agentic retrieval.", - ) RETRIEVAL_PLANNER_MODEL: str = Field( default="", description="Reasoning-capable model used by the workflow query planner.", @@ -72,13 +68,6 @@ class AIConfig(BaseModel): default=3, description="Maximum concurrent workflow steps in the same DAG batch.", ) - RETRIEVAL_AGENTIC_INLINE_TABLE_CHAR_LIMIT: int = Field( - default=10000, - description=( - "Maximum table HTML characters inlined into agentic evidence_text. " - "Larger tables are represented by asset URL, path, summary, and keywords." - ), - ) # Runtime LLM controls. LLM_MOCK_ENABLED: bool = Field( diff --git a/packages/shared-python/shared/core/config/database.py b/packages/shared-python/shared/core/config/database.py index d78629284..a9baac5f5 100644 --- a/packages/shared-python/shared/core/config/database.py +++ b/packages/shared-python/shared/core/config/database.py @@ -25,8 +25,10 @@ class DatabaseConfig(BaseModel): ) # Async database pool configuration for the API. - DB_POOL_SIZE: int = Field(default=20, description="Connection-pool size") - DB_MAX_OVERFLOW: int = Field(default=30, description="Maximum overflow connections") + # Defaults sized for legal job-poll bursts after nested-checkout hygiene + # (pool_size + max_overflow = 100 checkouts per API process). + DB_POOL_SIZE: int = Field(default=50, description="Connection-pool size") + DB_MAX_OVERFLOW: int = Field(default=50, description="Maximum overflow connections") DB_POOL_RECYCLE: int = Field( default=1800, description="Connection recycle interval in seconds" ) diff --git a/packages/shared-python/shared/core/config/storage.py b/packages/shared-python/shared/core/config/storage.py index 1706b2d2f..f8a31c393 100644 --- a/packages/shared-python/shared/core/config/storage.py +++ b/packages/shared-python/shared/core/config/storage.py @@ -82,9 +82,11 @@ class StorageConfig(BaseModel): PDF_PROFILE_TOC_ENABLED: bool = Field( default=False, description=( - "Enable page-owned PDF TOC profiling during parser-entry DOC_PROFILE. " - "When disabled, PDF parsing treats documents as no-TOC and does not " - "fall back to Markdown TOC detection." + "Enable page-owned PDF TOC profiling for the CHUNK track during " + "parser-entry DOC_PROFILE. When disabled, chunk-track PDF parsing " + "treats documents as no-TOC and does not fall back to Markdown TOC " + "detection. NOTE: the page-memory track always forces TOC profiling " + "on regardless of this flag, since its sections are TOC-anchored." ), ) MINERU_SHARD_CONCURRENCY: int = Field( diff --git a/packages/shared-python/shared/core/database.py b/packages/shared-python/shared/core/database.py index cd8a03d6e..bfea49902 100644 --- a/packages/shared-python/shared/core/database.py +++ b/packages/shared-python/shared/core/database.py @@ -12,7 +12,7 @@ create_async_engine, ) from sqlalchemy.orm import declarative_base -from sqlalchemy.pool import NullPool +from sqlalchemy.pool import NullPool, QueuePool from shared.core.config import settings from shared.core.constants import ProcessingConstants @@ -25,7 +25,7 @@ os.getenv("DB_USE_NULL_POOL", "false").lower() == "true" ) engine_options: dict[str, Any] = { - "pool_recycle": ProcessingConstants.DB_POOL_RECYCLE, + "pool_recycle": settings.DB_POOL_RECYCLE, "pool_pre_ping": ProcessingConstants.DB_POOL_PRE_PING, "pool_reset_on_return": ProcessingConstants.DB_POOL_RESET_ON_RETURN, "connect_args": { @@ -46,9 +46,9 @@ else: engine_options.update( { - "pool_size": ProcessingConstants.DB_POOL_SIZE, - "max_overflow": ProcessingConstants.DB_MAX_OVERFLOW, - "pool_timeout": ProcessingConstants.DB_POOL_TIMEOUT, + "pool_size": settings.DB_POOL_SIZE, + "max_overflow": settings.DB_MAX_OVERFLOW, + "pool_timeout": settings.DB_POOL_TIMEOUT, } ) @@ -204,12 +204,36 @@ def on_connect(dbapi_connection, connection_record): """Handle new connection events.""" logger.info("New database connection established") - @event.listens_for(engine.sync_engine, "checkout") + @event.listens_for(engine.sync_engine.pool, "checkout") def on_checkout(dbapi_connection, connection_record, connection_proxy): - """Handle connection checkout events.""" - logger.debug("Connection checked out from pool") - - @event.listens_for(engine.sync_engine, "checkin") + """Handle connection checkout events and surface pool pressure.""" + pool = engine.sync_engine.pool + if not isinstance(pool, QueuePool): + logger.debug("Connection checked out from pool") + return + checked_out = pool.checkedout() + overflow = pool.overflow() + pool_size = pool.size() + logger.debug( + "Connection checked out from pool " + "(checkedout=%s overflow=%s pool_size=%s)", + checked_out, + overflow, + pool_size, + ) + # QueuePool does not expose a first-party wait-started hook; treat + # checkedout >= pool_size (overflow in use) as pressure / likely wait. + if checked_out >= pool_size: + logger.warning( + "Database pool under pressure: checkedout=%s overflow=%s " + "pool_size=%s max_overflow=%s (checkout waits may exceed 1s)", + checked_out, + overflow, + pool_size, + settings.DB_MAX_OVERFLOW, + ) + + @event.listens_for(engine.sync_engine.pool, "checkin") def on_checkin(dbapi_connection, connection_record): """Handle connection check-in events.""" logger.debug("Connection checked in to pool") @@ -219,6 +243,7 @@ def on_invalidate(dbapi_connection, connection_record, exception): """Handle connection invalidation events.""" logger.warning(f"Database connection invalidated: {exception}") + setup_pool_event_listeners() @@ -231,7 +256,7 @@ async def prewarm_connection_pool(): logger.info("Starting connection pool prewarming...") try: # Warm the base connection pool. - connections_to_warm = min(ProcessingConstants.DB_POOL_SIZE, 5) + connections_to_warm = min(settings.DB_POOL_SIZE, 5) tasks = [] for _ in range(connections_to_warm): diff --git a/packages/shared-python/shared/core/state_machine/service_sync.py b/packages/shared-python/shared/core/state_machine/service_sync.py index 6d8ba1cc1..c820ef4b0 100644 --- a/packages/shared-python/shared/core/state_machine/service_sync.py +++ b/packages/shared-python/shared/core/state_machine/service_sync.py @@ -36,7 +36,10 @@ ) from shared.models.database.job import Job from shared.models.database.job_state_audit_log import JobStateAuditLog -from shared.services.redis.redis_sync_service import SyncRedisServiceFactory +from shared.services.redis.redis_sync_service import ( + SyncJobMetadataService, + SyncRedisServiceFactory, +) from shared.services.redis.key_builder import RedisKeyType, redis_key_builder @@ -412,6 +415,7 @@ def _update_job_error( "error_message": error_message, "error_code": error_code, } + metadata_updates: Dict[str, Any] = {} if error_details: import json as _json @@ -425,8 +429,14 @@ def _update_job_error( pg_array(["error_details"]), cast(literal(_json.dumps(error_details)), JSONB), ) + metadata_updates["error_details"] = error_details db.execute(update(Job).where(Job.job_id == job_id).values(**update_values)) + if metadata_updates: + SyncJobMetadataService(self.redis).update_metadata( + job_id, + metadata_updates, + ) except Exception as e: logger.error(f"Failed to update Job {job_id} error info: {e}") diff --git a/packages/shared-python/shared/models/schemas/job.py b/packages/shared-python/shared/models/schemas/job.py index 2a948f792..1ea775fde 100644 --- a/packages/shared-python/shared/models/schemas/job.py +++ b/packages/shared-python/shared/models/schemas/job.py @@ -38,6 +38,14 @@ class ParsingParams(BaseModel): "Increases parse time and API token cost." ), ) + top_summary_use_llm: bool = Field( + True, + description=( + "Use LLM for the document-level top_summary written to doc_nav. " + "Defaults to True because agentic document selection consumes this " + "field. Section-level summaries remain controlled by summary_use_llm." + ), + ) class JobCreateBase(BaseModel): diff --git a/packages/shared-python/shared/models/schemas/page_memory_config.py b/packages/shared-python/shared/models/schemas/page_memory_config.py index bbac71ed3..95c2122d1 100644 --- a/packages/shared-python/shared/models/schemas/page_memory_config.py +++ b/packages/shared-python/shared/models/schemas/page_memory_config.py @@ -3,7 +3,7 @@ from __future__ import annotations from dataclasses import asdict, dataclass -from typing import Literal, Self +from typing import Self @dataclass(frozen=True) @@ -15,12 +15,11 @@ class PageMemoryConfig: tag_concurrency: int = 4 title_detection_concurrency: int = 3 node_assembly_concurrency: int = 3 - tag_mode: Literal["vlm", "text"] = "vlm" fine_min_pages: int = 4 hierarchy_model: str | None = None hierarchy_max_tokens: int = 2000 max_heading_depth: int = 6 - asset_extraction_enabled: bool = False + asset_extraction_enabled: bool = True asset_summary_enabled: bool = False asset_model: str = "qwen3.6-flash" asset_max_pages: int | None = None @@ -29,11 +28,7 @@ class PageMemoryConfig: table_engine: str = "tabula" table_merge_enabled: bool = True node_summary_max_pages: int = 5 - page_locate_residual_agent_limit: int = 50 - page_locate_max_emit_depth: int = 5 - page_locate_min_emit_depth: int = 2 - page_locate_vlm_candidate_page_cap: int = 4 - page_locate_full_leaf_sections: bool = False + scan_direction: str = "top_to_bottom_left_to_right" @classmethod def default(cls) -> Self: @@ -64,10 +59,6 @@ def from_mapping(cls, value: object) -> Self: return cls.default() default = cls.default() - tag_mode = str(value.get("tag_mode", default.tag_mode)).strip().lower() - resolved_tag_mode: Literal["vlm", "text"] = ( - "text" if tag_mode == "text" else "vlm" - ) return cls( max_pages=_as_int(value.get("max_pages"), default.max_pages), scope_concurrency=_as_int( @@ -86,7 +77,6 @@ def from_mapping(cls, value: object) -> Self: value.get("node_assembly_concurrency"), default.node_assembly_concurrency, ), - tag_mode=resolved_tag_mode, fine_min_pages=_as_int( value.get("fine_min_pages"), default.fine_min_pages, @@ -127,25 +117,8 @@ def from_mapping(cls, value: object) -> Self: value.get("node_summary_max_pages"), default.node_summary_max_pages, ), - page_locate_residual_agent_limit=_as_int( - value.get("page_locate_residual_agent_limit"), - default.page_locate_residual_agent_limit, - ), - page_locate_max_emit_depth=_as_int( - value.get("page_locate_max_emit_depth"), - default.page_locate_max_emit_depth, - ), - page_locate_min_emit_depth=_as_int( - value.get("page_locate_min_emit_depth"), - default.page_locate_min_emit_depth, - ), - page_locate_vlm_candidate_page_cap=_as_int( - value.get("page_locate_vlm_candidate_page_cap"), - default.page_locate_vlm_candidate_page_cap, - ), - page_locate_full_leaf_sections=_as_bool( - value.get("page_locate_full_leaf_sections"), - default.page_locate_full_leaf_sections, + scan_direction=_as_str( + value.get("scan_direction"), default.scan_direction ), ) diff --git a/packages/shared-python/shared/services/ai/prompt_service.py b/packages/shared-python/shared/services/ai/prompt_service.py index 650864be7..f5152cab7 100755 --- a/packages/shared-python/shared/services/ai/prompt_service.py +++ b/packages/shared-python/shared/services/ai/prompt_service.py @@ -483,57 +483,44 @@ def build_prompt(task, texts, query, **kwargs): - Return ONLY the JSON object, with no markdown fences or extra text. """ - elif task == "page-memory-text-tag": + elif task == "page-memory-vlm-title": temperature = 0 top_p = 0.01 - max_tokens = kwargs.get("paras", {}).get("max_tokens", 600) - entity_line = _entity_instruction() - page_text = kwargs.get("paras", {}).get("page_text", "") - prompt = f"""\ - You are annotating a single document page for a document memory system. - The following is the extracted text from the page: - \"\"\" - {page_text} - \"\"\" - - Return one strict JSON object with exactly these keys: - - {{ - "summary": "", - "entities": [{{"text": "", "type": ""}}] - }} + paras = kwargs.get("paras", {}) + max_tokens = paras.get("max_tokens", 300) + scan_direction = paras.get("scan_direction", "top_to_bottom_left_to_right") - Rules: - - "summary": describe the main content in a few sentences, in the same - language as the text. If the text contains a table, state its topic - and key columns; if it describes a figure or chart, describe what it - depicts and any standout values. If the text is empty or carries no - meaningful content, set summary to an empty string. - {entity_line} - - Return ONLY the JSON object, with no markdown fences or extra text. - """ + if "right_to_left" in scan_direction: + reading_order_upper = "TOP-TO-BOTTOM, RIGHT-TO-LEFT" + column_order = "right to left (i.e. finish the right column before starting the left column)" + else: + reading_order_upper = "TOP-TO-BOTTOM, LEFT-TO-RIGHT" + column_order = "left to right (i.e. finish the left column before starting the right column)" - elif task == "page-memory-vlm-title": - temperature = 0 - top_p = 0.01 - max_tokens = kwargs.get("paras", {}).get("max_tokens", 300) - prompt = """\ + prompt = f"""\ You are extracting document-outline-level headings from a PDF page screenshot. - Your goal is to find ONLY the headings that would appear in a Table of Contents. - Most pages will have ZERO such headings — returning an empty list is expected - and correct for the majority of pages. + Your goal is to find ONLY the section headings that structure the document. + If no text on this page qualifies as a section heading, return an empty list. + + READING ORDER: + This page may contain one or more readable columns. + Within each column, read from top to bottom. + Between columns, read from {column_order}. + Return every qualifying heading on this page in that reading order. + Do not skip a heading just because it looks like a known section title; + extract all outline-level headings that appear on the page. Return strict JSON: - { + {{ "titles": [ - { + {{ "text": "", "prominence": <0.0-1.0>, "is_in_table": , "is_in_header_footer": - } + }} ] - } + }} ═══ MANDATORY BOOLEAN FLAGS (CRITICAL) ═══ For EVERY extracted heading, you MUST accurately evaluate these two flags: @@ -557,10 +544,11 @@ def build_prompt(task, texts, query, **kwargs): 3. VISUAL DISTINCTION (supporting): The text is visually set apart from body text — larger font, bold, - centered, or has extra vertical spacing. + centered, extra vertical spacing, or wrapped in a distinctive + background color block. "prominence": 1.0 = most prominent; 0.5 = medium; 0.1 = minor. - Return titles in TOP-TO-BOTTOM order. Text must be EXACT verbatim. + Return titles in {reading_order_upper} order. Text must be EXACT verbatim. ═══ WHAT TO EXCLUDE (critical — read carefully) ═══ @@ -576,8 +564,8 @@ def build_prompt(task, texts, query, **kwargs): organization/document names repeated as running headers, page numbers, book/volume titles used as running headers or footers. - 3. BODY TEXT — Numbered clauses, list items, paragraphs, or running - prose, even if bold or indented. + 3. INLINE TEXT — bullet list items, numbered clauses, or text that continues a paragraph. + These are content items, not section headings, even if bold. 4. CAPTIONS — Figure/table captions, footnotes. @@ -586,9 +574,9 @@ def build_prompt(task, texts, query, **kwargs): with page numbers — those entries are references, not headings. ═══ IMPORTANT ═══ - Many pages consist entirely of tables, body text, or appendix forms. - These pages have NO qualifying headings. Return {"titles": []} for them. - Do NOT force-extract table labels or body text as headings. + Many pages consist entirely of tables, numbered clauses, or appendix forms. + These pages have NO qualifying headings. Return {{"titles": []}} for them. + Do NOT force-extract table labels or numbered items as headings. Return ONLY the JSON object, no markdown fences. """ @@ -600,20 +588,29 @@ def build_prompt(task, texts, query, **kwargs): max_tokens = kwargs["paras"].get("max_tokens", 2000) coarse_context = kwargs["paras"].get("coarse_context", "") coarse_section = f""" -Confirmed coarse parent section: +Confirmed coarse parent section (the scope of this subtree): ''' {coarse_context} ''' +The input candidates already lie strictly INSIDE this coarse parent. The parent's +own title and the next coarse sibling title (if any) have already been removed. +Do NOT restate or invent those coarse endpoint titles. + +Level 1 means the first heading level under this coarse parent. Nest deeper +headings relative to that parent only. + """ if coarse_context else "" prompt = f""" You are constructing a fine-grained document hierarchy for ONE already-bounded PDF segment. The input rows are NOT raw body text. They are clean title -candidates observed directly from page screenshots by a VLM. +candidates observed directly from page screenshots by a VLM, then trimmed to +the interior of one coarse TOC leaf. Your task: - Assign a relative hierarchy level to each real section/table/form heading. -- Level 1 means top-level inside this segment, level 2 is its child, etc. +- Level 1 means top-level under the confirmed coarse parent, level 2 is its + child, etc. - Preserve all legitimate sibling headings. Consecutive same-level headings are normal and MUST NOT be demoted just because no body text appears between rows. - Use page order as reading order. The "prominence" value is visual strength, @@ -720,47 +717,7 @@ def build_prompt(task, texts, query, **kwargs): top_p = 0.01 max_tokens = kwargs.get("paras", {}).get("max_tokens", 1200) grid_size = kwargs.get("paras", {}).get("grid_size", 1000) - # Previous production prompt kept for comparison: - # prompt = f"""\ - # You are a precise document layout detector. The attached image is a single PDF - # page screenshot. - # - # Find visually distinct tables, charts, and figures that should become reusable - # document assets. Return strict JSON: - # {{ - # "regions": [ - # {{ - # "kind": "table|chart|figure", - # "bbox": [x1, y1, x2, y2], - # "caption": "", - # "title": "", - # "summary": "<1 sentence searchable summary>", - # "keywords": ["", ""], - # "confidence": 0.0 - # }} - # ] - # }} - # - # Coordinate system: - # - Treat the page image as a {grid_size}x{grid_size} grid. - # - Origin is the top-left corner. - # - bbox values must be integers in [0, {grid_size}]. - # - bbox must tightly include the whole asset: title, caption, legend, axes, - # labels, table headers, and footnotes that are part of the asset. - # - Exclude surrounding body paragraphs, page headers, page footers, and page - # numbers. - # - # Rules: - # - "table": rows/columns of data, forms, financial tables, appendix tables. - # - "chart": plotted data such as bar/line/pie/scatter charts. - # - "figure": distinct diagrams, flowcharts, architecture drawings, embedded images. - # - # - Do not transcribe entire tables. Summarize the topic and extreme values based on main columns or rows. - # - "keywords" must be an array of up to 5 strings in the same language as the visible asset text. - # - Use confidence 0.0-1.0. Only include assets you can localize. - # - If there are no assets, return {{"regions":[]}}. - # - Return ONLY the JSON object, no markdown fences or explanations. - # """ + prompt = f"""\ You are a precise document layout detector. The attached image is a single rendered PDF page. @@ -789,15 +746,38 @@ def build_prompt(task, texts, query, **kwargs): numbers. Rules: - - "table": data arranged in clear rows and columns - grid lines, cell - borders, or strongly aligned cells (data tables, forms, financial tables, - appendix tables). - - "figure": any non-table visual asset - bar/line/pie/scatter charts, - plots, diagrams, flowcharts, architecture drawings, schematics, or embedded - images. - - Do not mark ordinary paragraphs, bullet lists, title blocks, or loose - multi-line text as tables. - - Do not split a single coherent table or figure into sub-parts. + - "table": ONLY a conventional data table with visible grid and strongly regular cell alignment. + Require ALL of: + (1) an explicit header row and/or header column that labels the fields. + (2) one intact axis-aligned rectangular footprint: all four corners of + the table body are present, every data row spans that full width, and + the cell grid fills the rectangle without cutouts, protruding corner + panels, or L-shaped outlines. + typical table cases: forms, financial tables, data rows with field headers. + + - Do NOT mark as "table": process boards, flowchart-like matrices, cards + connected by arrows, multi-column visual layouts, comparison panels, or + any region whose meaning depends on icons/arrows/color blocks rather than + plain headered cells. Those must be "figure". + - If unsure whether it meets this bar, prefer "figure". + + - "figure": any non-table visual asset - charts, plots, diagrams, + flowcharts, architecture drawings, schematics, embedded images, and the + table-like visuals excluded above. + - Prefer one bbox for the whole figure. When multiple visual parts clearly + form one composition (shared caption or a multi-panel explanation of the + same concept/process), return them as a single figure, not separate images. + - Treat a flowchart or process diagram as one figure, including its nodes, + edges, labels, and legend when they belong together. + + - Do NOT extract page backgrounds, watermarks, stamps, or decorative underlays. + - Do NOT extract small logos, icons, bullets, or other scattered decorative + marks that are not standalone informative figures. + - Do NOT extract ornamental digits/letters placed before a heading title as figures. + + - Do NOT mark ordinary paragraphs, bullet lists, title blocks, or loose multi-line text as tables. + - Do NOT split a single coherent table or figure into sub-parts. + - "title" is one short label only (single line). Do not duplicate it into other fields and do not write a summary. Use an empty string when there is no visible title or caption. @@ -1023,6 +1003,14 @@ def build_prompt(task, texts, query, **kwargs): max_tokens = kwargs["paras"].get("max_tokens", 100) node_name = kwargs["paras"].get("node_name", "") lang = kwargs["paras"].get("lang") + has_self_only = bool(kwargs["paras"].get("has_self_only")) + child_titles = kwargs["paras"].get("child_titles") or [] + self_only_content = kwargs["paras"].get("self_only_content", "(none)") + covered_nodes = kwargs["paras"].get("covered_nodes", "(none)") + if isinstance(child_titles, list): + children_repr = ", ".join(str(t) for t in child_titles) if child_titles else "(none)" + else: + children_repr = str(child_titles) or "(none)" lang_directive = _language_directive(lang) lang_rule = ( f"- **LANGUAGE (HARD CONSTRAINT)**: {lang_directive}" @@ -1030,13 +1018,20 @@ def build_prompt(task, texts, query, **kwargs): else "- Your response must be in the SAME LANGUAGE as the input text" ) - prompt = f"""You will receive summaries of sub-sections from a document section called "{node_name}": - ''' - {texts} - ''' + prompt = f"""SCOPE_TITLE: {node_name} + SCOPE_STRUCTURE: + - self_only: {"yes" if has_self_only else "no"} + - children: [{children_repr}] + + SELF_ONLY_CONTENT: + {self_only_content} + + COVERED_NODES: + {covered_nodes} + Your task: {lang_rule} - - Produce ONE concise sentence summarizing ALL sub-sections, no more than {max_tokens} characters + - Produce ONE concise top-level summary of THIS scope (self_only content plus covered nodes), no more than {max_tokens} characters - Output the summary DIRECTLY, no prefixes, no explanations - If the input lacks meaningful text, return exactly: null """ diff --git a/packages/shared-python/shared/services/chunks/dataframe_chunk_converter.py b/packages/shared-python/shared/services/chunks/dataframe_chunk_converter.py index 5e4ec5905..e5ad3931f 100644 --- a/packages/shared-python/shared/services/chunks/dataframe_chunk_converter.py +++ b/packages/shared-python/shared/services/chunks/dataframe_chunk_converter.py @@ -337,11 +337,8 @@ def dataframe_to_chunks(df: _ParserDataFrame | None) -> list[Dict[str, JsonValue metadata["file_path"] = embedded_image_path metadata["original_name"] = image_name else: - normalized_path = path.replace("-->", "/") image_name = ( - os.path.basename(normalized_path) - if normalized_path - else f"image_{chunk_id}.jpg" + os.path.basename(path) if path else f"image_{chunk_id}.jpg" ) _, image_extension = os.path.splitext(image_name) if not image_extension: @@ -355,11 +352,8 @@ def dataframe_to_chunks(df: _ParserDataFrame | None) -> list[Dict[str, JsonValue if embedded_table_path: metadata["file_path"] = embedded_table_path else: - normalized_path = path.replace("-->", "/") table_name = ( - os.path.basename(normalized_path) - if normalized_path - else f"table_{chunk_id}.html" + os.path.basename(path) if path else f"table_{chunk_id}.html" ) metadata["file_path"] = f"tables/{table_name}" @@ -378,7 +372,7 @@ def dataframe_to_chunks(df: _ParserDataFrame | None) -> list[Dict[str, JsonValue for chunk in chunks: metadata = chunk["metadata"] relationship_refs = metadata.pop("_relationship_refs", []) - if chunk["type"] not in {"text", "page"}: + if chunk["type"] not in {"text", "page", "table"}: continue embed_connections = convert_refs_to_embed_connections( relationship_refs, resource_target_map diff --git a/packages/shared-python/shared/services/chunks/document_path.py b/packages/shared-python/shared/services/chunks/document_path.py index 37baa9943..a06ba231f 100644 --- a/packages/shared-python/shared/services/chunks/document_path.py +++ b/packages/shared-python/shared/services/chunks/document_path.py @@ -4,6 +4,8 @@ import os +from shared.services.chunks.path_segments import unescape_path_segment + _DOCUMENT_FILE_EXTENSIONS = { ".csv", ".atlas", @@ -38,7 +40,7 @@ def split_document_path( source_file_name: str | None = None, ) -> tuple[list[str], list[str]]: """Return ``(root_parts, section_parts)`` for new and legacy chunk paths.""" - parts = _split_path(path, source_file_name=source_file_name) + parts = _split_path(path) if not parts: return [], [] if parts[0] in _MEDIA_ROOT_SEGMENTS and not _is_legacy_namespace_path( @@ -51,63 +53,16 @@ def split_document_path( return parts[: document_index + 1], parts[document_index + 1 :] -def _split_path(path: str | None, *, source_file_name: str | None) -> list[str]: +def _split_path(path: str | None) -> list[str]: raw = str(path or "").strip() - raw_segments = raw.split("/") - source_segment = _normalize_document_file_name(source_file_name) - parts: list[str] = [] - for index, segment in enumerate(raw_segments): - parts.extend( - _split_arrow_document_segment( - segment, - can_split=_can_split_arrow_document_segment( - index=index, - raw_segments=raw_segments, - segment=segment, - source_segment=source_segment, - ), - ) - ) - return parts - - -def _can_split_arrow_document_segment( - *, - index: int, - raw_segments: list[str], - segment: str, - source_segment: str, -) -> bool: - if index == 0: - return True - if index != 1: - return False - - first_segment = raw_segments[0].strip() if raw_segments else "" - if not _is_document_file_segment(first_segment): - return True - - arrow_document_segment = _normalize_document_file_name( - segment.split("-->", 1)[0] - ) - return bool(source_segment and arrow_document_segment == source_segment) - - -def _split_arrow_document_segment(segment: str, *, can_split: bool) -> list[str]: - normalized_segment = segment.strip() - if not normalized_segment: - return [] - if not can_split or "-->" not in normalized_segment: - return [normalized_segment] - - arrow_parts = [ - part.strip() - for part in normalized_segment.split("-->") - if part.strip() + # Chunk paths use ``/`` as the sole hierarchy separator. Titles that contain + # a semantic ``/`` are escaped at construction time (``∕``) and restored here + # so publication/retrieval see one segment per title. + return [ + unescape_path_segment(segment.strip()) + for segment in raw.split("/") + if segment.strip() ] - if arrow_parts and _is_document_file_segment(arrow_parts[0]): - return arrow_parts - return [normalized_segment] def _find_document_index( diff --git a/packages/shared-python/shared/services/chunks/path_segments.py b/packages/shared-python/shared/services/chunks/path_segments.py new file mode 100644 index 000000000..ff4262e44 --- /dev/null +++ b/packages/shared-python/shared/services/chunks/path_segments.py @@ -0,0 +1,82 @@ +"""Escape semantic slashes in document hierarchy path segments. + +Parser chunk paths use ``/`` as a deterministic hierarchy separator +(``file.pdf/Section/Subsection``). Heading titles may themselves contain +``/`` (e.g. ``Symbols/Numbers``). Those semantic slashes must be escaped +before join and unescaped after split so they are not treated as levels. + +The escape character is U+2215 DIVISION SLASH (``∕``), matching the existing +markdown parse-state behavior. +""" + +from __future__ import annotations + +from collections.abc import Sequence + +# Deterministic hierarchy separator used in parser chunk / section paths. +DOCUMENT_PATH_SEP = "/" + +# Replacement for semantic ``/`` inside a single path segment. +# Must not equal DOCUMENT_PATH_SEP and should be rare in natural titles. +ESCAPED_DOCUMENT_PATH_SEP = "\u2215" # ∕ + + +def escape_path_segment(segment: object) -> str: + """Escape separator characters inside one hierarchy title/segment.""" + text = str(segment or "") + if not text: + return "" + # Escape the escape character first so round-trips stay injective. + text = text.replace(ESCAPED_DOCUMENT_PATH_SEP, ESCAPED_DOCUMENT_PATH_SEP * 2) + return text.replace(DOCUMENT_PATH_SEP, ESCAPED_DOCUMENT_PATH_SEP) + + +def unescape_path_segment(segment: object) -> str: + """Restore semantic slashes previously escaped by ``escape_path_segment``.""" + text = str(segment or "") + if not text: + return "" + out: list[str] = [] + i = 0 + esc = ESCAPED_DOCUMENT_PATH_SEP + while i < len(text): + if text.startswith(esc + esc, i): + out.append(esc) + i += 2 + continue + if text.startswith(esc, i): + out.append(DOCUMENT_PATH_SEP) + i += 1 + continue + out.append(text[i]) + i += 1 + return "".join(out) + + +def join_document_path(parts: Sequence[object]) -> str: + """Join hierarchy parts with ``/``, escaping each part first.""" + escaped = [escape_path_segment(part) for part in parts if str(part or "")] + return DOCUMENT_PATH_SEP.join(escaped) + + +def append_document_path(parent_path: object, *segments: object) -> str: + """Append title segments under an already-escaped parent path.""" + parent = str(parent_path or "") + extras = [escape_path_segment(segment) for segment in segments if str(segment or "")] + if not parent: + return DOCUMENT_PATH_SEP.join(extras) + if not extras: + return parent + return DOCUMENT_PATH_SEP.join([parent, *extras]) + + +def split_escaped_document_path(path: object) -> list[str]: + """Split a ``/``-joined hierarchy path and unescape each segment.""" + raw = str(path or "").strip() + if not raw: + return [] + return [ + unescape_path_segment(part) + for part in raw.split(DOCUMENT_PATH_SEP) + if part != "" + ] diff --git a/packages/shared-python/shared/services/jobs/lifecycle/publication.py b/packages/shared-python/shared/services/jobs/lifecycle/publication.py index f29ed6acb..a8a3e7473 100644 --- a/packages/shared-python/shared/services/jobs/lifecycle/publication.py +++ b/packages/shared-python/shared/services/jobs/lifecycle/publication.py @@ -51,6 +51,7 @@ def publish_result( job_result_id: str, chunks: list[dict[str, Any]], section_summaries: dict[str, str] | None, + document_top_summary: str | None = None, ) -> JobPublicationOutcome: previous_document_scope = self._retrieval_publication.get_existing_document_scope( db, @@ -69,6 +70,7 @@ def publish_result( db, job_id=job_id, job_result_id=job_result_id, + top_summary=document_top_summary, ) cache_invalidation = self._build_cache_invalidation( diff --git a/packages/shared-python/shared/services/jobs/lifecycle/service.py b/packages/shared-python/shared/services/jobs/lifecycle/service.py index 3683fca30..c7953a005 100644 --- a/packages/shared-python/shared/services/jobs/lifecycle/service.py +++ b/packages/shared-python/shared/services/jobs/lifecycle/service.py @@ -52,6 +52,7 @@ def finalize_job_success( stored_count: int = 0, delivery_mode: str = "url", section_summaries: Optional[Dict[str, str]] = None, + document_top_summary: Optional[str] = None, ) -> Dict[str, Any]: """Finalize a successful job in a single atomic transaction. @@ -78,6 +79,7 @@ def finalize_job_success( stored_count=stored_count, delivery_mode=delivery_mode, section_summaries=section_summaries, + document_top_summary=document_top_summary, ), should_commit=lambda finalization: finalization.response.should_commit(), build_response=lambda finalization: finalization.response.to_dict(), diff --git a/packages/shared-python/shared/services/jobs/lifecycle/success_finalizer.py b/packages/shared-python/shared/services/jobs/lifecycle/success_finalizer.py index d05b41436..2edfc2b66 100644 --- a/packages/shared-python/shared/services/jobs/lifecycle/success_finalizer.py +++ b/packages/shared-python/shared/services/jobs/lifecycle/success_finalizer.py @@ -79,6 +79,7 @@ def finalize( stored_count: int, delivery_mode: str, section_summaries: dict[str, str] | None, + document_top_summary: str | None = None, ) -> JobSuccessFinalization: job_result = self._result_writer.upsert_job_result( db, @@ -95,6 +96,7 @@ def finalize( job_result_id=job_result.id, chunks=chunks, section_summaries=section_summaries, + document_top_summary=document_top_summary, ) transition_outcome = self._state_machine.mark_completed_outcome( diff --git a/packages/shared-python/shared/services/redis/redis_service.py b/packages/shared-python/shared/services/redis/redis_service.py index fa9d39fa9..e83e3a7bc 100644 --- a/packages/shared-python/shared/services/redis/redis_service.py +++ b/packages/shared-python/shared/services/redis/redis_service.py @@ -111,6 +111,28 @@ async def _operation(): original_exception=e, ) + async def set_nx(self, key: str, value: str, ex: int) -> bool: + """Atomic SET NX EX — set only if the key does not already exist. + + Returns ``True`` if the key was written, ``False`` if it already existed. + Does not JSON-encode the value and does not fall back to a default TTL. + """ + try: + client = await self._get_client() + full_key = self._build_key(key) + + async def _operation(): + return await client.set(full_key, value, nx=True, ex=ex) + + return bool(await self._execute_with_retry(_operation)) + except Exception as e: + logger.error(f"Redis SET NX operation failed: {e}") + raise RedisOperationError( + internal_message=f"SET NX operation failed: {str(e)}", + operation="SET_NX", + original_exception=e, + ) + async def get(self, key: str, default: Any = None) -> Any: """Get a key value.""" try: diff --git a/packages/shared-python/shared/services/retrieval/agentic/evidence/renderer.py b/packages/shared-python/shared/services/retrieval/agentic/evidence/renderer.py index db9338027..3b8bd9c73 100644 --- a/packages/shared-python/shared/services/retrieval/agentic/evidence/renderer.py +++ b/packages/shared-python/shared/services/retrieval/agentic/evidence/renderer.py @@ -183,6 +183,9 @@ def render_leaf_chunks( table_lines = render_table_chunk_lines( target, display_ref=display_ref, + chunk_by_id=chunk_by_id, + asset_lookup=asset_lookup, + rendered_ids=rendered_ids, ) content = content.replace(ref_str, "\n" + "\n".join(table_lines) + "\n") elif target_type == "image": @@ -227,6 +230,9 @@ def render_leaf_chunks( for line in render_table_chunk_lines( chunk, display_ref=display_ref, + chunk_by_id=chunk_by_id, + asset_lookup=asset_lookup, + rendered_ids=rendered_ids, ): if line.strip(): parts.append(f"{indent}┈ {line}") @@ -262,6 +268,9 @@ def render_table_chunk_lines( chunk: dict[str, Any], *, display_ref: str, + chunk_by_id: dict[str, dict[str, Any]] | None = None, + asset_lookup: dict[str, AssetLookupValue] | None = None, + rendered_ids: set[str] | None = None, ) -> list[str]: header = f"[Table: {display_ref}]" if display_ref else "[Table]" lines = [header] @@ -292,6 +301,65 @@ def render_table_chunk_lines( if caption: lines.append("Caption:") lines.append(str(caption).strip()) + + lines.extend( + _render_table_embedded_image_lines( + chunk, + chunk_by_id=chunk_by_id or {}, + asset_lookup=asset_lookup, + rendered_ids=rendered_ids, + ) + ) + return lines + + +def _render_table_embedded_image_lines( + table_chunk: dict[str, Any], + *, + chunk_by_id: dict[str, dict[str, Any]], + asset_lookup: dict[str, AssetLookupValue] | None, + rendered_ids: set[str] | None, +) -> list[str]: + metadata = table_chunk.get("chunk_metadata") or table_chunk.get("metadata") or {} + if not isinstance(metadata, dict): + return [] + + lines: list[str] = [] + for connection in metadata.get("connect_to") or []: + if not isinstance(connection, dict): + continue + if str(connection.get("relation") or "").strip() != "embeds": + continue + target_id = str(connection.get("target") or "").strip() + if not target_id: + continue + target = chunk_by_id.get(target_id) + if not target: + continue + target_type = ( + target.get("chunk_type") or target.get("type") or "" + ).strip().lower() + if target_type != "image": + continue + if rendered_ids is not None: + rendered_ids.add(target_id) + + file_path = target.get("file_path") or "" + asset_url = _lookup_asset_url(asset_lookup, target_id) + display_ref = asset_url or file_path + image_description = str(target.get("content") or "").strip() + ref_str = str(connection.get("ref") or "").strip() + if ref_str and ref_str in image_description: + image_description = image_description.replace(ref_str, "").strip() + + if display_ref: + lines.append(f"[Image: {display_ref}]") + elif image_description: + lines.append("[Image description]") + if image_description: + lines.extend( + line for line in image_description.split("\n") if line.strip() + ) return lines diff --git a/packages/shared-python/shared/services/retrieval/graph/service.py b/packages/shared-python/shared/services/retrieval/graph/service.py index 35d05fec3..f5b9c51b6 100644 --- a/packages/shared-python/shared/services/retrieval/graph/service.py +++ b/packages/shared-python/shared/services/retrieval/graph/service.py @@ -75,6 +75,7 @@ def publish_document_graph( namespace: str, document_id: str, job_result_id: str, + top_summary: str | None = None, ) -> None: document = db.execute( select(Document).where(Document.document_id == document_id) @@ -103,7 +104,11 @@ def publish_document_graph( types_breakdown[chunk_type or 'text'] += 1 chunks_count = len(chunk_meta_rows) - top_summary = extract_document_top_summary(chunk_metadata_list) + resolved_top_summary = str(top_summary or "").strip() + if not resolved_top_summary: + # Backward compatible fallback for older chunks that still carry + # per-chunk document_top_summary copies. + resolved_top_summary = extract_document_top_summary(chunk_metadata_list) # ── Clean up old graph data for this document ── self.remove_document_graph( @@ -137,7 +142,7 @@ def publish_document_graph( 'top_entities': top_entities, 'chunks_count': chunks_count, 'types': dict(types_breakdown), - 'top_summary': top_summary, + 'top_summary': resolved_top_summary, }, ) ) diff --git a/packages/shared-python/shared/services/retrieval/hydration/connected.py b/packages/shared-python/shared/services/retrieval/hydration/connected.py index 2267b91c8..fcbc647bf 100644 --- a/packages/shared-python/shared/services/retrieval/hydration/connected.py +++ b/packages/shared-python/shared/services/retrieval/hydration/connected.py @@ -31,7 +31,7 @@ async def hydrate_connected_target_rows( } target_ids_by_revision: dict[tuple[str, str], set[str]] = {} for row in rows: - if normalize_chunk_type(row.get('chunk_type')) not in ('text', 'page'): + if normalize_chunk_type(row.get('chunk_type')) not in ('text', 'page', 'table'): continue document_id = str(row.get('document_id') or '').strip() job_result_id = str(row.get('job_result_id') or '').strip() diff --git a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py index 8c9a1b93e..96534bce3 100644 --- a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py +++ b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py @@ -60,28 +60,14 @@ async def assemble_retrieval_results( assembled_row['content'] = _page_summary(row) assembled_row['content_source'] = 'summary' elif chunk_type == 'table': - assembled_row['content'] = _table_summary_content(row) + assembled_row['content'] = _compose_table_content(row, rows_by_chunk_id) assembled_row['content_source'] = 'summary' elif chunk_type == 'text': - connected_targets: list[tuple[int, str]] = [] - for target_id in iter_connected_target_ids(row): - target_row = rows_by_chunk_id.get(target_id) - if not target_row: - continue - if normalize_chunk_type(target_row.get('chunk_type')) != 'table': - continue - target_content = _table_summary_content(target_row) - if target_content: - sort_key = int(target_row.get('sort_order', 0) or 0) - connected_targets.append((sort_key, target_content)) - connected_targets.sort(key=lambda item: item[0]) - related_parts = [content for _, content in connected_targets] - - # TODO: Dedicated Large Table Agent - # For the Notebook/Agent environment, consider introducing a dedicated - # "Large Table Agent" that can fetch and query oversized tables via URL. + related_parts = _connected_media_parts(row, rows_by_chunk_id) if base_content and related_parts: assembled_row['content'] = '\n\n'.join([base_content, *related_parts]) + elif related_parts: + assembled_row['content'] = '\n\n'.join(related_parts) else: assembled_row['content'] = base_content assembled_row['content_source'] = 'content' @@ -100,6 +86,71 @@ def _page_summary(row: dict[str, Any]) -> str: return str(metadata.get('summary') or '').strip() +def _compose_table_content( + row: dict[str, Any], + rows_by_chunk_id: dict[str, dict[str, Any]], +) -> str: + parts = [_table_summary_content(row)] + parts.extend(_connected_image_parts(row, rows_by_chunk_id)) + return '\n\n'.join(part for part in parts if part) + + +def _connected_media_parts( + row: dict[str, Any], + rows_by_chunk_id: dict[str, dict[str, Any]], +) -> list[str]: + connected_targets: list[tuple[int, str]] = [] + for target_id in iter_connected_target_ids(row): + target_row = rows_by_chunk_id.get(target_id) + if not target_row: + continue + target_type = normalize_chunk_type(target_row.get('chunk_type')) + if target_type == 'table': + target_content = _compose_table_content(target_row, rows_by_chunk_id) + elif target_type == 'image': + target_content = _image_display_content(target_row) + else: + continue + if target_content: + sort_key = int(target_row.get('sort_order', 0) or 0) + connected_targets.append((sort_key, target_content)) + connected_targets.sort(key=lambda item: item[0]) + return [content for _, content in connected_targets] + + +def _connected_image_parts( + row: dict[str, Any], + rows_by_chunk_id: dict[str, dict[str, Any]], +) -> list[str]: + parts: list[str] = [] + for target_id in iter_connected_target_ids(row): + target_row = rows_by_chunk_id.get(target_id) + if not target_row: + continue + if normalize_chunk_type(target_row.get('chunk_type')) != 'image': + continue + content = _image_display_content(target_row) + if content: + parts.append(content) + return parts + + +def _image_display_content(row: dict[str, Any]) -> str: + display_ref = ( + str(row.get('asset_url') or '').strip() + or str(row.get('file_path') or '').strip() + ) + description = str(row.get('content') or '').strip() + lines: list[str] = [] + if display_ref: + lines.append(f'[Image: {display_ref}]') + elif description: + lines.append('[Image description]') + if description: + lines.extend(line for line in description.split('\n') if line.strip()) + return '\n'.join(lines) + + def _table_summary_content(row: dict[str, Any]) -> str: metadata = row.get('chunk_metadata') or row.get('metadata') or {} if not isinstance(metadata, dict): diff --git a/packages/shared-python/shared/services/retrieval/publication_service.py b/packages/shared-python/shared/services/retrieval/publication_service.py index 6e70aa711..8911c3b7c 100644 --- a/packages/shared-python/shared/services/retrieval/publication_service.py +++ b/packages/shared-python/shared/services/retrieval/publication_service.py @@ -237,12 +237,18 @@ def publish_document_graph( *, job_id: str, job_result_id: str, + top_summary: str | None = None, ) -> None: job = db.execute(select(Job).where(Job.job_id == job_id)).scalar_one_or_none() if not job: raise RuntimeError(f"Job not found for graph publication: {job_id}") - self._publish_document_graph_for_job(db, job=job, job_result_id=job_result_id) + self._publish_document_graph_for_job( + db, + job=job, + job_result_id=job_result_id, + top_summary=top_summary, + ) def _publish_document_graph_for_job( self, @@ -250,6 +256,7 @@ def _publish_document_graph_for_job( *, job: Job, job_result_id: str, + top_summary: str | None = None, ) -> None: metadata = job.job_metadata or {} @@ -271,6 +278,7 @@ def _publish_document_graph_for_job( namespace=namespace, document_id=document_id, job_result_id=job_result_id, + top_summary=top_summary, ) def remove_document_graph( diff --git a/packages/shared-python/shared/services/retrieval/search/lexical_text.py b/packages/shared-python/shared/services/retrieval/search/lexical_text.py index a177978da..b6647a16d 100644 --- a/packages/shared-python/shared/services/retrieval/search/lexical_text.py +++ b/packages/shared-python/shared/services/retrieval/search/lexical_text.py @@ -8,6 +8,7 @@ from typing import Any, Optional from shared.services.chunks.document_path import split_document_path +from shared.services.chunks.path_segments import split_escaped_document_path from shared.utils.text_utils import tokenize_contents_for_retrieval _SAME_AS_RE = re.compile(r"\[SAME-AS [^\]]+\]") @@ -35,7 +36,7 @@ def split_section_path(path: Optional[str]) -> list[str]: return [] if " / " in raw: return [p.strip() for p in raw.split(" / ") if p.strip()] - return [p.strip() for p in raw.split("/") if p.strip()] + return split_escaped_document_path(raw) def build_lexical_text(value: str) -> str: diff --git a/packages/shared-python/shared/services/storage/zip_chunk_schema.py b/packages/shared-python/shared/services/storage/zip_chunk_schema.py index 2cb36b432..7dc0be09e 100644 --- a/packages/shared-python/shared/services/storage/zip_chunk_schema.py +++ b/packages/shared-python/shared/services/storage/zip_chunk_schema.py @@ -78,13 +78,19 @@ def format_chunks( if chunk_type == "text": metadata.update( - _format_text_metadata( - chunk=chunk, - chunk_type_str=chunk_type_str, - content=str(content), - existing_metadata=existing_metadata, - resource_target_map=resource_target_map, - ) + { + "tokens": existing_metadata.get("tokens") + or chunk.get("tokens", 0), + "keywords": existing_metadata.get("keywords") + or chunk.get("keywords", []), + "connect_to": _build_embed_connect_to( + chunk=chunk, + chunk_type_str=chunk_type_str, + content=str(content), + existing_metadata=existing_metadata, + resource_target_map=resource_target_map, + ), + } ) elif chunk_type == "image": if image_info: @@ -105,6 +111,13 @@ def format_chunks( "keywords", [] ) metadata["tokens"] = [] + metadata["connect_to"] = _build_embed_connect_to( + chunk=chunk, + chunk_type_str=chunk_type_str, + content=str(content), + existing_metadata=existing_metadata, + resource_target_map=resource_target_map, + ) elif chunk_type == "page": metadata["keywords"] = existing_metadata.get("keywords") or [] metadata["connect_to"] = existing_metadata.get("connect_to") or [] @@ -140,20 +153,17 @@ def _base_chunk_metadata( "summary": existing_metadata.get("summary") or chunk.get("summary", ""), "page_nums": existing_metadata.get("page_nums", []), } - document_top_summary = str(existing_metadata.get("document_top_summary") or "").strip() - if document_top_summary: - metadata["document_top_summary"] = document_top_summary return metadata -def _format_text_metadata( +def _build_embed_connect_to( *, chunk: dict[str, Any], chunk_type_str: Any, content: str, existing_metadata: dict[str, Any], resource_target_map: dict[str, str], -) -> dict[str, Any]: +) -> list[Any]: relationship_refs = parse_relationship_refs( chunk.get("type_raw") or chunk_type_str, content, @@ -168,11 +178,7 @@ def _format_text_metadata( or chunk.get("connectto"), resource_target_map, ) - return { - "tokens": existing_metadata.get("tokens") or chunk.get("tokens", 0), - "keywords": existing_metadata.get("keywords") or chunk.get("keywords", []), - "connect_to": merge_connections(embed_connections, related_connections), - } + return merge_connections(embed_connections, related_connections) def _resolve_table_file_path( diff --git a/packages/shared-python/shared/services/storage/zip_doc_navigation.py b/packages/shared-python/shared/services/storage/zip_doc_navigation.py index ff5f52ebf..938d62e58 100644 --- a/packages/shared-python/shared/services/storage/zip_doc_navigation.py +++ b/packages/shared-python/shared/services/storage/zip_doc_navigation.py @@ -5,6 +5,7 @@ from typing import Any from shared.services.chunks.document_path import split_document_path +from shared.services.chunks.path_segments import join_document_path from shared.utils.text_utils import truncate_content_preview @@ -144,7 +145,7 @@ def _build_section_tree( if key not in root_children: root_children[key] = { "title": "Root", - "path": "/".join(root_parts) if root_parts else path, + "path": join_document_path(root_parts) if root_parts else path, "summary": chunk.get("summary", ""), "chunk_count": 0, "_children_map": {}, @@ -161,7 +162,7 @@ def _build_section_tree( if part not in current_level: current_level[part] = { "title": part, - "path": "/".join(full_section_path_parts), + "path": join_document_path(full_section_path_parts), "summary": "", "chunk_count": 0, "_children_map": {}, diff --git a/packages/shared-python/shared/services/storage/zip_result_resources.py b/packages/shared-python/shared/services/storage/zip_result_resources.py index 797b36eed..58d88c667 100644 --- a/packages/shared-python/shared/services/storage/zip_result_resources.py +++ b/packages/shared-python/shared/services/storage/zip_result_resources.py @@ -95,8 +95,7 @@ def _collect_image_files( original_path = chunk.get("path", "") if original_path: _add_candidate(candidate_names, original_path) - normalized_path = original_path.replace("-->", "/") - original_name = os.path.basename(normalized_path) + original_name = os.path.basename(original_path) source_path, matched_name, ext = _resolve_image_source_path( image_files_map, @@ -178,8 +177,7 @@ def _collect_table_files( original_path = chunk.get("path", "") if original_path: - normalized_path = original_path.replace("-->", "/") - original_name = os.path.basename(normalized_path) + original_name = os.path.basename(original_path) _add_candidate(candidate_names, original_path) else: original_name = None @@ -317,7 +315,7 @@ def _normalize_page_citation_ref(value: Any) -> str | None: def _add_candidate(candidates: list[str], value: str | None) -> None: if not value: return - candidate = os.path.basename(str(value).strip().replace("-->", "/")) + candidate = os.path.basename(str(value).strip()) if not candidate: return if candidate.startswith("[") and candidate.endswith("]"): diff --git a/packages/shared-python/shared/services/storage/zip_result_service.py b/packages/shared-python/shared/services/storage/zip_result_service.py index fe1f0c6dc..8d6b229a2 100644 --- a/packages/shared-python/shared/services/storage/zip_result_service.py +++ b/packages/shared-python/shared/services/storage/zip_result_service.py @@ -6,6 +6,8 @@ from __future__ import annotations +import json +import os from typing import Any from loguru import logger @@ -69,6 +71,7 @@ def generate_zip_package( statistics = self._schema.calculate_statistics(formatted_chunks) doc_nav, hierarchy = self._build_navigation_outputs( + add_dir=add_dir, formatted_chunks=formatted_chunks, source_file_name=source_file_name, ) @@ -120,13 +123,42 @@ def generate_zip_package( def _build_navigation_outputs( self, *, + add_dir: str, formatted_chunks: list[dict[str, Any]], source_file_name: str, ) -> tuple[dict[str, Any] | None, dict[str, Any]]: + """Prefer the already-enriched on-disk doc_nav; rebuild only as fallback.""" try: + existing = self._load_existing_doc_nav(add_dir) + if existing is not None: + hierarchy = self._schema.build_hierarchy_dict( + existing.get("sections", []) + ) + logger.info("Using enriched on-disk doc_nav.json for ZIP package") + return existing, hierarchy + doc_nav = self._schema.build_doc_nav(formatted_chunks, source_file_name) hierarchy = self._schema.build_hierarchy_dict(doc_nav.get("sections", [])) return doc_nav, hierarchy except Exception as exc: logger.warning(f"generate doc_nav.json fail {exc}") return None, {} + + @staticmethod + def _load_existing_doc_nav(add_dir: str) -> dict[str, Any] | None: + if not add_dir: + return None + path = os.path.join(add_dir, "doc_nav.json") + if not os.path.isfile(path): + return None + try: + with open(path, encoding="utf-8") as handle: + payload = json.load(handle) + except Exception as exc: + logger.warning(f"Failed to read existing doc_nav.json: {exc}") + return None + if not isinstance(payload, dict): + return None + if not isinstance(payload.get("sections"), list): + return None + return payload diff --git a/packages/shared-python/shared/tests/test_database_pool_config.py b/packages/shared-python/shared/tests/test_database_pool_config.py new file mode 100644 index 000000000..83cac6680 --- /dev/null +++ b/packages/shared-python/shared/tests/test_database_pool_config.py @@ -0,0 +1,22 @@ +"""Unit tests for DatabaseConfig pool defaults.""" + +from __future__ import annotations + +import os + +os.environ.setdefault("DATABASE_URL", "postgresql+asyncpg://test:test@localhost/test") +os.environ.setdefault("TMP_PATH", "/tmp/knowhere-test") +os.environ.setdefault("S3_BUCKET_NAME", "test-uploads") +os.environ.setdefault("S3_ACCESS_KEY_ID", "test") +os.environ.setdefault("S3_SECRET_ACCESS_KEY", "test") +os.environ.setdefault("S3_TEMP_PATH", "/tmp") + +from shared.core.config.database import DatabaseConfig + + +def test_database_pool_defaults_are_fifty() -> None: + config = DatabaseConfig( + DATABASE_URL="postgresql+asyncpg://test:test@localhost/test", + ) + assert config.DB_POOL_SIZE == 50 + assert config.DB_MAX_OVERFLOW == 50 diff --git a/packages/shared-python/shared/tests/test_path_segments.py b/packages/shared-python/shared/tests/test_path_segments.py new file mode 100644 index 000000000..6a22d39ae --- /dev/null +++ b/packages/shared-python/shared/tests/test_path_segments.py @@ -0,0 +1,112 @@ +"""Tests for hierarchy path segment escaping (titles that contain ``/``).""" + +from __future__ import annotations + +import os + +os.environ.setdefault("DATABASE_URL", "postgresql+asyncpg://test:test@localhost/test") +os.environ.setdefault("TMP_PATH", "/tmp/knowhere-test") +os.environ.setdefault("S3_BUCKET_NAME", "test-uploads") +os.environ.setdefault("S3_ACCESS_KEY_ID", "test") +os.environ.setdefault("S3_SECRET_ACCESS_KEY", "test") +os.environ.setdefault("S3_TEMP_PATH", "/tmp") + +from shared.services.chunks.document_path import split_document_path +from shared.services.chunks.path_segments import ( + ESCAPED_DOCUMENT_PATH_SEP, + append_document_path, + escape_path_segment, + join_document_path, + split_escaped_document_path, + unescape_path_segment, +) +from shared.services.retrieval.search.lexical_text import ( + section_path_from_chunk_path, + split_section_path, +) +from shared.services.storage.zip_doc_navigation import ZipDocNavigationBuilder + + +def test_escape_round_trip_for_slash_in_title() -> None: + title = "Symbols/Numbers" + escaped = escape_path_segment(title) + assert ESCAPED_DOCUMENT_PATH_SEP in escaped + assert "/" not in escaped + assert unescape_path_segment(escaped) == title + + +def test_escape_round_trip_for_literal_escape_char() -> None: + title = f"A{ESCAPED_DOCUMENT_PATH_SEP}B/C" + assert unescape_path_segment(escape_path_segment(title)) == title + + +def test_join_and_split_keeps_slash_title_as_one_segment() -> None: + path = join_document_path(["manual.pdf", "Index", "Symbols/Numbers", "Detail"]) + assert path == ( + f"manual.pdf/Index/Symbols{ESCAPED_DOCUMENT_PATH_SEP}Numbers/Detail" + ) + assert split_escaped_document_path(path) == [ + "manual.pdf", + "Index", + "Symbols/Numbers", + "Detail", + ] + + +def test_append_under_escaped_parent() -> None: + parent = join_document_path(["manual.pdf", "Index"]) + child = append_document_path(parent, "Symbols/Numbers") + assert split_escaped_document_path(child)[-1] == "Symbols/Numbers" + assert child.count("/") == 2 + + +def test_split_document_path_unescapes_section_titles() -> None: + chunk_path = join_document_path( + ["manual.pdf", "Index", "Symbols/Numbers"] + ) + root_parts, section_parts = split_document_path( + chunk_path, + source_file_name="manual.pdf", + ) + assert root_parts == ["manual.pdf"] + assert section_parts == ["Index", "Symbols/Numbers"] + + +def test_section_path_from_chunk_path_preserves_slash_title() -> None: + chunk_path = join_document_path( + ["manual.pdf", "Index", "Symbols/Numbers"] + ) + assert ( + section_path_from_chunk_path( + chunk_path, + source_file_name="manual.pdf", + ) + == "Index / Symbols/Numbers" + ) + assert split_section_path("Index / Symbols/Numbers") == [ + "Index", + "Symbols/Numbers", + ] + + +def test_doc_nav_keeps_slash_title_as_single_section() -> None: + chunk_path = join_document_path( + ["manual.pdf", "Index", "Symbols/Numbers"] + ) + doc_nav = ZipDocNavigationBuilder().build_doc_nav( + [ + { + "chunk_id": "chunk_slash_title", + "type": "text", + "content": "index symbols", + "path": chunk_path, + "metadata": {"summary": "symbols"}, + } + ], + "manual.pdf", + ) + sections = doc_nav["sections"] + assert sections[0]["title"] == "Index" + assert sections[0]["children"][0]["title"] == "Symbols/Numbers" + assert sections[0]["children"][0]["path"] == chunk_path + assert sections[0]["children"][0]["children"] == [] diff --git a/page_memory_work_summary_20260624.md b/page_memory_work_summary_20260624.md deleted file mode 100644 index 7f7ed99d5..000000000 --- a/page_memory_work_summary_20260624.md +++ /dev/null @@ -1,437 +0,0 @@ -# Page-Memory 流程改造与测试进度汇总 - -日期:2026-06-24 -文档:`SJSYJ-SC-2024 企业制度汇编(上册).pdf` -当前 debug 目录:`/Users/wuchengke/.knowhere/_debug_parse/SJSYJ-SC-2024 企业制度汇编(上册).pdf/page_memory` - -## 1. 背景问题 - -这轮工作从“检查解析”对话中的异常开始: - -```text -The encrypted content Hand...run. could not be verified. -Reason: Encrypted content could not be decrypted or parsed. -traceid: f63bf5fe74f7bb1859be513dd7e54b22 -``` - -当时 debug 目录里只有: - -```text -page_memory_fine_hierarchy.json -``` - -但没有看到 PAGE-TAG 结果,也无法判断流程到底跑到哪一步。随后确认核心问题不是单个文件缺失,而是 page-memory 的中间产物、trace、scope 组织方式和生产链路不一致,导致 debug 不可控、不透明。 - -## 2. 已确认的目标流程 - -我们对齐后的 page-memory 生产/测试流程是: - -1. 先做 DOC_PROFILE,抽取文档页特征和 TOC hierarchy。 -2. 基于粗 TOC hierarchy 建立 coarse scopes。 -3. 对每个 coarse scope 构建 fine hierarchy。 -4. 只在 fine hierarchy 覆盖的页面范围内做 PAGE-TAG。 -5. 如开启图表资产提取,则同样在 hierarchy 覆盖范围内抽取资产。 -6. 最后汇总生成顶层 `hierarchy.json`、`page_tags.json`、`assets.json`,并继续支持后续 `chunks.json`、`doc_nav.json`、`manifest.json` 生成。 - -测试链路保持生产形状,但可以用 `--fat-only` 只选择最大的粗 scope 进行快速验证。 - -## 3. 已落地的主要改造 - -### 3.1 Trace 统一到顶层 - -已取消 `_doc_agent/trace.json` 的重复落盘,统一写到: - -```text -page_memory/trace.json -``` - -`_doc_agent/` 目前只保留 DOC_PROFILE/TOC 相关原始调试材料: - -```text -_doc_agent/anatomy_map.json -_doc_agent/toc_hierarchies.json -``` - -### 3.2 中间产物精简 - -当前 stop-at fine 的落盘结构已经收敛为: - -```text -page_memory/ - trace.json - hierarchy.json - page_tags.json - _doc_agent/ - anatomy_map.json - toc_hierarchies.json - scopes/ - p225-301/ - scope.json - fine_hierarchy.json - page_tags.json -``` - -不再输出大量重复的临时 JSON,例如旧的 `coarse_tag_scope.json`、`page_memory_fine_hierarchy.json`、`page_tags_pre_hierarchy.json` 等。 - -### 3.3 Scope 目录改为页码范围 - -scope 目录已从 hash/UUID 风格改为页码范围: - -```text -scopes/p225-301/ -``` - -这样 debug 时可以直接看出该 scope 覆盖的页域。 - -### 3.4 `hierarchy.json` 改为可读树形优先 - -顶层 `hierarchy.json` 和 scope 内 `fine_hierarchy.json` 都改成: - -```json -{ - "HIERARCHY": {}, - "nodes": [], - "stats": {} -} -``` - -其中 `HIERARCHY` 是第一字段,形态对齐最终 `manifest.json` 里的 `HIERARCHY` 字段,方便直接肉眼 debug。 - -scope 内的 `fine_hierarchy.json` 额外包含: - -```json -{ - "scope": {} -} -``` - -机器流程仍可继续使用 `nodes`。 - -### 3.5 页域字段精简 - -scope 和 trace 中不再同时记录 `page_ranges` 和完整 `pages` 列表。 - -当前约定: - -```json -{ - "document_page_count": 423, - "page_count": 77, - "page_ranges": [[225, 301]] -} -``` - -保留逐页 `page_index` 的地方仅限实体数据,例如: - -```text -page_tags.json -assets.json -``` - -因为这些文件本身就是逐页/逐资产记录。 - -## 4. 当前测试进度 - -本次从原始 PDF 重新开始测试: - -```text -/Users/wuchengke/Desktop/temp/test_docs/SJSYJ-SC-2024 企业制度汇编(上册).pdf -``` - -执行目标: - -```text -从头跑到最大粗 scope 的 fine hierarchy -``` - -实际命令: - -```bash -uv run python apps/worker/scripts/debug_page_memory.py \ - --file '/Users/wuchengke/Desktop/temp/test_docs/SJSYJ-SC-2024 企业制度汇编(上册).pdf' \ - --fat-only \ - --stop-at fine -``` - -结果:成功,最终状态为: - -```text -stopped_at_fine -``` - -### 4.1 DOC_PROFILE 结果 - -```text -page_count: 423 -toc_pages: [5, 6, 228] -native TOC TitleNode: 58 -native TOC leaf nodes: 44 -``` - -注意:C4 skeleton 阶段日志显示嵌入式目录页 `[228]` 被当前 global TOC 选择逻辑跳过: - -```text -embedded_toc_region_outside_front_cluster -``` - -这属于后续可优化点。 - -### 4.2 C4 Skeleton 定位 - -```text -skeleton_count: 25 -elapsed: 279.36s -``` - -残余定位阶段使用了多轮小窗口渲染 + VLM 确认,耗时较长。 - -### 4.3 最大粗 scope 选择 - -`--fat-only` 本次选中: - -```text -scope_id: p225-301 -page_ranges: [[225, 301]] -page_count: 77 -coarse skeletons before fine: 1 -``` - -### 4.4 Fine hierarchy 结果 - -title detection: - -```text -77 VLM calls -54 titles found -36 pages with observed_titles -``` - -fine hierarchy: - -```text -1 -> 52 skeletons -elapsed: 110.5s -``` - -最终 `hierarchy.json`: - -```json -{ - "stats": { - "node_count": 52, - "page_count": 77, - "page_ranges": [[225, 301]], - "max_depth": 6 - } -} -``` - -顶层 `HIERARCHY` 当前顶级节点: - -```text -安全类 -``` - -## 5. 当前已验证的文件 - -### 5.1 顶层 `hierarchy.json` - -路径: - -```text -page_memory/hierarchy.json -``` - -检查结果: - -```text -top_keys: HIERARCHY, nodes, stats -node_count: 52 -page_count: 77 -page_ranges: [[225, 301]] -max_depth: 6 -``` - -### 5.2 Scope `scope.json` - -路径: - -```text -page_memory/scopes/p225-301/scope.json -``` - -检查结果: - -```json -{ - "scope_id": "p225-301", - "strategy": "fat_only_coarse_scope:refined", - "document_page_count": 423, - "page_count": 77, - "page_ranges": [[225, 301]], - "skeleton_count": 52 -} -``` - -确认:没有 `pages` 长列表。 - -### 5.3 Scope `fine_hierarchy.json` - -路径: - -```text -page_memory/scopes/p225-301/fine_hierarchy.json -``` - -检查结果: - -```text -top_keys: HIERARCHY, nodes, stats, scope -node_count: 52 -page_count: 77 -page_ranges: [[225, 301]] -max_depth: 6 -``` - -### 5.4 `trace.json` - -路径: - -```text -page_memory/trace.json -``` - -检查结果: - -```text -final_status: stopped_at_fine -stage_count: 10 -summary.page_count: 423 -summary.scope_id: p225-301 -``` - -最后几个 stage 的 page_info 均为 compact range: - -```text -C4.coarse_scope page_count=77 page_ranges=[[225, 301]] -C1.render_pages.coarse page_count=77 page_ranges=[[225, 301]] -C2.page_plan.coarse page_count=77 page_ranges=[[225, 301]] -C3b.title_detection page_count=77 page_ranges=[[225, 301]] -C4b.fine_hierarchy fat_leaf.page_count=77 fat_leaf.page_ranges=[[225, 301]] -``` - -## 6. 已跑过的代码检查 - -最近一次相关检查通过: - -```bash -uv run ruff check \ - apps/worker/app/services/page_memory/memory_service.py \ - apps/worker/app/services/page_memory/fine_hierarchy.py \ - apps/worker/scripts/debug_page_memory.py -``` - -```bash -python -m py_compile \ - apps/worker/app/services/page_memory/memory_service.py \ - apps/worker/app/services/page_memory/fine_hierarchy.py \ - apps/worker/scripts/debug_page_memory.py -``` - -```bash -uv run pytest \ - apps/worker/tests/contract/test_page_memory_fine_hierarchy_contract.py \ - apps/worker/tests/contract/test_page_memory_node_assembler_contract.py \ - apps/worker/tests/contract/test_document_agent_budget_contract.py \ - -q -``` - -结果: - -```text -17 passed -``` - -## 7. 当前待讨论/后续优化点 - -### 7.1 `parent_paths` 仍偏长 - -`scope.json` 里目前仍保留 `parent_paths`,虽然不是 page 长列表,但对 debug 阅读来说有些臃肿。 - -可选优化: - -```json -{ - "root_path": "...", - "parent_path_count": 14 -} -``` - -完整 `parent_paths` 可以放入 `trace.json`。 - -### 7.2 C4 skeleton 残余定位耗时较长 - -本次 C4 skeleton 定位耗时约 279 秒,明显比 fine hierarchy 更重。 - -后续可考虑: - -1. 对已定位的 TOC 节点减少 residual VLM。 -2. 对 debug 模式增加更明确的 residual cap。 -3. 对多个 coarse scope 并发定位/处理。 -4. 复用 anatomy + skeleton cache 做快速迭代。 - -### 7.3 嵌入式 TOC 页 228 被跳过 - -本次 DOC_PROFILE 找到目录页 `[5, 6, 228]`,但 skeleton 阶段跳过了嵌入式目录区域 `[228]`。 - -这可能影响后续更细粒度 scope 的粗 hierarchy 完整性,需要单独评估: - -```text -embedded_toc_region_outside_front_cluster -``` - -### 7.4 下一步测试建议 - -建议下一步直接测试: - -```text -fine hierarchy -> PAGE-TAG -``` - -即跑到: - -```text ---stop-at tag -``` - -重点检查: - -1. `scopes/p225-301/page_tags.json` -2. 顶层 `page_tags.json` -3. `trace.json` 中 PAGE-TAG 是否只覆盖 `[[225, 301]]` -4. PAGE-TAG 是否能支撑后续 `chunks.json` 和 `doc_nav.json` - -之后再开启图表资产提取,验证: - -```text -assets.json -scopes/p225-301/assets.json -``` - -## 8. 当前涉及的主要代码文件 - -本轮 page-memory 相关核心改动集中在: - -```text -apps/worker/app/services/page_memory/memory_service.py -apps/worker/app/services/page_memory/fine_hierarchy.py -apps/worker/app/services/page_memory/page_renderer.py -apps/worker/app/services/page_memory/page_assets.py -apps/worker/app/services/page_memory/node_assembler.py -apps/worker/app/services/document_agent/trace.py -apps/worker/app/services/document_agent/visual.py -apps/worker/scripts/debug_page_memory.py -``` - -其中 `apps/worker/scripts/debug_page_memory.py` 是当前测试入口。 -